Compare commits

...
3 changed files with 107 additions and 36 deletions

No files matched your search

+32 -3
View File
@@ -19,19 +19,22 @@ Opens `similarities.json`, reverses the sort order, and saves it as
### `refresh_sources.lua`
Creates/Maintains a store of chunked data with embeddings based on configurable
sources. Example:
`sources.json`. Example:
```json
{
"source name":{
"embedding_model":"qwen3-embedding:0.6b",
"filters":{
"blacklist":[".git"],
"extension_whitelist":["md"]
},
"initialize_command":"git clone REMOTE .",
"max_chunk_size":32768,
"path":"will be created before initialize_command is run",
"strip_frontmatter":true,
"refresh_command":"git reset --hard origin/main"
"target_chunk_size":3072,
"refresh_command":"git fetch origin && git reset --hard origin/main"
}
}
```
@@ -41,6 +44,9 @@ The filters are based on `utility.tree`'s filter options (optional).
while `refresh_command` is run each time (optional).
Commands will be run in the specified `path`.
`strip_frontmatter` will remove YAML frontmatter (common in Markdown files).
`embedding_model` and `max_chunk_size` are self-explanatory (optional).
`target_chunk_size` (optional) allows for finer-grained chunks to be generated
alongside the maximum overviews for more precise retrieval.
The embeddings are stored like so:
@@ -55,7 +61,11 @@ The embeddings are stored like so:
}
```
TODO
SHA2 512-bit sums are used to link a text chunk with its embedding. All file
names reference a list of text chunks so they can handle being too large. Every
file that is too large for a single chunk has a whole-file embedding calculated
first (it is the first element of the array), so that if a whole file becomes
relevant, it can still show up instead of only chunks.
### `synopsis_generator.lua`
Chooses a random file within `notebook`, and generates a novel synopsis from it.
@@ -64,6 +74,25 @@ Arguments:
- `refresh_file_list`: Refreshes the cached file list to choose from.
- `export_ordered_list_of_prompts`: Makes an epub to review generated synopses.
## JSON config
This repo uses my utility library's config system, using a `config.json` file in
the repo root that is excluded from commits.
```json
{
"models":{
"embedding":{
"model":"qwen3-embedding:0.6b",
"max_chunk_size":32768
},
"initialized_sources":{}
}
}
```
`refresh_sources.lua` uses `models` to store default embedding model information
and `initialized_sources` to store which sources have been initialized.
## Tasks
- [ ] The whitelisting/blacklisting of tree should be in list too.
- [ ] synopsis_generator should be able to blacklist files it already tried?
+16 -9
View File
@@ -1,11 +1,8 @@
local stored_messages = {}
local display_levels = {}
local enable_logging = true
local message_limit = math.huge
-- USAGE:
-- -- set arbitrary log levels to be printed immediately
-- log{ info = true, warning = true, ducksauce = true, }
-- -- disable storage of messages depending on log level
-- log{ error = false, verbose = false, }
-- -- send anything (except nil) to a log level
-- log("bacom", true, "text", 5, function() end, {})
-- -- print all messages saved at a particular log level
@@ -14,13 +11,22 @@ local message_limit = math.huge
-- log(false) -- this deletes all stored messages
-- -- set a maximum number of stored messages per log level
-- log(50) -- warning: reaching large limits makes logging expensive
-- -- check if a log level is being displayed
-- if log.arbitrary_exit then log("arbitrary_exit") os.exit(1) end
-- -- do anything to the stored messages (this example deletes them all)
-- log(function(messages) return {} end)
return function(options, ...)
local microlog = {} -- display_levels are stored directly here
local stored_messages = {}
local enable_logging = true
local message_limit = math.huge
return setmetatable(microlog, {
__call = function(self, options, ...)
local options_type = type(options)
if (options_type == "string") and enable_logging then
if microlog[options] == false then return end
if not stored_messages[options] then stored_messages[options] = {} end
local message_table = stored_messages[options]
@@ -41,13 +47,13 @@ return function(options, ...)
-- store message (and print if in display_levels)
message_table[#message_table + 1] = current_message
if #message_table > message_limit then table.remove(message_table, 1) end
if display_levels[options] then print(current_message) end
if microlog[options] then print(current_message) end
end
-- set display_levels
elseif options_type == "table" then
for k,v in pairs(options) do
display_levels[k] = v
microlog[k] = v
end
elseif options_type == "number" then
@@ -72,4 +78,5 @@ return function(options, ...)
local result = options(stored_messages)
if type(result) == "table" then stored_messages = result end
end
end
end
})
+59 -24
View File
@@ -25,7 +25,7 @@ end
log{
info = true,
warning = true,
-- debug = true,
debug = false,
-- files = true, -- debugging why the wrong files are selected
-- sha = true, -- what the fuck is going on with sha sums?
}
@@ -43,7 +43,7 @@ if not utility.path_exists(embeddings_file_path) then
}, embeddings_file_path)
end
embeddings = utility.load_data(embeddings_file_path)
local function embeddings_debug()
if log.debug then
log("debug", "Embeddings loaded.", embeddings, embeddings.files, embeddings.vectors)
local file_count = 0
for k,v in pairs(embeddings.files) do
@@ -57,12 +57,11 @@ local function embeddings_debug()
log("debug", vector_count .. " vectors.")
-- os.exit(1)
end
embeddings_debug()
local refresh_file_list = function(source_name, data_source)
timing.mark("Assembling file list for " .. source_name .. ".")
timing.mark("Assembling file list for \"" .. source_name .. "\"")
local full_path = "PRIVATE_DATA" .. utility.path_separator .. data_source.path
if data_source.initialize_command and (not config.models.initialized_sources[data_source.path]) then
@@ -88,34 +87,26 @@ local refresh_file_list = function(source_name, data_source)
file_list[#file_list + 1] = file_name
end)
timing.mark("Finished assembling file list for " .. source_name .. ".")
timing.mark("Finished assembling file list for \"" .. source_name .. "\"")
return file_list
end
-- returns nothing when too much text is sent
local generate_embeddings = function(text)
if #text > config.models.embedding.max_chunk_size then
local generate_embeddings = function(data_source, text)
local max_chunk_size = data_source.max_chunk_size or config.models.embedding.max_chunk_size
local model = data_source.embedding_model or config.models.embedding.model
if #text > max_chunk_size then
return nil, "generate_embeddings() must only be passed appropriately-sized chunks!"
end
local result = utility.llm_prompt(text, config.models.embedding.model)
local result = utility.llm_prompt(text, model)
return json.decode(result)
end
-- returns nothing for empty files and errors
local process_file = function(data_source, file_name)
local text = utility.read_file(file_name)
local make_chunks = function(text, chunk_size)
assert(chunk_size, "make_chunks() requires chunk_size")
if data_source.strip_frontmatter then
text = text_processing.strip_frontmatter(text)
end
if #text == 0 then
log("empty", file_name .. "\n is empty and being skipped.")
return
end
local chunk_size = config.models.embedding.max_chunk_size
local half_chunk_size = math.floor(chunk_size / 2)
local chunks = { text }
@@ -133,17 +124,45 @@ local process_file = function(data_source, file_name)
end
end
return chunks
end
-- returns nothing for empty files and errors
local process_file = function(data_source, file_name)
local text = utility.read_file(file_name)
if data_source.strip_frontmatter then
text = text_processing.strip_frontmatter(text)
end
if #text == 0 then
log("empty", file_name .. "\n is empty and being skipped.")
return
end
local chunk_size = data_source.max_chunk_size or config.models.embedding.max_chunk_size
local chunks = make_chunks(text, chunk_size)
if data_source.target_chunk_size then
local extra_chunks = make_chunks(text, data_source.target_chunk_size)
for i = 2, #extra_chunks do
chunks[#chunks + 1] = extra_chunks[i]
end
end
local new_embeddings = {}
for i = 1, #chunks do
log("debug", "Embedding length:", #chunks[i])
new_embeddings[i] = generate_embeddings(chunks[i]) or {}
log("debug", "Embedding source length:", #chunks[i])
new_embeddings[i] = generate_embeddings(data_source, chunks[i]) or {}
end
if #new_embeddings[1] == 0 then
if log.debug then
log("debug", "Vector lengths:")
for e = 1, #new_embeddings do
log("debug", "", e, #new_embeddings[e])
end
end
if #new_embeddings == 1 then
-- Ollama very rarely errors with:
-- Error: do embedding request: Post "http://127.0.0.1:53441/v1/embeddings": EOF
@@ -198,6 +217,15 @@ local memorize_file = function(data_source, file_name)
return file_sums
end
local delete_memories = function(sha512sum_list)
-- if true then return end -- TEMP disabling removal to compare and verify function
for i = 1, #sha512sum_list do
local sha512sum = sha512sum_list[i]
embeddings.vectors[sha512sum] = nil
os.execute("rm " .. (memory_path .. utility.path_separator .. sha512sum):enquote())
end
end
local refresh_sources = function()
os.execute("mkdir -p " .. memory_path:enquote())
@@ -226,6 +254,7 @@ local refresh_sources = function()
local function loop()
local sha512sum = utility.sha512sum(file_name)
log("sha", file_name, sha512sum, "\n Sum present? " .. tostring(embeddings.vectors[sha512sum]))
if embeddings.vectors[sha512sum] then
log("debug", file_name .. "\n has already been embedded, skipping.")
-- add file reference if it was missing
@@ -235,7 +264,13 @@ local refresh_sources = function()
return
end
log("debug", "Generating embeddings for " .. file_name .. ".")
if embeddings.files[file_name] then
log("debug", file_name .. "\n Removing old embeddings for modified file.")
delete_memories(embeddings.files[file_name])
embeddings.files[file_name] = nil
end
log("debug", file_name .. "\n Generating embeddings..")
local file_sums = memorize_file(data_source, file_name)
embeddings.files[file_name] = file_sums
end