diff --git a/ReadMe.md b/ReadMe.md index 4ba1471..44c0106 100644 --- a/ReadMe.md +++ b/ReadMe.md @@ -19,19 +19,22 @@ Opens `similarities.json`, reverses the sort order, and saves it as ### `refresh_sources.lua` Creates/Maintains a store of chunked data with embeddings based on configurable -sources. Example: +`sources.json`. Example: ```json { "source name":{ + "embedding_model":"qwen3-embedding:0.6b", "filters":{ "blacklist":[".git"], "extension_whitelist":["md"] }, "initialize_command":"git clone REMOTE .", + "max_chunk_size":32768, "path":"will be created before initialize_command is run", "strip_frontmatter":true, - "refresh_command":"git reset --hard origin/main" + "target_chunk_size":3072, + "refresh_command":"git fetch origin && git reset --hard origin/main" } } ``` @@ -41,6 +44,9 @@ The filters are based on `utility.tree`'s filter options (optional). while `refresh_command` is run each time (optional). Commands will be run in the specified `path`. `strip_frontmatter` will remove YAML frontmatter (common in Markdown files). +`embedding_model` and `max_chunk_size` are self-explanatory (optional). +`target_chunk_size` (optional) allows for finer-grained chunks to be generated +alongside the maximum overviews for more precise retrieval. The embeddings are stored like so: @@ -55,7 +61,11 @@ The embeddings are stored like so: } ``` -TODO +SHA2 512-bit sums are used to link a text chunk with its embedding. All file +names reference a list of text chunks so they can handle being too large. Every +file that is too large for a single chunk has a whole-file embedding calculated +first (it is the first element of the array), so that if a whole file becomes +relevant, it can still show up instead of only chunks. ### `synopsis_generator.lua` Chooses a random file within `notebook`, and generates a novel synopsis from it. @@ -64,6 +74,25 @@ Arguments: - `refresh_file_list`: Refreshes the cached file list to choose from. - `export_ordered_list_of_prompts`: Makes an epub to review generated synopses. +## JSON config +This repo uses my utility library's config system, using a `config.json` file in +the repo root that is excluded from commits. + +```json +{ + "models":{ + "embedding":{ + "model":"qwen3-embedding:0.6b", + "max_chunk_size":32768 + }, + "initialized_sources":{} + } +} +``` + +`refresh_sources.lua` uses `models` to store default embedding model information +and `initialized_sources` to store which sources have been initialized. + ## Tasks - [ ] The whitelisting/blacklisting of tree should be in list too. - [ ] synopsis_generator should be able to blacklist files it already tried? diff --git a/refresh_sources.lua b/refresh_sources.lua index 4a8129e..7e38b36 100755 --- a/refresh_sources.lua +++ b/refresh_sources.lua @@ -61,7 +61,7 @@ end local refresh_file_list = function(source_name, data_source) - timing.mark("Assembling file list for " .. source_name .. ".") + timing.mark("Assembling file list for \"" .. source_name .. "\"") local full_path = "PRIVATE_DATA" .. utility.path_separator .. data_source.path if data_source.initialize_command and (not config.models.initialized_sources[data_source.path]) then @@ -87,7 +87,7 @@ local refresh_file_list = function(source_name, data_source) file_list[#file_list + 1] = file_name end) - timing.mark("Finished assembling file list for " .. source_name .. ".") + timing.mark("Finished assembling file list for \"" .. source_name .. "\"") return file_list end @@ -104,20 +104,9 @@ local generate_embeddings = function(data_source, text) return json.decode(result) end --- returns nothing for empty files and errors -local process_file = function(data_source, file_name) - local text = utility.read_file(file_name) +local make_chunks = function(text, chunk_size) + assert(chunk_size, "make_chunks() requires chunk_size") - if data_source.strip_frontmatter then - text = text_processing.strip_frontmatter(text) - end - - if #text == 0 then - log("empty", file_name .. "\n is empty and being skipped.") - return - end - - local chunk_size = data_source.max_chunk_size or config.models.embedding.max_chunk_size local half_chunk_size = math.floor(chunk_size / 2) local chunks = { text } @@ -135,6 +124,25 @@ local process_file = function(data_source, file_name) end end + return chunks +end + +-- returns nothing for empty files and errors +local process_file = function(data_source, file_name) + local text = utility.read_file(file_name) + + if data_source.strip_frontmatter then + text = text_processing.strip_frontmatter(text) + end + + if #text == 0 then + log("empty", file_name .. "\n is empty and being skipped.") + return + end + + local chunk_size = data_source.max_chunk_size or config.models.embedding.max_chunk_size + local chunks = make_chunks(text, chunk_size) + local new_embeddings = {} for i = 1, #chunks do log("debug", "Embedding length:", #chunks[i]) @@ -202,6 +210,15 @@ local memorize_file = function(data_source, file_name) return file_sums end +local delete_memories = function(sha512sum_list) + -- if true then return end -- TEMP disabling removal to compare and verify function + for i = 1, #sha512sum_list do + local sha512sum = sha512sum_list[i] + embeddings.vectors[sha512sum] = nil + os.execute("rm " .. (memory_path .. utility.path_separator .. sha512sum):enquote()) + end +end + local refresh_sources = function() os.execute("mkdir -p " .. memory_path:enquote()) @@ -230,6 +247,7 @@ local refresh_sources = function() local function loop() local sha512sum = utility.sha512sum(file_name) log("sha", file_name, sha512sum, "\n Sum present? " .. tostring(embeddings.vectors[sha512sum])) + if embeddings.vectors[sha512sum] then log("debug", file_name .. "\n has already been embedded, skipping.") -- add file reference if it was missing @@ -239,7 +257,13 @@ local refresh_sources = function() return end - log("debug", "Generating embeddings for " .. file_name .. ".") + if embeddings.files[file_name] then + log("debug", file_name .. "\n Removing old embeddings for modified file.") + delete_memories(embeddings.files[file_name]) + embeddings.files[file_name] = nil + end + + log("debug", file_name .. "\n Generating embeddings..") local file_sums = memorize_file(data_source, file_name) embeddings.files[file_name] = file_sums end