heading towards targeted chunk sizes, made sure updated files have old copies cleaned up

This commit is contained in:
2026-09-16 22:16:38 -06:00
parent 23c2db1046
commit 8d0c79b34d
2 changed files with 72 additions and 19 deletions
+32 -3
View File
@@ -19,19 +19,22 @@ Opens `similarities.json`, reverses the sort order, and saves it as
### `refresh_sources.lua` ### `refresh_sources.lua`
Creates/Maintains a store of chunked data with embeddings based on configurable Creates/Maintains a store of chunked data with embeddings based on configurable
sources. Example: `sources.json`. Example:
```json ```json
{ {
"source name":{ "source name":{
"embedding_model":"qwen3-embedding:0.6b",
"filters":{ "filters":{
"blacklist":[".git"], "blacklist":[".git"],
"extension_whitelist":["md"] "extension_whitelist":["md"]
}, },
"initialize_command":"git clone REMOTE .", "initialize_command":"git clone REMOTE .",
"max_chunk_size":32768,
"path":"will be created before initialize_command is run", "path":"will be created before initialize_command is run",
"strip_frontmatter":true, "strip_frontmatter":true,
"refresh_command":"git reset --hard origin/main" "target_chunk_size":3072,
"refresh_command":"git fetch origin && git reset --hard origin/main"
} }
} }
``` ```
@@ -41,6 +44,9 @@ The filters are based on `utility.tree`'s filter options (optional).
while `refresh_command` is run each time (optional). while `refresh_command` is run each time (optional).
Commands will be run in the specified `path`. Commands will be run in the specified `path`.
`strip_frontmatter` will remove YAML frontmatter (common in Markdown files). `strip_frontmatter` will remove YAML frontmatter (common in Markdown files).
`embedding_model` and `max_chunk_size` are self-explanatory (optional).
`target_chunk_size` (optional) allows for finer-grained chunks to be generated
alongside the maximum overviews for more precise retrieval.
The embeddings are stored like so: The embeddings are stored like so:
@@ -55,7 +61,11 @@ The embeddings are stored like so:
} }
``` ```
TODO SHA2 512-bit sums are used to link a text chunk with its embedding. All file
names reference a list of text chunks so they can handle being too large. Every
file that is too large for a single chunk has a whole-file embedding calculated
first (it is the first element of the array), so that if a whole file becomes
relevant, it can still show up instead of only chunks.
### `synopsis_generator.lua` ### `synopsis_generator.lua`
Chooses a random file within `notebook`, and generates a novel synopsis from it. Chooses a random file within `notebook`, and generates a novel synopsis from it.
@@ -64,6 +74,25 @@ Arguments:
- `refresh_file_list`: Refreshes the cached file list to choose from. - `refresh_file_list`: Refreshes the cached file list to choose from.
- `export_ordered_list_of_prompts`: Makes an epub to review generated synopses. - `export_ordered_list_of_prompts`: Makes an epub to review generated synopses.
## JSON config
This repo uses my utility library's config system, using a `config.json` file in
the repo root that is excluded from commits.
```json
{
"models":{
"embedding":{
"model":"qwen3-embedding:0.6b",
"max_chunk_size":32768
},
"initialized_sources":{}
}
}
```
`refresh_sources.lua` uses `models` to store default embedding model information
and `initialized_sources` to store which sources have been initialized.
## Tasks ## Tasks
- [ ] The whitelisting/blacklisting of tree should be in list too. - [ ] The whitelisting/blacklisting of tree should be in list too.
- [ ] synopsis_generator should be able to blacklist files it already tried? - [ ] synopsis_generator should be able to blacklist files it already tried?
+40 -16
View File
@@ -61,7 +61,7 @@ end
local refresh_file_list = function(source_name, data_source) local refresh_file_list = function(source_name, data_source)
timing.mark("Assembling file list for " .. source_name .. ".") timing.mark("Assembling file list for \"" .. source_name .. "\"")
local full_path = "PRIVATE_DATA" .. utility.path_separator .. data_source.path local full_path = "PRIVATE_DATA" .. utility.path_separator .. data_source.path
if data_source.initialize_command and (not config.models.initialized_sources[data_source.path]) then if data_source.initialize_command and (not config.models.initialized_sources[data_source.path]) then
@@ -87,7 +87,7 @@ local refresh_file_list = function(source_name, data_source)
file_list[#file_list + 1] = file_name file_list[#file_list + 1] = file_name
end) end)
timing.mark("Finished assembling file list for " .. source_name .. ".") timing.mark("Finished assembling file list for \"" .. source_name .. "\"")
return file_list return file_list
end end
@@ -104,20 +104,9 @@ local generate_embeddings = function(data_source, text)
return json.decode(result) return json.decode(result)
end end
-- returns nothing for empty files and errors local make_chunks = function(text, chunk_size)
local process_file = function(data_source, file_name) assert(chunk_size, "make_chunks() requires chunk_size")
local text = utility.read_file(file_name)
if data_source.strip_frontmatter then
text = text_processing.strip_frontmatter(text)
end
if #text == 0 then
log("empty", file_name .. "\n is empty and being skipped.")
return
end
local chunk_size = data_source.max_chunk_size or config.models.embedding.max_chunk_size
local half_chunk_size = math.floor(chunk_size / 2) local half_chunk_size = math.floor(chunk_size / 2)
local chunks = { text } local chunks = { text }
@@ -135,6 +124,25 @@ local process_file = function(data_source, file_name)
end end
end end
return chunks
end
-- returns nothing for empty files and errors
local process_file = function(data_source, file_name)
local text = utility.read_file(file_name)
if data_source.strip_frontmatter then
text = text_processing.strip_frontmatter(text)
end
if #text == 0 then
log("empty", file_name .. "\n is empty and being skipped.")
return
end
local chunk_size = data_source.max_chunk_size or config.models.embedding.max_chunk_size
local chunks = make_chunks(text, chunk_size)
local new_embeddings = {} local new_embeddings = {}
for i = 1, #chunks do for i = 1, #chunks do
log("debug", "Embedding length:", #chunks[i]) log("debug", "Embedding length:", #chunks[i])
@@ -202,6 +210,15 @@ local memorize_file = function(data_source, file_name)
return file_sums return file_sums
end end
local delete_memories = function(sha512sum_list)
-- if true then return end -- TEMP disabling removal to compare and verify function
for i = 1, #sha512sum_list do
local sha512sum = sha512sum_list[i]
embeddings.vectors[sha512sum] = nil
os.execute("rm " .. (memory_path .. utility.path_separator .. sha512sum):enquote())
end
end
local refresh_sources = function() local refresh_sources = function()
os.execute("mkdir -p " .. memory_path:enquote()) os.execute("mkdir -p " .. memory_path:enquote())
@@ -230,6 +247,7 @@ local refresh_sources = function()
local function loop() local function loop()
local sha512sum = utility.sha512sum(file_name) local sha512sum = utility.sha512sum(file_name)
log("sha", file_name, sha512sum, "\n Sum present? " .. tostring(embeddings.vectors[sha512sum])) log("sha", file_name, sha512sum, "\n Sum present? " .. tostring(embeddings.vectors[sha512sum]))
if embeddings.vectors[sha512sum] then if embeddings.vectors[sha512sum] then
log("debug", file_name .. "\n has already been embedded, skipping.") log("debug", file_name .. "\n has already been embedded, skipping.")
-- add file reference if it was missing -- add file reference if it was missing
@@ -239,7 +257,13 @@ local refresh_sources = function()
return return
end end
log("debug", "Generating embeddings for " .. file_name .. ".") if embeddings.files[file_name] then
log("debug", file_name .. "\n Removing old embeddings for modified file.")
delete_memories(embeddings.files[file_name])
embeddings.files[file_name] = nil
end
log("debug", file_name .. "\n Generating embeddings..")
local file_sums = memorize_file(data_source, file_name) local file_sums = memorize_file(data_source, file_name)
embeddings.files[file_name] = file_sums embeddings.files[file_name] = file_sums
end end