Compare commits
4
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
77298dad0c | ||
|
|
8d0c79b34d | ||
|
|
23c2db1046 | ||
|
|
6381a49d13 |
No files matched your search
@@ -19,19 +19,22 @@ Opens `similarities.json`, reverses the sort order, and saves it as
|
||||
|
||||
### `refresh_sources.lua`
|
||||
Creates/Maintains a store of chunked data with embeddings based on configurable
|
||||
sources. Example:
|
||||
`sources.json`. Example:
|
||||
|
||||
```json
|
||||
{
|
||||
"source name":{
|
||||
"embedding_model":"qwen3-embedding:0.6b",
|
||||
"filters":{
|
||||
"blacklist":[".git"],
|
||||
"extension_whitelist":["md"]
|
||||
},
|
||||
"initialize_command":"git clone REMOTE .",
|
||||
"max_chunk_size":32768,
|
||||
"path":"will be created before initialize_command is run",
|
||||
"strip_frontmatter":true,
|
||||
"refresh_command":"git reset --hard origin/main"
|
||||
"target_chunk_size":3072,
|
||||
"refresh_command":"git fetch origin && git reset --hard origin/main"
|
||||
}
|
||||
}
|
||||
```
|
||||
@@ -41,6 +44,9 @@ The filters are based on `utility.tree`'s filter options (optional).
|
||||
while `refresh_command` is run each time (optional).
|
||||
Commands will be run in the specified `path`.
|
||||
`strip_frontmatter` will remove YAML frontmatter (common in Markdown files).
|
||||
`embedding_model` and `max_chunk_size` are self-explanatory (optional).
|
||||
`target_chunk_size` (optional) allows for finer-grained chunks to be generated
|
||||
alongside the maximum overviews for more precise retrieval.
|
||||
|
||||
The embeddings are stored like so:
|
||||
|
||||
@@ -55,7 +61,11 @@ The embeddings are stored like so:
|
||||
}
|
||||
```
|
||||
|
||||
TODO
|
||||
SHA2 512-bit sums are used to link a text chunk with its embedding. All file
|
||||
names reference a list of text chunks so they can handle being too large. Every
|
||||
file that is too large for a single chunk has a whole-file embedding calculated
|
||||
first (it is the first element of the array), so that if a whole file becomes
|
||||
relevant, it can still show up instead of only chunks.
|
||||
|
||||
### `synopsis_generator.lua`
|
||||
Chooses a random file within `notebook`, and generates a novel synopsis from it.
|
||||
@@ -64,6 +74,25 @@ Arguments:
|
||||
- `refresh_file_list`: Refreshes the cached file list to choose from.
|
||||
- `export_ordered_list_of_prompts`: Makes an epub to review generated synopses.
|
||||
|
||||
## JSON config
|
||||
This repo uses my utility library's config system, using a `config.json` file in
|
||||
the repo root that is excluded from commits.
|
||||
|
||||
```json
|
||||
{
|
||||
"models":{
|
||||
"embedding":{
|
||||
"model":"qwen3-embedding:0.6b",
|
||||
"max_chunk_size":32768
|
||||
},
|
||||
"initialized_sources":{}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
`refresh_sources.lua` uses `models` to store default embedding model information
|
||||
and `initialized_sources` to store which sources have been initialized.
|
||||
|
||||
## Tasks
|
||||
- [ ] The whitelisting/blacklisting of tree should be in list too.
|
||||
- [ ] synopsis_generator should be able to blacklist files it already tried?
|
||||
+16
-9
@@ -1,11 +1,8 @@
|
||||
local stored_messages = {}
|
||||
local display_levels = {}
|
||||
local enable_logging = true
|
||||
local message_limit = math.huge
|
||||
|
||||
-- USAGE:
|
||||
-- -- set arbitrary log levels to be printed immediately
|
||||
-- log{ info = true, warning = true, ducksauce = true, }
|
||||
-- -- disable storage of messages depending on log level
|
||||
-- log{ error = false, verbose = false, }
|
||||
-- -- send anything (except nil) to a log level
|
||||
-- log("bacom", true, "text", 5, function() end, {})
|
||||
-- -- print all messages saved at a particular log level
|
||||
@@ -14,13 +11,22 @@ local message_limit = math.huge
|
||||
-- log(false) -- this deletes all stored messages
|
||||
-- -- set a maximum number of stored messages per log level
|
||||
-- log(50) -- warning: reaching large limits makes logging expensive
|
||||
-- -- check if a log level is being displayed
|
||||
-- if log.arbitrary_exit then log("arbitrary_exit") os.exit(1) end
|
||||
-- -- do anything to the stored messages (this example deletes them all)
|
||||
-- log(function(messages) return {} end)
|
||||
|
||||
return function(options, ...)
|
||||
local microlog = {} -- display_levels are stored directly here
|
||||
local stored_messages = {}
|
||||
local enable_logging = true
|
||||
local message_limit = math.huge
|
||||
|
||||
return setmetatable(microlog, {
|
||||
__call = function(self, options, ...)
|
||||
local options_type = type(options)
|
||||
|
||||
if (options_type == "string") and enable_logging then
|
||||
if microlog[options] == false then return end
|
||||
if not stored_messages[options] then stored_messages[options] = {} end
|
||||
local message_table = stored_messages[options]
|
||||
|
||||
@@ -41,13 +47,13 @@ return function(options, ...)
|
||||
-- store message (and print if in display_levels)
|
||||
message_table[#message_table + 1] = current_message
|
||||
if #message_table > message_limit then table.remove(message_table, 1) end
|
||||
if display_levels[options] then print(current_message) end
|
||||
if microlog[options] then print(current_message) end
|
||||
end
|
||||
|
||||
-- set display_levels
|
||||
elseif options_type == "table" then
|
||||
for k,v in pairs(options) do
|
||||
display_levels[k] = v
|
||||
microlog[k] = v
|
||||
end
|
||||
|
||||
elseif options_type == "number" then
|
||||
@@ -72,4 +78,5 @@ return function(options, ...)
|
||||
local result = options(stored_messages)
|
||||
if type(result) == "table" then stored_messages = result end
|
||||
end
|
||||
end
|
||||
end
|
||||
})
|
||||
+59
-24
@@ -25,7 +25,7 @@ end
|
||||
log{
|
||||
info = true,
|
||||
warning = true,
|
||||
-- debug = true,
|
||||
debug = false,
|
||||
-- files = true, -- debugging why the wrong files are selected
|
||||
-- sha = true, -- what the fuck is going on with sha sums?
|
||||
}
|
||||
@@ -43,7 +43,7 @@ if not utility.path_exists(embeddings_file_path) then
|
||||
}, embeddings_file_path)
|
||||
end
|
||||
embeddings = utility.load_data(embeddings_file_path)
|
||||
local function embeddings_debug()
|
||||
if log.debug then
|
||||
log("debug", "Embeddings loaded.", embeddings, embeddings.files, embeddings.vectors)
|
||||
local file_count = 0
|
||||
for k,v in pairs(embeddings.files) do
|
||||
@@ -57,12 +57,11 @@ local function embeddings_debug()
|
||||
log("debug", vector_count .. " vectors.")
|
||||
-- os.exit(1)
|
||||
end
|
||||
embeddings_debug()
|
||||
|
||||
|
||||
|
||||
local refresh_file_list = function(source_name, data_source)
|
||||
timing.mark("Assembling file list for " .. source_name .. ".")
|
||||
timing.mark("Assembling file list for \"" .. source_name .. "\"")
|
||||
|
||||
local full_path = "PRIVATE_DATA" .. utility.path_separator .. data_source.path
|
||||
if data_source.initialize_command and (not config.models.initialized_sources[data_source.path]) then
|
||||
@@ -88,34 +87,26 @@ local refresh_file_list = function(source_name, data_source)
|
||||
file_list[#file_list + 1] = file_name
|
||||
end)
|
||||
|
||||
timing.mark("Finished assembling file list for " .. source_name .. ".")
|
||||
timing.mark("Finished assembling file list for \"" .. source_name .. "\"")
|
||||
return file_list
|
||||
end
|
||||
|
||||
-- returns nothing when too much text is sent
|
||||
local generate_embeddings = function(text)
|
||||
if #text > config.models.embedding.max_chunk_size then
|
||||
local generate_embeddings = function(data_source, text)
|
||||
local max_chunk_size = data_source.max_chunk_size or config.models.embedding.max_chunk_size
|
||||
local model = data_source.embedding_model or config.models.embedding.model
|
||||
|
||||
if #text > max_chunk_size then
|
||||
return nil, "generate_embeddings() must only be passed appropriately-sized chunks!"
|
||||
end
|
||||
|
||||
local result = utility.llm_prompt(text, config.models.embedding.model)
|
||||
local result = utility.llm_prompt(text, model)
|
||||
return json.decode(result)
|
||||
end
|
||||
|
||||
-- returns nothing for empty files and errors
|
||||
local process_file = function(data_source, file_name)
|
||||
local text = utility.read_file(file_name)
|
||||
local make_chunks = function(text, chunk_size)
|
||||
assert(chunk_size, "make_chunks() requires chunk_size")
|
||||
|
||||
if data_source.strip_frontmatter then
|
||||
text = text_processing.strip_frontmatter(text)
|
||||
end
|
||||
|
||||
if #text == 0 then
|
||||
log("empty", file_name .. "\n is empty and being skipped.")
|
||||
return
|
||||
end
|
||||
|
||||
local chunk_size = config.models.embedding.max_chunk_size
|
||||
local half_chunk_size = math.floor(chunk_size / 2)
|
||||
local chunks = { text }
|
||||
|
||||
@@ -133,17 +124,45 @@ local process_file = function(data_source, file_name)
|
||||
end
|
||||
end
|
||||
|
||||
return chunks
|
||||
end
|
||||
|
||||
-- returns nothing for empty files and errors
|
||||
local process_file = function(data_source, file_name)
|
||||
local text = utility.read_file(file_name)
|
||||
|
||||
if data_source.strip_frontmatter then
|
||||
text = text_processing.strip_frontmatter(text)
|
||||
end
|
||||
|
||||
if #text == 0 then
|
||||
log("empty", file_name .. "\n is empty and being skipped.")
|
||||
return
|
||||
end
|
||||
|
||||
local chunk_size = data_source.max_chunk_size or config.models.embedding.max_chunk_size
|
||||
local chunks = make_chunks(text, chunk_size)
|
||||
|
||||
if data_source.target_chunk_size then
|
||||
local extra_chunks = make_chunks(text, data_source.target_chunk_size)
|
||||
for i = 2, #extra_chunks do
|
||||
chunks[#chunks + 1] = extra_chunks[i]
|
||||
end
|
||||
end
|
||||
|
||||
local new_embeddings = {}
|
||||
for i = 1, #chunks do
|
||||
log("debug", "Embedding length:", #chunks[i])
|
||||
new_embeddings[i] = generate_embeddings(chunks[i]) or {}
|
||||
log("debug", "Embedding source length:", #chunks[i])
|
||||
new_embeddings[i] = generate_embeddings(data_source, chunks[i]) or {}
|
||||
end
|
||||
|
||||
if #new_embeddings[1] == 0 then
|
||||
if log.debug then
|
||||
log("debug", "Vector lengths:")
|
||||
for e = 1, #new_embeddings do
|
||||
log("debug", "", e, #new_embeddings[e])
|
||||
end
|
||||
end
|
||||
if #new_embeddings == 1 then
|
||||
-- Ollama very rarely errors with:
|
||||
-- Error: do embedding request: Post "http://127.0.0.1:53441/v1/embeddings": EOF
|
||||
@@ -198,6 +217,15 @@ local memorize_file = function(data_source, file_name)
|
||||
return file_sums
|
||||
end
|
||||
|
||||
local delete_memories = function(sha512sum_list)
|
||||
-- if true then return end -- TEMP disabling removal to compare and verify function
|
||||
for i = 1, #sha512sum_list do
|
||||
local sha512sum = sha512sum_list[i]
|
||||
embeddings.vectors[sha512sum] = nil
|
||||
os.execute("rm " .. (memory_path .. utility.path_separator .. sha512sum):enquote())
|
||||
end
|
||||
end
|
||||
|
||||
local refresh_sources = function()
|
||||
os.execute("mkdir -p " .. memory_path:enquote())
|
||||
|
||||
@@ -226,6 +254,7 @@ local refresh_sources = function()
|
||||
local function loop()
|
||||
local sha512sum = utility.sha512sum(file_name)
|
||||
log("sha", file_name, sha512sum, "\n Sum present? " .. tostring(embeddings.vectors[sha512sum]))
|
||||
|
||||
if embeddings.vectors[sha512sum] then
|
||||
log("debug", file_name .. "\n has already been embedded, skipping.")
|
||||
-- add file reference if it was missing
|
||||
@@ -235,7 +264,13 @@ local refresh_sources = function()
|
||||
return
|
||||
end
|
||||
|
||||
log("debug", "Generating embeddings for " .. file_name .. ".")
|
||||
if embeddings.files[file_name] then
|
||||
log("debug", file_name .. "\n Removing old embeddings for modified file.")
|
||||
delete_memories(embeddings.files[file_name])
|
||||
embeddings.files[file_name] = nil
|
||||
end
|
||||
|
||||
log("debug", file_name .. "\n Generating embeddings..")
|
||||
local file_sums = memorize_file(data_source, file_name)
|
||||
embeddings.files[file_name] = file_sums
|
||||
end
|
||||
|
||||
Reference in new issue
Block a user