Compare commits
16
Commits
c5aed984f8
..
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
58107e0196 | ||
|
|
da37b4ea29 | ||
|
|
dd9ca71fc7 | ||
|
|
77298dad0c | ||
|
|
8d0c79b34d | ||
|
|
23c2db1046 | ||
|
|
6381a49d13 | ||
|
|
f1b5fbc615 | ||
|
|
f229e7331e | ||
|
|
470901d9b9 | ||
|
|
977920df61 | ||
|
|
98112bde4a | ||
|
|
214f82b0b1 | ||
|
|
11d0e356a2 | ||
|
|
d0bd237174 | ||
|
|
8ff72681e9 |
No files matched your search
+1
-1
@@ -1,3 +1,3 @@
|
||||
.DS_Store
|
||||
PRIVATE_DATA/**
|
||||
PRIVATE_DATA/
|
||||
config.json
|
||||
@@ -19,19 +19,22 @@ Opens `similarities.json`, reverses the sort order, and saves it as
|
||||
|
||||
### `refresh_sources.lua`
|
||||
Creates/Maintains a store of chunked data with embeddings based on configurable
|
||||
sources. Example:
|
||||
`sources.json`. Example:
|
||||
|
||||
```json
|
||||
{
|
||||
"source name":{
|
||||
"embedding_model":"qwen3-embedding:0.6b",
|
||||
"filters":{
|
||||
"blacklist":".git",
|
||||
"extension_whitelist":"md"
|
||||
"blacklist":[".git"],
|
||||
"extension_whitelist":["md"]
|
||||
},
|
||||
"initialize_command":"git clone REMOTE .",
|
||||
"max_chunk_size":32768,
|
||||
"path":"will be created before initialize_command is run",
|
||||
"strip_frontmatter":true,
|
||||
"refresh_command":"git reset --hard origin/main"
|
||||
"target_chunk_size":3072,
|
||||
"refresh_command":"git fetch origin && git reset --hard origin/main"
|
||||
}
|
||||
}
|
||||
```
|
||||
@@ -41,6 +44,28 @@ The filters are based on `utility.tree`'s filter options (optional).
|
||||
while `refresh_command` is run each time (optional).
|
||||
Commands will be run in the specified `path`.
|
||||
`strip_frontmatter` will remove YAML frontmatter (common in Markdown files).
|
||||
`embedding_model` and `max_chunk_size` are self-explanatory (optional).
|
||||
`target_chunk_size` (optional) allows for finer-grained chunks to be generated
|
||||
alongside the maximum overviews for more precise retrieval.
|
||||
|
||||
The embeddings are stored like so:
|
||||
|
||||
```json
|
||||
{
|
||||
"files":{
|
||||
"PRIVATE_DATA/source_path/path/to/file.ext":["sha512sum", "another sum"]
|
||||
},
|
||||
"vectors":{
|
||||
"sha512sum":[0.5, 0, 1, -0.5, -1, ...]
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
SHA2 512-bit sums are used to link a text chunk with its embedding. All file
|
||||
names reference a list of text chunks so they can handle being too large. Every
|
||||
file that is too large for a single chunk has a whole-file embedding calculated
|
||||
first (it is the first element of the array), so that if a whole file becomes
|
||||
relevant, it can still show up instead of only chunks.
|
||||
|
||||
### `synopsis_generator.lua`
|
||||
Chooses a random file within `notebook`, and generates a novel synopsis from it.
|
||||
@@ -49,6 +74,28 @@ Arguments:
|
||||
- `refresh_file_list`: Refreshes the cached file list to choose from.
|
||||
- `export_ordered_list_of_prompts`: Makes an epub to review generated synopses.
|
||||
|
||||
## JSON config
|
||||
This repo uses my utility library's config system, using a `config.json` file in
|
||||
the repo root that is excluded from commits.
|
||||
|
||||
```json
|
||||
{
|
||||
"models":{
|
||||
"embedding":{
|
||||
"model":"qwen3-embedding:0.6b",
|
||||
"max_chunk_size":32768
|
||||
},
|
||||
"initialized_sources":{}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
`refresh_sources.lua` uses `models` to store default embedding model information
|
||||
and `initialized_sources` to store which sources have been initialized.
|
||||
|
||||
## Tasks
|
||||
- [ ] The whitelisting/blacklisting of tree should be in list too.
|
||||
- [ ] synopsis_generator should be able to blacklist files it already tried?
|
||||
- [ ] `refresh_sources.lua` doesn't check for defined files that don't exist
|
||||
anymore, does it?
|
||||
- [ ] I think it checks for every other possibility, but this needs checking.
|
||||
+82
@@ -0,0 +1,82 @@
|
||||
-- USAGE:
|
||||
-- -- set arbitrary log levels to be printed immediately
|
||||
-- log{ info = true, warning = true, ducksauce = true, }
|
||||
-- -- disable storage of messages depending on log level
|
||||
-- log{ error = false, verbose = false, }
|
||||
-- -- send anything (except nil) to a log level
|
||||
-- log("bacom", true, "text", 5, function() end, {})
|
||||
-- -- print all messages saved at a particular log level
|
||||
-- log("error")
|
||||
-- -- toggle ALL logging on and off with a boolean
|
||||
-- log(false) -- this deletes all stored messages
|
||||
-- -- set a maximum number of stored messages per log level
|
||||
-- log(50) -- warning: reaching large limits makes logging expensive
|
||||
-- -- check if a log level is being displayed
|
||||
-- if log.arbitrary_exit then log("arbitrary_exit") os.exit(1) end
|
||||
-- -- do anything to the stored messages (this example deletes them all)
|
||||
-- log(function(messages) return {} end)
|
||||
|
||||
local microlog = {} -- display_levels are stored directly here
|
||||
local stored_messages = {}
|
||||
local enable_logging = true
|
||||
local message_limit = math.huge
|
||||
|
||||
return setmetatable(microlog, {
|
||||
__call = function(self, options, ...)
|
||||
local options_type = type(options)
|
||||
|
||||
if (options_type == "string") and enable_logging then
|
||||
if microlog[options] == false then return end
|
||||
if not stored_messages[options] then stored_messages[options] = {} end
|
||||
local message_table = stored_messages[options]
|
||||
|
||||
-- turn log message into text
|
||||
local tab = {...}
|
||||
for i = 1, #tab do
|
||||
tab[i] = tostring(tab[i])
|
||||
end
|
||||
local current_message = table.concat(tab, "\t")
|
||||
|
||||
if #current_message == 0 then
|
||||
-- print all stored_messages of specified level
|
||||
for i = 1, #message_table do
|
||||
print(message_table[i])
|
||||
end
|
||||
|
||||
else
|
||||
-- store message (and print if in display_levels)
|
||||
message_table[#message_table + 1] = current_message
|
||||
if #message_table > message_limit then table.remove(message_table, 1) end
|
||||
if microlog[options] then print(current_message) end
|
||||
end
|
||||
|
||||
-- set display_levels
|
||||
elseif options_type == "table" then
|
||||
for k,v in pairs(options) do
|
||||
microlog[k] = v
|
||||
end
|
||||
|
||||
elseif options_type == "number" then
|
||||
message_limit = options
|
||||
for k,v in pairs(stored_messages) do
|
||||
while #v > message_limit do
|
||||
table.remove(v, 1)
|
||||
end
|
||||
end
|
||||
|
||||
-- turn all logging off and delete stored messages; or turn it on
|
||||
elseif options_type == "boolean" then
|
||||
if options then
|
||||
enable_logging = true
|
||||
else
|
||||
enable_logging = false
|
||||
stored_messages = {}
|
||||
end
|
||||
|
||||
-- arbitrary access :D
|
||||
elseif options_type == "function" then
|
||||
local result = options(stored_messages)
|
||||
if type(result) == "table" then stored_messages = result end
|
||||
end
|
||||
end
|
||||
})
|
||||
+23
-8
@@ -1,18 +1,26 @@
|
||||
local timing = {}
|
||||
|
||||
local function human_readable_time(delta)
|
||||
if delta >= 2*7*24*60*60 then -- if more than 2 weeks
|
||||
delta = tostring(math.floor(delta/(7*24*60*60))/10) .. " weeks"
|
||||
elseif delta >= 2*24*60*60 then -- if more than 2 days
|
||||
delta = tostring(math.floor(delta/(24*60*60))/10) .. " days"
|
||||
elseif delta >= 2*60*60 then -- if more than 2 hours
|
||||
delta = tostring(math.floor(delta/(60*60/10))/10) .. " hours"
|
||||
elseif delta >= 2*60 then -- if more then 2 minutes
|
||||
delta = tostring(math.floor(delta/(60/10))/10) .. " minutes"
|
||||
else
|
||||
delta = tostring(delta) .. " seconds"
|
||||
end
|
||||
return delta
|
||||
end
|
||||
|
||||
timing.display = function(n)
|
||||
local function _display(n)
|
||||
local time = timing[n]
|
||||
local previous = timing[n - 1]
|
||||
|
||||
local delta = time.time - previous.time
|
||||
if delta >= 2*60*60 then -- if more than 2 hours
|
||||
delta = tostring(math.floor(delta/(60*60/10))/10) .. " hours"
|
||||
elseif delta >= 120 then -- if more then 2 minutes
|
||||
delta = tostring(math.floor(delta/(60/10))/10) .. " minutes"
|
||||
else
|
||||
delta = tostring(delta) .. " seconds"
|
||||
end
|
||||
local delta = human_readable_time(time.time - previous.time)
|
||||
|
||||
print(delta, previous.label)
|
||||
end
|
||||
@@ -32,6 +40,13 @@ timing.mark = function(label)
|
||||
if #timing > 1 then
|
||||
timing.display(#timing)
|
||||
end
|
||||
print("", "", label)
|
||||
end
|
||||
|
||||
timing.estimate = function(current_position, total_operations)
|
||||
local delta = os.time() - timing[#timing].time
|
||||
local estimate = delta * total_operations / current_position - delta
|
||||
return human_readable_time(estimate)
|
||||
end
|
||||
|
||||
return timing
|
||||
+57
-18
@@ -81,6 +81,10 @@ standard_library_addition(string, "split", function(s, delimiter)
|
||||
return result
|
||||
end)
|
||||
|
||||
utility.leftpad = function(text, length, character)
|
||||
return string.rep(character or " ", length - #(tostring(text))) .. text
|
||||
end
|
||||
|
||||
|
||||
|
||||
utility.require = function(...)
|
||||
@@ -217,7 +221,7 @@ utility.list = function(path, func)
|
||||
|
||||
local run = function(fn)
|
||||
for line in output:gmatch("[^\r\n]+") do -- thanks to https://stackoverflow.com/a/32847589
|
||||
if not (line == "." or line == "..") then
|
||||
if not ((line == ".") or (line == "..")) then
|
||||
fn(line)
|
||||
end
|
||||
end
|
||||
@@ -235,7 +239,8 @@ utility.ls = function(...)
|
||||
return utility.list(...)
|
||||
end
|
||||
|
||||
utility.tree = function(path, options, fn)
|
||||
local tree
|
||||
tree = function(path, options, fn)
|
||||
if type(options) == "function" then
|
||||
fn = options
|
||||
options = {}
|
||||
@@ -245,16 +250,25 @@ utility.tree = function(path, options, fn)
|
||||
if options.blacklist and options.blacklist[path_name] then return end
|
||||
if options.whitelist and (not options.whitelist[path_name]) then return end
|
||||
|
||||
if options.extension_blacklist or options.extension_whitelist then
|
||||
local _, _, extension = utility.split_path_components(path_name)
|
||||
if options.extension_blacklist and options.extension_blacklist[extension] then return end
|
||||
if options.extension_whitelist and (not options.extension_whitelist[extension]) then return end
|
||||
end
|
||||
|
||||
if utility.is_file(path_name) then
|
||||
if options.extension_blacklist or options.extension_whitelist then
|
||||
local _, _, extension = utility.split_path_components(path_name)
|
||||
if options.extension_blacklist and options.extension_blacklist[extension] then return end
|
||||
if options.extension_whitelist and (not options.extension_whitelist[extension]) then return end
|
||||
end
|
||||
|
||||
fn(path_name)
|
||||
else
|
||||
utility.tree(path .. utility.path_separator .. path_name, options, fn)
|
||||
tree(path .. utility.path_separator .. path_name, options, fn)
|
||||
end
|
||||
end)
|
||||
end
|
||||
utility.tree = function(path, options, fn)
|
||||
tree(path, options, function(path_name)
|
||||
if path_name:find(path) == 1 then
|
||||
fn(path_name)
|
||||
else
|
||||
fn(path .. utility.path_separator .. path_name)
|
||||
end
|
||||
end)
|
||||
end
|
||||
@@ -265,10 +279,11 @@ utility.read_file = function(file_name)
|
||||
end)
|
||||
end
|
||||
|
||||
utility.write_file = function(file_name, text)
|
||||
utility.write_file = function(file_name, ...)
|
||||
local text = table.concat{...}
|
||||
return utility.open(file_name, "w", function(file)
|
||||
file:write(text)
|
||||
file:write("\n")
|
||||
-- file:write("\n") -- I need to make sure /I/ handle this instead of trying to automate it
|
||||
end)
|
||||
end
|
||||
|
||||
@@ -301,10 +316,8 @@ end
|
||||
|
||||
utility.sha512sum = function(file_path)
|
||||
local sha512sum
|
||||
if utility.OS == "Linux" then
|
||||
sha512sum = os.capture_safe("shasum -p -t -a 512 " .. file_path:enquote()) -- TODO check this
|
||||
elseif utility.OS == "macOS" then
|
||||
sha512sum = os.capture_safe("shasum -U -a 512 " .. file_path:enquote())
|
||||
if (utility.OS == "Linux") or (utility.OS == "macOS") then
|
||||
sha512sum = utility.capture_safe("shasum -U -a 512 " .. file_path:enquote())
|
||||
elseif utility.OS == "Windows" then
|
||||
error("utility.sha512sum() not implemented for Windows.")
|
||||
end
|
||||
@@ -362,10 +375,10 @@ end
|
||||
|
||||
|
||||
|
||||
local config_path = utility.path .. "config.json"
|
||||
local config, config_lock
|
||||
utility.get_config = function(skip_lock)
|
||||
if not config then
|
||||
local config_path = utility.path .. "config.json"
|
||||
if utility.is_file(config_path) then
|
||||
if not skip_lock then
|
||||
config_lock = utility.get_lock(config_path)
|
||||
@@ -383,7 +396,6 @@ end
|
||||
|
||||
utility.save_config = function()
|
||||
if config then
|
||||
local config_path = utility.path .. "config.json"
|
||||
if not config_lock then
|
||||
print("Warning: A config lock file was not established.")
|
||||
end
|
||||
@@ -399,6 +411,31 @@ utility.save_config = function()
|
||||
end
|
||||
end
|
||||
|
||||
utility.get_config_with_defaults = function(defaults)
|
||||
local config = utility.get_config()
|
||||
|
||||
local loop, changes_made
|
||||
loop = function(config_level, default_level)
|
||||
for k,v in pairs(default_level) do
|
||||
if not config_level[k] then
|
||||
config_level[k] = v
|
||||
changes_made = true
|
||||
elseif type(v) == "table" then
|
||||
loop(config_level[k], v)
|
||||
end
|
||||
end
|
||||
end
|
||||
loop(config, defaults)
|
||||
|
||||
if changes_made then
|
||||
utility.save_config()
|
||||
else
|
||||
utility.release_lock(config_path)
|
||||
end
|
||||
|
||||
return config
|
||||
end
|
||||
|
||||
|
||||
|
||||
local data_file_locations = {}
|
||||
@@ -582,6 +619,8 @@ end
|
||||
|
||||
|
||||
|
||||
if utility.capture_safe("uname"):find("Darwin") == 1 then utility.OS = "macOS" end
|
||||
if (utility.OS == "Linux") and (utility.capture_safe("uname"):find("Darwin") == 1) then
|
||||
utility.OS = "macOS"
|
||||
end
|
||||
|
||||
return utility
|
||||
Executable
+19
@@ -0,0 +1,19 @@
|
||||
#!/usr/bin/env luajit
|
||||
|
||||
package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path
|
||||
local utility = require "utility"
|
||||
local json = utility.require("dkjson")
|
||||
|
||||
local search_path = arg[1]
|
||||
|
||||
local file_extensions = {}
|
||||
utility.tree(search_path, function(file_path)
|
||||
local _, _, extension = utility.split_path_components(file_path)
|
||||
if extension then
|
||||
file_extensions[extension] = true
|
||||
end
|
||||
end)
|
||||
|
||||
for extension in pairs(file_extensions) do
|
||||
print(extension)
|
||||
end
|
||||
Regular → Executable
+178
-64
@@ -4,38 +4,60 @@ package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" ..
|
||||
local utility = require "utility"
|
||||
|
||||
local json = utility.require("dkjson")
|
||||
local log = utility.require("log")
|
||||
local text_processing = utility.require("text_processing")
|
||||
local timing = utility.require("timing")
|
||||
|
||||
-- TODO make utility have a function for getting/setting defaults where locking is only used to set defaults if they aren't present
|
||||
-- TODO there should be a check function for a lock so a warning/error can be dumped?
|
||||
local config = utility.get_config("no-lock")
|
||||
if not config.models then
|
||||
config.models = {
|
||||
local config = utility.get_config_with_defaults{
|
||||
models = {
|
||||
embedding = {
|
||||
model = "qwen3-embedding:0.6b",
|
||||
max_chunk_size = 32768,
|
||||
},
|
||||
initialized_sources = {},
|
||||
}
|
||||
utility.save_config()
|
||||
end
|
||||
},
|
||||
}
|
||||
|
||||
log{
|
||||
info = true,
|
||||
warning = true,
|
||||
debug = false,
|
||||
-- files = true, -- debugging why the wrong files are selected
|
||||
-- sha = true, -- what the fuck is going on with sha sums?
|
||||
}
|
||||
|
||||
local memory_path = "PRIVATE_DATA" .. utility.path_separator .. "memory"
|
||||
local embeddings_file_path = memory_path .. utility.path_separator .. "+embeddings.json"
|
||||
local tmp_file_path = "PRIVATE_DATA" .. utility.path_separator .. ".tmp.2b65c19b-0883-49ca-8247-b1fe7760f922"
|
||||
|
||||
local embeddings
|
||||
local embeddings_file_path = "PRIVATE_DATA/memory/+embeddings.json"
|
||||
if not utility.path_exists(embeddings_file_path) then
|
||||
os.execute("mkdir -p PRIVATE_DATA/memory")
|
||||
os.execute("mkdir -p " .. memory_path:enquote())
|
||||
utility.save_data({
|
||||
files = {},
|
||||
vectors = {},
|
||||
}, embeddings_file_path)
|
||||
end
|
||||
embeddings = utility.load_data(embeddings_file_path)
|
||||
local embeddings = utility.load_data(embeddings_file_path)
|
||||
if log.debug then
|
||||
log("debug", "Embeddings loaded.", embeddings, embeddings.files, embeddings.vectors)
|
||||
local file_count = 0
|
||||
for k,v in pairs(embeddings.files) do
|
||||
file_count = file_count + 1
|
||||
end
|
||||
local vector_count = 0
|
||||
for k,v in pairs(embeddings.vectors) do
|
||||
vector_count = vector_count + 1
|
||||
end
|
||||
log("debug", file_count .. " files.")
|
||||
log("debug", vector_count .. " vectors.")
|
||||
-- os.exit(1)
|
||||
end
|
||||
|
||||
|
||||
|
||||
local refresh_file_list = function(source_name, data_source)
|
||||
timing.mark("Assembling file list for " .. source_name .. ".")
|
||||
timing.mark("Assembling file list for \"" .. source_name .. "\"")
|
||||
|
||||
local full_path = "PRIVATE_DATA" .. utility.path_separator .. data_source.path
|
||||
if data_source.initialize_command and (not config.models.initialized_sources[data_source.path]) then
|
||||
@@ -56,30 +78,52 @@ local refresh_file_list = function(source_name, data_source)
|
||||
end
|
||||
|
||||
local file_list = {}
|
||||
utility.tree(data_source.path, compiled_filters, function(file_name)
|
||||
utility.tree("PRIVATE_DATA" .. utility.path_separator .. data_source.path, compiled_filters, function(file_name)
|
||||
log("files", file_name)
|
||||
file_list[#file_list + 1] = file_name
|
||||
end)
|
||||
|
||||
timing.mark("Finished assembling file list for " .. source_name .. ".")
|
||||
timing.mark("Finished assembling file list for \"" .. source_name .. "\"")
|
||||
return file_list
|
||||
end
|
||||
|
||||
-- returns nothing when too much text is sent
|
||||
local generate_embeddings = function(text)
|
||||
if #text > config.models.embedding.max_chunk_size then
|
||||
local generate_embeddings = function(data_source, text)
|
||||
local max_chunk_size = data_source.max_chunk_size or config.models.embedding.max_chunk_size
|
||||
local model = data_source.embedding_model or config.models.embedding.model
|
||||
|
||||
if #text > max_chunk_size then
|
||||
return nil, "generate_embeddings() must only be passed appropriately-sized chunks!"
|
||||
end
|
||||
|
||||
local result = utility.llm_prompt(text, config.models.embedding.model)
|
||||
|
||||
result = setmetatable({}, {
|
||||
__tojson = function() return result end,
|
||||
})
|
||||
|
||||
local result = utility.llm_prompt(text, model)
|
||||
return json.decode(result)
|
||||
end
|
||||
|
||||
-- returns nothing for empty files
|
||||
local make_chunks = function(text, chunk_size)
|
||||
assert(chunk_size, "make_chunks() requires chunk_size")
|
||||
|
||||
local half_chunk_size = math.floor(chunk_size / 2)
|
||||
local chunks = { text }
|
||||
|
||||
while #text > chunk_size do
|
||||
local first_chunk = text:sub(1, chunk_size)
|
||||
local overlap_chunk = text:sub(half_chunk_size, chunk_size + half_chunk_size - 1)
|
||||
|
||||
chunks[#chunks + 1] = first_chunk
|
||||
chunks[#chunks + 1] = overlap_chunk
|
||||
|
||||
text = text:sub(chunk_size)
|
||||
if (#text > half_chunk_size) and (not (#text > chunk_size)) then
|
||||
-- last chunk would be skipped if we didn't handle this here
|
||||
chunks[#chunks + 1] = text
|
||||
end
|
||||
end
|
||||
|
||||
return chunks
|
||||
end
|
||||
|
||||
-- returns nothing for empty files and errors
|
||||
local process_file = function(data_source, file_name)
|
||||
local text = utility.read_file(file_name)
|
||||
|
||||
@@ -88,34 +132,41 @@ local process_file = function(data_source, file_name)
|
||||
end
|
||||
|
||||
if #text == 0 then
|
||||
print(file_name .. "\n is empty and being skipped.")
|
||||
log("empty", file_name .. "\n is empty and being skipped.")
|
||||
return
|
||||
end
|
||||
|
||||
local chunk_size = config.models.embedding.max_chunk_size
|
||||
local half_chunk_size = math.floor(chunk_size / 2)
|
||||
local chunks = { text }
|
||||
local chunk_size = data_source.max_chunk_size or config.models.embedding.max_chunk_size
|
||||
local chunks = make_chunks(text, chunk_size)
|
||||
|
||||
while #text > chunk_size do
|
||||
local first_chunk = text:sub(1, chunk_size)
|
||||
local overlap_chunk = text:sub(half_chunk_size, chunk_size + half_chunk_size)
|
||||
|
||||
chunks[#chunks + 1] = first_chunk
|
||||
chunks[#chunks + 1] = overlap_chunk
|
||||
|
||||
text = text:sub(chunk_size)
|
||||
if #text > half_chunk_size and (not #text > chunk_size) then
|
||||
-- last chunk would be skipped if we didn't handle this here
|
||||
chunks[#chunks + 1] = text
|
||||
if data_source.target_chunk_size then
|
||||
local extra_chunks = make_chunks(text, data_source.target_chunk_size)
|
||||
for i = 2, #extra_chunks do
|
||||
chunks[#chunks + 1] = extra_chunks[i]
|
||||
end
|
||||
end
|
||||
|
||||
local new_embeddings = {}
|
||||
for i = 1, #chunks do
|
||||
new_embeddings[i] = generate_embeddings(chunks[i]) or {}
|
||||
log("debug", "Embedding source length:", #chunks[i])
|
||||
new_embeddings[i] = generate_embeddings(data_source, chunks[i]) or {}
|
||||
end
|
||||
|
||||
if #new_embeddings[1] == 0 then
|
||||
if log.debug then
|
||||
log("debug", "Vector lengths:")
|
||||
for e = 1, #new_embeddings do
|
||||
log("debug", "", e, #new_embeddings[e])
|
||||
end
|
||||
end
|
||||
if #new_embeddings == 1 then
|
||||
-- Ollama very rarely errors with:
|
||||
-- Error: do embedding request: Post "http://127.0.0.1:53441/v1/embeddings": EOF
|
||||
-- but it is inconsistent and re-running will eventually fix it.
|
||||
log("warning", file_name .. "\n encountered an embedding error and will be skipped this run only.")
|
||||
return
|
||||
end
|
||||
|
||||
-- average all embeddings to make the core file embedding
|
||||
local count = #new_embeddings[2]
|
||||
for vector_index = 1, count do
|
||||
@@ -130,50 +181,113 @@ local process_file = function(data_source, file_name)
|
||||
return chunks, new_embeddings
|
||||
end
|
||||
|
||||
-- returns nothing for errors
|
||||
local memorize_file = function(data_source, file_name)
|
||||
local file_chunks, file_embeddings = process_file(data_source, file_name)
|
||||
if not file_chunks then return end
|
||||
|
||||
local file_sums = {}
|
||||
for i = 1, #file_chunks do
|
||||
local function loop()
|
||||
local text = file_chunks[i]
|
||||
local current_embedding = file_embeddings[i]
|
||||
|
||||
utility.write_file(tmp_file_path, text)
|
||||
|
||||
local sha512sum = utility.sha512sum(tmp_file_path)
|
||||
log("sha", "New sum? " .. sha512sum, tostring(embeddings.vectors[sha512sum]))
|
||||
file_sums[#file_sums + 1] = sha512sum
|
||||
if embeddings.vectors[sha512sum] then
|
||||
log("sha", "Detected previously extant sum, skipping.")
|
||||
return
|
||||
end
|
||||
|
||||
os.execute(utility.commands.move .. tmp_file_path:enquote()
|
||||
.. " " .. (memory_path .. utility.path_separator .. sha512sum):enquote())
|
||||
embeddings.vectors[sha512sum] = current_embedding
|
||||
log("sha", "Saved new sum!")
|
||||
end
|
||||
loop()
|
||||
end
|
||||
|
||||
return file_sums
|
||||
end
|
||||
|
||||
local delete_memories = function(sha512sum_list)
|
||||
-- if true then return end -- TEMP disabling removal to compare and verify function
|
||||
for i = 1, #sha512sum_list do
|
||||
local sha512sum = sha512sum_list[i]
|
||||
embeddings.vectors[sha512sum] = nil
|
||||
os.execute("rm " .. (memory_path .. utility.path_separator .. sha512sum):enquote())
|
||||
end
|
||||
end
|
||||
|
||||
local refresh_sources = function()
|
||||
local tmp_file_path = "PRIVATE_DATA/.tmp.2b65c19b-0883-49ca-8247-b1fe7760f922"
|
||||
os.execute("mkdir -p PRIVATE_DATA/memory")
|
||||
os.execute("mkdir -p " .. memory_path:enquote())
|
||||
|
||||
timing.mark("Checking memory for missing embeddings.")
|
||||
utility.list(memory_path, function(file_name)
|
||||
log("files", file_name)
|
||||
if (file_name == ".DS_Store") or (file_name == "+embeddings.json") then return end
|
||||
|
||||
local file_path = memory_path .. utility.path_separator .. file_name
|
||||
local sha512sum = utility.sha512sum(file_path)
|
||||
log("sha", "New sum? " .. sha512sum, tostring(embeddings.vectors[sha512sum]))
|
||||
if not embeddings.vectors[sha512sum] then
|
||||
log("sha", "Saved new sum!")
|
||||
memorize_file({}, file_path)
|
||||
end
|
||||
end)
|
||||
timing.mark("Finished checking memory for missing embeddings.")
|
||||
|
||||
local sources = utility.load_data("PRIVATE_DATA/sources.json")
|
||||
for source_name, data_source in pairs(sources) do
|
||||
local file_list = refresh_file_list(source_name, data_source)
|
||||
|
||||
timing.mark("Generating embeddings for " .. source_name .. ".")
|
||||
for _, file_name in ipairs(file_list) do
|
||||
timing.mark("Generating embeddings from source \"" .. source_name .. "\"")
|
||||
for f = 1, #file_list do
|
||||
local file_name = file_list[f]
|
||||
local function loop()
|
||||
local sha512sum = utility.sha512sum(tmp_file_path)
|
||||
if embeddings.vectors[sha512sum] then return end
|
||||
local sha512sum = utility.sha512sum(file_name)
|
||||
log("sha", file_name, sha512sum, "\n Sum present? " .. tostring(embeddings.vectors[sha512sum]))
|
||||
|
||||
local file_chunks, file_embeddings = process_file(data_source, file_name)
|
||||
|
||||
local file_sums = {}
|
||||
for i = 1, #file_chunks do
|
||||
local function loop()
|
||||
local text = file_chunks[i]
|
||||
local current_embedding = file_embeddings[i]
|
||||
|
||||
utility.write_file(tmp_file_path, text)
|
||||
|
||||
sha512sum = utility.sha512sum(tmp_file_path)
|
||||
file_sums[#file_sums + 1] = sha512sum
|
||||
if embeddings.vectors[sha512sum] then return end
|
||||
|
||||
os.execute(utility.commands.move .. tmp_file_path:enquote() .. " " .. "PRIVATE_DATA/memory/")
|
||||
embeddings.vectors[sha512sum] = current_embedding
|
||||
if embeddings.vectors[sha512sum] then
|
||||
log("debug", file_name .. "\n has already been embedded, skipping.")
|
||||
-- add file reference if it was missing (also handles recognition of duplicate files)
|
||||
if not embeddings.files[file_name] then
|
||||
embeddings.files[file_name] = { sha512sum }
|
||||
end
|
||||
loop()
|
||||
return
|
||||
end
|
||||
|
||||
if embeddings.files[file_name] then
|
||||
log("debug", file_name .. "\n Removing old embeddings for modified file.")
|
||||
delete_memories(embeddings.files[file_name])
|
||||
embeddings.files[file_name] = nil
|
||||
end
|
||||
|
||||
log("debug", file_name .. "\n Generating embeddings..")
|
||||
local file_sums = memorize_file(data_source, file_name)
|
||||
embeddings.files[file_name] = file_sums
|
||||
end
|
||||
loop()
|
||||
log("info", "Finished " .. utility.leftpad(f, #tostring(#file_list), "0")
|
||||
.. "/" .. #file_list
|
||||
.. " (" .. utility.leftpad(math.floor(f / #file_list * 100), 3, "0")
|
||||
.. "%) ETA: " .. timing.estimate(f, #file_list))
|
||||
end
|
||||
|
||||
timing.mark("Finished generating embeddings for " .. source_name .. ".")
|
||||
timing.mark("Finished generating embeddings from source \"" .. source_name .. "\"")
|
||||
end
|
||||
|
||||
utility.save_data(embeddings)
|
||||
os.execute("rm " .. tmp_file_path:enquote())
|
||||
if utility.path_exists(tmp_file_path) then
|
||||
os.execute("rm " .. tmp_file_path:enquote())
|
||||
end
|
||||
end
|
||||
|
||||
refresh_sources()
|
||||
timing.mark("Finished.")
|
||||
print("")
|
||||
timing.display()
|
||||
log("warning")
|
||||
+53
-49
@@ -7,49 +7,49 @@ local json = utility.require("dkjson")
|
||||
local prompts = utility.require("prompts")
|
||||
local text_processing = utility.require("text_processing")
|
||||
|
||||
local model = "gemma4:12b-mlx"
|
||||
local minimum_bytes = 1000
|
||||
local maximum_bytes = 40000
|
||||
local config = utility.get_config_with_defaults{
|
||||
models = {
|
||||
synopsis_generator = {
|
||||
model = "gemma4:12b-mlx",
|
||||
minimum_bytes = 1000,
|
||||
maximum_bytes = 40000,
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
local NOTEBOOK_PATH = "PRIVATE_DATA/notebook"
|
||||
local blacklist = utility.enumerate{ ".git", ".gitattributes", ".gitignore", ".gitkeep", ".DS_Store", }
|
||||
local extension_blacklist = utility.enumerate{ "gif", "jpg", "jpeg", "mp4", "pdf", "png", "webp", }
|
||||
|
||||
local data_location = "PRIVATE_DATA/synopsis_generator_file_list.json"
|
||||
local file_list
|
||||
if utility.path_exists(data_location) then
|
||||
file_list = utility.load_data(data_location)
|
||||
else
|
||||
file_list = {}
|
||||
end
|
||||
local memory_path = "PRIVATE_DATA" .. utility.path_separator .. "memory"
|
||||
local embeddings_file_path = memory_path .. utility.path_separator .. "+embeddings.json"
|
||||
local data_path = "PRIVATE_DATA" .. utility.path_separator .. "synopses"
|
||||
local data_location = data_path .. utility.path_separator .. "+file_list.json"
|
||||
|
||||
local refresh_file_list = function()
|
||||
os.execute("cd " .. NOTEBOOK_PATH:enquot() .. " && git pull origin")
|
||||
assert(utility.path_exists(embeddings_file_path), "Run \"./refresh_sources.lua\" to set up memory.")
|
||||
local embeddings = utility.load_data(embeddings_file_path)
|
||||
|
||||
local new_files_list = {}
|
||||
utility.tree(NOTEBOOK_PATH, {
|
||||
blacklist = blacklist,
|
||||
extension_blacklist = extension_blacklist,
|
||||
}, function(file_name)
|
||||
local minimum_bytes = config.models.synopsis_generator.minimum_bytes
|
||||
local maximum_bytes = config.models.synopsis_generator.maximum_bytes
|
||||
|
||||
local file_list = {}
|
||||
for file_name, sha512sum_list in pairs(embeddings.files) do
|
||||
local file_size = utility.file_size(file_name)
|
||||
if file_size >= minimum_bytes and file_size <= maximum_bytes then
|
||||
local text = text_processing.strip_frontmatter(utility.read_file(file_name))
|
||||
if #text >= minimum_bytes and #text <= maximum_bytes then
|
||||
new_files_list[#new_files_list + 1] = file_name
|
||||
file_list[#file_list + 1] = { file_name = file_name, }
|
||||
end
|
||||
end
|
||||
end)
|
||||
end
|
||||
|
||||
file_list = new_files_list
|
||||
utility.save_data(new_files_list, data_location)
|
||||
utility.save_data(file_list, data_location)
|
||||
return file_list
|
||||
end
|
||||
|
||||
local generate_and_score = function(file_name, text)
|
||||
print("Writing synopsis...")
|
||||
local synopsis = utility.llm_prompt(prompts.synopsis_prompt .. text, model)
|
||||
local synopsis = utility.llm_prompt(prompts.synopsis_prompt .. text, config.models.synopsis_generator.model)
|
||||
print(synopsis)
|
||||
print("Scoring synopsis...")
|
||||
local scoring, thinking = utility.llm_prompt(prompts.scoring_prompt .. synopsis, model)
|
||||
local scoring, thinking = utility.llm_prompt(prompts.scoring_prompt .. synopsis, config.models.synopsis_generator.model)
|
||||
print(scoring)
|
||||
local scoring_decoded = json.decode(scoring) -- likely will not work because it consistently returns Markdown instead of JSON
|
||||
if not scoring_decoded then
|
||||
@@ -62,23 +62,22 @@ local generate_and_score = function(file_name, text)
|
||||
thinking = thinking,
|
||||
}
|
||||
|
||||
utility.save_data(object, "PRIVATE_DATA/synopses/" .. utility.uuid() .. ".json")
|
||||
utility.save_data(object, data_path .. utility.path_separator .. utility.uuid() .. ".json")
|
||||
end
|
||||
|
||||
local export_ordered_list_of_prompts = function()
|
||||
local path = "PRIVATE_DATA/synopses"
|
||||
local items = {}
|
||||
local item_order = {}
|
||||
|
||||
utility.list(path, function(path_name)
|
||||
local full_path = path .. utility.path_separator .. path_name
|
||||
if path_name:find("%.json") then
|
||||
local object = utility.load_data(full_path)
|
||||
items[path_name] = object
|
||||
if type(object.scoring) == "table" then
|
||||
local mean_score, total_score = utility.mean(object.scoring)
|
||||
item_order[#item_order + 1] = { path_name = path_name, total_score = total_score, mean_score = mean_score, }
|
||||
end
|
||||
utility.list(data_path, function(path_name)
|
||||
local full_path = data_path .. utility.path_separator .. path_name
|
||||
if full_path == data_location then return end
|
||||
|
||||
local object = utility.load_data(full_path)
|
||||
items[path_name] = object
|
||||
if type(object.scoring) == "table" then
|
||||
local mean_score, total_score = utility.mean(object.scoring)
|
||||
item_order[#item_order + 1] = { path_name = path_name, total_score = total_score, mean_score = mean_score, }
|
||||
end
|
||||
end)
|
||||
|
||||
@@ -87,7 +86,7 @@ local export_ordered_list_of_prompts = function()
|
||||
local output = {
|
||||
"---",
|
||||
"title: Ordered Synopses (" .. #item_order .. " items)",
|
||||
"author: [\"" .. model .. "\", \"Tangent\", \"Ollama\"]",
|
||||
"author: [\"" .. config.models.synopsis_generator.model .. "\", \"Tangent\", \"Ollama\"]",
|
||||
"publisher: \"synopsis_generator.lua\"",
|
||||
"---",
|
||||
"",
|
||||
@@ -99,8 +98,11 @@ local export_ordered_list_of_prompts = function()
|
||||
local tab = text:split("\n")
|
||||
|
||||
for index, line in ipairs(tab) do
|
||||
if line:sub(1, 1) == "#" then
|
||||
tab[index] = "#" .. tab[index]
|
||||
-- ensure no output heading is an H1; make H6 bold text lines instead
|
||||
if line:sub(1, 7) == "###### " then
|
||||
tab[index] = "**" .. line:sub(8) .. "**"
|
||||
elseif line:sub(1, 1) == "#" then
|
||||
tab[index] = "#" .. line
|
||||
end
|
||||
end
|
||||
|
||||
@@ -108,31 +110,33 @@ local export_ordered_list_of_prompts = function()
|
||||
output[#output + 1] = "## Scoring\n\n```json\n" .. json.encode(item.scoring, { indent = true, }) .. "\n```\n"
|
||||
end
|
||||
|
||||
utility.write_file("PRIVATE_DATA/Ordered Synopses.md", table.concat(output, "\n"))
|
||||
utility.write_file("PRIVATE_DATA/Ordered Synopses.md", table.concat(output, "\n"), "\n")
|
||||
os.execute("pandoc \"PRIVATE_DATA/Ordered Synopses.md\" -o \"PRIVATE_DATA/Ordered Synopses.epub\"")
|
||||
end
|
||||
|
||||
|
||||
|
||||
os.execute("mkdir -p PRIVATE_DATA/synopses")
|
||||
os.execute("mkdir -p " .. data_path:enquote())
|
||||
|
||||
local file_list
|
||||
if utility.path_exists(data_location) then
|
||||
file_list = utility.load_data(data_location)
|
||||
end
|
||||
|
||||
if arg[1] == "refresh_file_list" then
|
||||
print("Refresing file list...")
|
||||
refresh_file_list()
|
||||
file_list = refresh_file_list()
|
||||
os.exit(0)
|
||||
elseif arg[1] == "export_ordered_list_of_prompts" then
|
||||
export_ordered_list_of_prompts()
|
||||
os.exit(0)
|
||||
end
|
||||
|
||||
print(#file_list .. " files to select from.")
|
||||
if #file_list == 0 then
|
||||
print("Run \"./synopsis_generator.lua refresh_file_list\" first.")
|
||||
os.exit(1)
|
||||
end
|
||||
assert(file_list and (#file_list > 0), "Run \"./synopsis_generator.lua refresh_file_list\" first.")
|
||||
|
||||
print(#file_list .. " files to select from.")
|
||||
while true do
|
||||
local file_name = file_list[math.random(1, #file_list)]
|
||||
local file_name = file_list[math.random(1, #file_list)].file_name
|
||||
local text = utility.read_file(file_name)
|
||||
print(file_name .. " chosen.")
|
||||
generate_and_score(file_name, text)
|
||||
|
||||
Reference in new issue
Block a user