Compare commits

...
28 Commits
Author SHA1 Message Date
tangent 58107e0196 fix formatting output synopses_generator 2026-09-17 03:39:13 -06:00
tangent da37b4ea29 synopsis_generator.lua updated, mostly 2026-09-17 03:08:11 -06:00
tangent dd9ca71fc7 config default set utility 2026-09-17 01:56:35 -06:00
tangent 77298dad0c target chunk size is possible now
closes #4
2026-09-16 23:16:29 -06:00
tangent 8d0c79b34d heading towards targeted chunk sizes, made sure updated files have old copies cleaned up 2026-09-16 22:16:38 -06:00
tangent 23c2db1046 better logging 2026-09-16 21:35:34 -06:00
tangent 6381a49d13 Merge pull request 'Refresh sources' (#2) from refresh_sources into main
Reviewed-on: #2
2026-09-16 18:01:27 -06:00
tangent f1b5fbc615 Lua got confused by varargs 2026-09-16 17:59:24 -06:00
tangent f229e7331e finally found my stupid trip-up, writing modified files 2026-09-16 17:56:47 -06:00
tangent 470901d9b9 very WIP, doesn't detect duplicates properly.. everything else seems fine? 2026-09-16 16:03:14 -06:00
tangent 977920df61 fix #9 shasum error 2026-09-16 12:48:43 -06:00
tangent 98112bde4a close #6 Darwin check only runs on 'Linux' detection
This prevents Windows from erroring when trying to run uname.
2026-09-15 22:30:46 -06:00
tangent 214f82b0b1 remove debug printing, workaround for EOF Ollama error #7 2026-09-15 22:28:53 -06:00
tangent 11d0e356a2 debug lines, timing improved/corrected 2026-09-15 19:11:38 -06:00
tangent d0bd237174 fixed all the errors 2026-09-15 13:16:57 -06:00
tangent 8ff72681e9 final bits 2026-09-15 11:16:12 -06:00
tangent c5aed984f8 should be ready to go 2026-09-15 11:02:54 -06:00
tangent 1e79c9ed99 macOS is weird 2026-09-15 09:27:46 -06:00
tangent 1b7c211aa2 Update test.lua 2026-09-15 09:25:04 -06:00
tangent 425cae4c20 Update .gitignore 2026-09-15 09:16:12 -06:00
tangent 915319ee1c Delete PRIVATE_DATA/.gitkeep 2026-09-15 09:16:00 -06:00
tangent 5b0520a7ce Update ReadMe.md 2026-09-15 08:37:03 -06:00
tangent bfe93e33f9 Update refresh_sources.lua 2026-09-15 08:36:19 -06:00
tangent dde5283aca Update ReadMe.md 2026-09-15 08:22:41 -06:00
tangent 012d3abe5a Update ReadMe.md 2026-09-15 08:22:02 -06:00
tangent 7b4608ec7e Update refresh_sources.lua 2026-09-15 08:14:35 -06:00
tangent 692d4f5690 Update .gitignore 2026-09-15 07:39:28 -06:00
tangent 20015af69c Add refresh_sources.lua
This file needs to be chmodded.
2026-09-15 07:39:15 -06:00
10 changed files with 619 additions and 121 deletions

No files matched your search

+2 -2
View File
@@ -1,3 +1,3 @@
.DS_Store
PRIVATE_DATA/**
!PRIVATE_DATA/.gitkeep
PRIVATE_DATA/
config.json
View File
Whitespace-only changes.
+76 -4
View File
@@ -4,26 +4,98 @@ Finding useful things to do with local LLMs.
All scripts work with data within `PRIVATE_DATA/` so that private data can't be
accidentally committed.
###### `cosine_similarity.lua`
### `cosine_similarity.lua`
Opens `embeddings.json` and creates `similarities.json` with a sorted list of
comparisons.
###### `generate_embeddings.lua`
### `generate_embeddings.lua`
Opens every file on its whitelist within the `notebook` directory, and generates
embeddings (placed in `embeddings.json`). When files are too long, it truncates
them.
###### `least_similar.lua`
### `least_similar.lua`
Opens `similarities.json`, reverses the sort order, and saves it as
`differences.json`.
###### `synopsis_generator.lua`
### `refresh_sources.lua`
Creates/Maintains a store of chunked data with embeddings based on configurable
`sources.json`. Example:
```json
{
"source name":{
"embedding_model":"qwen3-embedding:0.6b",
"filters":{
"blacklist":[".git"],
"extension_whitelist":["md"]
},
"initialize_command":"git clone REMOTE .",
"max_chunk_size":32768,
"path":"will be created before initialize_command is run",
"strip_frontmatter":true,
"target_chunk_size":3072,
"refresh_command":"git fetch origin && git reset --hard origin/main"
}
}
```
The filters are based on `utility.tree`'s filter options (optional).
`initialize_command` is only run the first time (optional),
while `refresh_command` is run each time (optional).
Commands will be run in the specified `path`.
`strip_frontmatter` will remove YAML frontmatter (common in Markdown files).
`embedding_model` and `max_chunk_size` are self-explanatory (optional).
`target_chunk_size` (optional) allows for finer-grained chunks to be generated
alongside the maximum overviews for more precise retrieval.
The embeddings are stored like so:
```json
{
"files":{
"PRIVATE_DATA/source_path/path/to/file.ext":["sha512sum", "another sum"]
},
"vectors":{
"sha512sum":[0.5, 0, 1, -0.5, -1, ...]
}
}
```
SHA2 512-bit sums are used to link a text chunk with its embedding. All file
names reference a list of text chunks so they can handle being too large. Every
file that is too large for a single chunk has a whole-file embedding calculated
first (it is the first element of the array), so that if a whole file becomes
relevant, it can still show up instead of only chunks.
### `synopsis_generator.lua`
Chooses a random file within `notebook`, and generates a novel synopsis from it.
Arguments:
- `refresh_file_list`: Refreshes the cached file list to choose from.
- `export_ordered_list_of_prompts`: Makes an epub to review generated synopses.
## JSON config
This repo uses my utility library's config system, using a `config.json` file in
the repo root that is excluded from commits.
```json
{
"models":{
"embedding":{
"model":"qwen3-embedding:0.6b",
"max_chunk_size":32768
},
"initialized_sources":{}
}
}
```
`refresh_sources.lua` uses `models` to store default embedding model information
and `initialized_sources` to store which sources have been initialized.
## Tasks
- [ ] The whitelisting/blacklisting of tree should be in list too.
- [ ] synopsis_generator should be able to blacklist files it already tried?
- [ ] `refresh_sources.lua` doesn't check for defined files that don't exist
anymore, does it?
- [ ] I think it checks for every other possibility, but this needs checking.
+82
View File
@@ -0,0 +1,82 @@
-- USAGE:
-- -- set arbitrary log levels to be printed immediately
-- log{ info = true, warning = true, ducksauce = true, }
-- -- disable storage of messages depending on log level
-- log{ error = false, verbose = false, }
-- -- send anything (except nil) to a log level
-- log("bacom", true, "text", 5, function() end, {})
-- -- print all messages saved at a particular log level
-- log("error")
-- -- toggle ALL logging on and off with a boolean
-- log(false) -- this deletes all stored messages
-- -- set a maximum number of stored messages per log level
-- log(50) -- warning: reaching large limits makes logging expensive
-- -- check if a log level is being displayed
-- if log.arbitrary_exit then log("arbitrary_exit") os.exit(1) end
-- -- do anything to the stored messages (this example deletes them all)
-- log(function(messages) return {} end)
local microlog = {} -- display_levels are stored directly here
local stored_messages = {}
local enable_logging = true
local message_limit = math.huge
return setmetatable(microlog, {
__call = function(self, options, ...)
local options_type = type(options)
if (options_type == "string") and enable_logging then
if microlog[options] == false then return end
if not stored_messages[options] then stored_messages[options] = {} end
local message_table = stored_messages[options]
-- turn log message into text
local tab = {...}
for i = 1, #tab do
tab[i] = tostring(tab[i])
end
local current_message = table.concat(tab, "\t")
if #current_message == 0 then
-- print all stored_messages of specified level
for i = 1, #message_table do
print(message_table[i])
end
else
-- store message (and print if in display_levels)
message_table[#message_table + 1] = current_message
if #message_table > message_limit then table.remove(message_table, 1) end
if microlog[options] then print(current_message) end
end
-- set display_levels
elseif options_type == "table" then
for k,v in pairs(options) do
microlog[k] = v
end
elseif options_type == "number" then
message_limit = options
for k,v in pairs(stored_messages) do
while #v > message_limit do
table.remove(v, 1)
end
end
-- turn all logging off and delete stored messages; or turn it on
elseif options_type == "boolean" then
if options then
enable_logging = true
else
enable_logging = false
stored_messages = {}
end
-- arbitrary access :D
elseif options_type == "function" then
local result = options(stored_messages)
if type(result) == "table" then stored_messages = result end
end
end
})
+23 -8
View File
@@ -1,18 +1,26 @@
local timing = {}
local function human_readable_time(delta)
if delta >= 2*7*24*60*60 then -- if more than 2 weeks
delta = tostring(math.floor(delta/(7*24*60*60))/10) .. " weeks"
elseif delta >= 2*24*60*60 then -- if more than 2 days
delta = tostring(math.floor(delta/(24*60*60))/10) .. " days"
elseif delta >= 2*60*60 then -- if more than 2 hours
delta = tostring(math.floor(delta/(60*60/10))/10) .. " hours"
elseif delta >= 2*60 then -- if more then 2 minutes
delta = tostring(math.floor(delta/(60/10))/10) .. " minutes"
else
delta = tostring(delta) .. " seconds"
end
return delta
end
timing.display = function(n)
local function _display(n)
local time = timing[n]
local previous = timing[n - 1]
local delta = time.time - previous.time
if delta >= 2*60*60 then -- if more than 2 hours
delta = tostring(math.floor(delta/(60*60/10))/10) .. " hours"
elseif delta >= 120 then -- if more then 2 minutes
delta = tostring(math.floor(delta/(60/10))/10) .. " minutes"
else
delta = tostring(delta) .. " seconds"
end
local delta = human_readable_time(time.time - previous.time)
print(delta, previous.label)
end
@@ -32,6 +40,13 @@ timing.mark = function(label)
if #timing > 1 then
timing.display(#timing)
end
print("", "", label)
end
timing.estimate = function(current_position, total_operations)
local delta = os.time() - timing[#timing].time
local estimate = delta * total_operations / current_position - delta
return human_readable_time(estimate)
end
return timing
+67 -14
View File
@@ -18,7 +18,7 @@ if package.config:sub(1, 1) == "\\" then
}
else
utility = {
OS = "UNIX-like",
OS = "Linux",
path_separator = "/",
temp_directory = "/tmp/",
commands = {
@@ -81,6 +81,10 @@ standard_library_addition(string, "split", function(s, delimiter)
return result
end)
utility.leftpad = function(text, length, character)
return string.rep(character or " ", length - #(tostring(text))) .. text
end
utility.require = function(...)
@@ -217,7 +221,7 @@ utility.list = function(path, func)
local run = function(fn)
for line in output:gmatch("[^\r\n]+") do -- thanks to https://stackoverflow.com/a/32847589
if not (line == "." or line == "..") then
if not ((line == ".") or (line == "..")) then
fn(line)
end
end
@@ -235,7 +239,8 @@ utility.ls = function(...)
return utility.list(...)
end
utility.tree = function(path, options, fn)
local tree
tree = function(path, options, fn)
if type(options) == "function" then
fn = options
options = {}
@@ -245,16 +250,25 @@ utility.tree = function(path, options, fn)
if options.blacklist and options.blacklist[path_name] then return end
if options.whitelist and (not options.whitelist[path_name]) then return end
if options.extension_blacklist or options.extension_whitelist then
local _, _, extension = utility.split_path_components(path_name)
if options.extension_blacklist and options.extension_blacklist[extension] then return end
if options.extension_whitelist and (not options.extension_whitelist[extension]) then return end
end
if utility.is_file(path_name) then
if options.extension_blacklist or options.extension_whitelist then
local _, _, extension = utility.split_path_components(path_name)
if options.extension_blacklist and options.extension_blacklist[extension] then return end
if options.extension_whitelist and (not options.extension_whitelist[extension]) then return end
end
fn(path_name)
else
utility.tree(path .. utility.path_separator .. path_name, options, fn)
tree(path .. utility.path_separator .. path_name, options, fn)
end
end)
end
utility.tree = function(path, options, fn)
tree(path, options, function(path_name)
if path_name:find(path) == 1 then
fn(path_name)
else
fn(path .. utility.path_separator .. path_name)
end
end)
end
@@ -265,10 +279,11 @@ utility.read_file = function(file_name)
end)
end
utility.write_file = function(file_name, text)
utility.write_file = function(file_name, ...)
local text = table.concat{...}
return utility.open(file_name, "w", function(file)
file:write(text)
file:write("\n")
-- file:write("\n") -- I need to make sure /I/ handle this instead of trying to automate it
end)
end
@@ -299,6 +314,16 @@ utility.file_size = function(file_path)
return utility.open(file_path, "rb", function(file) return file:seek("end") end)
end
utility.sha512sum = function(file_path)
local sha512sum
if (utility.OS == "Linux") or (utility.OS == "macOS") then
sha512sum = utility.capture_safe("shasum -U -a 512 " .. file_path:enquote())
elseif utility.OS == "Windows" then
error("utility.sha512sum() not implemented for Windows.")
end
return sha512sum:sub(1, 128)
end
utility.escape_quotes_and_escapes = function(input)
@@ -350,10 +375,10 @@ end
local config_path = utility.path .. "config.json"
local config, config_lock
utility.get_config = function(skip_lock)
if not config then
local config_path = utility.path .. "config.json"
if utility.is_file(config_path) then
if not skip_lock then
config_lock = utility.get_lock(config_path)
@@ -371,7 +396,6 @@ end
utility.save_config = function()
if config then
local config_path = utility.path .. "config.json"
if not config_lock then
print("Warning: A config lock file was not established.")
end
@@ -387,6 +411,31 @@ utility.save_config = function()
end
end
utility.get_config_with_defaults = function(defaults)
local config = utility.get_config()
local loop, changes_made
loop = function(config_level, default_level)
for k,v in pairs(default_level) do
if not config_level[k] then
config_level[k] = v
changes_made = true
elseif type(v) == "table" then
loop(config_level[k], v)
end
end
end
loop(config, defaults)
if changes_made then
utility.save_config()
else
utility.release_lock(config_path)
end
return config
end
local data_file_locations = {}
@@ -570,4 +619,8 @@ end
if (utility.OS == "Linux") and (utility.capture_safe("uname"):find("Darwin") == 1) then
utility.OS = "macOS"
end
return utility
+19
View File
@@ -0,0 +1,19 @@
#!/usr/bin/env luajit
package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path
local utility = require "utility"
local json = utility.require("dkjson")
local search_path = arg[1]
local file_extensions = {}
utility.tree(search_path, function(file_path)
local _, _, extension = utility.split_path_components(file_path)
if extension then
file_extensions[extension] = true
end
end)
for extension in pairs(file_extensions) do
print(extension)
end
+293
View File
@@ -0,0 +1,293 @@
#!/usr/bin/env luajit
package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path
local utility = require "utility"
local json = utility.require("dkjson")
local log = utility.require("log")
local text_processing = utility.require("text_processing")
local timing = utility.require("timing")
-- TODO there should be a check function for a lock so a warning/error can be dumped?
local config = utility.get_config_with_defaults{
models = {
embedding = {
model = "qwen3-embedding:0.6b",
max_chunk_size = 32768,
},
initialized_sources = {},
},
}
log{
info = true,
warning = true,
debug = false,
-- files = true, -- debugging why the wrong files are selected
-- sha = true, -- what the fuck is going on with sha sums?
}
local memory_path = "PRIVATE_DATA" .. utility.path_separator .. "memory"
local embeddings_file_path = memory_path .. utility.path_separator .. "+embeddings.json"
local tmp_file_path = "PRIVATE_DATA" .. utility.path_separator .. ".tmp.2b65c19b-0883-49ca-8247-b1fe7760f922"
if not utility.path_exists(embeddings_file_path) then
os.execute("mkdir -p " .. memory_path:enquote())
utility.save_data({
files = {},
vectors = {},
}, embeddings_file_path)
end
local embeddings = utility.load_data(embeddings_file_path)
if log.debug then
log("debug", "Embeddings loaded.", embeddings, embeddings.files, embeddings.vectors)
local file_count = 0
for k,v in pairs(embeddings.files) do
file_count = file_count + 1
end
local vector_count = 0
for k,v in pairs(embeddings.vectors) do
vector_count = vector_count + 1
end
log("debug", file_count .. " files.")
log("debug", vector_count .. " vectors.")
-- os.exit(1)
end
local refresh_file_list = function(source_name, data_source)
timing.mark("Assembling file list for \"" .. source_name .. "\"")
local full_path = "PRIVATE_DATA" .. utility.path_separator .. data_source.path
if data_source.initialize_command and (not config.models.initialized_sources[data_source.path]) then
-- NOTE this is where we'd want to check/obtain a lock on the config so we can safely run this
os.execute("mkdir -p " .. full_path:enquote() .. " && cd " .. full_path:enquote() .. " && " .. data_source.initialize_command)
config.models.initialized_sources[data_source.path] = true
utility.save_config()
end
if data_source.refresh_command then
os.execute("cd " .. full_path:enquote() .. " && " .. data_source.refresh_command)
end
local compiled_filters = {}
if data_source.filters then
for name, object in pairs(data_source.filters) do
compiled_filters[name] = utility.enumerate(object)
end
end
local file_list = {}
utility.tree("PRIVATE_DATA" .. utility.path_separator .. data_source.path, compiled_filters, function(file_name)
log("files", file_name)
file_list[#file_list + 1] = file_name
end)
timing.mark("Finished assembling file list for \"" .. source_name .. "\"")
return file_list
end
-- returns nothing when too much text is sent
local generate_embeddings = function(data_source, text)
local max_chunk_size = data_source.max_chunk_size or config.models.embedding.max_chunk_size
local model = data_source.embedding_model or config.models.embedding.model
if #text > max_chunk_size then
return nil, "generate_embeddings() must only be passed appropriately-sized chunks!"
end
local result = utility.llm_prompt(text, model)
return json.decode(result)
end
local make_chunks = function(text, chunk_size)
assert(chunk_size, "make_chunks() requires chunk_size")
local half_chunk_size = math.floor(chunk_size / 2)
local chunks = { text }
while #text > chunk_size do
local first_chunk = text:sub(1, chunk_size)
local overlap_chunk = text:sub(half_chunk_size, chunk_size + half_chunk_size - 1)
chunks[#chunks + 1] = first_chunk
chunks[#chunks + 1] = overlap_chunk
text = text:sub(chunk_size)
if (#text > half_chunk_size) and (not (#text > chunk_size)) then
-- last chunk would be skipped if we didn't handle this here
chunks[#chunks + 1] = text
end
end
return chunks
end
-- returns nothing for empty files and errors
local process_file = function(data_source, file_name)
local text = utility.read_file(file_name)
if data_source.strip_frontmatter then
text = text_processing.strip_frontmatter(text)
end
if #text == 0 then
log("empty", file_name .. "\n is empty and being skipped.")
return
end
local chunk_size = data_source.max_chunk_size or config.models.embedding.max_chunk_size
local chunks = make_chunks(text, chunk_size)
if data_source.target_chunk_size then
local extra_chunks = make_chunks(text, data_source.target_chunk_size)
for i = 2, #extra_chunks do
chunks[#chunks + 1] = extra_chunks[i]
end
end
local new_embeddings = {}
for i = 1, #chunks do
log("debug", "Embedding source length:", #chunks[i])
new_embeddings[i] = generate_embeddings(data_source, chunks[i]) or {}
end
if #new_embeddings[1] == 0 then
if log.debug then
log("debug", "Vector lengths:")
for e = 1, #new_embeddings do
log("debug", "", e, #new_embeddings[e])
end
end
if #new_embeddings == 1 then
-- Ollama very rarely errors with:
-- Error: do embedding request: Post "http://127.0.0.1:53441/v1/embeddings": EOF
-- but it is inconsistent and re-running will eventually fix it.
log("warning", file_name .. "\n encountered an embedding error and will be skipped this run only.")
return
end
-- average all embeddings to make the core file embedding
local count = #new_embeddings[2]
for vector_index = 1, count do
local total = 0
for chunk = 2, #new_embeddings do
total = total + new_embeddings[chunk][vector_index]
end
new_embeddings[1][vector_index] = total / count
end
end
return chunks, new_embeddings
end
-- returns nothing for errors
local memorize_file = function(data_source, file_name)
local file_chunks, file_embeddings = process_file(data_source, file_name)
if not file_chunks then return end
local file_sums = {}
for i = 1, #file_chunks do
local function loop()
local text = file_chunks[i]
local current_embedding = file_embeddings[i]
utility.write_file(tmp_file_path, text)
local sha512sum = utility.sha512sum(tmp_file_path)
log("sha", "New sum? " .. sha512sum, tostring(embeddings.vectors[sha512sum]))
file_sums[#file_sums + 1] = sha512sum
if embeddings.vectors[sha512sum] then
log("sha", "Detected previously extant sum, skipping.")
return
end
os.execute(utility.commands.move .. tmp_file_path:enquote()
.. " " .. (memory_path .. utility.path_separator .. sha512sum):enquote())
embeddings.vectors[sha512sum] = current_embedding
log("sha", "Saved new sum!")
end
loop()
end
return file_sums
end
local delete_memories = function(sha512sum_list)
-- if true then return end -- TEMP disabling removal to compare and verify function
for i = 1, #sha512sum_list do
local sha512sum = sha512sum_list[i]
embeddings.vectors[sha512sum] = nil
os.execute("rm " .. (memory_path .. utility.path_separator .. sha512sum):enquote())
end
end
local refresh_sources = function()
os.execute("mkdir -p " .. memory_path:enquote())
timing.mark("Checking memory for missing embeddings.")
utility.list(memory_path, function(file_name)
log("files", file_name)
if (file_name == ".DS_Store") or (file_name == "+embeddings.json") then return end
local file_path = memory_path .. utility.path_separator .. file_name
local sha512sum = utility.sha512sum(file_path)
log("sha", "New sum? " .. sha512sum, tostring(embeddings.vectors[sha512sum]))
if not embeddings.vectors[sha512sum] then
log("sha", "Saved new sum!")
memorize_file({}, file_path)
end
end)
timing.mark("Finished checking memory for missing embeddings.")
local sources = utility.load_data("PRIVATE_DATA/sources.json")
for source_name, data_source in pairs(sources) do
local file_list = refresh_file_list(source_name, data_source)
timing.mark("Generating embeddings from source \"" .. source_name .. "\"")
for f = 1, #file_list do
local file_name = file_list[f]
local function loop()
local sha512sum = utility.sha512sum(file_name)
log("sha", file_name, sha512sum, "\n Sum present? " .. tostring(embeddings.vectors[sha512sum]))
if embeddings.vectors[sha512sum] then
log("debug", file_name .. "\n has already been embedded, skipping.")
-- add file reference if it was missing (also handles recognition of duplicate files)
if not embeddings.files[file_name] then
embeddings.files[file_name] = { sha512sum }
end
return
end
if embeddings.files[file_name] then
log("debug", file_name .. "\n Removing old embeddings for modified file.")
delete_memories(embeddings.files[file_name])
embeddings.files[file_name] = nil
end
log("debug", file_name .. "\n Generating embeddings..")
local file_sums = memorize_file(data_source, file_name)
embeddings.files[file_name] = file_sums
end
loop()
log("info", "Finished " .. utility.leftpad(f, #tostring(#file_list), "0")
.. "/" .. #file_list
.. " (" .. utility.leftpad(math.floor(f / #file_list * 100), 3, "0")
.. "%) ETA: " .. timing.estimate(f, #file_list))
end
timing.mark("Finished generating embeddings from source \"" .. source_name .. "\"")
end
utility.save_data(embeddings)
if utility.path_exists(tmp_file_path) then
os.execute("rm " .. tmp_file_path:enquote())
end
end
refresh_sources()
timing.mark("Finished.")
print("")
timing.display()
log("warning")
+53 -49
View File
@@ -7,49 +7,49 @@ local json = utility.require("dkjson")
local prompts = utility.require("prompts")
local text_processing = utility.require("text_processing")
local model = "gemma4:12b-mlx"
local minimum_bytes = 1000
local maximum_bytes = 40000
local config = utility.get_config_with_defaults{
models = {
synopsis_generator = {
model = "gemma4:12b-mlx",
minimum_bytes = 1000,
maximum_bytes = 40000,
},
},
}
local NOTEBOOK_PATH = "PRIVATE_DATA/notebook"
local blacklist = utility.enumerate{ ".git", ".gitattributes", ".gitignore", ".gitkeep", ".DS_Store", }
local extension_blacklist = utility.enumerate{ "gif", "jpg", "jpeg", "mp4", "pdf", "png", "webp", }
local data_location = "PRIVATE_DATA/synopsis_generator_file_list.json"
local file_list
if utility.path_exists(data_location) then
file_list = utility.load_data(data_location)
else
file_list = {}
end
local memory_path = "PRIVATE_DATA" .. utility.path_separator .. "memory"
local embeddings_file_path = memory_path .. utility.path_separator .. "+embeddings.json"
local data_path = "PRIVATE_DATA" .. utility.path_separator .. "synopses"
local data_location = data_path .. utility.path_separator .. "+file_list.json"
local refresh_file_list = function()
os.execute("cd " .. NOTEBOOK_PATH:enquot() .. " && git pull origin")
assert(utility.path_exists(embeddings_file_path), "Run \"./refresh_sources.lua\" to set up memory.")
local embeddings = utility.load_data(embeddings_file_path)
local new_files_list = {}
utility.tree(NOTEBOOK_PATH, {
blacklist = blacklist,
extension_blacklist = extension_blacklist,
}, function(file_name)
local minimum_bytes = config.models.synopsis_generator.minimum_bytes
local maximum_bytes = config.models.synopsis_generator.maximum_bytes
local file_list = {}
for file_name, sha512sum_list in pairs(embeddings.files) do
local file_size = utility.file_size(file_name)
if file_size >= minimum_bytes and file_size <= maximum_bytes then
local text = text_processing.strip_frontmatter(utility.read_file(file_name))
if #text >= minimum_bytes and #text <= maximum_bytes then
new_files_list[#new_files_list + 1] = file_name
file_list[#file_list + 1] = { file_name = file_name, }
end
end
end)
end
file_list = new_files_list
utility.save_data(new_files_list, data_location)
utility.save_data(file_list, data_location)
return file_list
end
local generate_and_score = function(file_name, text)
print("Writing synopsis...")
local synopsis = utility.llm_prompt(prompts.synopsis_prompt .. text, model)
local synopsis = utility.llm_prompt(prompts.synopsis_prompt .. text, config.models.synopsis_generator.model)
print(synopsis)
print("Scoring synopsis...")
local scoring, thinking = utility.llm_prompt(prompts.scoring_prompt .. synopsis, model)
local scoring, thinking = utility.llm_prompt(prompts.scoring_prompt .. synopsis, config.models.synopsis_generator.model)
print(scoring)
local scoring_decoded = json.decode(scoring) -- likely will not work because it consistently returns Markdown instead of JSON
if not scoring_decoded then
@@ -62,23 +62,22 @@ local generate_and_score = function(file_name, text)
thinking = thinking,
}
utility.save_data(object, "PRIVATE_DATA/synopses/" .. utility.uuid() .. ".json")
utility.save_data(object, data_path .. utility.path_separator .. utility.uuid() .. ".json")
end
local export_ordered_list_of_prompts = function()
local path = "PRIVATE_DATA/synopses"
local items = {}
local item_order = {}
utility.list(path, function(path_name)
local full_path = path .. utility.path_separator .. path_name
if path_name:find("%.json") then
local object = utility.load_data(full_path)
items[path_name] = object
if type(object.scoring) == "table" then
local mean_score, total_score = utility.mean(object.scoring)
item_order[#item_order + 1] = { path_name = path_name, total_score = total_score, mean_score = mean_score, }
end
utility.list(data_path, function(path_name)
local full_path = data_path .. utility.path_separator .. path_name
if full_path == data_location then return end
local object = utility.load_data(full_path)
items[path_name] = object
if type(object.scoring) == "table" then
local mean_score, total_score = utility.mean(object.scoring)
item_order[#item_order + 1] = { path_name = path_name, total_score = total_score, mean_score = mean_score, }
end
end)
@@ -87,7 +86,7 @@ local export_ordered_list_of_prompts = function()
local output = {
"---",
"title: Ordered Synopses (" .. #item_order .. " items)",
"author: [\"" .. model .. "\", \"Tangent\", \"Ollama\"]",
"author: [\"" .. config.models.synopsis_generator.model .. "\", \"Tangent\", \"Ollama\"]",
"publisher: \"synopsis_generator.lua\"",
"---",
"",
@@ -99,8 +98,11 @@ local export_ordered_list_of_prompts = function()
local tab = text:split("\n")
for index, line in ipairs(tab) do
if line:sub(1, 1) == "#" then
tab[index] = "#" .. tab[index]
-- ensure no output heading is an H1; make H6 bold text lines instead
if line:sub(1, 7) == "###### " then
tab[index] = "**" .. line:sub(8) .. "**"
elseif line:sub(1, 1) == "#" then
tab[index] = "#" .. line
end
end
@@ -108,31 +110,33 @@ local export_ordered_list_of_prompts = function()
output[#output + 1] = "## Scoring\n\n```json\n" .. json.encode(item.scoring, { indent = true, }) .. "\n```\n"
end
utility.write_file("PRIVATE_DATA/Ordered Synopses.md", table.concat(output, "\n"))
utility.write_file("PRIVATE_DATA/Ordered Synopses.md", table.concat(output, "\n"), "\n")
os.execute("pandoc \"PRIVATE_DATA/Ordered Synopses.md\" -o \"PRIVATE_DATA/Ordered Synopses.epub\"")
end
os.execute("mkdir -p PRIVATE_DATA/synopses")
os.execute("mkdir -p " .. data_path:enquote())
local file_list
if utility.path_exists(data_location) then
file_list = utility.load_data(data_location)
end
if arg[1] == "refresh_file_list" then
print("Refresing file list...")
refresh_file_list()
file_list = refresh_file_list()
os.exit(0)
elseif arg[1] == "export_ordered_list_of_prompts" then
export_ordered_list_of_prompts()
os.exit(0)
end
print(#file_list .. " files to select from.")
if #file_list == 0 then
print("Run \"./synopsis_generator.lua refresh_file_list\" first.")
os.exit(1)
end
assert(file_list and (#file_list > 0), "Run \"./synopsis_generator.lua refresh_file_list\" first.")
print(#file_list .. " files to select from.")
while true do
local file_name = file_list[math.random(1, #file_list)]
local file_name = file_list[math.random(1, #file_list)].file_name
local text = utility.read_file(file_name)
print(file_name .. " chosen.")
generate_and_score(file_name, text)
+4 -44
View File
@@ -6,16 +6,9 @@ local json = utility.require("dkjson")
local file_name = arg[1]
print(utility.OS)
local function strip_markdown(text)
local tab = text:split("\n")
table.remove(tab, 1) -- codeblock opening
table.remove(tab, #tab) -- end of codeblock
print(table.concat(tab, "\n"))
end
os.exit(0)
local prompt = [[Return JSON: category, tags (array), summary (short)]]
-- local prompt = [[Return YAML: category, tags (array), summary (short)]]
@@ -24,39 +17,6 @@ if prompt:sub(-1) ~= "\n" then
prompt = prompt .. "\n\n"
end
local file_contents = utility.open(file_name, "r", function(file)
return file:read("*all")
end)
local file_contents = utility.read_file(file_name)
prompt = prompt .. file_contents
local model = "gemma3:4b"
-- local output = utility.capture_safe("ollama run " .. model .. " --nowordwrap " .. prompt:enquote())
-- local embedding_model = "nomic-embed-text"
-- local output = utility.capture_safe("ollama run " .. embedding_model .. " " .. file_contents:enquote())
-- output = output:sub(1, -2) -- strip extra newline from utility.capture_safe
-- strip YAML frontmatter (if present)
-- can error, will return nil & error message
local function strip_frontmatter(text)
local tab = text:split("\n")
if tab[1] == "---" then
table.remove(tab, 1)
while true do
local done = tab[1] == "---"
table.remove(tab, 1)
if done then
return table.concat(tab, "\n")
elseif #tab < 1 then
return nil, "Invalid YAML frontmatter."
end
end
end
return text
end
-- print(output)
print(strip_frontmatter(file_contents))
-- print(strip_markdown(output))
-- print(utility.llm_prompt(prompt))