Compare commits
31
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f1b5fbc615 | ||
|
|
f229e7331e | ||
|
|
470901d9b9 | ||
|
|
977920df61 | ||
|
|
98112bde4a | ||
|
|
214f82b0b1 | ||
|
|
11d0e356a2 | ||
|
|
d0bd237174 | ||
|
|
8ff72681e9 | ||
|
|
c5aed984f8 | ||
|
|
1e79c9ed99 | ||
|
|
1b7c211aa2 | ||
|
|
425cae4c20 | ||
|
|
915319ee1c | ||
|
|
5b0520a7ce | ||
|
|
bfe93e33f9 | ||
|
|
dde5283aca | ||
|
|
012d3abe5a | ||
|
|
7b4608ec7e | ||
|
|
692d4f5690 | ||
|
|
20015af69c | ||
|
|
399ad47a3d | ||
|
|
e7223d5c3e | ||
|
|
a966ea5bc6 | ||
|
|
f8c911bcd4 | ||
|
|
b5250e302d | ||
|
|
ac80118d5a | ||
|
|
4d4f95f54a | ||
|
|
16a0c6c93d | ||
|
|
c0bc8ca6a2 | ||
|
|
49a22eb2c4 |
No files matched your search
+2
-2
@@ -1,3 +1,3 @@
|
|||||||
.DS_Store
|
.DS_Store
|
||||||
PRIVATE_DATA/**
|
PRIVATE_DATA/
|
||||||
!PRIVATE_DATA/.gitkeep
|
config.json
|
||||||
Whitespace-only changes.
@@ -1,18 +1,69 @@
|
|||||||
# LLM Tricks
|
# LLM Tricks
|
||||||
Trying to find *anything* useful to do with local LLMs.
|
Finding useful things to do with local LLMs.
|
||||||
|
|
||||||
All scripts work with data within `PRIVATE_DATA/` so that private data can't be
|
All scripts work with data within `PRIVATE_DATA/` so that private data can't be
|
||||||
accidentally committed.
|
accidentally committed.
|
||||||
|
|
||||||
###### `cosine_similarity.lua`
|
### `cosine_similarity.lua`
|
||||||
Opens `embeddings.json` and creates `similarities.json` with a sorted list of
|
Opens `embeddings.json` and creates `similarities.json` with a sorted list of
|
||||||
comparisons.
|
comparisons.
|
||||||
|
|
||||||
###### `generate_embeddings.lua`
|
### `generate_embeddings.lua`
|
||||||
Opens every file on its whitelist within the `notebook` directory, and generates
|
Opens every file on its whitelist within the `notebook` directory, and generates
|
||||||
embeddings (placed in `embeddings.json`). When files are too long, it truncates
|
embeddings (placed in `embeddings.json`). When files are too long, it truncates
|
||||||
them.
|
them.
|
||||||
|
|
||||||
###### `least_similar.lua`
|
### `least_similar.lua`
|
||||||
Opens `similarities.json`, reverses the sort order, and saves it as
|
Opens `similarities.json`, reverses the sort order, and saves it as
|
||||||
`differences.json`.
|
`differences.json`.
|
||||||
|
|
||||||
|
### `refresh_sources.lua`
|
||||||
|
Creates/Maintains a store of chunked data with embeddings based on configurable
|
||||||
|
sources. Example:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"source name":{
|
||||||
|
"filters":{
|
||||||
|
"blacklist":[".git"],
|
||||||
|
"extension_whitelist":["md"]
|
||||||
|
},
|
||||||
|
"initialize_command":"git clone REMOTE .",
|
||||||
|
"path":"will be created before initialize_command is run",
|
||||||
|
"strip_frontmatter":true,
|
||||||
|
"refresh_command":"git reset --hard origin/main"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
The filters are based on `utility.tree`'s filter options (optional).
|
||||||
|
`initialize_command` is only run the first time (optional),
|
||||||
|
while `refresh_command` is run each time (optional).
|
||||||
|
Commands will be run in the specified `path`.
|
||||||
|
`strip_frontmatter` will remove YAML frontmatter (common in Markdown files).
|
||||||
|
|
||||||
|
The embeddings are stored like so:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"files":{
|
||||||
|
"PRIVATE_DATA/source_path/path/to/file.ext":["sha512sum", "another sum"]
|
||||||
|
},
|
||||||
|
"vectors":{
|
||||||
|
"sha512sum":[0.5, 0, 1, -0.5, -1, ...]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
TODO
|
||||||
|
|
||||||
|
### `synopsis_generator.lua`
|
||||||
|
Chooses a random file within `notebook`, and generates a novel synopsis from it.
|
||||||
|
|
||||||
|
Arguments:
|
||||||
|
- `refresh_file_list`: Refreshes the cached file list to choose from.
|
||||||
|
- `export_ordered_list_of_prompts`: Makes an epub to review generated synopses.
|
||||||
|
|
||||||
|
## Tasks
|
||||||
|
- [ ] The whitelisting/blacklisting of tree should be in list too.
|
||||||
|
- [ ] synopsis_generator should be able to blacklist files it already tried?
|
||||||
+19
-106
@@ -2,83 +2,21 @@
|
|||||||
|
|
||||||
package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path
|
package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path
|
||||||
local utility = require "utility"
|
local utility = require "utility"
|
||||||
local json = utility.require("dkjson")
|
|
||||||
|
|
||||||
local PATH = "PRIVATE_DATA/notebook"
|
local text_processing = utility.require("text_processing")
|
||||||
local whitelist = { md = true, }
|
local timing = utility.require("timing")
|
||||||
|
local json = utility.require("dkjson")
|
||||||
|
|
||||||
-- local embedding_model = "nomic-embed-text"
|
-- local embedding_model = "nomic-embed-text"
|
||||||
-- local maximum_file_size = 2048
|
-- local maximum_file_size = 2048
|
||||||
local embedding_model = "qwen3-embedding:0.6b"
|
local embedding_model = "qwen3-embedding:0.6b"
|
||||||
local maximum_file_size = 32768
|
local maximum_file_size = 32768
|
||||||
|
|
||||||
|
local NOTEBOOK_PATH = "PRIVATE_DATA/notebook"
|
||||||
|
local blacklist = utility.enumerate{ ".git", } -- accelerate processing by ignoring .git directory
|
||||||
|
local extension_whitelist = utility.enumerate{ "md", }
|
||||||
|
|
||||||
|
|
||||||
local timing = {}
|
|
||||||
local function display_timing(n)
|
|
||||||
local function _display(n)
|
|
||||||
local time = timing[n]
|
|
||||||
local previous = timing[n - 1]
|
|
||||||
|
|
||||||
local delta = time.time - previous.time
|
|
||||||
if delta >= 2*60*60 then -- 2 hours
|
|
||||||
delta = tostring(math.floor(delta/(60*60/10))/10) .. " hours"
|
|
||||||
elseif delta >= 120 then -- 2 minutes
|
|
||||||
delta = tostring(math.floor(delta/(60/10))/10) .. " minutes"
|
|
||||||
else
|
|
||||||
delta = tostring(delta) .. " seconds"
|
|
||||||
end
|
|
||||||
|
|
||||||
print(delta, previous.label)
|
|
||||||
end
|
|
||||||
|
|
||||||
if n then
|
|
||||||
_display(n)
|
|
||||||
else
|
|
||||||
print("All measured timings:")
|
|
||||||
for i = 2, #timing do
|
|
||||||
_display(i)
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
local function mark_timing(label)
|
|
||||||
timing[#timing + 1] = { label = label, time = os.time(), }
|
|
||||||
if #timing > 1 then
|
|
||||||
display_timing(#timing)
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
-- strip YAML frontmatter (if present)
|
|
||||||
-- can error, will return nil & error message
|
|
||||||
local function strip_frontmatter(text)
|
|
||||||
local tab = text:split("\n")
|
|
||||||
if tab[1] == "---" then
|
|
||||||
table.remove(tab, 1)
|
|
||||||
while true do
|
|
||||||
local done = tab[1] == "---"
|
|
||||||
table.remove(tab, 1)
|
|
||||||
if done then
|
|
||||||
return table.concat(tab, "\n")
|
|
||||||
elseif #tab < 1 then
|
|
||||||
return nil, "Invalid YAML frontmatter."
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return text
|
|
||||||
end
|
|
||||||
|
|
||||||
local tree
|
|
||||||
tree = function(path, fn)
|
|
||||||
utility.list(path or ".", function(path_name)
|
|
||||||
if path_name == ".git" then return end
|
|
||||||
if utility.is_file(path_name) then
|
|
||||||
fn(path_name)
|
|
||||||
else
|
|
||||||
tree(path .. utility.path_separator .. path_name, fn)
|
|
||||||
end
|
|
||||||
end)
|
|
||||||
end
|
|
||||||
|
|
||||||
local function leftpad(text, length)
|
local function leftpad(text, length)
|
||||||
return string.rep("0", length - #(tostring(text))) .. text
|
return string.rep("0", length - #(tostring(text))) .. text
|
||||||
@@ -89,36 +27,22 @@ end
|
|||||||
local file_list = {}
|
local file_list = {}
|
||||||
local embeddings = {}
|
local embeddings = {}
|
||||||
|
|
||||||
mark_timing("Assembling file list.")
|
timing.mark("Assembling file list.")
|
||||||
tree(PATH, function(file_name)
|
utility.tree(NOTEBOOK_PATH, {
|
||||||
local path, name, extension = utility.split_path_components(file_name)
|
blacklist = blacklist,
|
||||||
if not extension then
|
extension_whitelist = extension_whitelist,
|
||||||
return
|
}, function(file_name)
|
||||||
end
|
|
||||||
|
|
||||||
if not whitelist[extension] then
|
|
||||||
return
|
|
||||||
end
|
|
||||||
|
|
||||||
file_list[#file_list + 1] = file_name
|
file_list[#file_list + 1] = file_name
|
||||||
end)
|
end)
|
||||||
|
|
||||||
mark_timing("Generating embeddings.")
|
timing.mark("Generating embeddings.")
|
||||||
for i = 1, #file_list do
|
for i = 1, #file_list do
|
||||||
local function _run()
|
local function _run()
|
||||||
local file_name = file_list[i]
|
local file_name = file_list[i]
|
||||||
|
|
||||||
-- stripping YAML frontmatter before generating embeddings
|
-- stripping YAML frontmatter before generating embeddings
|
||||||
local file_contents = utility.open(file_name, "r", function(file)
|
local file_contents = utility.read_file(file_name)
|
||||||
return file:read("*all")
|
local tmp_file_contents = text_processing.strip_frontmatter(file_contents)
|
||||||
end)
|
|
||||||
|
|
||||||
local tmp_file_contents, error_message = strip_frontmatter(file_contents)
|
|
||||||
if tmp_file_contents == nil then
|
|
||||||
print("ERROR: " .. file_name .. " " .. error_message)
|
|
||||||
tmp_file_contents = file_contents
|
|
||||||
end
|
|
||||||
|
|
||||||
if #tmp_file_contents == 0 then
|
if #tmp_file_contents == 0 then
|
||||||
print(file_name .. "\n is empty and will be skipped.")
|
print(file_name .. "\n is empty and will be skipped.")
|
||||||
return
|
return
|
||||||
@@ -129,14 +53,7 @@ for i = 1, #file_list do
|
|||||||
tmp_file_contents = tmp_file_contents:sub(1, maximum_file_size)
|
tmp_file_contents = tmp_file_contents:sub(1, maximum_file_size)
|
||||||
end
|
end
|
||||||
|
|
||||||
local tmp_file_name = utility.tmp_file_name()
|
local output = utility.llm_prompt(tmp_file_contents, embedding_model)
|
||||||
utility.open(tmp_file_name, "w", function(file)
|
|
||||||
file:write(tmp_file_contents)
|
|
||||||
end)
|
|
||||||
|
|
||||||
local output = utility.capture_safe("cat " .. tmp_file_name:enquote() .. " | ollama run " .. embedding_model)
|
|
||||||
os.execute("rm " .. tmp_file_name)
|
|
||||||
output = output:sub(1, -2) -- strip extra newline from utility.capture_safe
|
|
||||||
|
|
||||||
if #output == 0 then
|
if #output == 0 then
|
||||||
print("Warning: " .. file_name .. " did not have embeddings generated.")
|
print("Warning: " .. file_name .. " did not have embeddings generated.")
|
||||||
@@ -157,13 +74,9 @@ for i = 1, #file_list do
|
|||||||
_run()
|
_run()
|
||||||
end
|
end
|
||||||
|
|
||||||
mark_timing("Outputting embeddings in JSON.")
|
timing.mark("Outputting embeddings in JSON.")
|
||||||
utility.open("PRIVATE_DATA/embeddings.json", "w", function(file)
|
utility.save_data(embeddings, "PRIVATE_DATA/embeddings.json")
|
||||||
local output = json.encode(embeddings, { indent = true })
|
|
||||||
file:write(output)
|
|
||||||
file:write("\n")
|
|
||||||
end)
|
|
||||||
|
|
||||||
mark_timing("Finished.")
|
timing.mark("Finished.")
|
||||||
print("")
|
print("")
|
||||||
display_timing()
|
timing.display()
|
||||||
+75
@@ -0,0 +1,75 @@
|
|||||||
|
local stored_messages = {}
|
||||||
|
local display_levels = {}
|
||||||
|
local enable_logging = true
|
||||||
|
local message_limit = math.huge
|
||||||
|
|
||||||
|
-- USAGE:
|
||||||
|
-- -- set arbitrary log levels to be printed immediately
|
||||||
|
-- log{ info = true, warning = true, ducksauce = true, }
|
||||||
|
-- -- send anything (except nil) to a log level
|
||||||
|
-- log("bacom", true, "text", 5, function() end, {})
|
||||||
|
-- -- print all messages saved at a particular log level
|
||||||
|
-- log("error")
|
||||||
|
-- -- toggle ALL logging on and off with a boolean
|
||||||
|
-- log(false) -- this deletes all stored messages
|
||||||
|
-- -- set a maximum number of stored messages per log level
|
||||||
|
-- log(50) -- warning: reaching large limits makes logging expensive
|
||||||
|
-- -- do anything to the stored messages (this example deletes them all)
|
||||||
|
-- log(function(messages) return {} end)
|
||||||
|
|
||||||
|
return function(options, ...)
|
||||||
|
local options_type = type(options)
|
||||||
|
|
||||||
|
if (options_type == "string") and enable_logging then
|
||||||
|
if not stored_messages[options] then stored_messages[options] = {} end
|
||||||
|
local message_table = stored_messages[options]
|
||||||
|
|
||||||
|
-- turn log message into text
|
||||||
|
local tab = {...}
|
||||||
|
for i = 1, #tab do
|
||||||
|
tab[i] = tostring(tab[i])
|
||||||
|
end
|
||||||
|
local current_message = table.concat(tab, "\t")
|
||||||
|
|
||||||
|
if #current_message == 0 then
|
||||||
|
-- print all stored_messages of specified level
|
||||||
|
for i = 1, #message_table do
|
||||||
|
print(message_table[i])
|
||||||
|
end
|
||||||
|
|
||||||
|
else
|
||||||
|
-- store message (and print if in display_levels)
|
||||||
|
message_table[#message_table + 1] = current_message
|
||||||
|
if #message_table > message_limit then table.remove(message_table, 1) end
|
||||||
|
if display_levels[options] then print(current_message) end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- set display_levels
|
||||||
|
elseif options_type == "table" then
|
||||||
|
for k,v in pairs(options) do
|
||||||
|
display_levels[k] = v
|
||||||
|
end
|
||||||
|
|
||||||
|
elseif options_type == "number" then
|
||||||
|
message_limit = options
|
||||||
|
for k,v in pairs(stored_messages) do
|
||||||
|
while #v > message_limit do
|
||||||
|
table.remove(v, 1)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
-- turn all logging off and delete stored messages; or turn it on
|
||||||
|
elseif options_type == "boolean" then
|
||||||
|
if options then
|
||||||
|
enable_logging = true
|
||||||
|
else
|
||||||
|
enable_logging = false
|
||||||
|
stored_messages = {}
|
||||||
|
end
|
||||||
|
|
||||||
|
-- arbitrary access :D
|
||||||
|
elseif options_type == "function" then
|
||||||
|
local result = options(stored_messages)
|
||||||
|
if type(result) == "table" then stored_messages = result end
|
||||||
|
end
|
||||||
|
end
|
||||||
@@ -0,0 +1,35 @@
|
|||||||
|
local text_processing = {}
|
||||||
|
|
||||||
|
-- strip YAML frontmatter (if present)
|
||||||
|
-- on error, will return nil & error message
|
||||||
|
text_processing.strip_frontmatter = function(text)
|
||||||
|
local tab = text:split("\n")
|
||||||
|
if tab[1] == "---" then
|
||||||
|
table.remove(tab, 1)
|
||||||
|
while true do
|
||||||
|
local done = tab[1] == "---"
|
||||||
|
table.remove(tab, 1)
|
||||||
|
if done then
|
||||||
|
return table.concat(tab, "\n")
|
||||||
|
elseif #tab < 1 then
|
||||||
|
return nil, "Invalid YAML frontmatter."
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return text
|
||||||
|
end
|
||||||
|
|
||||||
|
-- strip Markdown formatting of a JSON block
|
||||||
|
-- just returns the string if it doesn't match
|
||||||
|
text_processing.strip_markdown_codeblock = function(text)
|
||||||
|
local tab = text:split("\n")
|
||||||
|
if tab[1] == "```json" then
|
||||||
|
table.remove(tab, 1)
|
||||||
|
table.remove(tab, #tab)
|
||||||
|
return table.concat(tab, "\n")
|
||||||
|
else
|
||||||
|
return text
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return text_processing
|
||||||
@@ -0,0 +1,52 @@
|
|||||||
|
local timing = {}
|
||||||
|
|
||||||
|
local function human_readable_time(delta)
|
||||||
|
if delta >= 2*7*24*60*60 then -- if more than 2 weeks
|
||||||
|
delta = tostring(math.floor(delta/(7*24*60*60))/10) .. " weeks"
|
||||||
|
elseif delta >= 2*24*60*60 then -- if more than 2 days
|
||||||
|
delta = tostring(math.floor(delta/(24*60*60))/10) .. " days"
|
||||||
|
elseif delta >= 2*60*60 then -- if more than 2 hours
|
||||||
|
delta = tostring(math.floor(delta/(60*60/10))/10) .. " hours"
|
||||||
|
elseif delta >= 2*60 then -- if more then 2 minutes
|
||||||
|
delta = tostring(math.floor(delta/(60/10))/10) .. " minutes"
|
||||||
|
else
|
||||||
|
delta = tostring(delta) .. " seconds"
|
||||||
|
end
|
||||||
|
return delta
|
||||||
|
end
|
||||||
|
|
||||||
|
timing.display = function(n)
|
||||||
|
local function _display(n)
|
||||||
|
local time = timing[n]
|
||||||
|
local previous = timing[n - 1]
|
||||||
|
|
||||||
|
local delta = human_readable_time(time.time - previous.time)
|
||||||
|
|
||||||
|
print(delta, previous.label)
|
||||||
|
end
|
||||||
|
|
||||||
|
if n then
|
||||||
|
_display(n)
|
||||||
|
else
|
||||||
|
print("All measured timings:")
|
||||||
|
for i = 2, #timing do
|
||||||
|
_display(i)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
timing.mark = function(label)
|
||||||
|
timing[#timing + 1] = { label = label, time = os.time(), }
|
||||||
|
if #timing > 1 then
|
||||||
|
timing.display(#timing)
|
||||||
|
end
|
||||||
|
print("", "", label)
|
||||||
|
end
|
||||||
|
|
||||||
|
timing.estimate = function(current_position, total_operations)
|
||||||
|
local delta = os.time() - timing[#timing].time
|
||||||
|
local estimate = delta * total_operations / current_position - delta
|
||||||
|
return human_readable_time(estimate)
|
||||||
|
end
|
||||||
|
|
||||||
|
return timing
|
||||||
+64
-6
@@ -18,7 +18,7 @@ if package.config:sub(1, 1) == "\\" then
|
|||||||
}
|
}
|
||||||
else
|
else
|
||||||
utility = {
|
utility = {
|
||||||
OS = "UNIX-like",
|
OS = "Linux",
|
||||||
path_separator = "/",
|
path_separator = "/",
|
||||||
temp_directory = "/tmp/",
|
temp_directory = "/tmp/",
|
||||||
commands = {
|
commands = {
|
||||||
@@ -81,6 +81,10 @@ standard_library_addition(string, "split", function(s, delimiter)
|
|||||||
return result
|
return result
|
||||||
end)
|
end)
|
||||||
|
|
||||||
|
utility.leftpad = function(text, length, character)
|
||||||
|
return string.rep(character or " ", length - #(tostring(text))) .. text
|
||||||
|
end
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
utility.require = function(...)
|
utility.require = function(...)
|
||||||
@@ -217,7 +221,7 @@ utility.list = function(path, func)
|
|||||||
|
|
||||||
local run = function(fn)
|
local run = function(fn)
|
||||||
for line in output:gmatch("[^\r\n]+") do -- thanks to https://stackoverflow.com/a/32847589
|
for line in output:gmatch("[^\r\n]+") do -- thanks to https://stackoverflow.com/a/32847589
|
||||||
if not (line == "." or line == "..") then
|
if not ((line == ".") or (line == "..")) then
|
||||||
fn(line)
|
fn(line)
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
@@ -235,7 +239,8 @@ utility.ls = function(...)
|
|||||||
return utility.list(...)
|
return utility.list(...)
|
||||||
end
|
end
|
||||||
|
|
||||||
utility.tree = function(path, options, fn)
|
local tree
|
||||||
|
tree = function(path, options, fn)
|
||||||
if type(options) == "function" then
|
if type(options) == "function" then
|
||||||
fn = options
|
fn = options
|
||||||
options = {}
|
options = {}
|
||||||
@@ -244,10 +249,26 @@ utility.tree = function(path, options, fn)
|
|||||||
utility.list(path or ".", function(path_name)
|
utility.list(path or ".", function(path_name)
|
||||||
if options.blacklist and options.blacklist[path_name] then return end
|
if options.blacklist and options.blacklist[path_name] then return end
|
||||||
if options.whitelist and (not options.whitelist[path_name]) then return end
|
if options.whitelist and (not options.whitelist[path_name]) then return end
|
||||||
|
|
||||||
if utility.is_file(path_name) then
|
if utility.is_file(path_name) then
|
||||||
|
if options.extension_blacklist or options.extension_whitelist then
|
||||||
|
local _, _, extension = utility.split_path_components(path_name)
|
||||||
|
if options.extension_blacklist and options.extension_blacklist[extension] then return end
|
||||||
|
if options.extension_whitelist and (not options.extension_whitelist[extension]) then return end
|
||||||
|
end
|
||||||
|
|
||||||
fn(path_name)
|
fn(path_name)
|
||||||
else
|
else
|
||||||
utility.tree(path .. utility.path_separator .. path_name, options, fn)
|
tree(path .. utility.path_separator .. path_name, options, fn)
|
||||||
|
end
|
||||||
|
end)
|
||||||
|
end
|
||||||
|
utility.tree = function(path, options, fn)
|
||||||
|
tree(path, options, function(path_name)
|
||||||
|
if path_name:find(path) == 1 then
|
||||||
|
fn(path_name)
|
||||||
|
else
|
||||||
|
fn(path .. utility.path_separator .. path_name)
|
||||||
end
|
end
|
||||||
end)
|
end)
|
||||||
end
|
end
|
||||||
@@ -258,10 +279,11 @@ utility.read_file = function(file_name)
|
|||||||
end)
|
end)
|
||||||
end
|
end
|
||||||
|
|
||||||
utility.write_file = function(file_name, text)
|
utility.write_file = function(file_name, ...)
|
||||||
|
local text = table.concat{...}
|
||||||
return utility.open(file_name, "w", function(file)
|
return utility.open(file_name, "w", function(file)
|
||||||
file:write(text)
|
file:write(text)
|
||||||
file:write("\n")
|
-- file:write("\n") -- I need to make sure /I/ handle this instead of trying to automate it
|
||||||
end)
|
end)
|
||||||
end
|
end
|
||||||
|
|
||||||
@@ -292,6 +314,16 @@ utility.file_size = function(file_path)
|
|||||||
return utility.open(file_path, "rb", function(file) return file:seek("end") end)
|
return utility.open(file_path, "rb", function(file) return file:seek("end") end)
|
||||||
end
|
end
|
||||||
|
|
||||||
|
utility.sha512sum = function(file_path)
|
||||||
|
local sha512sum
|
||||||
|
if (utility.OS == "Linux") or (utility.OS == "macOS") then
|
||||||
|
sha512sum = utility.capture_safe("shasum -U -a 512 " .. file_path:enquote())
|
||||||
|
elseif utility.OS == "Windows" then
|
||||||
|
error("utility.sha512sum() not implemented for Windows.")
|
||||||
|
end
|
||||||
|
return sha512sum:sub(1, 128)
|
||||||
|
end
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
utility.escape_quotes_and_escapes = function(input)
|
utility.escape_quotes_and_escapes = function(input)
|
||||||
@@ -541,4 +573,30 @@ end
|
|||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
-- additionally returns total and count
|
||||||
|
utility.mean = function(object)
|
||||||
|
local total, count = 0, 0
|
||||||
|
for _, value in pairs(object) do
|
||||||
|
total = total + value
|
||||||
|
count = count + 1
|
||||||
|
end
|
||||||
|
return total / count, total, count
|
||||||
|
end
|
||||||
|
|
||||||
|
-- additionally returns total
|
||||||
|
utility.median = function(object)
|
||||||
|
local tab = {}
|
||||||
|
for _, value in pairs(object) do
|
||||||
|
tab[#tab + 1] = value
|
||||||
|
end
|
||||||
|
table.sort(tab)
|
||||||
|
return tab[math.floor(#tab / 2)], #tab
|
||||||
|
end
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
if (utility.OS == "Linux") and (utility.capture_safe("uname"):find("Darwin") == 1) then
|
||||||
|
utility.OS = "macOS"
|
||||||
|
end
|
||||||
|
|
||||||
return utility
|
return utility
|
||||||
Executable
+262
@@ -0,0 +1,262 @@
|
|||||||
|
#!/usr/bin/env luajit
|
||||||
|
|
||||||
|
package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path
|
||||||
|
local utility = require "utility"
|
||||||
|
|
||||||
|
local json = utility.require("dkjson")
|
||||||
|
local log = utility.require("log")
|
||||||
|
local text_processing = utility.require("text_processing")
|
||||||
|
local timing = utility.require("timing")
|
||||||
|
|
||||||
|
-- TODO make utility have a function for getting/setting defaults where locking is only used to set defaults if they aren't present
|
||||||
|
-- TODO there should be a check function for a lock so a warning/error can be dumped?
|
||||||
|
local config = utility.get_config("no-lock")
|
||||||
|
if not config.models then
|
||||||
|
config.models = {
|
||||||
|
embedding = {
|
||||||
|
model = "qwen3-embedding:0.6b",
|
||||||
|
max_chunk_size = 32768,
|
||||||
|
},
|
||||||
|
initialized_sources = {},
|
||||||
|
}
|
||||||
|
utility.save_config()
|
||||||
|
end
|
||||||
|
|
||||||
|
log{
|
||||||
|
info = true,
|
||||||
|
warning = true,
|
||||||
|
-- debug = true,
|
||||||
|
-- files = true, -- debugging why the wrong files are selected
|
||||||
|
-- sha = true, -- what the fuck is going on with sha sums?
|
||||||
|
}
|
||||||
|
|
||||||
|
local memory_path = "PRIVATE_DATA" .. utility.path_separator .. "memory"
|
||||||
|
local embeddings_file_path = memory_path .. utility.path_separator .. "+embeddings.json"
|
||||||
|
local tmp_file_path = "PRIVATE_DATA" .. utility.path_separator .. ".tmp.2b65c19b-0883-49ca-8247-b1fe7760f922"
|
||||||
|
|
||||||
|
local embeddings
|
||||||
|
if not utility.path_exists(embeddings_file_path) then
|
||||||
|
os.execute("mkdir -p " .. memory_path:enquote())
|
||||||
|
utility.save_data({
|
||||||
|
files = {},
|
||||||
|
vectors = {},
|
||||||
|
}, embeddings_file_path)
|
||||||
|
end
|
||||||
|
embeddings = utility.load_data(embeddings_file_path)
|
||||||
|
local function embeddings_debug()
|
||||||
|
log("debug", "Embeddings loaded.", embeddings, embeddings.files, embeddings.vectors)
|
||||||
|
local file_count = 0
|
||||||
|
for k,v in pairs(embeddings.files) do
|
||||||
|
file_count = file_count + 1
|
||||||
|
end
|
||||||
|
local vector_count = 0
|
||||||
|
for k,v in pairs(embeddings.vectors) do
|
||||||
|
vector_count = vector_count + 1
|
||||||
|
end
|
||||||
|
log("debug", file_count .. " files.")
|
||||||
|
log("debug", vector_count .. " vectors.")
|
||||||
|
-- os.exit(1)
|
||||||
|
end
|
||||||
|
embeddings_debug()
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
local refresh_file_list = function(source_name, data_source)
|
||||||
|
timing.mark("Assembling file list for " .. source_name .. ".")
|
||||||
|
|
||||||
|
local full_path = "PRIVATE_DATA" .. utility.path_separator .. data_source.path
|
||||||
|
if data_source.initialize_command and (not config.models.initialized_sources[data_source.path]) then
|
||||||
|
-- NOTE this is where we'd want to check/obtain a lock on the config so we can safely run this
|
||||||
|
os.execute("mkdir -p " .. full_path:enquote() .. " && cd " .. full_path:enquote() .. " && " .. data_source.initialize_command)
|
||||||
|
config.models.initialized_sources[data_source.path] = true
|
||||||
|
utility.save_config()
|
||||||
|
end
|
||||||
|
if data_source.refresh_command then
|
||||||
|
os.execute("cd " .. full_path:enquote() .. " && " .. data_source.refresh_command)
|
||||||
|
end
|
||||||
|
|
||||||
|
local compiled_filters = {}
|
||||||
|
if data_source.filters then
|
||||||
|
for name, object in pairs(data_source.filters) do
|
||||||
|
compiled_filters[name] = utility.enumerate(object)
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
local file_list = {}
|
||||||
|
utility.tree("PRIVATE_DATA" .. utility.path_separator .. data_source.path, compiled_filters, function(file_name)
|
||||||
|
log("files", file_name)
|
||||||
|
file_list[#file_list + 1] = file_name
|
||||||
|
end)
|
||||||
|
|
||||||
|
timing.mark("Finished assembling file list for " .. source_name .. ".")
|
||||||
|
return file_list
|
||||||
|
end
|
||||||
|
|
||||||
|
-- returns nothing when too much text is sent
|
||||||
|
local generate_embeddings = function(text)
|
||||||
|
if #text > config.models.embedding.max_chunk_size then
|
||||||
|
return nil, "generate_embeddings() must only be passed appropriately-sized chunks!"
|
||||||
|
end
|
||||||
|
|
||||||
|
local result = utility.llm_prompt(text, config.models.embedding.model)
|
||||||
|
return json.decode(result)
|
||||||
|
end
|
||||||
|
|
||||||
|
-- returns nothing for empty files and errors
|
||||||
|
local process_file = function(data_source, file_name)
|
||||||
|
local text = utility.read_file(file_name)
|
||||||
|
|
||||||
|
if data_source.strip_frontmatter then
|
||||||
|
text = text_processing.strip_frontmatter(text)
|
||||||
|
end
|
||||||
|
|
||||||
|
if #text == 0 then
|
||||||
|
log("empty", file_name .. "\n is empty and being skipped.")
|
||||||
|
return
|
||||||
|
end
|
||||||
|
|
||||||
|
local chunk_size = config.models.embedding.max_chunk_size
|
||||||
|
local half_chunk_size = math.floor(chunk_size / 2)
|
||||||
|
local chunks = { text }
|
||||||
|
|
||||||
|
while #text > chunk_size do
|
||||||
|
local first_chunk = text:sub(1, chunk_size)
|
||||||
|
local overlap_chunk = text:sub(half_chunk_size, chunk_size + half_chunk_size - 1)
|
||||||
|
|
||||||
|
chunks[#chunks + 1] = first_chunk
|
||||||
|
chunks[#chunks + 1] = overlap_chunk
|
||||||
|
|
||||||
|
text = text:sub(chunk_size)
|
||||||
|
if (#text > half_chunk_size) and (not (#text > chunk_size)) then
|
||||||
|
-- last chunk would be skipped if we didn't handle this here
|
||||||
|
chunks[#chunks + 1] = text
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
local new_embeddings = {}
|
||||||
|
for i = 1, #chunks do
|
||||||
|
log("debug", "Embedding length:", #chunks[i])
|
||||||
|
new_embeddings[i] = generate_embeddings(chunks[i]) or {}
|
||||||
|
end
|
||||||
|
|
||||||
|
if #new_embeddings[1] == 0 then
|
||||||
|
log("debug", "Vector lengths:")
|
||||||
|
for e = 1, #new_embeddings do
|
||||||
|
log("debug", "", e, #new_embeddings[e])
|
||||||
|
end
|
||||||
|
if #new_embeddings == 1 then
|
||||||
|
-- Ollama very rarely errors with:
|
||||||
|
-- Error: do embedding request: Post "http://127.0.0.1:53441/v1/embeddings": EOF
|
||||||
|
-- but it is inconsistent and re-running will eventually fix it.
|
||||||
|
log("warning", file_name .. "\n encountered an embedding error and will be skipped this run only.")
|
||||||
|
return
|
||||||
|
end
|
||||||
|
|
||||||
|
-- average all embeddings to make the core file embedding
|
||||||
|
local count = #new_embeddings[2]
|
||||||
|
for vector_index = 1, count do
|
||||||
|
local total = 0
|
||||||
|
for chunk = 2, #new_embeddings do
|
||||||
|
total = total + new_embeddings[chunk][vector_index]
|
||||||
|
end
|
||||||
|
new_embeddings[1][vector_index] = total / count
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return chunks, new_embeddings
|
||||||
|
end
|
||||||
|
|
||||||
|
-- returns nothing for errors
|
||||||
|
local memorize_file = function(data_source, file_name)
|
||||||
|
local file_chunks, file_embeddings = process_file(data_source, file_name)
|
||||||
|
if not file_chunks then return end
|
||||||
|
|
||||||
|
local file_sums = {}
|
||||||
|
for i = 1, #file_chunks do
|
||||||
|
local function loop()
|
||||||
|
local text = file_chunks[i]
|
||||||
|
local current_embedding = file_embeddings[i]
|
||||||
|
|
||||||
|
utility.write_file(tmp_file_path, text)
|
||||||
|
|
||||||
|
local sha512sum = utility.sha512sum(tmp_file_path)
|
||||||
|
log("sha", "New sum? " .. sha512sum, tostring(embeddings.vectors[sha512sum]))
|
||||||
|
file_sums[#file_sums + 1] = sha512sum
|
||||||
|
if embeddings.vectors[sha512sum] then
|
||||||
|
log("sha", "Detected previously extant sum, skipping.")
|
||||||
|
return
|
||||||
|
end
|
||||||
|
|
||||||
|
os.execute(utility.commands.move .. tmp_file_path:enquote()
|
||||||
|
.. " " .. (memory_path .. utility.path_separator .. sha512sum):enquote())
|
||||||
|
embeddings.vectors[sha512sum] = current_embedding
|
||||||
|
log("sha", "Saved new sum!")
|
||||||
|
end
|
||||||
|
loop()
|
||||||
|
end
|
||||||
|
|
||||||
|
return file_sums
|
||||||
|
end
|
||||||
|
|
||||||
|
local refresh_sources = function()
|
||||||
|
os.execute("mkdir -p " .. memory_path:enquote())
|
||||||
|
|
||||||
|
timing.mark("Checking memory for missing embeddings.")
|
||||||
|
utility.list(memory_path, function(file_name)
|
||||||
|
log("files", file_name)
|
||||||
|
if (file_name == ".DS_Store") or (file_name == "+embeddings.json") then return end
|
||||||
|
|
||||||
|
local file_path = memory_path .. utility.path_separator .. file_name
|
||||||
|
local sha512sum = utility.sha512sum(file_path)
|
||||||
|
log("sha", "New sum? " .. sha512sum, tostring(embeddings.vectors[sha512sum]))
|
||||||
|
if not embeddings.vectors[sha512sum] then
|
||||||
|
log("sha", "Saved new sum!")
|
||||||
|
memorize_file({}, file_path)
|
||||||
|
end
|
||||||
|
end)
|
||||||
|
timing.mark("Finished checking memory for missing embeddings.")
|
||||||
|
|
||||||
|
local sources = utility.load_data("PRIVATE_DATA/sources.json")
|
||||||
|
for source_name, data_source in pairs(sources) do
|
||||||
|
local file_list = refresh_file_list(source_name, data_source)
|
||||||
|
|
||||||
|
timing.mark("Generating embeddings from source \"" .. source_name .. "\"")
|
||||||
|
for f = 1, #file_list do
|
||||||
|
local file_name = file_list[f]
|
||||||
|
local function loop()
|
||||||
|
local sha512sum = utility.sha512sum(file_name)
|
||||||
|
log("sha", file_name, sha512sum, "\n Sum present? " .. tostring(embeddings.vectors[sha512sum]))
|
||||||
|
if embeddings.vectors[sha512sum] then
|
||||||
|
log("debug", file_name .. "\n has already been embedded, skipping.")
|
||||||
|
-- add file reference if it was missing
|
||||||
|
if not embeddings.files[file_name] then
|
||||||
|
embeddings.files[file_name] = { sha512sum }
|
||||||
|
end
|
||||||
|
return
|
||||||
|
end
|
||||||
|
|
||||||
|
log("debug", "Generating embeddings for " .. file_name .. ".")
|
||||||
|
local file_sums = memorize_file(data_source, file_name)
|
||||||
|
embeddings.files[file_name] = file_sums
|
||||||
|
end
|
||||||
|
loop()
|
||||||
|
log("info", "Finished " .. utility.leftpad(f, #tostring(#file_list), "0")
|
||||||
|
.. "/" .. #file_list
|
||||||
|
.. " (" .. utility.leftpad(math.floor(f / #file_list * 100), 3, "0")
|
||||||
|
.. "%) ETA: " .. timing.estimate(f, #file_list))
|
||||||
|
end
|
||||||
|
|
||||||
|
timing.mark("Finished generating embeddings from source \"" .. source_name .. "\"")
|
||||||
|
end
|
||||||
|
|
||||||
|
utility.save_data(embeddings)
|
||||||
|
if utility.path_exists(tmp_file_path) then
|
||||||
|
os.execute("rm " .. tmp_file_path:enquote())
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
refresh_sources()
|
||||||
|
timing.mark("Finished.")
|
||||||
|
print("")
|
||||||
|
timing.display()
|
||||||
|
log("warning")
|
||||||
+80
-161
@@ -5,118 +5,55 @@ local utility = require "utility"
|
|||||||
|
|
||||||
local json = utility.require("dkjson")
|
local json = utility.require("dkjson")
|
||||||
local prompts = utility.require("prompts")
|
local prompts = utility.require("prompts")
|
||||||
|
local text_processing = utility.require("text_processing")
|
||||||
|
|
||||||
local default_model = "gemma4:12b-mlx"
|
local model = "gemma4:12b-mlx"
|
||||||
local minimum_bytes = 1000
|
local minimum_bytes = 1000
|
||||||
local maximum_bytes = 40000
|
local maximum_bytes = 40000
|
||||||
|
|
||||||
local files
|
local NOTEBOOK_PATH = "PRIVATE_DATA/notebook"
|
||||||
if utility.path_exists("PRIVATE_DATA/file_list.json") then
|
local blacklist = utility.enumerate{ ".git", ".gitattributes", ".gitignore", ".gitkeep", ".DS_Store", }
|
||||||
files = utility.load_data("PRIVATE_DATA/file_list.json")
|
local extension_blacklist = utility.enumerate{ "gif", "jpg", "jpeg", "mp4", "pdf", "png", "webp", }
|
||||||
|
|
||||||
|
local data_location = "PRIVATE_DATA/synopsis_generator_file_list.json"
|
||||||
|
local file_list
|
||||||
|
if utility.path_exists(data_location) then
|
||||||
|
file_list = utility.load_data(data_location)
|
||||||
else
|
else
|
||||||
files = {}
|
file_list = {}
|
||||||
end
|
|
||||||
|
|
||||||
-- strip YAML frontmatter (if present)
|
|
||||||
-- can error, will return nil & error message
|
|
||||||
local function strip_frontmatter(text)
|
|
||||||
local tab = text:split("\n")
|
|
||||||
if tab[1] == "---" then
|
|
||||||
table.remove(tab, 1)
|
|
||||||
while true do
|
|
||||||
local done = tab[1] == "---"
|
|
||||||
table.remove(tab, 1)
|
|
||||||
if done then
|
|
||||||
return table.concat(tab, "\n")
|
|
||||||
elseif #tab < 1 then
|
|
||||||
return nil, "Invalid YAML frontmatter."
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return text
|
|
||||||
end
|
|
||||||
|
|
||||||
-- strip Markdown formatting of a JSON block
|
|
||||||
-- just returns the string if it doesn't match
|
|
||||||
local function strip_markdown_codeblock(text)
|
|
||||||
local tab = text:split("\n")
|
|
||||||
if tab[1] == "```json" then
|
|
||||||
table.remove(tab, 1)
|
|
||||||
table.remove(tab, #tab)
|
|
||||||
return table.concat(tab, "\n")
|
|
||||||
else
|
|
||||||
return text
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
local strip_reasoning = function(text, reasoning_lines)
|
|
||||||
if not reasoning_lines then reasoning_lines = {} end
|
|
||||||
local tab = text:split("\n")
|
|
||||||
table.remove(tab, 1) -- remove "Thinking..."
|
|
||||||
|
|
||||||
while true do
|
|
||||||
local done = tab[1] == "...done thinking."
|
|
||||||
local line = table.remove(tab, 1)
|
|
||||||
if done then
|
|
||||||
table.remove(tab, 1) -- remove newline after end of thinking
|
|
||||||
return table.concat(tab, "\n")
|
|
||||||
elseif #tab < 1 then
|
|
||||||
return text -- no reasoning output
|
|
||||||
else
|
|
||||||
reasoning_lines[#reasoning_lines + 1] = line -- export thinking lines
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
local send_prompt = function(text, model)
|
|
||||||
if type(text) == "table" then text = table.concat(text, "\n") end
|
|
||||||
|
|
||||||
local tmp_file_name = utility.tmp_file_name()
|
|
||||||
utility.open(tmp_file_name, "w", function(file)
|
|
||||||
file:write(text)
|
|
||||||
end)
|
|
||||||
|
|
||||||
-- word wrap breaks the raw output badly, so I need to implement my own for terminal output somehow
|
|
||||||
local output = utility.capture_safe("cat " .. tmp_file_name:enquote() .. " | ollama run " .. (model or default_model) .. " --nowordwrap")
|
|
||||||
os.execute("ollama stop " .. (model or default_model)) -- NOTE this makes things slower, but more stable
|
|
||||||
os.execute("rm " .. tmp_file_name)
|
|
||||||
if not output then error("ollama failed to generate output") end
|
|
||||||
output = output:sub(1, -2) -- strip extra newline from utility.capture_safe
|
|
||||||
|
|
||||||
local thinking = {}
|
|
||||||
output = strip_reasoning(output, thinking)
|
|
||||||
|
|
||||||
return output, thinking
|
|
||||||
end
|
end
|
||||||
|
|
||||||
local refresh_file_list = function()
|
local refresh_file_list = function()
|
||||||
local blacklist = { -- I'm only blacklisting binary formats because funny results happen with really invalid texts
|
os.execute("cd " .. NOTEBOOK_PATH:enquot() .. " && git pull origin")
|
||||||
"jpg", "mp4", "pdf", "png", "webp", "jpeg", "gif",
|
|
||||||
} for _, name in ipairs(blacklist) do blacklist[name] = true end
|
|
||||||
|
|
||||||
local new_files_list = {}
|
local new_files_list = {}
|
||||||
utility.tree("PRIVATE_DATA/notebook", {
|
utility.tree(NOTEBOOK_PATH, {
|
||||||
blacklist = utility.enumerate{".git", ".gitignore", ".gitkeep", ".DS_Store"}
|
blacklist = blacklist,
|
||||||
|
extension_blacklist = extension_blacklist,
|
||||||
}, function(file_name)
|
}, function(file_name)
|
||||||
local _, _, extension = utility.split_path_components(file_name)
|
local file_size = utility.file_size(file_name)
|
||||||
if not blacklist[extension] then
|
if file_size >= minimum_bytes and file_size <= maximum_bytes then
|
||||||
|
local text = text_processing.strip_frontmatter(utility.read_file(file_name))
|
||||||
|
if #text >= minimum_bytes and #text <= maximum_bytes then
|
||||||
new_files_list[#new_files_list + 1] = file_name
|
new_files_list[#new_files_list + 1] = file_name
|
||||||
end
|
end
|
||||||
|
end
|
||||||
end)
|
end)
|
||||||
files = new_files_list
|
|
||||||
utility.save_data(new_files_list, "PRIVATE_DATA/file_list.json")
|
file_list = new_files_list
|
||||||
|
utility.save_data(new_files_list, data_location)
|
||||||
end
|
end
|
||||||
|
|
||||||
local generate_and_score = function(file_name, text)
|
local generate_and_score = function(file_name, text)
|
||||||
print("Writing synopsis...")
|
print("Writing synopsis...")
|
||||||
local synopsis = send_prompt(prompts.synopsis_prompt .. text)
|
local synopsis = utility.llm_prompt(prompts.synopsis_prompt .. text, model)
|
||||||
print(synopsis)
|
print(synopsis)
|
||||||
print("Scoring synopsis...")
|
print("Scoring synopsis...")
|
||||||
local scoring, thinking = send_prompt(prompts.scoring_prompt .. synopsis)
|
local scoring, thinking = utility.llm_prompt(prompts.scoring_prompt .. synopsis, model)
|
||||||
print(scoring)
|
print(scoring)
|
||||||
local scoring_decoded = json.decode(scoring) -- likely will not work because it consistently returns Markdown instead of JSON
|
local scoring_decoded = json.decode(scoring) -- likely will not work because it consistently returns Markdown instead of JSON
|
||||||
if not scoring_decoded then
|
if not scoring_decoded then
|
||||||
scoring_decoded = json.decode(strip_markdown_codeblock(scoring))
|
scoring_decoded = json.decode(text_processing.strip_markdown_codeblock(scoring))
|
||||||
end
|
end
|
||||||
|
|
||||||
local object = {
|
local object = {
|
||||||
@@ -128,6 +65,53 @@ local generate_and_score = function(file_name, text)
|
|||||||
utility.save_data(object, "PRIVATE_DATA/synopses/" .. utility.uuid() .. ".json")
|
utility.save_data(object, "PRIVATE_DATA/synopses/" .. utility.uuid() .. ".json")
|
||||||
end
|
end
|
||||||
|
|
||||||
|
local export_ordered_list_of_prompts = function()
|
||||||
|
local path = "PRIVATE_DATA/synopses"
|
||||||
|
local items = {}
|
||||||
|
local item_order = {}
|
||||||
|
|
||||||
|
utility.list(path, function(path_name)
|
||||||
|
local full_path = path .. utility.path_separator .. path_name
|
||||||
|
if path_name:find("%.json") then
|
||||||
|
local object = utility.load_data(full_path)
|
||||||
|
items[path_name] = object
|
||||||
|
if type(object.scoring) == "table" then
|
||||||
|
local mean_score, total_score = utility.mean(object.scoring)
|
||||||
|
item_order[#item_order + 1] = { path_name = path_name, total_score = total_score, mean_score = mean_score, }
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end)
|
||||||
|
|
||||||
|
table.sort(item_order, function(A,B) return A.mean_score > B.mean_score end)
|
||||||
|
|
||||||
|
local output = {
|
||||||
|
"---",
|
||||||
|
"title: Ordered Synopses (" .. #item_order .. " items)",
|
||||||
|
"author: [\"" .. model .. "\", \"Tangent\", \"Ollama\"]",
|
||||||
|
"publisher: \"synopsis_generator.lua\"",
|
||||||
|
"---",
|
||||||
|
"",
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, v in pairs(item_order) do
|
||||||
|
local item = items[v.path_name]
|
||||||
|
local text = item.synopsis
|
||||||
|
local tab = text:split("\n")
|
||||||
|
|
||||||
|
for index, line in ipairs(tab) do
|
||||||
|
if line:sub(1, 1) == "#" then
|
||||||
|
tab[index] = "#" .. tab[index]
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
output[#output + 1] = "# " .. v.path_name .. " (" .. v.total_score .. ")\n\n" .. table.concat(tab, "\n") .. "\n"
|
||||||
|
output[#output + 1] = "## Scoring\n\n```json\n" .. json.encode(item.scoring, { indent = true, }) .. "\n```\n"
|
||||||
|
end
|
||||||
|
|
||||||
|
utility.write_file("PRIVATE_DATA/Ordered Synopses.md", table.concat(output, "\n"), "\n")
|
||||||
|
os.execute("pandoc \"PRIVATE_DATA/Ordered Synopses.md\" -o \"PRIVATE_DATA/Ordered Synopses.epub\"")
|
||||||
|
end
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
os.execute("mkdir -p PRIVATE_DATA/synopses")
|
os.execute("mkdir -p PRIVATE_DATA/synopses")
|
||||||
@@ -135,86 +119,21 @@ os.execute("mkdir -p PRIVATE_DATA/synopses")
|
|||||||
if arg[1] == "refresh_file_list" then
|
if arg[1] == "refresh_file_list" then
|
||||||
print("Refresing file list...")
|
print("Refresing file list...")
|
||||||
refresh_file_list()
|
refresh_file_list()
|
||||||
end
|
os.exit(0)
|
||||||
|
elseif arg[1] == "export_ordered_list_of_prompts" then
|
||||||
if arg[1] == "repair_synopsis_exports" then
|
export_ordered_list_of_prompts()
|
||||||
local path = "PRIVATE_DATA/synopses"
|
|
||||||
utility.list(path, function(path_name)
|
|
||||||
path_name = path .. utility.path_separator .. path_name
|
|
||||||
if path_name:find("%.json") then
|
|
||||||
local object = utility.load_data(path_name)
|
|
||||||
if type(object.scoring) == "table" then return end -- don't fuck with working pieces
|
|
||||||
local decoded = json.decode(strip_markdown_codeblock(object.scoring))
|
|
||||||
if decoded then
|
|
||||||
object.scoring = decoded
|
|
||||||
utility.save_data(object, path_name)
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end)
|
|
||||||
os.exit(0)
|
os.exit(0)
|
||||||
end
|
end
|
||||||
|
|
||||||
if arg[1] == "export_ordered_list_of_prompts" then
|
print(#file_list .. " files to select from.")
|
||||||
local path = "PRIVATE_DATA/synopses"
|
if #file_list == 0 then
|
||||||
local items = {}
|
|
||||||
local item_order = {}
|
|
||||||
utility.list(path, function(path_name)
|
|
||||||
local full_path = path .. utility.path_separator .. path_name
|
|
||||||
if path_name:find("%.json") then
|
|
||||||
local object = utility.load_data(full_path)
|
|
||||||
items[path_name] = object
|
|
||||||
if type(object.scoring) == "table" then
|
|
||||||
local s = object.scoring
|
|
||||||
local total_score = s.conflict_potential + s.emotional_potential + s.character_potential + s.worldbuilding_potential + s.expansion_potential + s.overall_promise + s.memorability + s.originality + s.curiosity + s.hook
|
|
||||||
item_order[#item_order + 1] = { path_name = path_name, total_score = total_score, }
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end)
|
|
||||||
table.sort(item_order, function(A,B) return A.total_score > B.total_score end)
|
|
||||||
-- for k,v in pairs(item_order) do print(v.path_name,v.total_score) end
|
|
||||||
local output = {
|
|
||||||
"---",
|
|
||||||
"title: Ordered Synopses (" .. #item_order .. " items)",
|
|
||||||
"author: [\"Gemma4:12b-mlx\", \"Tangent\", \"Ollama\"]",
|
|
||||||
"publisher: Tangent",
|
|
||||||
"---",
|
|
||||||
"",
|
|
||||||
}
|
|
||||||
for _, v in pairs(item_order) do
|
|
||||||
local item = items[v.path_name]
|
|
||||||
local text = item.synopsis
|
|
||||||
local tab = text:split("\n")
|
|
||||||
for index, line in ipairs(tab) do
|
|
||||||
if line:sub(1, 1) == "#" then
|
|
||||||
tab[index] = "#" .. tab[index]
|
|
||||||
end
|
|
||||||
end
|
|
||||||
-- output[#output + 1] = "# " .. v.path_name .. " (" .. v.total_score .. ")\n\n" .. item.synopsis .. "\n"
|
|
||||||
output[#output + 1] = "# " .. v.path_name .. " (" .. v.total_score .. ")\n\n" .. table.concat(tab, "\n") .. "\n"
|
|
||||||
output[#output + 1] = "## Scoring\n\n```json\n" .. json.encode(item.scoring, { indent = true, }) .. "\n```\n"
|
|
||||||
end
|
|
||||||
utility.write_file("PRIVATE_DATA/Ordered Synopses.md", table.concat(output, "\n"))
|
|
||||||
os.execute("pandoc \"PRIVATE_DATA/Ordered Synopses.md\" -o \"PRIVATE_DATA/Ordered Synopses.epub\"")
|
|
||||||
os.exit(0)
|
|
||||||
end
|
|
||||||
|
|
||||||
print(#files .. " files to select from.")
|
|
||||||
if #files == 0 then
|
|
||||||
print("Run \"./synopsis_generator.lua refresh_file_list\" first.")
|
print("Run \"./synopsis_generator.lua refresh_file_list\" first.")
|
||||||
os.exit(1)
|
os.exit(1)
|
||||||
end
|
end
|
||||||
|
|
||||||
while true do
|
while true do
|
||||||
print("Selecting a file...")
|
local file_name = file_list[math.random(1, #file_list)]
|
||||||
local file_name = files[math.random(1, #files)]
|
|
||||||
local file_size = utility.file_size(file_name)
|
|
||||||
if file_size > minimum_bytes and file_size <= maximum_bytes then
|
|
||||||
local text = utility.read_file(file_name)
|
local text = utility.read_file(file_name)
|
||||||
text = strip_frontmatter(text)
|
|
||||||
if #text > minimum_bytes and #text <= maximum_bytes then
|
|
||||||
print(file_name .. " chosen.")
|
print(file_name .. " chosen.")
|
||||||
generate_and_score(file_name, text) -- kind of the main function, innit?
|
generate_and_score(file_name, text)
|
||||||
-- os.exit(0)
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
end
|
||||||
@@ -6,16 +6,9 @@ local json = utility.require("dkjson")
|
|||||||
|
|
||||||
local file_name = arg[1]
|
local file_name = arg[1]
|
||||||
|
|
||||||
|
print(utility.OS)
|
||||||
|
|
||||||
|
os.exit(0)
|
||||||
local function strip_markdown(text)
|
|
||||||
local tab = text:split("\n")
|
|
||||||
table.remove(tab, 1) -- codeblock opening
|
|
||||||
table.remove(tab, #tab) -- end of codeblock
|
|
||||||
print(table.concat(tab, "\n"))
|
|
||||||
end
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
local prompt = [[Return JSON: category, tags (array), summary (short)]]
|
local prompt = [[Return JSON: category, tags (array), summary (short)]]
|
||||||
-- local prompt = [[Return YAML: category, tags (array), summary (short)]]
|
-- local prompt = [[Return YAML: category, tags (array), summary (short)]]
|
||||||
@@ -24,39 +17,6 @@ if prompt:sub(-1) ~= "\n" then
|
|||||||
prompt = prompt .. "\n\n"
|
prompt = prompt .. "\n\n"
|
||||||
end
|
end
|
||||||
|
|
||||||
local file_contents = utility.open(file_name, "r", function(file)
|
local file_contents = utility.read_file(file_name)
|
||||||
return file:read("*all")
|
|
||||||
end)
|
|
||||||
prompt = prompt .. file_contents
|
prompt = prompt .. file_contents
|
||||||
|
-- print(utility.llm_prompt(prompt))
|
||||||
local model = "gemma3:4b"
|
|
||||||
-- local output = utility.capture_safe("ollama run " .. model .. " --nowordwrap " .. prompt:enquote())
|
|
||||||
|
|
||||||
-- local embedding_model = "nomic-embed-text"
|
|
||||||
-- local output = utility.capture_safe("ollama run " .. embedding_model .. " " .. file_contents:enquote())
|
|
||||||
|
|
||||||
-- output = output:sub(1, -2) -- strip extra newline from utility.capture_safe
|
|
||||||
|
|
||||||
-- strip YAML frontmatter (if present)
|
|
||||||
-- can error, will return nil & error message
|
|
||||||
local function strip_frontmatter(text)
|
|
||||||
local tab = text:split("\n")
|
|
||||||
if tab[1] == "---" then
|
|
||||||
table.remove(tab, 1)
|
|
||||||
while true do
|
|
||||||
local done = tab[1] == "---"
|
|
||||||
table.remove(tab, 1)
|
|
||||||
if done then
|
|
||||||
return table.concat(tab, "\n")
|
|
||||||
elseif #tab < 1 then
|
|
||||||
return nil, "Invalid YAML frontmatter."
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return text
|
|
||||||
end
|
|
||||||
|
|
||||||
-- print(output)
|
|
||||||
|
|
||||||
print(strip_frontmatter(file_contents))
|
|
||||||
-- print(strip_markdown(output))
|
|
||||||
Reference in new issue
Block a user