Compare commits
40
Commits
b245a29d3c
..
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
58107e0196 | ||
|
|
da37b4ea29 | ||
|
|
dd9ca71fc7 | ||
|
|
77298dad0c | ||
|
|
8d0c79b34d | ||
|
|
23c2db1046 | ||
|
|
6381a49d13 | ||
|
|
f1b5fbc615 | ||
|
|
f229e7331e | ||
|
|
470901d9b9 | ||
|
|
977920df61 | ||
|
|
98112bde4a | ||
|
|
214f82b0b1 | ||
|
|
11d0e356a2 | ||
|
|
d0bd237174 | ||
|
|
8ff72681e9 | ||
|
|
c5aed984f8 | ||
|
|
1e79c9ed99 | ||
|
|
1b7c211aa2 | ||
|
|
425cae4c20 | ||
|
|
915319ee1c | ||
|
|
5b0520a7ce | ||
|
|
bfe93e33f9 | ||
|
|
dde5283aca | ||
|
|
012d3abe5a | ||
|
|
7b4608ec7e | ||
|
|
692d4f5690 | ||
|
|
20015af69c | ||
|
|
399ad47a3d | ||
|
|
e7223d5c3e | ||
|
|
a966ea5bc6 | ||
|
|
f8c911bcd4 | ||
|
|
b5250e302d | ||
|
|
ac80118d5a | ||
|
|
4d4f95f54a | ||
|
|
16a0c6c93d | ||
|
|
c0bc8ca6a2 | ||
|
|
49a22eb2c4 | ||
|
|
707776c12d | ||
|
|
8708c896aa |
No files matched your search
+2
-2
@@ -1,3 +1,3 @@
|
||||
.DS_Store
|
||||
PRIVATE_DATA/**
|
||||
!PRIVATE_DATA/.gitkeep
|
||||
PRIVATE_DATA/
|
||||
config.json
|
||||
Whitespace-only changes.
@@ -1,18 +1,101 @@
|
||||
# LLM Tricks
|
||||
Trying to find *anything* useful to do with local LLMs.
|
||||
Finding useful things to do with local LLMs.
|
||||
|
||||
All scripts work with data within `PRIVATE_DATA/` so that private data can't be
|
||||
accidentally committed.
|
||||
|
||||
###### `cosine_similarity.lua`
|
||||
### `cosine_similarity.lua`
|
||||
Opens `embeddings.json` and creates `similarities.json` with a sorted list of
|
||||
comparisons.
|
||||
|
||||
###### `generate_embeddings.lua`
|
||||
### `generate_embeddings.lua`
|
||||
Opens every file on its whitelist within the `notebook` directory, and generates
|
||||
embeddings (placed in `embeddings.json`). When files are too long, it truncates
|
||||
them.
|
||||
|
||||
###### `least_similar.lua`
|
||||
### `least_similar.lua`
|
||||
Opens `similarities.json`, reverses the sort order, and saves it as
|
||||
`differences.json`.
|
||||
|
||||
### `refresh_sources.lua`
|
||||
Creates/Maintains a store of chunked data with embeddings based on configurable
|
||||
`sources.json`. Example:
|
||||
|
||||
```json
|
||||
{
|
||||
"source name":{
|
||||
"embedding_model":"qwen3-embedding:0.6b",
|
||||
"filters":{
|
||||
"blacklist":[".git"],
|
||||
"extension_whitelist":["md"]
|
||||
},
|
||||
"initialize_command":"git clone REMOTE .",
|
||||
"max_chunk_size":32768,
|
||||
"path":"will be created before initialize_command is run",
|
||||
"strip_frontmatter":true,
|
||||
"target_chunk_size":3072,
|
||||
"refresh_command":"git fetch origin && git reset --hard origin/main"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
The filters are based on `utility.tree`'s filter options (optional).
|
||||
`initialize_command` is only run the first time (optional),
|
||||
while `refresh_command` is run each time (optional).
|
||||
Commands will be run in the specified `path`.
|
||||
`strip_frontmatter` will remove YAML frontmatter (common in Markdown files).
|
||||
`embedding_model` and `max_chunk_size` are self-explanatory (optional).
|
||||
`target_chunk_size` (optional) allows for finer-grained chunks to be generated
|
||||
alongside the maximum overviews for more precise retrieval.
|
||||
|
||||
The embeddings are stored like so:
|
||||
|
||||
```json
|
||||
{
|
||||
"files":{
|
||||
"PRIVATE_DATA/source_path/path/to/file.ext":["sha512sum", "another sum"]
|
||||
},
|
||||
"vectors":{
|
||||
"sha512sum":[0.5, 0, 1, -0.5, -1, ...]
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
SHA2 512-bit sums are used to link a text chunk with its embedding. All file
|
||||
names reference a list of text chunks so they can handle being too large. Every
|
||||
file that is too large for a single chunk has a whole-file embedding calculated
|
||||
first (it is the first element of the array), so that if a whole file becomes
|
||||
relevant, it can still show up instead of only chunks.
|
||||
|
||||
### `synopsis_generator.lua`
|
||||
Chooses a random file within `notebook`, and generates a novel synopsis from it.
|
||||
|
||||
Arguments:
|
||||
- `refresh_file_list`: Refreshes the cached file list to choose from.
|
||||
- `export_ordered_list_of_prompts`: Makes an epub to review generated synopses.
|
||||
|
||||
## JSON config
|
||||
This repo uses my utility library's config system, using a `config.json` file in
|
||||
the repo root that is excluded from commits.
|
||||
|
||||
```json
|
||||
{
|
||||
"models":{
|
||||
"embedding":{
|
||||
"model":"qwen3-embedding:0.6b",
|
||||
"max_chunk_size":32768
|
||||
},
|
||||
"initialized_sources":{}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
`refresh_sources.lua` uses `models` to store default embedding model information
|
||||
and `initialized_sources` to store which sources have been initialized.
|
||||
|
||||
## Tasks
|
||||
- [ ] The whitelisting/blacklisting of tree should be in list too.
|
||||
- [ ] synopsis_generator should be able to blacklist files it already tried?
|
||||
- [ ] `refresh_sources.lua` doesn't check for defined files that don't exist
|
||||
anymore, does it?
|
||||
- [ ] I think it checks for every other possibility, but this needs checking.
|
||||
+19
-106
@@ -2,83 +2,21 @@
|
||||
|
||||
package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path
|
||||
local utility = require "utility"
|
||||
local json = utility.require("dkjson")
|
||||
|
||||
local PATH = "PRIVATE_DATA/notebook"
|
||||
local whitelist = { md = true, }
|
||||
local text_processing = utility.require("text_processing")
|
||||
local timing = utility.require("timing")
|
||||
local json = utility.require("dkjson")
|
||||
|
||||
-- local embedding_model = "nomic-embed-text"
|
||||
-- local maximum_file_size = 2048
|
||||
local embedding_model = "qwen3-embedding:0.6b"
|
||||
local maximum_file_size = 32768
|
||||
|
||||
local NOTEBOOK_PATH = "PRIVATE_DATA/notebook"
|
||||
local blacklist = utility.enumerate{ ".git", } -- accelerate processing by ignoring .git directory
|
||||
local extension_whitelist = utility.enumerate{ "md", }
|
||||
|
||||
|
||||
local timing = {}
|
||||
local function display_timing(n)
|
||||
local function _display(n)
|
||||
local time = timing[n]
|
||||
local previous = timing[n - 1]
|
||||
|
||||
local delta = time.time - previous.time
|
||||
if delta >= 2*60*60 then -- 2 hours
|
||||
delta = tostring(math.floor(delta/(60*60/10))/10) .. " hours"
|
||||
elseif delta >= 120 then -- 2 minutes
|
||||
delta = tostring(math.floor(delta/(60/10))/10) .. " minutes"
|
||||
else
|
||||
delta = tostring(delta) .. " seconds"
|
||||
end
|
||||
|
||||
print(delta, previous.label)
|
||||
end
|
||||
|
||||
if n then
|
||||
_display(n)
|
||||
else
|
||||
print("All measured timings:")
|
||||
for i = 2, #timing do
|
||||
_display(i)
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
local function mark_timing(label)
|
||||
timing[#timing + 1] = { label = label, time = os.time(), }
|
||||
if #timing > 1 then
|
||||
display_timing(#timing)
|
||||
end
|
||||
end
|
||||
|
||||
-- strip YAML frontmatter (if present)
|
||||
-- can error, will return nil & error message
|
||||
local function strip_frontmatter(text)
|
||||
local tab = text:split("\n")
|
||||
if tab[1] == "---" then
|
||||
table.remove(tab, 1)
|
||||
while true do
|
||||
local done = tab[1] == "---"
|
||||
table.remove(tab, 1)
|
||||
if done then
|
||||
return table.concat(tab, "\n")
|
||||
elseif #tab < 1 then
|
||||
return nil, "Invalid YAML frontmatter."
|
||||
end
|
||||
end
|
||||
end
|
||||
return text
|
||||
end
|
||||
|
||||
local tree
|
||||
tree = function(path, fn)
|
||||
utility.list(path or ".", function(path_name)
|
||||
if path_name == ".git" then return end
|
||||
if utility.is_file(path_name) then
|
||||
fn(path_name)
|
||||
else
|
||||
tree(path .. utility.path_separator .. path_name, fn)
|
||||
end
|
||||
end)
|
||||
end
|
||||
|
||||
local function leftpad(text, length)
|
||||
return string.rep("0", length - #(tostring(text))) .. text
|
||||
@@ -89,36 +27,22 @@ end
|
||||
local file_list = {}
|
||||
local embeddings = {}
|
||||
|
||||
mark_timing("Assembling file list.")
|
||||
tree(PATH, function(file_name)
|
||||
local path, name, extension = utility.split_path_components(file_name)
|
||||
if not extension then
|
||||
return
|
||||
end
|
||||
|
||||
if not whitelist[extension] then
|
||||
return
|
||||
end
|
||||
|
||||
timing.mark("Assembling file list.")
|
||||
utility.tree(NOTEBOOK_PATH, {
|
||||
blacklist = blacklist,
|
||||
extension_whitelist = extension_whitelist,
|
||||
}, function(file_name)
|
||||
file_list[#file_list + 1] = file_name
|
||||
end)
|
||||
|
||||
mark_timing("Generating embeddings.")
|
||||
timing.mark("Generating embeddings.")
|
||||
for i = 1, #file_list do
|
||||
local function _run()
|
||||
local file_name = file_list[i]
|
||||
|
||||
-- stripping YAML frontmatter before generating embeddings
|
||||
local file_contents = utility.open(file_name, "r", function(file)
|
||||
return file:read("*all")
|
||||
end)
|
||||
|
||||
local tmp_file_contents, error_message = strip_frontmatter(file_contents)
|
||||
if tmp_file_contents == nil then
|
||||
print("ERROR: " .. file_name .. " " .. error_message)
|
||||
tmp_file_contents = file_contents
|
||||
end
|
||||
|
||||
local file_contents = utility.read_file(file_name)
|
||||
local tmp_file_contents = text_processing.strip_frontmatter(file_contents)
|
||||
if #tmp_file_contents == 0 then
|
||||
print(file_name .. "\n is empty and will be skipped.")
|
||||
return
|
||||
@@ -129,14 +53,7 @@ for i = 1, #file_list do
|
||||
tmp_file_contents = tmp_file_contents:sub(1, maximum_file_size)
|
||||
end
|
||||
|
||||
local tmp_file_name = utility.tmp_file_name()
|
||||
utility.open(tmp_file_name, "w", function(file)
|
||||
file:write(tmp_file_contents)
|
||||
end)
|
||||
|
||||
local output = utility.capture_safe("cat " .. tmp_file_name:enquote() .. " | ollama run " .. embedding_model)
|
||||
os.execute("rm " .. tmp_file_name)
|
||||
output = output:sub(1, -2) -- strip extra newline from utility.capture_safe
|
||||
local output = utility.llm_prompt(tmp_file_contents, embedding_model)
|
||||
|
||||
if #output == 0 then
|
||||
print("Warning: " .. file_name .. " did not have embeddings generated.")
|
||||
@@ -157,13 +74,9 @@ for i = 1, #file_list do
|
||||
_run()
|
||||
end
|
||||
|
||||
mark_timing("Outputting embeddings in JSON.")
|
||||
utility.open("PRIVATE_DATA/embeddings.json", "w", function(file)
|
||||
local output = json.encode(embeddings, { indent = true })
|
||||
file:write(output)
|
||||
file:write("\n")
|
||||
end)
|
||||
timing.mark("Outputting embeddings in JSON.")
|
||||
utility.save_data(embeddings, "PRIVATE_DATA/embeddings.json")
|
||||
|
||||
mark_timing("Finished.")
|
||||
timing.mark("Finished.")
|
||||
print("")
|
||||
display_timing()
|
||||
timing.display()
|
||||
+371
@@ -0,0 +1,371 @@
|
||||
local _tl_compat; if (tonumber((_VERSION or ''):match('[%d.]*$')) or 0) < 5.3 then local p, m = pcall(require, 'compat53.module'); if p then _tl_compat = m end end; local math = _tl_compat and _tl_compat.math or math; local string = _tl_compat and _tl_compat.string or string; local table = _tl_compat and _tl_compat.table or table
|
||||
local inspect = {Options = {}, }
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
inspect._VERSION = 'inspect.lua 3.1.0'
|
||||
inspect._URL = 'http://github.com/kikito/inspect.lua'
|
||||
inspect._DESCRIPTION = 'human-readable representations of tables'
|
||||
inspect._LICENSE = [[
|
||||
MIT LICENSE
|
||||
|
||||
Copyright (c) 2022 Enrique García Cota
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a
|
||||
copy of this software and associated documentation files (the
|
||||
"Software"), to deal in the Software without restriction, including
|
||||
without limitation the rights to use, copy, modify, merge, publish,
|
||||
distribute, sublicense, and/or sell copies of the Software, and to
|
||||
permit persons to whom the Software is furnished to do so, subject to
|
||||
the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included
|
||||
in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS
|
||||
OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||
IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY
|
||||
CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT,
|
||||
TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE
|
||||
SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
]]
|
||||
inspect.KEY = setmetatable({}, { __tostring = function() return 'inspect.KEY' end })
|
||||
inspect.METATABLE = setmetatable({}, { __tostring = function() return 'inspect.METATABLE' end })
|
||||
|
||||
local tostring = tostring
|
||||
local rep = string.rep
|
||||
local match = string.match
|
||||
local char = string.char
|
||||
local gsub = string.gsub
|
||||
local fmt = string.format
|
||||
|
||||
local _rawget
|
||||
if rawget then
|
||||
_rawget = rawget
|
||||
else
|
||||
_rawget = function(t, k) return t[k] end
|
||||
end
|
||||
|
||||
local function rawpairs(t)
|
||||
return next, t, nil
|
||||
end
|
||||
|
||||
|
||||
|
||||
local function smartQuote(str)
|
||||
if match(str, '"') and not match(str, "'") then
|
||||
return "'" .. str .. "'"
|
||||
end
|
||||
return '"' .. gsub(str, '"', '\\"') .. '"'
|
||||
end
|
||||
|
||||
|
||||
local shortControlCharEscapes = {
|
||||
["\a"] = "\\a", ["\b"] = "\\b", ["\f"] = "\\f", ["\n"] = "\\n",
|
||||
["\r"] = "\\r", ["\t"] = "\\t", ["\v"] = "\\v", ["\127"] = "\\127",
|
||||
}
|
||||
local longControlCharEscapes = { ["\127"] = "\127" }
|
||||
for i = 0, 31 do
|
||||
local ch = char(i)
|
||||
if not shortControlCharEscapes[ch] then
|
||||
shortControlCharEscapes[ch] = "\\" .. i
|
||||
longControlCharEscapes[ch] = fmt("\\%03d", i)
|
||||
end
|
||||
end
|
||||
|
||||
local function escape(str)
|
||||
return (gsub(gsub(gsub(str, "\\", "\\\\"),
|
||||
"(%c)%f[0-9]", longControlCharEscapes),
|
||||
"%c", shortControlCharEscapes))
|
||||
end
|
||||
|
||||
local luaKeywords = {
|
||||
['and'] = true,
|
||||
['break'] = true,
|
||||
['do'] = true,
|
||||
['else'] = true,
|
||||
['elseif'] = true,
|
||||
['end'] = true,
|
||||
['false'] = true,
|
||||
['for'] = true,
|
||||
['function'] = true,
|
||||
['goto'] = true,
|
||||
['if'] = true,
|
||||
['in'] = true,
|
||||
['local'] = true,
|
||||
['nil'] = true,
|
||||
['not'] = true,
|
||||
['or'] = true,
|
||||
['repeat'] = true,
|
||||
['return'] = true,
|
||||
['then'] = true,
|
||||
['true'] = true,
|
||||
['until'] = true,
|
||||
['while'] = true,
|
||||
}
|
||||
|
||||
local function isIdentifier(str)
|
||||
return type(str) == "string" and
|
||||
not not str:match("^[_%a][_%a%d]*$") and
|
||||
not luaKeywords[str]
|
||||
end
|
||||
|
||||
local flr = math.floor
|
||||
local function isSequenceKey(k, sequenceLength)
|
||||
return type(k) == "number" and
|
||||
flr(k) == k and
|
||||
1 <= (k) and
|
||||
k <= sequenceLength
|
||||
end
|
||||
|
||||
local defaultTypeOrders = {
|
||||
['number'] = 1, ['boolean'] = 2, ['string'] = 3, ['table'] = 4,
|
||||
['function'] = 5, ['userdata'] = 6, ['thread'] = 7,
|
||||
}
|
||||
|
||||
local function sortKeys(a, b)
|
||||
local ta, tb = type(a), type(b)
|
||||
|
||||
|
||||
if ta == tb and (ta == 'string' or ta == 'number') then
|
||||
return (a) < (b)
|
||||
end
|
||||
|
||||
local dta = defaultTypeOrders[ta] or 100
|
||||
local dtb = defaultTypeOrders[tb] or 100
|
||||
|
||||
|
||||
return dta == dtb and ta < tb or dta < dtb
|
||||
end
|
||||
|
||||
local function getKeys(t)
|
||||
|
||||
local seqLen = 1
|
||||
while _rawget(t, seqLen) ~= nil do
|
||||
seqLen = seqLen + 1
|
||||
end
|
||||
seqLen = seqLen - 1
|
||||
|
||||
local keys, keysLen = {}, 0
|
||||
for k in rawpairs(t) do
|
||||
if not isSequenceKey(k, seqLen) then
|
||||
keysLen = keysLen + 1
|
||||
keys[keysLen] = k
|
||||
end
|
||||
end
|
||||
table.sort(keys, sortKeys)
|
||||
return keys, keysLen, seqLen
|
||||
end
|
||||
|
||||
local function countCycles(x, cycles)
|
||||
if type(x) == "table" then
|
||||
if cycles[x] then
|
||||
cycles[x] = cycles[x] + 1
|
||||
else
|
||||
cycles[x] = 1
|
||||
for k, v in rawpairs(x) do
|
||||
countCycles(k, cycles)
|
||||
countCycles(v, cycles)
|
||||
end
|
||||
countCycles(getmetatable(x), cycles)
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
local function makePath(path, a, b)
|
||||
local newPath = {}
|
||||
local len = #path
|
||||
for i = 1, len do newPath[i] = path[i] end
|
||||
|
||||
newPath[len + 1] = a
|
||||
newPath[len + 2] = b
|
||||
|
||||
return newPath
|
||||
end
|
||||
|
||||
|
||||
local function processRecursive(process,
|
||||
item,
|
||||
path,
|
||||
visited)
|
||||
if item == nil then return nil end
|
||||
if visited[item] then return visited[item] end
|
||||
|
||||
local processed = process(item, path)
|
||||
if type(processed) == "table" then
|
||||
local processedCopy = {}
|
||||
visited[item] = processedCopy
|
||||
local processedKey
|
||||
|
||||
for k, v in rawpairs(processed) do
|
||||
processedKey = processRecursive(process, k, makePath(path, k, inspect.KEY), visited)
|
||||
if processedKey ~= nil then
|
||||
processedCopy[processedKey] = processRecursive(process, v, makePath(path, processedKey), visited)
|
||||
end
|
||||
end
|
||||
|
||||
local mt = processRecursive(process, getmetatable(processed), makePath(path, inspect.METATABLE), visited)
|
||||
if type(mt) ~= 'table' then mt = nil end
|
||||
setmetatable(processedCopy, mt)
|
||||
processed = processedCopy
|
||||
end
|
||||
return processed
|
||||
end
|
||||
|
||||
local function puts(buf, str)
|
||||
buf.n = buf.n + 1
|
||||
buf[buf.n] = str
|
||||
end
|
||||
|
||||
|
||||
|
||||
local Inspector = {}
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
local Inspector_mt = { __index = Inspector }
|
||||
|
||||
local function tabify(inspector)
|
||||
puts(inspector.buf, inspector.newline .. rep(inspector.indent, inspector.level))
|
||||
end
|
||||
|
||||
function Inspector:getId(v)
|
||||
local id = self.ids[v]
|
||||
local ids = self.ids
|
||||
if not id then
|
||||
local tv = type(v)
|
||||
id = (ids[tv] or 0) + 1
|
||||
ids[v], ids[tv] = id, id
|
||||
end
|
||||
return tostring(id)
|
||||
end
|
||||
|
||||
function Inspector:putValue(v)
|
||||
local buf = self.buf
|
||||
local tv = type(v)
|
||||
if tv == 'string' then
|
||||
puts(buf, smartQuote(escape(v)))
|
||||
elseif tv == 'number' or tv == 'boolean' or tv == 'nil' or
|
||||
tv == 'cdata' or tv == 'ctype' then
|
||||
puts(buf, tostring(v))
|
||||
elseif tv == 'table' and not self.ids[v] then
|
||||
local t = v
|
||||
|
||||
if t == inspect.KEY or t == inspect.METATABLE then
|
||||
puts(buf, tostring(t))
|
||||
elseif self.level >= self.depth then
|
||||
puts(buf, '{...}')
|
||||
else
|
||||
if self.cycles[t] > 1 then puts(buf, fmt('<%d>', self:getId(t))) end
|
||||
|
||||
local keys, keysLen, seqLen = getKeys(t)
|
||||
|
||||
puts(buf, '{')
|
||||
self.level = self.level + 1
|
||||
|
||||
for i = 1, seqLen + keysLen do
|
||||
if i > 1 then puts(buf, ',') end
|
||||
if i <= seqLen then
|
||||
puts(buf, ' ')
|
||||
self:putValue(t[i])
|
||||
else
|
||||
local k = keys[i - seqLen]
|
||||
tabify(self)
|
||||
if isIdentifier(k) then
|
||||
puts(buf, k)
|
||||
else
|
||||
puts(buf, "[")
|
||||
self:putValue(k)
|
||||
puts(buf, "]")
|
||||
end
|
||||
puts(buf, ' = ')
|
||||
self:putValue(t[k])
|
||||
end
|
||||
end
|
||||
|
||||
local mt = getmetatable(t)
|
||||
if type(mt) == 'table' then
|
||||
if seqLen + keysLen > 0 then puts(buf, ',') end
|
||||
tabify(self)
|
||||
puts(buf, '<metatable> = ')
|
||||
self:putValue(mt)
|
||||
end
|
||||
|
||||
self.level = self.level - 1
|
||||
|
||||
if keysLen > 0 or type(mt) == 'table' then
|
||||
tabify(self)
|
||||
elseif seqLen > 0 then
|
||||
puts(buf, ' ')
|
||||
end
|
||||
|
||||
puts(buf, '}')
|
||||
end
|
||||
|
||||
else
|
||||
puts(buf, fmt('<%s %d>', tv, self:getId(v)))
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
|
||||
|
||||
function inspect.inspect(root, options)
|
||||
options = options or {}
|
||||
|
||||
local depth = options.depth or (math.huge)
|
||||
local newline = options.newline or '\n'
|
||||
local indent = options.indent or ' '
|
||||
local process = options.process
|
||||
|
||||
if process then
|
||||
root = processRecursive(process, root, {}, {})
|
||||
end
|
||||
|
||||
local cycles = {}
|
||||
countCycles(root, cycles)
|
||||
|
||||
local inspector = setmetatable({
|
||||
buf = { n = 0 },
|
||||
ids = {},
|
||||
cycles = cycles,
|
||||
depth = depth,
|
||||
level = 0,
|
||||
newline = newline,
|
||||
indent = indent,
|
||||
}, Inspector_mt)
|
||||
|
||||
inspector:putValue(root)
|
||||
|
||||
return table.concat(inspector.buf)
|
||||
end
|
||||
|
||||
setmetatable(inspect, {
|
||||
__call = function(_, root, options)
|
||||
return inspect.inspect(root, options)
|
||||
end,
|
||||
})
|
||||
|
||||
return inspect
|
||||
+82
@@ -0,0 +1,82 @@
|
||||
-- USAGE:
|
||||
-- -- set arbitrary log levels to be printed immediately
|
||||
-- log{ info = true, warning = true, ducksauce = true, }
|
||||
-- -- disable storage of messages depending on log level
|
||||
-- log{ error = false, verbose = false, }
|
||||
-- -- send anything (except nil) to a log level
|
||||
-- log("bacom", true, "text", 5, function() end, {})
|
||||
-- -- print all messages saved at a particular log level
|
||||
-- log("error")
|
||||
-- -- toggle ALL logging on and off with a boolean
|
||||
-- log(false) -- this deletes all stored messages
|
||||
-- -- set a maximum number of stored messages per log level
|
||||
-- log(50) -- warning: reaching large limits makes logging expensive
|
||||
-- -- check if a log level is being displayed
|
||||
-- if log.arbitrary_exit then log("arbitrary_exit") os.exit(1) end
|
||||
-- -- do anything to the stored messages (this example deletes them all)
|
||||
-- log(function(messages) return {} end)
|
||||
|
||||
local microlog = {} -- display_levels are stored directly here
|
||||
local stored_messages = {}
|
||||
local enable_logging = true
|
||||
local message_limit = math.huge
|
||||
|
||||
return setmetatable(microlog, {
|
||||
__call = function(self, options, ...)
|
||||
local options_type = type(options)
|
||||
|
||||
if (options_type == "string") and enable_logging then
|
||||
if microlog[options] == false then return end
|
||||
if not stored_messages[options] then stored_messages[options] = {} end
|
||||
local message_table = stored_messages[options]
|
||||
|
||||
-- turn log message into text
|
||||
local tab = {...}
|
||||
for i = 1, #tab do
|
||||
tab[i] = tostring(tab[i])
|
||||
end
|
||||
local current_message = table.concat(tab, "\t")
|
||||
|
||||
if #current_message == 0 then
|
||||
-- print all stored_messages of specified level
|
||||
for i = 1, #message_table do
|
||||
print(message_table[i])
|
||||
end
|
||||
|
||||
else
|
||||
-- store message (and print if in display_levels)
|
||||
message_table[#message_table + 1] = current_message
|
||||
if #message_table > message_limit then table.remove(message_table, 1) end
|
||||
if microlog[options] then print(current_message) end
|
||||
end
|
||||
|
||||
-- set display_levels
|
||||
elseif options_type == "table" then
|
||||
for k,v in pairs(options) do
|
||||
microlog[k] = v
|
||||
end
|
||||
|
||||
elseif options_type == "number" then
|
||||
message_limit = options
|
||||
for k,v in pairs(stored_messages) do
|
||||
while #v > message_limit do
|
||||
table.remove(v, 1)
|
||||
end
|
||||
end
|
||||
|
||||
-- turn all logging off and delete stored messages; or turn it on
|
||||
elseif options_type == "boolean" then
|
||||
if options then
|
||||
enable_logging = true
|
||||
else
|
||||
enable_logging = false
|
||||
stored_messages = {}
|
||||
end
|
||||
|
||||
-- arbitrary access :D
|
||||
elseif options_type == "function" then
|
||||
local result = options(stored_messages)
|
||||
if type(result) == "table" then stored_messages = result end
|
||||
end
|
||||
end
|
||||
})
|
||||
@@ -0,0 +1,44 @@
|
||||
local prompts = {}
|
||||
|
||||
prompts.synopsis_prompt = [[
|
||||
Generate a lengthy novel synopsis from the following:
|
||||
]]
|
||||
|
||||
prompts.scoring_prompt = [[
|
||||
You are evaluating novel synopses for development priority. The goal is NOT to judge writing quality, grammar, or polish. The synopsis is only a rough idea. Assign an integer score from 1–100 for each category. Be extremely harsh with your scoring.
|
||||
|
||||
1. Hook
|
||||
2. Originality
|
||||
3. Memorability
|
||||
4. Expansion Potential
|
||||
5. Conflict Potential
|
||||
6. Character Potential
|
||||
7. Worldbuilding Potential
|
||||
8. Emotional Potential
|
||||
9. Curiosity
|
||||
10. Overall Promise
|
||||
|
||||
Guidelines:
|
||||
- Avoid clustering scores near the middle. Use the full 1–100 range.
|
||||
- A score around 50 represents an average publishable premise.
|
||||
- Scores above 85 should be rare and reserved for genuinely exceptional ideas.
|
||||
- Scores below 25 should represent ideas with major conceptual weaknesses.
|
||||
- Return only valid JSON.
|
||||
|
||||
Output format:
|
||||
|
||||
{
|
||||
"hook": 0,
|
||||
"originality": 0,
|
||||
"memorability": 0,
|
||||
"expansion_potential": 0,
|
||||
"conflict_potential": 0,
|
||||
"character_potential": 0,
|
||||
"worldbuilding_potential": 0,
|
||||
"emotional_potential": 0,
|
||||
"curiosity": 0,
|
||||
"overall_promise": 0
|
||||
}
|
||||
]]
|
||||
|
||||
return prompts
|
||||
@@ -0,0 +1,35 @@
|
||||
local text_processing = {}
|
||||
|
||||
-- strip YAML frontmatter (if present)
|
||||
-- on error, will return nil & error message
|
||||
text_processing.strip_frontmatter = function(text)
|
||||
local tab = text:split("\n")
|
||||
if tab[1] == "---" then
|
||||
table.remove(tab, 1)
|
||||
while true do
|
||||
local done = tab[1] == "---"
|
||||
table.remove(tab, 1)
|
||||
if done then
|
||||
return table.concat(tab, "\n")
|
||||
elseif #tab < 1 then
|
||||
return nil, "Invalid YAML frontmatter."
|
||||
end
|
||||
end
|
||||
end
|
||||
return text
|
||||
end
|
||||
|
||||
-- strip Markdown formatting of a JSON block
|
||||
-- just returns the string if it doesn't match
|
||||
text_processing.strip_markdown_codeblock = function(text)
|
||||
local tab = text:split("\n")
|
||||
if tab[1] == "```json" then
|
||||
table.remove(tab, 1)
|
||||
table.remove(tab, #tab)
|
||||
return table.concat(tab, "\n")
|
||||
else
|
||||
return text
|
||||
end
|
||||
end
|
||||
|
||||
return text_processing
|
||||
@@ -0,0 +1,52 @@
|
||||
local timing = {}
|
||||
|
||||
local function human_readable_time(delta)
|
||||
if delta >= 2*7*24*60*60 then -- if more than 2 weeks
|
||||
delta = tostring(math.floor(delta/(7*24*60*60))/10) .. " weeks"
|
||||
elseif delta >= 2*24*60*60 then -- if more than 2 days
|
||||
delta = tostring(math.floor(delta/(24*60*60))/10) .. " days"
|
||||
elseif delta >= 2*60*60 then -- if more than 2 hours
|
||||
delta = tostring(math.floor(delta/(60*60/10))/10) .. " hours"
|
||||
elseif delta >= 2*60 then -- if more then 2 minutes
|
||||
delta = tostring(math.floor(delta/(60/10))/10) .. " minutes"
|
||||
else
|
||||
delta = tostring(delta) .. " seconds"
|
||||
end
|
||||
return delta
|
||||
end
|
||||
|
||||
timing.display = function(n)
|
||||
local function _display(n)
|
||||
local time = timing[n]
|
||||
local previous = timing[n - 1]
|
||||
|
||||
local delta = human_readable_time(time.time - previous.time)
|
||||
|
||||
print(delta, previous.label)
|
||||
end
|
||||
|
||||
if n then
|
||||
_display(n)
|
||||
else
|
||||
print("All measured timings:")
|
||||
for i = 2, #timing do
|
||||
_display(i)
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
timing.mark = function(label)
|
||||
timing[#timing + 1] = { label = label, time = os.time(), }
|
||||
if #timing > 1 then
|
||||
timing.display(#timing)
|
||||
end
|
||||
print("", "", label)
|
||||
end
|
||||
|
||||
timing.estimate = function(current_position, total_operations)
|
||||
local delta = os.time() - timing[#timing].time
|
||||
local estimate = delta * total_operations / current_position - delta
|
||||
return human_readable_time(estimate)
|
||||
end
|
||||
|
||||
return timing
|
||||
+249
-49
@@ -18,7 +18,7 @@ if package.config:sub(1, 1) == "\\" then
|
||||
}
|
||||
else
|
||||
utility = {
|
||||
OS = "UNIX-like",
|
||||
OS = "Linux",
|
||||
path_separator = "/",
|
||||
temp_directory = "/tmp/",
|
||||
commands = {
|
||||
@@ -32,7 +32,7 @@ else
|
||||
}
|
||||
end
|
||||
|
||||
utility.version = "1.4.0"
|
||||
utility.version = "1.6.0-modified"
|
||||
-- WARNING: This will return "./" if the original script is called locally instead of with an absolute path!
|
||||
if arg[0] ~= nil then
|
||||
utility.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) -- inspired by discussion in https://stackoverflow.com/q/6380820
|
||||
@@ -40,6 +40,53 @@ else
|
||||
utility.path = "./"
|
||||
end
|
||||
|
||||
|
||||
|
||||
local function standard_library_addition(tab, name, func)
|
||||
if tab[name] then
|
||||
print("WARNING: " .. tab .. "." .. name .. " was defined by another library. lua-utility may encounter errors due to a differing implementation.")
|
||||
else
|
||||
tab[name] = func
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
|
||||
-- trim6 from Lua users wiki (best all-round pure Lua performance)
|
||||
standard_library_addition(string, "trim", function(s)
|
||||
return s:match'^()%s*$' and '' or s:match'^%s*(.*%S)'
|
||||
end)
|
||||
|
||||
standard_library_addition(string, "enquote", function(s)
|
||||
return "\"" .. s:gsub("\"", "\\\"") .. "\""
|
||||
end)
|
||||
|
||||
standard_library_addition(string, "gsplit", function(s, delimiter)
|
||||
local function escape_special_characters(s)
|
||||
local special_characters = "[()%%.[^$%]*+%-?]"
|
||||
if s == nil then return end
|
||||
return (s:gsub(special_characters, "%%%1"))
|
||||
end
|
||||
|
||||
delimiter = delimiter or ","
|
||||
if s:sub(-#delimiter) ~= delimiter then s = s .. delimiter end
|
||||
return s:gmatch("(.-)" .. escape_special_characters(delimiter))
|
||||
end)
|
||||
|
||||
standard_library_addition(string, "split", function(s, delimiter)
|
||||
local result = {}
|
||||
for item in s:gsplit(delimiter) do
|
||||
result[#result + 1] = item
|
||||
end
|
||||
return result
|
||||
end)
|
||||
|
||||
utility.leftpad = function(text, length, character)
|
||||
return string.rep(character or " ", length - #(tostring(text))) .. text
|
||||
end
|
||||
|
||||
|
||||
|
||||
utility.require = function(...)
|
||||
-- if libraries adjacent to this one aren't already loadable, make sure they are!
|
||||
if not package.path:find(utility.path, 1, true) then
|
||||
@@ -105,47 +152,6 @@ end
|
||||
|
||||
|
||||
|
||||
local function standard_library_addition(tab, name, func)
|
||||
if tab[name] then
|
||||
print("WARNING: " .. tab .. "." .. name .. " was defined by another library. lua-utility may encounter errors due to a differing implementation.")
|
||||
else
|
||||
tab[name] = func
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
|
||||
-- trim6 from Lua users wiki (best all-round pure Lua performance)
|
||||
standard_library_addition(string, "trim", function(s)
|
||||
return s:match'^()%s*$' and '' or s:match'^%s*(.*%S)'
|
||||
end)
|
||||
|
||||
standard_library_addition(string, "enquote", function(s)
|
||||
return "\"" .. s:gsub("\"", "\\\"") .. "\""
|
||||
end)
|
||||
|
||||
standard_library_addition(string, "split", function(s, delimiter)
|
||||
local result = {}
|
||||
for item in s:gsplit(delimiter) do
|
||||
result[#result + 1] = item
|
||||
end
|
||||
return result
|
||||
end)
|
||||
|
||||
standard_library_addition(string, "gsplit", function(s, delimiter)
|
||||
local function escape_special_characters(s)
|
||||
local special_characters = "[()%%.[^$%]*+%-?]"
|
||||
if s == nil then return end
|
||||
return (s:gsub(special_characters, "%%%1"))
|
||||
end
|
||||
|
||||
delimiter = delimiter or ","
|
||||
if s:sub(-#delimiter) ~= delimiter then s = s .. delimiter end
|
||||
return s:gmatch("(.-)" .. escape_special_characters(delimiter))
|
||||
end)
|
||||
|
||||
|
||||
|
||||
-- modified from my fork of lume
|
||||
utility.uuid = function()
|
||||
local fn = function(x)
|
||||
@@ -215,7 +221,7 @@ utility.list = function(path, func)
|
||||
|
||||
local run = function(fn)
|
||||
for line in output:gmatch("[^\r\n]+") do -- thanks to https://stackoverflow.com/a/32847589
|
||||
if not (line == "." or line == "..") then
|
||||
if not ((line == ".") or (line == "..")) then
|
||||
fn(line)
|
||||
end
|
||||
end
|
||||
@@ -233,6 +239,54 @@ utility.ls = function(...)
|
||||
return utility.list(...)
|
||||
end
|
||||
|
||||
local tree
|
||||
tree = function(path, options, fn)
|
||||
if type(options) == "function" then
|
||||
fn = options
|
||||
options = {}
|
||||
end
|
||||
|
||||
utility.list(path or ".", function(path_name)
|
||||
if options.blacklist and options.blacklist[path_name] then return end
|
||||
if options.whitelist and (not options.whitelist[path_name]) then return end
|
||||
|
||||
if utility.is_file(path_name) then
|
||||
if options.extension_blacklist or options.extension_whitelist then
|
||||
local _, _, extension = utility.split_path_components(path_name)
|
||||
if options.extension_blacklist and options.extension_blacklist[extension] then return end
|
||||
if options.extension_whitelist and (not options.extension_whitelist[extension]) then return end
|
||||
end
|
||||
|
||||
fn(path_name)
|
||||
else
|
||||
tree(path .. utility.path_separator .. path_name, options, fn)
|
||||
end
|
||||
end)
|
||||
end
|
||||
utility.tree = function(path, options, fn)
|
||||
tree(path, options, function(path_name)
|
||||
if path_name:find(path) == 1 then
|
||||
fn(path_name)
|
||||
else
|
||||
fn(path .. utility.path_separator .. path_name)
|
||||
end
|
||||
end)
|
||||
end
|
||||
|
||||
utility.read_file = function(file_name)
|
||||
return utility.open(file_name, "r", function(file)
|
||||
return file:read("*all")
|
||||
end)
|
||||
end
|
||||
|
||||
utility.write_file = function(file_name, ...)
|
||||
local text = table.concat{...}
|
||||
return utility.open(file_name, "w", function(file)
|
||||
file:write(text)
|
||||
-- file:write("\n") -- I need to make sure /I/ handle this instead of trying to automate it
|
||||
end)
|
||||
end
|
||||
|
||||
utility.path_exists = function(file_name)
|
||||
local file = io.open(file_name, "r")
|
||||
if file then file:close() return true else return false end
|
||||
@@ -260,6 +314,16 @@ utility.file_size = function(file_path)
|
||||
return utility.open(file_path, "rb", function(file) return file:seek("end") end)
|
||||
end
|
||||
|
||||
utility.sha512sum = function(file_path)
|
||||
local sha512sum
|
||||
if (utility.OS == "Linux") or (utility.OS == "macOS") then
|
||||
sha512sum = utility.capture_safe("shasum -U -a 512 " .. file_path:enquote())
|
||||
elseif utility.OS == "Windows" then
|
||||
error("utility.sha512sum() not implemented for Windows.")
|
||||
end
|
||||
return sha512sum:sub(1, 128)
|
||||
end
|
||||
|
||||
|
||||
|
||||
utility.escape_quotes_and_escapes = function(input)
|
||||
@@ -301,7 +365,7 @@ utility.release_lock = function(file_path, lock_uuid)
|
||||
local lock_file_path = file_path .. ".lock"
|
||||
if lock_uuid then
|
||||
utility.open(lock_file_path, "r", function(file)
|
||||
if not file:read("*all") == lock_uuid then
|
||||
if not (file:read("*all") == lock_uuid) then
|
||||
error("\n\n Lock UUID changed while lock was obtained. Data loss may have occurred. \n\n")
|
||||
end
|
||||
end)
|
||||
@@ -311,10 +375,10 @@ end
|
||||
|
||||
|
||||
|
||||
local config_path = utility.path .. "config.json"
|
||||
local config, config_lock
|
||||
utility.get_config = function(skip_lock)
|
||||
if not config then
|
||||
local config_path = utility.path .. "config.json"
|
||||
if utility.is_file(config_path) then
|
||||
if not skip_lock then
|
||||
config_lock = utility.get_lock(config_path)
|
||||
@@ -332,7 +396,6 @@ end
|
||||
|
||||
utility.save_config = function()
|
||||
if config then
|
||||
local config_path = utility.path .. "config.json"
|
||||
if not config_lock then
|
||||
print("Warning: A config lock file was not established.")
|
||||
end
|
||||
@@ -348,12 +411,74 @@ utility.save_config = function()
|
||||
end
|
||||
end
|
||||
|
||||
utility.get_config_with_defaults = function(defaults)
|
||||
local config = utility.get_config()
|
||||
|
||||
local loop, changes_made
|
||||
loop = function(config_level, default_level)
|
||||
for k,v in pairs(default_level) do
|
||||
if not config_level[k] then
|
||||
config_level[k] = v
|
||||
changes_made = true
|
||||
elseif type(v) == "table" then
|
||||
loop(config_level[k], v)
|
||||
end
|
||||
end
|
||||
end
|
||||
loop(config, defaults)
|
||||
|
||||
if changes_made then
|
||||
utility.save_config()
|
||||
else
|
||||
utility.release_lock(config_path)
|
||||
end
|
||||
|
||||
return config
|
||||
end
|
||||
|
||||
|
||||
|
||||
local data_file_locations = {}
|
||||
utility.load_data = function(file_path)
|
||||
local data = utility.open(file_path, "r", function(data_file)
|
||||
local json = utility.require("dkjson")
|
||||
return json.decode(data_file:read("*all"))
|
||||
end)
|
||||
data_file_locations[data] = file_path
|
||||
return data
|
||||
end
|
||||
|
||||
utility.save_data = function(data, file_path)
|
||||
local keys, loop = {}
|
||||
loop = function(tab)
|
||||
if type(tab) == "table" then
|
||||
for k,v in pairs(tab) do
|
||||
if not (type(k) == "number") then
|
||||
keys[k] = true
|
||||
end
|
||||
loop(v)
|
||||
end
|
||||
end
|
||||
end
|
||||
loop(data)
|
||||
local order = {}
|
||||
for k in pairs(keys) do order[#order + 1] = k end
|
||||
table.sort(order)
|
||||
|
||||
file_path = file_path or data_file_locations[data]
|
||||
assert(file_path, "The object must have been loaded by utility.load_data or you must pass a path as the second argument.")
|
||||
utility.open(file_path, "w", function(data_file)
|
||||
local json = utility.require("dkjson")
|
||||
data_file:write(json.encode(data, { indent = true, keyorder = order, }))
|
||||
data_file:write("\n")
|
||||
end)
|
||||
end
|
||||
|
||||
|
||||
|
||||
utility.deepcopy = function(tab)
|
||||
local _type = type(tab)
|
||||
local copy
|
||||
if _type == "table" then
|
||||
if type(tab) == "table" then
|
||||
copy = {}
|
||||
for key, value in next, tab, nil do
|
||||
copy[utility.deepcopy(key)] = utility.deepcopy(value)
|
||||
@@ -423,4 +548,79 @@ utility.curl_read = function(download_url, curl_options)
|
||||
return file_contents
|
||||
end
|
||||
|
||||
|
||||
|
||||
utility.llm_prompt = function(text, model, stabilize)
|
||||
utility.required_program("ollama")
|
||||
|
||||
-- if stabilize == nil then stabilize = true end -- I may switch this back to defaulting on
|
||||
if type(text) == "table" then text = table.concat(text, "\n") end
|
||||
|
||||
local config = utility.get_config("read-only")
|
||||
model = model or (config.ollama and config.ollama.default_model) or "gemma4:12b-mlx"
|
||||
|
||||
local tmp_file_name = utility.tmp_file_name()
|
||||
utility.open(tmp_file_name, "w", function(file)
|
||||
file:write(text)
|
||||
end)
|
||||
|
||||
-- word wrap fucks up usability of output in other things badly
|
||||
local output = utility.capture_safe("cat " .. tmp_file_name:enquote() .. " | ollama run " .. model .. " --nowordwrap")
|
||||
-- in previous testing, this improved system stability at the cost of speed by ensuring memory was not overtaxed by repeated calls to an LLM
|
||||
if stabilize then os.execute("ollama stop " .. model) end
|
||||
|
||||
os.execute("rm " .. tmp_file_name)
|
||||
assert(output, "ollama failed to generate output")
|
||||
|
||||
local strip_reasoning = function(text)
|
||||
local reasoning_lines = {}
|
||||
local tab = text:split("\n")
|
||||
table.remove(tab, 1) -- remove "Thinking..."
|
||||
|
||||
while true do
|
||||
local done = tab[1] == "...done thinking."
|
||||
local line = table.remove(tab, 1)
|
||||
if done then
|
||||
table.remove(tab, 1) -- remove blank line after reasoning
|
||||
return table.concat(tab, "\n"), table.concat(reasoning_lines, "\n")
|
||||
elseif #tab < 1 then
|
||||
return text -- can only be reached if theere was no reasoning output
|
||||
else
|
||||
reasoning_lines[#reasoning_lines + 1] = line
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
output = output:sub(1, -2) -- strip extra newline from utility.capture_safe
|
||||
return strip_reasoning(output) -- this returns the text AND reasoning
|
||||
end
|
||||
|
||||
|
||||
|
||||
-- additionally returns total and count
|
||||
utility.mean = function(object)
|
||||
local total, count = 0, 0
|
||||
for _, value in pairs(object) do
|
||||
total = total + value
|
||||
count = count + 1
|
||||
end
|
||||
return total / count, total, count
|
||||
end
|
||||
|
||||
-- additionally returns total
|
||||
utility.median = function(object)
|
||||
local tab = {}
|
||||
for _, value in pairs(object) do
|
||||
tab[#tab + 1] = value
|
||||
end
|
||||
table.sort(tab)
|
||||
return tab[math.floor(#tab / 2)], #tab
|
||||
end
|
||||
|
||||
|
||||
|
||||
if (utility.OS == "Linux") and (utility.capture_safe("uname"):find("Darwin") == 1) then
|
||||
utility.OS = "macOS"
|
||||
end
|
||||
|
||||
return utility
|
||||
Executable
+19
@@ -0,0 +1,19 @@
|
||||
#!/usr/bin/env luajit
|
||||
|
||||
package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path
|
||||
local utility = require "utility"
|
||||
local json = utility.require("dkjson")
|
||||
|
||||
local search_path = arg[1]
|
||||
|
||||
local file_extensions = {}
|
||||
utility.tree(search_path, function(file_path)
|
||||
local _, _, extension = utility.split_path_components(file_path)
|
||||
if extension then
|
||||
file_extensions[extension] = true
|
||||
end
|
||||
end)
|
||||
|
||||
for extension in pairs(file_extensions) do
|
||||
print(extension)
|
||||
end
|
||||
Executable
+293
@@ -0,0 +1,293 @@
|
||||
#!/usr/bin/env luajit
|
||||
|
||||
package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path
|
||||
local utility = require "utility"
|
||||
|
||||
local json = utility.require("dkjson")
|
||||
local log = utility.require("log")
|
||||
local text_processing = utility.require("text_processing")
|
||||
local timing = utility.require("timing")
|
||||
|
||||
-- TODO there should be a check function for a lock so a warning/error can be dumped?
|
||||
local config = utility.get_config_with_defaults{
|
||||
models = {
|
||||
embedding = {
|
||||
model = "qwen3-embedding:0.6b",
|
||||
max_chunk_size = 32768,
|
||||
},
|
||||
initialized_sources = {},
|
||||
},
|
||||
}
|
||||
|
||||
log{
|
||||
info = true,
|
||||
warning = true,
|
||||
debug = false,
|
||||
-- files = true, -- debugging why the wrong files are selected
|
||||
-- sha = true, -- what the fuck is going on with sha sums?
|
||||
}
|
||||
|
||||
local memory_path = "PRIVATE_DATA" .. utility.path_separator .. "memory"
|
||||
local embeddings_file_path = memory_path .. utility.path_separator .. "+embeddings.json"
|
||||
local tmp_file_path = "PRIVATE_DATA" .. utility.path_separator .. ".tmp.2b65c19b-0883-49ca-8247-b1fe7760f922"
|
||||
|
||||
if not utility.path_exists(embeddings_file_path) then
|
||||
os.execute("mkdir -p " .. memory_path:enquote())
|
||||
utility.save_data({
|
||||
files = {},
|
||||
vectors = {},
|
||||
}, embeddings_file_path)
|
||||
end
|
||||
local embeddings = utility.load_data(embeddings_file_path)
|
||||
if log.debug then
|
||||
log("debug", "Embeddings loaded.", embeddings, embeddings.files, embeddings.vectors)
|
||||
local file_count = 0
|
||||
for k,v in pairs(embeddings.files) do
|
||||
file_count = file_count + 1
|
||||
end
|
||||
local vector_count = 0
|
||||
for k,v in pairs(embeddings.vectors) do
|
||||
vector_count = vector_count + 1
|
||||
end
|
||||
log("debug", file_count .. " files.")
|
||||
log("debug", vector_count .. " vectors.")
|
||||
-- os.exit(1)
|
||||
end
|
||||
|
||||
|
||||
|
||||
local refresh_file_list = function(source_name, data_source)
|
||||
timing.mark("Assembling file list for \"" .. source_name .. "\"")
|
||||
|
||||
local full_path = "PRIVATE_DATA" .. utility.path_separator .. data_source.path
|
||||
if data_source.initialize_command and (not config.models.initialized_sources[data_source.path]) then
|
||||
-- NOTE this is where we'd want to check/obtain a lock on the config so we can safely run this
|
||||
os.execute("mkdir -p " .. full_path:enquote() .. " && cd " .. full_path:enquote() .. " && " .. data_source.initialize_command)
|
||||
config.models.initialized_sources[data_source.path] = true
|
||||
utility.save_config()
|
||||
end
|
||||
if data_source.refresh_command then
|
||||
os.execute("cd " .. full_path:enquote() .. " && " .. data_source.refresh_command)
|
||||
end
|
||||
|
||||
local compiled_filters = {}
|
||||
if data_source.filters then
|
||||
for name, object in pairs(data_source.filters) do
|
||||
compiled_filters[name] = utility.enumerate(object)
|
||||
end
|
||||
end
|
||||
|
||||
local file_list = {}
|
||||
utility.tree("PRIVATE_DATA" .. utility.path_separator .. data_source.path, compiled_filters, function(file_name)
|
||||
log("files", file_name)
|
||||
file_list[#file_list + 1] = file_name
|
||||
end)
|
||||
|
||||
timing.mark("Finished assembling file list for \"" .. source_name .. "\"")
|
||||
return file_list
|
||||
end
|
||||
|
||||
-- returns nothing when too much text is sent
|
||||
local generate_embeddings = function(data_source, text)
|
||||
local max_chunk_size = data_source.max_chunk_size or config.models.embedding.max_chunk_size
|
||||
local model = data_source.embedding_model or config.models.embedding.model
|
||||
|
||||
if #text > max_chunk_size then
|
||||
return nil, "generate_embeddings() must only be passed appropriately-sized chunks!"
|
||||
end
|
||||
|
||||
local result = utility.llm_prompt(text, model)
|
||||
return json.decode(result)
|
||||
end
|
||||
|
||||
local make_chunks = function(text, chunk_size)
|
||||
assert(chunk_size, "make_chunks() requires chunk_size")
|
||||
|
||||
local half_chunk_size = math.floor(chunk_size / 2)
|
||||
local chunks = { text }
|
||||
|
||||
while #text > chunk_size do
|
||||
local first_chunk = text:sub(1, chunk_size)
|
||||
local overlap_chunk = text:sub(half_chunk_size, chunk_size + half_chunk_size - 1)
|
||||
|
||||
chunks[#chunks + 1] = first_chunk
|
||||
chunks[#chunks + 1] = overlap_chunk
|
||||
|
||||
text = text:sub(chunk_size)
|
||||
if (#text > half_chunk_size) and (not (#text > chunk_size)) then
|
||||
-- last chunk would be skipped if we didn't handle this here
|
||||
chunks[#chunks + 1] = text
|
||||
end
|
||||
end
|
||||
|
||||
return chunks
|
||||
end
|
||||
|
||||
-- returns nothing for empty files and errors
|
||||
local process_file = function(data_source, file_name)
|
||||
local text = utility.read_file(file_name)
|
||||
|
||||
if data_source.strip_frontmatter then
|
||||
text = text_processing.strip_frontmatter(text)
|
||||
end
|
||||
|
||||
if #text == 0 then
|
||||
log("empty", file_name .. "\n is empty and being skipped.")
|
||||
return
|
||||
end
|
||||
|
||||
local chunk_size = data_source.max_chunk_size or config.models.embedding.max_chunk_size
|
||||
local chunks = make_chunks(text, chunk_size)
|
||||
|
||||
if data_source.target_chunk_size then
|
||||
local extra_chunks = make_chunks(text, data_source.target_chunk_size)
|
||||
for i = 2, #extra_chunks do
|
||||
chunks[#chunks + 1] = extra_chunks[i]
|
||||
end
|
||||
end
|
||||
|
||||
local new_embeddings = {}
|
||||
for i = 1, #chunks do
|
||||
log("debug", "Embedding source length:", #chunks[i])
|
||||
new_embeddings[i] = generate_embeddings(data_source, chunks[i]) or {}
|
||||
end
|
||||
|
||||
if #new_embeddings[1] == 0 then
|
||||
if log.debug then
|
||||
log("debug", "Vector lengths:")
|
||||
for e = 1, #new_embeddings do
|
||||
log("debug", "", e, #new_embeddings[e])
|
||||
end
|
||||
end
|
||||
if #new_embeddings == 1 then
|
||||
-- Ollama very rarely errors with:
|
||||
-- Error: do embedding request: Post "http://127.0.0.1:53441/v1/embeddings": EOF
|
||||
-- but it is inconsistent and re-running will eventually fix it.
|
||||
log("warning", file_name .. "\n encountered an embedding error and will be skipped this run only.")
|
||||
return
|
||||
end
|
||||
|
||||
-- average all embeddings to make the core file embedding
|
||||
local count = #new_embeddings[2]
|
||||
for vector_index = 1, count do
|
||||
local total = 0
|
||||
for chunk = 2, #new_embeddings do
|
||||
total = total + new_embeddings[chunk][vector_index]
|
||||
end
|
||||
new_embeddings[1][vector_index] = total / count
|
||||
end
|
||||
end
|
||||
|
||||
return chunks, new_embeddings
|
||||
end
|
||||
|
||||
-- returns nothing for errors
|
||||
local memorize_file = function(data_source, file_name)
|
||||
local file_chunks, file_embeddings = process_file(data_source, file_name)
|
||||
if not file_chunks then return end
|
||||
|
||||
local file_sums = {}
|
||||
for i = 1, #file_chunks do
|
||||
local function loop()
|
||||
local text = file_chunks[i]
|
||||
local current_embedding = file_embeddings[i]
|
||||
|
||||
utility.write_file(tmp_file_path, text)
|
||||
|
||||
local sha512sum = utility.sha512sum(tmp_file_path)
|
||||
log("sha", "New sum? " .. sha512sum, tostring(embeddings.vectors[sha512sum]))
|
||||
file_sums[#file_sums + 1] = sha512sum
|
||||
if embeddings.vectors[sha512sum] then
|
||||
log("sha", "Detected previously extant sum, skipping.")
|
||||
return
|
||||
end
|
||||
|
||||
os.execute(utility.commands.move .. tmp_file_path:enquote()
|
||||
.. " " .. (memory_path .. utility.path_separator .. sha512sum):enquote())
|
||||
embeddings.vectors[sha512sum] = current_embedding
|
||||
log("sha", "Saved new sum!")
|
||||
end
|
||||
loop()
|
||||
end
|
||||
|
||||
return file_sums
|
||||
end
|
||||
|
||||
local delete_memories = function(sha512sum_list)
|
||||
-- if true then return end -- TEMP disabling removal to compare and verify function
|
||||
for i = 1, #sha512sum_list do
|
||||
local sha512sum = sha512sum_list[i]
|
||||
embeddings.vectors[sha512sum] = nil
|
||||
os.execute("rm " .. (memory_path .. utility.path_separator .. sha512sum):enquote())
|
||||
end
|
||||
end
|
||||
|
||||
local refresh_sources = function()
|
||||
os.execute("mkdir -p " .. memory_path:enquote())
|
||||
|
||||
timing.mark("Checking memory for missing embeddings.")
|
||||
utility.list(memory_path, function(file_name)
|
||||
log("files", file_name)
|
||||
if (file_name == ".DS_Store") or (file_name == "+embeddings.json") then return end
|
||||
|
||||
local file_path = memory_path .. utility.path_separator .. file_name
|
||||
local sha512sum = utility.sha512sum(file_path)
|
||||
log("sha", "New sum? " .. sha512sum, tostring(embeddings.vectors[sha512sum]))
|
||||
if not embeddings.vectors[sha512sum] then
|
||||
log("sha", "Saved new sum!")
|
||||
memorize_file({}, file_path)
|
||||
end
|
||||
end)
|
||||
timing.mark("Finished checking memory for missing embeddings.")
|
||||
|
||||
local sources = utility.load_data("PRIVATE_DATA/sources.json")
|
||||
for source_name, data_source in pairs(sources) do
|
||||
local file_list = refresh_file_list(source_name, data_source)
|
||||
|
||||
timing.mark("Generating embeddings from source \"" .. source_name .. "\"")
|
||||
for f = 1, #file_list do
|
||||
local file_name = file_list[f]
|
||||
local function loop()
|
||||
local sha512sum = utility.sha512sum(file_name)
|
||||
log("sha", file_name, sha512sum, "\n Sum present? " .. tostring(embeddings.vectors[sha512sum]))
|
||||
|
||||
if embeddings.vectors[sha512sum] then
|
||||
log("debug", file_name .. "\n has already been embedded, skipping.")
|
||||
-- add file reference if it was missing (also handles recognition of duplicate files)
|
||||
if not embeddings.files[file_name] then
|
||||
embeddings.files[file_name] = { sha512sum }
|
||||
end
|
||||
return
|
||||
end
|
||||
|
||||
if embeddings.files[file_name] then
|
||||
log("debug", file_name .. "\n Removing old embeddings for modified file.")
|
||||
delete_memories(embeddings.files[file_name])
|
||||
embeddings.files[file_name] = nil
|
||||
end
|
||||
|
||||
log("debug", file_name .. "\n Generating embeddings..")
|
||||
local file_sums = memorize_file(data_source, file_name)
|
||||
embeddings.files[file_name] = file_sums
|
||||
end
|
||||
loop()
|
||||
log("info", "Finished " .. utility.leftpad(f, #tostring(#file_list), "0")
|
||||
.. "/" .. #file_list
|
||||
.. " (" .. utility.leftpad(math.floor(f / #file_list * 100), 3, "0")
|
||||
.. "%) ETA: " .. timing.estimate(f, #file_list))
|
||||
end
|
||||
|
||||
timing.mark("Finished generating embeddings from source \"" .. source_name .. "\"")
|
||||
end
|
||||
|
||||
utility.save_data(embeddings)
|
||||
if utility.path_exists(tmp_file_path) then
|
||||
os.execute("rm " .. tmp_file_path:enquote())
|
||||
end
|
||||
end
|
||||
|
||||
refresh_sources()
|
||||
timing.mark("Finished.")
|
||||
print("")
|
||||
timing.display()
|
||||
log("warning")
|
||||
+83
-236
@@ -2,197 +2,58 @@
|
||||
|
||||
package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path
|
||||
local utility = require "utility"
|
||||
|
||||
local json = utility.require("dkjson")
|
||||
local prompts = utility.require("prompts")
|
||||
local text_processing = utility.require("text_processing")
|
||||
|
||||
local default_model = "gemma4:12b-mlx"
|
||||
local minimum_bytes = 1000
|
||||
local maximum_bytes = 40000
|
||||
|
||||
local synopsis_prompt = [[
|
||||
Generate a lengthy novel synopsis from the following:
|
||||
]]
|
||||
|
||||
local scoring_prompt = [[
|
||||
You are evaluating novel synopses for development priority. The goal is NOT to judge writing quality, grammar, or polish. The synopsis is only a rough idea. Assign an integer score from 1–100 for each category. Be extremely harsh with your scoring.
|
||||
|
||||
1. Hook
|
||||
2. Originality
|
||||
3. Memorability
|
||||
4. Expansion Potential
|
||||
5. Conflict Potential
|
||||
6. Character Potential
|
||||
7. Worldbuilding Potential
|
||||
8. Emotional Potential
|
||||
9. Curiosity
|
||||
10. Overall Promise
|
||||
|
||||
Guidelines:
|
||||
- Avoid clustering scores near the middle. Use the full 1–100 range.
|
||||
- A score around 50 represents an average publishable premise.
|
||||
- Scores above 85 should be rare and reserved for genuinely exceptional ideas.
|
||||
- Scores below 25 should represent ideas with major conceptual weaknesses.
|
||||
- Return only valid JSON.
|
||||
|
||||
Output format:
|
||||
|
||||
{
|
||||
"hook": 0,
|
||||
"originality": 0,
|
||||
"memorability": 0,
|
||||
"expansion_potential": 0,
|
||||
"conflict_potential": 0,
|
||||
"character_potential": 0,
|
||||
"worldbuilding_potential": 0,
|
||||
"emotional_potential": 0,
|
||||
"curiosity": 0,
|
||||
"overall_promise": 0
|
||||
local config = utility.get_config_with_defaults{
|
||||
models = {
|
||||
synopsis_generator = {
|
||||
model = "gemma4:12b-mlx",
|
||||
minimum_bytes = 1000,
|
||||
maximum_bytes = 40000,
|
||||
},
|
||||
},
|
||||
}
|
||||
]]
|
||||
|
||||
local read_all = function(file_name)
|
||||
return utility.open(file_name, "r", function(file)
|
||||
return file:read("*all")
|
||||
end)
|
||||
end
|
||||
local write_all = function(file_name, text)
|
||||
return utility.open(file_name, "w", function(file)
|
||||
file:write(text)
|
||||
file:write("\n")
|
||||
end)
|
||||
end
|
||||
local memory_path = "PRIVATE_DATA" .. utility.path_separator .. "memory"
|
||||
local embeddings_file_path = memory_path .. utility.path_separator .. "+embeddings.json"
|
||||
local data_path = "PRIVATE_DATA" .. utility.path_separator .. "synopses"
|
||||
local data_location = data_path .. utility.path_separator .. "+file_list.json"
|
||||
|
||||
local files
|
||||
if utility.path_exists("PRIVATE_DATA/file_list.json") then
|
||||
files = json.decode(read_all("PRIVATE_DATA/file_list.json"))
|
||||
else
|
||||
files = {}
|
||||
end
|
||||
local refresh_file_list = function()
|
||||
assert(utility.path_exists(embeddings_file_path), "Run \"./refresh_sources.lua\" to set up memory.")
|
||||
local embeddings = utility.load_data(embeddings_file_path)
|
||||
|
||||
-- strip YAML frontmatter (if present)
|
||||
-- can error, will return nil & error message
|
||||
local function strip_frontmatter(text)
|
||||
local tab = text:split("\n")
|
||||
if tab[1] == "---" then
|
||||
table.remove(tab, 1)
|
||||
while true do
|
||||
local done = tab[1] == "---"
|
||||
table.remove(tab, 1)
|
||||
if done then
|
||||
return table.concat(tab, "\n")
|
||||
elseif #tab < 1 then
|
||||
return nil, "Invalid YAML frontmatter."
|
||||
local minimum_bytes = config.models.synopsis_generator.minimum_bytes
|
||||
local maximum_bytes = config.models.synopsis_generator.maximum_bytes
|
||||
|
||||
local file_list = {}
|
||||
for file_name, sha512sum_list in pairs(embeddings.files) do
|
||||
local file_size = utility.file_size(file_name)
|
||||
if file_size >= minimum_bytes and file_size <= maximum_bytes then
|
||||
local text = text_processing.strip_frontmatter(utility.read_file(file_name))
|
||||
if #text >= minimum_bytes and #text <= maximum_bytes then
|
||||
file_list[#file_list + 1] = { file_name = file_name, }
|
||||
end
|
||||
end
|
||||
end
|
||||
return text
|
||||
end
|
||||
|
||||
-- strip Markdown formatting of a JSON block
|
||||
-- just returns the string if it doesn't match
|
||||
local function strip_markdown_codeblock(text)
|
||||
local tab = text:split("\n")
|
||||
if tab[1] == "```json" then
|
||||
table.remove(tab, 1)
|
||||
table.remove(tab, #tab)
|
||||
return table.concat(tab, "\n")
|
||||
else
|
||||
return text
|
||||
end
|
||||
end
|
||||
|
||||
local strip_reasoning = function(text, reasoning_lines)
|
||||
if not reasoning_lines then reasoning_lines = {} end
|
||||
local tab = text:split("\n")
|
||||
table.remove(tab, 1) -- remove "Thinking..."
|
||||
|
||||
while true do
|
||||
local done = tab[1] == "...done thinking."
|
||||
local line = table.remove(tab, 1)
|
||||
if done then
|
||||
table.remove(tab, 1) -- remove newline after end of thinking
|
||||
return table.concat(tab, "\n")
|
||||
elseif #tab < 1 then
|
||||
return text -- no reasoning output
|
||||
else
|
||||
reasoning_lines[#reasoning_lines + 1] = line -- export thinking lines
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
local send_prompt = function(text, model)
|
||||
if type(text) == "table" then text = table.concat(text, "\n") end
|
||||
|
||||
local tmp_file_name = utility.tmp_file_name()
|
||||
utility.open(tmp_file_name, "w", function(file)
|
||||
file:write(text)
|
||||
end)
|
||||
|
||||
-- word wrap breaks the raw output badly, so I need to implement my own for terminal output somehow
|
||||
local output = utility.capture_safe("cat " .. tmp_file_name:enquote() .. " | ollama run " .. (model or default_model) .. " --nowordwrap")
|
||||
os.execute("ollama stop " .. (model or default_model)) -- NOTE this makes things slower, but more stable
|
||||
os.execute("rm " .. tmp_file_name)
|
||||
if not output then error("ollama failed to generate output") end
|
||||
output = output:sub(1, -2) -- strip extra newline from utility.capture_safe
|
||||
|
||||
local thinking = {}
|
||||
output = strip_reasoning(output, thinking)
|
||||
|
||||
return output, thinking
|
||||
end
|
||||
|
||||
local tree
|
||||
tree = function(path, fn)
|
||||
utility.list(path or ".", function(path_name)
|
||||
local blacklist = {
|
||||
[".git"] = true,
|
||||
[".gitkeep"] = true,
|
||||
[".gitignore"] = true,
|
||||
[".DS_Store"] = true,
|
||||
}
|
||||
if blacklist[path_name] then return end
|
||||
if utility.is_file(path_name) then
|
||||
fn(path_name)
|
||||
else
|
||||
tree(path .. utility.path_separator .. path_name, fn)
|
||||
end
|
||||
end)
|
||||
end
|
||||
|
||||
local get_file_size = function(file_name)
|
||||
local file = io.open(file_name, "rb")
|
||||
if file then
|
||||
local size = file:seek("end")
|
||||
file:close()
|
||||
return size
|
||||
end
|
||||
end
|
||||
|
||||
local refresh_file_list = function()
|
||||
local blacklist = { -- I'm only blacklisting binary formats because funny results happen with really invalid texts
|
||||
"jpg", "mp4", "pdf", "png", "webp", "jpeg", "gif",
|
||||
} for _, name in ipairs(blacklist) do blacklist[name] = true end
|
||||
|
||||
local new_files_list = {}
|
||||
tree("PRIVATE_DATA/notebook", function(file_name)
|
||||
local _, _, extension = utility.split_path_components(file_name)
|
||||
if not blacklist[extension] then
|
||||
new_files_list[#new_files_list + 1] = file_name
|
||||
end
|
||||
end)
|
||||
files = new_files_list
|
||||
write_all("PRIVATE_DATA/file_list.json", json.encode(new_files_list, { indent = true }))
|
||||
utility.save_data(file_list, data_location)
|
||||
return file_list
|
||||
end
|
||||
|
||||
local generate_and_score = function(file_name, text)
|
||||
print("Writing synopsis...")
|
||||
local synopsis = send_prompt(synopsis_prompt .. text)
|
||||
local synopsis = utility.llm_prompt(prompts.synopsis_prompt .. text, config.models.synopsis_generator.model)
|
||||
print(synopsis)
|
||||
print("Scoring synopsis...")
|
||||
local scoring, thinking = send_prompt(scoring_prompt .. synopsis)
|
||||
local scoring, thinking = utility.llm_prompt(prompts.scoring_prompt .. synopsis, config.models.synopsis_generator.model)
|
||||
print(scoring)
|
||||
local scoring_decoded = json.decode(scoring) -- likely will not work because it consistently returns Markdown instead of JSON
|
||||
if not scoring_decoded then
|
||||
scoring_decoded = json.decode(strip_markdown_codeblock(scoring))
|
||||
scoring_decoded = json.decode(text_processing.strip_markdown_codeblock(scoring))
|
||||
end
|
||||
|
||||
local object = {
|
||||
@@ -201,96 +62,82 @@ local generate_and_score = function(file_name, text)
|
||||
thinking = thinking,
|
||||
}
|
||||
|
||||
write_all("PRIVATE_DATA/synopses/" .. utility.uuid() .. ".json", json.encode(object, { indent = true }))
|
||||
utility.save_data(object, data_path .. utility.path_separator .. utility.uuid() .. ".json")
|
||||
end
|
||||
|
||||
|
||||
|
||||
os.execute("mkdir -p PRIVATE_DATA/synopses")
|
||||
|
||||
if arg[1] == "refresh_file_list" then
|
||||
print("Refresing file list...")
|
||||
refresh_file_list()
|
||||
end
|
||||
|
||||
if arg[1] == "repair_synopsis_exports" then
|
||||
local path = "PRIVATE_DATA/synopses"
|
||||
utility.list(path, function(path_name)
|
||||
path_name = path .. utility.path_separator .. path_name
|
||||
if path_name:find("%.json") then
|
||||
local object = json.decode(read_all(path_name))
|
||||
if type(object.scoring) == "table" then return end -- don't fuck with working pieces
|
||||
local decoded = json.decode(strip_markdown_codeblock(object.scoring))
|
||||
if decoded then
|
||||
object.scoring = decoded
|
||||
write_all(path_name, json.encode(object, { indent = true, }))
|
||||
end
|
||||
end
|
||||
end)
|
||||
os.exit(0)
|
||||
end
|
||||
|
||||
if arg[1] == "export_ordered_list_of_prompts" then
|
||||
local path = "PRIVATE_DATA/synopses"
|
||||
local export_ordered_list_of_prompts = function()
|
||||
local items = {}
|
||||
local item_order = {}
|
||||
utility.list(path, function(path_name)
|
||||
local full_path = path .. utility.path_separator .. path_name
|
||||
if path_name:find("%.json") then
|
||||
local object = json.decode(read_all(full_path))
|
||||
items[path_name] = object
|
||||
if type(object.scoring) == "table" then
|
||||
local s = object.scoring
|
||||
local total_score = s.conflict_potential + s.emotional_potential + s.character_potential + s.worldbuilding_potential + s.expansion_potential + s.overall_promise + s.memorability + s.originality + s.curiosity + s.hook
|
||||
item_order[#item_order + 1] = { path_name = path_name, total_score = total_score, }
|
||||
end
|
||||
|
||||
utility.list(data_path, function(path_name)
|
||||
local full_path = data_path .. utility.path_separator .. path_name
|
||||
if full_path == data_location then return end
|
||||
|
||||
local object = utility.load_data(full_path)
|
||||
items[path_name] = object
|
||||
if type(object.scoring) == "table" then
|
||||
local mean_score, total_score = utility.mean(object.scoring)
|
||||
item_order[#item_order + 1] = { path_name = path_name, total_score = total_score, mean_score = mean_score, }
|
||||
end
|
||||
end)
|
||||
table.sort(item_order, function(A,B) return A.total_score > B.total_score end)
|
||||
-- for k,v in pairs(item_order) do print(v.path_name,v.total_score) end
|
||||
|
||||
table.sort(item_order, function(A,B) return A.mean_score > B.mean_score end)
|
||||
|
||||
local output = {
|
||||
"---",
|
||||
"title: Ordered Synopses (" .. #item_order .. " items)",
|
||||
"author: [\"Gemma4:12b-mlx\", \"Tangent\", \"Ollama\"]",
|
||||
"publisher: Tangent",
|
||||
"author: [\"" .. config.models.synopsis_generator.model .. "\", \"Tangent\", \"Ollama\"]",
|
||||
"publisher: \"synopsis_generator.lua\"",
|
||||
"---",
|
||||
"",
|
||||
}
|
||||
|
||||
for _, v in pairs(item_order) do
|
||||
local item = items[v.path_name]
|
||||
local text = item.synopsis
|
||||
local tab = text:split("\n")
|
||||
|
||||
for index, line in ipairs(tab) do
|
||||
if line:sub(1, 1) == "#" then
|
||||
tab[index] = "#" .. tab[index]
|
||||
-- ensure no output heading is an H1; make H6 bold text lines instead
|
||||
if line:sub(1, 7) == "###### " then
|
||||
tab[index] = "**" .. line:sub(8) .. "**"
|
||||
elseif line:sub(1, 1) == "#" then
|
||||
tab[index] = "#" .. line
|
||||
end
|
||||
end
|
||||
-- output[#output + 1] = "# " .. v.path_name .. " (" .. v.total_score .. ")\n\n" .. item.synopsis .. "\n"
|
||||
|
||||
output[#output + 1] = "# " .. v.path_name .. " (" .. v.total_score .. ")\n\n" .. table.concat(tab, "\n") .. "\n"
|
||||
output[#output + 1] = "## Scoring\n\n```json\n" .. json.encode(item.scoring, { indent = true, }) .. "\n```\n"
|
||||
end
|
||||
write_all("PRIVATE_DATA/Ordered Synopses.md", table.concat(output, "\n"))
|
||||
|
||||
utility.write_file("PRIVATE_DATA/Ordered Synopses.md", table.concat(output, "\n"), "\n")
|
||||
os.execute("pandoc \"PRIVATE_DATA/Ordered Synopses.md\" -o \"PRIVATE_DATA/Ordered Synopses.epub\"")
|
||||
end
|
||||
|
||||
|
||||
|
||||
os.execute("mkdir -p " .. data_path:enquote())
|
||||
|
||||
local file_list
|
||||
if utility.path_exists(data_location) then
|
||||
file_list = utility.load_data(data_location)
|
||||
end
|
||||
|
||||
if arg[1] == "refresh_file_list" then
|
||||
print("Refresing file list...")
|
||||
file_list = refresh_file_list()
|
||||
os.exit(0)
|
||||
elseif arg[1] == "export_ordered_list_of_prompts" then
|
||||
export_ordered_list_of_prompts()
|
||||
os.exit(0)
|
||||
end
|
||||
|
||||
print(#files .. " files to select from.")
|
||||
if #files == 0 then
|
||||
print("Run \"./synopsis_generator.lua refresh_file_list\" first.")
|
||||
os.exit(1)
|
||||
end
|
||||
assert(file_list and (#file_list > 0), "Run \"./synopsis_generator.lua refresh_file_list\" first.")
|
||||
|
||||
print(#file_list .. " files to select from.")
|
||||
while true do
|
||||
print("Selecting a file...")
|
||||
local file_name = files[math.random(1, #files)]
|
||||
local file_size = get_file_size(file_name)
|
||||
if file_size > minimum_bytes and file_size <= maximum_bytes then
|
||||
local text = read_all(file_name)
|
||||
text = strip_frontmatter(text)
|
||||
if #text > minimum_bytes and #text <= maximum_bytes then
|
||||
print(file_name .. " chosen.")
|
||||
generate_and_score(file_name, text) -- kind of the main function, innit?
|
||||
-- os.exit(0)
|
||||
end
|
||||
end
|
||||
local file_name = file_list[math.random(1, #file_list)].file_name
|
||||
local text = utility.read_file(file_name)
|
||||
print(file_name .. " chosen.")
|
||||
generate_and_score(file_name, text)
|
||||
end
|
||||
@@ -6,16 +6,9 @@ local json = utility.require("dkjson")
|
||||
|
||||
local file_name = arg[1]
|
||||
|
||||
print(utility.OS)
|
||||
|
||||
|
||||
local function strip_markdown(text)
|
||||
local tab = text:split("\n")
|
||||
table.remove(tab, 1) -- codeblock opening
|
||||
table.remove(tab, #tab) -- end of codeblock
|
||||
print(table.concat(tab, "\n"))
|
||||
end
|
||||
|
||||
|
||||
os.exit(0)
|
||||
|
||||
local prompt = [[Return JSON: category, tags (array), summary (short)]]
|
||||
-- local prompt = [[Return YAML: category, tags (array), summary (short)]]
|
||||
@@ -24,39 +17,6 @@ if prompt:sub(-1) ~= "\n" then
|
||||
prompt = prompt .. "\n\n"
|
||||
end
|
||||
|
||||
local file_contents = utility.open(file_name, "r", function(file)
|
||||
return file:read("*all")
|
||||
end)
|
||||
local file_contents = utility.read_file(file_name)
|
||||
prompt = prompt .. file_contents
|
||||
|
||||
local model = "gemma3:4b"
|
||||
-- local output = utility.capture_safe("ollama run " .. model .. " --nowordwrap " .. prompt:enquote())
|
||||
|
||||
-- local embedding_model = "nomic-embed-text"
|
||||
-- local output = utility.capture_safe("ollama run " .. embedding_model .. " " .. file_contents:enquote())
|
||||
|
||||
-- output = output:sub(1, -2) -- strip extra newline from utility.capture_safe
|
||||
|
||||
-- strip YAML frontmatter (if present)
|
||||
-- can error, will return nil & error message
|
||||
local function strip_frontmatter(text)
|
||||
local tab = text:split("\n")
|
||||
if tab[1] == "---" then
|
||||
table.remove(tab, 1)
|
||||
while true do
|
||||
local done = tab[1] == "---"
|
||||
table.remove(tab, 1)
|
||||
if done then
|
||||
return table.concat(tab, "\n")
|
||||
elseif #tab < 1 then
|
||||
return nil, "Invalid YAML frontmatter."
|
||||
end
|
||||
end
|
||||
end
|
||||
return text
|
||||
end
|
||||
|
||||
-- print(output)
|
||||
|
||||
print(strip_frontmatter(file_contents))
|
||||
-- print(strip_markdown(output))
|
||||
-- print(utility.llm_prompt(prompt))
|
||||
Reference in new issue
Block a user