Compare commits
11
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b245a29d3c | ||
|
|
fe19c8ccfb | ||
|
|
83edc35af9 | ||
|
|
e2238ec1dc | ||
|
|
cdc2890810 | ||
|
|
570f98ecc9 | ||
|
|
f15765f995 | ||
|
|
5260aa7ffb | ||
|
|
57c4fa7a6d | ||
|
|
719951a348 | ||
|
|
9b4d9bf497 |
No files matched your search
+2
-1
@@ -1,2 +1,3 @@
|
||||
.DS_Store
|
||||
notebook/**
|
||||
PRIVATE_DATA/**
|
||||
!PRIVATE_DATA/.gitkeep
|
||||
Whitespace-only changes.
@@ -0,0 +1,18 @@
|
||||
# LLM Tricks
|
||||
Trying to find *anything* useful to do with local LLMs.
|
||||
|
||||
All scripts work with data within `PRIVATE_DATA/` so that private data can't be
|
||||
accidentally committed.
|
||||
|
||||
###### `cosine_similarity.lua`
|
||||
Opens `embeddings.json` and creates `similarities.json` with a sorted list of
|
||||
comparisons.
|
||||
|
||||
###### `generate_embeddings.lua`
|
||||
Opens every file on its whitelist within the `notebook` directory, and generates
|
||||
embeddings (placed in `embeddings.json`). When files are too long, it truncates
|
||||
them.
|
||||
|
||||
###### `least_similar.lua`
|
||||
Opens `similarities.json`, reverses the sort order, and saves it as
|
||||
`differences.json`.
|
||||
Executable
+67
@@ -0,0 +1,67 @@
|
||||
#!/usr/bin/env luajit
|
||||
|
||||
package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path
|
||||
local utility = require "utility"
|
||||
local json = utility.require("dkjson")
|
||||
|
||||
local function normalizing_cosine_similarity(a, b)
|
||||
local dot, normalized_a, normalized_b = 0, 0, 0
|
||||
|
||||
for i = 1, #a do
|
||||
dot = dot + a[i] * b[i]
|
||||
normalized_a = normalized_a + a[i] * a[i]
|
||||
normalized_b = normalized_b + b[i] * b[i]
|
||||
end
|
||||
|
||||
normalized_a = math.sqrt(normalized_a)
|
||||
normalized_b = math.sqrt(normalized_b)
|
||||
|
||||
if normalized_a == 0 or normalized_b == 0 then
|
||||
return 0
|
||||
end
|
||||
|
||||
return dot / (normalized_a * normalized_b)
|
||||
end
|
||||
|
||||
local function cosine_similarity(a, b)
|
||||
local dot = 0
|
||||
|
||||
for i = 1, #a do
|
||||
dot = dot + a[i] * b[i]
|
||||
end
|
||||
|
||||
return dot
|
||||
end
|
||||
|
||||
|
||||
|
||||
local embeddings = utility.open("PRIVATE_DATA/embeddings.json", "r", function(file)
|
||||
return json.decode(file:read("*all"))
|
||||
end)
|
||||
|
||||
local embeddings_list = {}
|
||||
for file_name, tab in pairs(embeddings) do
|
||||
table.insert(embeddings_list, { file_name = file_name, vector = tab.vector, })
|
||||
end
|
||||
|
||||
local total_comparisons = #embeddings_list * (#embeddings_list - 1) / 2
|
||||
local comparisons_completed = 0
|
||||
|
||||
local relations_list = {}
|
||||
for i = 1, #embeddings_list - 1 do
|
||||
for j = i + 1, #embeddings_list do
|
||||
local a, b = embeddings_list[i], embeddings_list[j]
|
||||
local similarity = cosine_similarity(a.vector, b.vector)
|
||||
|
||||
table.insert(relations_list, { similarity = similarity, a = a.file_name, b = b.file_name, })
|
||||
|
||||
comparisons_completed = comparisons_completed + 1
|
||||
print(comparisons_completed .. "/" .. total_comparisons)
|
||||
end
|
||||
end
|
||||
|
||||
table.sort(relations_list, function(a, b) return a.similarity > b.similarity end)
|
||||
|
||||
utility.open("PRIVATE_DATA/similarities.json", "w", function(file)
|
||||
file:write(json.encode(relations_list, { indent = true }))
|
||||
end)
|
||||
@@ -4,11 +4,50 @@ package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" ..
|
||||
local utility = require "utility"
|
||||
local json = utility.require("dkjson")
|
||||
|
||||
local PATH = "notebook"
|
||||
local PATH = "PRIVATE_DATA/notebook"
|
||||
local whitelist = { md = true, }
|
||||
|
||||
local embedding_model = "nomic-embed-text"
|
||||
local maximum_file_size = 2048
|
||||
-- local embedding_model = "nomic-embed-text"
|
||||
-- local maximum_file_size = 2048
|
||||
local embedding_model = "qwen3-embedding:0.6b"
|
||||
local maximum_file_size = 32768
|
||||
|
||||
|
||||
|
||||
local timing = {}
|
||||
local function display_timing(n)
|
||||
local function _display(n)
|
||||
local time = timing[n]
|
||||
local previous = timing[n - 1]
|
||||
|
||||
local delta = time.time - previous.time
|
||||
if delta >= 2*60*60 then -- 2 hours
|
||||
delta = tostring(math.floor(delta/(60*60/10))/10) .. " hours"
|
||||
elseif delta >= 120 then -- 2 minutes
|
||||
delta = tostring(math.floor(delta/(60/10))/10) .. " minutes"
|
||||
else
|
||||
delta = tostring(delta) .. " seconds"
|
||||
end
|
||||
|
||||
print(delta, previous.label)
|
||||
end
|
||||
|
||||
if n then
|
||||
_display(n)
|
||||
else
|
||||
print("All measured timings:")
|
||||
for i = 2, #timing do
|
||||
_display(i)
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
local function mark_timing(label)
|
||||
timing[#timing + 1] = { label = label, time = os.time(), }
|
||||
if #timing > 1 then
|
||||
display_timing(#timing)
|
||||
end
|
||||
end
|
||||
|
||||
-- strip YAML frontmatter (if present)
|
||||
-- can error, will return nil & error message
|
||||
@@ -32,6 +71,7 @@ end
|
||||
local tree
|
||||
tree = function(path, fn)
|
||||
utility.list(path or ".", function(path_name)
|
||||
if path_name == ".git" then return end
|
||||
if utility.is_file(path_name) then
|
||||
fn(path_name)
|
||||
else
|
||||
@@ -49,6 +89,7 @@ end
|
||||
local file_list = {}
|
||||
local embeddings = {}
|
||||
|
||||
mark_timing("Assembling file list.")
|
||||
tree(PATH, function(file_name)
|
||||
local path, name, extension = utility.split_path_components(file_name)
|
||||
if not extension then
|
||||
@@ -62,6 +103,7 @@ tree(PATH, function(file_name)
|
||||
file_list[#file_list + 1] = file_name
|
||||
end)
|
||||
|
||||
mark_timing("Generating embeddings.")
|
||||
for i = 1, #file_list do
|
||||
local function _run()
|
||||
local file_name = file_list[i]
|
||||
@@ -115,8 +157,13 @@ for i = 1, #file_list do
|
||||
_run()
|
||||
end
|
||||
|
||||
utility.open("embeddings.json", "w", function(file)
|
||||
mark_timing("Outputting embeddings in JSON.")
|
||||
utility.open("PRIVATE_DATA/embeddings.json", "w", function(file)
|
||||
local output = json.encode(embeddings, { indent = true })
|
||||
file:write(output)
|
||||
file:write("\n")
|
||||
end)
|
||||
|
||||
mark_timing("Finished.")
|
||||
print("")
|
||||
display_timing()
|
||||
Executable
+15
@@ -0,0 +1,15 @@
|
||||
#!/usr/bin/env luajit
|
||||
|
||||
package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path
|
||||
local utility = require "utility"
|
||||
local json = utility.require("dkjson")
|
||||
|
||||
local similarities = utility.open("PRIVATE_DATA/similarities.json", "r", function(file)
|
||||
return json.decode(file:read("*all"))
|
||||
end)
|
||||
|
||||
table.sort(similarities, function(a, b) return a.similarity < b.similarity end)
|
||||
|
||||
utility.open("PRIVATE_DATA/differences.json", "w", function(file)
|
||||
file:write(json.encode(similarities, { indent = true }))
|
||||
end)
|
||||
Executable
+296
@@ -0,0 +1,296 @@
|
||||
#!/usr/bin/env luajit
|
||||
|
||||
package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path
|
||||
local utility = require "utility"
|
||||
local json = utility.require("dkjson")
|
||||
|
||||
local default_model = "gemma4:12b-mlx"
|
||||
local minimum_bytes = 1000
|
||||
local maximum_bytes = 40000
|
||||
|
||||
local synopsis_prompt = [[
|
||||
Generate a lengthy novel synopsis from the following:
|
||||
]]
|
||||
|
||||
local scoring_prompt = [[
|
||||
You are evaluating novel synopses for development priority. The goal is NOT to judge writing quality, grammar, or polish. The synopsis is only a rough idea. Assign an integer score from 1–100 for each category. Be extremely harsh with your scoring.
|
||||
|
||||
1. Hook
|
||||
2. Originality
|
||||
3. Memorability
|
||||
4. Expansion Potential
|
||||
5. Conflict Potential
|
||||
6. Character Potential
|
||||
7. Worldbuilding Potential
|
||||
8. Emotional Potential
|
||||
9. Curiosity
|
||||
10. Overall Promise
|
||||
|
||||
Guidelines:
|
||||
- Avoid clustering scores near the middle. Use the full 1–100 range.
|
||||
- A score around 50 represents an average publishable premise.
|
||||
- Scores above 85 should be rare and reserved for genuinely exceptional ideas.
|
||||
- Scores below 25 should represent ideas with major conceptual weaknesses.
|
||||
- Return only valid JSON.
|
||||
|
||||
Output format:
|
||||
|
||||
{
|
||||
"hook": 0,
|
||||
"originality": 0,
|
||||
"memorability": 0,
|
||||
"expansion_potential": 0,
|
||||
"conflict_potential": 0,
|
||||
"character_potential": 0,
|
||||
"worldbuilding_potential": 0,
|
||||
"emotional_potential": 0,
|
||||
"curiosity": 0,
|
||||
"overall_promise": 0
|
||||
}
|
||||
]]
|
||||
|
||||
local read_all = function(file_name)
|
||||
return utility.open(file_name, "r", function(file)
|
||||
return file:read("*all")
|
||||
end)
|
||||
end
|
||||
local write_all = function(file_name, text)
|
||||
return utility.open(file_name, "w", function(file)
|
||||
file:write(text)
|
||||
file:write("\n")
|
||||
end)
|
||||
end
|
||||
|
||||
local files
|
||||
if utility.path_exists("PRIVATE_DATA/file_list.json") then
|
||||
files = json.decode(read_all("PRIVATE_DATA/file_list.json"))
|
||||
else
|
||||
files = {}
|
||||
end
|
||||
|
||||
-- strip YAML frontmatter (if present)
|
||||
-- can error, will return nil & error message
|
||||
local function strip_frontmatter(text)
|
||||
local tab = text:split("\n")
|
||||
if tab[1] == "---" then
|
||||
table.remove(tab, 1)
|
||||
while true do
|
||||
local done = tab[1] == "---"
|
||||
table.remove(tab, 1)
|
||||
if done then
|
||||
return table.concat(tab, "\n")
|
||||
elseif #tab < 1 then
|
||||
return nil, "Invalid YAML frontmatter."
|
||||
end
|
||||
end
|
||||
end
|
||||
return text
|
||||
end
|
||||
|
||||
-- strip Markdown formatting of a JSON block
|
||||
-- just returns the string if it doesn't match
|
||||
local function strip_markdown_codeblock(text)
|
||||
local tab = text:split("\n")
|
||||
if tab[1] == "```json" then
|
||||
table.remove(tab, 1)
|
||||
table.remove(tab, #tab)
|
||||
return table.concat(tab, "\n")
|
||||
else
|
||||
return text
|
||||
end
|
||||
end
|
||||
|
||||
local strip_reasoning = function(text, reasoning_lines)
|
||||
if not reasoning_lines then reasoning_lines = {} end
|
||||
local tab = text:split("\n")
|
||||
table.remove(tab, 1) -- remove "Thinking..."
|
||||
|
||||
while true do
|
||||
local done = tab[1] == "...done thinking."
|
||||
local line = table.remove(tab, 1)
|
||||
if done then
|
||||
table.remove(tab, 1) -- remove newline after end of thinking
|
||||
return table.concat(tab, "\n")
|
||||
elseif #tab < 1 then
|
||||
return text -- no reasoning output
|
||||
else
|
||||
reasoning_lines[#reasoning_lines + 1] = line -- export thinking lines
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
local send_prompt = function(text, model)
|
||||
if type(text) == "table" then text = table.concat(text, "\n") end
|
||||
|
||||
local tmp_file_name = utility.tmp_file_name()
|
||||
utility.open(tmp_file_name, "w", function(file)
|
||||
file:write(text)
|
||||
end)
|
||||
|
||||
-- word wrap breaks the raw output badly, so I need to implement my own for terminal output somehow
|
||||
local output = utility.capture_safe("cat " .. tmp_file_name:enquote() .. " | ollama run " .. (model or default_model) .. " --nowordwrap")
|
||||
os.execute("ollama stop " .. (model or default_model)) -- NOTE this makes things slower, but more stable
|
||||
os.execute("rm " .. tmp_file_name)
|
||||
if not output then error("ollama failed to generate output") end
|
||||
output = output:sub(1, -2) -- strip extra newline from utility.capture_safe
|
||||
|
||||
local thinking = {}
|
||||
output = strip_reasoning(output, thinking)
|
||||
|
||||
return output, thinking
|
||||
end
|
||||
|
||||
local tree
|
||||
tree = function(path, fn)
|
||||
utility.list(path or ".", function(path_name)
|
||||
local blacklist = {
|
||||
[".git"] = true,
|
||||
[".gitkeep"] = true,
|
||||
[".gitignore"] = true,
|
||||
[".DS_Store"] = true,
|
||||
}
|
||||
if blacklist[path_name] then return end
|
||||
if utility.is_file(path_name) then
|
||||
fn(path_name)
|
||||
else
|
||||
tree(path .. utility.path_separator .. path_name, fn)
|
||||
end
|
||||
end)
|
||||
end
|
||||
|
||||
local get_file_size = function(file_name)
|
||||
local file = io.open(file_name, "rb")
|
||||
if file then
|
||||
local size = file:seek("end")
|
||||
file:close()
|
||||
return size
|
||||
end
|
||||
end
|
||||
|
||||
local refresh_file_list = function()
|
||||
local blacklist = { -- I'm only blacklisting binary formats because funny results happen with really invalid texts
|
||||
"jpg", "mp4", "pdf", "png", "webp", "jpeg", "gif",
|
||||
} for _, name in ipairs(blacklist) do blacklist[name] = true end
|
||||
|
||||
local new_files_list = {}
|
||||
tree("PRIVATE_DATA/notebook", function(file_name)
|
||||
local _, _, extension = utility.split_path_components(file_name)
|
||||
if not blacklist[extension] then
|
||||
new_files_list[#new_files_list + 1] = file_name
|
||||
end
|
||||
end)
|
||||
files = new_files_list
|
||||
write_all("PRIVATE_DATA/file_list.json", json.encode(new_files_list, { indent = true }))
|
||||
end
|
||||
|
||||
local generate_and_score = function(file_name, text)
|
||||
print("Writing synopsis...")
|
||||
local synopsis = send_prompt(synopsis_prompt .. text)
|
||||
print(synopsis)
|
||||
print("Scoring synopsis...")
|
||||
local scoring, thinking = send_prompt(scoring_prompt .. synopsis)
|
||||
print(scoring)
|
||||
local scoring_decoded = json.decode(scoring) -- likely will not work because it consistently returns Markdown instead of JSON
|
||||
if not scoring_decoded then
|
||||
scoring_decoded = json.decode(strip_markdown_codeblock(scoring))
|
||||
end
|
||||
|
||||
local object = {
|
||||
synopsis = synopsis,
|
||||
scoring = scoring_decoded or scoring, -- either the correct values or a string of output that isn't JSON
|
||||
thinking = thinking,
|
||||
}
|
||||
|
||||
write_all("PRIVATE_DATA/synopses/" .. utility.uuid() .. ".json", json.encode(object, { indent = true }))
|
||||
end
|
||||
|
||||
|
||||
|
||||
os.execute("mkdir -p PRIVATE_DATA/synopses")
|
||||
|
||||
if arg[1] == "refresh_file_list" then
|
||||
print("Refresing file list...")
|
||||
refresh_file_list()
|
||||
end
|
||||
|
||||
if arg[1] == "repair_synopsis_exports" then
|
||||
local path = "PRIVATE_DATA/synopses"
|
||||
utility.list(path, function(path_name)
|
||||
path_name = path .. utility.path_separator .. path_name
|
||||
if path_name:find("%.json") then
|
||||
local object = json.decode(read_all(path_name))
|
||||
if type(object.scoring) == "table" then return end -- don't fuck with working pieces
|
||||
local decoded = json.decode(strip_markdown_codeblock(object.scoring))
|
||||
if decoded then
|
||||
object.scoring = decoded
|
||||
write_all(path_name, json.encode(object, { indent = true, }))
|
||||
end
|
||||
end
|
||||
end)
|
||||
os.exit(0)
|
||||
end
|
||||
|
||||
if arg[1] == "export_ordered_list_of_prompts" then
|
||||
local path = "PRIVATE_DATA/synopses"
|
||||
local items = {}
|
||||
local item_order = {}
|
||||
utility.list(path, function(path_name)
|
||||
local full_path = path .. utility.path_separator .. path_name
|
||||
if path_name:find("%.json") then
|
||||
local object = json.decode(read_all(full_path))
|
||||
items[path_name] = object
|
||||
if type(object.scoring) == "table" then
|
||||
local s = object.scoring
|
||||
local total_score = s.conflict_potential + s.emotional_potential + s.character_potential + s.worldbuilding_potential + s.expansion_potential + s.overall_promise + s.memorability + s.originality + s.curiosity + s.hook
|
||||
item_order[#item_order + 1] = { path_name = path_name, total_score = total_score, }
|
||||
end
|
||||
end
|
||||
end)
|
||||
table.sort(item_order, function(A,B) return A.total_score > B.total_score end)
|
||||
-- for k,v in pairs(item_order) do print(v.path_name,v.total_score) end
|
||||
local output = {
|
||||
"---",
|
||||
"title: Ordered Synopses (" .. #item_order .. " items)",
|
||||
"author: [\"Gemma4:12b-mlx\", \"Tangent\", \"Ollama\"]",
|
||||
"publisher: Tangent",
|
||||
"---",
|
||||
"",
|
||||
}
|
||||
for _, v in pairs(item_order) do
|
||||
local item = items[v.path_name]
|
||||
local text = item.synopsis
|
||||
local tab = text:split("\n")
|
||||
for index, line in ipairs(tab) do
|
||||
if line:sub(1, 1) == "#" then
|
||||
tab[index] = "#" .. tab[index]
|
||||
end
|
||||
end
|
||||
-- output[#output + 1] = "# " .. v.path_name .. " (" .. v.total_score .. ")\n\n" .. item.synopsis .. "\n"
|
||||
output[#output + 1] = "# " .. v.path_name .. " (" .. v.total_score .. ")\n\n" .. table.concat(tab, "\n") .. "\n"
|
||||
output[#output + 1] = "## Scoring\n\n```json\n" .. json.encode(item.scoring, { indent = true, }) .. "\n```\n"
|
||||
end
|
||||
write_all("PRIVATE_DATA/Ordered Synopses.md", table.concat(output, "\n"))
|
||||
os.execute("pandoc \"PRIVATE_DATA/Ordered Synopses.md\" -o \"PRIVATE_DATA/Ordered Synopses.epub\"")
|
||||
os.exit(0)
|
||||
end
|
||||
|
||||
print(#files .. " files to select from.")
|
||||
if #files == 0 then
|
||||
print("Run \"./synopsis_generator.lua refresh_file_list\" first.")
|
||||
os.exit(1)
|
||||
end
|
||||
|
||||
while true do
|
||||
print("Selecting a file...")
|
||||
local file_name = files[math.random(1, #files)]
|
||||
local file_size = get_file_size(file_name)
|
||||
if file_size > minimum_bytes and file_size <= maximum_bytes then
|
||||
local text = read_all(file_name)
|
||||
text = strip_frontmatter(text)
|
||||
if #text > minimum_bytes and #text <= maximum_bytes then
|
||||
print(file_name .. " chosen.")
|
||||
generate_and_score(file_name, text) -- kind of the main function, innit?
|
||||
-- os.exit(0)
|
||||
end
|
||||
end
|
||||
end
|
||||
Reference in new issue
Block a user