Compare commits
4
Commits
707776c12d
...
4d4f95f54a
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4d4f95f54a | ||
|
|
16a0c6c93d | ||
|
|
c0bc8ca6a2 | ||
|
|
49a22eb2c4 |
No files matched your search
@@ -16,3 +16,14 @@ them.
|
|||||||
###### `least_similar.lua`
|
###### `least_similar.lua`
|
||||||
Opens `similarities.json`, reverses the sort order, and saves it as
|
Opens `similarities.json`, reverses the sort order, and saves it as
|
||||||
`differences.json`.
|
`differences.json`.
|
||||||
|
|
||||||
|
###### `synopsis_generator.lua`
|
||||||
|
Chooses a random file within `notebook`, and generates a novel synopsis from it.
|
||||||
|
|
||||||
|
Arguments:
|
||||||
|
- `refresh_file_list`: Refreshes the cached file list to choose from.
|
||||||
|
- `export_ordered_list_of_prompts`: Makes an epub to review generated synopses.
|
||||||
|
|
||||||
|
## Tasks
|
||||||
|
- [ ] synopsis_generator cache should include file sizes to make script running easier/faster
|
||||||
|
- [ ] The whitelisting/blacklisting of tree should be in list too.
|
||||||
@@ -0,0 +1,35 @@
|
|||||||
|
local text_processing = {}
|
||||||
|
|
||||||
|
-- strip YAML frontmatter (if present)
|
||||||
|
-- on error, will return nil & error message
|
||||||
|
text_processing.strip_frontmatter = function(text)
|
||||||
|
local tab = text:split("\n")
|
||||||
|
if tab[1] == "---" then
|
||||||
|
table.remove(tab, 1)
|
||||||
|
while true do
|
||||||
|
local done = tab[1] == "---"
|
||||||
|
table.remove(tab, 1)
|
||||||
|
if done then
|
||||||
|
return table.concat(tab, "\n")
|
||||||
|
elseif #tab < 1 then
|
||||||
|
return nil, "Invalid YAML frontmatter."
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return text
|
||||||
|
end
|
||||||
|
|
||||||
|
-- strip Markdown formatting of a JSON block
|
||||||
|
-- just returns the string if it doesn't match
|
||||||
|
text_processing.strip_markdown_codeblock = function(text)
|
||||||
|
local tab = text:split("\n")
|
||||||
|
if tab[1] == "```json" then
|
||||||
|
table.remove(tab, 1)
|
||||||
|
table.remove(tab, #tab)
|
||||||
|
return table.concat(tab, "\n")
|
||||||
|
else
|
||||||
|
return text
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
return text_processing
|
||||||
@@ -244,6 +244,13 @@ utility.tree = function(path, options, fn)
|
|||||||
utility.list(path or ".", function(path_name)
|
utility.list(path or ".", function(path_name)
|
||||||
if options.blacklist and options.blacklist[path_name] then return end
|
if options.blacklist and options.blacklist[path_name] then return end
|
||||||
if options.whitelist and (not options.whitelist[path_name]) then return end
|
if options.whitelist and (not options.whitelist[path_name]) then return end
|
||||||
|
|
||||||
|
if options.extension_blacklist or options.extension_whitelist then
|
||||||
|
local _, _, extension = utility.split_path_components(path_name)
|
||||||
|
if options.extension_blacklist and options.extension_blacklist[extension] then return end
|
||||||
|
if options.extension_whitelist and (not options.extension_whitelist[extension]) then return end
|
||||||
|
end
|
||||||
|
|
||||||
if utility.is_file(path_name) then
|
if utility.is_file(path_name) then
|
||||||
fn(path_name)
|
fn(path_name)
|
||||||
else
|
else
|
||||||
@@ -541,4 +548,26 @@ end
|
|||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
-- additionally returns total and count
|
||||||
|
utility.mean = function(object)
|
||||||
|
local total, count = 0, 0
|
||||||
|
for _, value in pairs(object) do
|
||||||
|
total = total = value
|
||||||
|
count = count + 1
|
||||||
|
end
|
||||||
|
return total / count, total, count
|
||||||
|
end
|
||||||
|
|
||||||
|
-- additionally returns total
|
||||||
|
utility.median = function(object)
|
||||||
|
local tab = {}
|
||||||
|
for _, value in pairs(object) do
|
||||||
|
tab[#tab + 1] = value
|
||||||
|
end
|
||||||
|
table.sort(tab)
|
||||||
|
return tab[math.floor(#tab / 2)], #tab
|
||||||
|
end
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
return utility
|
return utility
|
||||||
+58
-145
@@ -5,8 +5,9 @@ local utility = require "utility"
|
|||||||
|
|
||||||
local json = utility.require("dkjson")
|
local json = utility.require("dkjson")
|
||||||
local prompts = utility.require("prompts")
|
local prompts = utility.require("prompts")
|
||||||
|
local text_processing = utility.require("text_processing")
|
||||||
|
|
||||||
local default_model = "gemma4:12b-mlx"
|
local model = "gemma4:12b-mlx"
|
||||||
local minimum_bytes = 1000
|
local minimum_bytes = 1000
|
||||||
local maximum_bytes = 40000
|
local maximum_bytes = 40000
|
||||||
|
|
||||||
@@ -17,91 +18,13 @@ else
|
|||||||
files = {}
|
files = {}
|
||||||
end
|
end
|
||||||
|
|
||||||
-- strip YAML frontmatter (if present)
|
|
||||||
-- can error, will return nil & error message
|
|
||||||
local function strip_frontmatter(text)
|
|
||||||
local tab = text:split("\n")
|
|
||||||
if tab[1] == "---" then
|
|
||||||
table.remove(tab, 1)
|
|
||||||
while true do
|
|
||||||
local done = tab[1] == "---"
|
|
||||||
table.remove(tab, 1)
|
|
||||||
if done then
|
|
||||||
return table.concat(tab, "\n")
|
|
||||||
elseif #tab < 1 then
|
|
||||||
return nil, "Invalid YAML frontmatter."
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
return text
|
|
||||||
end
|
|
||||||
|
|
||||||
-- strip Markdown formatting of a JSON block
|
|
||||||
-- just returns the string if it doesn't match
|
|
||||||
local function strip_markdown_codeblock(text)
|
|
||||||
local tab = text:split("\n")
|
|
||||||
if tab[1] == "```json" then
|
|
||||||
table.remove(tab, 1)
|
|
||||||
table.remove(tab, #tab)
|
|
||||||
return table.concat(tab, "\n")
|
|
||||||
else
|
|
||||||
return text
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
local strip_reasoning = function(text, reasoning_lines)
|
|
||||||
if not reasoning_lines then reasoning_lines = {} end
|
|
||||||
local tab = text:split("\n")
|
|
||||||
table.remove(tab, 1) -- remove "Thinking..."
|
|
||||||
|
|
||||||
while true do
|
|
||||||
local done = tab[1] == "...done thinking."
|
|
||||||
local line = table.remove(tab, 1)
|
|
||||||
if done then
|
|
||||||
table.remove(tab, 1) -- remove newline after end of thinking
|
|
||||||
return table.concat(tab, "\n")
|
|
||||||
elseif #tab < 1 then
|
|
||||||
return text -- no reasoning output
|
|
||||||
else
|
|
||||||
reasoning_lines[#reasoning_lines + 1] = line -- export thinking lines
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end
|
|
||||||
|
|
||||||
local send_prompt = function(text, model)
|
|
||||||
if type(text) == "table" then text = table.concat(text, "\n") end
|
|
||||||
|
|
||||||
local tmp_file_name = utility.tmp_file_name()
|
|
||||||
utility.open(tmp_file_name, "w", function(file)
|
|
||||||
file:write(text)
|
|
||||||
end)
|
|
||||||
|
|
||||||
-- word wrap breaks the raw output badly, so I need to implement my own for terminal output somehow
|
|
||||||
local output = utility.capture_safe("cat " .. tmp_file_name:enquote() .. " | ollama run " .. (model or default_model) .. " --nowordwrap")
|
|
||||||
os.execute("ollama stop " .. (model or default_model)) -- NOTE this makes things slower, but more stable
|
|
||||||
os.execute("rm " .. tmp_file_name)
|
|
||||||
if not output then error("ollama failed to generate output") end
|
|
||||||
output = output:sub(1, -2) -- strip extra newline from utility.capture_safe
|
|
||||||
|
|
||||||
local thinking = {}
|
|
||||||
output = strip_reasoning(output, thinking)
|
|
||||||
|
|
||||||
return output, thinking
|
|
||||||
end
|
|
||||||
|
|
||||||
local refresh_file_list = function()
|
local refresh_file_list = function()
|
||||||
local blacklist = { -- I'm only blacklisting binary formats because funny results happen with really invalid texts
|
|
||||||
"jpg", "mp4", "pdf", "png", "webp", "jpeg", "gif",
|
|
||||||
} for _, name in ipairs(blacklist) do blacklist[name] = true end
|
|
||||||
|
|
||||||
local new_files_list = {}
|
local new_files_list = {}
|
||||||
utility.tree("PRIVATE_DATA/notebook", {
|
utility.tree("PRIVATE_DATA/notebook", {
|
||||||
blacklist = utility.enumerate{".git", ".gitignore", ".gitkeep", ".DS_Store"}
|
blacklist = utility.enumerate{ ".git", ".gitattributes", ".gitignore", ".gitkeep", ".DS_Store", },
|
||||||
|
extension_blacklist = utility.enumerate{ "gif", "jpg", "jpeg", "mp4", "pdf", "png", "webp", },
|
||||||
}, function(file_name)
|
}, function(file_name)
|
||||||
local _, _, extension = utility.split_path_components(file_name)
|
new_files_list[#new_files_list + 1] = file_name
|
||||||
if not blacklist[extension] then
|
|
||||||
new_files_list[#new_files_list + 1] = file_name
|
|
||||||
end
|
|
||||||
end)
|
end)
|
||||||
files = new_files_list
|
files = new_files_list
|
||||||
utility.save_data(new_files_list, "PRIVATE_DATA/file_list.json")
|
utility.save_data(new_files_list, "PRIVATE_DATA/file_list.json")
|
||||||
@@ -109,14 +32,14 @@ end
|
|||||||
|
|
||||||
local generate_and_score = function(file_name, text)
|
local generate_and_score = function(file_name, text)
|
||||||
print("Writing synopsis...")
|
print("Writing synopsis...")
|
||||||
local synopsis = send_prompt(prompts.synopsis_prompt .. text)
|
local synopsis = utility.llm_prompt(prompts.synopsis_prompt .. text, model)
|
||||||
print(synopsis)
|
print(synopsis)
|
||||||
print("Scoring synopsis...")
|
print("Scoring synopsis...")
|
||||||
local scoring, thinking = send_prompt(prompts.scoring_prompt .. synopsis)
|
local scoring, thinking = utility.llm_prompt(prompts.scoring_prompt .. synopsis, model)
|
||||||
print(scoring)
|
print(scoring)
|
||||||
local scoring_decoded = json.decode(scoring) -- likely will not work because it consistently returns Markdown instead of JSON
|
local scoring_decoded = json.decode(scoring) -- likely will not work because it consistently returns Markdown instead of JSON
|
||||||
if not scoring_decoded then
|
if not scoring_decoded then
|
||||||
scoring_decoded = json.decode(strip_markdown_codeblock(scoring))
|
scoring_decoded = json.decode(text_processing.strip_markdown_codeblock(scoring))
|
||||||
end
|
end
|
||||||
|
|
||||||
local object = {
|
local object = {
|
||||||
@@ -128,6 +51,53 @@ local generate_and_score = function(file_name, text)
|
|||||||
utility.save_data(object, "PRIVATE_DATA/synopses/" .. utility.uuid() .. ".json")
|
utility.save_data(object, "PRIVATE_DATA/synopses/" .. utility.uuid() .. ".json")
|
||||||
end
|
end
|
||||||
|
|
||||||
|
local export_ordered_list_of_prompts = function()
|
||||||
|
local path = "PRIVATE_DATA/synopses"
|
||||||
|
local items = {}
|
||||||
|
local item_order = {}
|
||||||
|
|
||||||
|
utility.list(path, function(path_name)
|
||||||
|
local full_path = path .. utility.path_separator .. path_name
|
||||||
|
if path_name:find("%.json") then
|
||||||
|
local object = utility.load_data(full_path)
|
||||||
|
items[path_name] = object
|
||||||
|
if type(object.scoring) == "table" then
|
||||||
|
local mean_score, total_score = utility.mean(object.scoring)
|
||||||
|
item_order[#item_order + 1] = { path_name = path_name, total_score = total_score, mean_score = mean_score, }
|
||||||
|
end
|
||||||
|
end
|
||||||
|
end)
|
||||||
|
|
||||||
|
table.sort(item_order, function(A,B) return A.mean_score > B.mean_score end)
|
||||||
|
|
||||||
|
local output = {
|
||||||
|
"---",
|
||||||
|
"title: Ordered Synopses (" .. #item_order .. " items)",
|
||||||
|
"author: [\"" .. model .. "\", \"Tangent\", \"Ollama\"]",
|
||||||
|
"publisher: \"synopsis_generator.lua\"",
|
||||||
|
"---",
|
||||||
|
"",
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, v in pairs(item_order) do
|
||||||
|
local item = items[v.path_name]
|
||||||
|
local text = item.synopsis
|
||||||
|
local tab = text:split("\n")
|
||||||
|
|
||||||
|
for index, line in ipairs(tab) do
|
||||||
|
if line:sub(1, 1) == "#" then
|
||||||
|
tab[index] = "#" .. tab[index]
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
output[#output + 1] = "# " .. v.path_name .. " (" .. v.total_score .. ")\n\n" .. table.concat(tab, "\n") .. "\n"
|
||||||
|
output[#output + 1] = "## Scoring\n\n```json\n" .. json.encode(item.scoring, { indent = true, }) .. "\n```\n"
|
||||||
|
end
|
||||||
|
|
||||||
|
utility.write_file("PRIVATE_DATA/Ordered Synopses.md", table.concat(output, "\n"))
|
||||||
|
os.execute("pandoc \"PRIVATE_DATA/Ordered Synopses.md\" -o \"PRIVATE_DATA/Ordered Synopses.epub\"")
|
||||||
|
end
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
os.execute("mkdir -p PRIVATE_DATA/synopses")
|
os.execute("mkdir -p PRIVATE_DATA/synopses")
|
||||||
@@ -135,66 +105,9 @@ os.execute("mkdir -p PRIVATE_DATA/synopses")
|
|||||||
if arg[1] == "refresh_file_list" then
|
if arg[1] == "refresh_file_list" then
|
||||||
print("Refresing file list...")
|
print("Refresing file list...")
|
||||||
refresh_file_list()
|
refresh_file_list()
|
||||||
end
|
|
||||||
|
|
||||||
if arg[1] == "repair_synopsis_exports" then
|
|
||||||
local path = "PRIVATE_DATA/synopses"
|
|
||||||
utility.list(path, function(path_name)
|
|
||||||
path_name = path .. utility.path_separator .. path_name
|
|
||||||
if path_name:find("%.json") then
|
|
||||||
local object = utility.load_data(path_name)
|
|
||||||
if type(object.scoring) == "table" then return end -- don't fuck with working pieces
|
|
||||||
local decoded = json.decode(strip_markdown_codeblock(object.scoring))
|
|
||||||
if decoded then
|
|
||||||
object.scoring = decoded
|
|
||||||
utility.save_data(object, path_name)
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end)
|
|
||||||
os.exit(0)
|
os.exit(0)
|
||||||
end
|
elseif arg[1] == "export_ordered_list_of_prompts" then
|
||||||
|
export_ordered_list_of_prompts()
|
||||||
if arg[1] == "export_ordered_list_of_prompts" then
|
|
||||||
local path = "PRIVATE_DATA/synopses"
|
|
||||||
local items = {}
|
|
||||||
local item_order = {}
|
|
||||||
utility.list(path, function(path_name)
|
|
||||||
local full_path = path .. utility.path_separator .. path_name
|
|
||||||
if path_name:find("%.json") then
|
|
||||||
local object = utility.load_data(full_path)
|
|
||||||
items[path_name] = object
|
|
||||||
if type(object.scoring) == "table" then
|
|
||||||
local s = object.scoring
|
|
||||||
local total_score = s.conflict_potential + s.emotional_potential + s.character_potential + s.worldbuilding_potential + s.expansion_potential + s.overall_promise + s.memorability + s.originality + s.curiosity + s.hook
|
|
||||||
item_order[#item_order + 1] = { path_name = path_name, total_score = total_score, }
|
|
||||||
end
|
|
||||||
end
|
|
||||||
end)
|
|
||||||
table.sort(item_order, function(A,B) return A.total_score > B.total_score end)
|
|
||||||
-- for k,v in pairs(item_order) do print(v.path_name,v.total_score) end
|
|
||||||
local output = {
|
|
||||||
"---",
|
|
||||||
"title: Ordered Synopses (" .. #item_order .. " items)",
|
|
||||||
"author: [\"Gemma4:12b-mlx\", \"Tangent\", \"Ollama\"]",
|
|
||||||
"publisher: Tangent",
|
|
||||||
"---",
|
|
||||||
"",
|
|
||||||
}
|
|
||||||
for _, v in pairs(item_order) do
|
|
||||||
local item = items[v.path_name]
|
|
||||||
local text = item.synopsis
|
|
||||||
local tab = text:split("\n")
|
|
||||||
for index, line in ipairs(tab) do
|
|
||||||
if line:sub(1, 1) == "#" then
|
|
||||||
tab[index] = "#" .. tab[index]
|
|
||||||
end
|
|
||||||
end
|
|
||||||
-- output[#output + 1] = "# " .. v.path_name .. " (" .. v.total_score .. ")\n\n" .. item.synopsis .. "\n"
|
|
||||||
output[#output + 1] = "# " .. v.path_name .. " (" .. v.total_score .. ")\n\n" .. table.concat(tab, "\n") .. "\n"
|
|
||||||
output[#output + 1] = "## Scoring\n\n```json\n" .. json.encode(item.scoring, { indent = true, }) .. "\n```\n"
|
|
||||||
end
|
|
||||||
utility.write_file("PRIVATE_DATA/Ordered Synopses.md", table.concat(output, "\n"))
|
|
||||||
os.execute("pandoc \"PRIVATE_DATA/Ordered Synopses.md\" -o \"PRIVATE_DATA/Ordered Synopses.epub\"")
|
|
||||||
os.exit(0)
|
os.exit(0)
|
||||||
end
|
end
|
||||||
|
|
||||||
@@ -210,7 +123,7 @@ while true do
|
|||||||
local file_size = utility.file_size(file_name)
|
local file_size = utility.file_size(file_name)
|
||||||
if file_size > minimum_bytes and file_size <= maximum_bytes then
|
if file_size > minimum_bytes and file_size <= maximum_bytes then
|
||||||
local text = utility.read_file(file_name)
|
local text = utility.read_file(file_name)
|
||||||
text = strip_frontmatter(text)
|
text = text_processing.strip_frontmatter(text)
|
||||||
if #text > minimum_bytes and #text <= maximum_bytes then
|
if #text > minimum_bytes and #text <= maximum_bytes then
|
||||||
print(file_name .. " chosen.")
|
print(file_name .. " chosen.")
|
||||||
generate_and_score(file_name, text) -- kind of the main function, innit?
|
generate_and_score(file_name, text) -- kind of the main function, innit?
|
||||||
|
|||||||
Reference in new issue
Block a user