From 8ddd8fdf6398fbefb2b19497bbf96d774397f799 Mon Sep 17 00:00:00 2001 From: tangent Date: Fri, 21 Aug 2026 12:55:54 -0600 Subject: [PATCH 01/15] first level intermediates can be saved and loaded --- make_cover_briefs.lua | 23 +++++++++++++++++++++-- 1 file changed, 21 insertions(+), 2 deletions(-) diff --git a/make_cover_briefs.lua b/make_cover_briefs.lua index fa21bc0..1894192 100755 --- a/make_cover_briefs.lua +++ b/make_cover_briefs.lua @@ -177,20 +177,39 @@ tree("extracted_texts", function(file_name) local text = read_all(file_name) local result + -- TODO there is no way to recognize a dangling intermediate and pickup from there print(name) local outputs = {} if #text > maximum_bytes then -- TODO check if this would work just as well starting from 1 instead.. for i = 0, #text/maximum_bytes do print("Processing section " .. (i + 1) .. " of " .. math.floor(#text/maximum_bytes + 1)) - local piece = text:sub(i * maximum_bytes, (i + 1) * maximum_bytes - 1) - result = send_prompt(partial_prompt .. piece) + + local intermediate_name = "intermediates/" .. name:sub(1, -5) .. "-" .. tostring(#outputs + 1) .. ".txt" + if utility.path_exists(intermediate_name) then + result = read_all(intermediate_name) + print("Loaded from intermediates.") + else + local piece = text:sub(i * maximum_bytes, (i + 1) * maximum_bytes - 1) + result = send_prompt(partial_prompt .. piece) + write_all(intermediate_name, result) + end + outputs[#outputs + 1] = result print(#result .. " characters added to intermediate context.") end text = table.concat(outputs, "\n\n") write_all("intermediates/" .. name, text) + print("Saved full intermediate prompt.") + + for i = 0, #text/maximum_bytes do + local intermediate_name = "intermediates/" .. name:sub(1, -5) .. "-" .. tostring(#outputs + 1) .. ".txt" + if utility.path_exists(intermediate_name) then + os.execute("rm " .. intermediate_name:enquote()) + end + print("Removed excess intermediate texts.") + end end -- handle too much context (works with up to 100 slices) -- 2.54.0 From eab714a5747af42cbea3175230e5f4f113195494 Mon Sep 17 00:00:00 2001 From: tangent Date: Fri, 21 Aug 2026 13:18:09 -0600 Subject: [PATCH 02/15] incomplete function extraction attempt --- make_cover_briefs.lua | 76 ++++++++++++++++++++++++++----------------- 1 file changed, 46 insertions(+), 30 deletions(-) diff --git a/make_cover_briefs.lua b/make_cover_briefs.lua index 1894192..225804a 100755 --- a/make_cover_briefs.lua +++ b/make_cover_briefs.lua @@ -170,6 +170,43 @@ tree = function(path, fn) end) end +local process_intermediates = function(text, name) + local result + local outputs = {} + + -- TODO check if this would work just as well starting from 1 instead.. + for i = 0, #text/maximum_bytes do + print("Processing section " .. (i + 1) .. " of " .. math.floor(#text/maximum_bytes + 1)) + + local intermediate_name = "intermediates/" .. name:sub(1, -5) .. "-" .. tostring(#outputs + 1) .. ".txt" + if utility.path_exists(intermediate_name) then + result = read_all(intermediate_name) + print("Loaded from file.") + else + local piece = text:sub(i * maximum_bytes, (i + 1) * maximum_bytes - 1) + result = send_prompt(partial_prompt .. piece) + write_all(intermediate_name, result) + end + + outputs[#outputs + 1] = result + print(#result .. " characters added to intermediate context.") + end + + text = table.concat(outputs, "\n\n") + write_all("intermediates/" .. name, text) + print("Saved primary intermediate prompt.") + + for i = 0, #text/maximum_bytes do + local intermediate_name = "intermediates/" .. name:sub(1, -5) .. "-" .. tostring(#outputs + 1) .. ".txt" + if utility.path_exists(intermediate_name) then + os.execute("rm " .. intermediate_name:enquote()) + end + end + print("Removed redundant intermediate prompts.") + + return result, outputs, text +end + tree("extracted_texts", function(file_name) local path, name, extension = utility.split_path_components(file_name) if utility.path_exists("cover_briefs/" .. name) then return end @@ -179,37 +216,16 @@ tree("extracted_texts", function(file_name) -- TODO there is no way to recognize a dangling intermediate and pickup from there print(name) - local outputs = {} + local outputs + + -- restore previous intermediate + -- (this is where I realized I should redo how I'm processing all of this) + if utility.path_exists("intermediates/" .. name) then + end + if #text > maximum_bytes then - -- TODO check if this would work just as well starting from 1 instead.. - for i = 0, #text/maximum_bytes do - print("Processing section " .. (i + 1) .. " of " .. math.floor(#text/maximum_bytes + 1)) - - local intermediate_name = "intermediates/" .. name:sub(1, -5) .. "-" .. tostring(#outputs + 1) .. ".txt" - if utility.path_exists(intermediate_name) then - result = read_all(intermediate_name) - print("Loaded from intermediates.") - else - local piece = text:sub(i * maximum_bytes, (i + 1) * maximum_bytes - 1) - result = send_prompt(partial_prompt .. piece) - write_all(intermediate_name, result) - end - - outputs[#outputs + 1] = result - print(#result .. " characters added to intermediate context.") - end - - text = table.concat(outputs, "\n\n") - write_all("intermediates/" .. name, text) - print("Saved full intermediate prompt.") - - for i = 0, #text/maximum_bytes do - local intermediate_name = "intermediates/" .. name:sub(1, -5) .. "-" .. tostring(#outputs + 1) .. ".txt" - if utility.path_exists(intermediate_name) then - os.execute("rm " .. intermediate_name:enquote()) - end - print("Removed excess intermediate texts.") - end + -- previously overwrites result, text, outputs + result, outputs, text = process_intermediates(text, name) end -- handle too much context (works with up to 100 slices) -- 2.54.0 From 4820e904f0195f69537dd3f8ef5388a081ca85e9 Mon Sep 17 00:00:00 2001 From: tangent Date: Fri, 21 Aug 2026 13:19:37 -0600 Subject: [PATCH 03/15] update utility to v1.5.2 --- lib/utility.lua | 59 +++++++++++++++++++++++++++++++++++++++---------- 1 file changed, 47 insertions(+), 12 deletions(-) diff --git a/lib/utility.lua b/lib/utility.lua index 88e7e20..3eda5f8 100644 --- a/lib/utility.lua +++ b/lib/utility.lua @@ -32,7 +32,7 @@ else } end -utility.version = "1.4.0" +utility.version = "1.5.2" -- WARNING: This will return "./" if the original script is called locally instead of with an absolute path! if arg[0] ~= nil then utility.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) -- inspired by discussion in https://stackoverflow.com/q/6380820 @@ -124,14 +124,6 @@ standard_library_addition(string, "enquote", function(s) return "\"" .. s:gsub("\"", "\\\"") .. "\"" end) -standard_library_addition(string, "split", function(s, delimiter) - local result = {} - for item in s:gsplit(delimiter) do - result[#result + 1] = item - end - return result -end) - standard_library_addition(string, "gsplit", function(s, delimiter) local function escape_special_characters(s) local special_characters = "[()%%.[^$%]*+%-?]" @@ -144,6 +136,14 @@ standard_library_addition(string, "gsplit", function(s, delimiter) return s:gmatch("(.-)" .. escape_special_characters(delimiter)) end) +standard_library_addition(string, "split", function(s, delimiter) + local result = {} + for item in s:gsplit(delimiter) do + result[#result + 1] = item + end + return result +end) + -- modified from my fork of lume @@ -301,7 +301,7 @@ utility.release_lock = function(file_path, lock_uuid) local lock_file_path = file_path .. ".lock" if lock_uuid then utility.open(lock_file_path, "r", function(file) - if not file:read("*all") == lock_uuid then + if not (file:read("*all") == lock_uuid) then error("\n\n Lock UUID changed while lock was obtained. Data loss may have occurred. \n\n") end end) @@ -348,12 +348,47 @@ utility.save_config = function() end end +local data_file_locations = {} +utility.load_data = function(file_path) + local data = utility.open(file_path, "r", function(data_file) + local json = utility.require("dkjson") + return json.decode(data_file:read("*all")) + end) + data_file_locations[data] = file_path + return data +end + +utility.save_data = function(data, file_path) + local keys, loop = {} + loop = function(tab) + if type(tab) == "table" then + for k,v in pairs(tab) do + if not (type(k) == "number") then + keys[k] = true + end + loop(v) + end + end + end + loop(data) + local order = {} + for k in pairs(keys) do order[#order + 1] = k end + table.sort(order) + + file_path = file_path or data_file_locations[data] + assert(file_path, "The object must have been loaded by utility.load_data or you must pass a path as the second argument.") + utility.open(file_path, "w", function(data_file) + local json = utility.require("dkjson") + data_file:write(json.encode(data, { indent = true, keyorder = order, })) + data_file:write("\n") + end) +end + utility.deepcopy = function(tab) - local _type = type(tab) local copy - if _type == "table" then + if type(tab) == "table" then copy = {} for key, value in next, tab, nil do copy[utility.deepcopy(key)] = utility.deepcopy(value) -- 2.54.0 From c2c00e77901567526d11f9bf80db27abd0c3c7cc Mon Sep 17 00:00:00 2001 From: Rose Liverman Date: Fri, 21 Aug 2026 16:25:43 -0600 Subject: [PATCH 04/15] I fucking lost the plot --- lib/prompts.lua | 70 ++++++++++++++++++++++ make_cover_briefs.lua | 136 ++++++++++++++++-------------------------- 2 files changed, 120 insertions(+), 86 deletions(-) create mode 100644 lib/prompts.lua diff --git a/lib/prompts.lua b/lib/prompts.lua new file mode 100644 index 0000000..718c874 --- /dev/null +++ b/lib/prompts.lua @@ -0,0 +1,70 @@ +local prompts = {} + +prompts.section_summarization = [[You are extracting information for a later cover design process. + +This is only one section of a larger story. + +Do NOT attempt to design a cover. +Do NOT decide what should appear on the cover. +Do NOT write a summary. + +Extract only information that may be useful when creating a cover after all story sections have been analyzed. + +Return: + +GENRE: +- up to 5 items + +SETTING: +- locations +- environments +- time period + +CHARACTERS: +- names +- physical descriptions +- distinctive visual traits + +CREATURES: +- notable creatures + +OBJECTS: +- important recurring items + +VISUAL MOTIFS: +- recurring imagery +- symbols +- repeated visual elements + +THEMES: +- major themes + +MOOD: +- emotional tone + +COLORS: +- colors strongly associated with scenes or imagery + +MEMORABLE VISUAL SCENES: +- 3-10 visually striking moments + +CONFIDENCE: +- how central each item appears to be + +]] + +prompts.combine_summaries = [[Take the common portions from the following text and produce a single simplified list of information: + +]] + +prompts.cover_brief = [[Generate a cover brief from the following: + +]] + +prompts.for_gemini = [[ + +Generate an ebook cover inset within an empty white border 10% larger than the cover using the following cover brief. It should be a flat graphic design file, with no artifacts of a physical object. The aspect ratio of an ebook is tall, not wide. The white border is very important, and should take up 10% of the area of the image. The title and author should only appear once on the cover. It must be sutiable for printing. Do not include the word "by" when adding the author's name to the cover. + +]] + +return prompts diff --git a/make_cover_briefs.lua b/make_cover_briefs.lua index 225804a..5c650e8 100755 --- a/make_cover_briefs.lua +++ b/make_cover_briefs.lua @@ -2,88 +2,11 @@ package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path local utility = require "utility" +local prompts = require "prompts" local default_model = "gemma4:12b-mlx" local maximum_bytes = 40000 -local partial_prompt = [[You are extracting information for a later cover design process. - -This is only one section of a larger story. - -Do NOT attempt to design a cover. -Do NOT decide what should appear on the cover. -Do NOT write a summary. - -Extract only information that may be useful when creating a cover after all story sections have been analyzed. - -Return: - -GENRE: -- up to 5 items - -SETTING: -- locations -- environments -- time period - -CHARACTERS: -- names -- physical descriptions -- distinctive visual traits - -CREATURES: -- notable creatures - -OBJECTS: -- important recurring items - -VISUAL MOTIFS: -- recurring imagery -- symbols -- repeated visual elements - -THEMES: -- major themes - -MOOD: -- emotional tone - -COLORS: -- colors strongly associated with scenes or imagery - -MEMORABLE VISUAL SCENES: -- 3-10 visually striking moments - -CONFIDENCE: -- how central each item appears to be - -]] - -local squish_partials_prompt = [[Take the common portions from the following text and produce a single simplified list of information: - -]] - -local cover_brief_prompt = [[Generate a cover brief from the following: - -]] - -local segmented_cover_brief_prompt = [[Generate a single cover brief. - -Prefer elements that: -- appear repeatedly across chunks -- have highest confidence -- best represent the entire story - -Avoid minor plot events. - -]] - -local gemini_prompt = [[ - -Generate an ebook cover inset within an empty white border 10% larger than the cover using the following cover brief. It should be a flat graphic design file, with no artifacts of a physical object. The aspect ratio of an ebook is tall, not wide. The white border is very important, and should take up 10% of the area of the image. The title and author should only appear once on the cover. It must be sutiable for printing. Do not include the word "by" when adding the author's name to the cover. - -]] - local read_all = function(file_name) return utility.open(file_name, "r", function(file) return file:read("*all") @@ -170,6 +93,51 @@ tree = function(path, fn) end) end +local generate_cover_brief = function(name) + local text = read_all("intermediates/" .. name) + print("Generating cover brief.") + text = send_prompt(prompts.cover_brief .. text) + write_all("cover_briefs/" .. name, text) + write_all("for_gemini/" .. name, "Title: " .. name:sub(1, -5) .. prompts.for_gemini .. text) +end + +local combine_summaries = function(name) + local summaries, i = {}, 1 + while true do + local summary_path = "intermediates/" .. name:sub(1, -5) .. "-summaries-" .. i .. ".txt" + if utility.path_exists(summary_path) then + print("Restored summary " .. i .. ".") + summaries[i] = read_all(summary_path) + else + -- TODO how do we determine whether we ran out or needed to keep going? + end + end + -- TODO I lost track of what I'm trying to do here +end + +os.execute("mkdir -p extracted_texts" .. utility.commands.silence_output) +os.execute("mkdir -p intermediates" .. utility.commands.silence_output) +os.execute("mkdir -p cover_briefs" .. utility.commands.silence_output) +os.execute("mkdir -p for_gemini" .. utility.commands.silence_output) + +tree("extracted_texts", function(file_name) + local _, name, extension = utility.split_path_components(file_name) + if utility.path_exists("cover_briefs/" .. name) then return end + + if utility.path_exists("intermediates/" .. name) then + generate_cover_brief(name) + return + end + + if utility.path_exists("intermediates/" .. name:sub(1, -5) .. "-summaries-1.txt") then + combine_summaries(name) + generate_cover_brief(name) + return + end +end) + + + local process_intermediates = function(text, name) local result local outputs = {} @@ -184,7 +152,7 @@ local process_intermediates = function(text, name) print("Loaded from file.") else local piece = text:sub(i * maximum_bytes, (i + 1) * maximum_bytes - 1) - result = send_prompt(partial_prompt .. piece) + result = send_prompt(prompts.section_summarization .. piece) write_all(intermediate_name, result) end @@ -232,15 +200,15 @@ tree("extracted_texts", function(file_name) if #outputs > 10 then local context_slices = {} while #outputs > 2 do - local output_slices = {} + local summary_pieces = {} for i = 1, 10 do if #outputs >= 1 then - output_slices[#output_slices + 1] = table.remove(outputs, 1) + summary_pieces[#summary_pieces + 1] = table.remove(outputs, 1) end end print("Squishing context. " .. #outputs .. " samples remaining.") - result = send_prompt(squish_partials_prompt .. table.concat(output_slices, "\n\n")) + result = send_prompt(prompts.combine_summaries .. table.concat(summary_pieces, "\n\n")) context_slices[#context_slices + 1] = result print(#result .. " characters added to final context.") end @@ -249,10 +217,6 @@ tree("extracted_texts", function(file_name) write_all("intermediates/" .. name:sub(1, -5) .. " CONDENSED.txt", text) end - print("Generating cover brief.") - result = send_prompt(cover_brief_prompt .. text) - write_all("cover_briefs/" .. name, result) - write_all("for_gemini/" .. name, "Title: " .. name:sub(1, -5) .. gemini_prompt .. result) end) timing_report() -- 2.54.0 From 0a2380e60b261afad8846991b8dbd5f2289e73c0 Mon Sep 17 00:00:00 2001 From: Rose Liverman Date: Sat, 22 Aug 2026 21:32:13 -0600 Subject: [PATCH 05/15] finished? --- lib/timing.lua | 26 ++++++++++++++++------ rewrite.lua | 58 ++++++++++++++++++++++++++++++++++++-------------- 2 files changed, 61 insertions(+), 23 deletions(-) diff --git a/lib/timing.lua b/lib/timing.lua index 60ff108..ad4139c 100644 --- a/lib/timing.lua +++ b/lib/timing.lua @@ -1,9 +1,9 @@ -local timing = {} +local timing = { + entries = {} +} -function timing.timer() - return setmetatable({ - entries = {}, - }, timing) +local function to_minutes(t) -- floored to tenths + return math.floor( t / 60 * 10 ) / 10 end function timing:start() @@ -15,8 +15,20 @@ function timing:stop() local entry = self.entries[#self.entries] entry.stop_time = os.time() - -- delta in minutes, floored to tenths - return math.floor( (entry.stop_time - entry.start_time) / 60 * 10 ) / 10 + return to_minutes(entry.stop_time - entry.start_time) -- return delta +end + +function timing:print_report() + local minimum, maximum, sum = math.huge, -math.huge, 0 + for _, data in pairs(self.entries) do + local delta = data.stop_time - data.start_time + if delta > maximum then maximum = delta end + if delta < minimum then minimum = delta end + sum = sum + delta + end + print("Total: " .. to_minutes(sum) .. " minutes.") + print("Averge: " .. to_minutes(sum / #self.entries) .. " minutes. Fastest: " + .. to_minutes(minimum) .. " minutes. Slowest: " .. to_minutes(maximum) .. " minutes.") end return timing diff --git a/rewrite.lua b/rewrite.lua index 3d953bc..678cc1f 100644 --- a/rewrite.lua +++ b/rewrite.lua @@ -2,7 +2,8 @@ package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path local utility = require "utility" --- TODO import timing library +local prompts = require "prompts" +local timing = require "timing" local config = utility.get_config("read-only") if not config.llm_covers then config.llm_covers = {} end @@ -10,7 +11,17 @@ if not config.llm_covers.calibre_location then config.llm_covers.calibre_locatio if not config.llm_covers.default_model then config.llm_covers.default_model = "gemma4:12b-mlx" end if not config.llm_covers.maximum_bytes then config.llm_covers.maximum_bytes = 40000 end --- TODO import prompts, use new names from fork +local read_all = function(file_name) + return utility.open(file_name, "r", function(file) + return file:read("*all") + end) +end +local write_all = function(file_name, text) + return utility.open(file_name, "w", function(file) + file:write(text) + file:write("\n") + end) +end local strip_reasoning = function(text, reasoning_lines) if not reasoning_lines then reasoning_lines = {} end @@ -31,18 +42,27 @@ local strip_reasoning = function(text, reasoning_lines) end end --- TODO import and rewrite send_prompt to use new timing library +local send_prompt = function(text, model) + if type(text) == "table" then text = table.concat(text, "\n") end -local read_all = function(file_name) - return utility.open(file_name, "r", function(file) - return file:read("*all") - end) -end -local write_all = function(file_name, text) - return utility.open(file_name, "w", function(file) + local tmp_file_name = utility.tmp_file_name() + utility.open(tmp_file_name, "w", function(file) file:write(text) - file:write("\n") end) + + timing:start() + -- word wrap breaks the raw output badly, so I need to implement my own for terminal output somehow + local output = utility.capture_safe("cat " .. tmp_file_name:enquote() .. " | ollama run " .. (model or default_model) .. " --nowordwrap") + os.execute("ollama stop " .. (model or default_model)) -- NOTE this makes things slower, but more stable + local delta = timing:stop() + + print("Took " .. delta .. " minutes.") + os.execute("rm " .. tmp_file_name) + if not output then error("ollama failed to generate output") end + + output = output:sub(1, -2) -- strip extra newline from utility.capture_safe + output = strip_reasoning(output) + return output end local make_directories = function() @@ -166,7 +186,7 @@ local condense_summary = function(name, data, sections) print("Loaded condensed summary segment. " .. #sections .. " sections remaining.") else print("Condensing summary. " .. #sections .. " sections remaining.") - local text = send_prompt(squish_partials_prompt .. table.concat(sections_slice, "\n\n")) + local text = send_prompt(prompts.combine_summaries .. table.concat(sections_slice, "\n\n")) condensed_sections[#condensed_sections + 1] = text print(#text .. " characters added to final context.") write_all(intermediate_condensation_file, text) @@ -199,7 +219,7 @@ local generate_summary = function(name, data) else print("Processing section " .. (i + 1) .. " of " .. math.floor(data.extracted_file_size/config.llm_covers.maximum_bytes + 1)) local section = text:sub(i * config.llm_covers.maximum_bytes, (i + 1) * config.llm_covers.maximum_bytes - 1) - section = send_prompt(partial_prompt .. section) + section = send_prompt(prompts.section_summarization .. section) sections[#sections + 1] = section print(#section .. " characters added to intermediate context.") write_all(intermediate_section_file, section) @@ -238,17 +258,23 @@ local generate_cover_briefs = function(books) end print("Generating cover brief.") - text = send_prompt(cover_brief_prompt .. text) + text = send_prompt(prompts.cover_brief .. text) local cover_brief_file = "cover_briefs/" .. name .. ".txt" write_all(cover_brief_file, text) - write_all("for_gemini/" .. name .. ".txt", "Title: " .. name .. gemini_prompt .. text) + write_all("for_gemini/" .. name .. ".txt", "Title: " .. name .. prompts.for_gemini .. text) data.cover_brief_file = cover_brief_file end end make_directories() -generate_cover_briefs(get_books()) + +local books = get_books() +convert_ebooks(books) +generate_cover_briefs(books) + +print("") +timing:print_report() -- ebook_file, extracted_file, extracted_file_size, -- intermediate_file, condensed_intermediate_file, cover_brief_file -- 2.54.0 From 4aa417eaaac3154abdeefb2c113d7e9dce2c9d21 Mon Sep 17 00:00:00 2001 From: Rose Liverman Date: Sat, 22 Aug 2026 21:33:24 -0600 Subject: [PATCH 06/15] replace old file --- make_cover_briefs.lua | 314 +++++++++++++++++++++++++----------------- rewrite.lua | 280 ------------------------------------- 2 files changed, 186 insertions(+), 408 deletions(-) delete mode 100644 rewrite.lua diff --git a/make_cover_briefs.lua b/make_cover_briefs.lua index 5c650e8..678cc1f 100755 --- a/make_cover_briefs.lua +++ b/make_cover_briefs.lua @@ -3,9 +3,13 @@ package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path local utility = require "utility" local prompts = require "prompts" +local timing = require "timing" -local default_model = "gemma4:12b-mlx" -local maximum_bytes = 40000 +local config = utility.get_config("read-only") +if not config.llm_covers then config.llm_covers = {} end +if not config.llm_covers.calibre_location then config.llm_covers.calibre_location = "/Applications/calibre.app" end +if not config.llm_covers.default_model then config.llm_covers.default_model = "gemma4:12b-mlx" end +if not config.llm_covers.maximum_bytes then config.llm_covers.maximum_bytes = 40000 end local read_all = function(file_name) return utility.open(file_name, "r", function(file) @@ -19,19 +23,6 @@ local write_all = function(file_name, text) end) end -local timings = {} -local timing_report = function() - local minimum, maximum, sum = math.huge, 0, 0 - for _, delta in ipairs(timings) do - if delta > maximum then maximum = delta end - if delta < minimum then minimum = delta end - sum = sum + delta - end - print("") - print("Prompts took a total of " .. sum .. " minutes.") - print("Average: " .. math.floor(sum / #timings) .. " minutes. Fastest: " .. minimum .. " minutes. Slowest: " .. maximum .. " minutes.") -end - local strip_reasoning = function(text, reasoning_lines) if not reasoning_lines then reasoning_lines = {} end local tab = text:split("\n") @@ -59,22 +50,29 @@ local send_prompt = function(text, model) file:write(text) end) - local start_time = os.time() + timing:start() -- word wrap breaks the raw output badly, so I need to implement my own for terminal output somehow local output = utility.capture_safe("cat " .. tmp_file_name:enquote() .. " | ollama run " .. (model or default_model) .. " --nowordwrap") os.execute("ollama stop " .. (model or default_model)) -- NOTE this makes things slower, but more stable - local delta = math.floor( (os.time() - start_time) / 60 * 10 ) / 10 + local delta = timing:stop() + print("Took " .. delta .. " minutes.") - timings[#timings + 1] = delta os.execute("rm " .. tmp_file_name) if not output then error("ollama failed to generate output") end + output = output:sub(1, -2) -- strip extra newline from utility.capture_safe - output = strip_reasoning(output) - return output end +local make_directories = function() + os.execute("mkdir -p raw_ebooks" .. utility.commands.silence_output) + os.execute("mkdir -p extracted_texts" .. utility.commands.silence_output) + os.execute("mkdir -p intermediates" .. utility.commands.silence_output) + os.execute("mkdir -p cover_briefs" .. utility.commands.silence_output) + os.execute("mkdir -p for_gemini" .. utility.commands.silence_output) +end + local tree tree = function(path, fn) utility.list(path or ".", function(path_name) @@ -93,130 +91,190 @@ tree = function(path, fn) end) end -local generate_cover_brief = function(name) - local text = read_all("intermediates/" .. name) - print("Generating cover brief.") - text = send_prompt(prompts.cover_brief .. text) - write_all("cover_briefs/" .. name, text) - write_all("for_gemini/" .. name, "Title: " .. name:sub(1, -5) .. prompts.for_gemini .. text) -end +local find_ebook_files = function(books) + if not books then books = {} end -local combine_summaries = function(name) - local summaries, i = {}, 1 - while true do - local summary_path = "intermediates/" .. name:sub(1, -5) .. "-summaries-" .. i .. ".txt" - if utility.path_exists(summary_path) then - print("Restored summary " .. i .. ".") - summaries[i] = read_all(summary_path) - else - -- TODO how do we determine whether we ran out or needed to keep going? - end - end - -- TODO I lost track of what I'm trying to do here -end - -os.execute("mkdir -p extracted_texts" .. utility.commands.silence_output) -os.execute("mkdir -p intermediates" .. utility.commands.silence_output) -os.execute("mkdir -p cover_briefs" .. utility.commands.silence_output) -os.execute("mkdir -p for_gemini" .. utility.commands.silence_output) - -tree("extracted_texts", function(file_name) - local _, name, extension = utility.split_path_components(file_name) - if utility.path_exists("cover_briefs/" .. name) then return end - - if utility.path_exists("intermediates/" .. name) then - generate_cover_brief(name) - return - end - - if utility.path_exists("intermediates/" .. name:sub(1, -5) .. "-summaries-1.txt") then - combine_summaries(name) - generate_cover_brief(name) - return - end -end) - - - -local process_intermediates = function(text, name) - local result - local outputs = {} - - -- TODO check if this would work just as well starting from 1 instead.. - for i = 0, #text/maximum_bytes do - print("Processing section " .. (i + 1) .. " of " .. math.floor(#text/maximum_bytes + 1)) - - local intermediate_name = "intermediates/" .. name:sub(1, -5) .. "-" .. tostring(#outputs + 1) .. ".txt" - if utility.path_exists(intermediate_name) then - result = read_all(intermediate_name) - print("Loaded from file.") - else - local piece = text:sub(i * maximum_bytes, (i + 1) * maximum_bytes - 1) - result = send_prompt(prompts.section_summarization .. piece) - write_all(intermediate_name, result) + tree("raw_ebooks", function(file_name) + local _, name, extension = utility.split_path_components(file_name) + if extension then + name = name:sub(1, -(#extension + 2)) end - outputs[#outputs + 1] = result - print(#result .. " characters added to intermediate context.") - end - - text = table.concat(outputs, "\n\n") - write_all("intermediates/" .. name, text) - print("Saved primary intermediate prompt.") - - for i = 0, #text/maximum_bytes do - local intermediate_name = "intermediates/" .. name:sub(1, -5) .. "-" .. tostring(#outputs + 1) .. ".txt" - if utility.path_exists(intermediate_name) then - os.execute("rm " .. intermediate_name:enquote()) + if not books[name] then + books[name] = {} end - end - print("Removed redundant intermediate prompts.") + books[name].ebook_file = file_name + end) - return result, outputs, text + return books end -tree("extracted_texts", function(file_name) - local path, name, extension = utility.split_path_components(file_name) - if utility.path_exists("cover_briefs/" .. name) then return end +local find_extracted_texts = function(books) + if not books then books = {} end - local text = read_all(file_name) - local result + tree("extracted_texts", function(file_name) + local _, name, extension = utility.split_path_components(file_name) + if extension then + name = name:sub(1, -(#extension + 2)) + end - -- TODO there is no way to recognize a dangling intermediate and pickup from there - print(name) - local outputs + if not books[name] then + books[name] = {} + end + books[name].extracted_file = file_name + books[name].extracted_file_size = utility.file_size(file_name) + end) - -- restore previous intermediate - -- (this is where I realized I should redo how I'm processing all of this) - if utility.path_exists("intermediates/" .. name) then + return books +end + +local find_cover_briefs = function(books) + if not books then books = {} end + + tree("cover_briefs", function(file_name) + local _, name, extension = utility.split_path_components(file_name) + if extension then + name = name:sub(1, -(#extension + 2)) + end + + if not books[name] then + books[name] = {} + end + books[name].cover_brief_file = file_name + end) + + return books +end + +local get_books = function() + local books = {} + find_ebook_files(books) + find_extracted_texts(books) + find_cover_briefs(books) + return books +end + +local convert_ebooks = function(books) + for name, data in pairs(books) do + if data.extracted_file then return end + if not data.ebook_file then return end + + local extracted_file = "extracted_texts/" .. name .. ".txt" + os.execute(config.llm_covers.calibre_location .. "/Contents/MacOS/ebook-convert " + .. data.ebook_file:enquote() .. " " .. extracted_file:enquote()) + + if utility.path_exists(extracted_file) then + data.extracted_file = extracted_file + end end +end - if #text > maximum_bytes then - -- previously overwrites result, text, outputs - result, outputs, text = process_intermediates(text, name) - end +local condense_summary = function(name, data, sections) + local condensed_sections = {} - -- handle too much context (works with up to 100 slices) - if #outputs > 10 then - local context_slices = {} - while #outputs > 2 do - local summary_pieces = {} - for i = 1, 10 do - if #outputs >= 1 then - summary_pieces[#summary_pieces + 1] = table.remove(outputs, 1) - end + while #sections > 2 do -- this intentionally can skip the end of a book? + local sections_slice = {} + for i = 1, 10 do + if #sections >= 1 then + sections_slice[#sections_slice + 1] = table.remove(sections, 1) end - - print("Squishing context. " .. #outputs .. " samples remaining.") - result = send_prompt(prompts.combine_summaries .. table.concat(summary_pieces, "\n\n")) - context_slices[#context_slices + 1] = result - print(#result .. " characters added to final context.") end - text = table.concat(context_slices, "\n\n") - write_all("intermediates/" .. name:sub(1, -5) .. " CONDENSED.txt", text) + local intermediate_condensation_file = "intermediates/" .. name .. "-" .. (#condensed_sections + 1) .. "-condensed.txt" + if utility.path_exists(intermediate_condensation_file) then + condensed_sections[#condensed_sections + 1] = read_all(intermediate_condensation_file) + print("Loaded condensed summary segment. " .. #sections .. " sections remaining.") + else + print("Condensing summary. " .. #sections .. " sections remaining.") + local text = send_prompt(prompts.combine_summaries .. table.concat(sections_slice, "\n\n")) + condensed_sections[#condensed_sections + 1] = text + print(#text .. " characters added to final context.") + write_all(intermediate_condensation_file, text) + end end -end) + text = table.concat(condensed_sections, "\n\n") + local condensed_intermediate_file = "intermediates/" .. name .. " CONDENSED.txt" + write_all(condensed_intermediate_file, text) + data.condensed_intermediate_file = condensed_intermediate_file -timing_report() + -- if deleted any sooner, could break resuming + for i = 1, #condensed_sections do + local intermediate_condensation_file = "intermediates/" .. name .. "-" .. i .. "-condensed.txt" + os.execute("rm " .. intermediate_condensation_file:enquote()) + end + + return text +end + +local generate_summary = function(name, data) + local text = read_all(data.extracted_file) + local sections = {} + + for i = 0, data.extracted_file_size/config.llm_covers.maximum_bytes do + local intermediate_section_file = "intermediates/" .. name .. "-" .. i .. ".txt" + if utility.path_exists(intermediate_section_file) then + sections[#sections + 1] = read_all(intermediate_section_file) + print("Loaded section " .. (i + 1) .. ".") + else + print("Processing section " .. (i + 1) .. " of " .. math.floor(data.extracted_file_size/config.llm_covers.maximum_bytes + 1)) + local section = text:sub(i * config.llm_covers.maximum_bytes, (i + 1) * config.llm_covers.maximum_bytes - 1) + section = send_prompt(prompts.section_summarization .. section) + sections[#sections + 1] = section + print(#section .. " characters added to intermediate context.") + write_all(intermediate_section_file, section) + end + end + + text = table.concat(sections, "\n\n") + local intermediate_file = "intermediates/" .. name .. ".txt" + write_all(intermediate_file, text) + data.intermediate_file = intermediate_file + + if #sections > 10 then + text = condense_summary(name, data, sections) + end + + -- if these temporary files are removed before condense_summary is called, + -- sections cannot be individually loaded, which breaks condense_summary + for i = 0, data.extracted_file_size/config.llm_covers.maximum_bytes do + local intermediate_section_file = "intermediates/" .. name .. "-" .. i .. ".txt" + os.execute("rm " .. intermediate_section_file:enquote()) + end + + return text +end + +local generate_cover_briefs = function(books) + for name, data in pairs(books) do + if data.cover_brief_file then return end + print(name) + + local text + if data.extracted_file_size > config.llm_covers.maximum_bytes then + text = generate_summary(name, data) + else + text = read_all(extracted_file) + end + + print("Generating cover brief.") + text = send_prompt(prompts.cover_brief .. text) + + local cover_brief_file = "cover_briefs/" .. name .. ".txt" + write_all(cover_brief_file, text) + write_all("for_gemini/" .. name .. ".txt", "Title: " .. name .. prompts.for_gemini .. text) + data.cover_brief_file = cover_brief_file + end +end + +make_directories() + +local books = get_books() +convert_ebooks(books) +generate_cover_briefs(books) + +print("") +timing:print_report() + +-- ebook_file, extracted_file, extracted_file_size, +-- intermediate_file, condensed_intermediate_file, cover_brief_file diff --git a/rewrite.lua b/rewrite.lua deleted file mode 100644 index 678cc1f..0000000 --- a/rewrite.lua +++ /dev/null @@ -1,280 +0,0 @@ -#!/usr/bin/env luajit - -package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path -local utility = require "utility" -local prompts = require "prompts" -local timing = require "timing" - -local config = utility.get_config("read-only") -if not config.llm_covers then config.llm_covers = {} end -if not config.llm_covers.calibre_location then config.llm_covers.calibre_location = "/Applications/calibre.app" end -if not config.llm_covers.default_model then config.llm_covers.default_model = "gemma4:12b-mlx" end -if not config.llm_covers.maximum_bytes then config.llm_covers.maximum_bytes = 40000 end - -local read_all = function(file_name) - return utility.open(file_name, "r", function(file) - return file:read("*all") - end) -end -local write_all = function(file_name, text) - return utility.open(file_name, "w", function(file) - file:write(text) - file:write("\n") - end) -end - -local strip_reasoning = function(text, reasoning_lines) - if not reasoning_lines then reasoning_lines = {} end - local tab = text:split("\n") - table.remove(tab, 1) -- remove "Thinking..." - - while true do - local done = tab[1] == "...done thinking." - local line = table.remove(tab, 1) - if done then - table.remove(tab, 1) -- remove newline after end of thinking - return table.concat(tab, "\n") - elseif #tab < 1 then - return text -- no reasoning output - else - reasoning_lines[#reasoning_lines + 1] = line -- export thinking lines - end - end -end - -local send_prompt = function(text, model) - if type(text) == "table" then text = table.concat(text, "\n") end - - local tmp_file_name = utility.tmp_file_name() - utility.open(tmp_file_name, "w", function(file) - file:write(text) - end) - - timing:start() - -- word wrap breaks the raw output badly, so I need to implement my own for terminal output somehow - local output = utility.capture_safe("cat " .. tmp_file_name:enquote() .. " | ollama run " .. (model or default_model) .. " --nowordwrap") - os.execute("ollama stop " .. (model or default_model)) -- NOTE this makes things slower, but more stable - local delta = timing:stop() - - print("Took " .. delta .. " minutes.") - os.execute("rm " .. tmp_file_name) - if not output then error("ollama failed to generate output") end - - output = output:sub(1, -2) -- strip extra newline from utility.capture_safe - output = strip_reasoning(output) - return output -end - -local make_directories = function() - os.execute("mkdir -p raw_ebooks" .. utility.commands.silence_output) - os.execute("mkdir -p extracted_texts" .. utility.commands.silence_output) - os.execute("mkdir -p intermediates" .. utility.commands.silence_output) - os.execute("mkdir -p cover_briefs" .. utility.commands.silence_output) - os.execute("mkdir -p for_gemini" .. utility.commands.silence_output) -end - -local tree -tree = function(path, fn) - utility.list(path or ".", function(path_name) - local blacklist = { - [".git"] = true, - [".gitkeep"] = true, - [".gitignore"] = true, - [".DS_Store"] = true, - } - if blacklist[path_name] then return end - if utility.is_file(path_name) then - fn(path_name) - else - tree(path .. utility.path_separator .. path_name, fn) - end - end) -end - -local find_ebook_files = function(books) - if not books then books = {} end - - tree("raw_ebooks", function(file_name) - local _, name, extension = utility.split_path_components(file_name) - if extension then - name = name:sub(1, -(#extension + 2)) - end - - if not books[name] then - books[name] = {} - end - books[name].ebook_file = file_name - end) - - return books -end - -local find_extracted_texts = function(books) - if not books then books = {} end - - tree("extracted_texts", function(file_name) - local _, name, extension = utility.split_path_components(file_name) - if extension then - name = name:sub(1, -(#extension + 2)) - end - - if not books[name] then - books[name] = {} - end - books[name].extracted_file = file_name - books[name].extracted_file_size = utility.file_size(file_name) - end) - - return books -end - -local find_cover_briefs = function(books) - if not books then books = {} end - - tree("cover_briefs", function(file_name) - local _, name, extension = utility.split_path_components(file_name) - if extension then - name = name:sub(1, -(#extension + 2)) - end - - if not books[name] then - books[name] = {} - end - books[name].cover_brief_file = file_name - end) - - return books -end - -local get_books = function() - local books = {} - find_ebook_files(books) - find_extracted_texts(books) - find_cover_briefs(books) - return books -end - -local convert_ebooks = function(books) - for name, data in pairs(books) do - if data.extracted_file then return end - if not data.ebook_file then return end - - local extracted_file = "extracted_texts/" .. name .. ".txt" - os.execute(config.llm_covers.calibre_location .. "/Contents/MacOS/ebook-convert " - .. data.ebook_file:enquote() .. " " .. extracted_file:enquote()) - - if utility.path_exists(extracted_file) then - data.extracted_file = extracted_file - end - end -end - -local condense_summary = function(name, data, sections) - local condensed_sections = {} - - while #sections > 2 do -- this intentionally can skip the end of a book? - local sections_slice = {} - for i = 1, 10 do - if #sections >= 1 then - sections_slice[#sections_slice + 1] = table.remove(sections, 1) - end - end - - local intermediate_condensation_file = "intermediates/" .. name .. "-" .. (#condensed_sections + 1) .. "-condensed.txt" - if utility.path_exists(intermediate_condensation_file) then - condensed_sections[#condensed_sections + 1] = read_all(intermediate_condensation_file) - print("Loaded condensed summary segment. " .. #sections .. " sections remaining.") - else - print("Condensing summary. " .. #sections .. " sections remaining.") - local text = send_prompt(prompts.combine_summaries .. table.concat(sections_slice, "\n\n")) - condensed_sections[#condensed_sections + 1] = text - print(#text .. " characters added to final context.") - write_all(intermediate_condensation_file, text) - end - end - - text = table.concat(condensed_sections, "\n\n") - local condensed_intermediate_file = "intermediates/" .. name .. " CONDENSED.txt" - write_all(condensed_intermediate_file, text) - data.condensed_intermediate_file = condensed_intermediate_file - - -- if deleted any sooner, could break resuming - for i = 1, #condensed_sections do - local intermediate_condensation_file = "intermediates/" .. name .. "-" .. i .. "-condensed.txt" - os.execute("rm " .. intermediate_condensation_file:enquote()) - end - - return text -end - -local generate_summary = function(name, data) - local text = read_all(data.extracted_file) - local sections = {} - - for i = 0, data.extracted_file_size/config.llm_covers.maximum_bytes do - local intermediate_section_file = "intermediates/" .. name .. "-" .. i .. ".txt" - if utility.path_exists(intermediate_section_file) then - sections[#sections + 1] = read_all(intermediate_section_file) - print("Loaded section " .. (i + 1) .. ".") - else - print("Processing section " .. (i + 1) .. " of " .. math.floor(data.extracted_file_size/config.llm_covers.maximum_bytes + 1)) - local section = text:sub(i * config.llm_covers.maximum_bytes, (i + 1) * config.llm_covers.maximum_bytes - 1) - section = send_prompt(prompts.section_summarization .. section) - sections[#sections + 1] = section - print(#section .. " characters added to intermediate context.") - write_all(intermediate_section_file, section) - end - end - - text = table.concat(sections, "\n\n") - local intermediate_file = "intermediates/" .. name .. ".txt" - write_all(intermediate_file, text) - data.intermediate_file = intermediate_file - - if #sections > 10 then - text = condense_summary(name, data, sections) - end - - -- if these temporary files are removed before condense_summary is called, - -- sections cannot be individually loaded, which breaks condense_summary - for i = 0, data.extracted_file_size/config.llm_covers.maximum_bytes do - local intermediate_section_file = "intermediates/" .. name .. "-" .. i .. ".txt" - os.execute("rm " .. intermediate_section_file:enquote()) - end - - return text -end - -local generate_cover_briefs = function(books) - for name, data in pairs(books) do - if data.cover_brief_file then return end - print(name) - - local text - if data.extracted_file_size > config.llm_covers.maximum_bytes then - text = generate_summary(name, data) - else - text = read_all(extracted_file) - end - - print("Generating cover brief.") - text = send_prompt(prompts.cover_brief .. text) - - local cover_brief_file = "cover_briefs/" .. name .. ".txt" - write_all(cover_brief_file, text) - write_all("for_gemini/" .. name .. ".txt", "Title: " .. name .. prompts.for_gemini .. text) - data.cover_brief_file = cover_brief_file - end -end - -make_directories() - -local books = get_books() -convert_ebooks(books) -generate_cover_briefs(books) - -print("") -timing:print_report() - --- ebook_file, extracted_file, extracted_file_size, --- intermediate_file, condensed_intermediate_file, cover_brief_file -- 2.54.0 From 0e69ea9aaf53bf57af7e438157b8487b6aa6faf8 Mon Sep 17 00:00:00 2001 From: tangent Date: Sat, 22 Aug 2026 21:38:57 -0600 Subject: [PATCH 07/15] Update ReadMe.md --- ReadMe.md | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/ReadMe.md b/ReadMe.md index f14c0f2..a086997 100644 --- a/ReadMe.md +++ b/ReadMe.md @@ -33,12 +33,13 @@ to check accuracy. ## Tasks - [ ] Add timestamps to when a prompt is sent. -- [ ] Specify ETA when sending a prompt (1.5-8 minutes depending on length). (Show as end time as well as relative time.) +- [ ] Specify ETA when sending a prompt (2.5-8 minutes depending on length). (Show as end time as well as relative time.) - [ ] option to delete input files or intermediate files at end of each operation step -- [ ] make script resumable instead of having to restart per book +- [x] make script resumable instead of having to restart per book - [ ] instead of placeholder directories, `mkdir -p` should be used to only make them when necessary - [ ] dry run to estimate total time before running full script, print starting ETA -- [ ] Build a list of files to work on and current states before doing anything by looping over source directories -- [ ] Combine everything into one script to manage things +- [x] Build a list of files to work on and current states before doing anything by looping over source directories +- [x] Combine everything into one script to manage things - [ ] Allow specifiying a filter on what to process - [ ] Allow changing the maximum bytes per prompt dynamically (run smaller while I'm doing other things, run full while I'm asleep) +- [ ] specify config format -- 2.54.0 From 4073c057120710d7bcb9ee4762dfdd612b68960b Mon Sep 17 00:00:00 2001 From: Rose Liverman Date: Sat, 22 Aug 2026 21:41:02 -0600 Subject: [PATCH 08/15] remove placeholder directories and create them as needed --- .gitignore | 4 ---- ReadMe.md | 2 +- cover_briefs/.gitkeep | 0 extracted_texts/.gitkeep | 0 for_gemini/.gitkeep | 0 intermediates/.gitkeep | 0 6 files changed, 1 insertion(+), 5 deletions(-) delete mode 100644 cover_briefs/.gitkeep delete mode 100644 extracted_texts/.gitkeep delete mode 100644 for_gemini/.gitkeep delete mode 100644 intermediates/.gitkeep diff --git a/.gitignore b/.gitignore index bea6c7c..84a89e8 100644 --- a/.gitignore +++ b/.gitignore @@ -1,12 +1,8 @@ .DS_Store extracted_texts/** -!extracted_texts/.gitkeep intermediates/** -!intermediates/.gitkeep cover_briefs/** -!cover_briefs/.gitkeep for_gemini/** -!for_gemini/.gitkeep raw_ebooks/** config.json config.json.lock diff --git a/ReadMe.md b/ReadMe.md index f14c0f2..56c1ac1 100644 --- a/ReadMe.md +++ b/ReadMe.md @@ -36,7 +36,7 @@ to check accuracy. - [ ] Specify ETA when sending a prompt (1.5-8 minutes depending on length). (Show as end time as well as relative time.) - [ ] option to delete input files or intermediate files at end of each operation step - [ ] make script resumable instead of having to restart per book -- [ ] instead of placeholder directories, `mkdir -p` should be used to only make them when necessary +- [x] instead of placeholder directories, `mkdir -p` should be used to only make them when necessary - [ ] dry run to estimate total time before running full script, print starting ETA - [ ] Build a list of files to work on and current states before doing anything by looping over source directories - [ ] Combine everything into one script to manage things diff --git a/cover_briefs/.gitkeep b/cover_briefs/.gitkeep deleted file mode 100644 index e69de29..0000000 diff --git a/extracted_texts/.gitkeep b/extracted_texts/.gitkeep deleted file mode 100644 index e69de29..0000000 diff --git a/for_gemini/.gitkeep b/for_gemini/.gitkeep deleted file mode 100644 index e69de29..0000000 diff --git a/intermediates/.gitkeep b/intermediates/.gitkeep deleted file mode 100644 index e69de29..0000000 -- 2.54.0 From a931967dbd8551b98203fa68b2e899740d3f20fa Mon Sep 17 00:00:00 2001 From: Rose Liverman Date: Sat, 22 Aug 2026 22:33:17 -0600 Subject: [PATCH 09/15] presents estimated time to run before running --- ReadMe.md | 46 ++++++++++++++++++++++++++----------------- lib/timing.lua | 36 +++++++++++++++++---------------- make_cover_briefs.lua | 22 ++++++++++++++++++--- 3 files changed, 66 insertions(+), 38 deletions(-) diff --git a/ReadMe.md b/ReadMe.md index 221330c..ce5c611 100644 --- a/ReadMe.md +++ b/ReadMe.md @@ -5,31 +5,41 @@ A simple tool for using local LLMs to generate cover briefs for ebooks. - A Mac mini M4 base model (original version) or better M-series computer - Ollama and `gemma4:12b-mlx` - LuaJIT +- calibre ## Usage -1. Use a tool like [audiobook-creator](https://github.com/prakharsr/audiobook-creator) to extract the text of an audiobook. -2. Place it in `extracted_texts/TITLE by AUTHOR.txt`. (The format of this doesn't matter except that the file name sans extension will be used in the output.) -3. Run `make_cover_briefs.lua`. -4. Cover briefs will be placed in `cover_briefs`. -5. A prompt ready to paste directly into Gemini will be in `for_gemini`. An extra border is prompted for to make cropping easy to remove the Gemini watermark. (There are other small changes to increase likelihood of Gemini following the prompt correctly and making a good output.) +1. Create `config.json` if necessary (see format below). +2. Run `make_cover_briefs.lua` once to create necessary directories. +3. Put ebooks in `raw_ebooks`. +4. Run `make_cover_briefs.lua` + +Cover briefs will be placed in `cover_briefs`. A prompt ready to paste directly +into Gemini will be in `for_gemini`. An extra border is prompted for to make +cropping easy to remove the Gemini watermark. (There are other small changes to +increase likelihood of Gemini following the prompt correctly and making a good +output.) ### Gemini Issues -None of this works unless you **disable Gemini's personalization/memory**, because it pollutes context -extremely badly and will randomly add elements and rejections from different conversations everywhere. +None of this works unless you **disable Gemini's personalization/memory**, +because it pollutes context extremely badly and will randomly add elements and +rejections from different conversations everywhere. -If Gemini rejects the prompt, adding Remove any elements that may go against guidelines before generating the image. +If Gemini rejects the prompt, adding +Remove any elements that may go against guidelines before generating the image. to the opening paragraph can fix that issue. -- This suddenly got a lot less effective. I recommend instead telling it to rewrite the prompt and then - in a separate conversation, use that instead. I also recommend one retry before modifying it. +- This suddenly got a lot less effective. I recommend instead telling it to + rewrite the prompt and then in a separate conversation, use that instead. I + also recommend one retry before modifying it. -Gemini is really bad about adding hardcover seams. Despite being instructed not to, it shows up sometimes. -More strong prompting breaks other requested features, so the best way to deal with it is to request its -removal after the initial image is generated. +Gemini is really bad about adding hardcover seams. Despite being instructed not +to, it shows up sometimes. More strong prompting breaks other requested +features, so the best way to deal with it is to request its removal after the +initial image is generated. -The local model sometimes completely forgets critical details of characters, -so it's important you have some familiarity with the work before running the generator -to check accuracy. +The local model sometimes completely forgets critical details of characters, so +it's important you have some familiarity with the work before running the +generator to check accuracy. ## Tasks - [ ] Add timestamps to when a prompt is sent. @@ -37,9 +47,9 @@ to check accuracy. - [ ] option to delete input files or intermediate files at end of each operation step - [x] instead of placeholder directories, `mkdir -p` should be used to only make them when necessary - [x] make script resumable instead of having to restart per book -- [ ] dry run to estimate total time before running full script, print starting ETA +- [x] dry run to estimate total time before running full script, print starting ETA - [x] Build a list of files to work on and current states before doing anything by looping over source directories - [x] Combine everything into one script to manage things -- [ ] Allow specifiying a filter on what to process +- [ ] Allow specifying a filter on what to process - [ ] Allow changing the maximum bytes per prompt dynamically (run smaller while I'm doing other things, run full while I'm asleep) - [ ] specify config format diff --git a/lib/timing.lua b/lib/timing.lua index ad4139c..e770603 100644 --- a/lib/timing.lua +++ b/lib/timing.lua @@ -1,34 +1,36 @@ -local timing = { - entries = {} -} +local timing = {} +local entries = {} -local function to_minutes(t) -- floored to tenths - return math.floor( t / 60 * 10 ) / 10 +function timing.seconds_to_minutes(seconds) -- floored to tenths + return math.floor( seconds / 60 * 10 ) / 10 end -function timing:start() - -- if not self then self = {} end - self.entries[#self.entries + 1] = { start_time = os.time() } +function timing.start() + entries[#entries + 1] = { start_time = os.time() } end -function timing:stop() - local entry = self.entries[#self.entries] +function timing.stop() + local entry = entries[#entries] entry.stop_time = os.time() + entry.delta = entry.stop_time - entry.start_time - return to_minutes(entry.stop_time - entry.start_time) -- return delta + return timing.seconds_to_minutes(entry.delta) end -function timing:print_report() +function timing.print_report() local minimum, maximum, sum = math.huge, -math.huge, 0 - for _, data in pairs(self.entries) do - local delta = data.stop_time - data.start_time + for _, data in pairs(entries) do + local delta = data.delta if delta > maximum then maximum = delta end if delta < minimum then minimum = delta end sum = sum + delta end - print("Total: " .. to_minutes(sum) .. " minutes.") - print("Averge: " .. to_minutes(sum / #self.entries) .. " minutes. Fastest: " - .. to_minutes(minimum) .. " minutes. Slowest: " .. to_minutes(maximum) .. " minutes.") + + print("Total: " .. timing.seconds_to_minutes(sum) .. " minutes.") + print("Averge: " .. timing.seconds_to_minutes(sum / #entries) + .. " minutes. Fastest: " .. timing.seconds_to_minutes(minimum) + .. " minutes. Slowest: " .. timing.seconds_to_minutes(maximum) + .. " minutes.") end return timing diff --git a/make_cover_briefs.lua b/make_cover_briefs.lua index 678cc1f..1b4279b 100755 --- a/make_cover_briefs.lua +++ b/make_cover_briefs.lua @@ -50,11 +50,11 @@ local send_prompt = function(text, model) file:write(text) end) - timing:start() + timing.start() -- word wrap breaks the raw output badly, so I need to implement my own for terminal output somehow local output = utility.capture_safe("cat " .. tmp_file_name:enquote() .. " | ollama run " .. (model or default_model) .. " --nowordwrap") os.execute("ollama stop " .. (model or default_model)) -- NOTE this makes things slower, but more stable - local delta = timing:stop() + local delta = timing.stop() print("Took " .. delta .. " minutes.") os.execute("rm " .. tmp_file_name) @@ -267,14 +267,30 @@ local generate_cover_briefs = function(books) end end +local estimate_time_required = function(books) + local sum, count = 0, 0 + for name, data in pairs(books) do + if data.cover_brief_file then return end + sum = sum + 60 * (2 + data.extracted_file_size/(config.llm_covers.maximum_bytes/7.5)) + count = count + 1 + end + + print(count .. " cover briefs to generate. Estimated time: " + .. timing.seconds_to_minutes(sum) .. " minutes.") +end + make_directories() local books = get_books() convert_ebooks(books) + +estimate_time_required(books) +print("") + generate_cover_briefs(books) print("") -timing:print_report() +timing.print_report() -- ebook_file, extracted_file, extracted_file_size, -- intermediate_file, condensed_intermediate_file, cover_brief_file -- 2.54.0 From 886c5c44267b5bf0bd86953cfb1f4e94f17c30c0 Mon Sep 17 00:00:00 2001 From: Rose Liverman Date: Sat, 22 Aug 2026 22:40:26 -0600 Subject: [PATCH 10/15] config format specified --- ReadMe.md | 18 ++++++++++++++++-- 1 file changed, 16 insertions(+), 2 deletions(-) diff --git a/ReadMe.md b/ReadMe.md index ce5c611..1bf832f 100644 --- a/ReadMe.md +++ b/ReadMe.md @@ -19,6 +19,20 @@ cropping easy to remove the Gemini watermark. (There are other small changes to increase likelihood of Gemini following the prompt correctly and making a good output.) +### Config file format +```json +{ + "llm_covers": { + "calibre_location": "/Applications/calibre.app", + "default_model": "gemma4:12b-mlx", + "maximum_bytes": 40000 + } +} +``` + +The above are the default settings. If you need something different, you only +need to specify the changes. + ### Gemini Issues None of this works unless you **disable Gemini's personalization/memory**, because it pollutes context extremely badly and will randomly add elements and @@ -44,7 +58,7 @@ generator to check accuracy. ## Tasks - [ ] Add timestamps to when a prompt is sent. - [ ] Specify ETA when sending a prompt (2.5-8 minutes depending on length). (Show as end time as well as relative time.) -- [ ] option to delete input files or intermediate files at end of each operation step +- [ ] option to delete input files or intermediate files at end of each major operation segment (or after its finisher..) - [x] instead of placeholder directories, `mkdir -p` should be used to only make them when necessary - [x] make script resumable instead of having to restart per book - [x] dry run to estimate total time before running full script, print starting ETA @@ -52,4 +66,4 @@ generator to check accuracy. - [x] Combine everything into one script to manage things - [ ] Allow specifying a filter on what to process - [ ] Allow changing the maximum bytes per prompt dynamically (run smaller while I'm doing other things, run full while I'm asleep) -- [ ] specify config format +- [x] specify config format -- 2.54.0 From 906ed0628376cdaa1eda5117821a2339ba246e9e Mon Sep 17 00:00:00 2001 From: Rose Liverman Date: Sat, 22 Aug 2026 22:41:04 -0600 Subject: [PATCH 11/15] rm duplicate script :D --- ebook_to_txt.lua | 25 ------------------------- 1 file changed, 25 deletions(-) delete mode 100755 ebook_to_txt.lua diff --git a/ebook_to_txt.lua b/ebook_to_txt.lua deleted file mode 100755 index 5e58090..0000000 --- a/ebook_to_txt.lua +++ /dev/null @@ -1,25 +0,0 @@ -#!/usr/bin/env luajit - -package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path -local utility = require "utility" - -local calibre_location = "/Applications/calibre.app" -local config = utility.get_config("read-only") -if config and config.llm_covers and config.llm_covers.calibre_location then - calibre_location = config.llm_covers.calibre_location -end - -os.execute("mkdir -p raw_ebooks" .. utility.commands.silence_output) -os.execute("mkdir -p extracted_texts" .. utility.commands.silence_output) - -utility.ls("raw_ebooks", function(file_name) - local _, name, extension = utility.split_path_components(file_name) - if extension then - name = name:sub(1, -(#extension + 2)) - end - - local output_name = "extracted_texts/" .. name .. ".txt" - if utility.path_exists(output_name) then return end - - os.execute(calibre_location .. "/Contents/MacOS/ebook-convert " .. ("raw_ebooks/" .. file_name):enquote() .. " " .. output_name:enquote()) -end) -- 2.54.0 From 1595ac2b2c8bed82f4acbfcb11ccd7975863b5b3 Mon Sep 17 00:00:00 2001 From: Rose Liverman Date: Sat, 22 Aug 2026 22:43:11 -0600 Subject: [PATCH 12/15] fix count_characters.lua to use config if its present --- count_characters.lua | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/count_characters.lua b/count_characters.lua index dd88908..774dcb7 100755 --- a/count_characters.lua +++ b/count_characters.lua @@ -3,7 +3,9 @@ package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path local utility = require "utility" -local maximum_bytes = 40000 +local config = utility.get_config("read-only") +if not config.llm_covers then config.llm_covers = {} end +if not config.llm_covers.maximum_bytes then config.llm_covers.maximum_bytes = 40000 end local read_all = function(file_name) return utility.open(file_name, "r", function(file) @@ -25,7 +27,7 @@ for _, path in ipairs{ ".", "extracted_texts", } do local path, name, extension = utility.split_path_components(file_name) local text = read_all(file_name) - print(math.floor(#text/maximum_bytes + 1), math.floor(#text/1000) .. "k", name) + print(math.floor(#text/config.llm_covers.maximum_bytes + 1), math.floor(#text/1000) .. "k", name) end) print("Press enter.") io.read("*line") -- 2.54.0 From 83b26787fece03cd750627f4cac8dd40ba43a1c7 Mon Sep 17 00:00:00 2001 From: Tangent Date: Sun, 23 Aug 2026 01:23:51 -0600 Subject: [PATCH 13/15] fixing small errors / overlooked details --- ReadMe.md | 3 ++- config.json.example | 4 +++- lib/timing.lua | 2 ++ make_cover_briefs.lua | 5 +++-- 4 files changed, 10 insertions(+), 4 deletions(-) diff --git a/ReadMe.md b/ReadMe.md index 1bf832f..9d89be0 100644 --- a/ReadMe.md +++ b/ReadMe.md @@ -31,7 +31,8 @@ output.) ``` The above are the default settings. If you need something different, you only -need to specify the changes. +need to specify the changes. (There is a `config.json.example` file equivalent +to the above.) ### Gemini Issues None of this works unless you **disable Gemini's personalization/memory**, diff --git a/config.json.example b/config.json.example index cc22bbc..df371d6 100644 --- a/config.json.example +++ b/config.json.example @@ -1,5 +1,7 @@ { "llm_covers": { - "calibre_location": "/Applications/calibre.app" + "calibre_location": "/Applications/calibre.app", + "default_model": "gemma4:12b-mlx", + "maximum_bytes": 40000 } } diff --git a/lib/timing.lua b/lib/timing.lua index e770603..96ece80 100644 --- a/lib/timing.lua +++ b/lib/timing.lua @@ -18,6 +18,8 @@ function timing.stop() end function timing.print_report() + if #entries == 0 then return end + local minimum, maximum, sum = math.huge, -math.huge, 0 for _, data in pairs(entries) do local delta = data.delta diff --git a/make_cover_briefs.lua b/make_cover_briefs.lua index 1b4279b..40803f5 100755 --- a/make_cover_briefs.lua +++ b/make_cover_briefs.lua @@ -52,8 +52,8 @@ local send_prompt = function(text, model) timing.start() -- word wrap breaks the raw output badly, so I need to implement my own for terminal output somehow - local output = utility.capture_safe("cat " .. tmp_file_name:enquote() .. " | ollama run " .. (model or default_model) .. " --nowordwrap") - os.execute("ollama stop " .. (model or default_model)) -- NOTE this makes things slower, but more stable + local output = utility.capture_safe("cat " .. tmp_file_name:enquote() .. " | ollama run " .. (model or config.llm_covers.default_model) .. " --nowordwrap") + os.execute("ollama stop " .. (model or config.llm_covers.default_model)) -- NOTE this makes things slower, but more stable local delta = timing.stop() print("Took " .. delta .. " minutes.") @@ -165,6 +165,7 @@ local convert_ebooks = function(books) if utility.path_exists(extracted_file) then data.extracted_file = extracted_file + data.extracted_file_size = utility.file_size(extracted_file) end end end -- 2.54.0 From 994836cc05d30a6b9506424f3a3aada356d5b4a1 Mon Sep 17 00:00:00 2001 From: Tangent Date: Sun, 23 Aug 2026 02:40:19 -0600 Subject: [PATCH 14/15] fix loops, improve time estimator --- make_cover_briefs.lua | 63 +++++++++++++++++++++++++------------------ 1 file changed, 37 insertions(+), 26 deletions(-) diff --git a/make_cover_briefs.lua b/make_cover_briefs.lua index 40803f5..6369d5a 100755 --- a/make_cover_briefs.lua +++ b/make_cover_briefs.lua @@ -156,17 +156,20 @@ end local convert_ebooks = function(books) for name, data in pairs(books) do - if data.extracted_file then return end - if not data.ebook_file then return end + local function loop() + if data.extracted_file then return end + if not data.ebook_file then return end - local extracted_file = "extracted_texts/" .. name .. ".txt" - os.execute(config.llm_covers.calibre_location .. "/Contents/MacOS/ebook-convert " - .. data.ebook_file:enquote() .. " " .. extracted_file:enquote()) + local extracted_file = "extracted_texts/" .. name .. ".txt" + os.execute(config.llm_covers.calibre_location .. "/Contents/MacOS/ebook-convert " + .. data.ebook_file:enquote() .. " " .. extracted_file:enquote()) - if utility.path_exists(extracted_file) then - data.extracted_file = extracted_file - data.extracted_file_size = utility.file_size(extracted_file) + if utility.path_exists(extracted_file) then + data.extracted_file = extracted_file + data.extracted_file_size = utility.file_size(extracted_file) + end end + loop() end end @@ -248,32 +251,38 @@ end local generate_cover_briefs = function(books) for name, data in pairs(books) do - if data.cover_brief_file then return end - print(name) + local function loop() + if data.cover_brief_file then return end + print(name) - local text - if data.extracted_file_size > config.llm_covers.maximum_bytes then - text = generate_summary(name, data) - else - text = read_all(extracted_file) + local text + if data.extracted_file_size > config.llm_covers.maximum_bytes then + text = generate_summary(name, data) + else + text = read_all(data.extracted_file) + end + + print("Generating cover brief.") + text = send_prompt(prompts.cover_brief .. text) + + local cover_brief_file = "cover_briefs/" .. name .. ".txt" + write_all(cover_brief_file, text) + write_all("for_gemini/" .. name .. ".txt", "Title: " .. name .. prompts.for_gemini .. text) + data.cover_brief_file = cover_brief_file end - - print("Generating cover brief.") - text = send_prompt(prompts.cover_brief .. text) - - local cover_brief_file = "cover_briefs/" .. name .. ".txt" - write_all(cover_brief_file, text) - write_all("for_gemini/" .. name .. ".txt", "Title: " .. name .. prompts.for_gemini .. text) - data.cover_brief_file = cover_brief_file + loop() end end local estimate_time_required = function(books) local sum, count = 0, 0 for name, data in pairs(books) do - if data.cover_brief_file then return end - sum = sum + 60 * (2 + data.extracted_file_size/(config.llm_covers.maximum_bytes/7.5)) - count = count + 1 + local function loop() + if data.cover_brief_file then return end + sum = sum + 60 * (1.5 + data.extracted_file_size/(config.llm_covers.maximum_bytes/7.5)) + count = count + 1 + end + loop() end print(count .. " cover briefs to generate. Estimated time: " @@ -293,5 +302,7 @@ generate_cover_briefs(books) print("") timing.print_report() +-- utility.print_table(books) + -- ebook_file, extracted_file, extracted_file_size, -- intermediate_file, condensed_intermediate_file, cover_brief_file -- 2.54.0 From b95c0c21583bcfb981fa85a327d97e3924713961 Mon Sep 17 00:00:00 2001 From: Tangent Date: Sun, 23 Aug 2026 17:23:22 -0600 Subject: [PATCH 15/15] better time estimate? --- make_cover_briefs.lua | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/make_cover_briefs.lua b/make_cover_briefs.lua index 6369d5a..b7c744b 100755 --- a/make_cover_briefs.lua +++ b/make_cover_briefs.lua @@ -279,7 +279,7 @@ local estimate_time_required = function(books) for name, data in pairs(books) do local function loop() if data.cover_brief_file then return end - sum = sum + 60 * (1.5 + data.extracted_file_size/(config.llm_covers.maximum_bytes/7.5)) + sum = sum + 60 * (1.5 + data.extracted_file_size/(config.llm_covers.maximum_bytes/6)) count = count + 1 end loop() -- 2.54.0