diff --git a/.gitignore b/.gitignore index bea6c7c..84a89e8 100644 --- a/.gitignore +++ b/.gitignore @@ -1,12 +1,8 @@ .DS_Store extracted_texts/** -!extracted_texts/.gitkeep intermediates/** -!intermediates/.gitkeep cover_briefs/** -!cover_briefs/.gitkeep for_gemini/** -!for_gemini/.gitkeep raw_ebooks/** config.json config.json.lock diff --git a/ReadMe.md b/ReadMe.md index f14c0f2..9d89be0 100644 --- a/ReadMe.md +++ b/ReadMe.md @@ -5,40 +5,66 @@ A simple tool for using local LLMs to generate cover briefs for ebooks. - A Mac mini M4 base model (original version) or better M-series computer - Ollama and `gemma4:12b-mlx` - LuaJIT +- calibre ## Usage -1. Use a tool like [audiobook-creator](https://github.com/prakharsr/audiobook-creator) to extract the text of an audiobook. -2. Place it in `extracted_texts/TITLE by AUTHOR.txt`. (The format of this doesn't matter except that the file name sans extension will be used in the output.) -3. Run `make_cover_briefs.lua`. -4. Cover briefs will be placed in `cover_briefs`. -5. A prompt ready to paste directly into Gemini will be in `for_gemini`. An extra border is prompted for to make cropping easy to remove the Gemini watermark. (There are other small changes to increase likelihood of Gemini following the prompt correctly and making a good output.) +1. Create `config.json` if necessary (see format below). +2. Run `make_cover_briefs.lua` once to create necessary directories. +3. Put ebooks in `raw_ebooks`. +4. Run `make_cover_briefs.lua` + +Cover briefs will be placed in `cover_briefs`. A prompt ready to paste directly +into Gemini will be in `for_gemini`. An extra border is prompted for to make +cropping easy to remove the Gemini watermark. (There are other small changes to +increase likelihood of Gemini following the prompt correctly and making a good +output.) + +### Config file format +```json +{ + "llm_covers": { + "calibre_location": "/Applications/calibre.app", + "default_model": "gemma4:12b-mlx", + "maximum_bytes": 40000 + } +} +``` + +The above are the default settings. If you need something different, you only +need to specify the changes. (There is a `config.json.example` file equivalent +to the above.) ### Gemini Issues -None of this works unless you **disable Gemini's personalization/memory**, because it pollutes context -extremely badly and will randomly add elements and rejections from different conversations everywhere. +None of this works unless you **disable Gemini's personalization/memory**, +because it pollutes context extremely badly and will randomly add elements and +rejections from different conversations everywhere. -If Gemini rejects the prompt, adding Remove any elements that may go against guidelines before generating the image. +If Gemini rejects the prompt, adding +Remove any elements that may go against guidelines before generating the image. to the opening paragraph can fix that issue. -- This suddenly got a lot less effective. I recommend instead telling it to rewrite the prompt and then - in a separate conversation, use that instead. I also recommend one retry before modifying it. +- This suddenly got a lot less effective. I recommend instead telling it to + rewrite the prompt and then in a separate conversation, use that instead. I + also recommend one retry before modifying it. -Gemini is really bad about adding hardcover seams. Despite being instructed not to, it shows up sometimes. -More strong prompting breaks other requested features, so the best way to deal with it is to request its -removal after the initial image is generated. +Gemini is really bad about adding hardcover seams. Despite being instructed not +to, it shows up sometimes. More strong prompting breaks other requested +features, so the best way to deal with it is to request its removal after the +initial image is generated. -The local model sometimes completely forgets critical details of characters, -so it's important you have some familiarity with the work before running the generator -to check accuracy. +The local model sometimes completely forgets critical details of characters, so +it's important you have some familiarity with the work before running the +generator to check accuracy. ## Tasks - [ ] Add timestamps to when a prompt is sent. -- [ ] Specify ETA when sending a prompt (1.5-8 minutes depending on length). (Show as end time as well as relative time.) -- [ ] option to delete input files or intermediate files at end of each operation step -- [ ] make script resumable instead of having to restart per book -- [ ] instead of placeholder directories, `mkdir -p` should be used to only make them when necessary -- [ ] dry run to estimate total time before running full script, print starting ETA -- [ ] Build a list of files to work on and current states before doing anything by looping over source directories -- [ ] Combine everything into one script to manage things -- [ ] Allow specifiying a filter on what to process +- [ ] Specify ETA when sending a prompt (2.5-8 minutes depending on length). (Show as end time as well as relative time.) +- [ ] option to delete input files or intermediate files at end of each major operation segment (or after its finisher..) +- [x] instead of placeholder directories, `mkdir -p` should be used to only make them when necessary +- [x] make script resumable instead of having to restart per book +- [x] dry run to estimate total time before running full script, print starting ETA +- [x] Build a list of files to work on and current states before doing anything by looping over source directories +- [x] Combine everything into one script to manage things +- [ ] Allow specifying a filter on what to process - [ ] Allow changing the maximum bytes per prompt dynamically (run smaller while I'm doing other things, run full while I'm asleep) +- [x] specify config format diff --git a/config.json.example b/config.json.example index cc22bbc..df371d6 100644 --- a/config.json.example +++ b/config.json.example @@ -1,5 +1,7 @@ { "llm_covers": { - "calibre_location": "/Applications/calibre.app" + "calibre_location": "/Applications/calibre.app", + "default_model": "gemma4:12b-mlx", + "maximum_bytes": 40000 } } diff --git a/count_characters.lua b/count_characters.lua index dd88908..774dcb7 100755 --- a/count_characters.lua +++ b/count_characters.lua @@ -3,7 +3,9 @@ package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path local utility = require "utility" -local maximum_bytes = 40000 +local config = utility.get_config("read-only") +if not config.llm_covers then config.llm_covers = {} end +if not config.llm_covers.maximum_bytes then config.llm_covers.maximum_bytes = 40000 end local read_all = function(file_name) return utility.open(file_name, "r", function(file) @@ -25,7 +27,7 @@ for _, path in ipairs{ ".", "extracted_texts", } do local path, name, extension = utility.split_path_components(file_name) local text = read_all(file_name) - print(math.floor(#text/maximum_bytes + 1), math.floor(#text/1000) .. "k", name) + print(math.floor(#text/config.llm_covers.maximum_bytes + 1), math.floor(#text/1000) .. "k", name) end) print("Press enter.") io.read("*line") diff --git a/cover_briefs/.gitkeep b/cover_briefs/.gitkeep deleted file mode 100644 index e69de29..0000000 diff --git a/ebook_to_txt.lua b/ebook_to_txt.lua deleted file mode 100755 index 5e58090..0000000 --- a/ebook_to_txt.lua +++ /dev/null @@ -1,25 +0,0 @@ -#!/usr/bin/env luajit - -package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path -local utility = require "utility" - -local calibre_location = "/Applications/calibre.app" -local config = utility.get_config("read-only") -if config and config.llm_covers and config.llm_covers.calibre_location then - calibre_location = config.llm_covers.calibre_location -end - -os.execute("mkdir -p raw_ebooks" .. utility.commands.silence_output) -os.execute("mkdir -p extracted_texts" .. utility.commands.silence_output) - -utility.ls("raw_ebooks", function(file_name) - local _, name, extension = utility.split_path_components(file_name) - if extension then - name = name:sub(1, -(#extension + 2)) - end - - local output_name = "extracted_texts/" .. name .. ".txt" - if utility.path_exists(output_name) then return end - - os.execute(calibre_location .. "/Contents/MacOS/ebook-convert " .. ("raw_ebooks/" .. file_name):enquote() .. " " .. output_name:enquote()) -end) diff --git a/extracted_texts/.gitkeep b/extracted_texts/.gitkeep deleted file mode 100644 index e69de29..0000000 diff --git a/for_gemini/.gitkeep b/for_gemini/.gitkeep deleted file mode 100644 index e69de29..0000000 diff --git a/intermediates/.gitkeep b/intermediates/.gitkeep deleted file mode 100644 index e69de29..0000000 diff --git a/lib/prompts.lua b/lib/prompts.lua new file mode 100644 index 0000000..718c874 --- /dev/null +++ b/lib/prompts.lua @@ -0,0 +1,70 @@ +local prompts = {} + +prompts.section_summarization = [[You are extracting information for a later cover design process. + +This is only one section of a larger story. + +Do NOT attempt to design a cover. +Do NOT decide what should appear on the cover. +Do NOT write a summary. + +Extract only information that may be useful when creating a cover after all story sections have been analyzed. + +Return: + +GENRE: +- up to 5 items + +SETTING: +- locations +- environments +- time period + +CHARACTERS: +- names +- physical descriptions +- distinctive visual traits + +CREATURES: +- notable creatures + +OBJECTS: +- important recurring items + +VISUAL MOTIFS: +- recurring imagery +- symbols +- repeated visual elements + +THEMES: +- major themes + +MOOD: +- emotional tone + +COLORS: +- colors strongly associated with scenes or imagery + +MEMORABLE VISUAL SCENES: +- 3-10 visually striking moments + +CONFIDENCE: +- how central each item appears to be + +]] + +prompts.combine_summaries = [[Take the common portions from the following text and produce a single simplified list of information: + +]] + +prompts.cover_brief = [[Generate a cover brief from the following: + +]] + +prompts.for_gemini = [[ + +Generate an ebook cover inset within an empty white border 10% larger than the cover using the following cover brief. It should be a flat graphic design file, with no artifacts of a physical object. The aspect ratio of an ebook is tall, not wide. The white border is very important, and should take up 10% of the area of the image. The title and author should only appear once on the cover. It must be sutiable for printing. Do not include the word "by" when adding the author's name to the cover. + +]] + +return prompts diff --git a/lib/timing.lua b/lib/timing.lua index 60ff108..96ece80 100644 --- a/lib/timing.lua +++ b/lib/timing.lua @@ -1,22 +1,38 @@ local timing = {} +local entries = {} -function timing.timer() - return setmetatable({ - entries = {}, - }, timing) +function timing.seconds_to_minutes(seconds) -- floored to tenths + return math.floor( seconds / 60 * 10 ) / 10 end -function timing:start() - -- if not self then self = {} end - self.entries[#self.entries + 1] = { start_time = os.time() } +function timing.start() + entries[#entries + 1] = { start_time = os.time() } end -function timing:stop() - local entry = self.entries[#self.entries] +function timing.stop() + local entry = entries[#entries] entry.stop_time = os.time() + entry.delta = entry.stop_time - entry.start_time - -- delta in minutes, floored to tenths - return math.floor( (entry.stop_time - entry.start_time) / 60 * 10 ) / 10 + return timing.seconds_to_minutes(entry.delta) +end + +function timing.print_report() + if #entries == 0 then return end + + local minimum, maximum, sum = math.huge, -math.huge, 0 + for _, data in pairs(entries) do + local delta = data.delta + if delta > maximum then maximum = delta end + if delta < minimum then minimum = delta end + sum = sum + delta + end + + print("Total: " .. timing.seconds_to_minutes(sum) .. " minutes.") + print("Averge: " .. timing.seconds_to_minutes(sum / #entries) + .. " minutes. Fastest: " .. timing.seconds_to_minutes(minimum) + .. " minutes. Slowest: " .. timing.seconds_to_minutes(maximum) + .. " minutes.") end return timing diff --git a/lib/utility.lua b/lib/utility.lua index 88e7e20..3eda5f8 100644 --- a/lib/utility.lua +++ b/lib/utility.lua @@ -32,7 +32,7 @@ else } end -utility.version = "1.4.0" +utility.version = "1.5.2" -- WARNING: This will return "./" if the original script is called locally instead of with an absolute path! if arg[0] ~= nil then utility.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) -- inspired by discussion in https://stackoverflow.com/q/6380820 @@ -124,14 +124,6 @@ standard_library_addition(string, "enquote", function(s) return "\"" .. s:gsub("\"", "\\\"") .. "\"" end) -standard_library_addition(string, "split", function(s, delimiter) - local result = {} - for item in s:gsplit(delimiter) do - result[#result + 1] = item - end - return result -end) - standard_library_addition(string, "gsplit", function(s, delimiter) local function escape_special_characters(s) local special_characters = "[()%%.[^$%]*+%-?]" @@ -144,6 +136,14 @@ standard_library_addition(string, "gsplit", function(s, delimiter) return s:gmatch("(.-)" .. escape_special_characters(delimiter)) end) +standard_library_addition(string, "split", function(s, delimiter) + local result = {} + for item in s:gsplit(delimiter) do + result[#result + 1] = item + end + return result +end) + -- modified from my fork of lume @@ -301,7 +301,7 @@ utility.release_lock = function(file_path, lock_uuid) local lock_file_path = file_path .. ".lock" if lock_uuid then utility.open(lock_file_path, "r", function(file) - if not file:read("*all") == lock_uuid then + if not (file:read("*all") == lock_uuid) then error("\n\n Lock UUID changed while lock was obtained. Data loss may have occurred. \n\n") end end) @@ -348,12 +348,47 @@ utility.save_config = function() end end +local data_file_locations = {} +utility.load_data = function(file_path) + local data = utility.open(file_path, "r", function(data_file) + local json = utility.require("dkjson") + return json.decode(data_file:read("*all")) + end) + data_file_locations[data] = file_path + return data +end + +utility.save_data = function(data, file_path) + local keys, loop = {} + loop = function(tab) + if type(tab) == "table" then + for k,v in pairs(tab) do + if not (type(k) == "number") then + keys[k] = true + end + loop(v) + end + end + end + loop(data) + local order = {} + for k in pairs(keys) do order[#order + 1] = k end + table.sort(order) + + file_path = file_path or data_file_locations[data] + assert(file_path, "The object must have been loaded by utility.load_data or you must pass a path as the second argument.") + utility.open(file_path, "w", function(data_file) + local json = utility.require("dkjson") + data_file:write(json.encode(data, { indent = true, keyorder = order, })) + data_file:write("\n") + end) +end + utility.deepcopy = function(tab) - local _type = type(tab) local copy - if _type == "table" then + if type(tab) == "table" then copy = {} for key, value in next, tab, nil do copy[utility.deepcopy(key)] = utility.deepcopy(value) diff --git a/make_cover_briefs.lua b/make_cover_briefs.lua index fa21bc0..b7c744b 100755 --- a/make_cover_briefs.lua +++ b/make_cover_briefs.lua @@ -2,87 +2,14 @@ package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path local utility = require "utility" +local prompts = require "prompts" +local timing = require "timing" -local default_model = "gemma4:12b-mlx" -local maximum_bytes = 40000 - -local partial_prompt = [[You are extracting information for a later cover design process. - -This is only one section of a larger story. - -Do NOT attempt to design a cover. -Do NOT decide what should appear on the cover. -Do NOT write a summary. - -Extract only information that may be useful when creating a cover after all story sections have been analyzed. - -Return: - -GENRE: -- up to 5 items - -SETTING: -- locations -- environments -- time period - -CHARACTERS: -- names -- physical descriptions -- distinctive visual traits - -CREATURES: -- notable creatures - -OBJECTS: -- important recurring items - -VISUAL MOTIFS: -- recurring imagery -- symbols -- repeated visual elements - -THEMES: -- major themes - -MOOD: -- emotional tone - -COLORS: -- colors strongly associated with scenes or imagery - -MEMORABLE VISUAL SCENES: -- 3-10 visually striking moments - -CONFIDENCE: -- how central each item appears to be - -]] - -local squish_partials_prompt = [[Take the common portions from the following text and produce a single simplified list of information: - -]] - -local cover_brief_prompt = [[Generate a cover brief from the following: - -]] - -local segmented_cover_brief_prompt = [[Generate a single cover brief. - -Prefer elements that: -- appear repeatedly across chunks -- have highest confidence -- best represent the entire story - -Avoid minor plot events. - -]] - -local gemini_prompt = [[ - -Generate an ebook cover inset within an empty white border 10% larger than the cover using the following cover brief. It should be a flat graphic design file, with no artifacts of a physical object. The aspect ratio of an ebook is tall, not wide. The white border is very important, and should take up 10% of the area of the image. The title and author should only appear once on the cover. It must be sutiable for printing. Do not include the word "by" when adding the author's name to the cover. - -]] +local config = utility.get_config("read-only") +if not config.llm_covers then config.llm_covers = {} end +if not config.llm_covers.calibre_location then config.llm_covers.calibre_location = "/Applications/calibre.app" end +if not config.llm_covers.default_model then config.llm_covers.default_model = "gemma4:12b-mlx" end +if not config.llm_covers.maximum_bytes then config.llm_covers.maximum_bytes = 40000 end local read_all = function(file_name) return utility.open(file_name, "r", function(file) @@ -96,19 +23,6 @@ local write_all = function(file_name, text) end) end -local timings = {} -local timing_report = function() - local minimum, maximum, sum = math.huge, 0, 0 - for _, delta in ipairs(timings) do - if delta > maximum then maximum = delta end - if delta < minimum then minimum = delta end - sum = sum + delta - end - print("") - print("Prompts took a total of " .. sum .. " minutes.") - print("Average: " .. math.floor(sum / #timings) .. " minutes. Fastest: " .. minimum .. " minutes. Slowest: " .. maximum .. " minutes.") -end - local strip_reasoning = function(text, reasoning_lines) if not reasoning_lines then reasoning_lines = {} end local tab = text:split("\n") @@ -136,22 +50,29 @@ local send_prompt = function(text, model) file:write(text) end) - local start_time = os.time() + timing.start() -- word wrap breaks the raw output badly, so I need to implement my own for terminal output somehow - local output = utility.capture_safe("cat " .. tmp_file_name:enquote() .. " | ollama run " .. (model or default_model) .. " --nowordwrap") - os.execute("ollama stop " .. (model or default_model)) -- NOTE this makes things slower, but more stable - local delta = math.floor( (os.time() - start_time) / 60 * 10 ) / 10 + local output = utility.capture_safe("cat " .. tmp_file_name:enquote() .. " | ollama run " .. (model or config.llm_covers.default_model) .. " --nowordwrap") + os.execute("ollama stop " .. (model or config.llm_covers.default_model)) -- NOTE this makes things slower, but more stable + local delta = timing.stop() + print("Took " .. delta .. " minutes.") - timings[#timings + 1] = delta os.execute("rm " .. tmp_file_name) if not output then error("ollama failed to generate output") end + output = output:sub(1, -2) -- strip extra newline from utility.capture_safe - output = strip_reasoning(output) - return output end +local make_directories = function() + os.execute("mkdir -p raw_ebooks" .. utility.commands.silence_output) + os.execute("mkdir -p extracted_texts" .. utility.commands.silence_output) + os.execute("mkdir -p intermediates" .. utility.commands.silence_output) + os.execute("mkdir -p cover_briefs" .. utility.commands.silence_output) + os.execute("mkdir -p for_gemini" .. utility.commands.silence_output) +end + local tree tree = function(path, fn) utility.list(path or ".", function(path_name) @@ -170,54 +91,218 @@ tree = function(path, fn) end) end -tree("extracted_texts", function(file_name) - local path, name, extension = utility.split_path_components(file_name) - if utility.path_exists("cover_briefs/" .. name) then return end +local find_ebook_files = function(books) + if not books then books = {} end - local text = read_all(file_name) - local result - - print(name) - local outputs = {} - if #text > maximum_bytes then - -- TODO check if this would work just as well starting from 1 instead.. - for i = 0, #text/maximum_bytes do - print("Processing section " .. (i + 1) .. " of " .. math.floor(#text/maximum_bytes + 1)) - local piece = text:sub(i * maximum_bytes, (i + 1) * maximum_bytes - 1) - result = send_prompt(partial_prompt .. piece) - outputs[#outputs + 1] = result - print(#result .. " characters added to intermediate context.") + tree("raw_ebooks", function(file_name) + local _, name, extension = utility.split_path_components(file_name) + if extension then + name = name:sub(1, -(#extension + 2)) end - text = table.concat(outputs, "\n\n") - write_all("intermediates/" .. name, text) + if not books[name] then + books[name] = {} + end + books[name].ebook_file = file_name + end) + + return books +end + +local find_extracted_texts = function(books) + if not books then books = {} end + + tree("extracted_texts", function(file_name) + local _, name, extension = utility.split_path_components(file_name) + if extension then + name = name:sub(1, -(#extension + 2)) + end + + if not books[name] then + books[name] = {} + end + books[name].extracted_file = file_name + books[name].extracted_file_size = utility.file_size(file_name) + end) + + return books +end + +local find_cover_briefs = function(books) + if not books then books = {} end + + tree("cover_briefs", function(file_name) + local _, name, extension = utility.split_path_components(file_name) + if extension then + name = name:sub(1, -(#extension + 2)) + end + + if not books[name] then + books[name] = {} + end + books[name].cover_brief_file = file_name + end) + + return books +end + +local get_books = function() + local books = {} + find_ebook_files(books) + find_extracted_texts(books) + find_cover_briefs(books) + return books +end + +local convert_ebooks = function(books) + for name, data in pairs(books) do + local function loop() + if data.extracted_file then return end + if not data.ebook_file then return end + + local extracted_file = "extracted_texts/" .. name .. ".txt" + os.execute(config.llm_covers.calibre_location .. "/Contents/MacOS/ebook-convert " + .. data.ebook_file:enquote() .. " " .. extracted_file:enquote()) + + if utility.path_exists(extracted_file) then + data.extracted_file = extracted_file + data.extracted_file_size = utility.file_size(extracted_file) + end + end + loop() + end +end + +local condense_summary = function(name, data, sections) + local condensed_sections = {} + + while #sections > 2 do -- this intentionally can skip the end of a book? + local sections_slice = {} + for i = 1, 10 do + if #sections >= 1 then + sections_slice[#sections_slice + 1] = table.remove(sections, 1) + end + end + + local intermediate_condensation_file = "intermediates/" .. name .. "-" .. (#condensed_sections + 1) .. "-condensed.txt" + if utility.path_exists(intermediate_condensation_file) then + condensed_sections[#condensed_sections + 1] = read_all(intermediate_condensation_file) + print("Loaded condensed summary segment. " .. #sections .. " sections remaining.") + else + print("Condensing summary. " .. #sections .. " sections remaining.") + local text = send_prompt(prompts.combine_summaries .. table.concat(sections_slice, "\n\n")) + condensed_sections[#condensed_sections + 1] = text + print(#text .. " characters added to final context.") + write_all(intermediate_condensation_file, text) + end end - -- handle too much context (works with up to 100 slices) - if #outputs > 10 then - local context_slices = {} - while #outputs > 2 do - local output_slices = {} - for i = 1, 10 do - if #outputs >= 1 then - output_slices[#output_slices + 1] = table.remove(outputs, 1) - end + text = table.concat(condensed_sections, "\n\n") + local condensed_intermediate_file = "intermediates/" .. name .. " CONDENSED.txt" + write_all(condensed_intermediate_file, text) + data.condensed_intermediate_file = condensed_intermediate_file + + -- if deleted any sooner, could break resuming + for i = 1, #condensed_sections do + local intermediate_condensation_file = "intermediates/" .. name .. "-" .. i .. "-condensed.txt" + os.execute("rm " .. intermediate_condensation_file:enquote()) + end + + return text +end + +local generate_summary = function(name, data) + local text = read_all(data.extracted_file) + local sections = {} + + for i = 0, data.extracted_file_size/config.llm_covers.maximum_bytes do + local intermediate_section_file = "intermediates/" .. name .. "-" .. i .. ".txt" + if utility.path_exists(intermediate_section_file) then + sections[#sections + 1] = read_all(intermediate_section_file) + print("Loaded section " .. (i + 1) .. ".") + else + print("Processing section " .. (i + 1) .. " of " .. math.floor(data.extracted_file_size/config.llm_covers.maximum_bytes + 1)) + local section = text:sub(i * config.llm_covers.maximum_bytes, (i + 1) * config.llm_covers.maximum_bytes - 1) + section = send_prompt(prompts.section_summarization .. section) + sections[#sections + 1] = section + print(#section .. " characters added to intermediate context.") + write_all(intermediate_section_file, section) + end + end + + text = table.concat(sections, "\n\n") + local intermediate_file = "intermediates/" .. name .. ".txt" + write_all(intermediate_file, text) + data.intermediate_file = intermediate_file + + if #sections > 10 then + text = condense_summary(name, data, sections) + end + + -- if these temporary files are removed before condense_summary is called, + -- sections cannot be individually loaded, which breaks condense_summary + for i = 0, data.extracted_file_size/config.llm_covers.maximum_bytes do + local intermediate_section_file = "intermediates/" .. name .. "-" .. i .. ".txt" + os.execute("rm " .. intermediate_section_file:enquote()) + end + + return text +end + +local generate_cover_briefs = function(books) + for name, data in pairs(books) do + local function loop() + if data.cover_brief_file then return end + print(name) + + local text + if data.extracted_file_size > config.llm_covers.maximum_bytes then + text = generate_summary(name, data) + else + text = read_all(data.extracted_file) end - print("Squishing context. " .. #outputs .. " samples remaining.") - result = send_prompt(squish_partials_prompt .. table.concat(output_slices, "\n\n")) - context_slices[#context_slices + 1] = result - print(#result .. " characters added to final context.") - end + print("Generating cover brief.") + text = send_prompt(prompts.cover_brief .. text) - text = table.concat(context_slices, "\n\n") - write_all("intermediates/" .. name:sub(1, -5) .. " CONDENSED.txt", text) + local cover_brief_file = "cover_briefs/" .. name .. ".txt" + write_all(cover_brief_file, text) + write_all("for_gemini/" .. name .. ".txt", "Title: " .. name .. prompts.for_gemini .. text) + data.cover_brief_file = cover_brief_file + end + loop() + end +end + +local estimate_time_required = function(books) + local sum, count = 0, 0 + for name, data in pairs(books) do + local function loop() + if data.cover_brief_file then return end + sum = sum + 60 * (1.5 + data.extracted_file_size/(config.llm_covers.maximum_bytes/6)) + count = count + 1 + end + loop() end - print("Generating cover brief.") - result = send_prompt(cover_brief_prompt .. text) - write_all("cover_briefs/" .. name, result) - write_all("for_gemini/" .. name, "Title: " .. name:sub(1, -5) .. gemini_prompt .. result) -end) + print(count .. " cover briefs to generate. Estimated time: " + .. timing.seconds_to_minutes(sum) .. " minutes.") +end -timing_report() +make_directories() + +local books = get_books() +convert_ebooks(books) + +estimate_time_required(books) +print("") + +generate_cover_briefs(books) + +print("") +timing.print_report() + +-- utility.print_table(books) + +-- ebook_file, extracted_file, extracted_file_size, +-- intermediate_file, condensed_intermediate_file, cover_brief_file diff --git a/rewrite.lua b/rewrite.lua deleted file mode 100644 index 3d953bc..0000000 --- a/rewrite.lua +++ /dev/null @@ -1,254 +0,0 @@ -#!/usr/bin/env luajit - -package.path = (arg[0]:match("@?(.*/)") or arg[0]:match("@?(.*\\)")) .. "lib" .. package.config:sub(1, 1) .. "?.lua;" .. package.path -local utility = require "utility" --- TODO import timing library - -local config = utility.get_config("read-only") -if not config.llm_covers then config.llm_covers = {} end -if not config.llm_covers.calibre_location then config.llm_covers.calibre_location = "/Applications/calibre.app" end -if not config.llm_covers.default_model then config.llm_covers.default_model = "gemma4:12b-mlx" end -if not config.llm_covers.maximum_bytes then config.llm_covers.maximum_bytes = 40000 end - --- TODO import prompts, use new names from fork - -local strip_reasoning = function(text, reasoning_lines) - if not reasoning_lines then reasoning_lines = {} end - local tab = text:split("\n") - table.remove(tab, 1) -- remove "Thinking..." - - while true do - local done = tab[1] == "...done thinking." - local line = table.remove(tab, 1) - if done then - table.remove(tab, 1) -- remove newline after end of thinking - return table.concat(tab, "\n") - elseif #tab < 1 then - return text -- no reasoning output - else - reasoning_lines[#reasoning_lines + 1] = line -- export thinking lines - end - end -end - --- TODO import and rewrite send_prompt to use new timing library - -local read_all = function(file_name) - return utility.open(file_name, "r", function(file) - return file:read("*all") - end) -end -local write_all = function(file_name, text) - return utility.open(file_name, "w", function(file) - file:write(text) - file:write("\n") - end) -end - -local make_directories = function() - os.execute("mkdir -p raw_ebooks" .. utility.commands.silence_output) - os.execute("mkdir -p extracted_texts" .. utility.commands.silence_output) - os.execute("mkdir -p intermediates" .. utility.commands.silence_output) - os.execute("mkdir -p cover_briefs" .. utility.commands.silence_output) - os.execute("mkdir -p for_gemini" .. utility.commands.silence_output) -end - -local tree -tree = function(path, fn) - utility.list(path or ".", function(path_name) - local blacklist = { - [".git"] = true, - [".gitkeep"] = true, - [".gitignore"] = true, - [".DS_Store"] = true, - } - if blacklist[path_name] then return end - if utility.is_file(path_name) then - fn(path_name) - else - tree(path .. utility.path_separator .. path_name, fn) - end - end) -end - -local find_ebook_files = function(books) - if not books then books = {} end - - tree("raw_ebooks", function(file_name) - local _, name, extension = utility.split_path_components(file_name) - if extension then - name = name:sub(1, -(#extension + 2)) - end - - if not books[name] then - books[name] = {} - end - books[name].ebook_file = file_name - end) - - return books -end - -local find_extracted_texts = function(books) - if not books then books = {} end - - tree("extracted_texts", function(file_name) - local _, name, extension = utility.split_path_components(file_name) - if extension then - name = name:sub(1, -(#extension + 2)) - end - - if not books[name] then - books[name] = {} - end - books[name].extracted_file = file_name - books[name].extracted_file_size = utility.file_size(file_name) - end) - - return books -end - -local find_cover_briefs = function(books) - if not books then books = {} end - - tree("cover_briefs", function(file_name) - local _, name, extension = utility.split_path_components(file_name) - if extension then - name = name:sub(1, -(#extension + 2)) - end - - if not books[name] then - books[name] = {} - end - books[name].cover_brief_file = file_name - end) - - return books -end - -local get_books = function() - local books = {} - find_ebook_files(books) - find_extracted_texts(books) - find_cover_briefs(books) - return books -end - -local convert_ebooks = function(books) - for name, data in pairs(books) do - if data.extracted_file then return end - if not data.ebook_file then return end - - local extracted_file = "extracted_texts/" .. name .. ".txt" - os.execute(config.llm_covers.calibre_location .. "/Contents/MacOS/ebook-convert " - .. data.ebook_file:enquote() .. " " .. extracted_file:enquote()) - - if utility.path_exists(extracted_file) then - data.extracted_file = extracted_file - end - end -end - -local condense_summary = function(name, data, sections) - local condensed_sections = {} - - while #sections > 2 do -- this intentionally can skip the end of a book? - local sections_slice = {} - for i = 1, 10 do - if #sections >= 1 then - sections_slice[#sections_slice + 1] = table.remove(sections, 1) - end - end - - local intermediate_condensation_file = "intermediates/" .. name .. "-" .. (#condensed_sections + 1) .. "-condensed.txt" - if utility.path_exists(intermediate_condensation_file) then - condensed_sections[#condensed_sections + 1] = read_all(intermediate_condensation_file) - print("Loaded condensed summary segment. " .. #sections .. " sections remaining.") - else - print("Condensing summary. " .. #sections .. " sections remaining.") - local text = send_prompt(squish_partials_prompt .. table.concat(sections_slice, "\n\n")) - condensed_sections[#condensed_sections + 1] = text - print(#text .. " characters added to final context.") - write_all(intermediate_condensation_file, text) - end - end - - text = table.concat(condensed_sections, "\n\n") - local condensed_intermediate_file = "intermediates/" .. name .. " CONDENSED.txt" - write_all(condensed_intermediate_file, text) - data.condensed_intermediate_file = condensed_intermediate_file - - -- if deleted any sooner, could break resuming - for i = 1, #condensed_sections do - local intermediate_condensation_file = "intermediates/" .. name .. "-" .. i .. "-condensed.txt" - os.execute("rm " .. intermediate_condensation_file:enquote()) - end - - return text -end - -local generate_summary = function(name, data) - local text = read_all(data.extracted_file) - local sections = {} - - for i = 0, data.extracted_file_size/config.llm_covers.maximum_bytes do - local intermediate_section_file = "intermediates/" .. name .. "-" .. i .. ".txt" - if utility.path_exists(intermediate_section_file) then - sections[#sections + 1] = read_all(intermediate_section_file) - print("Loaded section " .. (i + 1) .. ".") - else - print("Processing section " .. (i + 1) .. " of " .. math.floor(data.extracted_file_size/config.llm_covers.maximum_bytes + 1)) - local section = text:sub(i * config.llm_covers.maximum_bytes, (i + 1) * config.llm_covers.maximum_bytes - 1) - section = send_prompt(partial_prompt .. section) - sections[#sections + 1] = section - print(#section .. " characters added to intermediate context.") - write_all(intermediate_section_file, section) - end - end - - text = table.concat(sections, "\n\n") - local intermediate_file = "intermediates/" .. name .. ".txt" - write_all(intermediate_file, text) - data.intermediate_file = intermediate_file - - if #sections > 10 then - text = condense_summary(name, data, sections) - end - - -- if these temporary files are removed before condense_summary is called, - -- sections cannot be individually loaded, which breaks condense_summary - for i = 0, data.extracted_file_size/config.llm_covers.maximum_bytes do - local intermediate_section_file = "intermediates/" .. name .. "-" .. i .. ".txt" - os.execute("rm " .. intermediate_section_file:enquote()) - end - - return text -end - -local generate_cover_briefs = function(books) - for name, data in pairs(books) do - if data.cover_brief_file then return end - print(name) - - local text - if data.extracted_file_size > config.llm_covers.maximum_bytes then - text = generate_summary(name, data) - else - text = read_all(extracted_file) - end - - print("Generating cover brief.") - text = send_prompt(cover_brief_prompt .. text) - - local cover_brief_file = "cover_briefs/" .. name .. ".txt" - write_all(cover_brief_file, text) - write_all("for_gemini/" .. name .. ".txt", "Title: " .. name .. gemini_prompt .. text) - data.cover_brief_file = cover_brief_file - end -end - -make_directories() -generate_cover_briefs(get_books()) - --- ebook_file, extracted_file, extracted_file_size, --- intermediate_file, condensed_intermediate_file, cover_brief_file