diff --git a/ReadMe.md b/ReadMe.md index 221330c..ce5c611 100644 --- a/ReadMe.md +++ b/ReadMe.md @@ -5,31 +5,41 @@ A simple tool for using local LLMs to generate cover briefs for ebooks. - A Mac mini M4 base model (original version) or better M-series computer - Ollama and `gemma4:12b-mlx` - LuaJIT +- calibre ## Usage -1. Use a tool like [audiobook-creator](https://github.com/prakharsr/audiobook-creator) to extract the text of an audiobook. -2. Place it in `extracted_texts/TITLE by AUTHOR.txt`. (The format of this doesn't matter except that the file name sans extension will be used in the output.) -3. Run `make_cover_briefs.lua`. -4. Cover briefs will be placed in `cover_briefs`. -5. A prompt ready to paste directly into Gemini will be in `for_gemini`. An extra border is prompted for to make cropping easy to remove the Gemini watermark. (There are other small changes to increase likelihood of Gemini following the prompt correctly and making a good output.) +1. Create `config.json` if necessary (see format below). +2. Run `make_cover_briefs.lua` once to create necessary directories. +3. Put ebooks in `raw_ebooks`. +4. Run `make_cover_briefs.lua` + +Cover briefs will be placed in `cover_briefs`. A prompt ready to paste directly +into Gemini will be in `for_gemini`. An extra border is prompted for to make +cropping easy to remove the Gemini watermark. (There are other small changes to +increase likelihood of Gemini following the prompt correctly and making a good +output.) ### Gemini Issues -None of this works unless you **disable Gemini's personalization/memory**, because it pollutes context -extremely badly and will randomly add elements and rejections from different conversations everywhere. +None of this works unless you **disable Gemini's personalization/memory**, +because it pollutes context extremely badly and will randomly add elements and +rejections from different conversations everywhere. -If Gemini rejects the prompt, adding Remove any elements that may go against guidelines before generating the image. +If Gemini rejects the prompt, adding +Remove any elements that may go against guidelines before generating the image. to the opening paragraph can fix that issue. -- This suddenly got a lot less effective. I recommend instead telling it to rewrite the prompt and then - in a separate conversation, use that instead. I also recommend one retry before modifying it. +- This suddenly got a lot less effective. I recommend instead telling it to + rewrite the prompt and then in a separate conversation, use that instead. I + also recommend one retry before modifying it. -Gemini is really bad about adding hardcover seams. Despite being instructed not to, it shows up sometimes. -More strong prompting breaks other requested features, so the best way to deal with it is to request its -removal after the initial image is generated. +Gemini is really bad about adding hardcover seams. Despite being instructed not +to, it shows up sometimes. More strong prompting breaks other requested +features, so the best way to deal with it is to request its removal after the +initial image is generated. -The local model sometimes completely forgets critical details of characters, -so it's important you have some familiarity with the work before running the generator -to check accuracy. +The local model sometimes completely forgets critical details of characters, so +it's important you have some familiarity with the work before running the +generator to check accuracy. ## Tasks - [ ] Add timestamps to when a prompt is sent. @@ -37,9 +47,9 @@ to check accuracy. - [ ] option to delete input files or intermediate files at end of each operation step - [x] instead of placeholder directories, `mkdir -p` should be used to only make them when necessary - [x] make script resumable instead of having to restart per book -- [ ] dry run to estimate total time before running full script, print starting ETA +- [x] dry run to estimate total time before running full script, print starting ETA - [x] Build a list of files to work on and current states before doing anything by looping over source directories - [x] Combine everything into one script to manage things -- [ ] Allow specifiying a filter on what to process +- [ ] Allow specifying a filter on what to process - [ ] Allow changing the maximum bytes per prompt dynamically (run smaller while I'm doing other things, run full while I'm asleep) - [ ] specify config format diff --git a/lib/timing.lua b/lib/timing.lua index ad4139c..e770603 100644 --- a/lib/timing.lua +++ b/lib/timing.lua @@ -1,34 +1,36 @@ -local timing = { - entries = {} -} +local timing = {} +local entries = {} -local function to_minutes(t) -- floored to tenths - return math.floor( t / 60 * 10 ) / 10 +function timing.seconds_to_minutes(seconds) -- floored to tenths + return math.floor( seconds / 60 * 10 ) / 10 end -function timing:start() - -- if not self then self = {} end - self.entries[#self.entries + 1] = { start_time = os.time() } +function timing.start() + entries[#entries + 1] = { start_time = os.time() } end -function timing:stop() - local entry = self.entries[#self.entries] +function timing.stop() + local entry = entries[#entries] entry.stop_time = os.time() + entry.delta = entry.stop_time - entry.start_time - return to_minutes(entry.stop_time - entry.start_time) -- return delta + return timing.seconds_to_minutes(entry.delta) end -function timing:print_report() +function timing.print_report() local minimum, maximum, sum = math.huge, -math.huge, 0 - for _, data in pairs(self.entries) do - local delta = data.stop_time - data.start_time + for _, data in pairs(entries) do + local delta = data.delta if delta > maximum then maximum = delta end if delta < minimum then minimum = delta end sum = sum + delta end - print("Total: " .. to_minutes(sum) .. " minutes.") - print("Averge: " .. to_minutes(sum / #self.entries) .. " minutes. Fastest: " - .. to_minutes(minimum) .. " minutes. Slowest: " .. to_minutes(maximum) .. " minutes.") + + print("Total: " .. timing.seconds_to_minutes(sum) .. " minutes.") + print("Averge: " .. timing.seconds_to_minutes(sum / #entries) + .. " minutes. Fastest: " .. timing.seconds_to_minutes(minimum) + .. " minutes. Slowest: " .. timing.seconds_to_minutes(maximum) + .. " minutes.") end return timing diff --git a/make_cover_briefs.lua b/make_cover_briefs.lua index 678cc1f..1b4279b 100755 --- a/make_cover_briefs.lua +++ b/make_cover_briefs.lua @@ -50,11 +50,11 @@ local send_prompt = function(text, model) file:write(text) end) - timing:start() + timing.start() -- word wrap breaks the raw output badly, so I need to implement my own for terminal output somehow local output = utility.capture_safe("cat " .. tmp_file_name:enquote() .. " | ollama run " .. (model or default_model) .. " --nowordwrap") os.execute("ollama stop " .. (model or default_model)) -- NOTE this makes things slower, but more stable - local delta = timing:stop() + local delta = timing.stop() print("Took " .. delta .. " minutes.") os.execute("rm " .. tmp_file_name) @@ -267,14 +267,30 @@ local generate_cover_briefs = function(books) end end +local estimate_time_required = function(books) + local sum, count = 0, 0 + for name, data in pairs(books) do + if data.cover_brief_file then return end + sum = sum + 60 * (2 + data.extracted_file_size/(config.llm_covers.maximum_bytes/7.5)) + count = count + 1 + end + + print(count .. " cover briefs to generate. Estimated time: " + .. timing.seconds_to_minutes(sum) .. " minutes.") +end + make_directories() local books = get_books() convert_ebooks(books) + +estimate_time_required(books) +print("") + generate_cover_briefs(books) print("") -timing:print_report() +timing.print_report() -- ebook_file, extracted_file, extracted_file_size, -- intermediate_file, condensed_intermediate_file, cover_brief_file