diff options
| -rw-r--r-- | README.md | 4 | ||||
| -rw-r--r-- | docs/superpowers/specs/2026-10-04-fanfictioner-design.md | 11 | ||||
| -rwxr-xr-x | fanfictioner | 29 |
3 files changed, 35 insertions, 9 deletions
@@ -26,7 +26,9 @@ Plan review keys: `c` confirm, `e` edit in `$EDITOR`, `g` ask Gemma to revise, `q` quit. Image review keys: `1`-`3` pick, `r` regenerate, `p` edit prompt, `s` skip edit (edit stage only), `q` quit. -A page can override the base model with a `Model: krea2` line in plan.md. To +Page 1 is the cover: its `Title:` line in plan.md is drawn as title lettering, +the only text in the book. Remove the line for a cover without text. A page can +override the base model with a `Model: krea2` line in plan.md. To redo a page, delete `NNN.png` from the library and its `cand/pNNN_*` folder. To redo a reference sheet, delete `refs/<Name>.png`; pages already done keep the old look. A prompt edited with `p` applies until you pick and is not diff --git a/docs/superpowers/specs/2026-10-04-fanfictioner-design.md b/docs/superpowers/specs/2026-10-04-fanfictioner-design.md index 6d054e6..9a34ef6 100644 --- a/docs/superpowers/specs/2026-10-04-fanfictioner-design.md +++ b/docs/superpowers/specs/2026-10-04-fanfictioner-design.md @@ -19,7 +19,8 @@ reviews and picks among candidates at every image stage. - llama-server runs in router mode on `localhost:8181`, model `Gemma4-12B-qat-mtp`, `--models-max 1`. Unload with `POST /models/unload {"model": ...}`; it reloads on the next request. -- No text in images: no speech bubbles, captions or labels. +- No text in images: no speech bubbles, captions or labels. Only exception: a + page with a `Title:` line (the cover) gets that title as lettering. ## Pipeline @@ -79,6 +80,7 @@ Characters: <Name>, <Name> (or "none") Pose: <Name>: <position in frame, orientation, pose>; <Name>: <...> Framing: <camera angle, shot size>, portrait|landscape Model: krea2 (optional, overrides the base model for this page) +Title: <book title> (cover only: lettering drawn on this page) ``` Validation before confirm: title, Series, Book, at least one character and @@ -95,7 +97,8 @@ All via `POST localhost:8181/v1/chat/completions`, `model: Gemma4-12B-qat-mtp`. 1. **Draft**: system prompt holds the plan.md format and rules (no text in images, fixed character descriptions starting with who they are, characters are people only while animals and objects belong to the scene, one page = one - image, a scene may span several pages). User message: `story.md`. Output: plan.md. + image, a scene may span several pages, page 1 is the cover with a `Title:` + line). User message: `story.md`. Output: plan.md. 2. **Revise**: current plan.md + user's instruction -> complete new plan.md. A revision that fails validation is saved as `plan.rejected.md` and plan.md is kept; an accepted one keeps the previous version as `plan.md.bak`. @@ -124,7 +127,9 @@ Templates: light-grey background, even studio lighting. <Style>. No text, no labels." - `base_prompt`: Style, page text, then for each character present its description followed by its pose, then Framing, then "No text, no speech - bubbles." + bubbles." A page with `Title:` instead ends with "Title lettering at the top + of the image reading exactly "<title>". No other text, no speech bubbles.", + and its edits also keep the title lettering unchanged. - `edits`: one entry per group of at most 2 characters (Qwen takes at most 3 reference images: the scene plus 2 refs), in `Characters:` order. Per character: "In image 1, change only <Name>'s head: give her/him/them the face, diff --git a/fanfictioner b/fanfictioner index 24d3a41..e1f5af4 100755 --- a/fanfictioner +++ b/fanfictioner @@ -66,7 +66,9 @@ MIN_RAM_KB = 2 * 1024 * 1024 # kill sd-cli below 2 GB MemAvailable, before the DRAFT_SYS = """You turn a story idea into a page plan for a text-free illustrated book. Each page is one full-page image. A scene may span several pages. Images never contain text: no speech bubbles, captions, signs or labels. Tell the story -through action, expression and setting. +through action, expression and setting. The only exception is the cover. +Page 1 is the book cover: the main characters in a striking composition, portrait, with a +Title: line holding the book title. No other page has a Title: line. Give every character one fixed visual description that starts with who they are (e.g. "Young woman", "Old man", "Little girl"), then age, hair, face, eyes, build, outfit, and never vary it between pages. @@ -93,6 +95,7 @@ Pronoun: <she, he or they> Characters: <Name>, <Name> (or: none) Pose: <Name>: <position in the frame, body orientation and pose>; <Name>: <...> Framing: <camera angle, shot size>, <portrait or landscape> +Title: <book title, cover page only> """ OBJ = {"she": "her", "he": "him", "they": "them"} # pronoun -> object form for edit prompts @@ -145,7 +148,8 @@ def parse_plan(text): elif section == "pages": m = re.match(r"(\d+)\.\s*(.*)", head) cur = {"n": int(m[1]) if m else 0, "title": m[2] if m else head, "text": [], - "heading": "" if m else head, "characters": None, "pose": "", "framing": "", "model": ""} + "heading": "" if m else head, "characters": None, "pose": "", "framing": "", "model": "", + "lettering": ""} plan["pages"].append(cur) elif section is None and (m := re.match(r"[-*\s]*\**(Series|Book|Style)\**:\**\s*(.*)", s)): plan[m[1].lower()] = m[2].strip().strip("*").strip() @@ -153,8 +157,10 @@ def parse_plan(text): (m := re.match(r"[-*\s]*\**Pronoun\**:\**\s*(.*)", s)): plan["pronouns"][cname] = m[1].strip().strip("*").strip().lower() elif section == "pages" and cur is not None and \ - (m := re.match(r"[-*\s]*\**(Characters|Pose|Framing|Model)\**:\**\s*(.*)", s)): + (m := re.match(r"[-*\s]*\**(Characters|Pose|Framing|Model|Title)\**:\**\s*(.*)", s)): key, val = m[1].lower(), m[2].strip().strip("*").strip() + if key == "title": # the page heading already uses "title" + key, val = "lettering", val.strip('"\u201c\u201d').strip() if key == "characters": val = [] if val.lower() == "none" else [c.strip() for c in val.split(",") if c.strip()] elif key == "model": @@ -266,12 +272,18 @@ def build_json(plan): poses = parse_poses(p["pose"]) parts = [style, sentence(p["text"])] parts += [f"{sentence(plan['characters'][n])} {sentence(poses[n])}" for n in p["characters"]] - parts += [sentence(p["framing"]), "No text, no speech bubbles."] + parts.append(sentence(p["framing"])) + if p["lettering"]: + parts.append(f'Title lettering at the top of the image reading exactly "{p["lettering"]}". ' + "No other text, no speech bubbles.") + else: + parts.append("No text, no speech bubbles.") + keep = "background, lighting" + (", title lettering" if p["lettering"] else "") edits = [{"characters": g, "prompt": " ".join( f"In image 1, change only {n}'s head: give {OBJ[plan['pronouns'][n]]} the face, eyes and hair " f"of the person in image {i + 2}. Keep {n}'s exact pose from image 1: {poses[n]}." for i, n in enumerate(g)) - + " Keep bodies, clothing, other people, background, lighting and art style of image 1 unchanged."} + + f" Keep bodies, clothing, other people, {keep} and art style of image 1 unchanged."} for g in edit_groups(p["characters"])] pages.append({"n": p["n"], "orientation": "landscape" if "landscape" in p["framing"].lower() else "portrait", "model": p["model"], "base_prompt": " ".join(x for x in parts if x), "edits": edits}) @@ -598,6 +610,13 @@ def selftest(): "In image 1, change only Jon's head: give him the face, eyes and hair of the person in image 3. " "Keep Jon's exact pose from image 1: center, facing camera. " "Keep bodies, clothing, other people, background, lighting and art style of image 1 unchanged.") + cover = build_json(parse_plan(SAMPLE_PLAN.replace("Framing: wide shot, eye level, landscape", + 'Framing: wide shot, eye level, landscape\nTitle: "Rooftop"'))) + c1, c2 = cover["pages"] + assert c1["base_prompt"].endswith('landscape. Title lettering at the top of the image reading exactly ' + '"Rooftop". No other text, no speech bubbles.'), c1["base_prompt"] + assert c1["edits"][1]["prompt"].endswith("background, lighting, title lettering and art style of image 1 unchanged.") + assert c2["base_prompt"].endswith("No text, no speech bubbles.") and c1["n"] == 1 assert "Kit's head: give her the face, eyes and hair of the person in image 2. " \ "Keep Kit's exact pose from image 1: right, sitting." in j1["edits"][1]["prompt"] q = sd_args("qwen_edit", "edit it", Path("c/7_%d.png"), (832, 1216), 7, |
