aboutsummaryrefslogtreecommitdiffstats
diff options
context:
space:
mode:
authorDanilo M. <danix@danix.xyz>2026-10-05 09:45:30 +0200
committerDanilo M. <danix@danix.xyz>2026-10-05 09:45:30 +0200
commit7b73710c6b963e3e4c797a44f7cb296d127d42ba (patch)
treeeee009f004921c8268221af408fc0d88745d4dc4
parent12d55dc34e638ca89a40e1f65f9713a297f1a6a2 (diff)
downloadfanfictioner-7b73710c6b963e3e4c797a44f7cb296d127d42ba.tar.gz
fanfictioner-7b73710c6b963e3e4c797a44f7cb296d127d42ba.zip
Draft, review and compile the plan with Gemma
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
-rwxr-xr-xfanfictioner161
1 files changed, 157 insertions, 4 deletions
diff --git a/fanfictioner b/fanfictioner
index 991f079..34ae021 100755
--- a/fanfictioner
+++ b/fanfictioner
@@ -61,6 +61,77 @@ SIZES = {"portrait": (832, 1216), "landscape": (1216, 832), "turnaround": (1216,
SHRINK = {"portrait": "576x832", "landscape": "832x576", "ref": "560x384"}
MIN_RAM_KB = 2 * 1024 * 1024 # kill sd-cli below 2 GB MemAvailable, before the OOM killer hits the desktop
+DRAFT_SYS = """You turn a story idea into a page plan for a text-free illustrated book.
+Each page is one full-page image. A scene may span several pages.
+Images never contain text: no speech bubbles, captions, signs or labels. Tell the story
+through action, expression and setting.
+Give every character one fixed visual description (age, hair, face, eyes, build, outfit)
+and never vary it between pages.
+If a page has a Model: line, keep it unchanged.
+Write the plan in exactly this Markdown format, with nothing before or after it:
+
+# <Title>
+Series: <series folder name>
+Book: <book folder name>
+Style: <one line describing the art style of every page>
+
+## Characters
+
+### <Name>
+<fixed visual description>
+
+## Pages
+
+### 1. <short title>
+<what happens, setting, mood>
+Characters: <Name>, <Name> (or: none)
+Pose: <body orientation and pose of each named character>
+Framing: <camera angle, shot size, portrait or landscape>
+"""
+
+COMPILE_SYS = """You convert an illustrated-book plan into image-generation prompts, as JSON.
+
+characters: one entry per character of the plan, in plan order, name exactly as written.
+turnaround_prompt: "Character turnaround reference sheet. The same <full visual description>
+shown three times side by side, full body: front view, side view, back view. Neutral standing
+pose, arms relaxed. Plain light-grey background, even studio lighting. <Style line>.
+No text, no labels."
+
+pages: one entry per page, n as in the plan. orientation: portrait or landscape, from Framing.
+base_prompt: the Style line, then the scene, then the full visual description of every
+character present, then their Pose, then the Framing, ending with
+"No text, no speech bubbles."
+
+edits: exactly the edit groups listed after the plan, in that order, names exactly as written.
+In an edit, image 1 is the scene, image 2 is the reference sheet of the first listed character,
+image 3 that of the second. For each character the prompt says:
+"In image 1, change only <name>'s head: give her/him the face, eyes and hair of the person in
+image <2 or 3>. Keep <name>'s exact pose from image 1: <that character's Pose>."
+and it ends with: "Keep bodies, clothing, other people, background, lighting and art style of
+image 1 unchanged."
+A page with no characters has an empty edits list.
+"""
+
+SCHEMA = {
+ "type": "object", "required": ["characters", "pages"],
+ "properties": {
+ "characters": {"type": "array", "items": {
+ "type": "object", "required": ["name", "turnaround_prompt"],
+ "properties": {"name": {"type": "string"}, "turnaround_prompt": {"type": "string"}}}},
+ "pages": {"type": "array", "items": {
+ "type": "object", "required": ["n", "orientation", "base_prompt", "edits"],
+ "properties": {
+ "n": {"type": "integer"},
+ "orientation": {"enum": ["portrait", "landscape"]},
+ "base_prompt": {"type": "string"},
+ "edits": {"type": "array", "items": {
+ "type": "object", "required": ["characters", "prompt"],
+ "properties": {"characters": {"type": "array", "items": {"type": "string"},
+ "maxItems": 2},
+ "prompt": {"type": "string"}}}}}}},
+ },
+}
+
def load_conf(path=CONF):
conf = {}
@@ -265,10 +336,6 @@ def editor(path):
subprocess.run([*shlex.split(os.environ.get("EDITOR") or "vi"), str(path)])
-def unload_llm():
- pass
-
-
def generate(d, profile, prompt, size, refs=()):
"""One sd-cli call producing 3 candidates in d. Returns (files, seed, error)."""
unload_llm()
@@ -330,6 +397,90 @@ def review(d, profile, prompt, size, refs=(), skip=False):
sys.exit("quit, run again to resume")
+def llm(path, body=None):
+ data = json.dumps(body).encode() if body is not None else None
+ req = urllib.request.Request(LLM_URL + path, data=data, headers={"Content-Type": "application/json"})
+ with urllib.request.urlopen(req, timeout=900) as r:
+ return json.load(r)
+
+
+def check_llm():
+ try:
+ ids = [m["id"] for m in llm("/v1/models")["data"]]
+ except OSError as e:
+ sys.exit(f"llama-server unreachable at {LLM_URL}: {e}")
+ if LLM_MODEL not in ids:
+ sys.exit(f"llama-server does not offer {LLM_MODEL} (has: {', '.join(ids)})")
+
+
+def unload_llm():
+ """Free VRAM for sd-cli. Idempotent: an unloaded or unreachable server is fine."""
+ try:
+ llm("/models/unload", {"model": LLM_MODEL})
+ except OSError:
+ pass
+
+
+def chat(system, user, schema=None):
+ body = {"model": LLM_MODEL,
+ "messages": [{"role": "system", "content": system}, {"role": "user", "content": user}]}
+ if schema:
+ body["response_format"] = {"type": "json_schema", "json_schema": {"name": "plan", "schema": schema}}
+ print("asking Gemma ...", flush=True)
+ return llm("/v1/chat/completions", body)["choices"][0]["message"]["content"]
+
+
+def strip_fences(s):
+ return re.sub(r"^```\w*\n|\n```$", "", s.strip())
+
+
+def compile_plan(md, plan):
+ """plan.md -> plan.json dict, or None when Gemma twice returns groups that do not match."""
+ for _ in range(2):
+ pj = json.loads(chat(COMPILE_SYS, f"{md}\n\nEdit groups, in order:\n{edit_hint(plan)}", SCHEMA))
+ errs = check_json(pj, plan)
+ if not errs:
+ break
+ print("compile mismatch:", *errs, sep="\n ")
+ else:
+ return None
+ models = {p["n"]: p["model"] for p in plan["pages"]}
+ pj.update(title=plan["title"], series=plan["series"], book=plan["book"])
+ for p in pj["pages"]:
+ p["model"] = models[p["n"]]
+ return pj
+
+
+def plan_stage(story, wd):
+ md_path, js_path = wd / "plan.md", wd / "plan.json"
+ if js_path.exists() and md_path.exists() and js_path.stat().st_mtime >= md_path.stat().st_mtime:
+ return json.loads(js_path.read_text())
+ check_llm()
+ wd.mkdir(exist_ok=True)
+ if not md_path.exists():
+ md_path.write_text(strip_fences(chat(DRAFT_SYS, story.read_text())) + "\n")
+ while True:
+ md = md_path.read_text()
+ plan = parse_plan(md)
+ errs = validate_plan(plan)
+ print(f"\n{md}\n--- {md_path}: {len(plan['characters'])} characters, {len(plan['pages'])} pages")
+ for e in errs:
+ print(" !", e)
+ k = input("[c]onfirm [e]dit [g]emma revise [q]uit > ").strip().lower()
+ if k == "c" and not errs:
+ pj = compile_plan(md, plan)
+ if pj:
+ js_path.write_text(json.dumps(pj, indent=2) + "\n")
+ return pj
+ elif k == "e":
+ editor(md_path)
+ elif k == "g":
+ ins = input("instruction for Gemma: ")
+ md_path.write_text(strip_fences(chat(
+ DRAFT_SYS, f"Current plan:\n\n{md}\n\nRevise it: {ins}\nReturn the complete new plan.")) + "\n")
+ elif k == "q":
+ sys.exit(0)
+
def selftest():
plan = parse_plan(SAMPLE_PLAN)
assert (plan["title"], plan["series"], plan["book"]) == ("Rooftop", "Test Series", "Book One"), plan
@@ -413,6 +564,8 @@ def selftest():
k = sd_args("krea2", "x", Path("o_%d.png"), SIZES["portrait"], 1)
assert k[k.index("--llm") + 1] == str(SD_DIR / "Qwen3VL-4B-Instruct-Q8_0.gguf")
assert k[k.index("--steps") + 1] == "8" and k[k.index("--cfg-scale") + 1] == "1.0"
+ assert strip_fences("```markdown\n# T\nx\n```\n") == "# T\nx"
+ assert strip_fences("# T\nx") == "# T\nx"
with tempfile.TemporaryDirectory() as t:
t = Path(t)
(t / "c.conf").write_text('# comment\nLIBRARY="/data/my lib"\nPORT=8642\n')