Files
manga-recap-pipeline/worker_vision.py
T
kami 0d281016e7 Cut the dead attic workers, and file what the audit left standing
attic/worker_ocr.py and attic/worker_parse.py are 224 lines imported by nothing
and named in no doc. The OCR stage was removed when narration moved to the
director beat. The two design notes in attic/ stay, they are history.

worker_vision._panel_size had one reference and it was the definition.

The audit's larger finding is filed rather than fixed: call_gemma4,
_extract_json and _strip_thought exist in both worker_vision and worker_script
and have already diverged. That matters because the JSON repair pass can
fabricate dialogue, so a fix would land in one copy and not the other. It is
caveats/audit-open.md#gemma-helpers-duplicated with its revisit trigger.

HANDOFF.md carries the rest: _wrap2 against textwrap, the duplicated ONNX
preprocessing, and worker_layers pointing at a legacy/ directory that was never
tracked in git.

Checked: ruff clean, worker_vision and worker_render self-checks pass.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-13 23:06:34 +04:00

1284 lines
72 KiB
Python

# worker_vision.py — gemma4 vision worker :8002. Hosts the split panel-understanding stages that
# all share the ONE warm gemma4 server (session manager owns its lifecycle on :8090):
# /vision — character DETECT: who's in the panel (bbox, appearance, gender). feeds identity.
# /dialogue — bubbles + text cleanup + speaker attribution + entities. gets the clean cast.
# /direct — camera + transition direction. runs off the panel + scene understanding.
# /vision/same — same-person adjudication for the reconcile stage (compares two ref crops).
# each stage writes its keys into the same per-panel vision_results blob on the orchestrator, so
# scene/script/render still read one merged {characters,dialogue,entities,camera,transition,scene}.
import os, uuid, re, json, base64, time, hashlib
from fastapi import FastAPI
from pydantic import BaseModel
import requests
import transport
app = FastAPI()
transport.install_logging(app, "vision")
SHM = "/dev/shm"
GEMMA4_URL = os.environ.get("GEMMA4_URL", "http://127.0.0.1:8090")
# set-of-mark speaker attribution: detect text regions with comic-text-detector (ONNX, CPU) and draw
# numbered boxes on the panel so gemma transcribes/attributes grounded regions instead of eyeballing
# the whole image. ON by default: face->identity pairing is now gated on containment, so an unmatched
# face is labelled "unknown" instead of borrowing the nearest name (see _pair_faces_to_present).
# SOM_ATTRIBUTION=0 falls back to the holistic path (see bubble_detect.py).
try:
import bubble_detect
except Exception: # onnxruntime/model absent -> feature simply stays unavailable
bubble_detect = None
try:
import face_detect
except Exception:
face_detect = None
SOM = os.environ.get("SOM_ATTRIBUTION", "1") == "1"
def _pair_faces_to_present(det_faces: list, present: list, margin: float = 0.25) -> list:
"""det_faces = real detector boxes (grounded but identity-less); present = characters gemma placed
in the panel (identity + a coarse, imprecise bbox). Pair a face with a present character ONLY when
the face centre falls inside that character's bbox grown by `margin` of its size — gemma's boxes are
imprecise but not arbitrary. An unpaired face stays unknown instead of borrowing whoever happens to
be nearest; the label it carries becomes `speaker_method="som_face"`, the highest-trust provenance
the pipeline has, so an ungated guess used to launder itself into evidence.
Pairs are taken globally shortest-first, so the first face processed cannot claim a character that
fits a later face far better. Each face and each character is used at most once."""
def cx_cy(b): return ((b[0] + b[2]) / 2, (b[1] + b[3]) / 2)
def contains(face_box, char_box) -> bool:
w, h = char_box[2] - char_box[0], char_box[3] - char_box[1]
if w <= 0 or h <= 0:
return False
fx, fy = cx_cy(face_box)
return (char_box[0] - margin * w <= fx <= char_box[2] + margin * w
and char_box[1] - margin * h <= fy <= char_box[3] + margin * h)
pairs = []
for i, f in enumerate(det_faces):
fx, fy = cx_cy(f["bbox"])
for j, c in enumerate(present):
b = c.get("bbox")
if not b or len(b) != 4 or not contains(f["bbox"], b):
continue
px, py = cx_cy(b)
pairs.append(((px - fx) ** 2 + (py - fy) ** 2, i, j))
taken_f, taken_c, match = set(), set(), {}
for _, i, j in sorted(pairs):
if i in taken_f or j in taken_c:
continue
taken_f.add(i)
taken_c.add(j)
match[i] = j
out = []
for i, f in enumerate(det_faces):
c = present[match[i]] if i in match else {}
out.append({"label": f"P{i + 1}", "bbox": f["bbox"], "local_id": c.get("local_id"),
"who": c.get("name") or c.get("desc") or "unknown", "gender": c.get("gender")})
return out
def _set_of_mark(local_path: str, present: list):
"""returns (image_path_to_send, prompt_note, label_map). With SoM on: draws numbered RED boxes on
detected text regions (grounds transcription -> no cross-panel bleed, no phantom lines) and labelled
GREEN boxes P1/P2/... on real detector faces (grounds attribution -> gemma names a visible face, not
a guess). label_map maps "P1" -> that character's local_id (None for an unidentified face).
Off / unavailable / nothing to mark -> (original path, "", {}) so the caller uses the holistic path."""
if not (SOM and bubble_detect):
return local_path, "", {}
import cv2
img = cv2.imread(local_path)
if img is None:
return local_path, "", {}
try:
regions = bubble_detect.detect_text_regions(img)
except Exception as e: # detector missing/broken must not fail the dialogue call
print(f"[dialogue] set-of-mark detect failed, holistic fallback: {e}", flush=True)
return local_path, "", {}
det_faces = []
if face_detect:
try:
det_faces = face_detect.detect_faces(img)
except Exception as e:
print(f"[dialogue] face detect failed, text marks only: {e}", flush=True)
faces = _pair_faces_to_present(det_faces, present)
if not regions and not faces:
return local_path, "", {}
vis = bubble_detect.draw_region_marks(img, regions) # copies img (region boxes, or a bare copy)
if faces:
bubble_detect.draw_face_marks(vis, faces) # in place
marked = f"{SHM}/som_{uuid.uuid4().hex[:8]}.png"
cv2.imwrite(marked, vis)
parts = []
if regions:
parts.append(f"{len(regions)} text region(s) are numbered with red boxes (#1..#{len(regions)}) "
"in reading order. These are the ONLY real dialogue/caption text here — transcribe "
"EXACTLY these, one output line per region in order; do NOT invent, merge away, or "
"omit text outside them.")
if faces:
leg = "; ".join(f["label"] + " = " + f["who"]
+ (f" ({f['gender']})" if f.get("gender") and f["gender"] != "unknown" else "")
for f in faces)
parts.append("Character faces are boxed in green and labelled: " + leg + ". For each line set "
"speaker to the LABEL of the face that says it (e.g. \"P1\"); or a NAME from the "
"conversation flow if the speaker is off-panel; or \"unknown\" if genuinely unclear.")
return marked, "\n" + "\n".join(parts) + "\n", {f["label"]: f["local_id"] or "unknown" for f in faces}
_GENDER_SUFFIX = re.compile(r"\s*\((?:m|f|male|female|man|woman|unknown)\)\s*$", re.IGNORECASE)
_ID_SHAPED = re.compile(r"^(?:person|char|face|p)[\s_-]*\d+$", re.IGNORECASE)
def _present_keys(present: list) -> dict:
"""Every string the dialogue prompt shows for a present character, mapped to its local_id.
`build_dialogue_prompt` renders each one as `- person_1: brown ponytail, green dress (f)`, and gemma
answers with any part of that line, most often the description. Those answers used to fall through
`normalize_speaker` as free-form NAMES and never matched the registry, losing 28 of 51 speech lines
on job 778297bc. A key shared by two present characters is dropped: it cannot identify either."""
keys: dict = {}
for c in present:
lid = (c.get("local_id") or "").strip()
if not lid:
continue
name, desc, gender = c.get("name") or "", c.get("desc") or "", c.get("gender") or ""
shown = name or desc or "unknown"
forms = {lid, name, desc, shown}
if gender and gender != "unknown":
forms |= {f"{f} ({gender})" for f in (name, desc, shown) if f}
for f in forms:
k = f.strip().casefold()
if not k or k == "unknown": # the prompt prints "unknown" for a nameless character
continue
keys[k] = lid if keys.get(k, lid) == lid else None # ambiguous key -> unusable
return {k: v for k, v in keys.items() if v}
def _apply_speaker_labels(dialogue: list, label_map: dict, present: list | None = None) -> list:
"""Map gemma's speaker answer back to a panel-local id.
Three answer shapes reach here: a set-of-mark face label ("P1"), a local_id, and any string the
prompt showed for a present character. An id-shaped answer that names nobody present is junk and
becomes "unknown" rather than a name claim (invariant 6). A trailing gender marker is stripped, so
"Seonho (m)" can still match the registry name "Seonho" off-panel."""
keys = _present_keys(present or [])
for d in dialogue:
s = (d.get("speaker") or "").strip()
if not s:
continue
if s in label_map:
lid = label_map[s]
# _set_of_mark labels a detected face `unknown` when gated pairing matched it to no present
# character. An answer pointing at such a mark grounds nothing, so it must not carry
# `som_face`, the highest-trust label. 7 of 7 som_face lines on the first 36 panels of the
# 2026-08-11 rerun were this case.
d["speaker"] = lid
if lid != "unknown":
d["speaker_method"] = "som_face"
continue
bare = _GENDER_SUFFIX.sub("", s).strip()
lid = keys.get(s.casefold()) or keys.get(bare.casefold())
if lid:
d["speaker"] = lid
elif _ID_SHAPED.match(bare):
d["speaker"] = "unknown" # an id for nobody present: never mint a name from it
elif bare != s:
d["speaker"] = bare # off-panel name, gender marker stripped
return dialogue
def _strip_thought(text: str) -> str:
# gemma4 wraps CoT as `<|channel>thought ... <channel|>`; the answer follows the close marker.
parts = re.split(r"<\|?channel\|?>", text)
return parts[-1].strip() if len(parts) > 1 else text
def _extract_json(raw: str) -> dict:
"""strip gemma4 thought, pull the first JSON object, parse it."""
text = _strip_thought(raw)
i = text.find("{")
if i < 0:
raise ValueError(f"no json in response: {text[:200]}")
# raw_decode stops at the end of the FIRST object. The old greedy `\{.*\}` ran to the LAST brace in
# the reply, so a second object or any trailing braced prose produced an unparseable span and burned
# a repair call on a response that was already fine.
try:
obj, _ = json.JSONDecoder().raw_decode(text[i:])
except json.JSONDecodeError as e:
raise ValueError(f"bad json in response: {text[:200]}") from e
return obj
def _img_part(image_path: str) -> dict:
b64 = base64.b64encode(open(image_path, "rb").read()).decode()
return {"type": "image_url", "image_url": {"url": f"data:image/png;base64,{b64}"}}
def call_gemma4(content: list, temperature: float = 0.2, max_tokens: int = 768) -> str:
"""content = the user message's content list (text + image parts)."""
payload = {"messages": [{"role": "user", "content": content}],
"temperature": temperature, "max_tokens": max_tokens}
# The gemma4 server (session-manager-owned on :8090) can drop the connection or refuse it when it's
# busy or gets restarted after an OOM crash. Both are transient: the session manager supervises the
# subprocess and respawns it. Retry until a deadline so one call rides across a full model reload
# (~20-60s for the 12B) instead of 500-ing and costing the orchestrator a whole panel.
deadline = time.time() + 120
attempt = 0
while True:
try:
r = requests.post(f"{GEMMA4_URL}/v1/chat/completions", json=payload, timeout=300)
r.raise_for_status()
return r.json()["choices"][0]["message"]["content"]
except (requests.ConnectionError, requests.Timeout) as e:
if time.time() >= deadline:
raise
attempt += 1
print(f"[gemma4] {type(e).__name__}, retry {attempt} (server restarting?)", flush=True)
time.sleep(min(5 * attempt, 15))
def call_gemma4_json(content: list, temperature: float = 0.2, max_tokens: int = 768) -> dict:
"""call + parse JSON, with ONE repair retry before giving up (182). a malformed/truncated/
thought-wrapped answer is a TECHNICAL failure, not an editorial one — callers must not turn it
into a skip. raises ValueError if even the repair pass can't be parsed."""
raw = call_gemma4(content, temperature, max_tokens)
try:
return _extract_json(raw)
except (ValueError, json.JSONDecodeError):
pass
# repair pass: hand the model its own broken output and ask for JSON only, more room, temp 0.
repair = [{"type": "text", "text":
"Your previous reply was not valid JSON. Return ONLY the JSON object it should have "
"been — no prose, no thoughts, no markdown fences, complete and parseable.\n\n"
"Previous reply:\n" + raw[:2000]}]
return _extract_json(call_gemma4(repair, temperature=0.0, max_tokens=max_tokens + 256))
# ---------------------------------------------------------------------------
# /vision — character DETECT only. who is in the panel + how they look. no dialogue, no direction.
# ---------------------------------------------------------------------------
def build_detect_prompt(known_characters) -> str:
known = ", ".join(f"{c['name']} ({c.get('description','')})" for c in known_characters
if c.get("name")) or "none"
return (
"You are analyzing a single manga panel. Identify every distinct CHARACTER (person/creature) "
"visibly present. Known characters in this story (reuse the exact name + treat their look as a "
"reference if one clearly appears): " + known + ".\n\n"
"DO NOT count decorative / non-story figures as characters: simplified chibi mascots, "
"icons, avatars, emoji, or the little figures inside an infographic, chart, diagram, "
"compatibility grid, poster, or UI/app screenshot. These are scenery — leave them out of "
"`characters` entirely. Only real people/creatures acting IN the scene count.\n\n"
"For each character give:\n"
"- local_id: a panel-local tag person_1, person_2, ... (top-to-bottom, left-to-right).\n"
"- bbox: pixel bounding box [x1,y1,x2,y2] (top-left, bottom-right corners).\n"
"- appearance: TERSE and non-redundant. hair = one short phrase; clothing = the single most "
"distinctive garment only (never list overlapping items, never repeat a colour/word); "
"features = at most 1-2 truly distinctive marks (glasses, scar). brevity beats completeness — "
"this builds a short label.\n"
"- species: human | cat | dog | bird | horse | creature | other. default human; an animal or "
"non-human creature gets its species so it is named as one (\"the black cat\"), not by clothes.\n"
"- gender: m | f | unknown. read it from presentation; use unknown when genuinely unclear.\n"
"- emotion: one word. action: a short phrase of what they are doing.\n"
"- name: ONLY when this character is explicitly identified in THIS panel (addressed by name, a "
"name tag/caption, or self-introduction). If you are guessing, leave it \"\". A LABEL on a "
"chart/diagram/infographic (e.g. \"TETO GUY\", \"OPTION A\") is NOT a character name — ignore "
"it. Prefer the exact spelling of a known character above if it clearly matches.\n\n"
"Also set skip=true ONLY when the panel carries no narrative content on its own: pure "
"scenery/establishing art, a transition/mood panel, or sfx-only with no characters and no "
"meaningful action. If anyone speaks or acts meaningfully, skip=false.\n"
"Also give the scene: location (short) and time (day/night/unknown).\n\n"
"Respond with ONLY this JSON, no prose:\n"
'{"skip":false,"characters":[{"local_id":"person_1","name":"","species":"human","gender":"unknown",'
'"appearance":{"hair":"","clothing":"","features":[]},"emotion":"","action":"","bbox":[0,0,0,0]}],'
'"scene":{"location":"","time":""}}'
)
class VisionInput(BaseModel):
panel_uri: str
known_characters: list = []
known_entities: list = [] # unused by detect (entities live in /dialogue now)
panel_id: str = ""
session_id: str = ""
BBOX_GRID = 1000 # gemma's native normalized box grid
def _bbox_to_pixels(chars: list, w: int, h: int) -> list:
"""Rewrite every character box from gemma's 0-1000 grid to pixels on this panel.
The prompt asks for pixels. The model answers on its own normalized grid regardless. Measured over
the 113 detections of job 778297bc: 47 boxes had x2 beyond the 900px panel width, and not one had y2
beyond 1000 on panels 1257 to 2307px tall. Consumed as pixels the boxes collapse into the top-left
corner of the panel, which is how identity came to embed crops of speech balloons and window frames
and match them at 0.9, and why gated face pairing returned 7 unknowns out of 7 real faces.
Convert once here so every consumer sees pixels: `_crop_bbox` in identity, the face pairing below,
the set-of-mark boxes, and the review UI's client-side crop.
ponytail: the prompt still says "pixel bounding box". Rewording it would change what the model
emits and needs a GPU run to re-verify, so the boundary converts instead. If a future model really
does answer in pixels, this scales them down -- check the box range before swapping models.
"""
for c in chars:
b = c.get("bbox")
if not (isinstance(b, list) and len(b) == 4 and all(isinstance(v, (int, float)) for v in b)):
continue
xs = sorted((min(w, max(0, round(b[0] * w / BBOX_GRID))),
min(w, max(0, round(b[2] * w / BBOX_GRID)))))
ys = sorted((min(h, max(0, round(b[1] * h / BBOX_GRID))),
min(h, max(0, round(b[3] * h / BBOX_GRID)))))
# ponytail: the model swapped corners on 1 of 117 boxes (p007 person_1, x1 > x2).
# Ordering here is enough because a zero-area box still crops to nothing downstream.
c["bbox"] = [xs[0], ys[0], xs[1], ys[1]]
return chars
def _mark_has_face(img, chars: list) -> list:
"""Stamp `has_face` on every character a real detected face lands inside.
gemma calls every person-shaped region a character. On panel 7 of job 778297bc that was two
people and four background extras, one of which is an empty window mullion. All six reached
identity as equal candidates, and after the bbox fix an extra took `Seonho` at confidence 1.00,
which puts an extra's crop into the lead's reference set. A faceless crop is a bad embedding as
well as a bad reference: the back of a head or a patch of coat matches almost anything.
Reuses `_pair_faces_to_present`, so containment, its margin, and the global shortest-first
assignment are exactly the rules the speaker path already uses. Requires pixel boxes, so call it
after `_bbox_to_pixels`.
Fails open. A missing or broken detector marks everything `True`, because dropping every
character is worse than the over-detection this gate exists to trim.
"""
def _all(v):
for c in chars:
c["has_face"] = v
return chars
if face_detect is None or not chars:
return _all(True)
try:
faces = face_detect.detect_faces(img)
except Exception as e:
print(f"[vision/detect] face detect failed, has_face gate off for this panel: {e}", flush=True)
return _all(True)
paired = {p["local_id"] for p in _pair_faces_to_present(faces, chars) if p.get("local_id")}
for c in chars:
c["has_face"] = c.get("local_id") in paired
return chars
@app.post("/vision")
async def vision(data: VisionInput):
local = transport.get(data.panel_uri, f"{SHM}/vision_{uuid.uuid4().hex[:8]}.png")
prompt = build_detect_prompt(data.known_characters)
try:
# a crowded panel emits one object per character; 768 tokens truncates the array mid-JSON
# (parse fails, retries forever with the same clip). give detect real headroom.
result = call_gemma4_json([{"type": "text", "text": prompt}, _img_part(local)], max_tokens=1536)
except (ValueError, json.JSONDecodeError) as e:
# 182: a parse failure is NOT an editorial skip. Flag parse_failed + skip=false so the
# orchestrator records a stage error and retries this panel on resume, instead of silently
# deleting the panel + its dialogue/narration/audio/video.
print(f"[vision/detect] parse failed for {data.panel_id} after repair retry: {e}", flush=True)
result = {"skip": False, "parse_failed": True, "characters": [], "scene": {}}
finally:
import cv2
img = cv2.imread(local) # read once: the size and the face gate both need it
os.remove(local)
pw, ph = (img.shape[1], img.shape[0]) if img is not None else (0, 0)
result.setdefault("characters", [])
if pw and ph:
_bbox_to_pixels(result["characters"], pw, ph)
_mark_has_face(img, result["characters"])
else:
print(f"[vision/detect] panel size unreadable for {data.panel_id}, boxes left normalized",
flush=True)
result["panel_id"] = data.panel_id
return result
# ---------------------------------------------------------------------------
# /dialogue — bubble classify + text cleanup + speaker attribution + entities. gets the clean cast.
# gemma reads the bubble text straight from the panel image now (no external OCR feed).
# ---------------------------------------------------------------------------
def build_dialogue_prompt(present_characters, recent=None, roster_cast=None) -> str:
# present_characters: [{"local_id","name","gender","desc"}] resolved from the merged registry.
# recent: [{"speaker","text"}] — the last few attributed lines of the chapter, for turn-taking
# when a bubble's speaker is drawn off-panel (very common in webtoons: bubble-only panels).
# roster_cast: [name] — other named chapter characters, so a bubble-only panel still has a cast
# to attribute an off-panel speaker to by name (181).
who = "\n".join(
f"- {c['local_id']}: {c.get('name') or c.get('desc') or 'unknown'}"
+ (f" ({c['gender']})" if c.get("gender") and c["gender"] != "unknown" else "")
for c in present_characters) or "- (none detected)"
convo = "\n".join(f"- {r.get('speaker') or 'someone'}: \"{r.get('text','')}\""
for r in (recent or [])) or "- (start of scene)"
cast = ", ".join(roster_cast) if roster_cast else ""
return (
"You are transcribing the text of a single manga panel. Read the bubble text directly from the "
"panel image (lettering is usually all-caps). The characters present (with their "
"panel-local ids) are:\n" + who + "\n\n"
+ (f"Other named characters in this chapter (may speak from off-panel): {cast}\n\n" if cast else "")
+ "The conversation SO FAR this scene, attributed (authoritative speaker history — use it to "
"track who has been speaking and whose turn it is):\n" + convo + "\n\n"
"For each text bubble, classify its type by shape: speech (rounded), thought (cloud/dashed), "
"shout (jagged/spiky), narration (rectangular box), or sfx (sound effect). When ONE sentence "
"is split across joined/adjacent bubbles from the same speaker, merge them into a SINGLE line. "
"Rewrite the text in normal sentence case, fixing garbled lettering and capitalizing names/"
"proper nouns. Transcribe ONLY English text. DROP entirely — do not include as a line — any text that "
"is:\n"
" - not in the Latin alphabet (Korean/Japanese/Chinese glyphs, e.g. 에겐남/척/하): this is "
"decorative art, labels, or onomatopoeia, NEVER dialogue — omit it, do not transliterate it;\n"
" - a lone syllable or 1-3 character fragment (\"bl\", \"kk\", \"Et\", \"ha\"): garbled sfx, "
"never speech;\n"
" - background sfx / sound effects with no speaker.\n"
"Keep a line ONLY if it is a real, readable English word or phrase actually spoken or narrated.\n"
"SPEAKER attribution — a bubble often belongs to a character NOT drawn in this panel "
"(off-panel speaker; a panel may be ONLY a floating bubble). Use the tail direction and the "
"recent conversation's turn-taking. Set speaker to:\n"
" - the local_id of the present character clearly saying it; OR\n"
" - a character NAME from the recent conversation when turn-taking points off-panel; OR\n"
" - \"unknown\" when it is clearly speech but you cannot tell who (narrated as \"someone\"); OR\n"
" - \"\" for narration boxes / sfx.\n"
"If EXACTLY ONE character is present, a speech/shout/thought line is almost certainly theirs — "
"give it that local_id, do NOT hedge to \"unknown\". With TWO OR MORE present, attribute only "
"when the tail or turn-taking supports it; otherwise prefer a NAME from the recent conversation "
"or \"unknown\" over guessing between them. Set confidence 0..1 (1.0 clear tail / lone speaker, "
"~0.3 when guessing).\n"
"Also extract any named places, organizations, shops, or brands visible in the panel.\n\n"
"NAMING — when a bubble explicitly identifies one of the present characters (someone is "
"addressed by name, introduces themselves, or is named in a caption pointing at them), add "
"{target_local_id,name,evidence_type,confidence} to `named`. Use evidence_type address, "
"self_intro, name_tag, or caption. Give the BARE personal name only — strip Korean/Japanese "
"honorifics and address suffixes (-ssi, -nim, -yang, -kun, -chan, -san, hyung, noona, sunbae, "
"oppa, unnie). Only include a naming you are confident maps to that visible local_id; omit "
"guesses and off-panel names.\n\n"
"Respond with ONLY this JSON, no prose:\n"
'{"dialogue":[{"speaker":"person_1","confidence":1.0,"type":"speech","text":""}],'
'"named":[{"target_local_id":"person_1","name":"","evidence_type":"address",'
'"confidence":0.9}],"entities":[{"name":"","kind":"shop"}]}'
)
class DialogueInput(BaseModel):
panel_uri: str
present_characters: list = [] # [{local_id,name,gender,desc}] from the merged registry
recent: list = [] # [{speaker,text}] last few attributed lines (cross-panel context)
roster_cast: list = [] # [name] other named chapter characters, for off-panel attribution
panel_id: str = ""
session_id: str = ""
_SPEECH = {"speech", "shout", "thought"}
def resolve_speakers(dialogue: list, present: list) -> list:
"""Deterministic backstop AFTER gemma's attribution, mutates+returns `dialogue`. The dominant
real panel is a SOLO reaction shot: one character present, a speech/shout/thought bubble, and
gemma hedges to "unknown"/"" — that line is almost certainly the one present character's, so
assign it (the p109 "someone else"-is-actually-him bug). Left untouched: narration/sfx, and any
panel with 0 or 2+ present (a wrong guess between faces is worse than "someone").
ponytail: solo-only. Multi-character attribution needs per-balloon geometry (set-of-mark on a
detected balloon layer) — add that when the balloon detector lands."""
if len(present) != 1:
_annotate_speaker_methods(dialogue, present)
return dialogue
only = (present[0].get("local_id") or "").strip()
if not only:
_annotate_speaker_methods(dialogue, present)
return dialogue
for d in dialogue:
if d.get("type", "speech") not in _SPEECH:
continue
if (d.get("speaker") or "").strip().lower() in ("", "unknown"):
d["speaker"] = only
d["confidence"] = 0.7 # inferred from sole presence, not a read tail
d["speaker_method"] = "solo_prior"
_annotate_speaker_methods(dialogue, present)
return dialogue
def _annotate_speaker_methods(dialogue: list, present: list) -> list:
"""Fill provenance for model-attributed lines without overwriting grounded/backstop methods.
A speaker matching a present local_id used to be stamped `tail`, the highest-trust label, at
gemma's own confidence of 1.0. No balloon geometry was ever read. On the 2026-08-11 chapter every
sampled two-character panel had the speakers swapped
(`caveats/speaker-attribution.md#tail-is-not-geometry`), so with 2+ present the guess is dropped
rather than shipped as truth. With one present it is the same claim as the solo backstop, so it is
kept and named for what it is.
ponytail: drop-on-crowd is the honest floor, not the fix. Bind by tail geometry when the balloon
detector lands, then this branch reads a tail for real."""
local_ids = {c.get("local_id") for c in present if c.get("local_id")}
crowded = len(present) > 1
for d in dialogue:
if d.get("speaker_method"):
continue
speaker = (d.get("speaker") or "").strip()
if d.get("type", "speech") not in _SPEECH or not speaker or speaker == "unknown":
d["speaker_method"] = "unknown"
elif speaker in local_ids and crowded:
d["speaker"] = "unknown"
d["confidence"] = 0.0
d["speaker_method"] = "unknown"
elif speaker in local_ids:
d["speaker_method"] = "model_solo"
d["confidence"] = min(float(d.get("confidence") or 0.7), 0.7)
else:
d["speaker_method"] = "turn_taking"
return dialogue
def _dialogue_envelope(expected_ids: list, parsed, parse_failed: bool = False) -> dict:
"""Build the shared fail-loud status envelope without manufacturing silent panel rows."""
rows = parsed if isinstance(parsed, list) else []
returned = {d.get("panel_id") for d in rows if isinstance(d, dict) and d.get("panel_id")}
missing = [pid for pid in expected_ids if pid not in returned]
warnings = [f"missing panel_id: {pid}" for pid in missing]
if parse_failed:
status = "failed"
warnings.insert(0, "dialogue response was not valid JSON")
elif missing:
status = "partial"
else:
status = "ok"
return {"status": status, "parse_failed": parse_failed,
"expected_items": len(expected_ids), "returned_items": len(returned),
"warnings": warnings}
_CLAIM_EVIDENCE = {"address", "self_intro", "name_tag", "caption", "roster_hint"}
def _normalize_claims(rows: list, panel_id: str) -> list:
"""Make model name evidence typed and resume-stable; never turn a claim into identity here."""
out = []
for i, row in enumerate(rows or []):
if not isinstance(row, dict):
continue
name = str(row.get("name") or "").strip()
target = str(row.get("target_local_id") or row.get("local_id") or "").strip()
evidence = row.get("evidence_type", "roster_hint")
if not name or not target or evidence not in _CLAIM_EVIDENCE:
continue
seed = f"{panel_id}\0{target}\0{name.casefold()}\0{evidence}\0{i}"
out.append({"claim_id": "claim_" + hashlib.sha256(seed.encode()).hexdigest()[:16],
"panel_id": panel_id, "name": name, "target_local_id": target,
"evidence_type": evidence,
"confidence": max(0.0, min(1.0, float(row.get("confidence", 0.5))))})
return out
@app.post("/dialogue")
async def dialogue(data: DialogueInput):
local = transport.get(data.panel_uri, f"{SHM}/dlg_{uuid.uuid4().hex[:8]}.png")
send_path, note, label_map = _set_of_mark(local, data.present_characters)
prompt = build_dialogue_prompt(data.present_characters, data.recent, data.roster_cast) + note
parse_failed = False
try:
# same truncation risk as detect: a text-heavy panel with many bubbles + entities.
result = call_gemma4_json([{"type": "text", "text": prompt}, _img_part(send_path)], max_tokens=1536)
except (ValueError, json.JSONDecodeError) as e:
print(f"[dialogue] parse failed for {data.panel_id}: {e}", flush=True)
result = {"dialogue": [], "entities": []}
parse_failed = True
finally:
os.remove(local)
if send_path != local:
os.remove(send_path)
result.setdefault("dialogue", [])
result.setdefault("entities", [])
result.setdefault("named", []) # [{local_id, name}] explicit namings -> bound to character_id by orchestrator
result["named"] = _normalize_claims(result["named"], data.panel_id)
_apply_speaker_labels(result["dialogue"], label_map, data.present_characters)
resolve_speakers(result["dialogue"], data.present_characters)
result["panel_id"] = data.panel_id
result.update(_dialogue_envelope([data.panel_id], [] if parse_failed else [result], parse_failed))
return result
# /dialogue/window — the same transcription as /dialogue, but over a short WINDOW of consecutive
# panels in ONE multi-image call. The model reads the whole exchange at once, so turn-taking /
# off-panel speaker attribution is decided from the actual conversational flow across panels rather
# than a text-only `recent` summary fed one panel at a time. keep the window small: each panel image
# is ~256 tokens and the transcript output is large, so 8 panels would overflow the 4096 gemma ctx.
def build_dialogue_window_prompt(panels_meta: list, recent=None, roster_cast=None) -> str:
"""panels_meta: [{panel_id, present:[{local_id,name,gender,desc}]}] in reading order (images
appended separately, one per panel, labelled by panel id)."""
convo = "\n".join(f"- {r.get('speaker') or 'someone'}: \"{r.get('text','')}\""
for r in (recent or [])) or "- (start of scene)"
cast = ", ".join(roster_cast) if roster_cast else ""
who_blocks = []
for m in panels_meta:
who = "\n".join(
f" - {c['local_id']}: {c.get('name') or c.get('desc') or 'unknown'}"
+ (f" ({c['gender']})" if c.get("gender") and c["gender"] != "unknown" else "")
for c in m.get("present", [])) or " - (none detected)"
who_blocks.append(f"Panel id={m['panel_id']} characters present:\n{who}")
rosters = "\n\n".join(who_blocks)
return (
"You are transcribing the text of CONSECUTIVE manga panels in reading order (their images "
"follow, one per panel, each labelled with its panel id). Read the bubble text directly from "
"each panel image (lettering is usually all-caps).\n\n"
+ rosters + "\n\n"
+ (f"Other named characters in this chapter (may speak from off-panel): {cast}\n\n" if cast else "")
+ "The conversation SO FAR this scene, attributed (authoritative speaker history — use it to "
"track who has been speaking and whose turn it is; a new line usually continues the current "
"turn-taking):\n" + convo + "\n\n"
"For each text bubble, classify its type by shape: speech (rounded), thought (cloud/dashed), "
"shout (jagged/spiky), narration (rectangular box), or sfx (sound effect). When ONE sentence "
"is split across joined/adjacent bubbles from the same speaker, merge them into a SINGLE line. "
"Rewrite the text in normal sentence case, fixing garbled lettering and capitalizing names/"
"proper nouns. Transcribe ONLY English text. DROP entirely — do not include as a line — any text that "
"is:\n"
" - not in the Latin alphabet (Korean/Japanese/Chinese glyphs): decorative art, labels, or "
"onomatopoeia, NEVER dialogue — omit it, do not transliterate it;\n"
" - a lone syllable or 1-3 character fragment (\"bl\", \"kk\", \"Et\", \"ha\"): garbled sfx;\n"
" - background sfx / sound effects with no speaker.\n"
"Keep a line ONLY if it is a real, readable English word or phrase actually spoken or narrated.\n"
"SPEAKER attribution — a bubble often belongs to a character NOT drawn in that panel (off-panel "
"speaker; a panel may be ONLY a floating bubble). Use the tail direction AND the flow of the "
"conversation ACROSS these panels. Set speaker to:\n"
" - the local_id of the present character in THAT panel clearly saying it; OR\n"
" - a character NAME (from a prior panel's cast or the recent conversation) when turn-taking "
"points off-panel; OR\n"
" - \"unknown\" when it is clearly speech but you cannot tell who; OR\n"
" - \"\" for narration boxes / sfx.\n"
"Attribution priors: (a) if EXACTLY ONE character is present in a panel, a speech/shout/thought "
"line there is almost certainly theirs — give it that local_id, do NOT hedge to \"unknown\"; "
"(b) a panel showing no face (only a body, hands, an object, or background) usually CONTINUES "
"the previous panel's speaker — carry that speaker forward; (c) attribute each line to the panel "
"whose image actually shows that bubble — NEVER move a line onto a neighbouring panel. With two "
"or more faces present and no clear tail, prefer a NAME from the flow or \"unknown\" over a "
"wrong guess. Set confidence 0..1 (1.0 clear tail / lone speaker, ~0.3 when guessing).\n"
"Also extract any named places, organizations, shops, or brands visible in the panels.\n\n"
"NAMING — when a bubble explicitly identifies a present character (addressed by name, "
"self-introduction, or a caption pointing at them), add "
"{target_local_id,name,evidence_type,confidence} to that panel's "
"`named`. Give the BARE personal name only — strip honorifics (-ssi, -nim, -yang, -kun, -chan, "
"-san, hyung, noona, sunbae, oppa, unnie). Omit guesses and off-panel names.\n\n"
"Return a result for EVERY panel id, in reading order. Respond with ONLY this JSON, no prose:\n"
'{"panels":[{"panel_id":"...","dialogue":[{"speaker":"person_1","confidence":1.0,'
'"type":"speech","text":""}],"named":[{"target_local_id":"person_1","name":"",'
'"evidence_type":"address","confidence":0.9}],'
'"entities":[{"name":"","kind":"shop"}]}]}'
)
class DialogueWindowInput(BaseModel):
panels: list = [] # [{panel_id, panel_uri, present:[{local_id,name,gender,desc}]}]
recent: list = [] # [{speaker,text}] last few attributed lines before the window
roster_cast: list = [] # [name] other named chapter characters
session_id: str = ""
@app.post("/dialogue/window")
async def dialogue_window(data: DialogueWindowInput):
ids_in_order = [p["panel_id"] for p in data.panels]
content = [{"type": "text", "text": build_dialogue_window_prompt(
data.panels, data.recent, data.roster_cast)}]
locals_: list = []
label_maps: dict = {}
try:
for i, p in enumerate(data.panels, 1):
local = transport.get(p["panel_uri"], f"{SHM}/dlgw_{uuid.uuid4().hex[:8]}.png")
locals_.append(local)
send_path, note, label_maps[p["panel_id"]] = _set_of_mark(local, p.get("present", []))
if send_path != local:
locals_.append(send_path)
content.append({"type": "text", "text": f'Panel {i} (id={p["panel_id"]}):{note}'})
content.append(_img_part(send_path))
parse_failed = False
try:
parsed = call_gemma4_json(content, max_tokens=2048).get("panels", [])
except (ValueError, json.JSONDecodeError) as e:
print(f"[dialogue/window] parse failed: {e}", flush=True)
parsed = []
parse_failed = True
finally:
for l in locals_:
os.remove(l)
by_id = {d.get("panel_id"): d for d in parsed if isinstance(d, dict)}
present_by_id = {p["panel_id"]: p.get("present", []) for p in data.panels}
out = []
for pid in ids_in_order:
d = by_id.get(pid)
if d is None: # missing is unresolved, never manufactured as a silent success
continue
_apply_speaker_labels(d.get("dialogue", []), label_maps.get(pid, {}), present_by_id.get(pid, []))
out.append({
"panel_id": pid,
"dialogue": resolve_speakers(d.get("dialogue", []), present_by_id.get(pid, [])),
"named": _normalize_claims(d.get("named", []), pid),
"entities": d.get("entities", []),
})
return {"panels": out, **_dialogue_envelope(ids_in_order, parsed, parse_failed)}
# ---------------------------------------------------------------------------
# /roster — chapter-level coarse cast + premise from a spread of page images (181). read once before
# detection; the orchestrator threads it back as NAME/appearance hints (never as identity evidence).
# ---------------------------------------------------------------------------
def build_roster_prompt(n_pages: int) -> str:
return (
f"You are given {n_pages} representative page images sampled from ONE manga chapter, in reading "
"order. Read the pages (text included) and build a COARSE chapter roster — hints only, you will "
"not see every page.\n\n"
"Give:\n"
"- premise: 1-2 sentences on what this chapter is about (for narration tone/continuity).\n"
"- characters: the few clearly RECURRING named or central characters. For each:\n"
" - name: bare personal name, honorifics/address-suffixes stripped (-ssi,-nim,-kun,-chan,-san,"
" hyung, noona, sunbae, oppa, unnie).\n"
" - aliases: other names/titles they are called (include honorific forms here).\n"
" - gender: m | f | unknown. - species: human | cat | dog | creature | other.\n"
" - description: one short phrase of their look (hair + most distinctive garment).\n"
"Only include characters you are confident recur; omit one-off background people and guesses.\n\n"
"Respond with ONLY this JSON, no prose:\n"
'{"premise":"","characters":[{"name":"","aliases":[],"gender":"unknown","species":"human","description":""}]}'
)
class RosterInput(BaseModel):
page_uris: list = [] # sampled page image uris, in reading order
session_id: str = ""
@app.post("/roster")
async def roster(data: RosterInput):
locals_ = [transport.get(u, f"{SHM}/roster_{uuid.uuid4().hex[:8]}.png") for u in data.page_uris]
content = [{"type": "text", "text": build_roster_prompt(len(locals_))}] + [_img_part(p) for p in locals_]
try:
result = call_gemma4_json(content, max_tokens=1024)
except (ValueError, json.JSONDecodeError) as e:
print(f"[roster] parse failed: {e}", flush=True)
result = {"premise": "", "characters": []}
finally:
for p in locals_:
os.remove(p)
result.setdefault("premise", "")
result.setdefault("characters", [])
return result
# ---------------------------------------------------------------------------
# /direct — camera + transition. cinematography, off the panel + a compact beat summary.
# ---------------------------------------------------------------------------
_NEW_SCENE_RULE = (
"new_scene — true if the panel STARTS a new narrative beat/moment (location change, time skip, "
"new speaker turn, topic shift); false if it continues the SAME moment as the panel before it (a "
"reaction, a second angle, the next line of the same exchange). Consecutive false-panels are shown "
"together as one transitioned scene, so only mark true at real beat boundaries.\n")
_CAMERA_RULE = (
"camera.effect — pick the single best fit: static (calm/quiet beat), zoom_in (rising tension, "
"focus), zoom_out (reveal, aftermath), dolly_to_subject (push in on one reacting character — "
"set to=[x,y] as the normalized 0..1 centre of their face), pan_left/pan_right (horizontal "
"action or a wide panel), pan_up/pan_down (tall panel or vertical motion), shake "
"(impact/shock), orbit (dramatic emphasis). Default zoom_in when unsure. Set to only for "
"dolly_to_subject.\n")
_TRANSITION_RULE = (
"transition — how the cut OUT of the panel feels: cut (default, most panels — snappy), "
"crossfade/dissolve (soft time/place change), fade_black (scene break/end of a beat), "
"fade_white (flashback, shock, blast), wipe_left/wipe_right (parallel action), push (energetic "
"scene change). Prefer cut; reserve the rest for real scene boundaries.\n")
def build_direct_prompt(beat: str, prev_beat: str = "") -> str:
prev = ("Previous panel's beat: " + prev_beat + "\n"
if prev_beat else "This is the first panel of the chapter.\n")
return (
"You are the director for a manga recap video. Given this manga panel and a one-line "
"beat summary, choose how to shoot it and whether it opens a new scene.\n\n"
+ prev +
"Beat: " + (beat or "(no dialogue; read the panel)") + "\n\n"
+ _NEW_SCENE_RULE + "The first panel is always a new scene.\n"
+ _CAMERA_RULE + _TRANSITION_RULE + "\n"
"Respond with ONLY this JSON, no prose:\n"
'{"new_scene":true,"camera":{"effect":"zoom_in","to":[0.5,0.35]},"transition":"cut"}'
)
# the window director is TWO focused passes over the same panel images, not one combined call:
# grouping precision drops when the model also has to design shots in the same breath (it re-merged a
# character-entrance panel into the prior beat). one call decides beats, one decides shots.
def build_direct_group_prompt(panels: list, story: str = "", prev_beat: str = "") -> str:
"""panels: [{panel_id, beat}] in reading order (images appended separately). grouping ONLY."""
ctx = (f"Story so far (what has happened up to now): {story}\n" if story else "")
prev = (f"The beat immediately before this window: {prev_beat}\n" if prev_beat else
"This window starts the chapter.\n")
return (
"These are CONSECUTIVE manga panels in reading order, each with a short beat description and "
"its image. Group them into distinct SCENES/BEATS: consecutive panels showing the SAME moment "
"belong to ONE group — same location and continuous action, INCLUDING back-and-forth dialogue, "
"reaction shots, and close-ups that are part of the same exchange. Prefer grouping: a speaker "
"turn or a reaction within the same conversation is NOT a new beat. Only start a new group on a "
"real break — a change of location, a time skip, or a clear jump to a different subject or event. "
"A lone solo panel should be the exception, not the rule.\n\n"
+ ctx + prev + "\n"
"Set continues_previous true only when the FIRST panel continues the beat immediately before "
"this window; false when it starts the chapter or a genuinely new beat. List every panel id "
"exactly once, in reading order. For each group give a short 'why' — this "
"reasoning is what makes the grouping accurate, so fill it in. Respond with ONLY this JSON:\n"
'{"continues_previous":false,"groups":[{"panels":["p001","p002"],"why":"..."}]}'
)
def _parse_direct_groups(parsed: dict, ids_in_order: list, has_previous: bool) -> tuple[dict, bool]:
"""Translate grouping output into boundaries while preserving the cross-window edge."""
new_scene = {pid: True for pid in ids_in_order}
for g in parsed.get("groups", []):
members = [m for m in (g if isinstance(g, list) else g.get("panels", [])) if m in new_scene]
for k, pid in enumerate(members):
new_scene[pid] = (k == 0)
continues_previous = bool(has_previous and ids_in_order and parsed.get("continues_previous") is True)
if continues_previous:
new_scene[ids_in_order[0]] = False
return new_scene, continues_previous
def build_direct_shot_prompt(panels: list) -> str:
"""panels: [{panel_id, beat}] in reading order. camera + transition per panel ONLY."""
ids = ", ".join(p["panel_id"] for p in panels)
return (
"These are CONSECUTIVE manga panels in reading order, each with a beat and its image. For EACH "
"panel choose how to shoot it.\n\n"
+ _CAMERA_RULE + _TRANSITION_RULE + "\n"
f"Return a decision for every panel id ({ids}), in order. Respond with ONLY this JSON:\n"
'{"panels":[{"panel_id":"...","camera":{"effect":"zoom_in","to":[0.5,0.35]},"transition":"cut"}]}'
)
class DirectWindowInput(BaseModel):
panels: list = [] # [{panel_id, panel_uri, beat}] in reading order
story: str = "" # running "story so far" for cross-window continuity
prev_beat: str = "" # last beat of the previous window (boundary continuity)
session_id: str = ""
def _window_content(prompt: str, panels: list, locals_: list) -> list:
content = [{"type": "text", "text": prompt}]
for i, p in enumerate(panels, 1):
local = transport.get(p["panel_uri"], f"{SHM}/dw_{uuid.uuid4().hex[:8]}.png")
locals_.append(local)
content.append({"type": "text", "text": f'Panel {i} (id={p["panel_id"]}): {p.get("beat","")}'})
content.append(_img_part(local))
return content
@app.post("/direct/window")
async def direct_window(data: DirectWindowInput):
ids_in_order = [p["panel_id"] for p in data.panels]
# pass 1 — grouping. first panel of each group starts a new scene.
new_scene = {pid: True for pid in ids_in_order} # default: solo (safe if a panel is dropped)
continues_previous = False
locals_: list = []
try:
raw = call_gemma4(_window_content(
build_direct_group_prompt(data.panels, data.story, data.prev_beat), data.panels, locals_),
temperature=0.2, max_tokens=1024)
new_scene, continues_previous = _parse_direct_groups(
_extract_json(raw), ids_in_order, bool(data.prev_beat))
except (ValueError, json.JSONDecodeError) as e:
print(f"[direct/window] grouping parse failed: {e}", flush=True)
finally:
for l in locals_:
os.remove(l)
# pass 2 — shot design (camera + transition per panel).
shot: dict = {}
locals_ = []
try:
raw = call_gemma4(_window_content(build_direct_shot_prompt(data.panels), data.panels, locals_),
temperature=0.3, max_tokens=1024)
shot = {d.get("panel_id"): d for d in _extract_json(raw).get("panels", []) if isinstance(d, dict)}
except (ValueError, json.JSONDecodeError) as e:
print(f"[direct/window] shot parse failed: {e}", flush=True)
finally:
for l in locals_:
os.remove(l)
out = []
for pid in ids_in_order:
d = shot.get(pid, {})
row = {
"panel_id": pid,
"new_scene": new_scene[pid],
"camera": d.get("camera") or {"effect": "zoom_in"},
"transition": d.get("transition", "cut"),
}
if pid == ids_in_order[0]:
row["continues_previous"] = continues_previous
out.append(row)
# collage diagnostic: how many panels the director actually merged this window. all-solo (grouped=0)
# means every beat is single-panel -> render never reaches the collage path. see grouping.plan_groups.
grouped = sum(1 for pid in ids_in_order if not new_scene[pid])
print(f"[direct/window] {len(ids_in_order)} panels, {grouped} grouped "
f"(-> {len(ids_in_order) - grouped} beats)", flush=True)
return {"panels": out}
class DirectInput(BaseModel):
panel_uri: str
beat: str = "" # compact summary the orchestrator builds from characters+dialogue
prev_beat: str = "" # previous panel's beat, for the new_scene boundary decision
panel_id: str = ""
session_id: str = ""
@app.post("/direct")
async def direct(data: DirectInput):
local = transport.get(data.panel_uri, f"{SHM}/dir_{uuid.uuid4().hex[:8]}.png")
prompt = build_direct_prompt(data.beat, data.prev_beat)
try:
raw = call_gemma4([{"type": "text", "text": prompt}, _img_part(local)],
temperature=0.3, max_tokens=128)
result = _extract_json(raw)
except (ValueError, json.JSONDecodeError) as e:
print(f"[direct] parse failed for {data.panel_id}, default cut/zoom_in: {e}", flush=True)
result = {"new_scene": True, "camera": {"effect": "zoom_in"}, "transition": "cut"}
finally:
os.remove(local)
result.setdefault("camera", {"effect": "zoom_in"})
result.setdefault("transition", "cut")
result["new_scene"] = bool(result.get("new_scene", True))
result["panel_id"] = data.panel_id
return result
# ---------------------------------------------------------------------------
# /vision/same — reconcile: are these two reference crops the same character?
# ---------------------------------------------------------------------------
SAME_PROMPT = (
"These are two cropped images of manga characters, image A then image B. Are they the SAME "
"character (same person/creature), judging by hair, face, build, and outfit — allowing for a "
"different pose, expression, or panel? Two different people who merely share a hair colour are "
"NOT the same. Respond with ONLY this JSON, no prose:\n"
'{"same":true,"confidence":0.0,"reason":""}'
)
class SameInput(BaseModel):
ref_a_uri: str
ref_b_uri: str
session_id: str = ""
@app.post("/vision/same")
async def vision_same(data: SameInput):
a = transport.get(data.ref_a_uri, f"{SHM}/sameA_{uuid.uuid4().hex[:8]}.png")
b = transport.get(data.ref_b_uri, f"{SHM}/sameB_{uuid.uuid4().hex[:8]}.png")
try:
raw = call_gemma4([{"type": "text", "text": SAME_PROMPT}, _img_part(a), _img_part(b)],
temperature=0.1, max_tokens=128)
result = _extract_json(raw)
except (ValueError, json.JSONDecodeError) as e:
print(f"[vision/same] parse failed, treating as NOT same: {e}", flush=True)
result = {"same": False, "confidence": 0.0, "reason": "parse_failed"}
finally:
os.remove(a); os.remove(b)
result.setdefault("same", False)
result.setdefault("confidence", 0.0)
return result
# ---------------------------------------------------------------------------
# /vision/resolve — TIER-2 identity decider. Given ONE person crop and a shortlist of known-character
# sheets (the gender-gated top-K by cosine, supplied by /identity/resolve), gemma decides which
# known character this crop IS, or NONE (a new character). This replaces siglip-cosine as the PRIMARY
# identity signal: cosine only picks the shortlist; gemma makes the call. Runs in a gemma-resident
# stage (like /vision/same), NOT inside the siglip-resident identity stage — the GPU mutex forbids both.
# ---------------------------------------------------------------------------
def _sheet(c: dict) -> str:
"""one compact text character-sheet line from a known-character row."""
ap = c.get("appearance") or c.get("description") or {}
if isinstance(ap, str): # description arrives as a JSON string from the roster
try:
ap = json.loads(ap)
except (ValueError, TypeError):
pass
if isinstance(ap, dict):
hair = ap.get("hair", "")
cloth = ap.get("clothing", "")
feats = ", ".join(ap.get("features", []) or [])
ap = "; ".join(p for p in (f"hair {hair}" if hair else "", cloth, feats) if p)
bits = [b for b in (c.get("name", ""), f"{c.get('gender','')}".strip(), c.get("species", ""), str(ap)) if b]
return ", ".join(bits) or "(no description)"
def build_resolve_prompt(candidates: list) -> str:
"""candidates: [{character_id, name, gender, species, appearance/description}] — the cosine shortlist."""
lines = "\n".join(f"{i+1}. {_sheet(c)}" for i, c in enumerate(candidates))
return (
"This is a cropped image of ONE manga character. Below are known characters from this story, "
"each with a short description followed by zero or more labelled reference images. Decide which "
"known character this crop IS — the SAME person/creature, "
"allowing for a different pose, expression, or panel — or NONE if it is not any of them.\n"
"Judge face shape/features first; use body, hair, and outfit only as secondary evidence. Two different people who merely share a hair "
"colour or setting are NOT the same. When unsure, answer 0 (NONE) rather than guess.\n\n"
"Known characters:\n" + lines + "\n\n"
"Reply with ONLY this JSON: the number of the matching character, or 0 for NONE.\n"
'{"choice":0,"confidence":0.0,"reason":""}'
)
REF_BUDGET = 12 # total reference images one /vision/resolve prompt may carry, beside the query crop
def _refs_per_candidate(n: int) -> int:
return max(1, min(3, REF_BUDGET // max(1, n)))
class ResolveInput(BaseModel):
crop_uri: str
candidates: list = [] # gender-gated gallery [{character_id, name, gender, ...}]
session_id: str = ""
def _resolve_content(crop: str, candidates: list, reference_paths: list[tuple[int, str]]) -> list:
content = [{"type": "text", "text": build_resolve_prompt(candidates)},
{"type": "text", "text": "QUERY CROP:"}, _img_part(crop)]
for candidate_no, path in reference_paths:
content.extend([{"type": "text", "text": f"REFERENCE FOR CANDIDATE {candidate_no}:"},
_img_part(path)])
return content
@app.post("/vision/resolve")
async def vision_resolve(data: ResolveInput):
if not data.candidates:
return {"character_id": None, "confidence": 0.0, "reason": "no_candidates"}
crop = transport.get(data.crop_uri, f"{SHM}/resolve_{uuid.uuid4().hex[:8]}.png")
refs = []
# image budget, not a per-candidate rule. The gallery is the live cast now, not a cosine top-k, so
# 3 references each was 27 images plus the query on a 9-character cast. Spread REF_BUDGET across the
# candidates instead: 3 references while the cast is small, 1 apiece once it is not.
per = _refs_per_candidate(len(data.candidates))
try:
for i, candidate in enumerate(data.candidates, 1):
for uri in (candidate.get("reference_image_uris") or [])[:per]:
try:
refs.append((i, transport.get(uri, f"{SHM}/ref_{uuid.uuid4().hex[:8]}.png")))
except Exception as e:
print(f"[vision/resolve] reference unavailable {uri}: {e}", flush=True)
raw = call_gemma4(_resolve_content(crop, data.candidates, refs), temperature=0.1, max_tokens=128)
result = _extract_json(raw)
except (ValueError, json.JSONDecodeError) as e:
print(f"[vision/resolve] parse failed, treating as NONE: {e}", flush=True)
result = {"choice": 0, "confidence": 0.0, "reason": "parse_failed"}
finally:
os.remove(crop)
for _, path in refs:
os.remove(path)
# map gemma's 1-based choice back to a character_id. An explicit 0 means NONE -> a new character.
# An OUT-OF-RANGE index is a hallucination, not an answer: it must be `unresolved` like a parse
# failure, or a bad index silently mints a brand new character in the permanent registry.
choice = result.get("choice", 0)
in_range = isinstance(choice, int) and 1 <= choice <= len(data.candidates)
cid = data.candidates[choice - 1]["character_id"] if in_range else None
if in_range:
state = "known"
elif choice == 0 and result.get("reason") != "parse_failed":
state = "new"
else:
state = "unresolved"
return {"character_id": cid, "state": state, "confidence": float(result.get("confidence", 0.0)),
"reason": result.get("reason", "")}
@app.get("/health")
async def health():
return {"status": "ok"}
if __name__ == "__main__":
# self-check: each prompt builder mentions its own job; each parser strips thought + reads json.
dp = build_detect_prompt([{"name": "Gojo", "description": "white hair"}])
assert "Gojo" in dp and "gender" in dp and "local_id" in dp
assert "dialogue" not in dp.lower() and "camera" not in dp.lower() # detect stays detect
canned = ("<|channel>thought\nrambling\n<channel|>"
'{"skip":false,"characters":[{"local_id":"person_1","name":"","gender":"m",'
'"appearance":{"hair":"white","clothing":"coat","features":["blindfold"]},'
'"emotion":"calm","action":"standing","bbox":[10,20,100,200]}],'
'"scene":{"location":"street","time":"night"}}')
r = _extract_json(canned)
assert r["characters"][0]["bbox"] == [10, 20, 100, 200] and r["characters"][0]["gender"] == "m"
assert r["scene"]["location"] == "street"
lp = build_dialogue_prompt([{"local_id": "person_1", "name": "Gojo", "gender": "m"}],
recent=[{"speaker": "Nanami", "text": "You're late."}])
assert "person_1: Gojo (m)" in lp and "hi" in lp and "speaker" in lp
assert "Nanami" in lp and "unknown" in lp and "off-panel" in lp # cross-panel attribution wired
# bubble-only panel: no present cast still yields a usable prompt (off-panel/someone path)
assert "(none detected)" in build_dialogue_prompt([])
rd = _extract_json('{"dialogue":[{"speaker":"person_1","type":"thought","text":"hm"}],'
'"entities":[{"name":"Everyday","kind":"shop"}]}')
assert rd["dialogue"][0]["type"] == "thought" and rd["entities"][0]["name"] == "Everyday"
# solo backstop (p109): one present char + an unattributed spoken line -> assigned to them.
solo = [{"local_id": "person_1"}]
dlg = resolve_speakers([{"speaker": "unknown", "type": "speech", "text": "you're kidding me"},
{"speaker": "", "type": "shout", "text": "ms. haeseon"},
{"speaker": "", "type": "narration", "text": "years passed"},
{"speaker": "person_1", "type": "speech", "text": "hi"}], solo)
assert dlg[0]["speaker"] == "person_1" and dlg[1]["speaker"] == "person_1" # unknown+"" -> solo
assert dlg[0]["speaker_method"] == "solo_prior" and dlg[0]["confidence"] == 0.7
assert dlg[2]["speaker"] == "" and dlg[3]["speaker"] == "person_1" # narration & set stay
# 2+ present or 0 present -> never guess (a wrong face is worse than "someone")
two = [{"local_id": "person_1"}, {"local_id": "person_2"}]
assert resolve_speakers([{"speaker": "unknown", "type": "speech", "text": "x"}], two)[0]["speaker"] == "unknown"
assert resolve_speakers([{"speaker": "unknown", "type": "speech", "text": "x"}], [])[0]["speaker"] == "unknown"
# a present-local_id match is gemma's guess, not a read tail: dropped when 2+ are present,
# kept but capped at 0.7 when only one is (caveats/speaker-attribution.md#tail-is-not-geometry)
crowd = resolve_speakers([{"speaker": "person_1", "type": "speech", "text": "x", "confidence": 1.0}], two)[0]
assert crowd["speaker"] == "unknown" and crowd["speaker_method"] == "unknown" and crowd["confidence"] == 0.0
lone = resolve_speakers([{"speaker": "person_1", "type": "speech", "text": "x", "confidence": 1.0}], solo)[0]
assert lone["speaker"] == "person_1" and lone["speaker_method"] == "model_solo" and lone["confidence"] == 0.7
# set-of-mark: gemma answers a face label -> remapped to local_id; a name/unknown passes through
lbl = _apply_speaker_labels([{"speaker": "P1"}, {"speaker": "Aria"}, {"speaker": "unknown"}],
{"P1": "person_3"})
assert lbl[0]["speaker"] == "person_3" and lbl[1]["speaker"] == "Aria" and lbl[2]["speaker"] == "unknown"
assert lbl[0]["speaker_method"] == "som_face"
# a mark whose face paired to nobody present grounds nothing, so it gets no som_face label
unpaired = _apply_speaker_labels([{"speaker": "P2"}], {"P2": "unknown"})[0]
assert unpaired["speaker"] == "unknown" and "speaker_method" not in unpaired, unpaired
# gemma answers with what the prompt SHOWED, not the id: description, name+gender, bare id, junk id.
# Each shape cost real lines on job 778297bc by falling through as a free-form name.
shown = [{"local_id": "person_1", "desc": "brown ponytail, green dress", "gender": "f"},
{"local_id": "person_2", "name": "Seonho", "gender": "m"}]
got = _apply_speaker_labels([{"speaker": "brown ponytail, green dress (f)"}, {"speaker": "Seonho"},
{"speaker": "person_2"}, {"speaker": "person_9"},
{"speaker": "Haeseon (f)"}, {"speaker": "unknown"}], {}, shown)
assert [d["speaker"] for d in got] == ["person_1", "person_2", "person_2", "unknown",
"Haeseon", "unknown"], got
# a description shared by two present characters identifies neither
twins = [{"local_id": "person_1", "desc": "schoolgirl"}, {"local_id": "person_2", "desc": "schoolgirl"}]
assert _apply_speaker_labels([{"speaker": "schoolgirl"}], {}, twins)[0]["speaker"] == "schoolgirl"
# a nameless present character is shown as "unknown"; that must not become an id
assert _apply_speaker_labels([{"speaker": "unknown"}], {}, [{"local_id": "person_1"}])[0]["speaker"] == "unknown"
# real detector face -> nearest present char's identity; each present char claimed once; leftover unknown
pf = _pair_faces_to_present(
[{"bbox": [10, 10, 50, 50]}, {"bbox": [200, 10, 240, 50]}, {"bbox": [400, 10, 440, 50]}],
[{"local_id": "person_2", "bbox": [190, 5, 250, 60]}, {"local_id": "person_1", "bbox": [0, 0, 60, 60]}])
assert [f["local_id"] for f in pf] == ["person_1", "person_2", None] # closest wins, 3rd face unmatched
# solo prior wired into the prompt; harmful "never default" instruction is gone
assert "EXACTLY ONE character is present" in lp and "just because they are present" not in lp
drp = build_direct_prompt("Gojo shouts a warning")
assert "camera" in drp and "transition" in drp and "Gojo shouts a warning" in drp
rr = _extract_json('{"camera":{"effect":"shake"},"transition":"fade_white"}')
assert rr["camera"]["effect"] == "shake" and rr["transition"] == "fade_white"
# A transport-window edge is not a story boundary: explicitly carry continuity from prev_beat.
ns, cp = _parse_direct_groups(
{"continues_previous": True,
"groups": [{"panels": ["p9", "p10"], "why": "same conversation"}]},
["p9", "p10"], has_previous=True)
assert cp is True and ns == {"p9": False, "p10": False}
# The model cannot continue before the chapter begins, even if it emits a malformed true value.
ns0, cp0 = _parse_direct_groups(
{"continues_previous": True, "groups": [{"panels": ["p1"]}]}, ["p1"], has_previous=False)
assert cp0 is False and ns0["p1"] is True
rs = _extract_json('{"same":true,"confidence":0.9,"reason":"same face+hair"}')
assert rs["same"] is True and rs["confidence"] == 0.9
# tier-2 resolve: sheet is compact + candidates are numbered; choice maps 1-based -> id, 0 -> NONE.
cands = [{"character_id": "c1", "name": "Gojo", "gender": "m",
"appearance": {"hair": "white", "clothing": "coat", "features": ["blindfold"]}},
{"character_id": "c2", "name": "Choi Haeseon", "gender": "f", "description": "black hair"}]
rp = build_resolve_prompt(cands)
assert "1. Gojo" in rp and "2. Choi Haeseon" in rp and "white" in rp and "NONE" in rp
_map = lambda ch: (cands[ch - 1]["character_id"] if isinstance(ch, int) and 1 <= ch <= len(cands) else None)
assert _map(1) == "c1" and _map(2) == "c2" and _map(0) is None and _map(9) is None
assert "reference images" in rp and "face shape/features first" in rp
# reference budget: a small gallery keeps 3 refs each, a cast-sized one drops to 1 and stays under
# REF_BUDGET images however big the cast gets.
assert _refs_per_candidate(1) == 3 and _refs_per_candidate(4) == 3
assert _refs_per_candidate(9) == 1 and _refs_per_candidate(0) == 3
# the budget holds up to REF_BUDGET candidates; past that the floor of 1 apiece wins, which is why
# the orchestrator caps the gallery (GALLERY_CAP) rather than relying on this alone.
assert all(n * _refs_per_candidate(n) <= REF_BUDGET for n in range(1, REF_BUDGET + 1))
# dialogue parsing is fail-loud; an omitted requested panel is partial, never silent-empty.
bad = _dialogue_envelope(["p1"], [], parse_failed=True)
assert bad["status"] == "failed" and bad["parse_failed"]
partial = _dialogue_envelope(["p1", "p2"], [{"panel_id": "p1"}])
assert partial["status"] == "partial" and "p2" in " ".join(partial["warnings"])
claims = _normalize_claims([
{"target_local_id": "person_2", "name": "Seonho", "evidence_type": "address", "confidence": .8},
{"target_local_id": "person_1", "name": "Mina", "evidence_type": "self_intro", "confidence": .95},
], "p7")
assert claims[0]["evidence_type"] == "address" and claims[0]["target_local_id"] == "person_2"
assert claims[1]["evidence_type"] == "self_intro" and claims[1]["target_local_id"] == "person_1"
same = _normalize_claims([{"target_local_id": "person_2", "name": "Seonho",
"evidence_type": "address", "confidence": .8}], "p7")
assert claims[0]["claim_id"] == same[0]["claim_id"]
# truncated json -> raises (caught by the endpoints as a skip/default)
try:
_extract_json('{"characters":[{"local_id":"person_1"')
assert False
except ValueError:
pass
# trailing braced prose after a complete object parses (used to burn a repair call)
assert _extract_json('{"skip":false}\nnote {see above}')["skip"] is False
# gemma's boxes arrive on a 0-1000 grid and leave /vision as pixels. The 900x1650 panel below is
# panel 7 of job 778297bc: person_5 is Seonho in the foreground, and read as pixels his box lands in
# the top sixth of the panel, inside a speech balloon, which is what identity embedded.
_ch = [{"local_id": "person_5", "bbox": [222, 405, 654, 1000]},
{"local_id": "edge", "bbox": [0, 0, 1000, 1000]},
{"local_id": "junk", "bbox": "nope"},
# p007 person_1 of run B: x1 > x2, so the crop came out empty and the detection was lost
{"local_id": "swapped", "bbox": [226, 417, 130, 551]}]
_bbox_to_pixels(_ch, 900, 1650)
assert _ch[3]["bbox"] == [117, 688, 203, 909], _ch[3]["bbox"] # corners ordered, non-empty crop
assert _ch[0]["bbox"] == [200, 668, 589, 1650], _ch[0]["bbox"]
assert _ch[1]["bbox"] == [0, 0, 900, 1650], _ch[1]["bbox"] # a clamped box spans the whole panel
assert _ch[2]["bbox"] == "nope", _ch[2]["bbox"] # unparseable is left alone, not crashed
# the box must now cover the lower half of a tall panel, which the raw grid value never can
assert _ch[0]["bbox"][3] > 1000 > _ch[0]["bbox"][1]
# the has_face gate, on panel 7's real converted boxes. Seonho and the colleague carry a detected
# face; the two background extras do not, and identity must skip them rather than embed scenery.
_p7 = [{"local_id": "person_1", "bbox": [457, 657, 642, 937]}, # Seonho, foreground
{"local_id": "person_2", "bbox": [669, 591, 763, 822]}, # the colleague
{"local_id": "person_5", "bbox": [31, 554, 94, 728]}, # extra that took Seonho at 1.00
{"local_id": "person_6", "bbox": [34, 414, 122, 564]}] # extra
_faces = [{"bbox": [520, 690, 590, 770]}, {"bbox": [695, 610, 740, 660]}]
class _FD:
@staticmethod
def detect_faces(_img):
return _faces
_real_fd, globals()["face_detect"] = face_detect, _FD
try:
_mark_has_face(None, _p7)
assert [c["has_face"] for c in _p7] == [True, True, False, False], _p7
# fail open: a detector that raises must not silently drop every character in the panel
class _Boom:
@staticmethod
def detect_faces(_img):
raise RuntimeError("onnx died")
globals()["face_detect"] = _Boom
_mark_has_face(None, _p7)
assert all(c["has_face"] for c in _p7), _p7
globals()["face_detect"] = None
_mark_has_face(None, _p7)
assert all(c["has_face"] for c in _p7), _p7
finally:
globals()["face_detect"] = _real_fd
# face->identity pairing is GATED on containment: a face outside every gemma bbox stays unknown.
faces = [{"bbox": [10, 10, 30, 30]}, {"bbox": [900, 900, 920, 920]}]
present = [{"local_id": "person_1", "name": "Teto", "bbox": [0, 0, 100, 200]}]
paired = _pair_faces_to_present(faces, present)
assert paired[0]["local_id"] == "person_1" and paired[0]["who"] == "Teto"
assert paired[1]["local_id"] is None and paired[1]["who"] == "unknown", paired[1]
# globally shortest-first: the first face must not claim a character that fits the second better.
faces2 = [{"bbox": [95, 95, 105, 105]}, {"bbox": [8, 8, 12, 12]}]
present2 = [{"local_id": "a", "bbox": [0, 0, 20, 20]}, {"local_id": "b", "bbox": [80, 80, 120, 120]}]
p2 = _pair_faces_to_present(faces2, present2)
assert [f["local_id"] for f in p2] == ["b", "a"], p2
# no present characters at all -> every face unknown, never a phantom identity
assert _pair_faces_to_present(faces, [])[0]["local_id"] is None
print("worker_vision self-check ok")