54c456801a
Panels, wavs, layers, clips, and the chapter mp4 leave the `manga` bucket for `panels`, `audio`, `layers`, and `video`. The key under the bucket is unchanged, so every reader that derives the bucket from the first path segment keeps working. The orchestrator half moves in the same commit, per invariant 7. The 2026-08-11 chapter run proves the split for `raw` and `panels` and produced the first quality read on speaker attribution, which is wrong in every sampled multi-character panel. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01XD7cAy81MZrc7gCr6aZGWr
360 lines
16 KiB
Python
360 lines
16 KiB
Python
# worker_crop.py — stage 2 panel detector. FastAPI :8000. cpu/opencv, no GPU, no session.
|
|
# one page in -> N panel crops (reading order) uploaded to minio, uris + bboxes returned.
|
|
# routes by aspect: framed pages -> kumiko; tall webtoon strips -> whitespace slicer.
|
|
import sys, os, uuid, logging
|
|
from fastapi import FastAPI, HTTPException
|
|
from pydantic import BaseModel
|
|
import cv2
|
|
import numpy as np
|
|
import transport
|
|
try: # CPU-EP ONNX detectors; absent -> merge pass is a no-op
|
|
import face_detect, bubble_detect
|
|
except Exception:
|
|
face_detect = bubble_detect = None
|
|
|
|
log = logging.getLogger("crop")
|
|
|
|
sys.path.insert(0, os.path.expanduser("~/Programs/kumiko"))
|
|
|
|
app = FastAPI()
|
|
transport.install_logging(app, "crop")
|
|
SHM = "/dev/shm"
|
|
WEBTOON_RATIO = 2.0 # ponytail: manga ~1.4, webtoons 2.5-10+; tune if a tall page misroutes
|
|
|
|
|
|
class CropInput(BaseModel):
|
|
page_uri: str
|
|
manga_id: str
|
|
chapter_id: str
|
|
page_index: int = 0
|
|
job_id: str = ""
|
|
rtl: bool = True # manga reads right-to-left; set false for western comics/webtoons
|
|
|
|
|
|
class WebtoonInput(BaseModel):
|
|
# webtoon scrapers deliver one episode as arbitrary fixed-height tiles cut mid-panel.
|
|
# restitch the whole chapter into one strip, THEN slice — per-tile slicing severs panels.
|
|
page_uris: list # all tiles of the chapter, in order
|
|
manga_id: str
|
|
chapter_id: str
|
|
job_id: str = ""
|
|
|
|
|
|
def slice_webtoon(img, bg_thresh=235, min_gap=20, min_seg=64, max_seg=2500, blank_frac=0.995):
|
|
"""Cut a vertical strip on blank row-bands. Returns (crop, bbox[x,y,w,h]) top-to-bottom.
|
|
Defaults tuned on real webtoon strips: 235/20 finds true gutters over full-color art;
|
|
max_seg=2500 (~one phone screen) caps full-bleed art that has no internal gutter.
|
|
186: a row is 'blank' when >=blank_frac of its pixels are bright — NOT every pixel (min>=thresh),
|
|
because one dark speck/border pixel/stray letter in a true gutter used to defeat the whole cut."""
|
|
gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) if img.ndim == 3 else img
|
|
row_blank = (gray >= bg_thresh).mean(axis=1) >= blank_frac
|
|
h, w = gray.shape[:2]
|
|
cuts, i = [0], 0
|
|
while i < h:
|
|
if row_blank[i]:
|
|
j = i
|
|
while j < h and row_blank[j]:
|
|
j += 1
|
|
if j - i >= min_gap:
|
|
cuts.append((i + j) // 2)
|
|
i = j
|
|
else:
|
|
i += 1
|
|
cuts.append(h)
|
|
out = []
|
|
for a, b in zip(cuts, cuts[1:]):
|
|
if b - a < min_seg:
|
|
continue
|
|
n = -(-(b - a) // max_seg)
|
|
step = (b - a) // n
|
|
for s in range(n):
|
|
y0 = a + s * step
|
|
y1 = b if s == n - 1 else a + (s + 1) * step
|
|
out.append((img[y0:y1], [0, y0, w, y1 - y0]))
|
|
return out or [(img, [0, 0, w, h])]
|
|
|
|
|
|
def _has_face(img):
|
|
try:
|
|
return bool(face_detect.detect_faces(img))
|
|
except Exception:
|
|
return True # detector broke -> assume a face so we do NOT merge (conservative)
|
|
|
|
|
|
def _has_text(img):
|
|
try:
|
|
return bool(bubble_detect.detect_text_regions(img))
|
|
except Exception:
|
|
return False
|
|
|
|
|
|
def _merge_plan(faces, texts):
|
|
"""Pure grouping decision (testable without a real face). faces/texts: per-fragment booleans,
|
|
top-to-bottom. A 'stranded caption' = has text but no face; its line has no face to attribute to.
|
|
Attach it to an adjacent face fragment — preferring the face ABOVE (the speaker usually appears,
|
|
then their caption follows), else the face below. Scenery (no text) and face fragments stay solo.
|
|
Returns a list of index-groups covering 0..n-1 in order; each group with >1 index gets vstacked."""
|
|
n = len(faces)
|
|
groups, i = [], 0
|
|
while i < n:
|
|
if faces[i]: # a face swallows the run of stranded captions below it
|
|
g = [i]
|
|
while i + 1 < n and texts[i + 1] and not faces[i + 1]:
|
|
i += 1
|
|
g.append(i)
|
|
groups.append(g)
|
|
elif texts[i] and not faces[i] and i + 1 < n and faces[i + 1]: # no face above -> ride the one below
|
|
groups.append([i, i + 1])
|
|
i += 1
|
|
else:
|
|
groups.append([i])
|
|
i += 1
|
|
return groups
|
|
|
|
|
|
def merge_faceless_captions(crops):
|
|
"""Post-slice pass (webtoon only): fold a stranded caption fragment into the vertically nearest
|
|
face-bearing fragment so the windowed vision/dialogue call sees the caption next to a real face,
|
|
instead of on a faceless panel where attribution has nothing to anchor to. This ONLY combines
|
|
images — it never assigns a speaker, so it can't misbind a monologue to the nearer face; gemma
|
|
still decides from the window's flow. Bonus: fewer panels -> fits gemma's window-count limit.
|
|
ponytail: runs face+text ONNX on every fragment (CPU); fine for batch, cache the models if a
|
|
huge chapter makes crop the bottleneck."""
|
|
if not (face_detect and bubble_detect) or len(crops) < 2:
|
|
return crops
|
|
faces = [_has_face(c) for c, _ in crops]
|
|
texts = [_has_text(c) for c, _ in crops]
|
|
plan = _merge_plan(faces, texts)
|
|
if len(plan) == len(crops):
|
|
return crops # nothing stranded -> untouched
|
|
out = []
|
|
for g in plan:
|
|
if len(g) == 1:
|
|
out.append(crops[g[0]])
|
|
continue
|
|
merged = np.vstack([crops[k][0] for k in g])
|
|
x, y, w, _ = crops[g[0]][1]
|
|
out.append((merged, [x, y, w, merged.shape[0]]))
|
|
return out
|
|
|
|
|
|
def context_fragment_links(crops):
|
|
"""Return non-destructive caption↔face links keyed by source-fragment index.
|
|
|
|
Both fragments remain independent panels with their original bboxes; the dialogue stage may use
|
|
the linked image as context, while the orchestrator remains free to accept/reject the linkage.
|
|
"""
|
|
links = {i: [] for i in range(len(crops))}
|
|
if not (face_detect and bubble_detect) or len(crops) < 2:
|
|
return links
|
|
faces = [_has_face(c) for c, _ in crops]
|
|
texts = [_has_text(c) for c, _ in crops]
|
|
for group in _merge_plan(faces, texts):
|
|
face_idxs = [i for i in group if faces[i]]
|
|
if not face_idxs:
|
|
continue
|
|
anchor = face_idxs[0]
|
|
for i in group:
|
|
if i == anchor or not (texts[i] and not faces[i]):
|
|
continue
|
|
links[i].append({"fragment_index": anchor, "bbox": crops[anchor][1],
|
|
"link_reason": "adjacent_face_context"})
|
|
links[anchor].append({"fragment_index": i, "bbox": crops[i][1],
|
|
"link_reason": "adjacent_text_context"})
|
|
return links
|
|
|
|
|
|
def reading_order(panels, rtl=True):
|
|
"""186: deterministic manga/comic reading order, replacing reliance on kumiko's neighbour-based
|
|
comparator (which emitted this RTL title left-to-right on irregular/overlapping layouts).
|
|
Group panels into horizontal row-bands by vertical overlap, order bands top-to-bottom, then within
|
|
a band order by x — right-to-left for rtl manga, left-to-right otherwise.
|
|
panels: [(crop, [x,y,w,h])]. ponytail: greedy vertical banding; a deeply interleaved splash
|
|
collage can still misband — upgrade to a full topological pass if a fixture proves it needed."""
|
|
if not panels:
|
|
return panels
|
|
bands = [] # each: {"y0","y1","items"}
|
|
for it in sorted(panels, key=lambda p: p[1][1]): # seed bands top-down
|
|
x, y, w, h = it[1]
|
|
for band in bands:
|
|
ov = max(0, min(y + h, band["y1"]) - max(y, band["y0"]))
|
|
if ov >= 0.5 * min(h, band["y1"] - band["y0"]): # substantial vertical overlap = same row
|
|
band["items"].append(it)
|
|
band["y0"], band["y1"] = min(band["y0"], y), max(band["y1"], y + h)
|
|
break
|
|
else:
|
|
bands.append({"y0": y, "y1": y + h, "items": [it]})
|
|
bands.sort(key=lambda b: b["y0"])
|
|
out = []
|
|
for band in bands:
|
|
band["items"].sort(key=lambda p: p[1][0], reverse=rtl) # x; rtl -> rightmost first
|
|
out.extend(band["items"])
|
|
return out
|
|
|
|
|
|
def kumiko_panels(path, rtl=True):
|
|
"""Framed-page panels via kumiko contour detection, then a DETERMINISTIC reading-order sort
|
|
(186 — do not trust kumiko's implicit comparator for RTL/irregular layouts)."""
|
|
from kumikolib import Kumiko
|
|
k = Kumiko({"rtl": rtl})
|
|
k.parse_image(path)
|
|
page = k.page_list[-1]
|
|
full = page.img
|
|
panels = [(full[p.y:p.b, p.x:p.r], [p.x, p.y, p.r - p.x, p.b - p.y]) for p in page.panels]
|
|
return reading_order(panels, rtl)
|
|
|
|
|
|
def flag_overlaps(panels, page_index=0):
|
|
"""Warn on panel pairs whose bboxes overlap by >50% of the smaller panel's area —
|
|
a sign of ambiguous splitting/ordering that silently corrupts narration sequence.
|
|
Returns the list of ambiguous (i, j) index pairs (also for the self-check)."""
|
|
def _ov(a, b):
|
|
ax, ay, aw, ah = a; bx, by, bw, bh = b
|
|
ix = max(0, min(ax + aw, bx + bw) - max(ax, bx))
|
|
iy = max(0, min(ay + ah, by + bh) - max(ay, by))
|
|
inter = ix * iy
|
|
return inter / max(1, min(aw * ah, bw * bh))
|
|
ambiguous = []
|
|
for i in range(len(panels)):
|
|
for j in range(i + 1, len(panels)):
|
|
if _ov(panels[i][1], panels[j][1]) > 0.5:
|
|
ambiguous.append((i, j))
|
|
if ambiguous:
|
|
log.warning("page=%d ambiguous panel overlap, order may be wrong: pairs=%s",
|
|
page_index, ambiguous)
|
|
return ambiguous
|
|
|
|
|
|
def restitch(paths):
|
|
"""Stack chapter tiles into one vertical strip. Tiles share width (scraper output)."""
|
|
imgs = [cv2.imread(p) for p in paths]
|
|
if any(i is None for i in imgs):
|
|
raise HTTPException(400, "a webtoon tile was not readable")
|
|
w = min(i.shape[1] for i in imgs)
|
|
imgs = [i if i.shape[1] == w else cv2.resize(i, (w, int(i.shape[0] * w / i.shape[1]))) for i in imgs]
|
|
return np.vstack(imgs)
|
|
|
|
|
|
@app.post("/crop/webtoon")
|
|
async def crop_webtoon(data: WebtoonInput):
|
|
tag = uuid.uuid4().hex[:8]
|
|
locals_ = [transport.get(u, f"{SHM}/wt_{tag}_{i:04d}.png") for i, u in enumerate(data.page_uris)]
|
|
strip = restitch(locals_)
|
|
crops = slice_webtoon(strip)
|
|
context_links = context_fragment_links(crops)
|
|
panels = []
|
|
for idx, (crop_img, bbox) in enumerate(crops):
|
|
uri = f"s3://panels/{data.manga_id}/{data.chapter_id}/panels/p{idx:03d}.png"
|
|
# slicing is deterministic, so on a resume the same idx -> same key; skip re-upload.
|
|
if not transport.exists(uri):
|
|
out = f"{SHM}/wt_{tag}_p{idx:03d}.png"
|
|
cv2.imwrite(out, crop_img)
|
|
transport.put(out, uri)
|
|
os.remove(out)
|
|
panels.append({"panel_index": idx, "uri": uri, "bbox": bbox,
|
|
"context_fragments": context_links[idx]})
|
|
for p in locals_:
|
|
os.remove(p)
|
|
return {"page_index": 0, "panels": panels}
|
|
|
|
|
|
@app.post("/crop")
|
|
async def crop(data: CropInput):
|
|
local = transport.get(data.page_uri, f"{SHM}/crop_{uuid.uuid4().hex[:8]}.png")
|
|
img = cv2.imread(local)
|
|
if img is None:
|
|
raise HTTPException(400, f"page not readable: {data.page_uri}")
|
|
h, w = img.shape[:2]
|
|
webtoon = h / w >= WEBTOON_RATIO
|
|
crops = slice_webtoon(img) if webtoon else kumiko_panels(local, data.rtl)
|
|
context_links = context_fragment_links(crops) if webtoon else {i: [] for i in range(len(crops))}
|
|
ambiguous = flag_overlaps(crops, data.page_index)
|
|
|
|
panels = []
|
|
for idx, (crop_img, bbox) in enumerate(crops):
|
|
out = f"{SHM}/pg{data.page_index:03d}_p{idx:02d}.png"
|
|
cv2.imwrite(out, crop_img)
|
|
uri = f"s3://panels/{data.manga_id}/{data.chapter_id}/panels/pg{data.page_index:03d}_p{idx:02d}.png"
|
|
transport.put(out, uri)
|
|
os.remove(out)
|
|
panels.append({"panel_index": idx, "uri": uri, "bbox": bbox,
|
|
"context_fragments": context_links[idx]})
|
|
os.remove(local)
|
|
# 186: surface anomalies so review can see a suspect page instead of it silently completing.
|
|
warnings = [f"ambiguous overlap {p}" for p in ambiguous]
|
|
if not panels:
|
|
warnings.append("no panels detected")
|
|
return {"page_index": data.page_index, "panels": panels, "warnings": warnings}
|
|
|
|
|
|
@app.get("/health")
|
|
async def health():
|
|
return {"status": "ok"}
|
|
|
|
|
|
if __name__ == "__main__":
|
|
# self-check: synthetic 2-band strip -> 2 panels top-to-bottom with sane bboxes.
|
|
strip = np.full((300, 100, 3), 255, np.uint8)
|
|
strip[50:100] = 0
|
|
strip[200:250] = 0
|
|
segs = slice_webtoon(strip)
|
|
assert len(segs) == 2, len(segs)
|
|
assert segs[0][1][1] < segs[1][1][1], "reading order top-to-bottom"
|
|
assert all(len(b) == 4 for _, b in segs)
|
|
|
|
# max_seg cap: a 6000px gutterless block must split into ceil(6000/2500)=3 beats
|
|
solid = np.zeros((6000, 100, 3), np.uint8)
|
|
capped = slice_webtoon(solid)
|
|
assert len(capped) == 3, len(capped)
|
|
|
|
# a true gutter with one dark speck must still cut (186: fraction, not min-pixel).
|
|
speckled = np.full((300, 100, 3), 255, np.uint8)
|
|
speckled[50:100] = 0; speckled[200:250] = 0
|
|
speckled[75, 40] = 0 # stray dark pixel inside the top band — used to defeat min()>=thresh
|
|
assert len(slice_webtoon(speckled)) == 2, "gutter detection must tolerate a speck"
|
|
|
|
# faceless-caption merge plan (pure grouping). F=face T=text per fragment, top-to-bottom.
|
|
assert _merge_plan([True, False], [False, True]) == [[0, 1]] # caption below a face -> absorbed up
|
|
assert _merge_plan([False, True], [True, False]) == [[0, 1]] # caption above a face -> absorbed down
|
|
assert _merge_plan([False], [True]) == [[0]] # lone caption, no face -> left alone
|
|
assert _merge_plan([False], [False]) == [[0]] # scenery (no text) -> left alone
|
|
assert _merge_plan([True, False, False], [False, True, True]) == [[0, 1, 2]] # face swallows caption run
|
|
assert _merge_plan([True, True], [False, False]) == [[0], [1]] # two faces -> untouched
|
|
# vstack path: a real merge stacks images and reports the combined height
|
|
a = (np.zeros((30, 10, 3), np.uint8), [0, 0, 10, 30])
|
|
b = (np.zeros((20, 10, 3), np.uint8), [0, 40, 10, 20])
|
|
import types
|
|
_fd, _bd = face_detect, bubble_detect
|
|
face_detect = types.SimpleNamespace(detect_faces=lambda im: [1] if im.shape[0] == 30 else [])
|
|
bubble_detect = types.SimpleNamespace(detect_text_regions=lambda im: [1] if im.shape[0] == 20 else [])
|
|
merged = merge_faceless_captions([a, b])
|
|
assert len(merged) == 1 and merged[0][0].shape[0] == 50, merged[0][0].shape
|
|
linked = context_fragment_links([a, b])
|
|
assert linked[0][0]["fragment_index"] == 1 and linked[1][0]["fragment_index"] == 0
|
|
assert linked[0][0]["bbox"] == b[1] and linked[1][0]["bbox"] == a[1]
|
|
face_detect, bubble_detect = _fd, _bd
|
|
|
|
# overlap flag: disjoint panels are clean; a >50% overlapping pair is flagged ambiguous.
|
|
clean = [(None, [0, 0, 10, 10]), (None, [20, 0, 10, 10])]
|
|
assert flag_overlaps(clean) == []
|
|
overlap = [(None, [0, 0, 10, 10]), (None, [2, 2, 10, 10])] # ~64% of the smaller area
|
|
assert flag_overlaps(overlap) == [(0, 1)]
|
|
|
|
# 186 reading order — bbox = [x,y,w,h]. ids track source panels so we can assert final order.
|
|
def ro_ids(boxes, rtl):
|
|
tagged = [(i, b) for i, b in enumerate(boxes)]
|
|
return [i for i, _ in reading_order(tagged, rtl)]
|
|
# two panels on one row: RTL reads the right one (x=60) first, LTR the left (x=0) first.
|
|
two = [[0, 0, 50, 100], [60, 0, 50, 100]]
|
|
assert ro_ids(two, rtl=True) == [1, 0], "RTL: rightmost first"
|
|
assert ro_ids(two, rtl=False) == [0, 1], "LTR: leftmost first"
|
|
# 2x2 grid: top row R->L then bottom row R->L. sources: 0=TL 1=TR 2=BL 3=BR.
|
|
grid = [[0, 0, 50, 50], [60, 0, 50, 50], [0, 60, 50, 50], [60, 60, 50, 50]]
|
|
assert ro_ids(grid, rtl=True) == [1, 0, 3, 2], ro_ids(grid, rtl=True)
|
|
# staggered rows (slightly offset y) still band by vertical overlap, not exact y.
|
|
stag = [[0, 0, 50, 100], [60, 5, 50, 100], [0, 200, 100, 80]]
|
|
assert ro_ids(stag, rtl=True) == [1, 0, 2], ro_ids(stag, rtl=True)
|
|
# splash (single full panel) -> unchanged.
|
|
assert ro_ids([[0, 0, 200, 300]], rtl=True) == [0]
|
|
print("worker_crop self-check ok")
|