#!/usr/bin/env python3 """Manifest-driven Live2D-style raster parts harness. The layout JSON is the single source of truth. This script can: * render a slot template for the image-generation prompt, * write a prompt that names every required part and slot, * extract full-canvas transparent PNG layers from a generated sheet, * render offline recomposition previews without touching app public assets. """ from __future__ import annotations import argparse import hashlib import json import shutil import subprocess from pathlib import Path from typing import Any, Iterable import numpy as np from PIL import Image, ImageDraw, ImageFilter, ImageFont from scipy import ndimage as ndi HERE = Path(__file__).resolve().parent LAYOUT = HERE / "layout-v3.json" OUT = HERE / "parts" AI_WORK = HERE / "ai-cutout-work" OBJECT_SEPARATE_PY = Path("C:/Users/encep/.agents/skills/object-separation/scripts/separate_object.py") OBJECT_SEPARATE_VENV_PY = Path("C:/Users/encep/.venvs/object-separation/Scripts/python.exe") Box = tuple[int, int, int, int] Inset = tuple[int, int, int, int] def load_layout() -> dict[str, Any]: return json.loads(LAYOUT.read_text(encoding="utf-8")) def as_box(value: list[int] | tuple[int, int, int, int]) -> Box: return int(value[0]), int(value[1]), int(value[2]), int(value[3]) def as_inset(value: int | list[int] | tuple[int, int, int, int] | None) -> Inset: if value is None: return 0, 0, 0, 0 if isinstance(value, int): return value, value, value, value return int(value[0]), int(value[1]), int(value[2]), int(value[3]) def inset_box(box: Box, inset: Inset) -> Box: x1, y1, x2, y2 = box left, top, right, bottom = inset nx1, ny1 = x1 + left, y1 + top nx2, ny2 = x2 - right, y2 - bottom if nx2 <= nx1 + 2 or ny2 <= ny1 + 2: return box return nx1, ny1, nx2, ny2 def slot_by_id(layout: dict[str, Any]) -> dict[str, dict[str, Any]]: return {slot["id"]: slot for slot in layout["slots"]} def alpha_bbox(im: Image.Image, threshold: int = 8) -> Box | None: alpha = np.array(im.getchannel("A")) ys, xs = np.where(alpha > threshold) if len(xs) == 0: return None return int(xs.min()), int(ys.min()), int(xs.max() + 1), int(ys.max() + 1) def remove_white_bg(im: Image.Image, *, lo: float = 10.0, hi: float = 58.0) -> Image.Image: arr = np.array(im.convert("RGBA")).astype(np.float32) rgb = arr[:, :, :3] border = np.concatenate( [ rgb[:6, :, :].reshape(-1, 3), rgb[-6:, :, :].reshape(-1, 3), rgb[:, :6, :].reshape(-1, 3), rgb[:, -6:, :].reshape(-1, 3), ], axis=0, ) bg = np.median(border, axis=0) dist_bg = np.sqrt(((rgb - bg.reshape(1, 1, 3)) ** 2).sum(axis=2)) dist_white = np.sqrt(((255.0 - rgb) ** 2).sum(axis=2)) alpha = np.minimum( np.clip((dist_bg - lo) * 255.0 / max(1.0, hi - lo), 0, 255), np.clip((dist_white - 12.0) * 255.0 / 48.0, 0, 255), ) luma = rgb[:, :, 0] * 0.2126 + rgb[:, :, 1] * 0.7152 + rgb[:, :, 2] * 0.0722 chroma = rgb.max(axis=2) - rgb.min(axis=2) neutral_artifact = (luma > 175) & (chroma < 20) & (dist_white < 110) alpha = np.where(neutral_artifact, 0, alpha) keep = ((luma < 225) | (chroma > 14)) & ~neutral_artifact alpha = np.where(keep & (alpha > 28), np.maximum(alpha, 235), alpha) arr[:, :, 3] = np.minimum(alpha, arr[:, :, 3]) arr[arr[:, :, 3] < 1, :3] = 0 return Image.fromarray(np.clip(arr, 0, 255).astype(np.uint8), "RGBA") def soft_detail_alpha(im: Image.Image, *, lo: float = 2.0, hi: float = 36.0) -> Image.Image: arr = np.array(im.convert("RGBA")).astype(np.float32) rgb = arr[:, :, :3] border = np.concatenate( [ rgb[:4, :, :].reshape(-1, 3), rgb[-4:, :, :].reshape(-1, 3), rgb[:, :4, :].reshape(-1, 3), rgb[:, -4:, :].reshape(-1, 3), ], axis=0, ) bg = np.median(border, axis=0) dist_bg = np.sqrt(((rgb - bg.reshape(1, 1, 3)) ** 2).sum(axis=2)) alpha = np.clip((dist_bg - lo) * 255.0 / max(1.0, hi - lo), 0, 255) arr[:, :, 3] = np.minimum(alpha, arr[:, :, 3]) arr[arr[:, :, 3] < 1, :3] = 0 return Image.fromarray(np.clip(arr, 0, 255).astype(np.uint8), "RGBA") def ai_cutout(im: Image.Image, slot: dict[str, Any], defaults: dict[str, Any]) -> Image.Image: AI_WORK.mkdir(parents=True, exist_ok=True) safe_id = slot["id"].replace(".", "-") src = AI_WORK / f"{safe_id}-input.png" out = AI_WORK / f"{safe_id}-cutout.png" meta = AI_WORK / f"{safe_id}.sha256" im.convert("RGB").save(src, optimize=True) digest = hashlib.sha256(src.read_bytes()).hexdigest() if out.exists() and meta.exists() and meta.read_text(encoding="utf-8") == digest: return Image.open(out).convert("RGBA") backend = slot.get("aiBackend", defaults.get("aiBackend", "ben2")) object_separate = shutil.which("object-separate") if object_separate: cmd = [object_separate, str(src), str(out), "--backend", str(backend)] else: cmd = [str(OBJECT_SEPARATE_VENV_PY), str(OBJECT_SEPARATE_PY), str(src), str(out), "--backend", str(backend)] if slot.get("aiRefine", defaults.get("aiRefine", True)): cmd.append("--refine") result = subprocess.run(cmd, cwd=HERE, text=True, capture_output=True) if result.returncode != 0: print(f"ai-cutout failed for {slot['id']}; falling back to matte") if result.stderr: print(result.stderr.strip()) return remove_white_bg(im) meta.write_text(digest, encoding="utf-8") return Image.open(out).convert("RGBA") def neutralize_crop_frame(im: Image.Image, px: int) -> Image.Image: """Erase atlas slot borders before alpha extraction. The generated sheet obeys the requested slot layout, so its faint rectangle guides are useful for coordinates but must never become character pixels. """ if px <= 0: return im arr = np.array(im.convert("RGBA")) px = min(px, max(0, arr.shape[0] // 3), max(0, arr.shape[1] // 3)) if px <= 0: return im arr[:px, :, :3] = 255 arr[-px:, :, :3] = 255 arr[:, :px, :3] = 255 arr[:, -px:, :3] = 255 arr[:, :, 3] = 255 return Image.fromarray(arr, "RGBA") def dark_alpha(im: Image.Image) -> Image.Image: arr = np.array(im.convert("RGBA")).astype(np.float32) rgb = arr[:, :, :3] luma = rgb[:, :, 0] * 0.2126 + rgb[:, :, 1] * 0.7152 + rgb[:, :, 2] * 0.0722 chroma = rgb.max(axis=2) - rgb.min(axis=2) alpha = np.clip((210.0 - luma) * 255.0 / 150.0, 0, 255) alpha = np.where((luma < 210) | (chroma > 18), alpha, 0) arr[:, :, 3] = np.minimum(alpha, arr[:, :, 3]) arr[arr[:, :, 3] < 1, :3] = 0 return Image.fromarray(np.clip(arr, 0, 255).astype(np.uint8), "RGBA") def shape_mask(size: tuple[int, int], shape: str) -> Image.Image: w, h = size mask = Image.new("L", size, 0) draw = ImageDraw.Draw(mask) if shape == "circle": draw.ellipse((1, 1, w - 2, h - 2), fill=255) elif shape == "eye": # Almond-like mask. This deliberately preserves white sclera pixels that # ordinary white-background removal would erase. pts = [ (1, h // 2), (w // 5, h // 5), (w // 2, 2), (w * 4 // 5, h // 5), (w - 2, h // 2), (w * 4 // 5, h * 4 // 5), (w // 2, h - 2), (w // 5, h * 4 // 5), ] draw.polygon(pts, fill=255) mask = mask.filter(ImageFilter.GaussianBlur(0.9)) elif shape == "neck-fill": pts = [ (w * 36 // 100, 1), (w * 64 // 100, 1), (w * 70 // 100, h * 54 // 100), (w * 94 // 100, h * 74 // 100), (w * 84 // 100, h - 2), (w * 16 // 100, h - 2), (w * 6 // 100, h * 74 // 100), (w * 30 // 100, h * 54 // 100), ] draw.polygon(pts, fill=255) mask = mask.filter(ImageFilter.GaussianBlur(1.2)) elif shape == "neck-gap": pts = [ (w * 36 // 100, 1), (w * 64 // 100, 1), (w * 67 // 100, h - 2), (w * 33 // 100, h - 2), ] draw.polygon(pts, fill=255) mask = mask.filter(ImageFilter.GaussianBlur(1.4)) else: draw.rounded_rectangle((1, 1, w - 2, h - 2), radius=max(2, h // 3), fill=255) return mask def apply_shape_mask(im: Image.Image, shape: str) -> Image.Image: part = im.convert("RGBA") arr = np.array(part) mask = np.array(shape_mask(part.size, shape)) arr[:, :, 3] = np.minimum(arr[:, :, 3], mask) arr[arr[:, :, 3] == 0, :3] = 0 return Image.fromarray(arr, "RGBA") def shape_crop(im: Image.Image, shape: str) -> Image.Image: return apply_shape_mask(im, shape) def filter_components(im: Image.Image, *, min_area: int = 0, keep_largest: bool = False) -> Image.Image: if min_area <= 0 and not keep_largest: return im arr = np.array(im.convert("RGBA")) alpha = arr[:, :, 3] > 8 labels, count = ndi.label(alpha) if count == 0: return im areas = np.bincount(labels.reshape(-1)) areas[0] = 0 if keep_largest: keep = labels == int(areas.argmax()) else: keep_labels = np.where(areas >= min_area)[0] keep = np.isin(labels, keep_labels) arr[:, :, 3] = np.where(keep, arr[:, :, 3], 0) arr[arr[:, :, 3] == 0, :3] = 0 return Image.fromarray(arr, "RGBA") def decontaminate_edges(im: Image.Image, *, opaque_threshold: int = 210) -> Image.Image: arr = np.array(im.convert("RGBA")) alpha = arr[:, :, 3] opaque = alpha >= opaque_threshold if not opaque.any(): return im indices = ndi.distance_transform_edt(~opaque, return_distances=False, return_indices=True) fringe = (alpha > 0) & (alpha < opaque_threshold) if fringe.any(): arr[fringe, :3] = arr[indices[0][fringe], indices[1][fringe], :3] arr[alpha == 0, :3] = 0 return Image.fromarray(arr, "RGBA") def suppress_edge_artifacts( im: Image.Image, *, edge_width: int = 3, luma_min: float = 150.0, chroma_max: float = 30.0, ) -> Image.Image: if edge_width <= 0: return im arr = np.array(im.convert("RGBA")) alpha = arr[:, :, 3] mask = alpha > 8 if not mask.any(): return im eroded = ndi.binary_erosion(mask, iterations=edge_width, border_value=0) edge = mask & ~eroded rgb = arr[:, :, :3].astype(np.float32) luma = rgb[:, :, 0] * 0.2126 + rgb[:, :, 1] * 0.7152 + rgb[:, :, 2] * 0.0722 chroma = rgb.max(axis=2) - rgb.min(axis=2) artifact = edge & (luma > luma_min) & (chroma < chroma_max) if artifact.any(): arr[artifact, 3] = 0 arr[artifact, :3] = 0 return Image.fromarray(arr, "RGBA") def clean_iris_detail(im: Image.Image) -> Image.Image: arr = np.array(im.convert("RGBA")) alpha = arr[:, :, 3] if not (alpha > 8).any(): return im h, w = alpha.shape yy, xx = np.mgrid[:h, :w] cx = (w - 1) / 2.0 cy = (h - 1) / 2.0 radius = np.sqrt(((xx - cx) / max(1.0, w * 0.46)) ** 2 + ((yy - cy) / max(1.0, h * 0.46)) ** 2) rgb = arr[:, :, :3].astype(np.float32) luma = rgb[:, :, 0] * 0.2126 + rgb[:, :, 1] * 0.7152 + rgb[:, :, 2] * 0.0722 chroma = rgb.max(axis=2) - rgb.min(axis=2) pupil = (radius < 0.62) & (luma < 48) highlight = (radius < 0.85) & (luma > 235) & (chroma < 30) pupil_or_highlight = pupil | highlight arr[pupil_or_highlight, 3] = 0 arr[pupil_or_highlight, :3] = 0 return Image.fromarray(arr, "RGBA") def resize_rgba(im: Image.Image, size: tuple[int, int]) -> Image.Image: """Resize RGBA in premultiplied-alpha space to avoid bright/dark halos.""" if im.size == size: return im arr = np.array(im.convert("RGBA")).astype(np.float32) alpha = arr[:, :, 3:4] / 255.0 premultiplied = arr.copy() premultiplied[:, :, :3] *= alpha resized = Image.fromarray(np.clip(premultiplied, 0, 255).astype(np.uint8), "RGBA").resize( size, Image.Resampling.LANCZOS, ) out = np.array(resized).astype(np.float32) out_alpha = out[:, :, 3:4] / 255.0 out[:, :, :3] = np.where(out_alpha > 0.001, out[:, :, :3] / np.maximum(out_alpha, 0.001), 0) return Image.fromarray(np.clip(out, 0, 255).astype(np.uint8), "RGBA") def place(part: Image.Image, dst_box: Box, canvas_size: tuple[int, int]) -> Image.Image: x1, y1, x2, y2 = dst_box resized = resize_rgba(part, (x2 - x1, y2 - y1)) canvas = Image.new("RGBA", canvas_size, (0, 0, 0, 0)) canvas.alpha_composite(resized, (x1, y1)) return canvas def draw_template(layout: dict[str, Any], *, clean: bool) -> Path: sheet = layout["sheet"] width, height = int(sheet["width"]), int(sheet["height"]) image = Image.new("RGB", (width, height), (255, 255, 255)) draw = ImageDraw.Draw(image) colors = { "reference-only": (80, 120, 180), "base": (220, 145, 70), "body": (90, 130, 170), "hair": (95, 70, 45), "brow": (90, 70, 50), "eye-mask": (80, 150, 210), "eye-iris": (90, 90, 150), "eye-pupil": (30, 30, 30), "eye-highlight": (160, 160, 160), "eye-line": (50, 50, 50), "blink": (120, 80, 150), "face-detail": (210, 130, 90), "mouth": (200, 90, 90), } for slot in layout["slots"]: if slot.get("derived", False): continue box = as_box(slot["sheetBox"]) role = slot["role"] color = colors.get(role, (120, 120, 120)) draw.rectangle(box, outline=color, width=3) if not clean: label = f"{slot['id']}\n{box[0]},{box[1]}-{box[2]},{box[3]}" draw.multiline_text((box[0] + 6, box[1] + 6), label, fill=color) out = HERE / ("layout-template-v3-clean.png" if clean else "layout-template-v3.png") image.save(out, optimize=True) return out def build_prompt(layout: dict[str, Any]) -> str: lines: list[str] = [] sheet = layout["sheet"] character = layout["character"] lines.extend( [ "Use case: stylized-concept", "Asset type: Live2D-ready raster character parts atlas", f"Primary request: Create a {sheet['width']}x{sheet['height']} pixel white-background parts sheet for {character['name']}.", f"Subject: {character['description']}.", "Style/medium: polished soft semi-realistic anime illustration, clean raster edges, consistent lighting, consistent scale.", "Composition/framing: Place each item inside its exact assigned rectangular slot. Keep generous whitespace between slots.", "Critical Live2D constraints:", ] ) for rule in layout["rules"]: lines.append(f"- {rule}") lines.append("") lines.append("Exact slot map. Put only the named item in that rectangle:") for slot in layout["slots"]: if slot.get("derived", False): continue box = as_box(slot["sheetBox"]) name = slot["promptName"] extract = "reference only" if not slot.get("extract", True) else f"extracts to {slot.get('file')}" lines.append(f"- {slot['id']}: x={box[0]} y={box[1]} w={box[2]-box[0]} h={box[3]-box[1]}: {name}; {extract}.") lines.extend( [ "", "Eye construction must be layer-ready: white sclera pieces must contain no iris; iris pieces must contain no sclera; pupils must be black-only; highlights must be separate white dots; lashes must be separate line art.", "Avoid: off-center mouth, mismatched eye scale, combined eyes, facial features on the faceless head, hair holes, side hair drawn as ponytails, labels, captions, watermark, decorative background, shadows, gradients, paper texture, creepy or distorted anatomy.", ] ) return "\n".join(lines) + "\n" def default_for_alpha(defaults: dict[str, Any], key: str, alpha_mode: str, fallback: Any) -> Any: by_alpha = defaults.get(f"{key}ByAlpha", {}) return by_alpha.get(alpha_mode, defaults.get(key, fallback)) def extract_part( sheet: Image.Image, slot: dict[str, Any], canvas_size: tuple[int, int], defaults: dict[str, Any], ) -> Image.Image: alpha_mode = slot.get("alpha", "matte") default_inset = default_for_alpha(defaults, "sourceInset", alpha_mode, 0) src = inset_box(as_box(slot["sheetBox"]), as_inset(slot.get("sourceInset", default_inset))) dst = as_box(slot["targetBox"]) crop = sheet.crop(src).convert("RGBA") frame_erase = int(slot.get("frameErase", default_for_alpha(defaults, "frameErase", alpha_mode, 0))) crop = neutralize_crop_frame(crop, frame_erase) if slot.get("aiCutout", False): part = ai_cutout(crop, slot, defaults) elif alpha_mode == "shape": part = shape_crop(crop, slot.get("shape", "round")) elif alpha_mode == "soft": part = soft_detail_alpha(crop) elif alpha_mode == "dark": part = dark_alpha(crop) else: part = remove_white_bg(crop) if "maskShape" in slot: part = apply_shape_mask(part, slot["maskShape"]) if slot.get("cleanIrisDetail", False): part = clean_iris_detail(part) part = filter_components( part, min_area=int(slot.get("minArea", 0)), keep_largest=bool(slot.get("keepLargest", False)), ) suppress_artifacts = slot.get( "suppressEdgeArtifacts", default_for_alpha(defaults, "suppressEdgeArtifacts", alpha_mode, alpha_mode == "matte"), ) if suppress_artifacts: part = suppress_edge_artifacts( part, edge_width=int(slot.get("edgeArtifactWidth", defaults.get("edgeArtifactWidth", 3))), luma_min=float(slot.get("edgeArtifactLuma", defaults.get("edgeArtifactLuma", 150))), chroma_max=float(slot.get("edgeArtifactChroma", defaults.get("edgeArtifactChroma", 30))), ) if defaults.get("defringe", True) and slot.get("defringe", True): part = decontaminate_edges(part) if "alphaScale" in slot: arr = np.array(part.convert("RGBA")) arr[:, :, 3] = np.clip(arr[:, :, 3].astype(np.float32) * float(slot["alphaScale"]), 0, 255).astype(np.uint8) arr[arr[:, :, 3] == 0, :3] = 0 part = Image.fromarray(arr, "RGBA") return place(part, dst, canvas_size) def extract_all(layout: dict[str, Any], sheet_path: Path) -> None: if not sheet_path.exists(): raise FileNotFoundError(sheet_path) OUT.mkdir(parents=True, exist_ok=True) sheet = Image.open(sheet_path).convert("RGBA") canvas_size = (int(layout["canvas"]["width"]), int(layout["canvas"]["height"])) defaults = layout.get("extraction", {}) for slot in layout["slots"]: if not slot.get("extract", True): continue im = extract_part(sheet, slot, canvas_size, defaults) out = OUT / slot["file"] im.save(out, optimize=True) print(f"{slot['id']:22} -> {slot['file']:26} bbox={alpha_bbox(im)}") def composite(layout: dict[str, Any], order_name: str, out_path: Path) -> Image.Image: slots = slot_by_id(layout) canvas_size = (int(layout["canvas"]["width"]), int(layout["canvas"]["height"])) bg = Image.new("RGBA", canvas_size, (30, 39, 36, 255)) for slot_id in layout[order_name]: slot = slots[slot_id] bg.alpha_composite(Image.open(OUT / slot["file"]).convert("RGBA")) bg.save(out_path, optimize=True) print(f"preview {out_path}") return bg def make_contact(layout: dict[str, Any]) -> None: slots = [slot for slot in layout["slots"] if slot.get("extract", True)] tile_w, tile_h = 180, 160 cols = 5 rows = (len(slots) + cols - 1) // cols sheet = Image.new("RGBA", (cols * tile_w, rows * tile_h), (30, 39, 36, 255)) draw = ImageDraw.Draw(sheet) for i, slot in enumerate(slots): layer = Image.open(OUT / slot["file"]).convert("RGBA") box = alpha_bbox(layer) thumb = Image.new("RGBA", (1, 1), (0, 0, 0, 0)) if box is None else layer.crop(box) thumb.thumbnail((tile_w - 24, tile_h - 42), Image.Resampling.LANCZOS) x = (i % cols) * tile_w + (tile_w - thumb.width) // 2 y = (i // cols) * tile_h + 16 sheet.alpha_composite(thumb, (x, y)) draw.text(((i % cols) * tile_w + 8, (i // cols) * tile_h + tile_h - 22), slot["file"], fill=(200, 214, 208, 255)) out = HERE / "parts-contact-v3.png" sheet.save(out, optimize=True) print(f"contact {out}") def preview(layout: dict[str, Any]) -> None: composite(layout, "layerOrder", HERE / "preview-neutral-v3.png") composite(layout, "blinkLayerOrder", HERE / "preview-blink-v3.png") composite(layout, "speakingLayerOrder", HERE / "preview-speaking-v3.png") make_contact(layout) def main() -> int: parser = argparse.ArgumentParser() parser.add_argument("command", choices=["template", "prompt", "extract", "preview", "all"]) parser.add_argument("--sheet", default=None, help="Generated sheet path for extract/all") args = parser.parse_args() layout = load_layout() if args.command in {"template", "all"}: labelled = draw_template(layout, clean=False) clean = draw_template(layout, clean=True) print(f"template {labelled}") print(f"template-clean {clean}") if args.command in {"prompt", "all"}: prompt = build_prompt(layout) out = HERE / "prompt-live2d-v3.txt" out.write_text(prompt, encoding="utf-8") print(f"prompt {out}") if args.command in {"extract", "all"}: sheet_path = Path(args.sheet) if args.sheet else HERE / layout["sheet"]["generated"] extract_all(layout, sheet_path) if args.command in {"preview", "all"}: preview(layout) return 0 if __name__ == "__main__": raise SystemExit(main())