vignette/docs/avatar-art/linocut-pipeline/scripts/register_edit.py
Yun Chan 98577cf4ae 리노컷 공통 파이프라인 캐스트 확장 — 정렬·정규화·얼굴 해칭·원화 눈썹·눈 실측
- register_edit(편집본 ECC 정렬·챔퍼 지표), normalize_base(P1 머리 크기 정규화, 위쪽 종이색 채움)
- faceDetail region=face(얼굴 해칭을 원화 픽셀로 보존)·matchFacelessColor
- brow_sprite 단계: 원화 눈썹 잉크 스프라이트(hairFront 아래 제외), 윗눈꺼풀 두께·홍채 실측
- 리그 계약 RigBrow.halfWidth/sprite·RigEye.lineScale, 렌더러 눈썹 16띠·바깥쪽 눈꺼풀 굵기
- 실행기 UTF-8 하위 실행, brow_centerline 패딩본 자체 생성, QA 시트 촬영 도구(깜빡임 회피·고해상도 대조)
- P1은 새 필드를 모두 끄고 게시 자산·리그 바이트 동일
2026-10-01 22:38:27 +09:00

327 lines
15 KiB
Python

"""공통 리노컷 리그 — 캐스트 확장 0단계: codex 편집본을 원화 좌표계로 정렬.
ref는 편집의 참조가 된 원본(후보 원화, normalize_base.py 이전 — 아직 캔버스로
옮기지 않은 원래 크기)이고, edited는 그 ref를 참조해 codex가 편집한 결과다
(얼굴 없는 기본형 또는 body 재생성). 크기가 다를 수 있다.
정렬: edited를 ref 크기로 리사이즈한 뒤, cv2.findTransformECC(MOTION_AFFINE)로
회색조·가우시안 흐림(gaussFiltSize) 위에서 ref 좌표계에 정렬한다. ECC가 보는
영역은 모드별 마스크로 제한한다(전체 이미지를 다 보면 얼굴 생김새 차이·크로마키
배경이 정렬을 왜곡한다):
- faceless: 머리카락·옷·얼굴 윤곽처럼 두 이미지에서 모양이 같아야 하는 곳만
본다 — 랜드마크(ref 기준)로 잡은 눈·눈썹·입 상자를 25px 넓혀 뺀 뒤, 평평한
배경(지역 표준편차가 작은 곳)도 뺀다.
- body: 턱끝 +20px 아래 행 중 edited(리사이즈본)의 비초록(크로마키 전경) 픽셀만
본다 — 그 위는 얼굴이라 body 재생성과 무관하고, 초록 배경은 정렬 정보가 없다.
출력은 ref 크기로 맞춘 정렬된 편집본 1장이다. ECC로 못 채우는 테두리는
faceless면 테두리 복제(BORDER_REPLICATE), body면 크로마키 배경(#00ff00)이다.
사용: register_edit.py <ref.png> <edited.png> <out.png> --mode faceless|body [--report <json>]
"""
from __future__ import annotations
import argparse
import json
import subprocess
import sys
import tempfile
from pathlib import Path
import cv2
import numpy as np
from PIL import Image
from scipy.ndimage import distance_transform_edt
SCRIPTS_DIR = Path(__file__).resolve().parent
sys.path.insert(0, str(SCRIPTS_DIR))
from segmentation import chroma_key # noqa: E402 — 크로마키 전경/배경 판정 재사용(같은 임계치로 일관성 유지)
# --- mediapipe FaceLandmarker 인덱스(landmarks.py/segmentation.py와 같다) ---
RIGHT_EYEBROW_IDX = [46, 53, 52, 65, 55, 70, 63, 105, 66, 107]
LEFT_EYEBROW_IDX = [276, 283, 282, 295, 285, 300, 293, 334, 296, 336]
EYE_A_IDX = {"outer": 33, "inner": 133, "top": 159, "bottom": 145}
EYE_B_IDX = {"inner": 362, "outer": 263, "top": 386, "bottom": 374}
MOUTH_CORNER_A_IDX = 61
MOUTH_CORNER_B_IDX = 291
UPPER_LIP_TOP_IDX = 0
LOWER_LIP_BOTTOM_IDX = 17
CHIN_IDX = 152
FACE_OVAL_LOOP = [
10, 338, 297, 332, 284, 251, 389, 356, 454, 323, 361, 288, 397, 365, 379,
378, 400, 377, 152, 148, 176, 149, 150, 136, 172, 58, 132, 93, 234, 127,
162, 21, 54, 103, 67, 109,
]
FACE_OVAL_SCALE = 1.04
# --- ECC ---
ECC_EXCLUDE_PAD_PX = 25.0 # 눈·눈썹·입 상자를 이만큼 넓혀 ECC 마스크에서 뺀다
ECC_CHIN_MARGIN_PX = 20.0 # body: 턱끝 + 이만큼 아래부터 본다
FLAT_BG_WINDOW = 9 # 평평한 배경 판정용 지역 표준편차 창(px)
FLAT_BG_STD_THRESH = 6.0 # 이 미만이면 "평평"(정렬 정보 없음)으로 보고 뺀다
GREEN_ALPHA_THRESH = 127 # chroma_key 알파 임계치(이상이면 비초록 전경)
ECC_MIN_MASK_PX = 400 # 이보다 적으면 ECC가 의미 있게 수렴하지 않는다
ECC_GAUSS_FILT_SIZE = 7 # findTransformECC 내부 가우시안 흐림 창(홀수)
ECC_MAX_ITER = 5000
ECC_EPS = 1e-7
# --- 지표 ---
CANNY_BLUR_SIGMA = 1.0
CANNY_LOW, CANNY_HIGH = 40, 120
FACE_OVAL_BAND_PX = 12.0 # 얼굴 윤곽 띠 폭(±px)
INK_LUM_THRESH = 70.0 # "어두운 잉크" 휘도 임계치
def detect_face_landmarks(image_path: Path) -> list[tuple[float, float]]:
"""FaceLandmarker를 별도 프로세스로 실행한다(세그폴트 회피, segmentation.py와 같은 이유)."""
with tempfile.TemporaryDirectory() as td:
out_json = Path(td) / "landmarks.json"
proc = subprocess.run(
[sys.executable, "-u", str(SCRIPTS_DIR / "_run_face_landmarks.py"), str(image_path), str(out_json)],
capture_output=True, text=True,
)
print(proc.stdout.strip())
if proc.returncode != 0 or not out_json.exists():
raise SystemExit(f"[중단] {image_path.name}: FaceLandmarker 서브프로세스 실패.\n{proc.stderr}")
data = json.loads(out_json.read_text(encoding="utf-8"))
if not data.get("ok"):
raise SystemExit(f"[중단] {image_path.name}: FaceLandmarker가 얼굴을 찾지 못했다.")
return [(p[0], p[1]) for p in data["points"]]
def bbox_of(points: list[tuple[float, float]], idxs: list[int]) -> tuple[float, float, float, float]:
xs = [points[i][0] for i in idxs]
ys = [points[i][1] for i in idxs]
return (min(xs), min(ys), max(xs), max(ys))
def expand_box(box: tuple[float, float, float, float], pad: float, w: int, h: int) -> tuple[int, int, int, int]:
x0, y0, x1, y1 = box
return (
max(0, int(np.floor(x0 - pad))), max(0, int(np.floor(y0 - pad))),
min(w, int(np.ceil(x1 + pad))), min(h, int(np.ceil(y1 + pad))),
)
def build_face_boxes(points: list[tuple[float, float]]) -> dict[str, tuple[float, float, float, float]]:
"""눈(좌/우)·눈썹(좌/우)·입 상자(원화 좌표, 확장 전)."""
mouth_idx = [MOUTH_CORNER_A_IDX, MOUTH_CORNER_B_IDX, UPPER_LIP_TOP_IDX, LOWER_LIP_BOTTOM_IDX]
return {
"eyeA": bbox_of(points, list(EYE_A_IDX.values())),
"eyeB": bbox_of(points, list(EYE_B_IDX.values())),
"browA": bbox_of(points, RIGHT_EYEBROW_IDX),
"browB": bbox_of(points, LEFT_EYEBROW_IDX),
"mouth": bbox_of(points, mouth_idx),
}
def flat_background_mask(gray_u8: np.ndarray) -> np.ndarray:
"""지역(FLAT_BG_WINDOW) 표준편차가 FLAT_BG_STD_THRESH 미만인 곳 — 평평한 배경(정렬
정보가 없는 영역)으로 보고 ECC 포함 마스크에서 뺀다."""
g = gray_u8.astype(np.float32)
mean = cv2.boxFilter(g, -1, (FLAT_BG_WINDOW, FLAT_BG_WINDOW))
sqmean = cv2.boxFilter(g * g, -1, (FLAT_BG_WINDOW, FLAT_BG_WINDOW))
std = np.sqrt(np.clip(sqmean - mean * mean, 0.0, None))
return std < FLAT_BG_STD_THRESH
def face_oval_band_mask(points: list[tuple[float, float]], w: int, h: int, band_px: float) -> np.ndarray:
"""랜드마크 face oval(결정문 §8.2, segmentation.py/export_rig.py와 같은 루프·배율)
경계선에서 band_px 이내인 픽셀."""
from PIL import ImageDraw
loop_pts = [points[i] for i in FACE_OVAL_LOOP]
cx = float(np.mean([p[0] for p in loop_pts]))
cy = float(np.mean([p[1] for p in loop_pts]))
scaled = [((x - cx) * FACE_OVAL_SCALE + cx, (y - cy) * FACE_OVAL_SCALE + cy) for x, y in loop_pts]
outline_img = Image.new("L", (w, h), 0)
ImageDraw.Draw(outline_img).polygon(scaled, outline=255, width=1)
boundary = np.array(outline_img) > 0
dist_to_boundary = distance_transform_edt(~boundary)
return dist_to_boundary <= band_px
def chamfer_stats(ref_edges: np.ndarray, aligned_edges: np.ndarray, region_mask: np.ndarray) -> dict:
"""aligned_edges 중 region_mask 안에 있는 에지 픽셀들이 ref_edges의 가장 가까운
에지까지 떨어진 거리(챔퍼)의 중앙값·p90. ref_edges 쪽 거리장은 호출자가 한 번
계산해 재사용하도록 이미 distance_transform_edt를 적용한 배열을 받는다는 점에
주의 — 이 함수는 edge 불boolean 배열 두 개를 받아 내부에서 거리장을 계산한다."""
dist_to_ref_edge = distance_transform_edt(~ref_edges)
sample_mask = aligned_edges & region_mask
n = int(sample_mask.sum())
if n == 0:
return {"median": None, "p90": None, "nSamples": 0}
d = dist_to_ref_edge[sample_mask]
return {"median": round(float(np.median(d)), 3), "p90": round(float(np.percentile(d, 90)), 3), "nSamples": n}
def ink_ratio(gray_u8: np.ndarray, box: tuple[int, int, int, int]) -> float:
x0, y0, x1, y1 = box
region = gray_u8[y0:y1, x0:x1].astype(np.float64)
if region.size == 0:
return float("nan")
return float((region < INK_LUM_THRESH).mean())
def decompose_affine(warp_matrix: np.ndarray) -> dict:
a, b, tx = warp_matrix[0]
c, d, ty = warp_matrix[1]
det = a * d - b * c
scale = float(np.sqrt(abs(det)))
rotation_deg = float(np.degrees(np.arctan2(c, a)))
return {
"scaleSqrtDet": round(scale, 5),
"rotationDeg": round(rotation_deg, 3),
"translateX": round(float(tx), 3),
"translateY": round(float(ty), 3),
"matrix": [[round(float(v), 6) for v in row] for row in warp_matrix.tolist()],
"direction": "ref(template) 좌표 -> edited(resized, input) 좌표. 정렬된 출력은 이 행렬의 역변환(WARP_INVERSE_MAP)으로 만든다.",
}
def parse_args(argv: list[str]) -> argparse.Namespace:
parser = argparse.ArgumentParser(description="codex 편집본을 ECC 아핀 정렬로 원화(ref) 좌표계에 맞춘다.")
parser.add_argument("ref", help="편집의 참조가 된 원본(원화)")
parser.add_argument("edited", help="codex 편집본(크기가 ref와 다를 수 있다)")
parser.add_argument("out", help="정렬된 출력(ref 크기)")
parser.add_argument("--mode", required=True, choices=["faceless", "body"])
parser.add_argument("--report", help="지표 JSON 저장 경로(생략 가능, 콘솔에는 항상 출력)")
return parser.parse_args(argv)
def main(argv: list[str]) -> int:
args = parse_args(argv)
ref_path = Path(args.ref).resolve()
edited_path = Path(args.edited).resolve()
out_path = Path(args.out).resolve()
mode = args.mode
ref_img = Image.open(ref_path).convert("RGB")
w, h = ref_img.size
ref_arr = np.array(ref_img)
ref_gray = cv2.cvtColor(ref_arr, cv2.COLOR_RGB2GRAY)
edited_img = Image.open(edited_path).convert("RGB")
ew, eh = edited_img.size
print(f"ref 크기={w}x{h} edited 원본 크기={ew}x{eh} (리사이즈 후 ref 크기로 정렬)")
edited_resized = np.array(edited_img.resize((w, h), Image.LANCZOS))
edited_gray = cv2.cvtColor(edited_resized, cv2.COLOR_RGB2GRAY)
print("=== ref 랜드마크 검출 ===")
ref_points = detect_face_landmarks(ref_path)
chin_xy = ref_points[CHIN_IDX]
print(f"턱끝(152)={chin_xy}")
boxes_report: dict[str, list[int]] = {}
if mode == "faceless":
boxes = build_face_boxes(ref_points)
exclude = np.zeros((h, w), dtype=bool)
for name, box in boxes.items():
x0, y0, x1, y1 = expand_box(box, ECC_EXCLUDE_PAD_PX, w, h)
boxes_report[name] = [x0, y0, x1, y1]
exclude[y0:y1, x0:x1] = True
flat_bg = flat_background_mask(ref_gray)
include = ~exclude & ~flat_bg
print(f"제외 상자(25px 확장): {boxes_report}")
print(f"평평한 배경 제외 픽셀: {int(flat_bg.sum())} ({flat_bg.mean()*100:.1f}%)")
else:
_, edited_alpha = chroma_key(edited_resized)
nongreen = edited_alpha > GREEN_ALPHA_THRESH
chin_line = chin_xy[1] + ECC_CHIN_MARGIN_PX
below_chin = np.zeros((h, w), dtype=bool)
below_chin[int(np.ceil(chin_line)):, :] = True
include = below_chin & nongreen
print(f"턱끝+{ECC_CHIN_MARGIN_PX:.0f}px={chin_line:.1f} 아래 비초록 픽셀: {int(include.sum())}")
n_include = int(include.sum())
print(f"ECC 포함 마스크 픽셀 수: {n_include} / {w * h} ({n_include / (w * h) * 100:.2f}%)")
if n_include < ECC_MIN_MASK_PX:
raise SystemExit(f"[중단] ECC 포함 마스크 픽셀이 너무 적다({n_include} < {ECC_MIN_MASK_PX}).")
include_u8 = (include.astype(np.uint8)) * 255
warp0 = np.eye(2, 3, dtype=np.float32)
criteria = (cv2.TERM_CRITERIA_EPS | cv2.TERM_CRITERIA_COUNT, ECC_MAX_ITER, ECC_EPS)
try:
cc, warp_matrix = cv2.findTransformECC(
ref_gray.astype(np.float32), edited_gray.astype(np.float32),
warp0, cv2.MOTION_AFFINE, criteria, include_u8, ECC_GAUSS_FILT_SIZE,
)
except cv2.error as exc:
raise SystemExit(f"[중단] ECC 정렬이 수렴하지 않았다: {exc}")
print(f"ECC 상관계수={cc:.5f}")
print(f"warp_matrix=\n{warp_matrix}")
transform_report = decompose_affine(warp_matrix)
print(
f"변환: 배율(sqrt det)={transform_report['scaleSqrtDet']} 회전={transform_report['rotationDeg']}deg "
f"이동=({transform_report['translateX']}, {transform_report['translateY']})"
)
border_mode = cv2.BORDER_REPLICATE if mode == "faceless" else cv2.BORDER_CONSTANT
border_value = None if mode == "faceless" else (0, 255, 0)
warp_kwargs: dict = {"flags": cv2.INTER_LANCZOS4 | cv2.WARP_INVERSE_MAP, "borderMode": border_mode}
if border_value is not None:
warp_kwargs["borderValue"] = border_value
aligned = cv2.warpAffine(edited_resized, warp_matrix, (w, h), **warp_kwargs)
out_path.parent.mkdir(parents=True, exist_ok=True)
Image.fromarray(aligned, "RGB").save(out_path)
print(f"저장: {out_path}")
aligned_gray = cv2.cvtColor(aligned, cv2.COLOR_RGB2GRAY)
ref_edges = cv2.Canny(cv2.GaussianBlur(ref_gray, (0, 0), CANNY_BLUR_SIGMA), CANNY_LOW, CANNY_HIGH) > 0
aligned_edges = cv2.Canny(cv2.GaussianBlur(aligned_gray, (0, 0), CANNY_BLUR_SIGMA), CANNY_LOW, CANNY_HIGH) > 0
mask_chamfer = chamfer_stats(ref_edges, aligned_edges, include)
print(
f"마스크 안 Canny 챔퍼: 중앙값={mask_chamfer['median']} p90={mask_chamfer['p90']} "
f"(n={mask_chamfer['nSamples']})"
+ (f" 기준(중앙값<=1.0,p90<=2.5) {'OK' if mask_chamfer['median'] is not None and mask_chamfer['median'] <= 1.0 and mask_chamfer['p90'] <= 2.5 else '[미달]'}"
if mode == "faceless" else
f" 기준(중앙값<=1.5,p90<=3.0) {'OK' if mask_chamfer['median'] is not None and mask_chamfer['median'] <= 1.5 and mask_chamfer['p90'] <= 3.0 else '[미달]'}")
)
report: dict = {
"mode": mode,
"ref": str(ref_path),
"edited": str(edited_path),
"out": str(out_path),
"refSize": [w, h],
"editedOriginalSize": [ew, eh],
"eccCorrelation": round(float(cc), 5),
"transform": transform_report,
"eccIncludeMaskPixels": n_include,
"chamferInMask": mask_chamfer,
}
if mode == "faceless":
oval_band = face_oval_band_mask(ref_points, w, h, FACE_OVAL_BAND_PX)
oval_chamfer = chamfer_stats(ref_edges, aligned_edges, oval_band)
print(
f"얼굴 윤곽 띠(±{FACE_OVAL_BAND_PX:.0f}px) Canny 챔퍼: 중앙값={oval_chamfer['median']} "
f"p90={oval_chamfer['p90']} (n={oval_chamfer['nSamples']}) 기준(p90<=2.0) "
f"{'OK' if oval_chamfer['p90'] is not None and oval_chamfer['p90'] <= 2.0 else '[미달]'}"
)
report["faceOvalBandChamfer"] = oval_chamfer
ink_report: dict[str, dict] = {}
for name, box in boxes_report.items():
ref_ratio = ink_ratio(ref_gray, tuple(box))
edited_ratio = ink_ratio(aligned_gray, tuple(box))
ink_report[name] = {"box": box, "refInkRatio": round(ref_ratio, 4), "editedInkRatio": round(edited_ratio, 4)}
print(f" 잉크 비율(lum<{INK_LUM_THRESH:.0f}) [{name}] ref={ref_ratio:.4f} edited={edited_ratio:.4f}")
report["inkRatioByBox"] = ink_report
report["eccExcludeBoxes"] = boxes_report
if args.report:
report_path = Path(args.report).resolve()
report_path.parent.mkdir(parents=True, exist_ok=True)
report_path.write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8")
print(f"저장: {report_path}")
return 0
if __name__ == "__main__":
sys.exit(main(sys.argv[1:]))