아바타 v3 세션 연결 — P1 서연 리노컷 아바타와 음성 립싱크
Some checks are pending
API contract / OpenAPI type drift (push) Waiting to run

- ClientAvatar가 리그 레지스트리(v3/rigs)로 P1을 판정해 그림 영역을 ClientAvatarV3로 그린다(로드 중·실패 시 기존 SVG, VITE_AVATAR_V3=0이면 끔)
- 발화 구동 공용 모듈 speechDriver: Lab과 세션이 함께 쓰고, TTS AudioBuffer를 오디오 시계로 표본해 립싱크·발화 동반층·지문 cue를 돌린다
- Session: TTS 재생 시작 시 발화(원문·버퍼·시작 시각) 전달, 음성 실패 시 텍스트 타이밍 발화, 정지 경로에서 해제, 개방도·겉표정 강도 전달
- 정지 화면(시작 전·썸네일)도 한 프레임을 그리고, 박스 실측 폭으로 상반신/얼굴 크롭을 고른다
- 세션 발화 E2E(오디오·실패 경로), P1 레이아웃 게이트, P1 표정 시험을 v3 기대로 갱신
This commit is contained in:
Yun Chan 2026-10-01 16:12:24 +09:00
parent ecb36d123f
commit 613bcb603e
14 changed files with 1000 additions and 182 deletions

View file

@ -14,18 +14,9 @@ import { CHANNEL_IDS } from "../components/avatar/engine/channels";
import { createAvatarEngine, type AvatarEngine, type DebugSnapshot } from "../components/avatar/engine/engine";
import { REACTION_CLIPS, REACTION_CLIP_IDS, type ReactionClipId } from "../components/avatar/engine/clipCatalog";
import { demeanorFor } from "../components/avatar/engine/demeanorDefaults";
import { buildPerformance, type Performance, type PerformanceCue } from "../components/avatar/engine/performance";
import {
buildSpeechTimeline,
currentViseme,
sampleSpeech,
type PhraseKind,
type SpeechShape,
type SpeechTimeline,
type VisemeId,
} from "../components/avatar/engine/lipsync";
import { buildCoSpeechPlan, isStressPulseActive, sampleCoSpeech, type CoSpeechPlan } from "../components/avatar/engine/coSpeech";
import { computeEnvelope, type SpeechEnvelope } from "../components/avatar/engine/speechEnvelope";
import type { Performance, PerformanceCue } from "../components/avatar/engine/performance";
import type { PhraseKind, SpeechShape, VisemeId } from "../components/avatar/engine/lipsync";
import { startSpeech, type SpeechHandle } from "../components/avatar/engine/speechDriver";
import { AVATAR_EXPRESSION_LIBRARY, type AvatarExpression, type AvatarState } from "../components/avatar/persona";
import "./avatar-lab.css";
@ -83,16 +74,6 @@ function createClock(): Clock {
};
}
/** 표시용: localMs 시점까지 시작한 가장 최근 구의 종류. 구 사이 휴지 중에는 그 직전 구를 보인다. */
function currentPhraseKind(timeline: SpeechTimeline, localMs: number): PhraseKind | null {
let kind: PhraseKind | null = null;
for (const ph of timeline.phrases) {
if (ph.startMs > localMs) break;
kind = ph.kind;
}
return kind;
}
function seedFromQuery(): number {
const raw = new URLSearchParams(window.location.search).get("seed");
const parsed = raw === null ? NaN : Number(raw);
@ -132,7 +113,7 @@ export default function AvatarLab() {
const [unmatched, setUnmatched] = useState<string[]>([]);
const [snapshot, setSnapshot] = useState<DebugSnapshot | null>(null);
const speechRafRef = useRef<number>(0);
const speechHandleRef = useRef<SpeechHandle | null>(null);
const meterRefs = useRef<Record<string, HTMLTableCellElement | null>>({});
/* 발화층(립싱크) — 원시 SpeechShape·비짐 표시용. 채널 미터와 달리 매 프레임 갱신하지
@ -213,7 +194,7 @@ export default function AvatarLab() {
useEffect(() => {
return () => {
cancelAnimationFrame(speechRafRef.current);
speechHandleRef.current?.stop();
try {
audioSourceRef.current?.stop();
} catch {
@ -247,76 +228,38 @@ export default function AvatarLab() {
coSpeechDisplayRef.current = { phraseKind: null, stressed: false };
}
/** timeline을 nowFn() 시계로 매 프레임 표본해 engine.setSpeechShape·setSpeechMotion에 흘려보낸다.
coSpeechPlan이 있으면 발화 동반층(§5.5)도 같은 프레임에 표본한다. */
function runSpeechTimeline(
timeline: SpeechTimeline,
speechStartMs: number,
nowFn: () => number,
articulation: number,
coSpeechPlan: CoSpeechPlan | null,
envelope?: SpeechEnvelope,
): void {
cancelAnimationFrame(speechRafRef.current);
const endMs = speechStartMs + timeline.totalDurationMs;
let prevLocalMs = -Infinity;
const step = () => {
const t = nowFn();
if (t < speechStartMs) {
speechRafRef.current = requestAnimationFrame(step);
return;
}
if (t >= endMs) {
endSpeech(clock.now());
return;
}
const localMs = t - speechStartMs;
const shape = sampleSpeech(timeline, localMs, articulation);
engine.setSpeechShape(shape);
speechDisplayRef.current = { viseme: currentViseme(timeline, localMs), shape };
if (coSpeechPlan) {
const sample = sampleCoSpeech(coSpeechPlan, localMs, prevLocalMs, envelope);
engine.setSpeechMotion(sample.delta);
if (sample.blinkNow) engine.requestSpeechBlink(clock.now());
coSpeechDisplayRef.current = {
phraseKind: currentPhraseKind(timeline, localMs),
stressed: isStressPulseActive(coSpeechPlan, localMs),
};
}
prevLocalMs = localMs;
speechRafRef.current = requestAnimationFrame(step);
};
speechRafRef.current = requestAnimationFrame(step);
}
function playSpeech(): void {
speechHandleRef.current?.stop();
const nowMs = clock.now();
const { performance: perf, unmatched: um } = buildPerformance({
const speech = demeanorFor(personaCode).speech;
setAvatarState("speaking");
engine.setState("speaking", nowMs);
const handle = startSpeech({
engine,
text: speechText,
expression: selectedExpression,
intensity,
openness,
seed,
speech,
now: clock.now,
onFrame: ({ viseme, shape, phraseKind, stressed }) => {
speechDisplayRef.current = { viseme, shape };
coSpeechDisplayRef.current = { phraseKind, stressed };
},
onEnd: () => endSpeech(clock.now()),
});
const speech = demeanorFor(personaCode).speech;
const timeline = buildSpeechTimeline({ text: speechText, syllablesPerSec: speech.syllablesPerSec });
const coSpeechPlan = buildCoSpeechPlan(timeline, speech, seed);
const speechStartMs = nowMs + 300;
setAvatarState("speaking");
engine.setState("speaking", nowMs);
engine.playPerformance(perf, { speechStartMs, speechDurationMs: timeline.totalDurationMs }, nowMs);
setParsedCues(perf.cues);
setUnmatched(um);
runSpeechTimeline(timeline, speechStartMs, clock.now, speech.articulation, coSpeechPlan);
speechHandleRef.current = handle;
setParsedCues(handle.cues);
setUnmatched(handle.unmatched);
}
function handleAudioFileChange(e: ChangeEvent<HTMLInputElement>): void {
setAudioFileName(e.target.files?.[0]?.name ?? null);
}
/** "오디오 파일로 말하기" — 로컬 오디오를 디코드해 포락선을 만들고, 오디오 시계로 표본한다. */
/** "오디오 파일로 말하기" — 로컬 오디오를 디코드해 speechDriver에 넘긴다(포락선 선분석·
오디오 시계 표본은 startSpeech 안에서 한다). */
async function playSpeechWithAudioFile(): Promise<void> {
const file = audioFileInputRef.current?.files?.[0];
if (!file) return;
@ -330,21 +273,6 @@ export default function AvatarLab() {
const arrayBuffer = await file.arrayBuffer();
const audioBuffer = await ctx.decodeAudioData(arrayBuffer.slice(0));
const channels: Float32Array[] = [];
for (let c = 0; c < audioBuffer.numberOfChannels; c++) channels.push(audioBuffer.getChannelData(c));
const envelope = computeEnvelope(channels, audioBuffer.sampleRate);
const nowMs = clock.now();
const { performance: perf, unmatched: um } = buildPerformance({
text: speechText,
expression: selectedExpression,
intensity,
openness,
seed,
});
const speech = demeanorFor(personaCode).speech;
const timeline = buildSpeechTimeline({ text: speechText, syllablesPerSec: speech.syllablesPerSec, envelope });
const coSpeechPlan = buildCoSpeechPlan(timeline, speech, seed);
try {
audioSourceRef.current?.stop();
@ -356,19 +284,33 @@ export default function AvatarLab() {
source.connect(ctx.destination);
audioSourceRef.current = source;
const speechStartMs = nowMs + 300;
setAvatarState("speaking");
engine.setState("speaking", nowMs);
engine.playPerformance(perf, { speechStartMs, speechDurationMs: timeline.totalDurationMs }, nowMs);
setParsedCues(perf.cues);
setUnmatched(um);
const startAtCtx = ctx.currentTime + 0.3;
source.start(startAtCtx);
/* 오디오 시계 기준(초 → ms). Lab의 clock(performance.now() 기반)과는 별개 시계다 —
재생 시작을 같은 300ms로 맞췄지만 독립 시계라 아주 긴 발화에서는 드리프트가 있을 수 있다. */
const nowFn = () => (ctx.currentTime - startAtCtx) * 1000 + speechStartMs;
runSpeechTimeline(timeline, speechStartMs, nowFn, speech.articulation, coSpeechPlan, envelope);
speechHandleRef.current?.stop();
const nowMs = clock.now();
const speech = demeanorFor(personaCode).speech;
setAvatarState("speaking");
engine.setState("speaking", nowMs);
const handle = startSpeech({
engine,
text: speechText,
expression: selectedExpression,
intensity,
openness,
seed,
speech,
now: clock.now,
audio: { buffer: audioBuffer, context: ctx, startAt: startAtCtx },
onFrame: ({ viseme, shape, phraseKind, stressed }) => {
speechDisplayRef.current = { viseme, shape };
coSpeechDisplayRef.current = { phraseKind, stressed };
},
onEnd: () => endSpeech(clock.now()),
});
speechHandleRef.current = handle;
setParsedCues(handle.cues);
setUnmatched(handle.unmatched);
}
function playLeakTest(): void {