아바타 v3 세션 연결 — P1 서연 리노컷 아바타와 음성 립싱크
Some checks are pending
API contract / OpenAPI type drift (push) Waiting to run
Some checks are pending
API contract / OpenAPI type drift (push) Waiting to run
- ClientAvatar가 리그 레지스트리(v3/rigs)로 P1을 판정해 그림 영역을 ClientAvatarV3로 그린다(로드 중·실패 시 기존 SVG, VITE_AVATAR_V3=0이면 끔) - 발화 구동 공용 모듈 speechDriver: Lab과 세션이 함께 쓰고, TTS AudioBuffer를 오디오 시계로 표본해 립싱크·발화 동반층·지문 cue를 돌린다 - Session: TTS 재생 시작 시 발화(원문·버퍼·시작 시각) 전달, 음성 실패 시 텍스트 타이밍 발화, 정지 경로에서 해제, 개방도·겉표정 강도 전달 - 정지 화면(시작 전·썸네일)도 한 프레임을 그리고, 박스 실측 폭으로 상반신/얼굴 크롭을 고른다 - 세션 발화 E2E(오디오·실패 경로), P1 레이아웃 게이트, P1 표정 시험을 v3 기대로 갱신
This commit is contained in:
parent
ecb36d123f
commit
613bcb603e
14 changed files with 1000 additions and 182 deletions
|
|
@ -14,18 +14,9 @@ import { CHANNEL_IDS } from "../components/avatar/engine/channels";
|
|||
import { createAvatarEngine, type AvatarEngine, type DebugSnapshot } from "../components/avatar/engine/engine";
|
||||
import { REACTION_CLIPS, REACTION_CLIP_IDS, type ReactionClipId } from "../components/avatar/engine/clipCatalog";
|
||||
import { demeanorFor } from "../components/avatar/engine/demeanorDefaults";
|
||||
import { buildPerformance, type Performance, type PerformanceCue } from "../components/avatar/engine/performance";
|
||||
import {
|
||||
buildSpeechTimeline,
|
||||
currentViseme,
|
||||
sampleSpeech,
|
||||
type PhraseKind,
|
||||
type SpeechShape,
|
||||
type SpeechTimeline,
|
||||
type VisemeId,
|
||||
} from "../components/avatar/engine/lipsync";
|
||||
import { buildCoSpeechPlan, isStressPulseActive, sampleCoSpeech, type CoSpeechPlan } from "../components/avatar/engine/coSpeech";
|
||||
import { computeEnvelope, type SpeechEnvelope } from "../components/avatar/engine/speechEnvelope";
|
||||
import type { Performance, PerformanceCue } from "../components/avatar/engine/performance";
|
||||
import type { PhraseKind, SpeechShape, VisemeId } from "../components/avatar/engine/lipsync";
|
||||
import { startSpeech, type SpeechHandle } from "../components/avatar/engine/speechDriver";
|
||||
import { AVATAR_EXPRESSION_LIBRARY, type AvatarExpression, type AvatarState } from "../components/avatar/persona";
|
||||
import "./avatar-lab.css";
|
||||
|
||||
|
|
@ -83,16 +74,6 @@ function createClock(): Clock {
|
|||
};
|
||||
}
|
||||
|
||||
/** 표시용: localMs 시점까지 시작한 가장 최근 구의 종류. 구 사이 휴지 중에는 그 직전 구를 보인다. */
|
||||
function currentPhraseKind(timeline: SpeechTimeline, localMs: number): PhraseKind | null {
|
||||
let kind: PhraseKind | null = null;
|
||||
for (const ph of timeline.phrases) {
|
||||
if (ph.startMs > localMs) break;
|
||||
kind = ph.kind;
|
||||
}
|
||||
return kind;
|
||||
}
|
||||
|
||||
function seedFromQuery(): number {
|
||||
const raw = new URLSearchParams(window.location.search).get("seed");
|
||||
const parsed = raw === null ? NaN : Number(raw);
|
||||
|
|
@ -132,7 +113,7 @@ export default function AvatarLab() {
|
|||
const [unmatched, setUnmatched] = useState<string[]>([]);
|
||||
const [snapshot, setSnapshot] = useState<DebugSnapshot | null>(null);
|
||||
|
||||
const speechRafRef = useRef<number>(0);
|
||||
const speechHandleRef = useRef<SpeechHandle | null>(null);
|
||||
const meterRefs = useRef<Record<string, HTMLTableCellElement | null>>({});
|
||||
|
||||
/* 발화층(립싱크) — 원시 SpeechShape·비짐 표시용. 채널 미터와 달리 매 프레임 갱신하지
|
||||
|
|
@ -213,7 +194,7 @@ export default function AvatarLab() {
|
|||
|
||||
useEffect(() => {
|
||||
return () => {
|
||||
cancelAnimationFrame(speechRafRef.current);
|
||||
speechHandleRef.current?.stop();
|
||||
try {
|
||||
audioSourceRef.current?.stop();
|
||||
} catch {
|
||||
|
|
@ -247,76 +228,38 @@ export default function AvatarLab() {
|
|||
coSpeechDisplayRef.current = { phraseKind: null, stressed: false };
|
||||
}
|
||||
|
||||
/** timeline을 nowFn() 시계로 매 프레임 표본해 engine.setSpeechShape·setSpeechMotion에 흘려보낸다.
|
||||
coSpeechPlan이 있으면 발화 동반층(§5.5)도 같은 프레임에 표본한다. */
|
||||
function runSpeechTimeline(
|
||||
timeline: SpeechTimeline,
|
||||
speechStartMs: number,
|
||||
nowFn: () => number,
|
||||
articulation: number,
|
||||
coSpeechPlan: CoSpeechPlan | null,
|
||||
envelope?: SpeechEnvelope,
|
||||
): void {
|
||||
cancelAnimationFrame(speechRafRef.current);
|
||||
const endMs = speechStartMs + timeline.totalDurationMs;
|
||||
let prevLocalMs = -Infinity;
|
||||
const step = () => {
|
||||
const t = nowFn();
|
||||
if (t < speechStartMs) {
|
||||
speechRafRef.current = requestAnimationFrame(step);
|
||||
return;
|
||||
}
|
||||
if (t >= endMs) {
|
||||
endSpeech(clock.now());
|
||||
return;
|
||||
}
|
||||
const localMs = t - speechStartMs;
|
||||
const shape = sampleSpeech(timeline, localMs, articulation);
|
||||
engine.setSpeechShape(shape);
|
||||
speechDisplayRef.current = { viseme: currentViseme(timeline, localMs), shape };
|
||||
|
||||
if (coSpeechPlan) {
|
||||
const sample = sampleCoSpeech(coSpeechPlan, localMs, prevLocalMs, envelope);
|
||||
engine.setSpeechMotion(sample.delta);
|
||||
if (sample.blinkNow) engine.requestSpeechBlink(clock.now());
|
||||
coSpeechDisplayRef.current = {
|
||||
phraseKind: currentPhraseKind(timeline, localMs),
|
||||
stressed: isStressPulseActive(coSpeechPlan, localMs),
|
||||
};
|
||||
}
|
||||
prevLocalMs = localMs;
|
||||
|
||||
speechRafRef.current = requestAnimationFrame(step);
|
||||
};
|
||||
speechRafRef.current = requestAnimationFrame(step);
|
||||
}
|
||||
|
||||
function playSpeech(): void {
|
||||
speechHandleRef.current?.stop();
|
||||
const nowMs = clock.now();
|
||||
const { performance: perf, unmatched: um } = buildPerformance({
|
||||
const speech = demeanorFor(personaCode).speech;
|
||||
setAvatarState("speaking");
|
||||
engine.setState("speaking", nowMs);
|
||||
const handle = startSpeech({
|
||||
engine,
|
||||
text: speechText,
|
||||
expression: selectedExpression,
|
||||
intensity,
|
||||
openness,
|
||||
seed,
|
||||
speech,
|
||||
now: clock.now,
|
||||
onFrame: ({ viseme, shape, phraseKind, stressed }) => {
|
||||
speechDisplayRef.current = { viseme, shape };
|
||||
coSpeechDisplayRef.current = { phraseKind, stressed };
|
||||
},
|
||||
onEnd: () => endSpeech(clock.now()),
|
||||
});
|
||||
const speech = demeanorFor(personaCode).speech;
|
||||
const timeline = buildSpeechTimeline({ text: speechText, syllablesPerSec: speech.syllablesPerSec });
|
||||
const coSpeechPlan = buildCoSpeechPlan(timeline, speech, seed);
|
||||
const speechStartMs = nowMs + 300;
|
||||
setAvatarState("speaking");
|
||||
engine.setState("speaking", nowMs);
|
||||
engine.playPerformance(perf, { speechStartMs, speechDurationMs: timeline.totalDurationMs }, nowMs);
|
||||
setParsedCues(perf.cues);
|
||||
setUnmatched(um);
|
||||
runSpeechTimeline(timeline, speechStartMs, clock.now, speech.articulation, coSpeechPlan);
|
||||
speechHandleRef.current = handle;
|
||||
setParsedCues(handle.cues);
|
||||
setUnmatched(handle.unmatched);
|
||||
}
|
||||
|
||||
function handleAudioFileChange(e: ChangeEvent<HTMLInputElement>): void {
|
||||
setAudioFileName(e.target.files?.[0]?.name ?? null);
|
||||
}
|
||||
|
||||
/** "오디오 파일로 말하기" — 로컬 오디오를 디코드해 포락선을 만들고, 오디오 시계로 표본한다. */
|
||||
/** "오디오 파일로 말하기" — 로컬 오디오를 디코드해 speechDriver에 넘긴다(포락선 선분석·
|
||||
오디오 시계 표본은 startSpeech 안에서 한다). */
|
||||
async function playSpeechWithAudioFile(): Promise<void> {
|
||||
const file = audioFileInputRef.current?.files?.[0];
|
||||
if (!file) return;
|
||||
|
|
@ -330,21 +273,6 @@ export default function AvatarLab() {
|
|||
|
||||
const arrayBuffer = await file.arrayBuffer();
|
||||
const audioBuffer = await ctx.decodeAudioData(arrayBuffer.slice(0));
|
||||
const channels: Float32Array[] = [];
|
||||
for (let c = 0; c < audioBuffer.numberOfChannels; c++) channels.push(audioBuffer.getChannelData(c));
|
||||
const envelope = computeEnvelope(channels, audioBuffer.sampleRate);
|
||||
|
||||
const nowMs = clock.now();
|
||||
const { performance: perf, unmatched: um } = buildPerformance({
|
||||
text: speechText,
|
||||
expression: selectedExpression,
|
||||
intensity,
|
||||
openness,
|
||||
seed,
|
||||
});
|
||||
const speech = demeanorFor(personaCode).speech;
|
||||
const timeline = buildSpeechTimeline({ text: speechText, syllablesPerSec: speech.syllablesPerSec, envelope });
|
||||
const coSpeechPlan = buildCoSpeechPlan(timeline, speech, seed);
|
||||
|
||||
try {
|
||||
audioSourceRef.current?.stop();
|
||||
|
|
@ -356,19 +284,33 @@ export default function AvatarLab() {
|
|||
source.connect(ctx.destination);
|
||||
audioSourceRef.current = source;
|
||||
|
||||
const speechStartMs = nowMs + 300;
|
||||
setAvatarState("speaking");
|
||||
engine.setState("speaking", nowMs);
|
||||
engine.playPerformance(perf, { speechStartMs, speechDurationMs: timeline.totalDurationMs }, nowMs);
|
||||
setParsedCues(perf.cues);
|
||||
setUnmatched(um);
|
||||
|
||||
const startAtCtx = ctx.currentTime + 0.3;
|
||||
source.start(startAtCtx);
|
||||
/* 오디오 시계 기준(초 → ms). Lab의 clock(performance.now() 기반)과는 별개 시계다 —
|
||||
재생 시작을 같은 300ms로 맞췄지만 독립 시계라 아주 긴 발화에서는 드리프트가 있을 수 있다. */
|
||||
const nowFn = () => (ctx.currentTime - startAtCtx) * 1000 + speechStartMs;
|
||||
runSpeechTimeline(timeline, speechStartMs, nowFn, speech.articulation, coSpeechPlan, envelope);
|
||||
|
||||
speechHandleRef.current?.stop();
|
||||
const nowMs = clock.now();
|
||||
const speech = demeanorFor(personaCode).speech;
|
||||
setAvatarState("speaking");
|
||||
engine.setState("speaking", nowMs);
|
||||
const handle = startSpeech({
|
||||
engine,
|
||||
text: speechText,
|
||||
expression: selectedExpression,
|
||||
intensity,
|
||||
openness,
|
||||
seed,
|
||||
speech,
|
||||
now: clock.now,
|
||||
audio: { buffer: audioBuffer, context: ctx, startAt: startAtCtx },
|
||||
onFrame: ({ viseme, shape, phraseKind, stressed }) => {
|
||||
speechDisplayRef.current = { viseme, shape };
|
||||
coSpeechDisplayRef.current = { phraseKind, stressed };
|
||||
},
|
||||
onEnd: () => endSpeech(clock.now()),
|
||||
});
|
||||
speechHandleRef.current = handle;
|
||||
setParsedCues(handle.cues);
|
||||
setUnmatched(handle.unmatched);
|
||||
}
|
||||
|
||||
function playLeakTest(): void {
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue