gemma-multi-avatar / src /avatar.js
bep40's picture
Upload src/avatar.js
0008f70 verified
Raw
History Blame Contribute Delete
14.2 kB
// @ts-check
/**
* The avatar stage: a TalkingHead 3D avatar with real-time audio-driven
* lip-sync (HeadAudio), plus static model fallback for custom GLBs.
*
* WHY CUSTOM GLBS FAIL (ROOT CAUSE):
* TalkingHead.showAvatar() expects morph targets (ARKit viseme blend shapes)
* and a Mixamo-compatible skeleton. Most downloaded GLB models (Sketchfab,
* free 3D scans, etc.) are STATIC — no morph targets, no skeleton.
* showAvatar() throws and the model never renders.
*
* FIX: Two-phase loading:
* Phase 1 — Try showAvatar() (works for ReadyPlayerMe, morph-target models)
* Phase 2 — On failure, load GLB directly via THREE.GLTFLoader as a static
* model in the TalkingHead scene. No lip-sync, but the model
* renders correctly with proper lighting and camera.
*/
import { TalkingHead } from "@met4citizen/talkinghead";
import { HeadAudio } from "./vendor/headaudio.min.mjs";
import { Box3, Vector3, HemisphereLight, DirectionalLight } from "three";
import { GLTFLoader } from "three/addons/loaders/GLTFLoader.js";
const HEADAUDIO_WORKLET_URL = "/vendor/headworklet.min.mjs";
const HEADAUDIO_MODEL_URL = "/vendor/model-en-mixed.bin";
export const AVATAR_MOODS = [
"neutral", "happy", "angry", "sad", "fear", "disgust", "love", "sleep",
];
export const AVATAR_GESTURES = [
"handup", "index", "ok", "thumbup", "thumbdown", "side", "shrug",
];
const HEAD_CONFIG = {
ttsEndpoint: "N/A",
lipsyncModules: [],
cameraView: "upper",
cameraDistance: -1.4,
cameraY: -0.15,
cameraX: -0.18,
cameraRotateEnable: false,
lightAmbientIntensity: 0,
lightDirectIntensity: 0,
lightSpotIntensity: 0,
avatarIdleEyeContact: 0.3,
avatarSpeakingEyeContact: 0.7,
};
/** Fully dispose all Three.js + AudioContext resources. */
function _fullyDispose(head) {
try {
if (head.scene) {
head.scene.traverse((obj) => {
if (obj.isMesh) {
obj.geometry?.dispose();
if (obj.material) {
(Array.isArray(obj.material) ? obj.material : [obj.material]).forEach((m) => m.dispose());
}
}
});
}
while (head.scene?.children?.length) head.scene.remove(head.scene.children[0]);
if (head.renderer) {
head.renderer.dispose();
const canvas = head.renderer.domElement;
if (canvas?.parentNode) canvas.parentNode.removeChild(canvas);
}
if (head.audioCtx && head.audioCtx.state !== "closed") {
head.audioCtx.close().catch(() => {});
}
head.dispose?.();
} catch (e) {
console.warn("[AvatarStage] dispose:", e);
}
}
export class AvatarStage {
/** @param {HTMLElement} container */
constructor(container) {
this._container = container;
/** @type {TalkingHead | null} */
this.head = null;
/** @type {any | null} */
this._headaudio = null;
this._lastSpeechEnded = 0;
this._loadingEl = document.getElementById("loading");
this._lastError = null;
/** @type {boolean} True when model is static (no morph targets) */
this._staticModel = false;
/** @type {THREE.Object3D | null} */
this._modelRoot = null;
/** @type {THREE.Light[]} */
this._staticLights = [];
}
/**
* Initialize the avatar. Returns true on success, false on failure.
* Two-phase: tries showAvatar() first, falls back to static GLTF loading.
*/
async init(opt = {}) {
// ── FULL DISPOSE ──────────────────────────────────────────────────
if (this.head) {
this._removeStaticLights();
_fullyDispose(this.head);
this.head = null;
this._headaudio = null;
this._modelRoot = null;
this._staticModel = false;
}
while (this._container.firstChild) this._container.removeChild(this._container.firstChild);
this._lastError = null;
const isDefault = !opt.avatarUrl;
this.head = new TalkingHead(this._container, {
...HEAD_CONFIG,
cameraDistance: isDefault ? HEAD_CONFIG.cameraDistance : -1.25,
cameraY: isDefault ? HEAD_CONFIG.cameraY : 0.0,
cameraX: isDefault ? HEAD_CONFIG.cameraX : 0.0,
});
console.log("[AvatarStage] Loading:", opt.avatarUrl ?? "default");
// ── Phase 1: Try showAvatar() ────────────────────────────────────
let morphOk = false;
if (isDefault) {
try {
await this._loadShowAvatar(opt);
morphOk = true;
} catch (err) {
this._lastError = err?.message || String(err);
console.error("[AvatarStage] Default showAvatar FAILED:", this._lastError);
_fullyDispose(this.head);
this.head = null;
return false;
}
} else {
try {
await this._loadShowAvatar(opt);
morphOk = true;
console.log("[AvatarStage] Custom model loaded WITH morph targets");
} catch (err) {
console.warn("[AvatarStage] showAvatar failed, trying static:", err?.message);
}
}
// ── Phase 2: Static fallback for custom models ────────────────────
if (!morphOk) {
try {
await this._loadStaticModel(/** @type {string} */ (opt.avatarUrl), opt);
this._staticModel = true;
this._addStaticLights();
console.log("[AvatarStage] Custom model loaded as STATIC");
} catch (err2) {
this._lastError = err2?.message || String(err2);
console.error("[AvatarStage] Static ALSO failed:", this._lastError);
_fullyDispose(this.head);
this.head = null;
return false;
}
}
// ── Camera / BBox adjust ─────────────────────────────────────────
this._adjustCamera(opt.avatarUrl);
// ── Lipsync (morph-target models only) ───────────────────────────
if (!this._staticModel) {
try { await this._initLipsync(); } catch (err) { console.warn("[AvatarStage] Lipsync:", err); }
} else {
console.log("[AvatarStage] Static model — lipsync skipped");
}
return true;
}
/** Load via TalkingHead showAvatar (morph-target models) */
async _loadShowAvatar(opt) {
const head = /** @type {TalkingHead} */ (this.head);
await head.showAvatar(
{ url: opt.avatarUrl ?? "/avatars/vuong.glb", body: opt.body ?? "F", avatarMood: "neutral" },
(ev) => {
if (ev.lengthComputable) {
const pct = Math.min(100, Math.round((ev.loaded / ev.total) * 100));
if (this._loadingEl) this._loadingEl.textContent = `Loading avatar ${pct}%`;
}
opt.onprogress?.(ev);
},
);
}
/** Load GLB as a static model via THREE.GLTFLoader */
_loadStaticModel(url, opt) {
return new Promise((resolve, reject) => {
const loader = new GLTFLoader();
const head = /** @type {TalkingHead} */ (this.head);
loader.load(
url,
(gltf) => {
const model = gltf.scene;
this._modelRoot = model;
// Remove any default placeholder from scene
for (let i = head.scene.children.length - 1; i >= 0; i--) {
const c = head.scene.children[i];
if (c !== head.camera) head.scene.remove(c);
}
// Center the model visually
model.visible = true;
head.scene.add(model);
if (this._loadingEl) this._loadingEl.textContent = "Positioning model...";
opt?.onprogress?.({ lengthComputable: true, loaded: 1, total: 1 });
resolve();
},
(ev) => {
if (ev.lengthComputable) {
const pct = Math.min(100, Math.round((ev.loaded / ev.total) * 100));
if (this._loadingEl) this._loadingEl.textContent = `Loading model ${pct}%`;
}
opt?.onprogress?.(ev);
},
(err) => reject(new Error(`GLTF load: ${err?.message || err}`)),
);
});
}
/** Add lighting for static models (TalkingHead lights are disabled at intensity 0) */
_addStaticLights() {
this._removeStaticLights();
const head = /** @type {TalkingHead} */ (this.head);
const lights = [
new HemisphereLight(0xffffff, 0x444444, 0.7),
new DirectionalLight(0xffffff, 0.9),
];
lights[1].position.set(5, 8, 5);
const fill = new DirectionalLight(0x8888ff, 0.3);
fill.position.set(-3, 2, -4);
lights.push(fill);
for (const l of lights) { head.scene.add(l); this._staticLights.push(l); }
console.log("[AvatarStage] Added", lights.length, "lights for static model");
}
_removeStaticLights() {
for (const l of this._staticLights) {
l.parent?.remove(l);
l.dispose?.();
}
this._staticLights = [];
}
/** Auto-adjust camera + model position based on bounding box */
_adjustCamera(avatarUrl) {
const head = /** @type {TalkingHead} */ (this.head);
if (!head.scene || !head.camera) return;
if (!avatarUrl && !this._staticModel) {
// Default morph model: just update projection
requestAnimationFrame(() => {
try { head.renderer?.render(head.scene, head.camera); head.camera.updateProjectionMatrix?.(); } catch {}
});
return;
}
try {
const box = new Box3().setFromObject(head.scene);
const size = box.getSize(new Vector3());
const center = box.getCenter(new Vector3());
console.log("[AvatarStage] BBox:", { size, center });
if (size.x === 0 && size.y === 0 && size.z === 0) {
console.warn("[AvatarStage] Empty bbox, skipping");
return;
}
if (size.y > 2.0) {
// Full-body: shift up + zoom out
head.scene.position.y = -(center.y - size.y * 0.35);
head.camera.position.z = -Math.max(size.z, size.y, 1.0) * 1.2;
head.camera.position.y = size.y * 0.3;
console.log("[AvatarStage] Full-body adjusted");
} else if (size.y > 0.3) {
// Head/shoulder: center
head.camera.position.z = -Math.max(size.z, 0.5) * 2.5;
head.camera.position.y = center.y * 0.5;
console.log("[AvatarStage] Head model adjusted");
}
requestAnimationFrame(() => {
try { head.renderer?.render(head.scene, head.camera); head.camera.updateProjectionMatrix?.(); } catch {}
});
box.dispose();
} catch (e) {
console.warn("[AvatarStage] Camera adjust:", e);
}
}
async _initLipsync() {
const head = /** @type {TalkingHead} */ (this.head);
if (!head.audioCtx) { console.warn("[AvatarStage] No AudioContext"); return; }
try { await head.audioCtx.audioWorklet.addModule(HEADAUDIO_WORKLET_URL); } catch (e) { console.warn("[AvatarStage] Worklet:", e); return; }
const headaudio = new HeadAudio(head.audioCtx);
await headaudio.loadModel(HEADAUDIO_MODEL_URL);
if (!head.audioSpeechGainNode) { console.warn("[AvatarStage] No audioSpeechGainNode"); return; }
head.audioSpeechGainNode.connect(headaudio);
headaudio.onvalue = (key, value) => {
const mt = head.mtAvatar?.[key];
if (mt) Object.assign(mt, { newvalue: value, needsUpdate: true });
};
head.opt.update = headaudio.update.bind(headaudio);
headaudio.onended = () => { this._lastSpeechEnded = Date.now(); };
headaudio.onstarted = () => {
if (Date.now() - this._lastSpeechEnded > 150) { head.lookAtCamera(500); head.speakWithHands(); }
};
this._headaudio = headaudio;
}
get audioCtx() { return this.head?.audioCtx ?? null; }
get voiceSink() { return this.head?.audioAnalyzerNode ?? null; }
get lastError() { return this._lastError; }
get isStaticModel() { return this._staticModel; }
/**
* Capture the current WebGL frame as a PNG data URL.
* Re-renders one frame then reads the buffer immediately (no await in
* between) so it works even when preserveDrawingBuffer is false.
* @returns {string|null} PNG data URL or null on failure.
*/
capturePNG() {
const head = this.head;
if (!head || !head.renderer || !head.scene || !head.camera) return null;
try {
head.renderer.render(head.scene, head.camera);
const canvas = head.renderer.domElement;
return canvas.toDataURL("image/png");
} catch (e) {
console.warn("[AvatarStage] capturePNG:", e);
return null;
}
}
/**
* Start the avatar animation loop and try to resume AudioContext.
* Returns a Promise that resolves when AudioContext resumes (if called
* from a user-gesture handler) or never (if called from boot).
*/
resume() {
this.head?.start();
if (this.head && this.head.audioCtx.state === "suspended") {
return this.head.audioCtx.resume().catch(() => {});
}
return Promise.resolve();
}
setConversationState(status) {
const h = this.head;
if (!h) return;
switch (status) {
case "user-speaking": h.isSpeaking = false; h.lookAtCamera(800); break;
case "ai-speaking": h.isSpeaking = true; break;
case "processing": h.isSpeaking = false; break;
case "closed": case "error": case "idle": h.isSpeaking = false; h.stopGesture(300); break;
default: h.isSpeaking = false;
}
}
runTool(name, args) {
const h = this.head;
if (!h) return null;
if (this._staticModel) {
if (name === "set_mood") return "Mood unavailable (static model)";
if (name === "make_hand_gesture") return "Gestures unavailable (static model)";
if (name === "make_facial_expression") return "Expressions unavailable (static model)";
}
if (name === "set_mood") {
if (!AVATAR_MOODS.includes(args.mood)) return `Unknown mood: ${args.mood}`;
h.setMood(args.mood); return `Mood set to ${args.mood}.`;
}
if (name === "make_hand_gesture") {
if (!AVATAR_GESTURES.includes(args.gesture)) return `Unknown gesture: ${args.gesture}`;
h.playGesture(args.gesture, 3); return `Playing gesture ${args.gesture}.`;
}
if (name === "make_facial_expression") {
if (!args.emoji) return "No emoji given.";
h.speakEmoji(args.emoji); return `Expressing ${args.emoji}.`;
}
return null;
}
}