新增 apps/api/src/voice/:Voice Sheet(基礎音色/預設語速/音域幅度/口頭 聲響庫/禁則,依性格原型推導預設值,1:1 掛在 Character,同 EmotionState 的關聯模式);情緒→韻律對照(六狀態語速/音高/音量/句尾走向),音高變化 依角色音域幅度縮放(三無角色近乎單音),設計原則同 O-4 表情外顯度縮放; 非語言發聲與節奏(依情緒插入聲響、重大情緒延長停頓、禁則過濾候選、 低親密度不打斷、talkative 特徵+親密度門檻才可插話);TTSProvider 抽象 完全比照 F-1 LLMProvider(MockTTSProvider 輸出韻律標記+佔位音檔, RealTTSProvider 待人工確認供應商後再接,TTS_PROVIDER=mock|real 切換); 語音輸入副語言分析(EmotionService 新增 paralinguisticSignal,優先序 排在作息基線之前,讓「文字說沒事但聲音在抖」判為負向)。 修正一個真實 bug:非語言發聲的停頓時長原本直接拿目前主導情緒的數值 當強度訊號,但平靜狀態下 calm 預設就是 100,導致「什麼事都沒發生」被 誤判成「強度最強的重大情緒事件」而觸發不該有的長停頓——修正為 CALM 一律視為強度 0(平靜是情緒事件的缺席,不是訊號)。 刻意簡化:副語言特徵(語速/音量/顫抖/停頓)採結構化輸入,不做真實音訊 分析,比照 M/N 群組「先用結構化輸入代替真實訊號處理」的一貫取捨。 Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
161 lines
7.5 KiB
JavaScript
161 lines
7.5 KiB
JavaScript
import { prisma } from "@kokorone/db";
|
||
|
||
const API_PORT = process.env.PORT_API ?? "3001";
|
||
const USER_ID = "seed-user-primary";
|
||
const GENKI_ID = "smoke-p-genki"; // 元氣:talkative、音域幅度大
|
||
const COOL_ID = "smoke-p-cool"; // 三無:非 talkative、音域幅度極小
|
||
|
||
async function post(path, body) {
|
||
const res = await fetch(`http://localhost:${API_PORT}${path}`, {
|
||
method: "POST",
|
||
headers: { "Content-Type": "application/json" },
|
||
body: JSON.stringify(body ?? {}),
|
||
});
|
||
if (!res.ok) {
|
||
throw new Error(`POST ${path} 回傳 ${res.status}:${await res.text()}`);
|
||
}
|
||
return res.json();
|
||
}
|
||
|
||
async function get(path) {
|
||
const res = await fetch(`http://localhost:${API_PORT}${path}`);
|
||
if (!res.ok) {
|
||
throw new Error(`GET ${path} 回傳 ${res.status}`);
|
||
}
|
||
return res.json();
|
||
}
|
||
|
||
async function createCharacter(id, archetype, intimacy) {
|
||
await post("/personality/characters", {
|
||
id,
|
||
source: "ORIGINAL",
|
||
buildStatus: "BUILT",
|
||
formalName: id,
|
||
basicInfo: "測試角色",
|
||
backgroundStory: "測試",
|
||
personalityArchetype: archetype,
|
||
likesDislikes: "測試",
|
||
goalsObsessions: "測試",
|
||
speechStyle: "第一人稱「我」",
|
||
initialRelationship: { userId: USER_ID, intimacy, trust: intimacy },
|
||
});
|
||
}
|
||
|
||
async function setEmotion(characterId, tag, value) {
|
||
const dims = { calm: 0, joy: 0, sad: 0, alert: 0, shy: 0, grumpy: 0 };
|
||
dims[tag.toLowerCase()] = value;
|
||
await prisma.emotionState.update({ where: { characterId }, data: dims });
|
||
}
|
||
|
||
export default async function smokeP() {
|
||
await prisma.character.deleteMany({ where: { id: { in: [GENKI_ID, COOL_ID] } } });
|
||
await createCharacter(GENKI_ID, "元氣", 10);
|
||
await createCharacter(COOL_ID, "三無", 10);
|
||
|
||
// P-1 Voice Sheet:可為種子角色寫入完整 Voice Sheet,且預設值依性格原型推導。
|
||
const genkiSheet = await post(`/voice/${GENKI_ID}/voice-sheet/seed-default`, { verbalTics: ["嘿嘿"], forbiddenSounds: ["欸嘿嘿"] });
|
||
const coolSheet = await post(`/voice/${COOL_ID}/voice-sheet/seed-default`, {});
|
||
if (!genkiSheet.baseTimbre || !genkiSheet.baseRate || genkiSheet.pitchRangeSemitones <= 0) {
|
||
throw new Error("Voice Sheet 應該包含基礎音色、預設語速、音域幅度");
|
||
}
|
||
if (!(genkiSheet.baseRate > coolSheet.baseRate) || !(genkiSheet.pitchRangeSemitones > coolSheet.pitchRangeSemitones)) {
|
||
throw new Error("元氣角色的預設語速與音域幅度應該大於三無角色(性格決定基準)");
|
||
}
|
||
|
||
// P-3 情緒 → 韻律對照:六種情緒狀態應該產生不同的韻律組合。
|
||
const emotionTags = ["CALM", "JOY", "SAD", "ALERT", "SHY", "GRUMPY"];
|
||
const prosodySignatures = new Set();
|
||
for (const tag of emotionTags) {
|
||
await setEmotion(GENKI_ID, tag, 80);
|
||
const prosody = await get(`/voice/${GENKI_ID}/${USER_ID}/prosody`);
|
||
if (prosody.emotionTag !== tag) {
|
||
throw new Error(`情緒設為 ${tag} 時,韻律解析應該反映相同的情緒`);
|
||
}
|
||
prosodySignatures.add(`${prosody.rateMultiplier.toFixed(3)}/${prosody.pitchShiftSemitones.toFixed(3)}/${prosody.contour}`);
|
||
}
|
||
if (prosodySignatures.size !== emotionTags.length) {
|
||
throw new Error("六種情緒狀態應該產生彼此不同的韻律標記組合");
|
||
}
|
||
|
||
// P-4 韻律 × 外顯度縮放:同一句話在不同角色身上,音域幅度小的角色音高變化幅度應該遠小於音域幅度大的角色。
|
||
await setEmotion(GENKI_ID, "JOY", 90);
|
||
await setEmotion(COOL_ID, "JOY", 90);
|
||
const genkiProsody = await get(`/voice/${GENKI_ID}/${USER_ID}/prosody`);
|
||
const coolProsody = await get(`/voice/${COOL_ID}/${USER_ID}/prosody`);
|
||
if (!(Math.abs(genkiProsody.pitchShiftSemitones) > Math.abs(coolProsody.pitchShiftSemitones) * 5)) {
|
||
throw new Error("三無角色(音域幅度窄)的音高變化應該遠小於元氣角色——近乎單音");
|
||
}
|
||
|
||
// P-4 TTS Provider 抽象:Mock 供應商應該輸出韻律標記與佔位音檔參照。
|
||
const synthesized = await post(`/voice/${GENKI_ID}/${USER_ID}/synthesize`, { text: "謝謝你!" });
|
||
if (!synthesized.audioRef?.startsWith("mock://") || !synthesized.prosodyMarkup || synthesized.durationEstimateMs <= 0) {
|
||
throw new Error("MockTTSProvider 應該輸出佔位音檔參照、韻律標記字串與時長估算");
|
||
}
|
||
|
||
// P-2 非語言發聲與節奏:不同情緒下插入的聲響與停頓應該不同;重大情緒(高強度)停頓應明顯變長。
|
||
await setEmotion(GENKI_ID, "CALM", 100);
|
||
const calmCue = await get(`/voice/${GENKI_ID}/${USER_ID}/nonverbal-cue`);
|
||
await setEmotion(GENKI_ID, "SAD", 90);
|
||
const sadCue = await get(`/voice/${GENKI_ID}/${USER_ID}/nonverbal-cue`);
|
||
if (calmCue.interjection === sadCue.interjection) {
|
||
throw new Error("不同情緒狀態插入的非語言發聲應該不同");
|
||
}
|
||
if (!(sadCue.pauseSeconds > calmCue.pauseSeconds + 1)) {
|
||
throw new Error("高強度的重大情緒,回答前的停頓應該明顯變長(1~2 秒等級)");
|
||
}
|
||
|
||
// P-1 禁則:標記為禁則的口頭聲響不應該出現在候選結果中。
|
||
await setEmotion(GENKI_ID, "JOY", 90);
|
||
const cueWithForbidden = await get(`/voice/${GENKI_ID}/${USER_ID}/nonverbal-cue`);
|
||
if (cueWithForbidden.interjection === "欸嘿嘿") {
|
||
throw new Error("已標記為禁則的口頭聲響不應該被選中");
|
||
}
|
||
|
||
// P-2 打斷與搶話:低親密度時絕不打斷;需要 talkative 特徵+親密度達門檻才可能插話。
|
||
const lowIntimacyCue = await get(`/voice/${GENKI_ID}/${USER_ID}/nonverbal-cue`);
|
||
if (lowIntimacyCue.canInterrupt) {
|
||
throw new Error("低親密度時不應該允許打斷使用者");
|
||
}
|
||
await prisma.relationship.update({
|
||
where: { characterId_userId: { characterId: GENKI_ID, userId: USER_ID } },
|
||
data: { intimacy: 60 },
|
||
});
|
||
const highIntimacyCue = await get(`/voice/${GENKI_ID}/${USER_ID}/nonverbal-cue`);
|
||
if (!highIntimacyCue.canInterrupt) {
|
||
throw new Error("高親密度+talkative 特徵(元氣)時應該允許打斷");
|
||
}
|
||
await prisma.relationship.update({
|
||
where: { characterId_userId: { characterId: COOL_ID, userId: USER_ID } },
|
||
data: { intimacy: 60 },
|
||
});
|
||
const coolHighIntimacyCue = await get(`/voice/${COOL_ID}/${USER_ID}/nonverbal-cue`);
|
||
if (coolHighIntimacyCue.canInterrupt) {
|
||
throw new Error("三無角色沒有 talkative 特徵,即使高親密度也不應該允許打斷");
|
||
}
|
||
|
||
// P-5 語音輸入與副語言分析:同一段文字配不同副語言特徵,應該得到不同的情緒標記——
|
||
// 「文字說沒事但聲音在抖」要判為負向,即使文字本身沒有任何負向關鍵字。
|
||
await setEmotion(GENKI_ID, "CALM", 100);
|
||
const trembling = await post(`/voice/${GENKI_ID}/analyze-input`, {
|
||
userId: USER_ID,
|
||
text: "我沒事,真的沒事。",
|
||
paralinguistic: { rate: 0.6, volume: 0.3, tremor: 0.7, pauseCount: 3 },
|
||
});
|
||
if (trembling.state.sad <= trembling.state.calm) {
|
||
throw new Error("文字沒有負向關鍵字,但聲音顫抖時應該被判為低落情緒");
|
||
}
|
||
|
||
await setEmotion(GENKI_ID, "CALM", 100);
|
||
const steady = await post(`/voice/${GENKI_ID}/analyze-input`, {
|
||
userId: USER_ID,
|
||
text: "我沒事,真的沒事。",
|
||
paralinguistic: { rate: 1.0, volume: 0.6, tremor: 0.05, pauseCount: 0 },
|
||
});
|
||
if (steady.state.sad > steady.state.calm) {
|
||
throw new Error("同樣的文字,聲音平穩時不應該被誤判為低落情緒");
|
||
}
|
||
|
||
// 清理本次測試建立的角色(cascade 會一併清掉 Voice Sheet、關係、情緒狀態等)。
|
||
await prisma.character.deleteMany({ where: { id: { in: [GENKI_ID, COOL_ID] } } });
|
||
}
|