Files
Kokorone/scripts/smoke/P.mjs
T
JefferyandClaude Sonnet 5 0e9401b891 feat: 完成 P 群組 — 語音子系統
新增 apps/api/src/voice/:Voice Sheet(基礎音色/預設語速/音域幅度/口頭
聲響庫/禁則,依性格原型推導預設值,1:1 掛在 Character,同 EmotionState
的關聯模式);情緒→韻律對照(六狀態語速/音高/音量/句尾走向),音高變化
依角色音域幅度縮放(三無角色近乎單音),設計原則同 O-4 表情外顯度縮放;
非語言發聲與節奏(依情緒插入聲響、重大情緒延長停頓、禁則過濾候選、
低親密度不打斷、talkative 特徵+親密度門檻才可插話);TTSProvider 抽象
完全比照 F-1 LLMProvider(MockTTSProvider 輸出韻律標記+佔位音檔,
RealTTSProvider 待人工確認供應商後再接,TTS_PROVIDER=mock|real 切換);
語音輸入副語言分析(EmotionService 新增 paralinguisticSignal,優先序
排在作息基線之前,讓「文字說沒事但聲音在抖」判為負向)。

修正一個真實 bug:非語言發聲的停頓時長原本直接拿目前主導情緒的數值
當強度訊號,但平靜狀態下 calm 預設就是 100,導致「什麼事都沒發生」被
誤判成「強度最強的重大情緒事件」而觸發不該有的長停頓——修正為 CALM
一律視為強度 0(平靜是情緒事件的缺席,不是訊號)。

刻意簡化:副語言特徵(語速/音量/顫抖/停頓)採結構化輸入,不做真實音訊
分析,比照 M/N 群組「先用結構化輸入代替真實訊號處理」的一貫取捨。

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-08-13 18:19:51 +08:00

161 lines
7.5 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import { prisma } from "@kokorone/db";
const API_PORT = process.env.PORT_API ?? "3001";
const USER_ID = "seed-user-primary";
const GENKI_ID = "smoke-p-genki"; // 元氣:talkative、音域幅度大
const COOL_ID = "smoke-p-cool"; // 三無:非 talkative、音域幅度極小
async function post(path, body) {
const res = await fetch(`http://localhost:${API_PORT}${path}`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify(body ?? {}),
});
if (!res.ok) {
throw new Error(`POST ${path} 回傳 ${res.status}:${await res.text()}`);
}
return res.json();
}
async function get(path) {
const res = await fetch(`http://localhost:${API_PORT}${path}`);
if (!res.ok) {
throw new Error(`GET ${path} 回傳 ${res.status}`);
}
return res.json();
}
async function createCharacter(id, archetype, intimacy) {
await post("/personality/characters", {
id,
source: "ORIGINAL",
buildStatus: "BUILT",
formalName: id,
basicInfo: "測試角色",
backgroundStory: "測試",
personalityArchetype: archetype,
likesDislikes: "測試",
goalsObsessions: "測試",
speechStyle: "第一人稱「我」",
initialRelationship: { userId: USER_ID, intimacy, trust: intimacy },
});
}
async function setEmotion(characterId, tag, value) {
const dims = { calm: 0, joy: 0, sad: 0, alert: 0, shy: 0, grumpy: 0 };
dims[tag.toLowerCase()] = value;
await prisma.emotionState.update({ where: { characterId }, data: dims });
}
export default async function smokeP() {
await prisma.character.deleteMany({ where: { id: { in: [GENKI_ID, COOL_ID] } } });
await createCharacter(GENKI_ID, "元氣", 10);
await createCharacter(COOL_ID, "三無", 10);
// P-1 Voice Sheet:可為種子角色寫入完整 Voice Sheet,且預設值依性格原型推導。
const genkiSheet = await post(`/voice/${GENKI_ID}/voice-sheet/seed-default`, { verbalTics: ["嘿嘿"], forbiddenSounds: ["欸嘿嘿"] });
const coolSheet = await post(`/voice/${COOL_ID}/voice-sheet/seed-default`, {});
if (!genkiSheet.baseTimbre || !genkiSheet.baseRate || genkiSheet.pitchRangeSemitones <= 0) {
throw new Error("Voice Sheet 應該包含基礎音色、預設語速、音域幅度");
}
if (!(genkiSheet.baseRate > coolSheet.baseRate) || !(genkiSheet.pitchRangeSemitones > coolSheet.pitchRangeSemitones)) {
throw new Error("元氣角色的預設語速與音域幅度應該大於三無角色(性格決定基準)");
}
// P-3 情緒 → 韻律對照:六種情緒狀態應該產生不同的韻律組合。
const emotionTags = ["CALM", "JOY", "SAD", "ALERT", "SHY", "GRUMPY"];
const prosodySignatures = new Set();
for (const tag of emotionTags) {
await setEmotion(GENKI_ID, tag, 80);
const prosody = await get(`/voice/${GENKI_ID}/${USER_ID}/prosody`);
if (prosody.emotionTag !== tag) {
throw new Error(`情緒設為 ${tag} 時,韻律解析應該反映相同的情緒`);
}
prosodySignatures.add(`${prosody.rateMultiplier.toFixed(3)}/${prosody.pitchShiftSemitones.toFixed(3)}/${prosody.contour}`);
}
if (prosodySignatures.size !== emotionTags.length) {
throw new Error("六種情緒狀態應該產生彼此不同的韻律標記組合");
}
// P-4 韻律 × 外顯度縮放:同一句話在不同角色身上,音域幅度小的角色音高變化幅度應該遠小於音域幅度大的角色。
await setEmotion(GENKI_ID, "JOY", 90);
await setEmotion(COOL_ID, "JOY", 90);
const genkiProsody = await get(`/voice/${GENKI_ID}/${USER_ID}/prosody`);
const coolProsody = await get(`/voice/${COOL_ID}/${USER_ID}/prosody`);
if (!(Math.abs(genkiProsody.pitchShiftSemitones) > Math.abs(coolProsody.pitchShiftSemitones) * 5)) {
throw new Error("三無角色(音域幅度窄)的音高變化應該遠小於元氣角色——近乎單音");
}
// P-4 TTS Provider 抽象:Mock 供應商應該輸出韻律標記與佔位音檔參照。
const synthesized = await post(`/voice/${GENKI_ID}/${USER_ID}/synthesize`, { text: "謝謝你!" });
if (!synthesized.audioRef?.startsWith("mock://") || !synthesized.prosodyMarkup || synthesized.durationEstimateMs <= 0) {
throw new Error("MockTTSProvider 應該輸出佔位音檔參照、韻律標記字串與時長估算");
}
// P-2 非語言發聲與節奏:不同情緒下插入的聲響與停頓應該不同;重大情緒(高強度)停頓應明顯變長。
await setEmotion(GENKI_ID, "CALM", 100);
const calmCue = await get(`/voice/${GENKI_ID}/${USER_ID}/nonverbal-cue`);
await setEmotion(GENKI_ID, "SAD", 90);
const sadCue = await get(`/voice/${GENKI_ID}/${USER_ID}/nonverbal-cue`);
if (calmCue.interjection === sadCue.interjection) {
throw new Error("不同情緒狀態插入的非語言發聲應該不同");
}
if (!(sadCue.pauseSeconds > calmCue.pauseSeconds + 1)) {
throw new Error("高強度的重大情緒,回答前的停頓應該明顯變長(1~2 秒等級)");
}
// P-1 禁則:標記為禁則的口頭聲響不應該出現在候選結果中。
await setEmotion(GENKI_ID, "JOY", 90);
const cueWithForbidden = await get(`/voice/${GENKI_ID}/${USER_ID}/nonverbal-cue`);
if (cueWithForbidden.interjection === "欸嘿嘿") {
throw new Error("已標記為禁則的口頭聲響不應該被選中");
}
// P-2 打斷與搶話:低親密度時絕不打斷;需要 talkative 特徵+親密度達門檻才可能插話。
const lowIntimacyCue = await get(`/voice/${GENKI_ID}/${USER_ID}/nonverbal-cue`);
if (lowIntimacyCue.canInterrupt) {
throw new Error("低親密度時不應該允許打斷使用者");
}
await prisma.relationship.update({
where: { characterId_userId: { characterId: GENKI_ID, userId: USER_ID } },
data: { intimacy: 60 },
});
const highIntimacyCue = await get(`/voice/${GENKI_ID}/${USER_ID}/nonverbal-cue`);
if (!highIntimacyCue.canInterrupt) {
throw new Error("高親密度+talkative 特徵(元氣)時應該允許打斷");
}
await prisma.relationship.update({
where: { characterId_userId: { characterId: COOL_ID, userId: USER_ID } },
data: { intimacy: 60 },
});
const coolHighIntimacyCue = await get(`/voice/${COOL_ID}/${USER_ID}/nonverbal-cue`);
if (coolHighIntimacyCue.canInterrupt) {
throw new Error("三無角色沒有 talkative 特徵,即使高親密度也不應該允許打斷");
}
// P-5 語音輸入與副語言分析:同一段文字配不同副語言特徵,應該得到不同的情緒標記——
// 「文字說沒事但聲音在抖」要判為負向,即使文字本身沒有任何負向關鍵字。
await setEmotion(GENKI_ID, "CALM", 100);
const trembling = await post(`/voice/${GENKI_ID}/analyze-input`, {
userId: USER_ID,
text: "我沒事,真的沒事。",
paralinguistic: { rate: 0.6, volume: 0.3, tremor: 0.7, pauseCount: 3 },
});
if (trembling.state.sad <= trembling.state.calm) {
throw new Error("文字沒有負向關鍵字,但聲音顫抖時應該被判為低落情緒");
}
await setEmotion(GENKI_ID, "CALM", 100);
const steady = await post(`/voice/${GENKI_ID}/analyze-input`, {
userId: USER_ID,
text: "我沒事,真的沒事。",
paralinguistic: { rate: 1.0, volume: 0.6, tremor: 0.05, pauseCount: 0 },
});
if (steady.state.sad > steady.state.calm) {
throw new Error("同樣的文字,聲音平穩時不應該被誤判為低落情緒");
}
// 清理本次測試建立的角色(cascade 會一併清掉 Voice Sheet、關係、情緒狀態等)。
await prisma.character.deleteMany({ where: { id: { in: [GENKI_ID, COOL_ID] } } });
}