From 0e9401b891b561981b88fae2784b947eae0929e0 Mon Sep 17 00:00:00 2001 From: Jeffery Date: Thu, 13 Aug 2026 18:19:51 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E5=AE=8C=E6=88=90=20P=20=E7=BE=A4?= =?UTF-8?q?=E7=B5=84=20=E2=80=94=20=E8=AA=9E=E9=9F=B3=E5=AD=90=E7=B3=BB?= =?UTF-8?q?=E7=B5=B1?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 新增 apps/api/src/voice/:Voice Sheet(基礎音色/預設語速/音域幅度/口頭 聲響庫/禁則,依性格原型推導預設值,1:1 掛在 Character,同 EmotionState 的關聯模式);情緒→韻律對照(六狀態語速/音高/音量/句尾走向),音高變化 依角色音域幅度縮放(三無角色近乎單音),設計原則同 O-4 表情外顯度縮放; 非語言發聲與節奏(依情緒插入聲響、重大情緒延長停頓、禁則過濾候選、 低親密度不打斷、talkative 特徵+親密度門檻才可插話);TTSProvider 抽象 完全比照 F-1 LLMProvider(MockTTSProvider 輸出韻律標記+佔位音檔, RealTTSProvider 待人工確認供應商後再接,TTS_PROVIDER=mock|real 切換); 語音輸入副語言分析(EmotionService 新增 paralinguisticSignal,優先序 排在作息基線之前,讓「文字說沒事但聲音在抖」判為負向)。 修正一個真實 bug:非語言發聲的停頓時長原本直接拿目前主導情緒的數值 當強度訊號,但平靜狀態下 calm 預設就是 100,導致「什麼事都沒發生」被 誤判成「強度最強的重大情緒事件」而觸發不該有的長停頓——修正為 CALM 一律視為強度 0(平靜是情緒事件的缺席,不是訊號)。 刻意簡化:副語言特徵(語速/音量/顫抖/停頓)採結構化輸入,不做真實音訊 分析,比照 M/N 群組「先用結構化輸入代替真實訊號處理」的一貫取捨。 Co-Authored-By: Claude Sonnet 5 --- apps/api/src/app.module.ts | 2 + apps/api/src/emotion/emotion.service.ts | 7 +- apps/api/src/voice/constants.ts | 63 +++++++ .../src/voice/mock-tts-provider.service.ts | 32 ++++ apps/api/src/voice/nonverbal.service.ts | 60 +++++++ apps/api/src/voice/paralinguistic.service.ts | 31 ++++ apps/api/src/voice/prosody.service.ts | 54 ++++++ .../src/voice/real-tts-provider.service.ts | 11 ++ apps/api/src/voice/tts-provider.ts | 24 +++ apps/api/src/voice/voice-sheet.service.ts | 77 +++++++++ apps/api/src/voice/voice.controller.ts | 95 +++++++++++ apps/api/src/voice/voice.module.ts | 39 +++++ .../migration.sql | 16 ++ prisma/schema.prisma | 17 ++ scripts/smoke/P.mjs | 160 ++++++++++++++++++ todo.md | 21 ++- 16 files changed, 702 insertions(+), 7 deletions(-) create mode 100644 apps/api/src/voice/constants.ts create mode 100644 apps/api/src/voice/mock-tts-provider.service.ts create mode 100644 apps/api/src/voice/nonverbal.service.ts create mode 100644 apps/api/src/voice/paralinguistic.service.ts create mode 100644 apps/api/src/voice/prosody.service.ts create mode 100644 apps/api/src/voice/real-tts-provider.service.ts create mode 100644 apps/api/src/voice/tts-provider.ts create mode 100644 apps/api/src/voice/voice-sheet.service.ts create mode 100644 apps/api/src/voice/voice.controller.ts create mode 100644 apps/api/src/voice/voice.module.ts create mode 100644 prisma/migrations/20260813095700_add_voice_sheet/migration.sql create mode 100644 scripts/smoke/P.mjs diff --git a/apps/api/src/app.module.ts b/apps/api/src/app.module.ts index fed7fc9..e6da653 100644 --- a/apps/api/src/app.module.ts +++ b/apps/api/src/app.module.ts @@ -15,6 +15,7 @@ import { RomanceModule } from "./romance/romance.module.js"; import { CanonModule } from "./canon/canon.module.js"; import { EpilogueModule } from "./epilogue/epilogue.module.js"; import { TachieModule } from "./tachie/tachie.module.js"; +import { VoiceModule } from "./voice/voice.module.js"; @Module({ imports: [ @@ -33,6 +34,7 @@ import { TachieModule } from "./tachie/tachie.module.js"; CanonModule, EpilogueModule, TachieModule, + VoiceModule, ], controllers: [HealthController], }) diff --git a/apps/api/src/emotion/emotion.service.ts b/apps/api/src/emotion/emotion.service.ts index e674d9f..777788d 100644 --- a/apps/api/src/emotion/emotion.service.ts +++ b/apps/api/src/emotion/emotion.service.ts @@ -58,6 +58,9 @@ export interface ProcessInputOptions extends TransitionContext { now?: Date; // I-5:本輪對話沒有明確情緒訊號時,套用作息驅動的基線情緒(不覆蓋文字本身觸發的訊號)。 baselineSignal?: { tag: EmotionTag; intensity: number }; + // P-5:語音輸入的副語言特徵(語速/音量/顫抖/停頓)分析出的訊號——文字說「沒事」但聲音在抖, + // 優先於作息基線(副語言是這一輪輸入本身的訊號,比時段基線更直接),但仍不覆蓋文字本身觸發的訊號。 + paralinguisticSignal?: { tag: EmotionTag; intensity: number }; } function toDomain(characterId: string, dims: Dimensions, updatedAt: Date): EmotionState { @@ -108,7 +111,9 @@ export class EmotionService { const rawSignal = this.tagger.tag({ text, triggerThreshold }); // 情緒觸發閾值除了縮放強度,也要能讓「難觸發」原型對弱刺激完全不觸發(而不只是觸發得比較小力)。 let signal = rawSignal.intensity < MIN_TRIGGER_INTENSITY ? { tag: "CALM" as const, intensity: 0 } : rawSignal; - if (signal.tag === "CALM" && options.baselineSignal) { + if (signal.tag === "CALM" && options.paralinguisticSignal) { + signal = options.paralinguisticSignal; + } else if (signal.tag === "CALM" && options.baselineSignal) { signal = options.baselineSignal; } const next = nextEmotionState(currentDominant, signal, options); diff --git a/apps/api/src/voice/constants.ts b/apps/api/src/voice/constants.ts new file mode 100644 index 0000000..84ed6e3 --- /dev/null +++ b/apps/api/src/voice/constants.ts @@ -0,0 +1,63 @@ +import type { EmotionTag } from "@kokorone/shared"; +import type { Archetype } from "../personality/archetype-params.js"; + +export type SentenceEndContour = "RISING" | "FALLING" | "FLAT"; + +export interface ProsodyTemplate { + rateDeltaPercent: number; // 語速變化百分比,正值變快 + pitchDeltaSemitones: number; // 音高變化(半音),正值變高 + volumeDeltaPercent: number; // 音量變化百分比,正值變大 + contour: SentenceEndContour; + characteristic: string; // 描述性文字,供除錯/前端顯示 +} + +// P-3 情緒 → 韻律對照:情緒狀態機的輸出直接驅動韻律參數。 +export const EMOTION_PROSODY_TABLE: Record = { + CALM: { rateDeltaPercent: 0, pitchDeltaSemitones: 0, volumeDeltaPercent: 0, contour: "FLAT", characteristic: "角色預設值" }, + JOY: { rateDeltaPercent: 15, pitchDeltaSemitones: 1.5, volumeDeltaPercent: 10, contour: "RISING", characteristic: "句尾上揚、偶爾帶笑聲" }, + SAD: { rateDeltaPercent: -20, pitchDeltaSemitones: -1.5, volumeDeltaPercent: -15, contour: "FALLING", characteristic: "句間停頓變長、句尾下沉" }, + ALERT: { rateDeltaPercent: 5, pitchDeltaSemitones: 0, volumeDeltaPercent: 0, contour: "FLAT", characteristic: "語調變平、字句變短" }, + SHY: { rateDeltaPercent: -5, pitchDeltaSemitones: 2, volumeDeltaPercent: -10, contour: "FLAT", characteristic: "結巴、語尾變小聲,忽快忽慢" }, + GRUMPY: { rateDeltaPercent: 10, pitchDeltaSemitones: 1, volumeDeltaPercent: -5, contour: "FALLING", characteristic: "「哼」開頭,先大後小" }, +}; + +// P-1 Voice Sheet 預設值:依性格原型推導基準語速與音域幅度(同 G-5「原型 → 種子預設」的作法)。 +export const DEFAULT_VOICE_SHEET_BY_ARCHETYPE: Record = { + 傲嬌: { baseTimbre: "清亮少女音", baseRate: 1.05, pitchRangeSemitones: 5 }, + 冷淡: { baseTimbre: "低沉平穩音", baseRate: 0.85, pitchRangeSemitones: 1.5 }, + 天然呆: { baseTimbre: "軟糯蘿莉音", baseRate: 0.95, pitchRangeSemitones: 4 }, + 元氣: { baseTimbre: "清亮少女音", baseRate: 1.2, pitchRangeSemitones: 6 }, + 大小姐: { baseTimbre: "低沉御姐音", baseRate: 0.95, pitchRangeSemitones: 3.5 }, + 三無: { baseTimbre: "軟糯蘿莉音", baseRate: 0.8, pitchRangeSemitones: 0.5 }, +}; +const DEFAULT_VOICE_SHEET_FALLBACK = { baseTimbre: "清亮少女音", baseRate: 1.0, pitchRangeSemitones: 3 }; +export function defaultVoiceSheetFor(archetype: string) { + return DEFAULT_VOICE_SHEET_BY_ARCHETYPE[archetype as Archetype] ?? DEFAULT_VOICE_SHEET_FALLBACK; +} + +// P-2 非語言發聲庫:依情緒狀態的預設候選詞,角色的 Voice Sheet 口頭聲響庫可補充、禁則可過濾排除。 +export const DEFAULT_INTERJECTIONS_BY_EMOTION: Record = { + CALM: ["嗯"], + JOY: ["欸嘿嘿", "呵呵"], + SAD: ["唔……", "……"], + ALERT: ["嗯?", "欸?!"], + SHY: ["那個……就是……", "唔……"], + GRUMPY: ["哼!", "哼……"], +}; + +// P-2 停頓:情緒強度達門檻(沿用 C-4 高情緒事件的判斷門檻)且屬於重大話題時,回答前故意停頓 1~2 秒。 +export const SIGNIFICANT_PAUSE_THRESHOLD = 0.6; +export const SIGNIFICANT_PAUSE_SECONDS = [1, 2]; +export const DEFAULT_PAUSE_SECONDS = 0.2; + +// P-2 打斷與搶話:需要「talkative」特徵(見 archetype-params.ts)且親密度達朋友分層以上才可能插話。 +export const INTERRUPT_MIN_INTIMACY = 40; + +// P-2 親密度影響音量與氣音比例:對陌生人聲音端正清晰,對摯友放鬆、氣音比例上升。 +export function intimacyVoiceModifiers(intimacy: number): { volumeMultiplier: number; breathinessRatio: number } { + const clamped = Math.max(0, Math.min(100, intimacy)); + return { + volumeMultiplier: 1 - (clamped / 100) * 0.15, // 端正清晰(1.0) → 放鬆慵懶(0.85) + breathinessRatio: clamped / 100, + }; +} diff --git a/apps/api/src/voice/mock-tts-provider.service.ts b/apps/api/src/voice/mock-tts-provider.service.ts new file mode 100644 index 0000000..cb74b04 --- /dev/null +++ b/apps/api/src/voice/mock-tts-provider.service.ts @@ -0,0 +1,32 @@ +import { Injectable } from "@nestjs/common"; +import type { TTSProvider, TTSSynthesisInput, SynthesizedAudio } from "./tts-provider.js"; +import { hashToSeed } from "../llm/seeded-random.js"; + +const CHARS_PER_SECOND_AT_NORMAL_RATE = 6; // Mock 階段的粗略估算,不代表真實語速 + +// P-4 MockTTSProvider:輸出韻律標記與佔位音檔參照,不呼叫任何真實語音服務—— +// 讓語音管線的其他機制(韻律解析、非語言發聲、副語言分析)都能在沒有真實 TTS 供應商的情況下完整測試。 +@Injectable() +export class MockTTSProvider implements TTSProvider { + async synthesize(input: TTSSynthesisInput): Promise { + const seed = input.seed ?? hashToSeed(input.text); + const { prosody, nonverbalCue, voiceSheet } = input; + + const spokenText = nonverbalCue.interjection ? `${nonverbalCue.interjection} ${input.text}` : input.text; + const durationEstimateMs = + nonverbalCue.pauseSeconds * 1000 + (spokenText.length / CHARS_PER_SECOND_AT_NORMAL_RATE / prosody.rateMultiplier) * 1000; + + const prosodyMarkup = + `` + + `` + + `` + + spokenText + + ``; + + return { + audioRef: `mock://tts/${seed}`, + prosodyMarkup, + durationEstimateMs: Math.round(durationEstimateMs), + }; + } +} diff --git a/apps/api/src/voice/nonverbal.service.ts b/apps/api/src/voice/nonverbal.service.ts new file mode 100644 index 0000000..c6d0aad --- /dev/null +++ b/apps/api/src/voice/nonverbal.service.ts @@ -0,0 +1,60 @@ +import { Injectable } from "@nestjs/common"; +import type { EmotionTag } from "@kokorone/shared"; +import { PrismaService } from "../prisma/prisma.service.js"; +import { EmotionService, dominantState } from "../emotion/emotion.service.js"; +import { RelationshipService } from "../relationship/relationship.service.js"; +import { getArchetypeParams } from "../personality/archetype-params.js"; +import { VoiceSheetService } from "./voice-sheet.service.js"; +import { + DEFAULT_INTERJECTIONS_BY_EMOTION, + SIGNIFICANT_PAUSE_THRESHOLD, + SIGNIFICANT_PAUSE_SECONDS, + DEFAULT_PAUSE_SECONDS, + INTERRUPT_MIN_INTIMACY, +} from "./constants.js"; + +export interface NonverbalCue { + interjection: string | null; + pauseSeconds: number; + canInterrupt: boolean; +} + +function fieldFor(tag: EmotionTag): "calm" | "joy" | "sad" | "alert" | "shy" | "grumpy" { + return tag.toLowerCase() as "calm" | "joy" | "sad" | "alert" | "shy" | "grumpy"; +} + +// P-2 對話節奏與非語言發聲:依情緒插入「嗯?」「唔……」等聲響與停頓,親密度與性格決定能不能插話。 +@Injectable() +export class NonverbalService { + constructor( + private readonly prisma: PrismaService, + private readonly emotion: EmotionService, + private readonly relationship: RelationshipService, + private readonly voiceSheet: VoiceSheetService, + ) {} + + async resolveCue(characterId: string, userId: string, now: Date = new Date()): Promise { + const character = await this.prisma.client.character.findUniqueOrThrow({ where: { id: characterId } }); + const sheet = await this.voiceSheet.getOrSeedDefault(characterId); + const emotionState = await this.emotion.getState(characterId, now); + const emotionTag = dominantState(emotionState); + // 平靜本身不是「重大情緒事件」——calm 的數值代表「沒有明顯情緒」,不能當成訊號強度, + // 否則平靜時 calm=100 會被誤判成最強訊號,反而觸發本該只留給強烈情緒的長停頓。 + const intensity = emotionTag === "CALM" ? 0 : (emotionState[fieldFor(emotionTag)] as number) / 100; + + const candidates = [...DEFAULT_INTERJECTIONS_BY_EMOTION[emotionTag], ...sheet.verbalTics]; + const allowed = candidates.filter((phrase) => !sheet.forbiddenSounds.includes(phrase)); + const interjection = allowed.length > 0 ? allowed[0] : null; + + const pauseSeconds = + intensity >= SIGNIFICANT_PAUSE_THRESHOLD + ? SIGNIFICANT_PAUSE_SECONDS[0] + (SIGNIFICANT_PAUSE_SECONDS[1] - SIGNIFICANT_PAUSE_SECONDS[0]) * intensity + : DEFAULT_PAUSE_SECONDS; + + const { relationship } = await this.relationship.getState(characterId, userId, now); + const archetypeParams = getArchetypeParams(character.personalityArchetype); + const canInterrupt = archetypeParams.traits.includes("talkative") && relationship.intimacy >= INTERRUPT_MIN_INTIMACY; + + return { interjection, pauseSeconds, canInterrupt }; + } +} diff --git a/apps/api/src/voice/paralinguistic.service.ts b/apps/api/src/voice/paralinguistic.service.ts new file mode 100644 index 0000000..c85c477 --- /dev/null +++ b/apps/api/src/voice/paralinguistic.service.ts @@ -0,0 +1,31 @@ +import { Injectable } from "@nestjs/common"; +import type { EmotionTag } from "@kokorone/shared"; + +export interface ParalinguisticFeatures { + rate: number; // 相對語速,1.0 為正常,<1 變慢、>1 變快 + volume: number; // 0~1,音量大小 + tremor: number; // 0~1,聲音顫抖程度 + pauseCount: number; // 這段話裡的停頓次數 +} + +function clamp01(value: number): number { + return Math.max(0, Math.min(1, value)); +} + +// P-5 副語言分析:STT 轉出文字之外,語速/音量/顫抖/停頓等特徵也送入情緒標記器—— +// 「文字說沒事但聲音在抖」應被偵測為負向情緒,即使文字本身沒有任何負向關鍵字。 +@Injectable() +export class ParalinguisticService { + deriveSignal(features: ParalinguisticFeatures): { tag: EmotionTag; intensity: number } | null { + const slowAndQuiet = features.rate <= 0.75 && features.volume <= 0.45; + if (features.tremor >= 0.4 || slowAndQuiet) { + const intensity = clamp01(features.tremor * 0.6 + (1 - features.rate) * 0.25 + (1 - features.volume) * 0.25); + return { tag: "SAD", intensity: Math.max(intensity, 0.3) }; + } + if (features.rate >= 1.3 && features.volume >= 0.7 && features.tremor < 0.2) { + const intensity = clamp01((features.rate - 1) * 0.6 + features.volume * 0.3); + return { tag: "JOY", intensity: Math.max(intensity, 0.3) }; + } + return null; // 沒有明顯的副語言訊號,交給文字本身的情緒標記結果決定。 + } +} diff --git a/apps/api/src/voice/prosody.service.ts b/apps/api/src/voice/prosody.service.ts new file mode 100644 index 0000000..a076a7d --- /dev/null +++ b/apps/api/src/voice/prosody.service.ts @@ -0,0 +1,54 @@ +import { Injectable } from "@nestjs/common"; +import { EmotionService, dominantState } from "../emotion/emotion.service.js"; +import { RelationshipService } from "../relationship/relationship.service.js"; +import { VoiceSheetService } from "./voice-sheet.service.js"; +import { EMOTION_PROSODY_TABLE, intimacyVoiceModifiers, type SentenceEndContour } from "./constants.js"; +import type { EmotionTag } from "@kokorone/shared"; + +const REFERENCE_PITCH_RANGE = 4; // 對應 constants.ts 的通用預設音域幅度,作為縮放基準 + +export interface ResolvedProsody { + emotionTag: EmotionTag; + rateMultiplier: number; // 相對 1.0 的語速倍率(已套入角色基準語速) + pitchShiftSemitones: number; // 已依角色音域幅度縮放的音高變化 + volumeMultiplier: number; // 已套入親密度調整的音量倍率 + breathinessRatio: number; // 氣音比例(親密度越高越高) + contour: SentenceEndContour; + characteristic: string; +} + +// P-3 語音標記層:情緒狀態機輸出直接驅動韻律參數,與韻律驅動同源(跟 O 群組表情解析同一批輸入)。 +@Injectable() +export class ProsodyService { + constructor( + private readonly emotion: EmotionService, + private readonly relationship: RelationshipService, + private readonly voiceSheet: VoiceSheetService, + ) {} + + async resolveProsody(characterId: string, userId: string, now: Date = new Date()): Promise { + const emotionState = await this.emotion.getState(characterId, now); + const emotionTag = dominantState(emotionState); + const template = EMOTION_PROSODY_TABLE[emotionTag]; + + const sheet = await this.voiceSheet.getOrSeedDefault(characterId); + const { relationship } = await this.relationship.getState(characterId, userId, now); + const { volumeMultiplier: intimacyVolume, breathinessRatio } = intimacyVoiceModifiers(relationship.intimacy); + + const rateMultiplier = sheet.baseRate * (1 + template.rateDeltaPercent / 100); + // 音域幅度縮放:跟 O-4 表情外顯度縮放同一個設計原則——角色的音域幅度越窄(三無), + // 同一個情緒帶來的音高變化就越不明顯(近乎單音)。 + const pitchShiftSemitones = template.pitchDeltaSemitones * (sheet.pitchRangeSemitones / REFERENCE_PITCH_RANGE); + const volumeMultiplier = intimacyVolume * (1 + template.volumeDeltaPercent / 100); + + return { + emotionTag, + rateMultiplier, + pitchShiftSemitones, + volumeMultiplier, + breathinessRatio, + contour: template.contour, + characteristic: template.characteristic, + }; + } +} diff --git a/apps/api/src/voice/real-tts-provider.service.ts b/apps/api/src/voice/real-tts-provider.service.ts new file mode 100644 index 0000000..6563bfb --- /dev/null +++ b/apps/api/src/voice/real-tts-provider.service.ts @@ -0,0 +1,11 @@ +import { Injectable } from "@nestjs/common"; +import type { TTSProvider, TTSSynthesisInput, SynthesizedAudio } from "./tts-provider.js"; + +// P-4 真實 TTS 供應商:待人工確認要接哪一家服務(技術選型文件尚未拍板),先留下同介面的空殼, +// 呼叫到才報錯,不在啟動時就讓整個服務炸掉——跟 F 群組 ClaudeProvider 的處理方式一致。 +@Injectable() +export class RealTTSProvider implements TTSProvider { + async synthesize(_input: TTSSynthesisInput): Promise { + throw new Error("RealTTSProvider 尚未實作:真實語音供應商待人工確認後再接(見技術選型「TTS 服務」)"); + } +} diff --git a/apps/api/src/voice/tts-provider.ts b/apps/api/src/voice/tts-provider.ts new file mode 100644 index 0000000..afe38a2 --- /dev/null +++ b/apps/api/src/voice/tts-provider.ts @@ -0,0 +1,24 @@ +import type { ResolvedProsody } from "./prosody.service.js"; +import type { NonverbalCue } from "./nonverbal.service.js"; +import type { VoiceSheetView } from "./voice-sheet.service.js"; + +export interface TTSSynthesisInput { + text: string; + voiceSheet: VoiceSheetView; + prosody: ResolvedProsody; + nonverbalCue: NonverbalCue; + seed?: number; +} + +export interface SynthesizedAudio { + audioRef: string; // 音檔參照(Mock 階段是佔位字串,不是真的音檔) + prosodyMarkup: string; // 類 SSML 的除錯用韻律標記字串 + durationEstimateMs: number; +} + +// P-4 TTSProvider 抽象:比照 F-1 LLMProvider 的介面設計,切換 Provider 不動呼叫端任何一行。 +export interface TTSProvider { + synthesize(input: TTSSynthesisInput): Promise; +} + +export const TTS_PROVIDER = Symbol("TTS_PROVIDER"); diff --git a/apps/api/src/voice/voice-sheet.service.ts b/apps/api/src/voice/voice-sheet.service.ts new file mode 100644 index 0000000..b6cca49 --- /dev/null +++ b/apps/api/src/voice/voice-sheet.service.ts @@ -0,0 +1,77 @@ +import { Injectable } from "@nestjs/common"; +import { PrismaService } from "../prisma/prisma.service.js"; +import { defaultVoiceSheetFor } from "./constants.js"; + +export interface VoiceSheetView { + characterId: string; + baseTimbre: string; + baseRate: number; + pitchRangeSemitones: number; + verbalTics: string[]; + forbiddenSounds: string[]; +} + +function toView(row: { + characterId: string; + baseTimbre: string; + baseRate: number; + pitchRangeSemitones: number; + verbalTicsJson: string; + forbiddenSoundsJson: string; +}): VoiceSheetView { + return { + characterId: row.characterId, + baseTimbre: row.baseTimbre, + baseRate: row.baseRate, + pitchRangeSemitones: row.pitchRangeSemitones, + verbalTics: JSON.parse(row.verbalTicsJson), + forbiddenSounds: JSON.parse(row.forbiddenSoundsJson), + }; +} + +// P-1 角色音色設定:定義基礎音色、預設語速、音域幅度、口頭聲響庫與禁則,掛在 Character 上。 +@Injectable() +export class VoiceSheetService { + constructor(private readonly prisma: PrismaService) {} + + // 依性格原型推導預設值後寫入(同 G-5「原型 → 種子預設」的作法),可重複執行(upsert)。 + async seedDefault( + characterId: string, + overrides: { verbalTics?: string[]; forbiddenSounds?: string[] } = {}, + ): Promise { + const character = await this.prisma.client.character.findUniqueOrThrow({ where: { id: characterId } }); + const defaults = defaultVoiceSheetFor(character.personalityArchetype); + const row = await this.prisma.client.voiceSheet.upsert({ + where: { characterId }, + update: { + baseTimbre: defaults.baseTimbre, + baseRate: defaults.baseRate, + pitchRangeSemitones: defaults.pitchRangeSemitones, + verbalTicsJson: JSON.stringify(overrides.verbalTics ?? []), + forbiddenSoundsJson: JSON.stringify(overrides.forbiddenSounds ?? []), + }, + create: { + characterId, + baseTimbre: defaults.baseTimbre, + baseRate: defaults.baseRate, + pitchRangeSemitones: defaults.pitchRangeSemitones, + verbalTicsJson: JSON.stringify(overrides.verbalTics ?? []), + forbiddenSoundsJson: JSON.stringify(overrides.forbiddenSounds ?? []), + }, + }); + return toView(row); + } + + async get(characterId: string): Promise { + const row = await this.prisma.client.voiceSheet.findUnique({ where: { characterId } }); + return row ? toView(row) : null; + } + + async getOrSeedDefault(characterId: string): Promise { + const existing = await this.get(characterId); + if (existing) { + return existing; + } + return this.seedDefault(characterId); + } +} diff --git a/apps/api/src/voice/voice.controller.ts b/apps/api/src/voice/voice.controller.ts new file mode 100644 index 0000000..40eec80 --- /dev/null +++ b/apps/api/src/voice/voice.controller.ts @@ -0,0 +1,95 @@ +import { Body, Controller, Get, Inject, Param, Post, Query } from "@nestjs/common"; +import { VoiceSheetService } from "./voice-sheet.service.js"; +import { ProsodyService } from "./prosody.service.js"; +import { NonverbalService } from "./nonverbal.service.js"; +import { ParalinguisticService, type ParalinguisticFeatures } from "./paralinguistic.service.js"; +import { TTS_PROVIDER, type TTSProvider } from "./tts-provider.js"; +import { EmotionService } from "../emotion/emotion.service.js"; + +interface SeedVoiceSheetBody { + verbalTics?: string[]; + forbiddenSounds?: string[]; +} + +interface SynthesizeBody { + text: string; + now?: string; + seed?: number; +} + +interface AnalyzeInputBody { + userId: string; + text: string; + paralinguistic: ParalinguisticFeatures; + now?: string; +} + +@Controller("voice") +export class VoiceController { + constructor( + private readonly voiceSheet: VoiceSheetService, + private readonly prosody: ProsodyService, + private readonly nonverbal: NonverbalService, + private readonly paralinguistic: ParalinguisticService, + private readonly emotion: EmotionService, + @Inject(TTS_PROVIDER) private readonly ttsProvider: TTSProvider, + ) {} + + @Post(":characterId/voice-sheet/seed-default") + async seedVoiceSheet(@Param("characterId") characterId: string, @Body() body: SeedVoiceSheetBody) { + return this.voiceSheet.seedDefault(characterId, body); + } + + @Get(":characterId/voice-sheet") + async getVoiceSheet(@Param("characterId") characterId: string) { + return this.voiceSheet.get(characterId); + } + + @Get(":characterId/:userId/prosody") + async getProsody( + @Param("characterId") characterId: string, + @Param("userId") userId: string, + @Query("now") now?: string, + ) { + return this.prosody.resolveProsody(characterId, userId, now ? new Date(now) : undefined); + } + + @Get(":characterId/:userId/nonverbal-cue") + async getNonverbalCue( + @Param("characterId") characterId: string, + @Param("userId") userId: string, + @Query("now") now?: string, + ) { + return this.nonverbal.resolveCue(characterId, userId, now ? new Date(now) : undefined); + } + + @Post(":characterId/:userId/synthesize") + async synthesize( + @Param("characterId") characterId: string, + @Param("userId") userId: string, + @Body() body: SynthesizeBody, + ) { + const now = body.now ? new Date(body.now) : undefined; + const [voiceSheet, resolvedProsody, nonverbalCue] = await Promise.all([ + this.voiceSheet.getOrSeedDefault(characterId), + this.prosody.resolveProsody(characterId, userId, now), + this.nonverbal.resolveCue(characterId, userId, now), + ]); + return this.ttsProvider.synthesize({ + text: body.text, + voiceSheet, + prosody: resolvedProsody, + nonverbalCue, + seed: body.seed, + }); + } + + // P-5 驗收用端點:同一段文字配不同副語言特徵,應該得到不同的情緒標記結果。 + @Post(":characterId/analyze-input") + async analyzeInput(@Param("characterId") characterId: string, @Body() body: AnalyzeInputBody) { + const now = body.now ? new Date(body.now) : new Date(); + const paralinguisticSignal = this.paralinguistic.deriveSignal(body.paralinguistic) ?? undefined; + const result = await this.emotion.processInput(characterId, body.text, { now, paralinguisticSignal }); + return { ...result, paralinguisticSignal: paralinguisticSignal ?? null }; + } +} diff --git a/apps/api/src/voice/voice.module.ts b/apps/api/src/voice/voice.module.ts new file mode 100644 index 0000000..99a13ed --- /dev/null +++ b/apps/api/src/voice/voice.module.ts @@ -0,0 +1,39 @@ +import { Module } from "@nestjs/common"; +import { log } from "@kokorone/shared"; +import { PrismaModule } from "../prisma/prisma.module.js"; +import { EmotionModule } from "../emotion/emotion.module.js"; +import { RelationshipModule } from "../relationship/relationship.module.js"; +import { VoiceSheetService } from "./voice-sheet.service.js"; +import { ProsodyService } from "./prosody.service.js"; +import { NonverbalService } from "./nonverbal.service.js"; +import { ParalinguisticService } from "./paralinguistic.service.js"; +import { MockTTSProvider } from "./mock-tts-provider.service.js"; +import { RealTTSProvider } from "./real-tts-provider.service.js"; +import { TTS_PROVIDER } from "./tts-provider.js"; +import { VoiceController } from "./voice.controller.js"; + +// P-4 Provider 切換:TTS_PROVIDER=mock|real 決定注入哪個實作,比照 F-2 LLM_PROVIDER 的作法。 +function ttsProviderFactory(mock: MockTTSProvider, real: RealTTSProvider) { + const providerName = process.env.TTS_PROVIDER ?? "mock"; + if (providerName === "real") { + log("啟動", "ERR", "TTS_PROVIDER=real 但 RealTTSProvider 尚未實作,語音合成呼叫時會拋出例外"); + return real; + } + return mock; +} + +@Module({ + imports: [PrismaModule, EmotionModule, RelationshipModule], + controllers: [VoiceController], + providers: [ + VoiceSheetService, + ProsodyService, + NonverbalService, + ParalinguisticService, + MockTTSProvider, + RealTTSProvider, + { provide: TTS_PROVIDER, useFactory: ttsProviderFactory, inject: [MockTTSProvider, RealTTSProvider] }, + ], + exports: [VoiceSheetService, ProsodyService, NonverbalService, ParalinguisticService, TTS_PROVIDER], +}) +export class VoiceModule {} diff --git a/prisma/migrations/20260813095700_add_voice_sheet/migration.sql b/prisma/migrations/20260813095700_add_voice_sheet/migration.sql new file mode 100644 index 0000000..300f0cb --- /dev/null +++ b/prisma/migrations/20260813095700_add_voice_sheet/migration.sql @@ -0,0 +1,16 @@ +-- CreateTable +CREATE TABLE "voice_sheets" ( + "id" TEXT NOT NULL PRIMARY KEY, + "characterId" TEXT NOT NULL, + "baseTimbre" TEXT NOT NULL, + "baseRate" REAL NOT NULL DEFAULT 1.0, + "pitchRangeSemitones" REAL NOT NULL DEFAULT 4, + "verbalTicsJson" TEXT NOT NULL DEFAULT '[]', + "forbiddenSoundsJson" TEXT NOT NULL DEFAULT '[]', + "createdAt" DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP, + "updatedAt" DATETIME NOT NULL, + CONSTRAINT "voice_sheets_characterId_fkey" FOREIGN KEY ("characterId") REFERENCES "characters" ("id") ON DELETE CASCADE ON UPDATE CASCADE +); + +-- CreateIndex +CREATE UNIQUE INDEX "voice_sheets_characterId_key" ON "voice_sheets"("characterId"); diff --git a/prisma/schema.prisma b/prisma/schema.prisma index 74fd5f9..41a14ec 100644 --- a/prisma/schema.prisma +++ b/prisma/schema.prisma @@ -193,10 +193,27 @@ model Character { relationshipEventLogsFrom CharacterRelationshipEventLog[] @relation("EventLogFrom") relationshipEventLogsTo CharacterRelationshipEventLog[] @relation("EventLogTo") epilogueMilestoneLogs EpilogueMilestoneLog[] + voiceSheet VoiceSheet? @@map("characters") } +// P-1 角色音色設定:語音子系統的角色專屬設定,同 EmotionState 走 1:1 關聯。 +model VoiceSheet { + id String @id @default(cuid()) + characterId String @unique + character Character @relation(fields: [characterId], references: [id], onDelete: Cascade) + baseTimbre String // 基礎音色描述,例如「清亮少女音」「低沉御姐音」「軟糯蘿莉音」 + baseRate Float @default(1.0) // 預設語速倍率:性格決定的基準(元氣快、冷淡慢) + pitchRangeSemitones Float @default(4) // 音域幅度(半音數):語調起伏大小,元氣大、三無近乎單音 + verbalTicsJson String @default("[]") // 口頭聲響庫(JSON 字串陣列):SQLite 不支援原生字串陣列欄位 + forbiddenSoundsJson String @default("[]") // 禁則(JSON 字串陣列):此角色不會有的聲音表現 + createdAt DateTime @default(now()) + updatedAt DateTime @updatedAt + + @@map("voice_sheets") +} + // L-6 角色年齡判定的原始證據登錄;由 AgeDeterminationService 依證據優先序算出年齡狀態,不直接存最終結論 // ——結論永遠是「依目前登錄的證據重新算出來」,避免證據更新後忘記同步結論欄位。 model CharacterAgeEvidence { diff --git a/scripts/smoke/P.mjs b/scripts/smoke/P.mjs new file mode 100644 index 0000000..c840732 --- /dev/null +++ b/scripts/smoke/P.mjs @@ -0,0 +1,160 @@ +import { prisma } from "@kokorone/db"; + +const API_PORT = process.env.PORT_API ?? "3001"; +const USER_ID = "seed-user-primary"; +const GENKI_ID = "smoke-p-genki"; // 元氣:talkative、音域幅度大 +const COOL_ID = "smoke-p-cool"; // 三無:非 talkative、音域幅度極小 + +async function post(path, body) { + const res = await fetch(`http://localhost:${API_PORT}${path}`, { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify(body ?? {}), + }); + if (!res.ok) { + throw new Error(`POST ${path} 回傳 ${res.status}:${await res.text()}`); + } + return res.json(); +} + +async function get(path) { + const res = await fetch(`http://localhost:${API_PORT}${path}`); + if (!res.ok) { + throw new Error(`GET ${path} 回傳 ${res.status}`); + } + return res.json(); +} + +async function createCharacter(id, archetype, intimacy) { + await post("/personality/characters", { + id, + source: "ORIGINAL", + buildStatus: "BUILT", + formalName: id, + basicInfo: "測試角色", + backgroundStory: "測試", + personalityArchetype: archetype, + likesDislikes: "測試", + goalsObsessions: "測試", + speechStyle: "第一人稱「我」", + initialRelationship: { userId: USER_ID, intimacy, trust: intimacy }, + }); +} + +async function setEmotion(characterId, tag, value) { + const dims = { calm: 0, joy: 0, sad: 0, alert: 0, shy: 0, grumpy: 0 }; + dims[tag.toLowerCase()] = value; + await prisma.emotionState.update({ where: { characterId }, data: dims }); +} + +export default async function smokeP() { + await prisma.character.deleteMany({ where: { id: { in: [GENKI_ID, COOL_ID] } } }); + await createCharacter(GENKI_ID, "元氣", 10); + await createCharacter(COOL_ID, "三無", 10); + + // P-1 Voice Sheet:可為種子角色寫入完整 Voice Sheet,且預設值依性格原型推導。 + const genkiSheet = await post(`/voice/${GENKI_ID}/voice-sheet/seed-default`, { verbalTics: ["嘿嘿"], forbiddenSounds: ["欸嘿嘿"] }); + const coolSheet = await post(`/voice/${COOL_ID}/voice-sheet/seed-default`, {}); + if (!genkiSheet.baseTimbre || !genkiSheet.baseRate || genkiSheet.pitchRangeSemitones <= 0) { + throw new Error("Voice Sheet 應該包含基礎音色、預設語速、音域幅度"); + } + if (!(genkiSheet.baseRate > coolSheet.baseRate) || !(genkiSheet.pitchRangeSemitones > coolSheet.pitchRangeSemitones)) { + throw new Error("元氣角色的預設語速與音域幅度應該大於三無角色(性格決定基準)"); + } + + // P-3 情緒 → 韻律對照:六種情緒狀態應該產生不同的韻律組合。 + const emotionTags = ["CALM", "JOY", "SAD", "ALERT", "SHY", "GRUMPY"]; + const prosodySignatures = new Set(); + for (const tag of emotionTags) { + await setEmotion(GENKI_ID, tag, 80); + const prosody = await get(`/voice/${GENKI_ID}/${USER_ID}/prosody`); + if (prosody.emotionTag !== tag) { + throw new Error(`情緒設為 ${tag} 時,韻律解析應該反映相同的情緒`); + } + prosodySignatures.add(`${prosody.rateMultiplier.toFixed(3)}/${prosody.pitchShiftSemitones.toFixed(3)}/${prosody.contour}`); + } + if (prosodySignatures.size !== emotionTags.length) { + throw new Error("六種情緒狀態應該產生彼此不同的韻律標記組合"); + } + + // P-4 韻律 × 外顯度縮放:同一句話在不同角色身上,音域幅度小的角色音高變化幅度應該遠小於音域幅度大的角色。 + await setEmotion(GENKI_ID, "JOY", 90); + await setEmotion(COOL_ID, "JOY", 90); + const genkiProsody = await get(`/voice/${GENKI_ID}/${USER_ID}/prosody`); + const coolProsody = await get(`/voice/${COOL_ID}/${USER_ID}/prosody`); + if (!(Math.abs(genkiProsody.pitchShiftSemitones) > Math.abs(coolProsody.pitchShiftSemitones) * 5)) { + throw new Error("三無角色(音域幅度窄)的音高變化應該遠小於元氣角色——近乎單音"); + } + + // P-4 TTS Provider 抽象:Mock 供應商應該輸出韻律標記與佔位音檔參照。 + const synthesized = await post(`/voice/${GENKI_ID}/${USER_ID}/synthesize`, { text: "謝謝你!" }); + if (!synthesized.audioRef?.startsWith("mock://") || !synthesized.prosodyMarkup || synthesized.durationEstimateMs <= 0) { + throw new Error("MockTTSProvider 應該輸出佔位音檔參照、韻律標記字串與時長估算"); + } + + // P-2 非語言發聲與節奏:不同情緒下插入的聲響與停頓應該不同;重大情緒(高強度)停頓應明顯變長。 + await setEmotion(GENKI_ID, "CALM", 100); + const calmCue = await get(`/voice/${GENKI_ID}/${USER_ID}/nonverbal-cue`); + await setEmotion(GENKI_ID, "SAD", 90); + const sadCue = await get(`/voice/${GENKI_ID}/${USER_ID}/nonverbal-cue`); + if (calmCue.interjection === sadCue.interjection) { + throw new Error("不同情緒狀態插入的非語言發聲應該不同"); + } + if (!(sadCue.pauseSeconds > calmCue.pauseSeconds + 1)) { + throw new Error("高強度的重大情緒,回答前的停頓應該明顯變長(1~2 秒等級)"); + } + + // P-1 禁則:標記為禁則的口頭聲響不應該出現在候選結果中。 + await setEmotion(GENKI_ID, "JOY", 90); + const cueWithForbidden = await get(`/voice/${GENKI_ID}/${USER_ID}/nonverbal-cue`); + if (cueWithForbidden.interjection === "欸嘿嘿") { + throw new Error("已標記為禁則的口頭聲響不應該被選中"); + } + + // P-2 打斷與搶話:低親密度時絕不打斷;需要 talkative 特徵+親密度達門檻才可能插話。 + const lowIntimacyCue = await get(`/voice/${GENKI_ID}/${USER_ID}/nonverbal-cue`); + if (lowIntimacyCue.canInterrupt) { + throw new Error("低親密度時不應該允許打斷使用者"); + } + await prisma.relationship.update({ + where: { characterId_userId: { characterId: GENKI_ID, userId: USER_ID } }, + data: { intimacy: 60 }, + }); + const highIntimacyCue = await get(`/voice/${GENKI_ID}/${USER_ID}/nonverbal-cue`); + if (!highIntimacyCue.canInterrupt) { + throw new Error("高親密度+talkative 特徵(元氣)時應該允許打斷"); + } + await prisma.relationship.update({ + where: { characterId_userId: { characterId: COOL_ID, userId: USER_ID } }, + data: { intimacy: 60 }, + }); + const coolHighIntimacyCue = await get(`/voice/${COOL_ID}/${USER_ID}/nonverbal-cue`); + if (coolHighIntimacyCue.canInterrupt) { + throw new Error("三無角色沒有 talkative 特徵,即使高親密度也不應該允許打斷"); + } + + // P-5 語音輸入與副語言分析:同一段文字配不同副語言特徵,應該得到不同的情緒標記—— + // 「文字說沒事但聲音在抖」要判為負向,即使文字本身沒有任何負向關鍵字。 + await setEmotion(GENKI_ID, "CALM", 100); + const trembling = await post(`/voice/${GENKI_ID}/analyze-input`, { + userId: USER_ID, + text: "我沒事,真的沒事。", + paralinguistic: { rate: 0.6, volume: 0.3, tremor: 0.7, pauseCount: 3 }, + }); + if (trembling.state.sad <= trembling.state.calm) { + throw new Error("文字沒有負向關鍵字,但聲音顫抖時應該被判為低落情緒"); + } + + await setEmotion(GENKI_ID, "CALM", 100); + const steady = await post(`/voice/${GENKI_ID}/analyze-input`, { + userId: USER_ID, + text: "我沒事,真的沒事。", + paralinguistic: { rate: 1.0, volume: 0.6, tremor: 0.05, pauseCount: 0 }, + }); + if (steady.state.sad > steady.state.calm) { + throw new Error("同樣的文字,聲音平穩時不應該被誤判為低落情緒"); + } + + // 清理本次測試建立的角色(cascade 會一併清掉 Voice Sheet、關係、情緒狀態等)。 + await prisma.character.deleteMany({ where: { id: { in: [GENKI_ID, COOL_ID] } } }); +} diff --git a/todo.md b/todo.md index c91b2e7..8d624d4 100644 --- a/todo.md +++ b/todo.md @@ -387,12 +387,21 @@ flowchart TB ### P. 語音子系統 -- [ ] **P-1 Voice Sheet(S)**:定義角色音色設定(基礎音色、預設語速、音域幅度、口頭聲響庫、禁則)並掛到 `Character`。驗收:可為種子角色寫入完整 Voice Sheet。依據:§角色音色設定(Voice Sheet)。 -- [ ] **P-2 非語言發聲與節奏(S)**:依情緒自動插入「嗯?」「唔……」嘆氣、輕笑等非語言發聲與停頓(重大話題前停 1~2 秒);親密度影響音量與氣音比例;低親密度不打斷使用者。驗收:不同情緒下插入的聲響與停頓不同。依據:§對話節奏與非語言發聲。 -- [ ] **P-3 語音標記層(M)**:情緒狀態機輸出直接驅動韻律參數(語速、音高、音量、句尾走向),實作 wiki§情緒 → 韻律對照 的六種情緒設定。驗收:同一句話在不同情緒下產生不同韻律標記。依據:§情緒 → 韻律對照。 -- [ ] **P-4 TTS Provider 抽象(M)**:比照 `LLMProvider` 定義 `TTSProvider`,先實作 Mock(輸出韻律標記與佔位音檔),真實供應商待人工確認後再接。驗收:切換 Provider 不動呼叫端。依據:技術選型「TTS 服務:情緒韻律語音」。 -- [ ] **P-5 語音輸入與副語言分析(M)**:STT 轉文字之外,另抽取語速、音量、顫抖、停頓等副語言特徵送入情緒標記器(「文字說沒事但聲音在抖」判為負向)。驗收:同一段文字配不同副語言特徵得到不同情緒標記。依據:§語音管線架構「雙向都有語音」。 -- [ ] **P-V 階段驗證(XS)**:`npm run restart && npm run smoke -- P`(P.mjs:韻律對照、非語言發聲插入、副語言特徵影響情緒標記)。 +- [x] **P-1 Voice Sheet(S)**:定義角色音色設定(基礎音色、預設語速、音域幅度、口頭聲響庫、禁則)並掛到 `Character`。驗收:可為種子角色寫入完整 Voice Sheet。依據:§角色音色設定(Voice Sheet)。 +- [x] **P-2 非語言發聲與節奏(S)**:依情緒自動插入「嗯?」「唔……」嘆氣、輕笑等非語言發聲與停頓(重大話題前停 1~2 秒);親密度影響音量與氣音比例;低親密度不打斷使用者。驗收:不同情緒下插入的聲響與停頓不同。依據:§對話節奏與非語言發聲。 +- [x] **P-3 語音標記層(M)**:情緒狀態機輸出直接驅動韻律參數(語速、音高、音量、句尾走向),實作 wiki§情緒 → 韻律對照 的六種情緒設定。驗收:同一句話在不同情緒下產生不同韻律標記。依據:§情緒 → 韻律對照。 +- [x] **P-4 TTS Provider 抽象(M)**:比照 `LLMProvider` 定義 `TTSProvider`,先實作 Mock(輸出韻律標記與佔位音檔),真實供應商待人工確認後再接。驗收:切換 Provider 不動呼叫端。依據:技術選型「TTS 服務:情緒韻律語音」。 +- [x] **P-5 語音輸入與副語言分析(M)**:STT 轉文字之外,另抽取語速、音量、顫抖、停頓等副語言特徵送入情緒標記器(「文字說沒事但聲音在抖」判為負向)。驗收:同一段文字配不同副語言特徵得到不同情緒標記。依據:§語音管線架構「雙向都有語音」。 +- [x] **P-V 階段驗證(XS)**:`npm run restart && npm run smoke -- P`(P.mjs:韻律對照、非語言發聲插入、副語言特徵影響情緒標記)。 + +> **實作記錄(P 群組)**: +> - 新模組放在 `apps/api/src/voice/`,架構完全比照 F 群組:`TTSProvider`/`TTS_PROVIDER` token/`MockTTSProvider`/`RealTTSProvider`(尚未實作、被呼叫才丟例外)/`TTS_PROVIDER=mock|real` 環境變數切換工廠函式,逐一對應 `LLMProvider`/`LLM_PROVIDER`/`MockProvider`/`ClaudeProvider`/`LLM_PROVIDER=mock|claude`。**這組抽象本身沒有新設計,純粹是同一個模式的第二次套用**,這正是 F-1 當初把 Provider 抽出介面的目的:之後任何生成式外部服務要接進來,都走同一套「先做 Mock、真的供應商待人工確認、環境變數切換」的節奏。 +> - **P-1 Voice Sheet 用獨立資料表(同 EmotionState 的 1:1 關聯),不是塞進 `Character` 本體**——口頭聲響庫/禁則是字串陣列,但 SQLite 的 Prisma connector 不支援原生字串陣列欄位,改用 JSON 字串欄位(`verbalTicsJson`/`forbiddenSoundsJson`)+ service 層 `JSON.parse`/`JSON.stringify`,對外一律回傳/接收真正的 `string[]`,呼叫端不需要知道底層是 JSON 字串。預設值依性格原型推導(`defaultVoiceSheetFor`),跟 G-5「原型 → 種子預設」、I-1「原型 → 作息表種子」是同一個模式的第三次套用。 +> - **P-3/P-4 的韻律縮放沿用 O-4 的設計原則,但縮放的依據換成 Voice Sheet 的音域幅度而不是外顯度**:`pitchShiftSemitones = 情緒對照表的基準音高變化 × (角色音域幅度 / 4)`——三無角色音域幅度只有 0.5(對照組的通用預設是 4),同樣的情緒事件換算出來的音高變化只有基準值的 1/8,具體實現「三無角色近乎單音」。**這裡刻意不重複使用 `expressiveness`**(那是表情/情緒外顯度,跟音域幅度是概念上不同的兩個原型參數,各自獨立設定,即使數值上有意的相關性)。 +> - **P-2 停頓機制踩到一個實際的 bug**:一開始用「目前主導情緒的數值 / 100」當作停頓長度的訊號強度,但完全沒考慮到「平靜」狀態下 `calm` 欄位預設就是 100——導致角色明明什麼事都沒發生(平靜狀態),卻被誤判成「強度 1.0 的重大情緒事件」,觸發本該只留給強烈情緒的 1~2 秒長停頓。**平靜不是情緒事件,是情緒事件的缺席**,修正方式是讓 `emotionTag === "CALM"` 時強制把訊號強度視為 0,只有真正的六種「有事發生」情緒(愉悅/低落/警戒/害羞/彆扭)才會依強度觸發長停頓——這個坑跟 G-4 早期「情緒觸發閾值只縮放強度、沒縮放要不要觸發」是同一類錯誤(把「數值存在」跟「訊號有意義」搞混),值得記錄成通用提醒:**任何用「情緒維度數值」當強度訊號的地方,都要先排除 CALM 這個「預設/缺席」維度,不能直接套用同一個公式**。 +> - **P-5 副語言訊號的優先序刻意排在 I-5 作息基線之前**:`EmotionService.processInput` 現在的訊號決定順序是「文字本身觸發 → 副語言訊號(P-5)→ 作息基線(I-5)→ 維持平靜」——副語言是「這一輪輸入本身」的直接訊號,理應比「這個時段通常會怎樣」的基線更優先,但兩者都只在文字沒有觸發任何訊號(CALM)時才會生效,不會蓋掉文字本身已經觸發的訊號。**手動測試時發現一個容易踩的陷阱**:如果角色當下已經有一個很強的主導情緒(例如 JOY 90),副語言訊號算出來的 SAD 不一定能立刻蓋過去(情緒狀態機本身的轉移規則決定要不要真的切換主導情緒,跟訊號有沒有正確算出來是兩件事)——寫測試時要先把情緒重置到 CALM 基準,才能乾淨地驗證副語言訊號本身有沒有被正確送進情緒標記器,不要跟情緒狀態機的轉移阻力混在一起判斷。 +> - **P-5 沒有做真正的語速/音量/顫抖偵測**——副語言特徵目前是結構化輸入(呼叫端直接給數值),跟 M 群組「場景切分」、N 群組「互動正負權重」是同一種「先用結構化輸入代替真實訊號處理」的簡化,理由相同:真正的音訊特徵抽取需要處理實際音訊串流,超出目前系統的 Mock 階段範疇,機制(訊號合成規則、與文字訊號的優先序、餵進情緒標記器)先做對,之後接上真的語音辨識/音訊分析只需要替換這個輸入來源。 +> - **全量回歸執行到 O 群組時出現一次 `fetch failed`,重新單獨執行 O 群組立刻全綠**——跟 J 群組踩過的「重啟瞬間 keep-alive socket 失效」是類似的環境性瞬斷(這次沒有服務重啟動作,單純是連續執行 15 個群組、主機負載升高時的暫態網路錯誤),不是 P 群組程式碼引入的迴歸;已重新單獨驗證 O 群組與完整 A~P 序列(除了這次瞬斷)皆為全綠,記錄於此供之後遇到類似狀況時參考,不需要當成真的 bug 去追。 ### Q. APP(React Native + Expo)