feat: 完成 P 群組 — 語音子系統
新增 apps/api/src/voice/:Voice Sheet(基礎音色/預設語速/音域幅度/口頭 聲響庫/禁則,依性格原型推導預設值,1:1 掛在 Character,同 EmotionState 的關聯模式);情緒→韻律對照(六狀態語速/音高/音量/句尾走向),音高變化 依角色音域幅度縮放(三無角色近乎單音),設計原則同 O-4 表情外顯度縮放; 非語言發聲與節奏(依情緒插入聲響、重大情緒延長停頓、禁則過濾候選、 低親密度不打斷、talkative 特徵+親密度門檻才可插話);TTSProvider 抽象 完全比照 F-1 LLMProvider(MockTTSProvider 輸出韻律標記+佔位音檔, RealTTSProvider 待人工確認供應商後再接,TTS_PROVIDER=mock|real 切換); 語音輸入副語言分析(EmotionService 新增 paralinguisticSignal,優先序 排在作息基線之前,讓「文字說沒事但聲音在抖」判為負向)。 修正一個真實 bug:非語言發聲的停頓時長原本直接拿目前主導情緒的數值 當強度訊號,但平靜狀態下 calm 預設就是 100,導致「什麼事都沒發生」被 誤判成「強度最強的重大情緒事件」而觸發不該有的長停頓——修正為 CALM 一律視為強度 0(平靜是情緒事件的缺席,不是訊號)。 刻意簡化:副語言特徵(語速/音量/顫抖/停頓)採結構化輸入,不做真實音訊 分析,比照 M/N 群組「先用結構化輸入代替真實訊號處理」的一貫取捨。 Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
0f3d16059c
commit
0e9401b891
@@ -15,6 +15,7 @@ import { RomanceModule } from "./romance/romance.module.js";
|
||||
import { CanonModule } from "./canon/canon.module.js";
|
||||
import { EpilogueModule } from "./epilogue/epilogue.module.js";
|
||||
import { TachieModule } from "./tachie/tachie.module.js";
|
||||
import { VoiceModule } from "./voice/voice.module.js";
|
||||
|
||||
@Module({
|
||||
imports: [
|
||||
@@ -33,6 +34,7 @@ import { TachieModule } from "./tachie/tachie.module.js";
|
||||
CanonModule,
|
||||
EpilogueModule,
|
||||
TachieModule,
|
||||
VoiceModule,
|
||||
],
|
||||
controllers: [HealthController],
|
||||
})
|
||||
|
||||
@@ -58,6 +58,9 @@ export interface ProcessInputOptions extends TransitionContext {
|
||||
now?: Date;
|
||||
// I-5:本輪對話沒有明確情緒訊號時,套用作息驅動的基線情緒(不覆蓋文字本身觸發的訊號)。
|
||||
baselineSignal?: { tag: EmotionTag; intensity: number };
|
||||
// P-5:語音輸入的副語言特徵(語速/音量/顫抖/停頓)分析出的訊號——文字說「沒事」但聲音在抖,
|
||||
// 優先於作息基線(副語言是這一輪輸入本身的訊號,比時段基線更直接),但仍不覆蓋文字本身觸發的訊號。
|
||||
paralinguisticSignal?: { tag: EmotionTag; intensity: number };
|
||||
}
|
||||
|
||||
function toDomain(characterId: string, dims: Dimensions, updatedAt: Date): EmotionState {
|
||||
@@ -108,7 +111,9 @@ export class EmotionService {
|
||||
const rawSignal = this.tagger.tag({ text, triggerThreshold });
|
||||
// 情緒觸發閾值除了縮放強度,也要能讓「難觸發」原型對弱刺激完全不觸發(而不只是觸發得比較小力)。
|
||||
let signal = rawSignal.intensity < MIN_TRIGGER_INTENSITY ? { tag: "CALM" as const, intensity: 0 } : rawSignal;
|
||||
if (signal.tag === "CALM" && options.baselineSignal) {
|
||||
if (signal.tag === "CALM" && options.paralinguisticSignal) {
|
||||
signal = options.paralinguisticSignal;
|
||||
} else if (signal.tag === "CALM" && options.baselineSignal) {
|
||||
signal = options.baselineSignal;
|
||||
}
|
||||
const next = nextEmotionState(currentDominant, signal, options);
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
import type { EmotionTag } from "@kokorone/shared";
|
||||
import type { Archetype } from "../personality/archetype-params.js";
|
||||
|
||||
export type SentenceEndContour = "RISING" | "FALLING" | "FLAT";
|
||||
|
||||
export interface ProsodyTemplate {
|
||||
rateDeltaPercent: number; // 語速變化百分比,正值變快
|
||||
pitchDeltaSemitones: number; // 音高變化(半音),正值變高
|
||||
volumeDeltaPercent: number; // 音量變化百分比,正值變大
|
||||
contour: SentenceEndContour;
|
||||
characteristic: string; // 描述性文字,供除錯/前端顯示
|
||||
}
|
||||
|
||||
// P-3 情緒 → 韻律對照:情緒狀態機的輸出直接驅動韻律參數。
|
||||
export const EMOTION_PROSODY_TABLE: Record<EmotionTag, ProsodyTemplate> = {
|
||||
CALM: { rateDeltaPercent: 0, pitchDeltaSemitones: 0, volumeDeltaPercent: 0, contour: "FLAT", characteristic: "角色預設值" },
|
||||
JOY: { rateDeltaPercent: 15, pitchDeltaSemitones: 1.5, volumeDeltaPercent: 10, contour: "RISING", characteristic: "句尾上揚、偶爾帶笑聲" },
|
||||
SAD: { rateDeltaPercent: -20, pitchDeltaSemitones: -1.5, volumeDeltaPercent: -15, contour: "FALLING", characteristic: "句間停頓變長、句尾下沉" },
|
||||
ALERT: { rateDeltaPercent: 5, pitchDeltaSemitones: 0, volumeDeltaPercent: 0, contour: "FLAT", characteristic: "語調變平、字句變短" },
|
||||
SHY: { rateDeltaPercent: -5, pitchDeltaSemitones: 2, volumeDeltaPercent: -10, contour: "FLAT", characteristic: "結巴、語尾變小聲,忽快忽慢" },
|
||||
GRUMPY: { rateDeltaPercent: 10, pitchDeltaSemitones: 1, volumeDeltaPercent: -5, contour: "FALLING", characteristic: "「哼」開頭,先大後小" },
|
||||
};
|
||||
|
||||
// P-1 Voice Sheet 預設值:依性格原型推導基準語速與音域幅度(同 G-5「原型 → 種子預設」的作法)。
|
||||
export const DEFAULT_VOICE_SHEET_BY_ARCHETYPE: Record<Archetype, { baseTimbre: string; baseRate: number; pitchRangeSemitones: number }> = {
|
||||
傲嬌: { baseTimbre: "清亮少女音", baseRate: 1.05, pitchRangeSemitones: 5 },
|
||||
冷淡: { baseTimbre: "低沉平穩音", baseRate: 0.85, pitchRangeSemitones: 1.5 },
|
||||
天然呆: { baseTimbre: "軟糯蘿莉音", baseRate: 0.95, pitchRangeSemitones: 4 },
|
||||
元氣: { baseTimbre: "清亮少女音", baseRate: 1.2, pitchRangeSemitones: 6 },
|
||||
大小姐: { baseTimbre: "低沉御姐音", baseRate: 0.95, pitchRangeSemitones: 3.5 },
|
||||
三無: { baseTimbre: "軟糯蘿莉音", baseRate: 0.8, pitchRangeSemitones: 0.5 },
|
||||
};
|
||||
const DEFAULT_VOICE_SHEET_FALLBACK = { baseTimbre: "清亮少女音", baseRate: 1.0, pitchRangeSemitones: 3 };
|
||||
export function defaultVoiceSheetFor(archetype: string) {
|
||||
return DEFAULT_VOICE_SHEET_BY_ARCHETYPE[archetype as Archetype] ?? DEFAULT_VOICE_SHEET_FALLBACK;
|
||||
}
|
||||
|
||||
// P-2 非語言發聲庫:依情緒狀態的預設候選詞,角色的 Voice Sheet 口頭聲響庫可補充、禁則可過濾排除。
|
||||
export const DEFAULT_INTERJECTIONS_BY_EMOTION: Record<EmotionTag, string[]> = {
|
||||
CALM: ["嗯"],
|
||||
JOY: ["欸嘿嘿", "呵呵"],
|
||||
SAD: ["唔……", "……"],
|
||||
ALERT: ["嗯?", "欸?!"],
|
||||
SHY: ["那個……就是……", "唔……"],
|
||||
GRUMPY: ["哼!", "哼……"],
|
||||
};
|
||||
|
||||
// P-2 停頓:情緒強度達門檻(沿用 C-4 高情緒事件的判斷門檻)且屬於重大話題時,回答前故意停頓 1~2 秒。
|
||||
export const SIGNIFICANT_PAUSE_THRESHOLD = 0.6;
|
||||
export const SIGNIFICANT_PAUSE_SECONDS = [1, 2];
|
||||
export const DEFAULT_PAUSE_SECONDS = 0.2;
|
||||
|
||||
// P-2 打斷與搶話:需要「talkative」特徵(見 archetype-params.ts)且親密度達朋友分層以上才可能插話。
|
||||
export const INTERRUPT_MIN_INTIMACY = 40;
|
||||
|
||||
// P-2 親密度影響音量與氣音比例:對陌生人聲音端正清晰,對摯友放鬆、氣音比例上升。
|
||||
export function intimacyVoiceModifiers(intimacy: number): { volumeMultiplier: number; breathinessRatio: number } {
|
||||
const clamped = Math.max(0, Math.min(100, intimacy));
|
||||
return {
|
||||
volumeMultiplier: 1 - (clamped / 100) * 0.15, // 端正清晰(1.0) → 放鬆慵懶(0.85)
|
||||
breathinessRatio: clamped / 100,
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,32 @@
|
||||
import { Injectable } from "@nestjs/common";
|
||||
import type { TTSProvider, TTSSynthesisInput, SynthesizedAudio } from "./tts-provider.js";
|
||||
import { hashToSeed } from "../llm/seeded-random.js";
|
||||
|
||||
const CHARS_PER_SECOND_AT_NORMAL_RATE = 6; // Mock 階段的粗略估算,不代表真實語速
|
||||
|
||||
// P-4 MockTTSProvider:輸出韻律標記與佔位音檔參照,不呼叫任何真實語音服務——
|
||||
// 讓語音管線的其他機制(韻律解析、非語言發聲、副語言分析)都能在沒有真實 TTS 供應商的情況下完整測試。
|
||||
@Injectable()
|
||||
export class MockTTSProvider implements TTSProvider {
|
||||
async synthesize(input: TTSSynthesisInput): Promise<SynthesizedAudio> {
|
||||
const seed = input.seed ?? hashToSeed(input.text);
|
||||
const { prosody, nonverbalCue, voiceSheet } = input;
|
||||
|
||||
const spokenText = nonverbalCue.interjection ? `${nonverbalCue.interjection} ${input.text}` : input.text;
|
||||
const durationEstimateMs =
|
||||
nonverbalCue.pauseSeconds * 1000 + (spokenText.length / CHARS_PER_SECOND_AT_NORMAL_RATE / prosody.rateMultiplier) * 1000;
|
||||
|
||||
const prosodyMarkup =
|
||||
`<voice timbre="${voiceSheet.baseTimbre}">` +
|
||||
`<break time="${nonverbalCue.pauseSeconds.toFixed(1)}s"/>` +
|
||||
`<prosody rate="${prosody.rateMultiplier.toFixed(2)}" pitch="${prosody.pitchShiftSemitones >= 0 ? "+" : ""}${prosody.pitchShiftSemitones.toFixed(1)}st" volume="${prosody.volumeMultiplier.toFixed(2)}" contour="${prosody.contour}">` +
|
||||
spokenText +
|
||||
`</prosody></voice>`;
|
||||
|
||||
return {
|
||||
audioRef: `mock://tts/${seed}`,
|
||||
prosodyMarkup,
|
||||
durationEstimateMs: Math.round(durationEstimateMs),
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
import { Injectable } from "@nestjs/common";
|
||||
import type { EmotionTag } from "@kokorone/shared";
|
||||
import { PrismaService } from "../prisma/prisma.service.js";
|
||||
import { EmotionService, dominantState } from "../emotion/emotion.service.js";
|
||||
import { RelationshipService } from "../relationship/relationship.service.js";
|
||||
import { getArchetypeParams } from "../personality/archetype-params.js";
|
||||
import { VoiceSheetService } from "./voice-sheet.service.js";
|
||||
import {
|
||||
DEFAULT_INTERJECTIONS_BY_EMOTION,
|
||||
SIGNIFICANT_PAUSE_THRESHOLD,
|
||||
SIGNIFICANT_PAUSE_SECONDS,
|
||||
DEFAULT_PAUSE_SECONDS,
|
||||
INTERRUPT_MIN_INTIMACY,
|
||||
} from "./constants.js";
|
||||
|
||||
export interface NonverbalCue {
|
||||
interjection: string | null;
|
||||
pauseSeconds: number;
|
||||
canInterrupt: boolean;
|
||||
}
|
||||
|
||||
function fieldFor(tag: EmotionTag): "calm" | "joy" | "sad" | "alert" | "shy" | "grumpy" {
|
||||
return tag.toLowerCase() as "calm" | "joy" | "sad" | "alert" | "shy" | "grumpy";
|
||||
}
|
||||
|
||||
// P-2 對話節奏與非語言發聲:依情緒插入「嗯?」「唔……」等聲響與停頓,親密度與性格決定能不能插話。
|
||||
@Injectable()
|
||||
export class NonverbalService {
|
||||
constructor(
|
||||
private readonly prisma: PrismaService,
|
||||
private readonly emotion: EmotionService,
|
||||
private readonly relationship: RelationshipService,
|
||||
private readonly voiceSheet: VoiceSheetService,
|
||||
) {}
|
||||
|
||||
async resolveCue(characterId: string, userId: string, now: Date = new Date()): Promise<NonverbalCue> {
|
||||
const character = await this.prisma.client.character.findUniqueOrThrow({ where: { id: characterId } });
|
||||
const sheet = await this.voiceSheet.getOrSeedDefault(characterId);
|
||||
const emotionState = await this.emotion.getState(characterId, now);
|
||||
const emotionTag = dominantState(emotionState);
|
||||
// 平靜本身不是「重大情緒事件」——calm 的數值代表「沒有明顯情緒」,不能當成訊號強度,
|
||||
// 否則平靜時 calm=100 會被誤判成最強訊號,反而觸發本該只留給強烈情緒的長停頓。
|
||||
const intensity = emotionTag === "CALM" ? 0 : (emotionState[fieldFor(emotionTag)] as number) / 100;
|
||||
|
||||
const candidates = [...DEFAULT_INTERJECTIONS_BY_EMOTION[emotionTag], ...sheet.verbalTics];
|
||||
const allowed = candidates.filter((phrase) => !sheet.forbiddenSounds.includes(phrase));
|
||||
const interjection = allowed.length > 0 ? allowed[0] : null;
|
||||
|
||||
const pauseSeconds =
|
||||
intensity >= SIGNIFICANT_PAUSE_THRESHOLD
|
||||
? SIGNIFICANT_PAUSE_SECONDS[0] + (SIGNIFICANT_PAUSE_SECONDS[1] - SIGNIFICANT_PAUSE_SECONDS[0]) * intensity
|
||||
: DEFAULT_PAUSE_SECONDS;
|
||||
|
||||
const { relationship } = await this.relationship.getState(characterId, userId, now);
|
||||
const archetypeParams = getArchetypeParams(character.personalityArchetype);
|
||||
const canInterrupt = archetypeParams.traits.includes("talkative") && relationship.intimacy >= INTERRUPT_MIN_INTIMACY;
|
||||
|
||||
return { interjection, pauseSeconds, canInterrupt };
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,31 @@
|
||||
import { Injectable } from "@nestjs/common";
|
||||
import type { EmotionTag } from "@kokorone/shared";
|
||||
|
||||
export interface ParalinguisticFeatures {
|
||||
rate: number; // 相對語速,1.0 為正常,<1 變慢、>1 變快
|
||||
volume: number; // 0~1,音量大小
|
||||
tremor: number; // 0~1,聲音顫抖程度
|
||||
pauseCount: number; // 這段話裡的停頓次數
|
||||
}
|
||||
|
||||
function clamp01(value: number): number {
|
||||
return Math.max(0, Math.min(1, value));
|
||||
}
|
||||
|
||||
// P-5 副語言分析:STT 轉出文字之外,語速/音量/顫抖/停頓等特徵也送入情緒標記器——
|
||||
// 「文字說沒事但聲音在抖」應被偵測為負向情緒,即使文字本身沒有任何負向關鍵字。
|
||||
@Injectable()
|
||||
export class ParalinguisticService {
|
||||
deriveSignal(features: ParalinguisticFeatures): { tag: EmotionTag; intensity: number } | null {
|
||||
const slowAndQuiet = features.rate <= 0.75 && features.volume <= 0.45;
|
||||
if (features.tremor >= 0.4 || slowAndQuiet) {
|
||||
const intensity = clamp01(features.tremor * 0.6 + (1 - features.rate) * 0.25 + (1 - features.volume) * 0.25);
|
||||
return { tag: "SAD", intensity: Math.max(intensity, 0.3) };
|
||||
}
|
||||
if (features.rate >= 1.3 && features.volume >= 0.7 && features.tremor < 0.2) {
|
||||
const intensity = clamp01((features.rate - 1) * 0.6 + features.volume * 0.3);
|
||||
return { tag: "JOY", intensity: Math.max(intensity, 0.3) };
|
||||
}
|
||||
return null; // 沒有明顯的副語言訊號,交給文字本身的情緒標記結果決定。
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
import { Injectable } from "@nestjs/common";
|
||||
import { EmotionService, dominantState } from "../emotion/emotion.service.js";
|
||||
import { RelationshipService } from "../relationship/relationship.service.js";
|
||||
import { VoiceSheetService } from "./voice-sheet.service.js";
|
||||
import { EMOTION_PROSODY_TABLE, intimacyVoiceModifiers, type SentenceEndContour } from "./constants.js";
|
||||
import type { EmotionTag } from "@kokorone/shared";
|
||||
|
||||
const REFERENCE_PITCH_RANGE = 4; // 對應 constants.ts 的通用預設音域幅度,作為縮放基準
|
||||
|
||||
export interface ResolvedProsody {
|
||||
emotionTag: EmotionTag;
|
||||
rateMultiplier: number; // 相對 1.0 的語速倍率(已套入角色基準語速)
|
||||
pitchShiftSemitones: number; // 已依角色音域幅度縮放的音高變化
|
||||
volumeMultiplier: number; // 已套入親密度調整的音量倍率
|
||||
breathinessRatio: number; // 氣音比例(親密度越高越高)
|
||||
contour: SentenceEndContour;
|
||||
characteristic: string;
|
||||
}
|
||||
|
||||
// P-3 語音標記層:情緒狀態機輸出直接驅動韻律參數,與韻律驅動同源(跟 O 群組表情解析同一批輸入)。
|
||||
@Injectable()
|
||||
export class ProsodyService {
|
||||
constructor(
|
||||
private readonly emotion: EmotionService,
|
||||
private readonly relationship: RelationshipService,
|
||||
private readonly voiceSheet: VoiceSheetService,
|
||||
) {}
|
||||
|
||||
async resolveProsody(characterId: string, userId: string, now: Date = new Date()): Promise<ResolvedProsody> {
|
||||
const emotionState = await this.emotion.getState(characterId, now);
|
||||
const emotionTag = dominantState(emotionState);
|
||||
const template = EMOTION_PROSODY_TABLE[emotionTag];
|
||||
|
||||
const sheet = await this.voiceSheet.getOrSeedDefault(characterId);
|
||||
const { relationship } = await this.relationship.getState(characterId, userId, now);
|
||||
const { volumeMultiplier: intimacyVolume, breathinessRatio } = intimacyVoiceModifiers(relationship.intimacy);
|
||||
|
||||
const rateMultiplier = sheet.baseRate * (1 + template.rateDeltaPercent / 100);
|
||||
// 音域幅度縮放:跟 O-4 表情外顯度縮放同一個設計原則——角色的音域幅度越窄(三無),
|
||||
// 同一個情緒帶來的音高變化就越不明顯(近乎單音)。
|
||||
const pitchShiftSemitones = template.pitchDeltaSemitones * (sheet.pitchRangeSemitones / REFERENCE_PITCH_RANGE);
|
||||
const volumeMultiplier = intimacyVolume * (1 + template.volumeDeltaPercent / 100);
|
||||
|
||||
return {
|
||||
emotionTag,
|
||||
rateMultiplier,
|
||||
pitchShiftSemitones,
|
||||
volumeMultiplier,
|
||||
breathinessRatio,
|
||||
contour: template.contour,
|
||||
characteristic: template.characteristic,
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
import { Injectable } from "@nestjs/common";
|
||||
import type { TTSProvider, TTSSynthesisInput, SynthesizedAudio } from "./tts-provider.js";
|
||||
|
||||
// P-4 真實 TTS 供應商:待人工確認要接哪一家服務(技術選型文件尚未拍板),先留下同介面的空殼,
|
||||
// 呼叫到才報錯,不在啟動時就讓整個服務炸掉——跟 F 群組 ClaudeProvider 的處理方式一致。
|
||||
@Injectable()
|
||||
export class RealTTSProvider implements TTSProvider {
|
||||
async synthesize(_input: TTSSynthesisInput): Promise<SynthesizedAudio> {
|
||||
throw new Error("RealTTSProvider 尚未實作:真實語音供應商待人工確認後再接(見技術選型「TTS 服務」)");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
import type { ResolvedProsody } from "./prosody.service.js";
|
||||
import type { NonverbalCue } from "./nonverbal.service.js";
|
||||
import type { VoiceSheetView } from "./voice-sheet.service.js";
|
||||
|
||||
export interface TTSSynthesisInput {
|
||||
text: string;
|
||||
voiceSheet: VoiceSheetView;
|
||||
prosody: ResolvedProsody;
|
||||
nonverbalCue: NonverbalCue;
|
||||
seed?: number;
|
||||
}
|
||||
|
||||
export interface SynthesizedAudio {
|
||||
audioRef: string; // 音檔參照(Mock 階段是佔位字串,不是真的音檔)
|
||||
prosodyMarkup: string; // 類 SSML 的除錯用韻律標記字串
|
||||
durationEstimateMs: number;
|
||||
}
|
||||
|
||||
// P-4 TTSProvider 抽象:比照 F-1 LLMProvider 的介面設計,切換 Provider 不動呼叫端任何一行。
|
||||
export interface TTSProvider {
|
||||
synthesize(input: TTSSynthesisInput): Promise<SynthesizedAudio>;
|
||||
}
|
||||
|
||||
export const TTS_PROVIDER = Symbol("TTS_PROVIDER");
|
||||
@@ -0,0 +1,77 @@
|
||||
import { Injectable } from "@nestjs/common";
|
||||
import { PrismaService } from "../prisma/prisma.service.js";
|
||||
import { defaultVoiceSheetFor } from "./constants.js";
|
||||
|
||||
export interface VoiceSheetView {
|
||||
characterId: string;
|
||||
baseTimbre: string;
|
||||
baseRate: number;
|
||||
pitchRangeSemitones: number;
|
||||
verbalTics: string[];
|
||||
forbiddenSounds: string[];
|
||||
}
|
||||
|
||||
function toView(row: {
|
||||
characterId: string;
|
||||
baseTimbre: string;
|
||||
baseRate: number;
|
||||
pitchRangeSemitones: number;
|
||||
verbalTicsJson: string;
|
||||
forbiddenSoundsJson: string;
|
||||
}): VoiceSheetView {
|
||||
return {
|
||||
characterId: row.characterId,
|
||||
baseTimbre: row.baseTimbre,
|
||||
baseRate: row.baseRate,
|
||||
pitchRangeSemitones: row.pitchRangeSemitones,
|
||||
verbalTics: JSON.parse(row.verbalTicsJson),
|
||||
forbiddenSounds: JSON.parse(row.forbiddenSoundsJson),
|
||||
};
|
||||
}
|
||||
|
||||
// P-1 角色音色設定:定義基礎音色、預設語速、音域幅度、口頭聲響庫與禁則,掛在 Character 上。
|
||||
@Injectable()
|
||||
export class VoiceSheetService {
|
||||
constructor(private readonly prisma: PrismaService) {}
|
||||
|
||||
// 依性格原型推導預設值後寫入(同 G-5「原型 → 種子預設」的作法),可重複執行(upsert)。
|
||||
async seedDefault(
|
||||
characterId: string,
|
||||
overrides: { verbalTics?: string[]; forbiddenSounds?: string[] } = {},
|
||||
): Promise<VoiceSheetView> {
|
||||
const character = await this.prisma.client.character.findUniqueOrThrow({ where: { id: characterId } });
|
||||
const defaults = defaultVoiceSheetFor(character.personalityArchetype);
|
||||
const row = await this.prisma.client.voiceSheet.upsert({
|
||||
where: { characterId },
|
||||
update: {
|
||||
baseTimbre: defaults.baseTimbre,
|
||||
baseRate: defaults.baseRate,
|
||||
pitchRangeSemitones: defaults.pitchRangeSemitones,
|
||||
verbalTicsJson: JSON.stringify(overrides.verbalTics ?? []),
|
||||
forbiddenSoundsJson: JSON.stringify(overrides.forbiddenSounds ?? []),
|
||||
},
|
||||
create: {
|
||||
characterId,
|
||||
baseTimbre: defaults.baseTimbre,
|
||||
baseRate: defaults.baseRate,
|
||||
pitchRangeSemitones: defaults.pitchRangeSemitones,
|
||||
verbalTicsJson: JSON.stringify(overrides.verbalTics ?? []),
|
||||
forbiddenSoundsJson: JSON.stringify(overrides.forbiddenSounds ?? []),
|
||||
},
|
||||
});
|
||||
return toView(row);
|
||||
}
|
||||
|
||||
async get(characterId: string): Promise<VoiceSheetView | null> {
|
||||
const row = await this.prisma.client.voiceSheet.findUnique({ where: { characterId } });
|
||||
return row ? toView(row) : null;
|
||||
}
|
||||
|
||||
async getOrSeedDefault(characterId: string): Promise<VoiceSheetView> {
|
||||
const existing = await this.get(characterId);
|
||||
if (existing) {
|
||||
return existing;
|
||||
}
|
||||
return this.seedDefault(characterId);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,95 @@
|
||||
import { Body, Controller, Get, Inject, Param, Post, Query } from "@nestjs/common";
|
||||
import { VoiceSheetService } from "./voice-sheet.service.js";
|
||||
import { ProsodyService } from "./prosody.service.js";
|
||||
import { NonverbalService } from "./nonverbal.service.js";
|
||||
import { ParalinguisticService, type ParalinguisticFeatures } from "./paralinguistic.service.js";
|
||||
import { TTS_PROVIDER, type TTSProvider } from "./tts-provider.js";
|
||||
import { EmotionService } from "../emotion/emotion.service.js";
|
||||
|
||||
interface SeedVoiceSheetBody {
|
||||
verbalTics?: string[];
|
||||
forbiddenSounds?: string[];
|
||||
}
|
||||
|
||||
interface SynthesizeBody {
|
||||
text: string;
|
||||
now?: string;
|
||||
seed?: number;
|
||||
}
|
||||
|
||||
interface AnalyzeInputBody {
|
||||
userId: string;
|
||||
text: string;
|
||||
paralinguistic: ParalinguisticFeatures;
|
||||
now?: string;
|
||||
}
|
||||
|
||||
@Controller("voice")
|
||||
export class VoiceController {
|
||||
constructor(
|
||||
private readonly voiceSheet: VoiceSheetService,
|
||||
private readonly prosody: ProsodyService,
|
||||
private readonly nonverbal: NonverbalService,
|
||||
private readonly paralinguistic: ParalinguisticService,
|
||||
private readonly emotion: EmotionService,
|
||||
@Inject(TTS_PROVIDER) private readonly ttsProvider: TTSProvider,
|
||||
) {}
|
||||
|
||||
@Post(":characterId/voice-sheet/seed-default")
|
||||
async seedVoiceSheet(@Param("characterId") characterId: string, @Body() body: SeedVoiceSheetBody) {
|
||||
return this.voiceSheet.seedDefault(characterId, body);
|
||||
}
|
||||
|
||||
@Get(":characterId/voice-sheet")
|
||||
async getVoiceSheet(@Param("characterId") characterId: string) {
|
||||
return this.voiceSheet.get(characterId);
|
||||
}
|
||||
|
||||
@Get(":characterId/:userId/prosody")
|
||||
async getProsody(
|
||||
@Param("characterId") characterId: string,
|
||||
@Param("userId") userId: string,
|
||||
@Query("now") now?: string,
|
||||
) {
|
||||
return this.prosody.resolveProsody(characterId, userId, now ? new Date(now) : undefined);
|
||||
}
|
||||
|
||||
@Get(":characterId/:userId/nonverbal-cue")
|
||||
async getNonverbalCue(
|
||||
@Param("characterId") characterId: string,
|
||||
@Param("userId") userId: string,
|
||||
@Query("now") now?: string,
|
||||
) {
|
||||
return this.nonverbal.resolveCue(characterId, userId, now ? new Date(now) : undefined);
|
||||
}
|
||||
|
||||
@Post(":characterId/:userId/synthesize")
|
||||
async synthesize(
|
||||
@Param("characterId") characterId: string,
|
||||
@Param("userId") userId: string,
|
||||
@Body() body: SynthesizeBody,
|
||||
) {
|
||||
const now = body.now ? new Date(body.now) : undefined;
|
||||
const [voiceSheet, resolvedProsody, nonverbalCue] = await Promise.all([
|
||||
this.voiceSheet.getOrSeedDefault(characterId),
|
||||
this.prosody.resolveProsody(characterId, userId, now),
|
||||
this.nonverbal.resolveCue(characterId, userId, now),
|
||||
]);
|
||||
return this.ttsProvider.synthesize({
|
||||
text: body.text,
|
||||
voiceSheet,
|
||||
prosody: resolvedProsody,
|
||||
nonverbalCue,
|
||||
seed: body.seed,
|
||||
});
|
||||
}
|
||||
|
||||
// P-5 驗收用端點:同一段文字配不同副語言特徵,應該得到不同的情緒標記結果。
|
||||
@Post(":characterId/analyze-input")
|
||||
async analyzeInput(@Param("characterId") characterId: string, @Body() body: AnalyzeInputBody) {
|
||||
const now = body.now ? new Date(body.now) : new Date();
|
||||
const paralinguisticSignal = this.paralinguistic.deriveSignal(body.paralinguistic) ?? undefined;
|
||||
const result = await this.emotion.processInput(characterId, body.text, { now, paralinguisticSignal });
|
||||
return { ...result, paralinguisticSignal: paralinguisticSignal ?? null };
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
import { Module } from "@nestjs/common";
|
||||
import { log } from "@kokorone/shared";
|
||||
import { PrismaModule } from "../prisma/prisma.module.js";
|
||||
import { EmotionModule } from "../emotion/emotion.module.js";
|
||||
import { RelationshipModule } from "../relationship/relationship.module.js";
|
||||
import { VoiceSheetService } from "./voice-sheet.service.js";
|
||||
import { ProsodyService } from "./prosody.service.js";
|
||||
import { NonverbalService } from "./nonverbal.service.js";
|
||||
import { ParalinguisticService } from "./paralinguistic.service.js";
|
||||
import { MockTTSProvider } from "./mock-tts-provider.service.js";
|
||||
import { RealTTSProvider } from "./real-tts-provider.service.js";
|
||||
import { TTS_PROVIDER } from "./tts-provider.js";
|
||||
import { VoiceController } from "./voice.controller.js";
|
||||
|
||||
// P-4 Provider 切換:TTS_PROVIDER=mock|real 決定注入哪個實作,比照 F-2 LLM_PROVIDER 的作法。
|
||||
function ttsProviderFactory(mock: MockTTSProvider, real: RealTTSProvider) {
|
||||
const providerName = process.env.TTS_PROVIDER ?? "mock";
|
||||
if (providerName === "real") {
|
||||
log("啟動", "ERR", "TTS_PROVIDER=real 但 RealTTSProvider 尚未實作,語音合成呼叫時會拋出例外");
|
||||
return real;
|
||||
}
|
||||
return mock;
|
||||
}
|
||||
|
||||
@Module({
|
||||
imports: [PrismaModule, EmotionModule, RelationshipModule],
|
||||
controllers: [VoiceController],
|
||||
providers: [
|
||||
VoiceSheetService,
|
||||
ProsodyService,
|
||||
NonverbalService,
|
||||
ParalinguisticService,
|
||||
MockTTSProvider,
|
||||
RealTTSProvider,
|
||||
{ provide: TTS_PROVIDER, useFactory: ttsProviderFactory, inject: [MockTTSProvider, RealTTSProvider] },
|
||||
],
|
||||
exports: [VoiceSheetService, ProsodyService, NonverbalService, ParalinguisticService, TTS_PROVIDER],
|
||||
})
|
||||
export class VoiceModule {}
|
||||
Reference in New Issue
Block a user