#!/usr/bin/env node // ============================================================================== // 用途:worklog 的 transcript 處理工具。負責 (1) 從 Claude Code/Codex/ // GitHub Copilot CLI 的 JSONL 抽出「本輪」對話片段(最後一筆使用者訊息 // 之後的全部內容),(2) 估算本輪花費時間,(3) 統計本輪 token 用量, // (4) 對文字做機密遮蔽(token/密碼/PII),作為寫入 wiki 前的第二道防線。 // 本檔 REDACT_PATTERNS 對應 /jsc-shared:spec-gitea『機密遮蔽實作』章節 // (其他工具的機密遮蔽規則以該章節為準)。 // 原 Python 版(transcript.py)的 extract/duration/redact 三個子命令 // 已逐一以同一份 transcript 對拍 diff 驗證輸出逐字元相同,token 統計與 // Copilot 支援為本次新增,Python 版沒有對應功能可供對拍。 // 更新時間:2026/08/11 16:51:56 // 相依:Node.js 標準內建功能,無外部套件。全程僅走 stdin/stdout,不寫任何檔案。 // ============================================================================== import fs from "node:fs"; // 單則工具結果/參數的擷取上限,避免整份 transcript 塞進摘要輸入 const TOOL_RESULT_LIMIT = 200; const TOOL_INPUT_LIMIT = 160; const TOTAL_LIMIT = 24000; // ------------------------------------------------------------------------------ // 機密遮蔽規則:命中一律換成 *** // ------------------------------------------------------------------------------ // 本檔遮蔽規則對應 shared/scripts/lib/redact-patterns.json(經 /jsc-shared:spec-gitea 收斂)。 // 依 A5-3/G1-6 裁決:doc repo 內自帶一份,不跨 repo 讀取 shared/scripts/lib/redact-patterns.json // (執行期無法跨 plugin 存取),異動一律先改 shared 那份規則檔,再回頭同步本檔。 // // 移植自 Python re 版時的三個地雷(逐一處理,對應驗收案例見 G1-2): // (a) Python re.sub 預設全域取代,JS String.replace 不是 —— 全部補上 g flag。 // (b) Python (?i) inline flag,JS 不支援 —— 改用 RegExp 的 i flag。 // (c) 反向參照:Python 用 \1,JS String.replace 用 $1。 const REDACT_PATTERNS = [ [/[A-Za-z0-9_-]*:[A-Za-z0-9_-]{16,}@/g, "***@"], // URL 內嵌憑證 user:token@ [/\b[0-9a-f]{40}\b/g, "***"], // Gitea 40 字元 token [/\bgh[pousr]_[A-Za-z0-9_]{16,}\b/g, "***"], // GitHub token [/\bsk-[A-Za-z0-9_-]{16,}\b/g, "***"], // API key [/\b(token|password|passwd|pwd|secret|api[_-]?key)\b\s*[:=]\s*\S+/gi, "$1=***"], [/Authorization:\s*(token|bearer)\s+\S+/gi, "Authorization: $1 ***"], [/[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}/g, "***"], // Email [/\b09\d{2}[-\s]?\d{3}[-\s]?\d{3}\b/g, "***"], // 台灣手機 [/\b[A-Z][12]\d{8}\b/g, "***"], // 身分證字號 ]; /** * 仿 Python `json.dumps(obj, ensure_ascii=False)` 預設格式序列化(`", "`/`": "` * 分隔符,逗號與冒號後都有空格):JS `JSON.stringify` 預設不加這些空格, * 若直接拿來取代會讓 `[tool:xxx]` 那行的參數字串跟 Python 版逐字元不同 * (已於 G1-1 對拍真實 subagent transcript 時抓到這個差異)。 */ function pyJsonDumps(value) { if (value === null || value === undefined) return "null"; if (typeof value === "boolean" || typeof value === "number") return String(value); if (typeof value === "string") return JSON.stringify(value); if (Array.isArray(value)) return `[${value.map(pyJsonDumps).join(", ")}]`; if (typeof value === "object") { const parts = Object.entries(value).map(([k, v]) => `${JSON.stringify(k)}: ${pyJsonDumps(v)}`); return `{${parts.join(", ")}}`; } return "null"; } /** Python 的 truthy 判斷:`0`/`""`/`None`/`False`/空 list/空 dict 皆為 falsy。 */ function isPyTruthy(value) { if (value === null || value === undefined || value === false || value === 0 || value === "") return false; if (Array.isArray(value)) return value.length > 0; if (typeof value === "object") return Object.keys(value).length > 0; return true; } /** * 仿 Python `repr()`:字串用單引號,list/dict 遞迴展開。用於重現 * `str(payload.get("output") or "")` 這類寫法在 `output` 是 list/dict(而非 * 字串)時的輸出——Python `str(list)` 會呼叫每個元素的 `repr()`,JS * `String(array)` 只會呼叫 `Array.prototype.toString`(物件變成 * `[object Object]`),兩者天差地遠(已於 G1-1 對拍真實 Codex transcript 時 * 抓到這個差異:某筆 `function_call_output` 的 `output` 是結構化 array 而非 * JSON 字串)。 */ function pyRepr(value) { if (value === null || value === undefined) return "None"; if (value === true) return "True"; if (value === false) return "False"; if (typeof value === "number") return String(value); if (typeof value === "string") { const quote = value.includes("'") && !value.includes('"') ? '"' : "'"; let out = ""; for (const ch of value) { const cp = ch.codePointAt(0); if (ch === "\\") out += "\\\\"; else if (ch === quote) out += `\\${quote}`; else if (ch === "\n") out += "\\n"; else if (ch === "\r") out += "\\r"; else if (ch === "\t") out += "\\t"; // Python repr() 對「不可印字元」一律跳脫成 \uXXXX/\UXXXXXXXX:這裡不追 // 完整 Unicode 分類表(那需要外部資料庫),只涵蓋最常見的兩類——C0/C1 // 控制字元、與私用區(Private Use Area,已於 G1-1 對拍時在真實 Codex // transcript 的引註標記裡遇到 U+E200 這個案例)。其餘罕見不可印分類 // (如某些格式控制符)維持原樣輸出,屬已知、影響範圍為零的簡化 // (這段文字只會餵給 LLM 摘要,不影響任何程式邏輯判讀)。 else if ((cp >= 0x00 && cp <= 0x1f) || cp === 0x7f || (cp >= 0x80 && cp <= 0x9f)) { out += `\\x${cp.toString(16).padStart(2, "0")}`; } else if ((cp >= 0xe000 && cp <= 0xf8ff) || (cp >= 0xf0000 && cp <= 0xffffd) || (cp >= 0x100000 && cp <= 0x10fffd)) { out += cp <= 0xffff ? `\\u${cp.toString(16).padStart(4, "0")}` : `\\U${cp.toString(16).padStart(8, "0")}`; } else out += ch; } return `${quote}${out}${quote}`; } if (Array.isArray(value)) return `[${value.map(pyRepr).join(", ")}]`; if (typeof value === "object") { const parts = Object.entries(value).map(([k, v]) => `${pyRepr(k)}: ${pyRepr(v)}`); return `{${parts.join(", ")}}`; } return String(value); } /** 仿 Python `str(value or fallback)`:字串型別的 `str()` 是自己本身,其餘型別走 `pyRepr`。 */ function pythonStrOr(value, fallback) { if (!isPyTruthy(value)) return fallback; return typeof value === "string" ? value : pyRepr(value); } /** 依 Unicode 碼點(非 UTF-16 code unit)取字串長度,對齊 Python str 的 len() 語意。 */ function codePointLength(text) { return Array.from(text).length; } /** * 依 Unicode 碼點切片,對齊 Python 字串切片語意(避免切斷代理對)。 * ⚠️ 必須用這個,不能直接 `str.slice()`:emoji 等 astral-plane 字元在 JS 是 * 兩個 UTF-16 code unit,`slice()` 按 code unit 數截斷會比 Python 按碼點數 * 截斷的版本少算字元(已於 G1-1 對拍真實 transcript 時抓到這個差異)。 */ function codePointSlice(text, start, end) { return Array.from(text).slice(start, end).join(""); } /** 對文字套用全部機密遮蔽規則,回傳遮蔽後的結果。 */ export function redact(text) { let out = text; for (const [pattern, replacement] of REDACT_PATTERNS) { out = out.replace(pattern, replacement); } return out; } /** 判斷 transcript 條目是否為真正的使用者輸入(排除工具回填與環境注入)。 */ function isRealUserMessage(entry) { const payload = entry.payload; if (payload && typeof payload === "object" && !Array.isArray(payload) && entry.type === "event_msg") { return payload.type === "user_message" && Boolean(String(payload.message ?? "").trim()); } // Copilot:type "user.message",取 data.content(不用 transformedContent,那裡混了注入內容) if (entry.type === "user.message") { const content = entry.data && typeof entry.data === "object" ? entry.data.content : undefined; return Boolean(String(content ?? "").trim()); } if (entry.type !== "user") return false; const content = entry.message && typeof entry.message === "object" ? entry.message.content : undefined; if (typeof content === "string") return Boolean(content.trim()); if (Array.isArray(content)) { return content.some((b) => b && typeof b === "object" && b.type === "text"); } return false; } /** 取出條目的 content blocks,統一為 array 形式。 */ function blocks(entry) { const content = entry.message && typeof entry.message === "object" ? entry.message.content : undefined; if (typeof content === "string") return [{ type: "text", text: content }]; return Array.isArray(content) ? content : []; } /** 把 Codex response_item 的 content blocks 轉成純文字片段。 */ function payloadTextBlocks(content) { if (typeof content === "string") return [content]; if (!Array.isArray(content)) return []; const texts = []; for (const block of content) { if (!block || typeof block !== "object") continue; if (["input_text", "output_text", "text"].includes(block.type)) { const text = String(block.text ?? "").trim(); if (text) texts.push(text); } } return texts; } /** 將 Codex session JSONL 的 payload 格式轉為摘要輸入用純文字。 */ function renderCodexPayload(entry) { const payload = entry.payload; if (!payload || typeof payload !== "object") return []; const lines = []; const entryType = entry.type; const payloadType = payload.type; if (entryType === "event_msg") { if (payloadType === "user_message") { const message = String(payload.message ?? "").trim(); if (message) lines.push(`[user] ${message}`); } else if (payloadType === "agent_message") { const message = String(payload.message ?? "").trim(); if (message) { const phase = payload.phase || "assistant"; lines.push(`[assistant:${phase}] ${message}`); } } return lines; } if (entryType !== "response_item") return lines; if (payloadType === "message") { const role = payload.role || "assistant"; if (role === "system" || role === "developer") return lines; for (const text of payloadTextBlocks(payload.content)) { // Codex 會把 skill 內容以 user role 注入;避免把整份 SKILL.md 當成本輪工作。 if (role === "user" && text.trimStart().startsWith("")) continue; if (role === "user" && text.trimStart().startsWith("")) continue; lines.push(`[${role}] ${text}`); } } else if (payloadType === "function_call") { const name = payload.name || "?"; const raw = pythonStrOr(payload.arguments, "").trim().replace(/\n/g, " "); lines.push(`[tool:${name}] ${codePointSlice(raw, 0, TOOL_INPUT_LIMIT)}`); } else if (payloadType === "function_call_output") { const raw = pythonStrOr(payload.output, "").trim().replace(/\n/g, " "); if (raw) lines.push(`[result] ${codePointSlice(raw, 0, TOOL_RESULT_LIMIT)}`); } return lines; } /** * 將 Copilot events.jsonl 條目轉為摘要輸入用純文字。 * ⚠️ user.message/assistant.message 的欄位形狀已於 2026/08/11 實測真實 Copilot * session 確認(見 todo.md H1-1 記錄);tool.execution_complete 與 assistant.message * 的 toolRequests 欄位形狀依規劃文件推斷,尚未實測到含工具呼叫的真實 session, * 若欄位名稱與實際不符請依實測結果修正。 */ function renderCopilotEvent(entry) { const data = entry.data; if (!data || typeof data !== "object") return []; if (entry.type === "user.message") { const text = String(data.content ?? "").trim(); return text ? [`[user] ${text}`] : []; } if (entry.type === "assistant.message") { const lines = []; const text = String(data.content ?? "").trim(); if (text) lines.push(`[assistant] ${text}`); const toolRequests = Array.isArray(data.toolRequests) ? data.toolRequests : []; for (const req of toolRequests) { const name = (req && (req.name || req.tool)) || "?"; const raw = pyJsonDumps((req && (req.input ?? req.arguments)) ?? {}); lines.push(`[tool:${name}] ${codePointSlice(raw, 0, TOOL_INPUT_LIMIT)}`); } return lines; } if (entry.type === "tool.execution_complete") { const raw = String(data.output ?? data.result ?? "").trim().replace(/\n/g, " "); return raw ? [`[result] ${codePointSlice(raw, 0, TOOL_RESULT_LIMIT)}`] : []; } return []; } /** 將單一 transcript 條目轉為摘要輸入用的純文字行(工具結果僅取前段)。 */ function render(entry) { const copilotLines = renderCopilotEvent(entry); if (copilotLines.length) return copilotLines; const codexLines = renderCodexPayload(entry); if (codexLines.length) return codexLines; const role = entry.type; const lines = []; for (const block of blocks(entry)) { if (!block || typeof block !== "object") continue; const kind = block.type; if (kind === "text") { const text = String(block.text ?? "").trim(); if (text) lines.push(`[${role}] ${text}`); } else if (kind === "tool_use") { const name = block.name ?? "?"; const raw = pyJsonDumps(block.input ?? {}); lines.push(`[tool:${name}] ${codePointSlice(raw, 0, TOOL_INPUT_LIMIT)}`); } else if (kind === "tool_result") { let raw = block.content; if (Array.isArray(raw)) { raw = raw .filter((b) => b && typeof b === "object" && b.type === "text") .map((b) => b.text ?? "") .join(" "); } raw = pythonStrOr(raw, "").trim().replace(/\n/g, " "); if (raw) lines.push(`[result] ${codePointSlice(raw, 0, TOOL_RESULT_LIMIT)}`); } } return lines; } /** 讀取 transcript JSONL,忽略無法解析的列。 */ function readEntries(filePath) { let raw; try { raw = fs.readFileSync(filePath, "utf8"); } catch { return []; } const entries = []; for (const rawLine of raw.split("\n")) { const line = rawLine.trim(); if (!line) continue; try { entries.push(JSON.parse(line)); } catch { continue; } } return entries; } /** 找出本輪起點:最後一筆真正使用者訊息的位置。 */ function turnStartIndex(entries) { let start = 0; for (let index = entries.length - 1; index >= 0; index--) { if (isRealUserMessage(entries[index])) { start = index; break; } } return start; } /** 解析常見 transcript timestamp 格式,回傳 epoch 毫秒;失敗回 null。 */ function parseTimestamp(value) { if (typeof value !== "string" || !value.trim()) return null; let raw = value.trim(); if (!/[zZ]$/.test(raw) && !/[+-]\d{2}:\d{2}$/.test(raw)) { raw += "Z"; // 沒有時區資訊時視為 UTC,等同 Python 版 tzinfo=timezone.utc 的退回值 } const ms = Date.parse(raw); return Number.isNaN(ms) ? null : ms; } /** 取出 transcript 條目的時間欄位(epoch 毫秒)。 */ function entryTimestamp(entry) { for (const key of ["timestamp", "created_at", "time"]) { const t = parseTimestamp(entry[key]); if (t !== null) return t; } const message = entry.message; if (message && typeof message === "object") { for (const key of ["timestamp", "created_at", "time"]) { const t = parseTimestamp(message[key]); if (t !== null) return t; } } return null; } /** 等同 Python round():四捨五入時「剛好 .5」採銀行家捨入(就近取偶)。 */ function pythonRound(x) { const floor = Math.floor(x); const diff = x - floor; if (diff < 0.5) return floor; if (diff > 0.5) return floor + 1; return floor % 2 === 0 ? floor : floor + 1; } /** 把秒數格式化為精簡中文耗時。 */ export function formatDuration(seconds) { if (seconds < 0) return "未判定"; const minutes = pythonRound(seconds / 60); if (minutes <= 0) return "1 分鐘內"; const hours = Math.floor(minutes / 60); const mins = minutes % 60; if (hours && mins) return `${hours} 小時 ${mins} 分鐘`; if (hours) return `${hours} 小時`; return `${mins} 分鐘`; } /** * 估算本輪花費時間:取本輪起點到最後一筆可解析 timestamp 的差距。 * transcript 無時間欄位或本輪少於兩個時間點時回「未判定」,避免臆測。 */ export function turnDuration(filePath) { const entries = readEntries(filePath); if (!entries.length) return "未判定"; const start = turnStartIndex(entries); const stamps = entries .slice(start) .map(entryTimestamp) .filter((t) => t !== null); if (stamps.length < 2) return "未判定"; const seconds = (Math.max(...stamps) - Math.min(...stamps)) / 1000; return formatDuration(seconds); } /** * 從 transcript JSONL 抽出本輪內容:最後一筆真正使用者訊息(含該筆)之後的全部條目。 * 不需任何狀態檔即可界定「本輪」,符合工作內容不落地的要求。 * 回傳純文字字串;讀取失敗或無內容時回空字串。 */ export function extractTurn(filePath) { const entries = readEntries(filePath); if (!entries.length) return ""; const start = turnStartIndex(entries); const lines = []; for (const entry of entries.slice(start)) { lines.push(...render(entry)); } let text = lines.join("\n").trim(); const totalLen = codePointLength(text); if (totalLen > TOTAL_LIMIT) { const half = Math.floor(TOTAL_LIMIT / 2); const head = codePointSlice(text, 0, half); const tail = codePointSlice(text, totalLen - half, totalLen); text = `${head}\n…(中段省略)…\n${tail}`; } return text; } /** 把 token 數格式化為精簡字串:<1000 直接輸出整數,>=1000 輸出一位小數的 k 表示。 */ export function formatTokens(n) { if (n < 1000) return String(n); return `${(n / 1000).toFixed(1)}k`; } /** * 統計本輪 token 用量,回傳 { input, output }(input 為 null 代表「未判定」, * 例如 Copilot 逐輪只有 outputTokens);本輪完全找不到任何用量欄位時回 null * (不可回 { input: 0, output: 0 },否則無法與「真的沒用到 token」區分)。 * * 三種格式互斥(依 CLI 各自的 transcript 結構,不會同時出現在同一份檔案), * 依序嘗試 Claude Code → Codex → Copilot,找到哪種格式的用量欄位就用哪種。 */ export function turnTokens(filePath) { const entries = readEntries(filePath); if (!entries.length) return null; const start = turnStartIndex(entries); const turn = entries.slice(start); // Claude Code:entry.type === "assistant",用量在 message.usage。 // input_tokens 只是「未命中快取的殘量」,必須加上 cache_creation_input_tokens // 與 cache_read_input_tokens 才是完整輸入量(本機實測曾見 input_tokens:2 但 // cache_creation_input_tokens:48200 的案例,單獨拿 input_tokens 會嚴重低估)。 const claudeUsages = turn.filter( (e) => e.type === "assistant" && e.message && typeof e.message === "object" && e.message.usage && typeof e.message.usage === "object", ); if (claudeUsages.length) { let input = 0; let output = 0; for (const e of claudeUsages) { const u = e.message.usage; input += Number(u.input_tokens || 0) + Number(u.cache_creation_input_tokens || 0) + Number(u.cache_read_input_tokens || 0); output += Number(u.output_tokens || 0); } return { input, output }; } // Codex:type === "event_msg" 且 payload.type === "token_count",用量在 // payload.info.last_token_usage。input_tokens 已包含 cached_input_tokens, // 不可再加一次;reasoning_output_tokens 是 output_tokens 的子集,不另計。 const codexUsages = turn.filter( (e) => e.type === "event_msg" && e.payload && typeof e.payload === "object" && e.payload.type === "token_count" && e.payload.info && typeof e.payload.info === "object" && e.payload.info.last_token_usage, ); if (codexUsages.length) { let input = 0; let output = 0; for (const e of codexUsages) { const u = e.payload.info.last_token_usage; input += Number(u.input_tokens || 0); output += Number(u.output_tokens || 0); } return { input, output }; } // Copilot:type === "assistant.message",只有 outputTokens,輸入固定「未判定」 // (依使用者裁示:不得改讀 session.shutdown.modelMetrics,那是整個 session // 結束才寫的累計值,語意不是「本輪」)。 const copilotUsages = turn.filter((e) => e.type === "assistant.message" && e.data && typeof e.data === "object" && typeof e.data.outputTokens === "number"); if (copilotUsages.length) { let output = 0; for (const e of copilotUsages) output += Number(e.data.outputTokens || 0); return { input: null, output }; } return null; } /** 把 turnTokens() 的結果格式化為條目用字串;找不到用量時回「未判定」。 */ export function formatTokenLine(usage) { if (!usage) return "未判定"; const inputStr = usage.input === null || usage.input === undefined ? "未判定" : formatTokens(usage.input); const outputStr = formatTokens(usage.output); return `輸入 ${inputStr}/輸出 ${outputStr}`; } const USAGE = `用法:transcript.mjs <子命令> [參數] extract 抽出本輪內容並遮蔽機密後輸出到 stdout duration 估算本輪花費時間,無法判定時輸出「未判定」 tokens 統計本輪 token 用量(輸入/輸出),無法判定時輸出「未判定」 redact 自 stdin 讀取文字,遮蔽機密後輸出到 stdout `; function readStdin() { try { return fs.readFileSync(0, "utf8"); } catch { return ""; } } /** CLI 進入點:解析子命令並執行抽取、估時、統計 token 或遮蔽。 */ function main(argv) { if (!argv.length || argv[0] === "-h" || argv[0] === "--help") { process.stdout.write(USAGE); return 0; } if (argv[0] === "extract") { if (argv.length < 2) return 2; const text = extractTurn(argv[1]); if (!text) return 1; process.stdout.write(redact(text)); return 0; } if (argv[0] === "duration") { if (argv.length < 2) return 2; process.stdout.write(turnDuration(argv[1])); return 0; } if (argv[0] === "tokens") { if (argv.length < 2) return 2; process.stdout.write(formatTokenLine(turnTokens(argv[1]))); return 0; } if (argv[0] === "redact") { process.stdout.write(redact(readStdin())); return 0; } process.stdout.write(USAGE); return 2; } if (import.meta.url === `file://${process.argv[1]}`) { process.exit(main(process.argv.slice(2))); }