import fs from 'fs'; import path from 'path'; import { chatJSON, mapWithConcurrency, LLM_CONCURRENCY } from './llm.js'; import { buildAnalysisPrompt, loadRole, buildVerdictPrompt, buildLocateLinePrompt } from './roles.js'; import { FINDINGS_PATH, EXCLUSIONS_PATH } from './config.js'; import { line, ok, warn } from './log.js'; const LEVELS = ['critical', 'warning', 'info']; /** * 用單一角色分析 diff,呼叫 LLM 取得該角色視角下的 code review 問題並回傳 findings 陣列。 * role 欄位一律以角色定義的 name 為準(覆寫 LLM 回傳值),避免 LLM 自行填入不一致的角色名稱。 * * @param {{name: string}} role - 審查角色定義物件,至少需含 name。 * @param {string} diff - 欲分析的 unified diff 文字內容。 * @returns {Promise>} 有效 findings 陣列(僅保留同時具備 level/location/suggestion 者), * 每筆皆補上 role(角色名稱)與 is_new: true。 * @throws 當 chatJSON 呼叫失敗(LLM 錯誤、額度限制等)時直接拋出例外,本函式不做降級處理。 */ export async function analyzeWithRole(role, diff) { line(`[${role.name}] 開始分析`); const findings = await chatJSON(buildAnalysisPrompt(role), `以下是 Git Diff 內容:\n\n${diff}`); const valid = findings.filter(f => f.level && f.location && f.suggestion) .map(f => ({ ...f, role: role.name, is_new: true })); ok(`[${role.name}] 找到 ${valid.length} 個問題`); return valid; } /** * 將排除設定(頂層陣列、{ exclusions: [] } 或 { excluded_findings: [] })正規化為條目陣列。 * * @param {Array|{exclusions?: Array, excluded_findings?: Array}|*} data - 任意形式的排除資料來源。 * @returns {Array} 對應的排除條目陣列;無法辨識時回傳空陣列。 * @remarks 與 detectExclusionSource 搭配,相容舊有多種 exclusions.json 結構。 */ function normalizeExclusions(data) { if (Array.isArray(data)) return data; if (data && Array.isArray(data.exclusions)) return data.exclusions; if (data && Array.isArray(data.excluded_findings)) return data.excluded_findings; return []; } /** * 偵測排除資料的原始容器格式,回傳格式標籤。 * * @param {Array|{exclusions?: *, excluded_findings?: *}|*} data - 任意形式的排除資料來源。 * @returns {('array'|'exclusions'|'excluded_findings'|'unknown')} 對應的格式標籤。 * @remarks 供 loadExclusions 判斷是否需把非陣列格式改寫成標準頂層陣列。 */ function detectExclusionSource(data) { if (Array.isArray(data)) return 'array'; if (data && Array.isArray(data.exclusions)) return 'exclusions'; if (data && Array.isArray(data.excluded_findings)) return 'excluded_findings'; return 'unknown'; } /** * 以標準格式(2 空白縮排 JSON 陣列、結尾換行、UTF-8)將排除條目寫回檔案,覆蓋原內容。 * * @param {string} fullPath - 目標檔案路徑;上層目錄須事先存在(本函式不建立目錄)。 * @param {Array} exclusions - 欲寫入的排除條目陣列。 * @returns {void} * @throws 檔案寫入失敗(權限不足、目錄不存在等)時拋出 fs 錯誤。 * @remarks 統一輸出格式,使 exclusions.json 永遠是可預期的頂層陣列。 */ function writeCanonicalExclusions(fullPath, exclusions) { fs.writeFileSync(fullPath, JSON.stringify(exclusions, null, 2) + '\n', 'utf8'); } /** * 將檔案 mtime(毫秒時間戳)格式化為 ISO 字串,無效值回傳 'unknown'。 * * @param {number} mtimeMs - 毫秒時間戳(通常為 fs.Stats.mtimeMs)。 * @returns {string} ISO 8601 時間字串,或在輸入非有限數時回傳 'unknown'。 * @remarks 僅用於診斷日誌,呈現舊 findings / exclusions 檔案的修改時間。 */ function formatFileTime(mtimeMs) { if (!Number.isFinite(mtimeMs)) return 'unknown'; return new Date(mtimeMs).toISOString(); } /** * 安全取字串:字串則去頭尾空白,其餘型別(含 null/undefined/數字)一律回傳空字串。 * * @param {*} value - 任意值。 * @returns {string} 去除頭尾空白後的字串,或空字串。 * @remarks 作為 normalizeText、toKeyText、getExclusionText 等的基礎防呆。 */ function cleanText(value) { return typeof value === 'string' ? value.trim() : ''; } const _normalizeTextCache = new Map(); /** * 將文字正規化為比對用形式:先以 cleanText 轉為安全字串,NFKC 正規化、轉小寫, * 並把所有標點/符號/空白字元壓縮成單一空白(再壓縮連續空白、去頭尾空白)。 * 因常對同一段文字重複呼叫(findings × exclusions 笛卡爾積比對), * 以模組層級 Map 對「字串輸入」做 memoization,避免重複執行 NFKC/正則運算。 * * @param {*} value - 任意值;非字串會先經 cleanText 轉為空字串(不會寫入快取)。 * @returns {string} 正規化後、以單一空白分隔的字串(可能為空字串)。 * @remarks 用於 finding 與排除條目文字的雙向「包含」比對(applyExclusions、appendExclusions)。 * 快取為模組層級、程序生命週期內不會清除,需人工確認長期執行(如常駐服務)情境下是否有記憶體成長風險; * 在本專案作為一次性 CI 腳本執行的用法下應無實際影響。 */ export function normalizeText(value) { if (typeof value === 'string' && _normalizeTextCache.has(value)) return _normalizeTextCache.get(value); const result = cleanText(value) .normalize('NFKC') .toLowerCase() .replace(/[\p{P}\p{S}\s]+/gu, ' ') .replace(/\s+/g, ' ') .trim(); if (typeof value === 'string') _normalizeTextCache.set(value, result); return result; } /** * 將文字壓縮成無分隔符的鍵值:NFKC 後移除所有標點/符號/空白。 * * @param {*} value - 任意值;非字串會先經 cleanText 轉為空字串。 * @returns {string} 去除所有分隔符的緊湊字串(可能為空字串)。 * @remarks 用於 normalizeExclusionEntry 的 textKey 與 fingerprint,以及群組鍵。 * 不確定:是否刻意不轉小寫(與 normalizeText 不同),需人工確認此差異是否預期。 */ function toKeyText(value) { return cleanText(value) .normalize('NFKC') .replace(/[\p{P}\p{S}\s]+/gu, '') .trim(); } /** * 從排除條目取出代表性文字,依優先序 original_finding > title > suggestion > reason > note 取第一個非空值。 * * @param {object|null|undefined} exclusion - 排除條目物件(可為 null/undefined)。 * @returns {string} 第一個非空的代表性文字,皆空時回傳空字串。 * @remarks 供 normalizeExclusionEntry 產生比對文字;相容多種人工撰寫的排除欄位命名。 */ function getExclusionText(exclusion) { return cleanText(exclusion?.original_finding) || cleanText(exclusion?.title) || cleanText(exclusion?.suggestion) || cleanText(exclusion?.reason) || cleanText(exclusion?.note); } /** * 正規化單一排除條目,補上 filePath、text、textKey 與唯一 fingerprint,保留原始欄位。 * * @param {object} exclusion - 原始排除條目(可能僅含部分欄位)。 * @param {number} index - 條目在來源陣列中的索引;無文字可用時用於產生 fallback 指紋(entry-N)。 * @returns {object} 合併原欄位與衍生欄位(location、filePath、role、text、textKey、fingerprint)的新物件。 * @remarks fingerprint 以 filePath|role|textKey 組成,缺值以 '*' 或 entry-N 補位,供 dedupeExclusions 去重。 */ function normalizeExclusionEntry(exclusion, index) { const location = cleanText(exclusion?.location); const filePath = location ? location.split(':')[0] : ''; const role = cleanText(exclusion?.role); const text = getExclusionText(exclusion); const textKey = toKeyText(text); const fingerprint = [filePath || '*', role || '*', textKey || `entry-${index + 1}`].join('|'); return { ...exclusion, location: location || null, filePath, role: role || null, text, textKey, fingerprint, }; } /** * 依 fingerprint 去除重複的排除條目,保留首次出現者並維持原順序。 * * @param {Array} exclusions - 已正規化(含 fingerprint)的排除條目陣列。 * @returns {Array} 去重後的排除條目陣列。 * @remarks 須先呼叫 normalizeExclusionEntry 補上 fingerprint,否則缺指紋的條目可能被誤併。 */ function dedupeExclusions(exclusions) { const seen = new Set(); return exclusions.filter(exclusion => { if (seen.has(exclusion.fingerprint)) return false; seen.add(exclusion.fingerprint); return true; }); } /** * 將排除條目依 textKey 分組統計,產生供 AI prompt 使用的群組摘要(含出現次數、涉及路徑與角色、樣本)。 * * @param {Array} exclusions - 已正規化(含 textKey、filePath、role、text、fingerprint)的排除條目。 * @returns {Array<{text: string, count: number, paths: string[], roles: string[], samples: string[]}>} * 依出現次數、涉及路徑數、文字字典序排序的群組摘要陣列。 * @remarks 每組最多保留 2 筆樣本,避免後續 prompt 過長;供 buildExclusionContext 取前 N 組組裝提示。 */ function groupExclusionsForAI(exclusions) { const groups = new Map(); for (const exclusion of exclusions) { const groupKey = exclusion.textKey || exclusion.fingerprint; if (!groups.has(groupKey)) { groups.set(groupKey, { key: groupKey, text: exclusion.text || exclusion.location || exclusion.fingerprint, count: 0, paths: new Set(), roles: new Set(), samples: [], }); } const group = groups.get(groupKey); group.count += 1; if (exclusion.filePath) group.paths.add(exclusion.filePath); if (exclusion.role) group.roles.add(exclusion.role); if (group.samples.length < 2 && exclusion.text) group.samples.push(exclusion.text); } return [...groups.values()] .sort((a, b) => b.count - a.count || b.paths.size - a.paths.size || a.text.localeCompare(b.text)) .map(group => ({ text: group.text, count: group.count, paths: [...group.paths].sort(), roles: [...group.roles].sort(), samples: group.samples, })); } /** * 由原始排除條目建立「已知誤報」上下文:正規化、去重、分組後,產生計數摘要與可直接嵌入 prompt 的文字。 * * @param {Array} exclusions - 原始(未正規化)排除條目陣列。 * @returns {{rawCount: number, uniqueCount: number, groupCount?: number, groups: Array, prompt: string}} * 含計數、前 12 組群組摘要與 prompt 字串;空輸入時 prompt 為空字串且不含 groupCount。 * @remarks 供 loadExclusions 日誌與 filterFalsePositivesWithAI 組裝防守方提示使用;prompt 最多展開 12 類群組。 */ function buildExclusionContext(exclusions) { if (exclusions.length === 0) { return { rawCount: 0, uniqueCount: 0, groups: [], prompt: '', }; } const normalized = exclusions.map((exclusion, index) => normalizeExclusionEntry(exclusion, index)); const unique = dedupeExclusions(normalized); const groups = groupExclusionsForAI(unique); const topGroups = groups.slice(0, 12).map(group => ({ text: group.text, count: group.count, paths: group.paths.slice(0, 4), roles: group.roles.slice(0, 3), samples: group.samples.slice(0, 2), })); const omitted = groups.length - topGroups.length; const promptLines = [ `已知誤報清單(原始 ${exclusions.length} 筆,整理後 ${unique.length} 筆,分成 ${groups.length} 類):`, ...topGroups.map((group, index) => { const parts = [ `${index + 1}. ${group.text}`, `count=${group.count}`, ]; if (group.paths.length > 0) parts.push(`paths=${group.paths.join(', ')}`); if (group.roles.length > 0) parts.push(`roles=${group.roles.join(', ')}`); if (group.samples.length > 0) parts.push(`samples=${group.samples.join(' | ')}`); return `- ${parts.join(' ; ')}`; }), ]; if (omitted > 0) { promptLines.push(`- 另有 ${omitted} 類相似排除條目未展開,請依上述群組規則推論。`); } return { rawCount: exclusions.length, uniqueCount: unique.length, groupCount: groups.length, groups: topGroups, prompt: promptLines.join('\n'), }; } /** * 讀取舊 findings(來源分支 cloned repoDir 下 FINDINGS_PATH 指向的檔案), * 同時相容舊版頂層陣列與新版 wrapper 物件;每筆項目一律標記 is_new: false *(代表非本次新產生),並記錄檔案大小/修改時間等診斷日誌。 * 檔案不存在或讀取失敗時視為空陣列,不拋例外。 * * @param {string} workspace - 來源分支 clone 出的工作目錄根路徑,FINDINGS_PATH 會相對此路徑解析。 * @returns {Array} 舊 findings 陣列,每筆皆含 is_new: false;讀取失敗或檔案不存在時回傳空陣列。 */ export function loadOldFindings(workspace) { const fullPath = path.join(workspace, FINDINGS_PATH); let old = []; if (fs.existsSync(fullPath)) { try { const stat = fs.statSync(fullPath); const data = JSON.parse(fs.readFileSync(fullPath, 'utf8')); const sourceFormat = Array.isArray(data) ? 'array' : (data && Array.isArray(data.findings) ? 'wrapper' : 'unknown'); const rawFindings = Array.isArray(data) ? data : (data && Array.isArray(data.findings) ? data.findings : []); old = rawFindings.map(f => ({ ...f, is_new: false })); line(`讀取舊 findings 檔案: ${fullPath}`); line(`舊 findings 檔案資訊: bytes=${stat.size} mtime=${formatFileTime(stat.mtimeMs)} source=${sourceFormat} path=${path.relative(workspace, fullPath) || fullPath}`); } catch (e) { warn(`讀取舊 findings 失敗: ${e.message},視為空: ${fullPath}`); old = []; } } else { warn(`舊 findings 檔案不存在: ${fullPath}`); } ok(`讀取舊 findings: ${old.length} 筆`); return old; } /** * 合併新舊 findings:以 (role + location + problem + suggestion 前 50 字) 組成的字串為 key, * 過濾掉 newFindings 中與 oldFindings(或 newFindings 自身先出現的項目)key 相同的重複項。 * oldFindings 本身不會互相去重(視為既有基準),回傳陣列為 [...oldFindings, ...去重後的 newFindings]。 * * @param {Array} oldFindings - 既有(上一輪)findings 陣列,作為去重比對基準,原樣保留於結果前段。 * @param {Array} newFindings - 本輪新產生的 findings 陣列,將依 key 去除與 oldFindings 重複者。 * @returns {Array} 合併後的 findings 陣列,不修改傳入的兩個陣列本身。 */ export function mergeFindings(oldFindings, newFindings) { const key = f => `${f.role}|${f.location}|${String(f.problem || '')}|${String(f.suggestion || '').slice(0, 50)}`; const seen = new Set(oldFindings.map(key)); const deduped = newFindings.filter(f => { if (seen.has(key(f))) return false; seen.add(key(f)); return true; }); const merged = [...oldFindings, ...deduped]; ok(`合併結果: 舊=${oldFindings.length} 新(去重後)=${deduped.length} 總計=${merged.length}`); return merged; } /** * 依等級排序(critical > warning > info,未知等級排最後),回傳新陣列,不修改傳入的 findings。 * * @param {Array} findings - 欲排序的 findings 陣列(各筆需含 level 欄位)。 * @returns {Array} 依 critical/warning/info 順序排序後的新陣列;未知等級會排在最後。 */ export function sortByLevel(findings) { const rank = (level) => { const index = LEVELS.indexOf(level); return index === -1 ? LEVELS.length : index; }; return [...findings].sort((a, b) => rank(a.level) - rank(b.level)); } /** * AI 呼叫失敗時的統一降級處理:記錄警告訊息後原樣回傳 findings(不做任何篩選), * 確保 AI(去重/誤報過濾等)暫時性失敗時不會誤刪合法問題。 * * @param {string} label - 用於警告訊息中識別此次失敗的處理名稱(例如「AI 去重」)。 * @param {Array} findings - 發生失敗前的 findings 陣列,將原樣回傳。 * @param {Error} e - 捕捉到的錯誤物件;若 e.response.status 為 402 或 429,訊息會顯示為「額度/限流」,否則顯示 e.message。 * @returns {Array} 原樣回傳的 findings(與傳入的參照相同,未複製)。 */ function fallback(label, findings, e) { const status = e.response?.status; const reason = (status === 402 || status === 429) ? `${status} 額度/限流` : e.message; warn(`${label}失敗(${reason}),降級:保留所有問題`); return findings; } const MAX_LOCATE_ATTEMPTS = 3; /** * 從 location(格式如「檔案:行號」或「檔案:起始行-結束行」)取出行號。 * * @param {string|null|undefined} location - finding 的 location 欄位。 * @returns {number|null} 解析出的(起始)行號;若 location 為空、包含逗號(代表多檔案) * 或不符合「檔案:數字」格式,回傳 null。範圍格式僅回傳起始行號,不回傳結束行號。 */ function findingLine(location) { const s = String(location || '').trim(); if (!s || s.includes(',')) return null; const m = /^(.+?):(\d+)(?:-\d+)?$/.exec(s); return m ? Number(m[2]) : null; } /** * 從整份 unified diff 擷取指定檔案的區段(依 `diff --git a/... b/...` 標頭切分);找不到對應區段時回退回傳整份 diff。 * * @param {string} diff - 完整的 unified diff 文字。 * @param {string} file - 欲擷取的檔案路徑(會以 includes 比對是否出現在 diff --git 標頭的 a/、b/ 路徑中)。 * @returns {string} 該檔案對應的 diff 區段文字;若無法定位,回退回傳原始 diff 字串。 * @remarks 檔名比對採子字串 includes,若 file 恰為另一檔案路徑的子字串,可能誤判擷取到錯誤區段,此為已知限制,需人工確認是否需要更嚴謹的邊界比對。 */ function extractFileDiff(diff, file) { const lines = String(diff || '').split('\n'); const out = []; let capturing = false; for (const l of lines) { if (l.startsWith('diff --git ')) capturing = l.includes(`b/${file}`) || l.includes(`a/${file}`); if (capturing) out.push(l); } return out.length ? out.join('\n') : String(diff || ''); } /** * 對缺行號的 findings 重新詢問原角色補上行號,成功時會就地更新 `location`。 * @param {Array} findings findings 陣列。 * @param {string} diff 完整 unified diff。 * @param {{chatFn?: Function, getRole?: Function, maxAttempts?: number, concurrency?: number}} [deps] * 測試用依賴注入。 * @returns {Promise>} 與傳入相同參照的 findings 陣列。 */ export async function resolveMissingLineNumbers(findings, diff, deps = {}) { const { chatFn = chatJSON, getRole = loadRole, maxAttempts = MAX_LOCATE_ATTEMPTS, concurrency = LLM_CONCURRENCY } = deps; // 只挑「缺行號且有檔名」的 finding;各自以獨立 LLM 子行程並行定位(併發上限見 concurrency)。 const pending = findings.filter(f => findingLine(f.location) == null && String(f.location || '').split(',')[0].split(':')[0].trim()); if (pending.length === 0) return findings; const outcomes = await mapWithConcurrency(pending, concurrency, async (f) => { const file = String(f.location || '').split(',')[0].split(':')[0].trim(); const systemPrompt = buildLocateLinePrompt(getRole(f.role) || { name: f.role }); const userContent = `${JSON.stringify({ file, problem: f.problem, suggestion: f.suggestion })}\n\n--- ${file} Git Diff ---\n${extractFileDiff(diff, file)}`; let located = null; for (let attempt = 1; attempt <= maxAttempts && located == null; attempt++) { try { const res = await chatFn(systemPrompt, userContent); const ln = Number(res?.line); if (Number.isInteger(ln) && ln > 0) located = ln; } catch (e) { warn(`[${f.role}] 行號定位失敗(第 ${attempt}/${maxAttempts} 次): ${e.message}`); } } if (located != null) { f.location = `${file}:${located}`; return true; } warn(`[${f.role}] ${maxAttempts} 次嘗試後仍無法定位行號,保留檔名: ${file}`); return false; }); ok(`補行號: ${outcomes.filter(Boolean).length}/${pending.length} 筆成功定位`); return findings; } /** * 將 findings 精簡為僅含 level、role、location、problem、suggestion 的物件,移除多餘欄位以節省 token。 * * @param {Array} findings - 完整 findings 陣列。 * @returns {Array<{level: *, role: *, location: *, problem: *, suggestion: *}>} 精簡後的 payload 陣列。 * @remarks 送往 LLM 前的瘦身步驟;原始欄位(如 is_new)需由呼叫端事後依鍵補回。 */ function toAIPayload(findings) { return findings.map(({ level, role, location, problem, suggestion }) => ({ level, role, location, problem, suggestion })); } /** * 呼叫 LLM(Paladin 角色)進行語意去重:合併「同位置+同問題本質」的重複 findings,重複者保留等級較高者。 * 為避免 LLM 幻覺出不存在的內容,回傳結果會逐筆以 (location + suggestion 前 50 字) 對應回原始 findings, * 對應不到、結果為空、非陣列或數量超過輸入筆數者,皆視為異常並整批降級為保留所有原始 findings(不篩選)。 * * @param {Array} findings - 欲去重的 findings 陣列;為空陣列時直接原樣回傳。 * @returns {Promise>} 去重後的原始 finding 物件陣列(非 LLM 回傳的精簡版); * AI 呼叫失敗或結果驗證異常時,降級回傳原始 findings(未經任何篩選)。 */ export async function deduplicateWithAI(findings) { if (findings.length === 0) return findings; const systemPrompt = `你是 🛡️ Paladin(聖騎士),這座程式碼競技場沉穩公正的裁判。攻擊方提出了一批程式碼審查問題(JSON 陣列)。請就事論事,把「同檔案位置 + 同問題本質」的重複指控合併,重複者只保留等級較高的一條(critical > warning > info)。只回傳去重後的 JSON 陣列,不要有其他文字。`; try { const result = await chatJSON(systemPrompt, JSON.stringify(toAIPayload(findings))); // 去重結果數量不得超過輸入(避免 LLM 無中生有),且每筆都必須能對應回原始 finding。 if (Array.isArray(result) && result.length > 0 && result.length <= findings.length) { const keyOf = f => `${f.location}|${String(f.suggestion).slice(0, 50)}`; const origMap = new Map(findings.map(f => [keyOf(f), f])); // 只保留能對應回原始 finding 的項目,丟棄無法對應(可能為幻覺)的結果 const mapped = result.map(r => origMap.get(keyOf(r))).filter(Boolean); if (mapped.length > 0) { ok(`AI 去重: ${findings.length} -> ${mapped.length} 筆`); return mapped; } } throw new Error('AI 去重結果異常(空、超量或無法對應原始 findings)'); } catch (e) { return fallback('AI 去重', findings, e); } } /** * 讀取排除問題檔案(來源分支 cloned repoDir 下 EXCLUSIONS_PATH),正規化並去重後回傳。 * 若偵測到檔案為舊格式(非頂層陣列,如 { exclusions: [...] } 或 { excluded_findings: [...] }), * 會就地把該檔案覆寫為標準頂層陣列格式(若提供 mirrorWorkspace 且路徑不同,也會同步寫入 mirror 目錄)。 * 檔案不存在或讀取/解析失敗時,皆視為空陣列,不拋出例外。 * * @param {string} workspace - 來源分支工作目錄根路徑,EXCLUSIONS_PATH 會相對此路徑解析。 * @param {object|null} [repoState] - 可選的來源分支狀態(branch/shortSha 或 headSha/commitTime),僅用於診斷日誌。 * @param {string|null} [mirrorWorkspace] - 可選的鏡像工作目錄;當原始格式非頂層陣列時,會同步覆寫此目錄下的 exclusions.json。 * @returns {Array} 正規化並去重後的排除條目陣列;讀取失敗或檔案不存在時回傳空陣列。 */ export function loadExclusions(workspace, repoState = null, mirrorWorkspace = null) { const fullPath = path.join(workspace, EXCLUSIONS_PATH); if (!fs.existsSync(fullPath)) { warn(`排除問題檔案不存在,視為空: ${fullPath}`); if (repoState) { const branch = repoState.branch || 'detached'; const shortSha = repoState.shortSha || repoState.headSha || 'unknown'; line(`來源分支狀態: branch=${branch} commit=${shortSha} commit_time=${repoState.commitTime || 'unknown'}`); } ok('讀取排除問題: raw=0 normalized=0 筆'); return []; } let exclusions = []; let rawCount = 0; try { const stat = fs.statSync(fullPath); const data = JSON.parse(fs.readFileSync(fullPath, 'utf8')); const sourceFormat = detectExclusionSource(data); const normalizedSource = normalizeExclusions(data); rawCount = normalizedSource.length; exclusions = dedupeExclusions(normalizedSource.map((exclusion, index) => normalizeExclusionEntry(exclusion, index))); const branch = repoState?.branch || 'detached'; const shortSha = repoState?.shortSha || repoState?.headSha || 'unknown'; const commitTime = repoState?.commitTime || 'unknown'; line(`讀取排除問題檔案: ${fullPath}`); line(`來源分支狀態: branch=${branch} commit=${shortSha} commit_time=${commitTime}`); line(`檔案資訊: bytes=${stat.size} mtime=${formatFileTime(stat.mtimeMs)} raw=${rawCount} normalized=${exclusions.length} path=${path.relative(workspace, fullPath) || fullPath}`); if (sourceFormat !== 'array') { writeCanonicalExclusions(fullPath, normalizedSource); if (mirrorWorkspace && path.resolve(mirrorWorkspace) !== path.resolve(workspace)) { const mirrorPath = path.join(mirrorWorkspace, EXCLUSIONS_PATH); fs.mkdirSync(path.dirname(mirrorPath), { recursive: true }); writeCanonicalExclusions(mirrorPath, normalizedSource); } line(`排除問題格式已修正為頂層陣列: source=${sourceFormat} -> array`); } } catch (e) { warn(`讀取排除問題失敗: ${e.message},視為空: ${fullPath}`); exclusions = []; } const summary = buildExclusionContext(exclusions); ok(`讀取排除問題: raw=${rawCount} normalized=${exclusions.length} groups=${summary.groupCount} 筆`); return exclusions; } /** * 把新的排除條目(raw 形式,未經 normalizeExclusionEntry 加工)append 到 exclusions.json, * 以「檔案路徑(location 冒號前段)+ normalizeText 後的原文」為簽名去重後, * 以頂層陣列格式寫回 workspace(及提供且路徑不同的 mirrorWorkspace)。 * * @param {string} workspace - 目標工作目錄,EXCLUSIONS_PATH 相對此路徑解析並寫入。 * @param {Array} newEntries - 欲新增的排除條目(raw 形式);為空或未提供時直接回傳 null(無操作)。 * @param {string|null} [mirrorWorkspace] - 可選鏡像目錄;提供且與 workspace 路徑不同時,會同步寫入相同內容。 * @returns {Array|null} 合併後的 raw 排除條目陣列;newEntries 為空時回傳 null; * 若 newEntries 皆與既有條目重複(無實際新增)則回傳既有陣列(未寫檔)。 */ export function appendExclusions(workspace, newEntries, mirrorWorkspace = null) { if (!newEntries || newEntries.length === 0) return null; const fileOf = loc => String(loc || '').split(':')[0].trim(); const sigOf = e => `${fileOf(e.location)}|${normalizeText(e.original_finding || e.suggestion || e.text || e.title || '')}`; const fullPath = path.join(workspace, EXCLUSIONS_PATH); let existing = []; if (fs.existsSync(fullPath)) { try { existing = normalizeExclusions(JSON.parse(fs.readFileSync(fullPath, 'utf8'))); } catch (e) { warn(`讀取排除問題以追加失敗,視為空: ${e.message}`); existing = []; } } const seen = new Set(existing.map(sigOf)); const additions = newEntries.filter(e => { const sig = sigOf(e); if (seen.has(sig)) return false; seen.add(sig); return true; }); if (additions.length === 0) { line(`誤報排除無新增(皆已存在): 候選 ${newEntries.length} 筆`); return existing; } const merged = [...existing, ...additions]; const targets = [workspace]; if (mirrorWorkspace && path.resolve(mirrorWorkspace) !== path.resolve(workspace)) targets.push(mirrorWorkspace); for (const dir of targets) { const target = path.join(dir, EXCLUSIONS_PATH); fs.mkdirSync(path.dirname(target), { recursive: true }); writeCanonicalExclusions(target, merged); } ok(`誤報寫入 exclusions: 新增 ${additions.length} 筆(總計 ${merged.length} 筆)`); return merged; } /** * 套用排除規則,過濾掉符合任一排除條件的 findings。 * exclusions 為空時原樣回傳 findings(新陣列,不修改原輸入)。 * * 比對規則(對每個 exclusion,locationMatches && roleMatches && (有指定 path 或 role ? 一律視為符合 : textMatches)): * - location 只比對檔案路徑(忽略行號),exclusion 未指定 filePath 時視為萬用; * - role 未指定時視為萬用,否則需與 finding.role 完全相等; * - 僅當 exclusion 同時未指定 filePath 與 role 時,才會實際比對正規化後文字(suggestion/title 等)是否互相包含。 * * @param {Array} findings - 欲過濾的 findings 陣列。 * @param {Array} exclusions - 排除條目陣列(建議為已正規化含 filePath 的條目)。 * @returns {Array} 過濾後的新陣列。 * @remarks 「只要 exclusion 指定了 filePath 或 role,文字比對即完全略過」是否為刻意設計,需人工確認; * 若非刻意,可能造成排除範圍比預期寬(例如同檔案下所有問題都被排除,而非僅特定描述的問題)。 */ export function applyExclusions(findings, exclusions) { if (exclusions.length === 0) return findings; const before = findings.length; const filtered = findings.filter(f => !exclusions.some(ex => { const fPath = String(f.location).split(':')[0]; const exPath = ex.filePath || (ex.location ? String(ex.location).split(':')[0] : null); const findingText = normalizeText(f.suggestion || f.title || ''); const exclusionText = normalizeText(ex.text || ex.original_finding || ex.suggestion || ex.title || ex.textKey || ''); const locationMatches = (!exPath || fPath === exPath); const roleMatches = (!ex.role || ex.role === f.role); const textMatches = !exclusionText || !findingText || findingText.includes(exclusionText) || exclusionText.includes(findingText); return locationMatches && roleMatches && (exPath || ex.role ? true : textMatches); })); ok(`排除過濾: ${before} -> ${filtered.length} 筆(排除 ${before - filtered.length} 筆)`); return filtered; } /** * 派一個「防守方」角色裁決單一 finding 是否為誤報;任何失敗都保守視為「成立」(即保留該問題)。 * * @param {object} finding - 欲裁決的單一 finding。 * @param {object} defender - 防守方角色定義(通常為 Paladin),供 buildVerdictPrompt 組系統提示。 * @param {string} exclusionHint - 已知誤報清單的提示文字(可為空字串),供 AI 判斷是否與已知誤報類似。 * @param {Function} chatFn - 實際呼叫 LLM 的函式(簽名同 chatJSON),供測試時注入替換。 * @returns {Promise} true 表示裁決為誤報(應剔除);false 表示成立或裁決失敗(保守保留)。 */ async function judgeFindingIsFalsePositive(finding, defender, exclusionHint, chatFn) { const systemPrompt = buildVerdictPrompt(defender, exclusionHint); try { const result = await chatFn(systemPrompt, JSON.stringify(toAIPayload([finding])[0])); return result?.verdict === 'false_positive'; } catch (e) { warn(`誤報裁決失敗(保守視為成立): ${finding.location} error=${e.message}`); return false; } } /** * 由「防守方」角色(固定為 Paladin)逐條裁決 findings 是否為誤報,剔除誤報、保留成立者。 * 多個問題時各派一個裁決任務平行處理(併發上限 LLM_CONCURRENCY);任一裁決失敗保守保留該問題,不中斷流程。 * * @param {Array} findings - 欲裁決的 findings 陣列;為空陣列時直接原樣回傳。 * @param {Array} [exclusions=[]] - 已知誤報排除條目,用於組裝提示,引導 AI 對相似的誤報更寬鬆判定。 * @param {Function} [chatFn=chatJSON] - 實際呼叫 LLM 的函式,供測試時注入替換。 * @returns {Promise>} 裁決為「非誤報」而保留下來的原始 finding 物件陣列。 */ export async function filterFalsePositivesWithAI(findings, exclusions = [], chatFn = chatJSON) { if (findings.length === 0) return findings; const defender = loadRole('Paladin'); const exclusionContext = buildExclusionContext(exclusions); const exclusionHint = exclusionContext.prompt ? `${exclusionContext.prompt}\n規則:若此 finding 與上述任何一類的路徑、角色或描述高度相似,優先視為誤報或不適用。` : ''; // 每條 finding 各派一個防守方 sub-agent 裁決;併發上限與其他 LLM 任務共用 LLM_CONCURRENCY(預設不限制)。 const verdicts = await mapWithConcurrency(findings, LLM_CONCURRENCY, async (f) => ({ f, isFP: await judgeFindingIsFalsePositive(f, defender, exclusionHint, chatFn), })); const kept = verdicts.filter(v => !v.isFP).map(v => v.f); ok(`AI 誤報過濾(防守方${findings.length > 1 ? '平行' : ''}裁決): ${findings.length} -> ${kept.length} 筆`); return kept; }