Files
ai-code-review/src/findings.js
T

676 lines
34 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import fs from 'fs';
import path from 'path';
import { chatJSON, mapWithConcurrency, LLM_CONCURRENCY } from './llm.js';
import { buildAnalysisPrompt, loadRole, buildVerdictPrompt, buildLocateLinePrompt } from './roles.js';
import { FINDINGS_PATH, EXCLUSIONS_PATH } from './config.js';
import { line, ok, warn } from './log.js';
const LEVELS = ['critical', 'warning', 'info'];
/**
* 用單一角色分析 diff,呼叫 LLM 取得該角色視角下的 code review 問題並回傳 findings 陣列。
* role 欄位一律以角色定義的 name 為準(覆寫 LLM 回傳值),避免 LLM 自行填入不一致的角色名稱。
*
* @param {{name: string}} role - 審查角色定義物件,至少需含 name。
* @param {string} diff - 欲分析的 unified diff 文字內容。
* @returns {Promise<Array<object>>} 有效 findings 陣列(僅保留同時具備 level/location/suggestion 者),
* 每筆皆補上 role(角色名稱)與 is_new: true。
* @throws 當 chatJSON 呼叫失敗(LLM 錯誤、額度限制等)時直接拋出例外,本函式不做降級處理。
*/
export async function analyzeWithRole(role, diff) {
line(`[${role.name}] 開始分析`);
const findings = await chatJSON(buildAnalysisPrompt(role), `以下是 Git Diff 內容:\n\n${diff}`);
const valid = findings.filter(f => f.level && f.location && f.suggestion)
.map(f => ({ ...f, role: role.name, is_new: true }));
ok(`[${role.name}] 找到 ${valid.length} 個問題`);
return valid;
}
/**
* 將排除設定(頂層陣列、{ exclusions: [] } 或 { excluded_findings: [] })正規化為條目陣列。
*
* @param {Array<object>|{exclusions?: Array<object>, excluded_findings?: Array<object>}|*} data - 任意形式的排除資料來源。
* @returns {Array<object>} 對應的排除條目陣列;無法辨識時回傳空陣列。
* @remarks 與 detectExclusionSource 搭配,相容舊有多種 exclusions.json 結構。
*/
function normalizeExclusions(data) {
if (Array.isArray(data)) return data;
if (data && Array.isArray(data.exclusions)) return data.exclusions;
if (data && Array.isArray(data.excluded_findings)) return data.excluded_findings;
return [];
}
/**
* 偵測排除資料的原始容器格式,回傳格式標籤。
*
* @param {Array<object>|{exclusions?: *, excluded_findings?: *}|*} data - 任意形式的排除資料來源。
* @returns {('array'|'exclusions'|'excluded_findings'|'unknown')} 對應的格式標籤。
* @remarks 供 loadExclusions 判斷是否需把非陣列格式改寫成標準頂層陣列。
*/
function detectExclusionSource(data) {
if (Array.isArray(data)) return 'array';
if (data && Array.isArray(data.exclusions)) return 'exclusions';
if (data && Array.isArray(data.excluded_findings)) return 'excluded_findings';
return 'unknown';
}
/**
* 以標準格式(2 空白縮排 JSON 陣列、結尾換行、UTF-8)將排除條目寫回檔案,覆蓋原內容。
*
* @param {string} fullPath - 目標檔案路徑;上層目錄須事先存在(本函式不建立目錄)。
* @param {Array<object>} exclusions - 欲寫入的排除條目陣列。
* @returns {void}
* @throws 檔案寫入失敗(權限不足、目錄不存在等)時拋出 fs 錯誤。
* @remarks 統一輸出格式,使 exclusions.json 永遠是可預期的頂層陣列。
*/
function writeCanonicalExclusions(fullPath, exclusions) {
fs.writeFileSync(fullPath, JSON.stringify(exclusions, null, 2) + '\n', 'utf8');
}
/**
* 將檔案 mtime(毫秒時間戳)格式化為 ISO 字串,無效值回傳 'unknown'。
*
* @param {number} mtimeMs - 毫秒時間戳(通常為 fs.Stats.mtimeMs)。
* @returns {string} ISO 8601 時間字串,或在輸入非有限數時回傳 'unknown'。
* @remarks 僅用於診斷日誌,呈現舊 findings / exclusions 檔案的修改時間。
*/
function formatFileTime(mtimeMs) {
if (!Number.isFinite(mtimeMs)) return 'unknown';
return new Date(mtimeMs).toISOString();
}
/**
* 安全取字串:字串則去頭尾空白,其餘型別(含 null/undefined/數字)一律回傳空字串。
*
* @param {*} value - 任意值。
* @returns {string} 去除頭尾空白後的字串,或空字串。
* @remarks 作為 normalizeText、toKeyText、getExclusionText 等的基礎防呆。
*/
function cleanText(value) {
return typeof value === 'string' ? value.trim() : '';
}
const _normalizeTextCache = new Map();
/**
* 將文字正規化為比對用形式:先以 cleanText 轉為安全字串,NFKC 正規化、轉小寫,
* 並把所有標點/符號/空白字元壓縮成單一空白(再壓縮連續空白、去頭尾空白)。
* 因常對同一段文字重複呼叫(findings × exclusions 笛卡爾積比對),
* 以模組層級 Map 對「字串輸入」做 memoization,避免重複執行 NFKC/正則運算。
*
* @param {*} value - 任意值;非字串會先經 cleanText 轉為空字串(不會寫入快取)。
* @returns {string} 正規化後、以單一空白分隔的字串(可能為空字串)。
* @remarks 用於 finding 與排除條目文字的雙向「包含」比對(applyExclusions、appendExclusions)。
* 快取為模組層級、程序生命週期內不會清除,需人工確認長期執行(如常駐服務)情境下是否有記憶體成長風險;
* 在本專案作為一次性 CI 腳本執行的用法下應無實際影響。
*/
export function normalizeText(value) {
if (typeof value === 'string' && _normalizeTextCache.has(value)) return _normalizeTextCache.get(value);
const result = cleanText(value)
.normalize('NFKC')
.toLowerCase()
.replace(/[\p{P}\p{S}\s]+/gu, ' ')
.replace(/\s+/g, ' ')
.trim();
if (typeof value === 'string') _normalizeTextCache.set(value, result);
return result;
}
/**
* 將文字壓縮成無分隔符的鍵值:NFKC 後移除所有標點/符號/空白。
*
* @param {*} value - 任意值;非字串會先經 cleanText 轉為空字串。
* @returns {string} 去除所有分隔符的緊湊字串(可能為空字串)。
* @remarks 用於 normalizeExclusionEntry 的 textKey 與 fingerprint,以及群組鍵。
* 不確定:是否刻意不轉小寫(與 normalizeText 不同),需人工確認此差異是否預期。
*/
function toKeyText(value) {
return cleanText(value)
.normalize('NFKC')
.replace(/[\p{P}\p{S}\s]+/gu, '')
.trim();
}
/**
* 從排除條目取出代表性文字,依優先序 original_finding > title > suggestion > reason > note 取第一個非空值。
*
* @param {object|null|undefined} exclusion - 排除條目物件(可為 null/undefined)。
* @returns {string} 第一個非空的代表性文字,皆空時回傳空字串。
* @remarks 供 normalizeExclusionEntry 產生比對文字;相容多種人工撰寫的排除欄位命名。
*/
function getExclusionText(exclusion) {
return cleanText(exclusion?.original_finding)
|| cleanText(exclusion?.title)
|| cleanText(exclusion?.suggestion)
|| cleanText(exclusion?.reason)
|| cleanText(exclusion?.note);
}
/**
* 正規化單一排除條目,補上 filePath、text、textKey 與唯一 fingerprint,保留原始欄位。
*
* @param {object} exclusion - 原始排除條目(可能僅含部分欄位)。
* @param {number} index - 條目在來源陣列中的索引;無文字可用時用於產生 fallback 指紋(entry-N)。
* @returns {object} 合併原欄位與衍生欄位(location、filePath、role、text、textKey、fingerprint)的新物件。
* @remarks fingerprint 以 filePath|role|textKey 組成,缺值以 '*' 或 entry-N 補位,供 dedupeExclusions 去重。
*/
function normalizeExclusionEntry(exclusion, index) {
const location = cleanText(exclusion?.location);
const filePath = location ? location.split(':')[0] : '';
const role = cleanText(exclusion?.role);
const text = getExclusionText(exclusion);
const textKey = toKeyText(text);
const fingerprint = [filePath || '*', role || '*', textKey || `entry-${index + 1}`].join('|');
return {
...exclusion,
location: location || null,
filePath,
role: role || null,
text,
textKey,
fingerprint,
};
}
/**
* 依 fingerprint 去除重複的排除條目,保留首次出現者並維持原順序。
*
* @param {Array<object>} exclusions - 已正規化(含 fingerprint)的排除條目陣列。
* @returns {Array<object>} 去重後的排除條目陣列。
* @remarks 須先呼叫 normalizeExclusionEntry 補上 fingerprint,否則缺指紋的條目可能被誤併。
*/
function dedupeExclusions(exclusions) {
const seen = new Set();
return exclusions.filter(exclusion => {
if (seen.has(exclusion.fingerprint)) return false;
seen.add(exclusion.fingerprint);
return true;
});
}
/**
* 將排除條目依 textKey 分組統計,產生供 AI prompt 使用的群組摘要(含出現次數、涉及路徑與角色、樣本)。
*
* @param {Array<object>} exclusions - 已正規化(含 textKey、filePath、role、text、fingerprint)的排除條目。
* @returns {Array<{text: string, count: number, paths: string[], roles: string[], samples: string[]}>}
* 依出現次數、涉及路徑數、文字字典序排序的群組摘要陣列。
* @remarks 每組最多保留 2 筆樣本,避免後續 prompt 過長;供 buildExclusionContext 取前 N 組組裝提示。
*/
function groupExclusionsForAI(exclusions) {
const groups = new Map();
for (const exclusion of exclusions) {
const groupKey = exclusion.textKey || exclusion.fingerprint;
if (!groups.has(groupKey)) {
groups.set(groupKey, {
key: groupKey,
text: exclusion.text || exclusion.location || exclusion.fingerprint,
count: 0,
paths: new Set(),
roles: new Set(),
samples: [],
});
}
const group = groups.get(groupKey);
group.count += 1;
if (exclusion.filePath) group.paths.add(exclusion.filePath);
if (exclusion.role) group.roles.add(exclusion.role);
if (group.samples.length < 2 && exclusion.text) group.samples.push(exclusion.text);
}
return [...groups.values()]
.sort((a, b) => b.count - a.count || b.paths.size - a.paths.size || a.text.localeCompare(b.text))
.map(group => ({
text: group.text,
count: group.count,
paths: [...group.paths].sort(),
roles: [...group.roles].sort(),
samples: group.samples,
}));
}
/**
* 由原始排除條目建立「已知誤報」上下文:正規化、去重、分組後,產生計數摘要與可直接嵌入 prompt 的文字。
*
* @param {Array<object>} exclusions - 原始(未正規化)排除條目陣列。
* @returns {{rawCount: number, uniqueCount: number, groupCount?: number, groups: Array<object>, prompt: string}}
* 含計數、前 12 組群組摘要與 prompt 字串;空輸入時 prompt 為空字串且不含 groupCount。
* @remarks 供 loadExclusions 日誌與 filterFalsePositivesWithAI 組裝防守方提示使用;prompt 最多展開 12 類群組。
*/
function buildExclusionContext(exclusions) {
if (exclusions.length === 0) {
return {
rawCount: 0,
uniqueCount: 0,
groups: [],
prompt: '',
};
}
const normalized = exclusions.map((exclusion, index) => normalizeExclusionEntry(exclusion, index));
const unique = dedupeExclusions(normalized);
const groups = groupExclusionsForAI(unique);
const topGroups = groups.slice(0, 12).map(group => ({
text: group.text,
count: group.count,
paths: group.paths.slice(0, 4),
roles: group.roles.slice(0, 3),
samples: group.samples.slice(0, 2),
}));
const omitted = groups.length - topGroups.length;
const promptLines = [
`已知誤報清單(原始 ${exclusions.length} 筆,整理後 ${unique.length} 筆,分成 ${groups.length} 類):`,
...topGroups.map((group, index) => {
const parts = [
`${index + 1}. ${group.text}`,
`count=${group.count}`,
];
if (group.paths.length > 0) parts.push(`paths=${group.paths.join(', ')}`);
if (group.roles.length > 0) parts.push(`roles=${group.roles.join(', ')}`);
if (group.samples.length > 0) parts.push(`samples=${group.samples.join(' | ')}`);
return `- ${parts.join(' ; ')}`;
}),
];
if (omitted > 0) {
promptLines.push(`- 另有 ${omitted} 類相似排除條目未展開,請依上述群組規則推論。`);
}
return {
rawCount: exclusions.length,
uniqueCount: unique.length,
groupCount: groups.length,
groups: topGroups,
prompt: promptLines.join('\n'),
};
}
/**
* 讀取舊 findings(來源分支 cloned repoDir 下 FINDINGS_PATH 指向的檔案),
* 同時相容舊版頂層陣列與新版 wrapper 物件;每筆項目一律標記 is_new: false
*(代表非本次新產生),並記錄檔案大小/修改時間等診斷日誌。
* 檔案不存在或讀取失敗時視為空陣列,不拋例外。
*
* @param {string} workspace - 來源分支 clone 出的工作目錄根路徑,FINDINGS_PATH 會相對此路徑解析。
* @returns {Array<object>} 舊 findings 陣列,每筆皆含 is_new: false;讀取失敗或檔案不存在時回傳空陣列。
*/
export function loadOldFindings(workspace) {
const fullPath = path.join(workspace, FINDINGS_PATH);
let old = [];
if (fs.existsSync(fullPath)) {
try {
const stat = fs.statSync(fullPath);
const data = JSON.parse(fs.readFileSync(fullPath, 'utf8'));
const sourceFormat = Array.isArray(data) ? 'array' : (data && Array.isArray(data.findings) ? 'wrapper' : 'unknown');
const rawFindings = Array.isArray(data) ? data : (data && Array.isArray(data.findings) ? data.findings : []);
old = rawFindings.map(f => ({ ...f, is_new: false }));
line(`讀取舊 findings 檔案: ${fullPath}`);
line(`舊 findings 檔案資訊: bytes=${stat.size} mtime=${formatFileTime(stat.mtimeMs)} source=${sourceFormat} path=${path.relative(workspace, fullPath) || fullPath}`);
} catch (e) {
warn(`讀取舊 findings 失敗: ${e.message},視為空: ${fullPath}`);
old = [];
}
} else {
warn(`舊 findings 檔案不存在: ${fullPath}`);
}
ok(`讀取舊 findings: ${old.length} 筆`);
return old;
}
/**
* 合併新舊 findings:以 (role + location + problem + suggestion 前 50 字) 組成的字串為 key,
* 過濾掉 newFindings 中與 oldFindings(或 newFindings 自身先出現的項目)key 相同的重複項。
* oldFindings 本身不會互相去重(視為既有基準),回傳陣列為 [...oldFindings, ...去重後的 newFindings]。
*
* @param {Array<object>} oldFindings - 既有(上一輪)findings 陣列,作為去重比對基準,原樣保留於結果前段。
* @param {Array<object>} newFindings - 本輪新產生的 findings 陣列,將依 key 去除與 oldFindings 重複者。
* @returns {Array<object>} 合併後的 findings 陣列,不修改傳入的兩個陣列本身。
*/
export function mergeFindings(oldFindings, newFindings) {
const key = f => `${f.role}|${f.location}|${String(f.problem || '')}|${String(f.suggestion || '').slice(0, 50)}`;
const seen = new Set(oldFindings.map(key));
const deduped = newFindings.filter(f => {
if (seen.has(key(f))) return false;
seen.add(key(f));
return true;
});
const merged = [...oldFindings, ...deduped];
ok(`合併結果: 舊=${oldFindings.length} 新(去重後)=${deduped.length} 總計=${merged.length}`);
return merged;
}
/**
* 依等級排序(critical > warning > info,未知等級排最後),回傳新陣列,不修改傳入的 findings。
*
* @param {Array<object>} findings - 欲排序的 findings 陣列(各筆需含 level 欄位)。
* @returns {Array<object>} 依 critical/warning/info 順序排序後的新陣列;未知等級會排在最後。
*/
export function sortByLevel(findings) {
const rank = (level) => {
const index = LEVELS.indexOf(level);
return index === -1 ? LEVELS.length : index;
};
return [...findings].sort((a, b) => rank(a.level) - rank(b.level));
}
/**
* AI 呼叫失敗時的統一降級處理:記錄警告訊息後原樣回傳 findings(不做任何篩選),
* 確保 AI(去重/誤報過濾等)暫時性失敗時不會誤刪合法問題。
*
* @param {string} label - 用於警告訊息中識別此次失敗的處理名稱(例如「AI 去重」)。
* @param {Array<object>} findings - 發生失敗前的 findings 陣列,將原樣回傳。
* @param {Error} e - 捕捉到的錯誤物件;若 e.response.status 為 402 或 429,訊息會顯示為「額度/限流」,否則顯示 e.message。
* @returns {Array<object>} 原樣回傳的 findings(與傳入的參照相同,未複製)。
*/
function fallback(label, findings, e) {
const status = e.response?.status;
const reason = (status === 402 || status === 429) ? `${status} 額度/限流` : e.message;
warn(`${label}失敗(${reason}),降級:保留所有問題`);
return findings;
}
const MAX_LOCATE_ATTEMPTS = 3;
/**
* 從 location(格式如「檔案:行號」或「檔案:起始行-結束行」)取出行號。
*
* @param {string|null|undefined} location - finding 的 location 欄位。
* @returns {number|null} 解析出的(起始)行號;若 location 為空、包含逗號(代表多檔案)
* 或不符合「檔案:數字」格式,回傳 null。範圍格式僅回傳起始行號,不回傳結束行號。
*/
function findingLine(location) {
const s = String(location || '').trim();
if (!s || s.includes(',')) return null;
const m = /^(.+?):(\d+)(?:-\d+)?$/.exec(s);
return m ? Number(m[2]) : null;
}
/**
* 從整份 unified diff 擷取指定檔案的區段(依 `diff --git a/... b/...` 標頭切分);找不到對應區段時回退回傳整份 diff。
*
* @param {string} diff - 完整的 unified diff 文字。
* @param {string} file - 欲擷取的檔案路徑(會以 includes 比對是否出現在 diff --git 標頭的 a/、b/ 路徑中)。
* @returns {string} 該檔案對應的 diff 區段文字;若無法定位,回退回傳原始 diff 字串。
* @remarks 檔名比對採子字串 includes,若 file 恰為另一檔案路徑的子字串,可能誤判擷取到錯誤區段,此為已知限制,需人工確認是否需要更嚴謹的邊界比對。
*/
function extractFileDiff(diff, file) {
const lines = String(diff || '').split('\n');
const out = [];
let capturing = false;
for (const l of lines) {
if (l.startsWith('diff --git ')) capturing = l.includes(`b/${file}`) || l.includes(`a/${file}`);
if (capturing) out.push(l);
}
return out.length ? out.join('\n') : String(diff || '');
}
/**
* 對缺行號的 findings 重新詢問原角色補上行號,成功時會就地更新 `location`。
* @param {Array<object>} findings findings 陣列。
* @param {string} diff 完整 unified diff。
* @param {{chatFn?: Function, getRole?: Function, maxAttempts?: number, concurrency?: number}} [deps]
* 測試用依賴注入。
* @returns {Promise<Array<object>>} 與傳入相同參照的 findings 陣列。
*/
export async function resolveMissingLineNumbers(findings, diff, deps = {}) {
const { chatFn = chatJSON, getRole = loadRole, maxAttempts = MAX_LOCATE_ATTEMPTS, concurrency = LLM_CONCURRENCY } = deps;
// 只挑「缺行號且有檔名」的 finding;各自以獨立 LLM 子行程並行定位(併發上限見 concurrency)。
const pending = findings.filter(f => findingLine(f.location) == null
&& String(f.location || '').split(',')[0].split(':')[0].trim());
if (pending.length === 0) return findings;
const outcomes = await mapWithConcurrency(pending, concurrency, async (f) => {
const file = String(f.location || '').split(',')[0].split(':')[0].trim();
const systemPrompt = buildLocateLinePrompt(getRole(f.role) || { name: f.role });
const userContent = `${JSON.stringify({ file, problem: f.problem, suggestion: f.suggestion })}\n\n--- ${file} Git Diff ---\n${extractFileDiff(diff, file)}`;
let located = null;
for (let attempt = 1; attempt <= maxAttempts && located == null; attempt++) {
try {
const res = await chatFn(systemPrompt, userContent);
const ln = Number(res?.line);
if (Number.isInteger(ln) && ln > 0) located = ln;
} catch (e) {
warn(`[${f.role}] 行號定位失敗(第 ${attempt}/${maxAttempts} 次): ${e.message}`);
}
}
if (located != null) {
f.location = `${file}:${located}`;
return true;
}
warn(`[${f.role}] ${maxAttempts} 次嘗試後仍無法定位行號,保留檔名: ${file}`);
return false;
});
ok(`補行號: ${outcomes.filter(Boolean).length}/${pending.length} 筆成功定位`);
return findings;
}
/**
* 將 findings 精簡為僅含 level、role、location、problem、suggestion 的物件,移除多餘欄位以節省 token。
*
* @param {Array<object>} findings - 完整 findings 陣列。
* @returns {Array<{level: *, role: *, location: *, problem: *, suggestion: *}>} 精簡後的 payload 陣列。
* @remarks 送往 LLM 前的瘦身步驟;原始欄位(如 is_new)需由呼叫端事後依鍵補回。
*/
function toAIPayload(findings) {
return findings.map(({ level, role, location, problem, suggestion }) => ({ level, role, location, problem, suggestion }));
}
/**
* 呼叫 LLM(Paladin 角色)進行語意去重:合併「同位置+同問題本質」的重複 findings,重複者保留等級較高者。
* 為避免 LLM 幻覺出不存在的內容,回傳結果會逐筆以 (location + suggestion 前 50 字) 對應回原始 findings,
* 對應不到、結果為空、非陣列或數量超過輸入筆數者,皆視為異常並整批降級為保留所有原始 findings(不篩選)。
*
* @param {Array<object>} findings - 欲去重的 findings 陣列;為空陣列時直接原樣回傳。
* @returns {Promise<Array<object>>} 去重後的原始 finding 物件陣列(非 LLM 回傳的精簡版);
* AI 呼叫失敗或結果驗證異常時,降級回傳原始 findings(未經任何篩選)。
*/
export async function deduplicateWithAI(findings) {
if (findings.length === 0) return findings;
const systemPrompt = `你是 🛡️ Paladin(聖騎士),這座程式碼競技場沉穩公正的裁判。攻擊方提出了一批程式碼審查問題(JSON 陣列)。請就事論事,把「同檔案位置 + 同問題本質」的重複指控合併,重複者只保留等級較高的一條(critical > warning > info)。只回傳去重後的 JSON 陣列,不要有其他文字。`;
try {
const result = await chatJSON(systemPrompt, JSON.stringify(toAIPayload(findings)));
// 去重結果數量不得超過輸入(避免 LLM 無中生有),且每筆都必須能對應回原始 finding。
if (Array.isArray(result) && result.length > 0 && result.length <= findings.length) {
const keyOf = f => `${f.location}|${String(f.suggestion).slice(0, 50)}`;
const origMap = new Map(findings.map(f => [keyOf(f), f]));
// 只保留能對應回原始 finding 的項目,丟棄無法對應(可能為幻覺)的結果
const mapped = result.map(r => origMap.get(keyOf(r))).filter(Boolean);
if (mapped.length > 0) {
ok(`AI 去重: ${findings.length} -> ${mapped.length} 筆`);
return mapped;
}
}
throw new Error('AI 去重結果異常(空、超量或無法對應原始 findings)');
} catch (e) {
return fallback('AI 去重', findings, e);
}
}
/**
* 讀取排除問題檔案(來源分支 cloned repoDir 下 EXCLUSIONS_PATH),正規化並去重後回傳。
* 若偵測到檔案為舊格式(非頂層陣列,如 { exclusions: [...] } 或 { excluded_findings: [...] }),
* 會就地把該檔案覆寫為標準頂層陣列格式(若提供 mirrorWorkspace 且路徑不同,也會同步寫入 mirror 目錄)。
* 檔案不存在或讀取/解析失敗時,皆視為空陣列,不拋出例外。
*
* @param {string} workspace - 來源分支工作目錄根路徑,EXCLUSIONS_PATH 會相對此路徑解析。
* @param {object|null} [repoState] - 可選的來源分支狀態(branch/shortSha 或 headSha/commitTime),僅用於診斷日誌。
* @param {string|null} [mirrorWorkspace] - 可選的鏡像工作目錄;當原始格式非頂層陣列時,會同步覆寫此目錄下的 exclusions.json。
* @returns {Array<object>} 正規化並去重後的排除條目陣列;讀取失敗或檔案不存在時回傳空陣列。
*/
export function loadExclusions(workspace, repoState = null, mirrorWorkspace = null) {
const fullPath = path.join(workspace, EXCLUSIONS_PATH);
if (!fs.existsSync(fullPath)) {
warn(`排除問題檔案不存在,視為空: ${fullPath}`);
if (repoState) {
const branch = repoState.branch || 'detached';
const shortSha = repoState.shortSha || repoState.headSha || 'unknown';
line(`來源分支狀態: branch=${branch} commit=${shortSha} commit_time=${repoState.commitTime || 'unknown'}`);
}
ok('讀取排除問題: raw=0 normalized=0 筆');
return [];
}
let exclusions = [];
let rawCount = 0;
try {
const stat = fs.statSync(fullPath);
const data = JSON.parse(fs.readFileSync(fullPath, 'utf8'));
const sourceFormat = detectExclusionSource(data);
const normalizedSource = normalizeExclusions(data);
rawCount = normalizedSource.length;
exclusions = dedupeExclusions(normalizedSource.map((exclusion, index) => normalizeExclusionEntry(exclusion, index)));
const branch = repoState?.branch || 'detached';
const shortSha = repoState?.shortSha || repoState?.headSha || 'unknown';
const commitTime = repoState?.commitTime || 'unknown';
line(`讀取排除問題檔案: ${fullPath}`);
line(`來源分支狀態: branch=${branch} commit=${shortSha} commit_time=${commitTime}`);
line(`檔案資訊: bytes=${stat.size} mtime=${formatFileTime(stat.mtimeMs)} raw=${rawCount} normalized=${exclusions.length} path=${path.relative(workspace, fullPath) || fullPath}`);
if (sourceFormat !== 'array') {
writeCanonicalExclusions(fullPath, normalizedSource);
if (mirrorWorkspace && path.resolve(mirrorWorkspace) !== path.resolve(workspace)) {
const mirrorPath = path.join(mirrorWorkspace, EXCLUSIONS_PATH);
fs.mkdirSync(path.dirname(mirrorPath), { recursive: true });
writeCanonicalExclusions(mirrorPath, normalizedSource);
}
line(`排除問題格式已修正為頂層陣列: source=${sourceFormat} -> array`);
}
} catch (e) {
warn(`讀取排除問題失敗: ${e.message},視為空: ${fullPath}`);
exclusions = [];
}
const summary = buildExclusionContext(exclusions);
ok(`讀取排除問題: raw=${rawCount} normalized=${exclusions.length} groups=${summary.groupCount} 筆`);
return exclusions;
}
/**
* 把新的排除條目(raw 形式,未經 normalizeExclusionEntry 加工)append 到 exclusions.json,
* 以「檔案路徑(location 冒號前段)+ normalizeText 後的原文」為簽名去重後,
* 以頂層陣列格式寫回 workspace(及提供且路徑不同的 mirrorWorkspace)。
*
* @param {string} workspace - 目標工作目錄,EXCLUSIONS_PATH 相對此路徑解析並寫入。
* @param {Array<object>} newEntries - 欲新增的排除條目(raw 形式);為空或未提供時直接回傳 null(無操作)。
* @param {string|null} [mirrorWorkspace] - 可選鏡像目錄;提供且與 workspace 路徑不同時,會同步寫入相同內容。
* @returns {Array<object>|null} 合併後的 raw 排除條目陣列;newEntries 為空時回傳 null;
* 若 newEntries 皆與既有條目重複(無實際新增)則回傳既有陣列(未寫檔)。
*/
export function appendExclusions(workspace, newEntries, mirrorWorkspace = null) {
if (!newEntries || newEntries.length === 0) return null;
const fileOf = loc => String(loc || '').split(':')[0].trim();
const sigOf = e => `${fileOf(e.location)}|${normalizeText(e.original_finding || e.suggestion || e.text || e.title || '')}`;
const fullPath = path.join(workspace, EXCLUSIONS_PATH);
let existing = [];
if (fs.existsSync(fullPath)) {
try {
existing = normalizeExclusions(JSON.parse(fs.readFileSync(fullPath, 'utf8')));
} catch (e) {
warn(`讀取排除問題以追加失敗,視為空: ${e.message}`);
existing = [];
}
}
const seen = new Set(existing.map(sigOf));
const additions = newEntries.filter(e => {
const sig = sigOf(e);
if (seen.has(sig)) return false;
seen.add(sig);
return true;
});
if (additions.length === 0) {
line(`誤報排除無新增(皆已存在): 候選 ${newEntries.length} 筆`);
return existing;
}
const merged = [...existing, ...additions];
const targets = [workspace];
if (mirrorWorkspace && path.resolve(mirrorWorkspace) !== path.resolve(workspace)) targets.push(mirrorWorkspace);
for (const dir of targets) {
const target = path.join(dir, EXCLUSIONS_PATH);
fs.mkdirSync(path.dirname(target), { recursive: true });
writeCanonicalExclusions(target, merged);
}
ok(`誤報寫入 exclusions: 新增 ${additions.length} 筆(總計 ${merged.length} 筆)`);
return merged;
}
/**
* 套用排除規則,過濾掉符合任一排除條件的 findings。
* exclusions 為空時原樣回傳 findings(新陣列,不修改原輸入)。
*
* 比對規則(對每個 exclusion,locationMatches && roleMatches && (有指定 path 或 role ? 一律視為符合 : textMatches)):
* - location 只比對檔案路徑(忽略行號),exclusion 未指定 filePath 時視為萬用;
* - role 未指定時視為萬用,否則需與 finding.role 完全相等;
* - 僅當 exclusion 同時未指定 filePath 與 role 時,才會實際比對正規化後文字(suggestion/title 等)是否互相包含。
*
* @param {Array<object>} findings - 欲過濾的 findings 陣列。
* @param {Array<object>} exclusions - 排除條目陣列(建議為已正規化含 filePath 的條目)。
* @returns {Array<object>} 過濾後的新陣列。
* @remarks 「只要 exclusion 指定了 filePath 或 role,文字比對即完全略過」是否為刻意設計,需人工確認;
* 若非刻意,可能造成排除範圍比預期寬(例如同檔案下所有問題都被排除,而非僅特定描述的問題)。
*/
export function applyExclusions(findings, exclusions) {
if (exclusions.length === 0) return findings;
const before = findings.length;
const filtered = findings.filter(f => !exclusions.some(ex => {
const fPath = String(f.location).split(':')[0];
const exPath = ex.filePath || (ex.location ? String(ex.location).split(':')[0] : null);
const findingText = normalizeText(f.suggestion || f.title || '');
const exclusionText = normalizeText(ex.text || ex.original_finding || ex.suggestion || ex.title || ex.textKey || '');
const locationMatches = (!exPath || fPath === exPath);
const roleMatches = (!ex.role || ex.role === f.role);
const textMatches = !exclusionText || !findingText || findingText.includes(exclusionText) || exclusionText.includes(findingText);
return locationMatches && roleMatches && (exPath || ex.role ? true : textMatches);
}));
ok(`排除過濾: ${before} -> ${filtered.length} 筆(排除 ${before - filtered.length} 筆)`);
return filtered;
}
/**
* 派一個「防守方」角色裁決單一 finding 是否為誤報;任何失敗都保守視為「成立」(即保留該問題)。
*
* @param {object} finding - 欲裁決的單一 finding。
* @param {object} defender - 防守方角色定義(通常為 Paladin),供 buildVerdictPrompt 組系統提示。
* @param {string} exclusionHint - 已知誤報清單的提示文字(可為空字串),供 AI 判斷是否與已知誤報類似。
* @param {Function} chatFn - 實際呼叫 LLM 的函式(簽名同 chatJSON),供測試時注入替換。
* @returns {Promise<boolean>} true 表示裁決為誤報(應剔除);false 表示成立或裁決失敗(保守保留)。
*/
async function judgeFindingIsFalsePositive(finding, defender, exclusionHint, chatFn) {
const systemPrompt = buildVerdictPrompt(defender, exclusionHint);
try {
const result = await chatFn(systemPrompt, JSON.stringify(toAIPayload([finding])[0]));
return result?.verdict === 'false_positive';
} catch (e) {
warn(`誤報裁決失敗(保守視為成立): ${finding.location} error=${e.message}`);
return false;
}
}
/**
* 由「防守方」角色(固定為 Paladin)逐條裁決 findings 是否為誤報,剔除誤報、保留成立者。
* 多個問題時各派一個裁決任務平行處理(併發上限 LLM_CONCURRENCY);任一裁決失敗保守保留該問題,不中斷流程。
*
* @param {Array<object>} findings - 欲裁決的 findings 陣列;為空陣列時直接原樣回傳。
* @param {Array<object>} [exclusions=[]] - 已知誤報排除條目,用於組裝提示,引導 AI 對相似的誤報更寬鬆判定。
* @param {Function} [chatFn=chatJSON] - 實際呼叫 LLM 的函式,供測試時注入替換。
* @returns {Promise<Array<object>>} 裁決為「非誤報」而保留下來的原始 finding 物件陣列。
*/
export async function filterFalsePositivesWithAI(findings, exclusions = [], chatFn = chatJSON) {
if (findings.length === 0) return findings;
const defender = loadRole('Paladin');
const exclusionContext = buildExclusionContext(exclusions);
const exclusionHint = exclusionContext.prompt
? `${exclusionContext.prompt}\n規則:若此 finding 與上述任何一類的路徑、角色或描述高度相似,優先視為誤報或不適用。`
: '';
// 每條 finding 各派一個防守方 sub-agent 裁決;併發上限與其他 LLM 任務共用 LLM_CONCURRENCY(預設不限制)。
const verdicts = await mapWithConcurrency(findings, LLM_CONCURRENCY, async (f) => ({
f,
isFP: await judgeFindingIsFalsePositive(f, defender, exclusionHint, chatFn),
}));
const kept = verdicts.filter(v => !v.isFP).map(v => v.f);
ok(`AI 誤報過濾(防守方${findings.length > 1 ? '平行' : ''}裁決): ${findings.length} -> ${kept.length} 筆`);
return kept;
}