Files
tea-sdlc/scripts/issue-body.js
T
jiantw83andClaude Opus 5 9582553c41 feat(議題解析): 勾選 checkbox 的精確替換,並統一清單項的文法
tickLine 把指定那一行的方框換成已勾,其餘一字不動。四件事決定它會不會靜靜改壞議題:

- **跳過圍欄。** 這是 issue-body.js 全檔的前提,而勾選是本檔唯一會寫回議題的路徑。
  工作包模板的架構圖就是一塊 fenced mermaid,裡面出現減號開頭的行是常態,
  把它當成待辦勾下去,改壞的是一張圖。
- **限定段落。** 待辦與整體驗收常有一模一樣的一句話,不限定就會回報「分不出來」,
  而使用者其實講得很清楚。理由與 upsertLineInSection 相同:弄錯的代價是靜靜改壞內容。
- **整行比對,認不出就交回 ambiguous。** 巢狀待辦底下常有一樣的驗收,賭第一個會讓
  進度條指著錯的那一項,而沒有人會去比對編輯紀錄。
- **[ ]、[x]、[X] 指的是同一行。** [X] 是合法的 GFM,Gitea 也渲染成已勾;只認小寫的話,
  中斷後重跑會硬失敗,錯誤訊息還會誣指「議題被改過」。

清單項的文法收斂成一份 LIST_ITEM,parseChecklistItem 與 tickLine 共用。先前兩端各寫一份,
鬆緊不一致:`- [ ]甲` 抽得出來卻勾不動,正本那句「一律用 wp-extract 給的 raw」就成了
做不到的指示。沒有方框的項目交回 no-checkbox,不再謊報「已經勾過」。

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-17 07:39:43 +00:00

399 lines
15 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* 議題 body 的 markdown 解析。
*
* 純函式,不碰網路也不碰檔案系統:抽取類腳本的解析全部走這裡,
* 解析規則只寫一次,需求議題與工作包議題共用同一套。
*
* 全檔的共同前提是「圍欄裡的東西不是內容」:議題裡放 mermaid 或程式碼是常態,
* 那裡面的井字號不是標題、減號不是清單項、直線不是表格。這件事只在 eachLine
* 處理一次,其餘函式都靠它。
*/
/**
* 逐行走過內容並標註圍欄狀態。
* 圍欄以 ``` 或 ~~~ 開啟,且要同一種標記才算關閉——混用時後者只是普通文字。
* 圍欄沒關就到結尾時,其後的內容一律算在圍欄內,這與 markdown 的實際渲染一致。
* @param {string} text
* @returns {Generator<{line: string, inFence: boolean, isFence: boolean}>}
*/
function* eachLine(text) {
let fence = null;
for (const line of (text ?? '').split('\n')) {
const marker = line.match(/^\s*(`{3,}|~{3,})/)?.[1]?.[0];
if (marker && fence === null) {
fence = marker;
yield { line, inFence: true, isFence: true };
continue;
}
if (marker && marker === fence) {
fence = null;
yield { line, inFence: true, isFence: true };
continue;
}
yield { line, inFence: fence !== null, isFence: false };
}
}
/**
* 依 `## 標題` 切出各段落。圍欄內的行原樣保留在段落內容裡
* ——流程圖那一段要的就是整個 mermaid 區塊。
* @param {string} body 議題 body
* @returns {Map<string, string>} 段落名稱 → 該段內容(前後空白已修掉)
*/
export function parseSections(body) {
const sections = new Map();
const buffer = [];
let current = null;
const flush = () => {
if (current !== null) sections.set(current, buffer.join('\n').trim());
buffer.length = 0;
};
for (const { line, inFence } of eachLine(body)) {
const heading = inFence ? null : line.match(/^##\s+(.+?)\s*$/);
if (heading) {
flush();
current = heading[1];
continue;
}
if (current !== null) buffer.push(line);
}
flush();
return sections;
}
/**
* 取出文字型段落的整段內容。段落不存在時回空字串,不報錯 ——
* 缺段落是模板的正常變體,不是解析失敗。
* @param {Map<string, string>} sections
* @param {string} name
* @returns {string}
*/
export function textSection(sections, name) {
return sections.get(name) ?? '';
}
/**
* 取出列表型段落的每一項。符號清單與編號清單一視同仁,checkbox 只留文字不留標記
* ——需求議題的驗收標準不追蹤勾選狀態。巢狀項目一律攤平:需求議題的模板沒有巢狀,
* 真的出現時寧可多帶一項,也不要無聲吃掉內容。
* @param {Map<string, string>} sections
* @param {string} name
* @returns {string[]}
*/
export function listSection(sections, name) {
const items = [];
for (const { line, inFence } of eachLine(sections.get(name))) {
if (inFence) continue;
// [\s\S] 而非 . :CRLF 的 body 逐行切開後行尾有 \r,而 . 不吃 \r,
// 用 . 會讓整行比不中,清單靜靜變成空的。後面的 trim 再把 \r 修掉。
const item = line.match(/^\s*(?:[-*+]|\d+\.)\s+([\s\S]*)$/);
if (!item) continue;
const text = item[1].replace(/^\[[ xX]\]\s*/, '').trim();
if (text !== '') items.push(text);
}
return items;
}
/**
* 取出巢狀的待辦清單:上層是待辦,縮排一層是該項自己的驗收。
*
* 只有這一支收 body 而不收切好的段落,因為它要交出 `raw`——下游靠 `raw` 在整份 body 上
* 做精確字串替換來勾選 checkbox,那一行必須逐字等於 body 裡的原樣,
* 連縮排、行尾空白與 \r 都不能動。段落切分會修掉前後空白,拿不到這種保證。
*
* 兩種畸形寫法都不丟內容,寧可放在稍微不對的位置也不要靜靜消失:
* - 巢狀超過一層 → 攤進所在待辦的驗收
* - 還沒有上層待辦就先出現縮排項目 → 升格成待辦
*
* @param {string} body 議題 body
* @param {string} section 段落名稱,例如 '待辦'
* @returns {{text: string, done: boolean, raw: string, 驗收: {text: string, done: boolean, raw: string}[]}[]}
*/
export function checklistInSection(body, section) {
const rows = [...eachLine(body)];
const { start, end } = sectionBounds(rows, section);
if (start === -1) return [];
const todos = [];
let topIndent = null;
for (let i = start + 1; i < end; i += 1) {
if (rows[i].inFence) continue;
const item = parseChecklistItem(rows[i].line);
if (!item) continue;
const nested = topIndent !== null && todos.length > 0 && item.indent > topIndent;
if (nested) {
todos.at(-1).驗收.push(item.value);
continue;
}
// 比目前認定的上層還淺時,把上層改認成更淺的那一層:第一項剛好縮排時,
// 後面出現的真正上層才不會被當成它的驗收。
topIndent = topIndent === null ? item.indent : Math.min(topIndent, item.indent);
todos.push({ ...item.value, 驗收: [] });
}
return todos;
}
/**
* 拆一行清單項。符號清單與編號清單一視同仁,checkbox 可有可無——
* 忘了寫 checkbox 的項目仍是一項待辦,只是 done 為 false。
* @returns {{indent: number, value: {text: string, done: boolean, raw: string}}|null}
*/
function parseChecklistItem(line) {
// 文法與 tickLine 共用 LIST_ITEM:抽得出來的行,勾選端就要收得下。
// text 靠 trim 修掉 CRLF 的 \r,raw 則原樣留著——它要逐字等於 body 裡的那一行。
const item = LIST_ITEM.exec(line);
if (!item) return null;
const text = item[3].trim();
if (text === '') return null;
return {
indent: item[1].length,
value: { text, done: item[2]?.toLowerCase() === '[x]', raw: line },
};
}
/**
* 在段落裡找出「標籤:#編號」那一行的編號,例如關聯段落的 `需求議題:#7`。
* 全形與半形冒號都認;找不到回 null——沒填不是解析失敗。
* @param {Map<string, string>} sections
* @param {string} name 段落名稱
* @param {string} label 標籤,例如 '需求議題'
* @returns {number|null}
*/
export function referencedIndex(sections, name, label) {
for (const { line, inFence } of eachLine(sections.get(name))) {
if (inFence) continue;
const at = line.indexOf(label);
if (at === -1) continue;
const value = line.slice(at + label.length).match(/^\s*[::]\s*#?(\d+)/);
if (value) return Number(value[1]);
}
return null;
}
/**
* 在段落裡找出「標籤:數字」那一行的數字,例如關聯段落的 `估算人天:3`。
* 與 referencedIndex 同形狀,差別只在這裡要的是數量而非議題編號,所以認小數。
* 全形與半形冒號都認;找不到回 null——沒填不是解析失敗,呼叫端要分得開「沒估」與「估 0」。
* @param {Map<string, string>} sections
* @param {string} name 段落名稱
* @param {string} label 標籤,例如 '估算人天'
* @returns {number|null}
*/
export function labelledNumber(sections, name, label) {
for (const { line, inFence } of eachLine(sections.get(name))) {
if (inFence) continue;
const at = line.indexOf(label);
if (at === -1) continue;
const value = line.slice(at + label.length).match(/^\s*[::]\s*(\d+(?:\.\d+)?)\s*$/);
if (value) return Number(value[1]);
}
return null;
}
/**
* 取出兩欄表格型段落,欄位固定命名為 term 與 def。
* 需求議題的領域名詞表用它;四欄的介面契約請用 tableRows。
* @param {Map<string, string>} sections
* @param {string} name
* @returns {{term: string, def: string}[]}
*/
export function tableSection(sections, name) {
return tableRows(sections, name, ['term', 'def']);
}
/**
* 取出表格型段落的資料列,欄位依 columns 命名。以分隔列(|---|---|)為界,
* 之後才是資料列;沒有分隔列就當成沒有資料,避免把表頭當成一筆資料。
*
* 資料列比 columns 短時補空字串而不是讓欄位消失——下游拿到的形狀要固定,
* 少一欄是內容的問題,不該變成「欄位不存在」讓下游多寫一種分支。
*
* @param {Map<string, string>} sections
* @param {string} name
* @param {string[]} columns 由左到右的欄位名稱;多出來的欄會被丟掉
* @returns {Record<string, string>[]}
*/
export function tableRows(sections, name, columns) {
const rows = [];
for (const { line, inFence } of eachLine(sections.get(name))) {
if (inFence || !line.trim().startsWith('|')) continue;
rows.push(splitRow(line));
}
const separator = rows.findIndex(isSeparator);
if (separator === -1) return [];
return rows
.slice(separator + 1)
// 同一段落裡若不慎貼了第二張表,它的分隔列不該變成一筆資料
.filter((cells) => !isSeparator(cells))
.filter((cells) => cells.some((cell) => cell !== ''))
.map((cells) => Object.fromEntries(columns.map((column, i) => [column, cells[i] ?? ''])));
}
/** 以未被逸脫的直線切欄,再把 `\|` 還原成內容裡的直線 */
function splitRow(line) {
return line
.trim()
.replace(/^\||\|$/g, '')
.split(/(?<!\\)\|/)
.map((cell) => cell.replace(/\\\|/g, '|').trim());
}
function isSeparator(cells) {
return cells.length > 0 && cells.every((cell) => /^:?-+:?$/.test(cell));
}
/**
* 一行清單項的文法:符號或編號清單,後面可以有一個 checkbox。
*
* 全檔只有這一份定義。抽取端(parseChecklistItem)與勾選端(tickLine)若各寫一份,
* 遲早會鬆緊不一——抽得出來卻勾不動的那一行,會讓「一律用 wp-extract 給的 raw」
* 變成做不到的指示。
*/
const LIST_ITEM = /^(\s*)(?:[-*+]|\d+\.)\s+(\[[ xX]\])?\s*([\s\S]*)$/;
/**
* 這一行是不是清單項(有沒有 checkbox 都算)。
*
* `--tick` 用它驗輸入,而且刻意不要求 checkbox:抽取端會把「忘了寫 checkbox 的待辦」
* 也收成一項待辦,那種 raw 要走到 tickLine 才能得到「去議題上補成 checkbox」這句話,
* 在入口就擋掉只會回一個看不出該怎麼辦的格式錯誤。
* @param {string} line
* @returns {boolean}
*/
export function isListItem(line) {
return LIST_ITEM.test(line);
}
/**
* 勾起一行 checkbox:把 `raw` 那一行的方框換成已勾,其餘一字不動。
*
* 三件事都限定在目標段落之內、且跳過圍欄,理由與 upsertLineInSection 相同——
* 弄錯的代價是靜靜改壞別人的內容。勾選是這個檔案裡唯一會寫回議題的路徑,
* 而工作包模板的架構圖就是一塊 fenced mermaid:裡面出現減號開頭的行是常態,
* 把它當成待辦勾下去,改壞的是一張圖。
*
* 用整行精確比對而不是「找那段文字」,因為巢狀待辦底下常有一模一樣的驗收
* (兩項待辦各有一條「加上測試」)。認不出是哪一行時交回 ambiguous 讓呼叫端報錯,
* 不賭第一個——猜錯的話議題上的進度條會指著錯的那一項,而沒有人會去比對編輯紀錄。
*
* 勾選狀態與大小寫都不影響比對:`[ ]`、`[x]`、`[X]` 指的是同一行,
* 已經勾過就交回 already,讓中斷後重跑是安靜的 no-op 而不是失敗。
*
* 本函式不拋錯——它是純解析,錯誤碼由呼叫端決定。
*
* @param {string} body 議題 body
* @param {string} raw 抽取契約交出的原始 markdown 行,逐字包含縮排與行尾的 \r
* @param {string} [section] 限定在這個段落內找;省略時找全文(圍欄照樣不算)
* @returns {{status: 'ticked'|'already'|'not-found'|'ambiguous'|'no-checkbox'|'no-section', body?: string, line?: string, count: number}}
*/
export function tickLine(body, raw, section) {
const item = LIST_ITEM.exec(raw);
if (!item || item[2] === undefined) return { status: 'no-checkbox', count: 0 };
const rows = [...eachLine(body)];
const { start, end } = section === undefined
? { start: -1, end: rows.length }
: sectionBounds(rows, section);
if (section !== undefined && start === -1) return { status: 'no-section', count: 0 };
/** 同一行的三種寫法都指向它自己:比對時一律正規化成未勾的小寫版本 */
const normalize = (line) => line.replace(/\[[ xX]\]/, '[ ]');
const wanted = normalize(raw);
const ticked = raw.replace(/\[[ xX]\]/, '[x]');
const hits = [];
for (let i = start + 1; i < end; i += 1) {
if (rows[i].inFence) continue;
if (normalize(rows[i].line) === wanted) hits.push(i);
}
if (hits.length === 0) return { status: 'not-found', count: 0 };
if (hits.length > 1) return { status: 'ambiguous', count: hits.length };
const [at] = hits;
// 已勾與否看方框本身,不比整行字串:`[X]` 是合法的 GFM,Gitea 也渲染成已勾,
// 用字串相等判斷會把它當成還沒勾,於是重跑時硬把大寫改成小寫
if (LIST_ITEM.exec(rows[at].line)[2].toLowerCase() === '[x]') {
return { status: 'already', line: rows[at].line, count: 1 };
}
const lines = rows.map((row) => row.line);
lines[at] = ticked;
return { status: 'ticked', body: lines.join('\n'), line: ticked, count: 1 };
}
/**
* 在指定段落裡就地更新(或補上)一行「前綴+值」。
*
* 用於人天估算這種「議題上只該有一行、重跑要覆蓋而不是累積」的欄位。
*
* 三件事都限定在目標段落之內,因為弄錯的代價是靜靜改壞別人的內容:
* - 標題要完全相同才算數,`## 關聯度說明` 不是 `## 關聯`。
* - 圍欄裡的假標題不算標題,沿用本檔共同的圍欄判斷。
* - 找既有那一行只在段落範圍內找,別的段落剛好有同前綴時不會被改掉。
*
* 段落不存在時補在 body 結尾——寧可放錯位置,也不要讓值靜靜消失。
*
* @param {string} body 議題 body
* @param {string} section 段落名稱,例如 '關聯'
* @param {string} line 完整的一行,例如 '估算人天:3'
* @returns {string} 更新後的 body;內容沒有變動時回傳原字串
*/
export function upsertLineInSection(body, section, line) {
const prefix = line.slice(0, line.indexOf(':') + 1);
const rows = [...eachLine(body)];
const { start, end } = sectionBounds(rows, section);
if (start === -1) {
return `${body.replace(/\n*$/, '')}\n\n${line}\n`;
}
const text = rows.map((row) => row.line);
for (let i = start + 1; i < end; i += 1) {
if (rows[i].inFence || !text[i].startsWith(prefix)) continue;
if (text[i] === line) return body;
text[i] = line;
return text.join('\n');
}
// 插在段落內容的結尾,跳過段落與段落之間的空行
let insertAt = end;
while (insertAt > start + 1 && text[insertAt - 1].trim() === '') insertAt -= 1;
text.splice(insertAt, 0, line);
return text.join('\n');
}
/** 找出段落的起訖行號;start 為標題那一行,end 為下一個標題(或結尾) */
function sectionBounds(rows, section) {
let start = -1;
for (let i = 0; i < rows.length; i += 1) {
if (rows[i].inFence) continue;
const heading = rows[i].line.match(/^##\s+(.+?)\s*$/);
if (!heading) continue;
if (start === -1) {
if (heading[1] === section) start = i;
continue;
}
return { start, end: i };
}
return { start, end: rows.length };
}