Files
tea-sdlc/scripts/issue-body.js
T
jiantw83 a8b816c324 feat(議題解析): 換掉一個段落的內容,標題與其餘段落一字不動
整併留言裡的決策時用它。不整份重寫的理由跟 upsertLineInSection 一樣,只是代價更大:
重寫會把別人在其他段落的編輯一起蓋掉,而議題的編輯紀錄沒有人會去比對。

三件事照著同檔既有的規矩做:圍欄裡的假標題不算段落;同名標題出現不只一次時交回 ambiguous
而不賭第一個(蓋掉的是一整段,猜錯的代價比 tickLine 更高);換行沿用 body 原本的那一種,
CRLF 的 body 裡混進 LF 會讓抽取契約交出的 raw 對不上原文,之後就勾不動那幾行。
2026-09-17 09:08:07 +00:00

458 lines
18 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* 議題 body 的 markdown 解析。
*
* 純函式,不碰網路也不碰檔案系統:抽取類腳本的解析全部走這裡,
* 解析規則只寫一次,需求議題與工作包議題共用同一套。
*
* 全檔的共同前提是「圍欄裡的東西不是內容」:議題裡放 mermaid 或程式碼是常態,
* 那裡面的井字號不是標題、減號不是清單項、直線不是表格。這件事只在 eachLine
* 處理一次,其餘函式都靠它。
*/
/**
* 逐行走過內容並標註圍欄狀態。
* 圍欄以 ``` 或 ~~~ 開啟,且要同一種標記才算關閉——混用時後者只是普通文字。
* 圍欄沒關就到結尾時,其後的內容一律算在圍欄內,這與 markdown 的實際渲染一致。
* @param {string} text
* @returns {Generator<{line: string, inFence: boolean, isFence: boolean}>}
*/
function* eachLine(text) {
let fence = null;
for (const line of (text ?? '').split('\n')) {
const marker = line.match(/^\s*(`{3,}|~{3,})/)?.[1]?.[0];
if (marker && fence === null) {
fence = marker;
yield { line, inFence: true, isFence: true };
continue;
}
if (marker && marker === fence) {
fence = null;
yield { line, inFence: true, isFence: true };
continue;
}
yield { line, inFence: fence !== null, isFence: false };
}
}
/**
* 依 `## 標題` 切出各段落。圍欄內的行原樣保留在段落內容裡
* ——流程圖那一段要的就是整個 mermaid 區塊。
* @param {string} body 議題 body
* @returns {Map<string, string>} 段落名稱 → 該段內容(前後空白已修掉)
*/
export function parseSections(body) {
const sections = new Map();
const buffer = [];
let current = null;
const flush = () => {
if (current !== null) sections.set(current, buffer.join('\n').trim());
buffer.length = 0;
};
for (const { line, inFence } of eachLine(body)) {
const heading = inFence ? null : line.match(/^##\s+(.+?)\s*$/);
if (heading) {
flush();
current = heading[1];
continue;
}
if (current !== null) buffer.push(line);
}
flush();
return sections;
}
/**
* 取出文字型段落的整段內容。段落不存在時回空字串,不報錯 ——
* 缺段落是模板的正常變體,不是解析失敗。
* @param {Map<string, string>} sections
* @param {string} name
* @returns {string}
*/
export function textSection(sections, name) {
return sections.get(name) ?? '';
}
/**
* 取出列表型段落的每一項。符號清單與編號清單一視同仁,checkbox 只留文字不留標記
* ——需求議題的驗收標準不追蹤勾選狀態。巢狀項目一律攤平:需求議題的模板沒有巢狀,
* 真的出現時寧可多帶一項,也不要無聲吃掉內容。
* @param {Map<string, string>} sections
* @param {string} name
* @returns {string[]}
*/
export function listSection(sections, name) {
const items = [];
for (const { line, inFence } of eachLine(sections.get(name))) {
if (inFence) continue;
// [\s\S] 而非 . :CRLF 的 body 逐行切開後行尾有 \r,而 . 不吃 \r,
// 用 . 會讓整行比不中,清單靜靜變成空的。後面的 trim 再把 \r 修掉。
const item = line.match(/^\s*(?:[-*+]|\d+\.)\s+([\s\S]*)$/);
if (!item) continue;
const text = item[1].replace(/^\[[ xX]\]\s*/, '').trim();
if (text !== '') items.push(text);
}
return items;
}
/**
* 取出巢狀的待辦清單:上層是待辦,縮排一層是該項自己的驗收。
*
* 只有這一支收 body 而不收切好的段落,因為它要交出 `raw`——下游靠 `raw` 在整份 body 上
* 做精確字串替換來勾選 checkbox,那一行必須逐字等於 body 裡的原樣,
* 連縮排、行尾空白與 \r 都不能動。段落切分會修掉前後空白,拿不到這種保證。
*
* 兩種畸形寫法都不丟內容,寧可放在稍微不對的位置也不要靜靜消失:
* - 巢狀超過一層 → 攤進所在待辦的驗收
* - 還沒有上層待辦就先出現縮排項目 → 升格成待辦
*
* @param {string} body 議題 body
* @param {string} section 段落名稱,例如 '待辦'
* @returns {{text: string, done: boolean, raw: string, 驗收: {text: string, done: boolean, raw: string}[]}[]}
*/
export function checklistInSection(body, section) {
const rows = [...eachLine(body)];
const { start, end } = sectionBounds(rows, section);
if (start === -1) return [];
const todos = [];
let topIndent = null;
for (let i = start + 1; i < end; i += 1) {
if (rows[i].inFence) continue;
const item = parseChecklistItem(rows[i].line);
if (!item) continue;
const nested = topIndent !== null && todos.length > 0 && item.indent > topIndent;
if (nested) {
todos.at(-1).驗收.push(item.value);
continue;
}
// 比目前認定的上層還淺時,把上層改認成更淺的那一層:第一項剛好縮排時,
// 後面出現的真正上層才不會被當成它的驗收。
topIndent = topIndent === null ? item.indent : Math.min(topIndent, item.indent);
todos.push({ ...item.value, 驗收: [] });
}
return todos;
}
/**
* 拆一行清單項。符號清單與編號清單一視同仁,checkbox 可有可無——
* 忘了寫 checkbox 的項目仍是一項待辦,只是 done 為 false。
* @returns {{indent: number, value: {text: string, done: boolean, raw: string}}|null}
*/
function parseChecklistItem(line) {
// 文法與 tickLine 共用 LIST_ITEM:抽得出來的行,勾選端就要收得下。
// text 靠 trim 修掉 CRLF 的 \r,raw 則原樣留著——它要逐字等於 body 裡的那一行。
const item = LIST_ITEM.exec(line);
if (!item) return null;
const text = item[3].trim();
if (text === '') return null;
return {
indent: item[1].length,
value: { text, done: item[2]?.toLowerCase() === '[x]', raw: line },
};
}
/**
* 在段落裡找出「標籤:#編號」那一行的編號,例如關聯段落的 `需求議題:#7`。
* 全形與半形冒號都認;找不到回 null——沒填不是解析失敗。
* @param {Map<string, string>} sections
* @param {string} name 段落名稱
* @param {string} label 標籤,例如 '需求議題'
* @returns {number|null}
*/
export function referencedIndex(sections, name, label) {
for (const { line, inFence } of eachLine(sections.get(name))) {
if (inFence) continue;
const at = line.indexOf(label);
if (at === -1) continue;
const value = line.slice(at + label.length).match(/^\s*[::]\s*#?(\d+)/);
if (value) return Number(value[1]);
}
return null;
}
/**
* 在段落裡找出「標籤:數字」那一行的數字,例如關聯段落的 `估算人天:3`。
* 與 referencedIndex 同形狀,差別只在這裡要的是數量而非議題編號,所以認小數。
* 全形與半形冒號都認;找不到回 null——沒填不是解析失敗,呼叫端要分得開「沒估」與「估 0」。
* @param {Map<string, string>} sections
* @param {string} name 段落名稱
* @param {string} label 標籤,例如 '估算人天'
* @returns {number|null}
*/
export function labelledNumber(sections, name, label) {
for (const { line, inFence } of eachLine(sections.get(name))) {
if (inFence) continue;
const at = line.indexOf(label);
if (at === -1) continue;
const value = line.slice(at + label.length).match(/^\s*[::]\s*(\d+(?:\.\d+)?)\s*$/);
if (value) return Number(value[1]);
}
return null;
}
/**
* 取出兩欄表格型段落,欄位固定命名為 term 與 def。
* 需求議題的領域名詞表用它;四欄的介面契約請用 tableRows。
* @param {Map<string, string>} sections
* @param {string} name
* @returns {{term: string, def: string}[]}
*/
export function tableSection(sections, name) {
return tableRows(sections, name, ['term', 'def']);
}
/**
* 取出表格型段落的資料列,欄位依 columns 命名。以分隔列(|---|---|)為界,
* 之後才是資料列;沒有分隔列就當成沒有資料,避免把表頭當成一筆資料。
*
* 資料列比 columns 短時補空字串而不是讓欄位消失——下游拿到的形狀要固定,
* 少一欄是內容的問題,不該變成「欄位不存在」讓下游多寫一種分支。
*
* @param {Map<string, string>} sections
* @param {string} name
* @param {string[]} columns 由左到右的欄位名稱;多出來的欄會被丟掉
* @returns {Record<string, string>[]}
*/
export function tableRows(sections, name, columns) {
const rows = [];
for (const { line, inFence } of eachLine(sections.get(name))) {
if (inFence || !line.trim().startsWith('|')) continue;
rows.push(splitRow(line));
}
const separator = rows.findIndex(isSeparator);
if (separator === -1) return [];
return rows
.slice(separator + 1)
// 同一段落裡若不慎貼了第二張表,它的分隔列不該變成一筆資料
.filter((cells) => !isSeparator(cells))
.filter((cells) => cells.some((cell) => cell !== ''))
.map((cells) => Object.fromEntries(columns.map((column, i) => [column, cells[i] ?? ''])));
}
/** 以未被逸脫的直線切欄,再把 `\|` 還原成內容裡的直線 */
function splitRow(line) {
return line
.trim()
.replace(/^\||\|$/g, '')
.split(/(?<!\\)\|/)
.map((cell) => cell.replace(/\\\|/g, '|').trim());
}
function isSeparator(cells) {
return cells.length > 0 && cells.every((cell) => /^:?-+:?$/.test(cell));
}
/**
* 一行清單項的文法:符號或編號清單,後面可以有一個 checkbox。
*
* 全檔只有這一份定義。抽取端(parseChecklistItem)與勾選端(tickLine)若各寫一份,
* 遲早會鬆緊不一——抽得出來卻勾不動的那一行,會讓「一律用 wp-extract 給的 raw」
* 變成做不到的指示。
*/
const LIST_ITEM = /^(\s*)(?:[-*+]|\d+\.)\s+(\[[ xX]\])?\s*([\s\S]*)$/;
/**
* 這一行是不是清單項(有沒有 checkbox 都算)。
*
* `--tick` 用它驗輸入,而且刻意不要求 checkbox:抽取端會把「忘了寫 checkbox 的待辦」
* 也收成一項待辦,那種 raw 要走到 tickLine 才能得到「去議題上補成 checkbox」這句話,
* 在入口就擋掉只會回一個看不出該怎麼辦的格式錯誤。
* @param {string} line
* @returns {boolean}
*/
export function isListItem(line) {
return LIST_ITEM.test(line);
}
/**
* 勾起一行 checkbox:把 `raw` 那一行的方框換成已勾,其餘一字不動。
*
* 三件事都限定在目標段落之內、且跳過圍欄,理由與 upsertLineInSection 相同——
* 弄錯的代價是靜靜改壞別人的內容。勾選是這個檔案裡唯一會寫回議題的路徑,
* 而工作包模板的架構圖就是一塊 fenced mermaid:裡面出現減號開頭的行是常態,
* 把它當成待辦勾下去,改壞的是一張圖。
*
* 用整行精確比對而不是「找那段文字」,因為巢狀待辦底下常有一模一樣的驗收
* (兩項待辦各有一條「加上測試」)。認不出是哪一行時交回 ambiguous 讓呼叫端報錯,
* 不賭第一個——猜錯的話議題上的進度條會指著錯的那一項,而沒有人會去比對編輯紀錄。
*
* 勾選狀態與大小寫都不影響比對:`[ ]`、`[x]`、`[X]` 指的是同一行,
* 已經勾過就交回 already,讓中斷後重跑是安靜的 no-op 而不是失敗。
*
* 本函式不拋錯——它是純解析,錯誤碼由呼叫端決定。
*
* @param {string} body 議題 body
* @param {string} raw 抽取契約交出的原始 markdown 行,逐字包含縮排與行尾的 \r
* @param {string} [section] 限定在這個段落內找;省略時找全文(圍欄照樣不算)
* @returns {{status: 'ticked'|'already'|'not-found'|'ambiguous'|'no-checkbox'|'no-section', body?: string, line?: string, count: number}}
*/
export function tickLine(body, raw, section) {
const item = LIST_ITEM.exec(raw);
if (!item || item[2] === undefined) return { status: 'no-checkbox', count: 0 };
const rows = [...eachLine(body)];
const { start, end } = section === undefined
? { start: -1, end: rows.length }
: sectionBounds(rows, section);
if (section !== undefined && start === -1) return { status: 'no-section', count: 0 };
/** 同一行的三種寫法都指向它自己:比對時一律正規化成未勾的小寫版本 */
const normalize = (line) => line.replace(/\[[ xX]\]/, '[ ]');
const wanted = normalize(raw);
const ticked = raw.replace(/\[[ xX]\]/, '[x]');
const hits = [];
for (let i = start + 1; i < end; i += 1) {
if (rows[i].inFence) continue;
if (normalize(rows[i].line) === wanted) hits.push(i);
}
if (hits.length === 0) return { status: 'not-found', count: 0 };
if (hits.length > 1) return { status: 'ambiguous', count: hits.length };
const [at] = hits;
// 已勾與否看方框本身,不比整行字串:`[X]` 是合法的 GFM,Gitea 也渲染成已勾,
// 用字串相等判斷會把它當成還沒勾,於是重跑時硬把大寫改成小寫
if (LIST_ITEM.exec(rows[at].line)[2].toLowerCase() === '[x]') {
return { status: 'already', line: rows[at].line, count: 1 };
}
const lines = rows.map((row) => row.line);
lines[at] = ticked;
return { status: 'ticked', body: lines.join('\n'), line: ticked, count: 1 };
}
/**
* 換掉一個段落的內容,標題與其餘段落一字不動。
*
* 整併留言裡的決策時用它。不整份重寫的理由跟 upsertLineInSection 一樣,只是代價更大:
* 重寫會把別人在其他段落的編輯一起蓋掉,而議題的編輯紀錄沒有人會去比對。
*
* 同名標題出現不只一次時交回 `ambiguous`,不賭第一個——理由與 tickLine 相同,
* 而這裡蓋掉的是一整段而不是一行,猜錯的代價更高。
*
* 段落不存在時交回 `not-found` 讓呼叫端報錯,不補在結尾:「找不到那一段」多半是段落名
* 打錯,這時把內容塞到議題末尾,比什麼都不做更難收拾。
*
* @param {string} body 議題 body
* @param {string} section 段落名稱,例如 '目標'
* @param {string} content 新的段落內容(不含 `## 標題` 那一行)
* @returns {{status: 'replaced'|'not-found'|'ambiguous', body?: string, count: number}}
*/
export function replaceSection(body, section, content) {
const rows = [...eachLine(body)];
const headings = [];
for (let i = 0; i < rows.length; i += 1) {
if (rows[i].inFence) continue;
if (rows[i].line.match(/^##\s+(.+?)\s*$/)?.[1] === section) headings.push(i);
}
if (headings.length === 0) return { status: 'not-found', count: 0 };
if (headings.length > 1) return { status: 'ambiguous', count: headings.length };
const [start] = headings;
const lines = rows.map((row) => row.line);
// 下一個段落的標題;沒有就是到結尾
let end = lines.length;
for (let i = start + 1; i < lines.length; i += 1) {
if (!rows[i].inFence && /^##\s+/.test(lines[i])) {
end = i;
break;
}
}
// 段落與段落之間的空行屬於版面,不屬於內容:換內容時把它留著。
// 原本就沒有空行(兩個標題緊貼)時補一個,免得新內容黏在下一個標題上。
let tail = end;
while (tail > start + 1 && lines[tail - 1].trim() === '') tail -= 1;
const spacer = end === lines.length || end > tail ? lines.slice(tail, end) : [''];
// 換行沿用 body 原本的那一種:CRLF 的 body 裡混進 LF,會讓抽取契約交出的 raw
// 對不上原文,之後就勾不動那幾行了
const eol = body.includes('\r\n') ? '\r\n' : '\n';
const normalized = content.trim().split(/\r?\n/);
return {
status: 'replaced',
body: [...lines.slice(0, start + 1), '', ...normalized, ...spacer, ...lines.slice(end)]
.map((line) => line.replace(/\r$/, ''))
.join(eol),
count: 1,
};
}
/**
* 在指定段落裡就地更新(或補上)一行「前綴+值」。
*
* 用於人天估算這種「議題上只該有一行、重跑要覆蓋而不是累積」的欄位。
*
* 三件事都限定在目標段落之內,因為弄錯的代價是靜靜改壞別人的內容:
* - 標題要完全相同才算數,`## 關聯度說明` 不是 `## 關聯`。
* - 圍欄裡的假標題不算標題,沿用本檔共同的圍欄判斷。
* - 找既有那一行只在段落範圍內找,別的段落剛好有同前綴時不會被改掉。
*
* 段落不存在時補在 body 結尾——寧可放錯位置,也不要讓值靜靜消失。
*
* @param {string} body 議題 body
* @param {string} section 段落名稱,例如 '關聯'
* @param {string} line 完整的一行,例如 '估算人天:3'
* @returns {string} 更新後的 body;內容沒有變動時回傳原字串
*/
export function upsertLineInSection(body, section, line) {
const prefix = line.slice(0, line.indexOf(':') + 1);
const rows = [...eachLine(body)];
const { start, end } = sectionBounds(rows, section);
if (start === -1) {
return `${body.replace(/\n*$/, '')}\n\n${line}\n`;
}
const text = rows.map((row) => row.line);
for (let i = start + 1; i < end; i += 1) {
if (rows[i].inFence || !text[i].startsWith(prefix)) continue;
if (text[i] === line) return body;
text[i] = line;
return text.join('\n');
}
// 插在段落內容的結尾,跳過段落與段落之間的空行
let insertAt = end;
while (insertAt > start + 1 && text[insertAt - 1].trim() === '') insertAt -= 1;
text.splice(insertAt, 0, line);
return text.join('\n');
}
/** 找出段落的起訖行號;start 為標題那一行,end 為下一個標題(或結尾) */
function sectionBounds(rows, section) {
let start = -1;
for (let i = 0; i < rows.length; i += 1) {
if (rows[i].inFence) continue;
const heading = rows[i].line.match(/^##\s+(.+?)\s*$/);
if (!heading) continue;
if (start === -1) {
if (heading[1] === section) start = i;
continue;
}
return { start, end: i };
}
return { start, end: rows.length };
}