Files
tea-sdlc/scripts/issue-body.js
T
jiantw83andClaude Opus 5 289c3e6b6a feat(議題解析): 解析巢狀待辦、多欄表格與關聯裡的議題編號
工作包議題比需求議題多三種結構,抽取契約(#9)要靠它們:

- checklistInSection 收 body 而不收切好的段落。它要交出的 `raw` 是下游勾選 checkbox
  時做精確字串替換的依據,那一行必須逐字等於 body 裡的原樣,連縮排與行尾的 \r 都不能
  動;段落切分會修掉前後空白,給不出這種保證。兩種畸形寫法都不丟內容:巢狀超過一層攤
  進所在待辦的驗收,還沒有上層待辦就先出現的縮排項目升格成待辦。
- tableRows 保留全部欄位,tableSection 改寫成它的兩欄版。介面契約是四欄,先前那一支
  只留兩欄。短的資料列補空字串——正本明講「不產出對外介面就寫一列『無』」,那一列不該
  與「沒有這一段」混為一談。
- referencedIndex 從關聯段落讀出 `需求議題:#N`。

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-17 14:25:02 +08:00

299 lines
11 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* 議題 body 的 markdown 解析。
*
* 純函式,不碰網路也不碰檔案系統:抽取類腳本的解析全部走這裡,
* 解析規則只寫一次,需求議題與工作包議題共用同一套。
*
* 全檔的共同前提是「圍欄裡的東西不是內容」:議題裡放 mermaid 或程式碼是常態,
* 那裡面的井字號不是標題、減號不是清單項、直線不是表格。這件事只在 eachLine
* 處理一次,其餘函式都靠它。
*/
/**
* 逐行走過內容並標註圍欄狀態。
* 圍欄以 ``` 或 ~~~ 開啟,且要同一種標記才算關閉——混用時後者只是普通文字。
* 圍欄沒關就到結尾時,其後的內容一律算在圍欄內,這與 markdown 的實際渲染一致。
* @param {string} text
* @returns {Generator<{line: string, inFence: boolean, isFence: boolean}>}
*/
function* eachLine(text) {
let fence = null;
for (const line of (text ?? '').split('\n')) {
const marker = line.match(/^\s*(`{3,}|~{3,})/)?.[1]?.[0];
if (marker && fence === null) {
fence = marker;
yield { line, inFence: true, isFence: true };
continue;
}
if (marker && marker === fence) {
fence = null;
yield { line, inFence: true, isFence: true };
continue;
}
yield { line, inFence: fence !== null, isFence: false };
}
}
/**
* 依 `## 標題` 切出各段落。圍欄內的行原樣保留在段落內容裡
* ——流程圖那一段要的就是整個 mermaid 區塊。
* @param {string} body 議題 body
* @returns {Map<string, string>} 段落名稱 → 該段內容(前後空白已修掉)
*/
export function parseSections(body) {
const sections = new Map();
const buffer = [];
let current = null;
const flush = () => {
if (current !== null) sections.set(current, buffer.join('\n').trim());
buffer.length = 0;
};
for (const { line, inFence } of eachLine(body)) {
const heading = inFence ? null : line.match(/^##\s+(.+?)\s*$/);
if (heading) {
flush();
current = heading[1];
continue;
}
if (current !== null) buffer.push(line);
}
flush();
return sections;
}
/**
* 取出文字型段落的整段內容。段落不存在時回空字串,不報錯 ——
* 缺段落是模板的正常變體,不是解析失敗。
* @param {Map<string, string>} sections
* @param {string} name
* @returns {string}
*/
export function textSection(sections, name) {
return sections.get(name) ?? '';
}
/**
* 取出列表型段落的每一項。符號清單與編號清單一視同仁,checkbox 只留文字不留標記
* ——需求議題的驗收標準不追蹤勾選狀態。巢狀項目一律攤平:需求議題的模板沒有巢狀,
* 真的出現時寧可多帶一項,也不要無聲吃掉內容。
* @param {Map<string, string>} sections
* @param {string} name
* @returns {string[]}
*/
export function listSection(sections, name) {
const items = [];
for (const { line, inFence } of eachLine(sections.get(name))) {
if (inFence) continue;
// [\s\S] 而非 . :CRLF 的 body 逐行切開後行尾有 \r,而 . 不吃 \r,
// 用 . 會讓整行比不中,清單靜靜變成空的。後面的 trim 再把 \r 修掉。
const item = line.match(/^\s*(?:[-*+]|\d+\.)\s+([\s\S]*)$/);
if (!item) continue;
const text = item[1].replace(/^\[[ xX]\]\s*/, '').trim();
if (text !== '') items.push(text);
}
return items;
}
/**
* 取出巢狀的待辦清單:上層是待辦,縮排一層是該項自己的驗收。
*
* 只有這一支收 body 而不收切好的段落,因為它要交出 `raw`——下游靠 `raw` 在整份 body 上
* 做精確字串替換來勾選 checkbox,那一行必須逐字等於 body 裡的原樣,
* 連縮排、行尾空白與 \r 都不能動。段落切分會修掉前後空白,拿不到這種保證。
*
* 兩種畸形寫法都不丟內容,寧可放在稍微不對的位置也不要靜靜消失:
* - 巢狀超過一層 → 攤進所在待辦的驗收
* - 還沒有上層待辦就先出現縮排項目 → 升格成待辦
*
* @param {string} body 議題 body
* @param {string} section 段落名稱,例如 '待辦'
* @returns {{text: string, done: boolean, raw: string, 驗收: {text: string, done: boolean, raw: string}[]}[]}
*/
export function checklistInSection(body, section) {
const rows = [...eachLine(body)];
const { start, end } = sectionBounds(rows, section);
if (start === -1) return [];
const todos = [];
let topIndent = null;
for (let i = start + 1; i < end; i += 1) {
if (rows[i].inFence) continue;
const item = parseChecklistItem(rows[i].line);
if (!item) continue;
const nested = topIndent !== null && todos.length > 0 && item.indent > topIndent;
if (nested) {
todos.at(-1).驗收.push(item.value);
continue;
}
// 比目前認定的上層還淺時,把上層改認成更淺的那一層:第一項剛好縮排時,
// 後面出現的真正上層才不會被當成它的驗收。
topIndent = topIndent === null ? item.indent : Math.min(topIndent, item.indent);
todos.push({ ...item.value, 驗收: [] });
}
return todos;
}
/**
* 拆一行清單項。符號清單與編號清單一視同仁,checkbox 可有可無——
* 忘了寫 checkbox 的項目仍是一項待辦,只是 done 為 false。
* @returns {{indent: number, value: {text: string, done: boolean, raw: string}}|null}
*/
function parseChecklistItem(line) {
// [\s\S] 而非 . 的理由同 listSection:CRLF 的 body 行尾有 \r,. 不吃它。
// text 靠 trim 修掉 \r,raw 則原樣留著——它要逐字等於 body 裡的那一行。
const item = line.match(/^(\s*)(?:[-*+]|\d+\.)\s+([\s\S]*)$/);
if (!item) return null;
const box = item[2].match(/^\[([ xX])\]\s*([\s\S]*)$/);
const text = (box ? box[2] : item[2]).trim();
if (text === '') return null;
return {
indent: item[1].length,
value: { text, done: box ? box[1].toLowerCase() === 'x' : false, raw: line },
};
}
/**
* 在段落裡找出「標籤:#編號」那一行的編號,例如關聯段落的 `需求議題:#7`。
* 全形與半形冒號都認;找不到回 null——沒填不是解析失敗。
* @param {Map<string, string>} sections
* @param {string} name 段落名稱
* @param {string} label 標籤,例如 '需求議題'
* @returns {number|null}
*/
export function referencedIndex(sections, name, label) {
for (const { line, inFence } of eachLine(sections.get(name))) {
if (inFence) continue;
const at = line.indexOf(label);
if (at === -1) continue;
const value = line.slice(at + label.length).match(/^\s*[::]\s*#?(\d+)/);
if (value) return Number(value[1]);
}
return null;
}
/**
* 取出兩欄表格型段落,欄位固定命名為 term 與 def。
* 需求議題的領域名詞表用它;四欄的介面契約請用 tableRows。
* @param {Map<string, string>} sections
* @param {string} name
* @returns {{term: string, def: string}[]}
*/
export function tableSection(sections, name) {
return tableRows(sections, name, ['term', 'def']);
}
/**
* 取出表格型段落的資料列,欄位依 columns 命名。以分隔列(|---|---|)為界,
* 之後才是資料列;沒有分隔列就當成沒有資料,避免把表頭當成一筆資料。
*
* 資料列比 columns 短時補空字串而不是讓欄位消失——下游拿到的形狀要固定,
* 少一欄是內容的問題,不該變成「欄位不存在」讓下游多寫一種分支。
*
* @param {Map<string, string>} sections
* @param {string} name
* @param {string[]} columns 由左到右的欄位名稱;多出來的欄會被丟掉
* @returns {Record<string, string>[]}
*/
export function tableRows(sections, name, columns) {
const rows = [];
for (const { line, inFence } of eachLine(sections.get(name))) {
if (inFence || !line.trim().startsWith('|')) continue;
rows.push(splitRow(line));
}
const separator = rows.findIndex(isSeparator);
if (separator === -1) return [];
return rows
.slice(separator + 1)
// 同一段落裡若不慎貼了第二張表,它的分隔列不該變成一筆資料
.filter((cells) => !isSeparator(cells))
.filter((cells) => cells.some((cell) => cell !== ''))
.map((cells) => Object.fromEntries(columns.map((column, i) => [column, cells[i] ?? ''])));
}
/** 以未被逸脫的直線切欄,再把 `\|` 還原成內容裡的直線 */
function splitRow(line) {
return line
.trim()
.replace(/^\||\|$/g, '')
.split(/(?<!\\)\|/)
.map((cell) => cell.replace(/\\\|/g, '|').trim());
}
function isSeparator(cells) {
return cells.length > 0 && cells.every((cell) => /^:?-+:?$/.test(cell));
}
/**
* 在指定段落裡就地更新(或補上)一行「前綴+值」。
*
* 用於人天估算這種「議題上只該有一行、重跑要覆蓋而不是累積」的欄位。
*
* 三件事都限定在目標段落之內,因為弄錯的代價是靜靜改壞別人的內容:
* - 標題要完全相同才算數,`## 關聯度說明` 不是 `## 關聯`。
* - 圍欄裡的假標題不算標題,沿用本檔共同的圍欄判斷。
* - 找既有那一行只在段落範圍內找,別的段落剛好有同前綴時不會被改掉。
*
* 段落不存在時補在 body 結尾——寧可放錯位置,也不要讓值靜靜消失。
*
* @param {string} body 議題 body
* @param {string} section 段落名稱,例如 '關聯'
* @param {string} line 完整的一行,例如 '估算人天:3'
* @returns {string} 更新後的 body;內容沒有變動時回傳原字串
*/
export function upsertLineInSection(body, section, line) {
const prefix = line.slice(0, line.indexOf(':') + 1);
const rows = [...eachLine(body)];
const { start, end } = sectionBounds(rows, section);
if (start === -1) {
return `${body.replace(/\n*$/, '')}\n\n${line}\n`;
}
const text = rows.map((row) => row.line);
for (let i = start + 1; i < end; i += 1) {
if (rows[i].inFence || !text[i].startsWith(prefix)) continue;
if (text[i] === line) return body;
text[i] = line;
return text.join('\n');
}
// 插在段落內容的結尾,跳過段落與段落之間的空行
let insertAt = end;
while (insertAt > start + 1 && text[insertAt - 1].trim() === '') insertAt -= 1;
text.splice(insertAt, 0, line);
return text.join('\n');
}
/** 找出段落的起訖行號;start 為標題那一行,end 為下一個標題(或結尾) */
function sectionBounds(rows, section) {
let start = -1;
for (let i = 0; i < rows.length; i += 1) {
if (rows[i].inFence) continue;
const heading = rows[i].line.match(/^##\s+(.+?)\s*$/);
if (!heading) continue;
if (start === -1) {
if (heading[1] === section) start = i;
continue;
}
return { start, end: i };
}
return { start, end: rows.length };
}