工作包議題比需求議題多三種結構,抽取契約(#9)要靠它們: - checklistInSection 收 body 而不收切好的段落。它要交出的 `raw` 是下游勾選 checkbox 時做精確字串替換的依據,那一行必須逐字等於 body 裡的原樣,連縮排與行尾的 \r 都不能 動;段落切分會修掉前後空白,給不出這種保證。兩種畸形寫法都不丟內容:巢狀超過一層攤 進所在待辦的驗收,還沒有上層待辦就先出現的縮排項目升格成待辦。 - tableRows 保留全部欄位,tableSection 改寫成它的兩欄版。介面契約是四欄,先前那一支 只留兩欄。短的資料列補空字串——正本明講「不產出對外介面就寫一列『無』」,那一列不該 與「沒有這一段」混為一談。 - referencedIndex 從關聯段落讀出 `需求議題:#N`。 Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
299 lines
11 KiB
JavaScript
299 lines
11 KiB
JavaScript
/**
|
||
* 議題 body 的 markdown 解析。
|
||
*
|
||
* 純函式,不碰網路也不碰檔案系統:抽取類腳本的解析全部走這裡,
|
||
* 解析規則只寫一次,需求議題與工作包議題共用同一套。
|
||
*
|
||
* 全檔的共同前提是「圍欄裡的東西不是內容」:議題裡放 mermaid 或程式碼是常態,
|
||
* 那裡面的井字號不是標題、減號不是清單項、直線不是表格。這件事只在 eachLine
|
||
* 處理一次,其餘函式都靠它。
|
||
*/
|
||
|
||
/**
|
||
* 逐行走過內容並標註圍欄狀態。
|
||
* 圍欄以 ``` 或 ~~~ 開啟,且要同一種標記才算關閉——混用時後者只是普通文字。
|
||
* 圍欄沒關就到結尾時,其後的內容一律算在圍欄內,這與 markdown 的實際渲染一致。
|
||
* @param {string} text
|
||
* @returns {Generator<{line: string, inFence: boolean, isFence: boolean}>}
|
||
*/
|
||
function* eachLine(text) {
|
||
let fence = null;
|
||
|
||
for (const line of (text ?? '').split('\n')) {
|
||
const marker = line.match(/^\s*(`{3,}|~{3,})/)?.[1]?.[0];
|
||
|
||
if (marker && fence === null) {
|
||
fence = marker;
|
||
yield { line, inFence: true, isFence: true };
|
||
continue;
|
||
}
|
||
if (marker && marker === fence) {
|
||
fence = null;
|
||
yield { line, inFence: true, isFence: true };
|
||
continue;
|
||
}
|
||
yield { line, inFence: fence !== null, isFence: false };
|
||
}
|
||
}
|
||
|
||
/**
|
||
* 依 `## 標題` 切出各段落。圍欄內的行原樣保留在段落內容裡
|
||
* ——流程圖那一段要的就是整個 mermaid 區塊。
|
||
* @param {string} body 議題 body
|
||
* @returns {Map<string, string>} 段落名稱 → 該段內容(前後空白已修掉)
|
||
*/
|
||
export function parseSections(body) {
|
||
const sections = new Map();
|
||
const buffer = [];
|
||
let current = null;
|
||
|
||
const flush = () => {
|
||
if (current !== null) sections.set(current, buffer.join('\n').trim());
|
||
buffer.length = 0;
|
||
};
|
||
|
||
for (const { line, inFence } of eachLine(body)) {
|
||
const heading = inFence ? null : line.match(/^##\s+(.+?)\s*$/);
|
||
if (heading) {
|
||
flush();
|
||
current = heading[1];
|
||
continue;
|
||
}
|
||
if (current !== null) buffer.push(line);
|
||
}
|
||
flush();
|
||
|
||
return sections;
|
||
}
|
||
|
||
/**
|
||
* 取出文字型段落的整段內容。段落不存在時回空字串,不報錯 ——
|
||
* 缺段落是模板的正常變體,不是解析失敗。
|
||
* @param {Map<string, string>} sections
|
||
* @param {string} name
|
||
* @returns {string}
|
||
*/
|
||
export function textSection(sections, name) {
|
||
return sections.get(name) ?? '';
|
||
}
|
||
|
||
/**
|
||
* 取出列表型段落的每一項。符號清單與編號清單一視同仁,checkbox 只留文字不留標記
|
||
* ——需求議題的驗收標準不追蹤勾選狀態。巢狀項目一律攤平:需求議題的模板沒有巢狀,
|
||
* 真的出現時寧可多帶一項,也不要無聲吃掉內容。
|
||
* @param {Map<string, string>} sections
|
||
* @param {string} name
|
||
* @returns {string[]}
|
||
*/
|
||
export function listSection(sections, name) {
|
||
const items = [];
|
||
|
||
for (const { line, inFence } of eachLine(sections.get(name))) {
|
||
if (inFence) continue;
|
||
// [\s\S] 而非 . :CRLF 的 body 逐行切開後行尾有 \r,而 . 不吃 \r,
|
||
// 用 . 會讓整行比不中,清單靜靜變成空的。後面的 trim 再把 \r 修掉。
|
||
const item = line.match(/^\s*(?:[-*+]|\d+\.)\s+([\s\S]*)$/);
|
||
if (!item) continue;
|
||
|
||
const text = item[1].replace(/^\[[ xX]\]\s*/, '').trim();
|
||
if (text !== '') items.push(text);
|
||
}
|
||
return items;
|
||
}
|
||
|
||
/**
|
||
* 取出巢狀的待辦清單:上層是待辦,縮排一層是該項自己的驗收。
|
||
*
|
||
* 只有這一支收 body 而不收切好的段落,因為它要交出 `raw`——下游靠 `raw` 在整份 body 上
|
||
* 做精確字串替換來勾選 checkbox,那一行必須逐字等於 body 裡的原樣,
|
||
* 連縮排、行尾空白與 \r 都不能動。段落切分會修掉前後空白,拿不到這種保證。
|
||
*
|
||
* 兩種畸形寫法都不丟內容,寧可放在稍微不對的位置也不要靜靜消失:
|
||
* - 巢狀超過一層 → 攤進所在待辦的驗收
|
||
* - 還沒有上層待辦就先出現縮排項目 → 升格成待辦
|
||
*
|
||
* @param {string} body 議題 body
|
||
* @param {string} section 段落名稱,例如 '待辦'
|
||
* @returns {{text: string, done: boolean, raw: string, 驗收: {text: string, done: boolean, raw: string}[]}[]}
|
||
*/
|
||
export function checklistInSection(body, section) {
|
||
const rows = [...eachLine(body)];
|
||
const { start, end } = sectionBounds(rows, section);
|
||
if (start === -1) return [];
|
||
|
||
const todos = [];
|
||
let topIndent = null;
|
||
|
||
for (let i = start + 1; i < end; i += 1) {
|
||
if (rows[i].inFence) continue;
|
||
const item = parseChecklistItem(rows[i].line);
|
||
if (!item) continue;
|
||
|
||
const nested = topIndent !== null && todos.length > 0 && item.indent > topIndent;
|
||
if (nested) {
|
||
todos.at(-1).驗收.push(item.value);
|
||
continue;
|
||
}
|
||
// 比目前認定的上層還淺時,把上層改認成更淺的那一層:第一項剛好縮排時,
|
||
// 後面出現的真正上層才不會被當成它的驗收。
|
||
topIndent = topIndent === null ? item.indent : Math.min(topIndent, item.indent);
|
||
todos.push({ ...item.value, 驗收: [] });
|
||
}
|
||
return todos;
|
||
}
|
||
|
||
/**
|
||
* 拆一行清單項。符號清單與編號清單一視同仁,checkbox 可有可無——
|
||
* 忘了寫 checkbox 的項目仍是一項待辦,只是 done 為 false。
|
||
* @returns {{indent: number, value: {text: string, done: boolean, raw: string}}|null}
|
||
*/
|
||
function parseChecklistItem(line) {
|
||
// [\s\S] 而非 . 的理由同 listSection:CRLF 的 body 行尾有 \r,. 不吃它。
|
||
// text 靠 trim 修掉 \r,raw 則原樣留著——它要逐字等於 body 裡的那一行。
|
||
const item = line.match(/^(\s*)(?:[-*+]|\d+\.)\s+([\s\S]*)$/);
|
||
if (!item) return null;
|
||
|
||
const box = item[2].match(/^\[([ xX])\]\s*([\s\S]*)$/);
|
||
const text = (box ? box[2] : item[2]).trim();
|
||
if (text === '') return null;
|
||
|
||
return {
|
||
indent: item[1].length,
|
||
value: { text, done: box ? box[1].toLowerCase() === 'x' : false, raw: line },
|
||
};
|
||
}
|
||
|
||
/**
|
||
* 在段落裡找出「標籤:#編號」那一行的編號,例如關聯段落的 `需求議題:#7`。
|
||
* 全形與半形冒號都認;找不到回 null——沒填不是解析失敗。
|
||
* @param {Map<string, string>} sections
|
||
* @param {string} name 段落名稱
|
||
* @param {string} label 標籤,例如 '需求議題'
|
||
* @returns {number|null}
|
||
*/
|
||
export function referencedIndex(sections, name, label) {
|
||
for (const { line, inFence } of eachLine(sections.get(name))) {
|
||
if (inFence) continue;
|
||
const at = line.indexOf(label);
|
||
if (at === -1) continue;
|
||
|
||
const value = line.slice(at + label.length).match(/^\s*[::]\s*#?(\d+)/);
|
||
if (value) return Number(value[1]);
|
||
}
|
||
return null;
|
||
}
|
||
|
||
/**
|
||
* 取出兩欄表格型段落,欄位固定命名為 term 與 def。
|
||
* 需求議題的領域名詞表用它;四欄的介面契約請用 tableRows。
|
||
* @param {Map<string, string>} sections
|
||
* @param {string} name
|
||
* @returns {{term: string, def: string}[]}
|
||
*/
|
||
export function tableSection(sections, name) {
|
||
return tableRows(sections, name, ['term', 'def']);
|
||
}
|
||
|
||
/**
|
||
* 取出表格型段落的資料列,欄位依 columns 命名。以分隔列(|---|---|)為界,
|
||
* 之後才是資料列;沒有分隔列就當成沒有資料,避免把表頭當成一筆資料。
|
||
*
|
||
* 資料列比 columns 短時補空字串而不是讓欄位消失——下游拿到的形狀要固定,
|
||
* 少一欄是內容的問題,不該變成「欄位不存在」讓下游多寫一種分支。
|
||
*
|
||
* @param {Map<string, string>} sections
|
||
* @param {string} name
|
||
* @param {string[]} columns 由左到右的欄位名稱;多出來的欄會被丟掉
|
||
* @returns {Record<string, string>[]}
|
||
*/
|
||
export function tableRows(sections, name, columns) {
|
||
const rows = [];
|
||
|
||
for (const { line, inFence } of eachLine(sections.get(name))) {
|
||
if (inFence || !line.trim().startsWith('|')) continue;
|
||
rows.push(splitRow(line));
|
||
}
|
||
|
||
const separator = rows.findIndex(isSeparator);
|
||
if (separator === -1) return [];
|
||
|
||
return rows
|
||
.slice(separator + 1)
|
||
// 同一段落裡若不慎貼了第二張表,它的分隔列不該變成一筆資料
|
||
.filter((cells) => !isSeparator(cells))
|
||
.filter((cells) => cells.some((cell) => cell !== ''))
|
||
.map((cells) => Object.fromEntries(columns.map((column, i) => [column, cells[i] ?? ''])));
|
||
}
|
||
|
||
/** 以未被逸脫的直線切欄,再把 `\|` 還原成內容裡的直線 */
|
||
function splitRow(line) {
|
||
return line
|
||
.trim()
|
||
.replace(/^\||\|$/g, '')
|
||
.split(/(?<!\\)\|/)
|
||
.map((cell) => cell.replace(/\\\|/g, '|').trim());
|
||
}
|
||
|
||
function isSeparator(cells) {
|
||
return cells.length > 0 && cells.every((cell) => /^:?-+:?$/.test(cell));
|
||
}
|
||
|
||
/**
|
||
* 在指定段落裡就地更新(或補上)一行「前綴+值」。
|
||
*
|
||
* 用於人天估算這種「議題上只該有一行、重跑要覆蓋而不是累積」的欄位。
|
||
*
|
||
* 三件事都限定在目標段落之內,因為弄錯的代價是靜靜改壞別人的內容:
|
||
* - 標題要完全相同才算數,`## 關聯度說明` 不是 `## 關聯`。
|
||
* - 圍欄裡的假標題不算標題,沿用本檔共同的圍欄判斷。
|
||
* - 找既有那一行只在段落範圍內找,別的段落剛好有同前綴時不會被改掉。
|
||
*
|
||
* 段落不存在時補在 body 結尾——寧可放錯位置,也不要讓值靜靜消失。
|
||
*
|
||
* @param {string} body 議題 body
|
||
* @param {string} section 段落名稱,例如 '關聯'
|
||
* @param {string} line 完整的一行,例如 '估算人天:3'
|
||
* @returns {string} 更新後的 body;內容沒有變動時回傳原字串
|
||
*/
|
||
export function upsertLineInSection(body, section, line) {
|
||
const prefix = line.slice(0, line.indexOf(':') + 1);
|
||
const rows = [...eachLine(body)];
|
||
const { start, end } = sectionBounds(rows, section);
|
||
|
||
if (start === -1) {
|
||
return `${body.replace(/\n*$/, '')}\n\n${line}\n`;
|
||
}
|
||
|
||
const text = rows.map((row) => row.line);
|
||
for (let i = start + 1; i < end; i += 1) {
|
||
if (rows[i].inFence || !text[i].startsWith(prefix)) continue;
|
||
if (text[i] === line) return body;
|
||
text[i] = line;
|
||
return text.join('\n');
|
||
}
|
||
|
||
// 插在段落內容的結尾,跳過段落與段落之間的空行
|
||
let insertAt = end;
|
||
while (insertAt > start + 1 && text[insertAt - 1].trim() === '') insertAt -= 1;
|
||
text.splice(insertAt, 0, line);
|
||
return text.join('\n');
|
||
}
|
||
|
||
/** 找出段落的起訖行號;start 為標題那一行,end 為下一個標題(或結尾) */
|
||
function sectionBounds(rows, section) {
|
||
let start = -1;
|
||
|
||
for (let i = 0; i < rows.length; i += 1) {
|
||
if (rows[i].inFence) continue;
|
||
const heading = rows[i].line.match(/^##\s+(.+?)\s*$/);
|
||
if (!heading) continue;
|
||
|
||
if (start === -1) {
|
||
if (heading[1] === section) start = i;
|
||
continue;
|
||
}
|
||
return { start, end: i };
|
||
}
|
||
return { start, end: rows.length };
|
||
}
|