wp-extract 與 wp-list 都在問「這顆工作包掛在哪顆需求底下」。規則寫兩份, 某天只會有一邊被改到,而分岔的樣子是「清單裡看得到、抽取卻說不是」。 順手把測試裡兩種取段落的寫法統一,並刪掉沒有人傳過的參數。 Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
471 lines
18 KiB
JavaScript
471 lines
18 KiB
JavaScript
/**
|
||
* 議題 body 的 markdown 解析。
|
||
*
|
||
* 純函式,不碰網路也不碰檔案系統:抽取類腳本的解析全部走這裡,
|
||
* 解析規則只寫一次,需求議題與工作包議題共用同一套。
|
||
*
|
||
* 全檔的共同前提是「圍欄裡的東西不是內容」:議題裡放 mermaid 或程式碼是常態,
|
||
* 那裡面的井字號不是標題、減號不是清單項、直線不是表格。這件事只在 eachLine
|
||
* 處理一次,其餘函式都靠它。
|
||
*/
|
||
|
||
/**
|
||
* 逐行走過內容並標註圍欄狀態。
|
||
* 圍欄以 ``` 或 ~~~ 開啟,且要同一種標記才算關閉——混用時後者只是普通文字。
|
||
* 圍欄沒關就到結尾時,其後的內容一律算在圍欄內,這與 markdown 的實際渲染一致。
|
||
* @param {string} text
|
||
* @returns {Generator<{line: string, inFence: boolean, isFence: boolean}>}
|
||
*/
|
||
function* eachLine(text) {
|
||
let fence = null;
|
||
|
||
for (const line of (text ?? '').split('\n')) {
|
||
const marker = line.match(/^\s*(`{3,}|~{3,})/)?.[1]?.[0];
|
||
|
||
if (marker && fence === null) {
|
||
fence = marker;
|
||
yield { line, inFence: true, isFence: true };
|
||
continue;
|
||
}
|
||
if (marker && marker === fence) {
|
||
fence = null;
|
||
yield { line, inFence: true, isFence: true };
|
||
continue;
|
||
}
|
||
yield { line, inFence: fence !== null, isFence: false };
|
||
}
|
||
}
|
||
|
||
/**
|
||
* 依 `## 標題` 切出各段落。圍欄內的行原樣保留在段落內容裡
|
||
* ——流程圖那一段要的就是整個 mermaid 區塊。
|
||
* @param {string} body 議題 body
|
||
* @returns {Map<string, string>} 段落名稱 → 該段內容(前後空白已修掉)
|
||
*/
|
||
export function parseSections(body) {
|
||
const sections = new Map();
|
||
const buffer = [];
|
||
let current = null;
|
||
|
||
const flush = () => {
|
||
if (current !== null) sections.set(current, buffer.join('\n').trim());
|
||
buffer.length = 0;
|
||
};
|
||
|
||
for (const { line, inFence } of eachLine(body)) {
|
||
const heading = inFence ? null : line.match(/^##\s+(.+?)\s*$/);
|
||
if (heading) {
|
||
flush();
|
||
current = heading[1];
|
||
continue;
|
||
}
|
||
if (current !== null) buffer.push(line);
|
||
}
|
||
flush();
|
||
|
||
return sections;
|
||
}
|
||
|
||
/**
|
||
* 取出文字型段落的整段內容。段落不存在時回空字串,不報錯 ——
|
||
* 缺段落是模板的正常變體,不是解析失敗。
|
||
* @param {Map<string, string>} sections
|
||
* @param {string} name
|
||
* @returns {string}
|
||
*/
|
||
export function textSection(sections, name) {
|
||
return sections.get(name) ?? '';
|
||
}
|
||
|
||
/**
|
||
* 取出列表型段落的每一項。符號清單與編號清單一視同仁,checkbox 只留文字不留標記
|
||
* ——需求議題的驗收標準不追蹤勾選狀態。巢狀項目一律攤平:需求議題的模板沒有巢狀,
|
||
* 真的出現時寧可多帶一項,也不要無聲吃掉內容。
|
||
* @param {Map<string, string>} sections
|
||
* @param {string} name
|
||
* @returns {string[]}
|
||
*/
|
||
export function listSection(sections, name) {
|
||
const items = [];
|
||
|
||
for (const { line, inFence } of eachLine(sections.get(name))) {
|
||
if (inFence) continue;
|
||
// [\s\S] 而非 . :CRLF 的 body 逐行切開後行尾有 \r,而 . 不吃 \r,
|
||
// 用 . 會讓整行比不中,清單靜靜變成空的。後面的 trim 再把 \r 修掉。
|
||
const item = line.match(/^\s*(?:[-*+]|\d+\.)\s+([\s\S]*)$/);
|
||
if (!item) continue;
|
||
|
||
const text = item[1].replace(/^\[[ xX]\]\s*/, '').trim();
|
||
if (text !== '') items.push(text);
|
||
}
|
||
return items;
|
||
}
|
||
|
||
/**
|
||
* 取出巢狀的待辦清單:上層是待辦,縮排一層是該項自己的驗收。
|
||
*
|
||
* 只有這一支收 body 而不收切好的段落,因為它要交出 `raw`——下游靠 `raw` 在整份 body 上
|
||
* 做精確字串替換來勾選 checkbox,那一行必須逐字等於 body 裡的原樣,
|
||
* 連縮排、行尾空白與 \r 都不能動。段落切分會修掉前後空白,拿不到這種保證。
|
||
*
|
||
* 兩種畸形寫法都不丟內容,寧可放在稍微不對的位置也不要靜靜消失:
|
||
* - 巢狀超過一層 → 攤進所在待辦的驗收
|
||
* - 還沒有上層待辦就先出現縮排項目 → 升格成待辦
|
||
*
|
||
* @param {string} body 議題 body
|
||
* @param {string} section 段落名稱,例如 '待辦'
|
||
* @returns {{text: string, done: boolean, raw: string, 驗收: {text: string, done: boolean, raw: string}[]}[]}
|
||
*/
|
||
export function checklistInSection(body, section) {
|
||
const rows = [...eachLine(body)];
|
||
const { start, end } = sectionBounds(rows, section);
|
||
if (start === -1) return [];
|
||
|
||
const todos = [];
|
||
let topIndent = null;
|
||
|
||
for (let i = start + 1; i < end; i += 1) {
|
||
if (rows[i].inFence) continue;
|
||
const item = parseChecklistItem(rows[i].line);
|
||
if (!item) continue;
|
||
|
||
const nested = topIndent !== null && todos.length > 0 && item.indent > topIndent;
|
||
if (nested) {
|
||
todos.at(-1).驗收.push(item.value);
|
||
continue;
|
||
}
|
||
// 比目前認定的上層還淺時,把上層改認成更淺的那一層:第一項剛好縮排時,
|
||
// 後面出現的真正上層才不會被當成它的驗收。
|
||
topIndent = topIndent === null ? item.indent : Math.min(topIndent, item.indent);
|
||
todos.push({ ...item.value, 驗收: [] });
|
||
}
|
||
return todos;
|
||
}
|
||
|
||
/**
|
||
* 拆一行清單項。符號清單與編號清單一視同仁,checkbox 可有可無——
|
||
* 忘了寫 checkbox 的項目仍是一項待辦,只是 done 為 false。
|
||
* @returns {{indent: number, value: {text: string, done: boolean, raw: string}}|null}
|
||
*/
|
||
function parseChecklistItem(line) {
|
||
// 文法與 tickLine 共用 LIST_ITEM:抽得出來的行,勾選端就要收得下。
|
||
// text 靠 trim 修掉 CRLF 的 \r,raw 則原樣留著——它要逐字等於 body 裡的那一行。
|
||
const item = LIST_ITEM.exec(line);
|
||
if (!item) return null;
|
||
|
||
const text = item[3].trim();
|
||
if (text === '') return null;
|
||
|
||
return {
|
||
indent: item[1].length,
|
||
value: { text, done: item[2]?.toLowerCase() === '[x]', raw: line },
|
||
};
|
||
}
|
||
|
||
/**
|
||
* 在段落裡找出「標籤:#編號」那一行的編號,例如關聯段落的 `需求議題:#7`。
|
||
* 全形與半形冒號都認;找不到回 null——沒填不是解析失敗。
|
||
* @param {Map<string, string>} sections
|
||
* @param {string} name 段落名稱
|
||
* @param {string} label 標籤,例如 '需求議題'
|
||
* @returns {number|null}
|
||
*/
|
||
export function referencedIndex(sections, name, label) {
|
||
for (const { line, inFence } of eachLine(sections.get(name))) {
|
||
if (inFence) continue;
|
||
const at = line.indexOf(label);
|
||
if (at === -1) continue;
|
||
|
||
const value = line.slice(at + label.length).match(/^\s*[::]\s*#?(\d+)/);
|
||
if (value) return Number(value[1]);
|
||
}
|
||
return null;
|
||
}
|
||
|
||
/**
|
||
* 工作包掛在哪一顆需求議題底下:關聯段落的 `需求議題:#7`。
|
||
*
|
||
* 這是「這顆議題是不是工作包、屬於誰」的唯一判準,抽取(wp-extract)與清單(wp-list)
|
||
* 共用同一個函式。兩邊各寫一次也跑得起來,但歸屬規則一旦有兩份,某天只會有一邊被改到,
|
||
* 而分岔的樣子是「清單裡看得到、抽取卻說不是」——那種不一致沒有人看得懂。
|
||
* @param {Map<string, string>} sections
|
||
* @returns {number|null} 沒填或不是工作包時為 null
|
||
*/
|
||
export function requirementIndex(sections) {
|
||
return referencedIndex(sections, '關聯', '需求議題');
|
||
}
|
||
|
||
/**
|
||
* 在段落裡找出「標籤:數字」那一行的數字,例如關聯段落的 `估算人天:3`。
|
||
* 與 referencedIndex 同形狀,差別只在這裡要的是數量而非議題編號,所以認小數。
|
||
* 全形與半形冒號都認;找不到回 null——沒填不是解析失敗,呼叫端要分得開「沒估」與「估 0」。
|
||
* @param {Map<string, string>} sections
|
||
* @param {string} name 段落名稱
|
||
* @param {string} label 標籤,例如 '估算人天'
|
||
* @returns {number|null}
|
||
*/
|
||
export function labelledNumber(sections, name, label) {
|
||
for (const { line, inFence } of eachLine(sections.get(name))) {
|
||
if (inFence) continue;
|
||
const at = line.indexOf(label);
|
||
if (at === -1) continue;
|
||
|
||
const value = line.slice(at + label.length).match(/^\s*[::]\s*(\d+(?:\.\d+)?)\s*$/);
|
||
if (value) return Number(value[1]);
|
||
}
|
||
return null;
|
||
}
|
||
|
||
/**
|
||
* 取出兩欄表格型段落,欄位固定命名為 term 與 def。
|
||
* 需求議題的領域名詞表用它;四欄的介面契約請用 tableRows。
|
||
* @param {Map<string, string>} sections
|
||
* @param {string} name
|
||
* @returns {{term: string, def: string}[]}
|
||
*/
|
||
export function tableSection(sections, name) {
|
||
return tableRows(sections, name, ['term', 'def']);
|
||
}
|
||
|
||
/**
|
||
* 取出表格型段落的資料列,欄位依 columns 命名。以分隔列(|---|---|)為界,
|
||
* 之後才是資料列;沒有分隔列就當成沒有資料,避免把表頭當成一筆資料。
|
||
*
|
||
* 資料列比 columns 短時補空字串而不是讓欄位消失——下游拿到的形狀要固定,
|
||
* 少一欄是內容的問題,不該變成「欄位不存在」讓下游多寫一種分支。
|
||
*
|
||
* @param {Map<string, string>} sections
|
||
* @param {string} name
|
||
* @param {string[]} columns 由左到右的欄位名稱;多出來的欄會被丟掉
|
||
* @returns {Record<string, string>[]}
|
||
*/
|
||
export function tableRows(sections, name, columns) {
|
||
const rows = [];
|
||
|
||
for (const { line, inFence } of eachLine(sections.get(name))) {
|
||
if (inFence || !line.trim().startsWith('|')) continue;
|
||
rows.push(splitRow(line));
|
||
}
|
||
|
||
const separator = rows.findIndex(isSeparator);
|
||
if (separator === -1) return [];
|
||
|
||
return rows
|
||
.slice(separator + 1)
|
||
// 同一段落裡若不慎貼了第二張表,它的分隔列不該變成一筆資料
|
||
.filter((cells) => !isSeparator(cells))
|
||
.filter((cells) => cells.some((cell) => cell !== ''))
|
||
.map((cells) => Object.fromEntries(columns.map((column, i) => [column, cells[i] ?? ''])));
|
||
}
|
||
|
||
/** 以未被逸脫的直線切欄,再把 `\|` 還原成內容裡的直線 */
|
||
function splitRow(line) {
|
||
return line
|
||
.trim()
|
||
.replace(/^\||\|$/g, '')
|
||
.split(/(?<!\\)\|/)
|
||
.map((cell) => cell.replace(/\\\|/g, '|').trim());
|
||
}
|
||
|
||
function isSeparator(cells) {
|
||
return cells.length > 0 && cells.every((cell) => /^:?-+:?$/.test(cell));
|
||
}
|
||
|
||
/**
|
||
* 一行清單項的文法:符號或編號清單,後面可以有一個 checkbox。
|
||
*
|
||
* 全檔只有這一份定義。抽取端(parseChecklistItem)與勾選端(tickLine)若各寫一份,
|
||
* 遲早會鬆緊不一——抽得出來卻勾不動的那一行,會讓「一律用 wp-extract 給的 raw」
|
||
* 變成做不到的指示。
|
||
*/
|
||
const LIST_ITEM = /^(\s*)(?:[-*+]|\d+\.)\s+(\[[ xX]\])?\s*([\s\S]*)$/;
|
||
|
||
/**
|
||
* 這一行是不是清單項(有沒有 checkbox 都算)。
|
||
*
|
||
* `--tick` 用它驗輸入,而且刻意不要求 checkbox:抽取端會把「忘了寫 checkbox 的待辦」
|
||
* 也收成一項待辦,那種 raw 要走到 tickLine 才能得到「去議題上補成 checkbox」這句話,
|
||
* 在入口就擋掉只會回一個看不出該怎麼辦的格式錯誤。
|
||
* @param {string} line
|
||
* @returns {boolean}
|
||
*/
|
||
export function isListItem(line) {
|
||
return LIST_ITEM.test(line);
|
||
}
|
||
|
||
/**
|
||
* 勾起一行 checkbox:把 `raw` 那一行的方框換成已勾,其餘一字不動。
|
||
*
|
||
* 三件事都限定在目標段落之內、且跳過圍欄,理由與 upsertLineInSection 相同——
|
||
* 弄錯的代價是靜靜改壞別人的內容。勾選是這個檔案裡唯一會寫回議題的路徑,
|
||
* 而工作包模板的架構圖就是一塊 fenced mermaid:裡面出現減號開頭的行是常態,
|
||
* 把它當成待辦勾下去,改壞的是一張圖。
|
||
*
|
||
* 用整行精確比對而不是「找那段文字」,因為巢狀待辦底下常有一模一樣的驗收
|
||
* (兩項待辦各有一條「加上測試」)。認不出是哪一行時交回 ambiguous 讓呼叫端報錯,
|
||
* 不賭第一個——猜錯的話議題上的進度條會指著錯的那一項,而沒有人會去比對編輯紀錄。
|
||
*
|
||
* 勾選狀態與大小寫都不影響比對:`[ ]`、`[x]`、`[X]` 指的是同一行,
|
||
* 已經勾過就交回 already,讓中斷後重跑是安靜的 no-op 而不是失敗。
|
||
*
|
||
* 本函式不拋錯——它是純解析,錯誤碼由呼叫端決定。
|
||
*
|
||
* @param {string} body 議題 body
|
||
* @param {string} raw 抽取契約交出的原始 markdown 行,逐字包含縮排與行尾的 \r
|
||
* @param {string} [section] 限定在這個段落內找;省略時找全文(圍欄照樣不算)
|
||
* @returns {{status: 'ticked'|'already'|'not-found'|'ambiguous'|'no-checkbox'|'no-section', body?: string, line?: string, count: number}}
|
||
*/
|
||
export function tickLine(body, raw, section) {
|
||
const item = LIST_ITEM.exec(raw);
|
||
if (!item || item[2] === undefined) return { status: 'no-checkbox', count: 0 };
|
||
|
||
const rows = [...eachLine(body)];
|
||
const { start, end } = section === undefined
|
||
? { start: -1, end: rows.length }
|
||
: sectionBounds(rows, section);
|
||
if (section !== undefined && start === -1) return { status: 'no-section', count: 0 };
|
||
|
||
/** 同一行的三種寫法都指向它自己:比對時一律正規化成未勾的小寫版本 */
|
||
const normalize = (line) => line.replace(/\[[ xX]\]/, '[ ]');
|
||
const wanted = normalize(raw);
|
||
const ticked = raw.replace(/\[[ xX]\]/, '[x]');
|
||
|
||
const hits = [];
|
||
for (let i = start + 1; i < end; i += 1) {
|
||
if (rows[i].inFence) continue;
|
||
if (normalize(rows[i].line) === wanted) hits.push(i);
|
||
}
|
||
|
||
if (hits.length === 0) return { status: 'not-found', count: 0 };
|
||
if (hits.length > 1) return { status: 'ambiguous', count: hits.length };
|
||
|
||
const [at] = hits;
|
||
// 已勾與否看方框本身,不比整行字串:`[X]` 是合法的 GFM,Gitea 也渲染成已勾,
|
||
// 用字串相等判斷會把它當成還沒勾,於是重跑時硬把大寫改成小寫
|
||
if (LIST_ITEM.exec(rows[at].line)[2].toLowerCase() === '[x]') {
|
||
return { status: 'already', line: rows[at].line, count: 1 };
|
||
}
|
||
|
||
const lines = rows.map((row) => row.line);
|
||
lines[at] = ticked;
|
||
return { status: 'ticked', body: lines.join('\n'), line: ticked, count: 1 };
|
||
}
|
||
|
||
/**
|
||
* 換掉一個段落的內容,標題與其餘段落一字不動。
|
||
*
|
||
* 整併留言裡的決策時用它。不整份重寫的理由跟 upsertLineInSection 一樣,只是代價更大:
|
||
* 重寫會把別人在其他段落的編輯一起蓋掉,而議題的編輯紀錄沒有人會去比對。
|
||
*
|
||
* 同名標題出現不只一次時交回 `ambiguous`,不賭第一個——理由與 tickLine 相同,
|
||
* 而這裡蓋掉的是一整段而不是一行,猜錯的代價更高。
|
||
*
|
||
* 段落不存在時交回 `not-found` 讓呼叫端報錯,不補在結尾:「找不到那一段」多半是段落名
|
||
* 打錯,這時把內容塞到議題末尾,比什麼都不做更難收拾。
|
||
*
|
||
* @param {string} body 議題 body
|
||
* @param {string} section 段落名稱,例如 '目標'
|
||
* @param {string} content 新的段落內容(不含 `## 標題` 那一行)
|
||
* @returns {{status: 'replaced'|'not-found'|'ambiguous', body?: string, count: number}}
|
||
*/
|
||
export function replaceSection(body, section, content) {
|
||
const rows = [...eachLine(body)];
|
||
const headings = [];
|
||
for (let i = 0; i < rows.length; i += 1) {
|
||
if (rows[i].inFence) continue;
|
||
if (rows[i].line.match(/^##\s+(.+?)\s*$/)?.[1] === section) headings.push(i);
|
||
}
|
||
|
||
if (headings.length === 0) return { status: 'not-found', count: 0 };
|
||
if (headings.length > 1) return { status: 'ambiguous', count: headings.length };
|
||
|
||
const [start] = headings;
|
||
const lines = rows.map((row) => row.line);
|
||
// 下一個段落的標題;沒有就是到結尾
|
||
let end = lines.length;
|
||
for (let i = start + 1; i < lines.length; i += 1) {
|
||
if (!rows[i].inFence && /^##\s+/.test(lines[i])) {
|
||
end = i;
|
||
break;
|
||
}
|
||
}
|
||
|
||
// 段落與段落之間的空行屬於版面,不屬於內容:換內容時把它留著。
|
||
// 原本就沒有空行(兩個標題緊貼)時補一個,免得新內容黏在下一個標題上。
|
||
let tail = end;
|
||
while (tail > start + 1 && lines[tail - 1].trim() === '') tail -= 1;
|
||
const spacer = end === lines.length || end > tail ? lines.slice(tail, end) : [''];
|
||
|
||
// 換行沿用 body 原本的那一種:CRLF 的 body 裡混進 LF,會讓抽取契約交出的 raw
|
||
// 對不上原文,之後就勾不動那幾行了
|
||
const eol = body.includes('\r\n') ? '\r\n' : '\n';
|
||
const normalized = content.trim().split(/\r?\n/);
|
||
|
||
return {
|
||
status: 'replaced',
|
||
body: [...lines.slice(0, start + 1), '', ...normalized, ...spacer, ...lines.slice(end)]
|
||
.map((line) => line.replace(/\r$/, ''))
|
||
.join(eol),
|
||
count: 1,
|
||
};
|
||
}
|
||
|
||
/**
|
||
* 在指定段落裡就地更新(或補上)一行「前綴+值」。
|
||
*
|
||
* 用於人天估算這種「議題上只該有一行、重跑要覆蓋而不是累積」的欄位。
|
||
*
|
||
* 三件事都限定在目標段落之內,因為弄錯的代價是靜靜改壞別人的內容:
|
||
* - 標題要完全相同才算數,`## 關聯度說明` 不是 `## 關聯`。
|
||
* - 圍欄裡的假標題不算標題,沿用本檔共同的圍欄判斷。
|
||
* - 找既有那一行只在段落範圍內找,別的段落剛好有同前綴時不會被改掉。
|
||
*
|
||
* 段落不存在時補在 body 結尾——寧可放錯位置,也不要讓值靜靜消失。
|
||
*
|
||
* @param {string} body 議題 body
|
||
* @param {string} section 段落名稱,例如 '關聯'
|
||
* @param {string} line 完整的一行,例如 '估算人天:3'
|
||
* @returns {string} 更新後的 body;內容沒有變動時回傳原字串
|
||
*/
|
||
export function upsertLineInSection(body, section, line) {
|
||
const prefix = line.slice(0, line.indexOf(':') + 1);
|
||
const rows = [...eachLine(body)];
|
||
const { start, end } = sectionBounds(rows, section);
|
||
|
||
if (start === -1) {
|
||
return `${body.replace(/\n*$/, '')}\n\n${line}\n`;
|
||
}
|
||
|
||
const text = rows.map((row) => row.line);
|
||
for (let i = start + 1; i < end; i += 1) {
|
||
if (rows[i].inFence || !text[i].startsWith(prefix)) continue;
|
||
if (text[i] === line) return body;
|
||
text[i] = line;
|
||
return text.join('\n');
|
||
}
|
||
|
||
// 插在段落內容的結尾,跳過段落與段落之間的空行
|
||
let insertAt = end;
|
||
while (insertAt > start + 1 && text[insertAt - 1].trim() === '') insertAt -= 1;
|
||
text.splice(insertAt, 0, line);
|
||
return text.join('\n');
|
||
}
|
||
|
||
/** 找出段落的起訖行號;start 為標題那一行,end 為下一個標題(或結尾) */
|
||
function sectionBounds(rows, section) {
|
||
let start = -1;
|
||
|
||
for (let i = 0; i < rows.length; i += 1) {
|
||
if (rows[i].inFence) continue;
|
||
const heading = rows[i].line.match(/^##\s+(.+?)\s*$/);
|
||
if (!heading) continue;
|
||
|
||
if (start === -1) {
|
||
if (heading[1] === section) start = i;
|
||
continue;
|
||
}
|
||
return { start, end: i };
|
||
}
|
||
return { start, end: rows.length };
|
||
}
|