/** * 議題 body 的 markdown 解析。 * * 純函式,不碰網路也不碰檔案系統:抽取類腳本的解析全部走這裡, * 解析規則只寫一次,需求議題與工作包議題共用同一套。 * * 全檔的共同前提是「圍欄裡的東西不是內容」:議題裡放 mermaid 或程式碼是常態, * 那裡面的井字號不是標題、減號不是清單項、直線不是表格。這件事只在 eachLine * 處理一次,其餘函式都靠它。 */ /** * 逐行走過內容並標註圍欄狀態。 * 圍欄以 ``` 或 ~~~ 開啟,且要同一種標記才算關閉——混用時後者只是普通文字。 * 圍欄沒關就到結尾時,其後的內容一律算在圍欄內,這與 markdown 的實際渲染一致。 * @param {string} text * @returns {Generator<{line: string, inFence: boolean, isFence: boolean}>} */ function* eachLine(text) { let fence = null; for (const line of (text ?? '').split('\n')) { const marker = line.match(/^\s*(`{3,}|~{3,})/)?.[1]?.[0]; if (marker && fence === null) { fence = marker; yield { line, inFence: true, isFence: true }; continue; } if (marker && marker === fence) { fence = null; yield { line, inFence: true, isFence: true }; continue; } yield { line, inFence: fence !== null, isFence: false }; } } /** * 依 `## 標題` 切出各段落。圍欄內的行原樣保留在段落內容裡 * ——流程圖那一段要的就是整個 mermaid 區塊。 * @param {string} body 議題 body * @returns {Map} 段落名稱 → 該段內容(前後空白已修掉) */ export function parseSections(body) { const sections = new Map(); const buffer = []; let current = null; const flush = () => { if (current !== null) sections.set(current, buffer.join('\n').trim()); buffer.length = 0; }; for (const { line, inFence } of eachLine(body)) { const heading = inFence ? null : line.match(/^##\s+(.+?)\s*$/); if (heading) { flush(); current = heading[1]; continue; } if (current !== null) buffer.push(line); } flush(); return sections; } /** * 取出文字型段落的整段內容。段落不存在時回空字串,不報錯 —— * 缺段落是模板的正常變體,不是解析失敗。 * @param {Map} sections * @param {string} name * @returns {string} */ export function textSection(sections, name) { return sections.get(name) ?? ''; } /** * 取出列表型段落的每一項。符號清單與編號清單一視同仁,checkbox 只留文字不留標記 * ——需求議題的驗收標準不追蹤勾選狀態。巢狀項目一律攤平:需求議題的模板沒有巢狀, * 真的出現時寧可多帶一項,也不要無聲吃掉內容。 * @param {Map} sections * @param {string} name * @returns {string[]} */ export function listSection(sections, name) { const items = []; for (const { line, inFence } of eachLine(sections.get(name))) { if (inFence) continue; // [\s\S] 而非 . :CRLF 的 body 逐行切開後行尾有 \r,而 . 不吃 \r, // 用 . 會讓整行比不中,清單靜靜變成空的。後面的 trim 再把 \r 修掉。 const item = line.match(/^\s*(?:[-*+]|\d+\.)\s+([\s\S]*)$/); if (!item) continue; const text = item[1].replace(/^\[[ xX]\]\s*/, '').trim(); if (text !== '') items.push(text); } return items; } /** * 取出巢狀的待辦清單:上層是待辦,縮排一層是該項自己的驗收。 * * 只有這一支收 body 而不收切好的段落,因為它要交出 `raw`——下游靠 `raw` 在整份 body 上 * 做精確字串替換來勾選 checkbox,那一行必須逐字等於 body 裡的原樣, * 連縮排、行尾空白與 \r 都不能動。段落切分會修掉前後空白,拿不到這種保證。 * * 兩種畸形寫法都不丟內容,寧可放在稍微不對的位置也不要靜靜消失: * - 巢狀超過一層 → 攤進所在待辦的驗收 * - 還沒有上層待辦就先出現縮排項目 → 升格成待辦 * * @param {string} body 議題 body * @param {string} section 段落名稱,例如 '待辦' * @returns {{text: string, done: boolean, raw: string, 驗收: {text: string, done: boolean, raw: string}[]}[]} */ export function checklistInSection(body, section) { const rows = [...eachLine(body)]; const { start, end } = sectionBounds(rows, section); if (start === -1) return []; const todos = []; let topIndent = null; for (let i = start + 1; i < end; i += 1) { if (rows[i].inFence) continue; const item = parseChecklistItem(rows[i].line); if (!item) continue; const nested = topIndent !== null && todos.length > 0 && item.indent > topIndent; if (nested) { todos.at(-1).驗收.push(item.value); continue; } // 比目前認定的上層還淺時,把上層改認成更淺的那一層:第一項剛好縮排時, // 後面出現的真正上層才不會被當成它的驗收。 topIndent = topIndent === null ? item.indent : Math.min(topIndent, item.indent); todos.push({ ...item.value, 驗收: [] }); } return todos; } /** * 拆一行清單項。符號清單與編號清單一視同仁,checkbox 可有可無—— * 忘了寫 checkbox 的項目仍是一項待辦,只是 done 為 false。 * @returns {{indent: number, value: {text: string, done: boolean, raw: string}}|null} */ function parseChecklistItem(line) { // 文法與 tickLine 共用 LIST_ITEM:抽得出來的行,勾選端就要收得下。 // text 靠 trim 修掉 CRLF 的 \r,raw 則原樣留著——它要逐字等於 body 裡的那一行。 const item = LIST_ITEM.exec(line); if (!item) return null; const text = item[3].trim(); if (text === '') return null; return { indent: item[1].length, value: { text, done: item[2]?.toLowerCase() === '[x]', raw: line }, }; } /** * 在段落裡找出「標籤:#編號」那一行的編號,例如關聯段落的 `需求議題:#7`。 * 全形與半形冒號都認;找不到回 null——沒填不是解析失敗。 * @param {Map} sections * @param {string} name 段落名稱 * @param {string} label 標籤,例如 '需求議題' * @returns {number|null} */ export function referencedIndex(sections, name, label) { for (const { line, inFence } of eachLine(sections.get(name))) { if (inFence) continue; const at = line.indexOf(label); if (at === -1) continue; const value = line.slice(at + label.length).match(/^\s*[::]\s*#?(\d+)/); if (value) return Number(value[1]); } return null; } /** * 在段落裡找出「標籤:數字」那一行的數字,例如關聯段落的 `估算人天:3`。 * 與 referencedIndex 同形狀,差別只在這裡要的是數量而非議題編號,所以認小數。 * 全形與半形冒號都認;找不到回 null——沒填不是解析失敗,呼叫端要分得開「沒估」與「估 0」。 * @param {Map} sections * @param {string} name 段落名稱 * @param {string} label 標籤,例如 '估算人天' * @returns {number|null} */ export function labelledNumber(sections, name, label) { for (const { line, inFence } of eachLine(sections.get(name))) { if (inFence) continue; const at = line.indexOf(label); if (at === -1) continue; const value = line.slice(at + label.length).match(/^\s*[::]\s*(\d+(?:\.\d+)?)\s*$/); if (value) return Number(value[1]); } return null; } /** * 取出兩欄表格型段落,欄位固定命名為 term 與 def。 * 需求議題的領域名詞表用它;四欄的介面契約請用 tableRows。 * @param {Map} sections * @param {string} name * @returns {{term: string, def: string}[]} */ export function tableSection(sections, name) { return tableRows(sections, name, ['term', 'def']); } /** * 取出表格型段落的資料列,欄位依 columns 命名。以分隔列(|---|---|)為界, * 之後才是資料列;沒有分隔列就當成沒有資料,避免把表頭當成一筆資料。 * * 資料列比 columns 短時補空字串而不是讓欄位消失——下游拿到的形狀要固定, * 少一欄是內容的問題,不該變成「欄位不存在」讓下游多寫一種分支。 * * @param {Map} sections * @param {string} name * @param {string[]} columns 由左到右的欄位名稱;多出來的欄會被丟掉 * @returns {Record[]} */ export function tableRows(sections, name, columns) { const rows = []; for (const { line, inFence } of eachLine(sections.get(name))) { if (inFence || !line.trim().startsWith('|')) continue; rows.push(splitRow(line)); } const separator = rows.findIndex(isSeparator); if (separator === -1) return []; return rows .slice(separator + 1) // 同一段落裡若不慎貼了第二張表,它的分隔列不該變成一筆資料 .filter((cells) => !isSeparator(cells)) .filter((cells) => cells.some((cell) => cell !== '')) .map((cells) => Object.fromEntries(columns.map((column, i) => [column, cells[i] ?? '']))); } /** 以未被逸脫的直線切欄,再把 `\|` 還原成內容裡的直線 */ function splitRow(line) { return line .trim() .replace(/^\||\|$/g, '') .split(/(? cell.replace(/\\\|/g, '|').trim()); } function isSeparator(cells) { return cells.length > 0 && cells.every((cell) => /^:?-+:?$/.test(cell)); } /** * 一行清單項的文法:符號或編號清單,後面可以有一個 checkbox。 * * 全檔只有這一份定義。抽取端(parseChecklistItem)與勾選端(tickLine)若各寫一份, * 遲早會鬆緊不一——抽得出來卻勾不動的那一行,會讓「一律用 wp-extract 給的 raw」 * 變成做不到的指示。 */ const LIST_ITEM = /^(\s*)(?:[-*+]|\d+\.)\s+(\[[ xX]\])?\s*([\s\S]*)$/; /** * 這一行是不是清單項(有沒有 checkbox 都算)。 * * `--tick` 用它驗輸入,而且刻意不要求 checkbox:抽取端會把「忘了寫 checkbox 的待辦」 * 也收成一項待辦,那種 raw 要走到 tickLine 才能得到「去議題上補成 checkbox」這句話, * 在入口就擋掉只會回一個看不出該怎麼辦的格式錯誤。 * @param {string} line * @returns {boolean} */ export function isListItem(line) { return LIST_ITEM.test(line); } /** * 勾起一行 checkbox:把 `raw` 那一行的方框換成已勾,其餘一字不動。 * * 三件事都限定在目標段落之內、且跳過圍欄,理由與 upsertLineInSection 相同—— * 弄錯的代價是靜靜改壞別人的內容。勾選是這個檔案裡唯一會寫回議題的路徑, * 而工作包模板的架構圖就是一塊 fenced mermaid:裡面出現減號開頭的行是常態, * 把它當成待辦勾下去,改壞的是一張圖。 * * 用整行精確比對而不是「找那段文字」,因為巢狀待辦底下常有一模一樣的驗收 * (兩項待辦各有一條「加上測試」)。認不出是哪一行時交回 ambiguous 讓呼叫端報錯, * 不賭第一個——猜錯的話議題上的進度條會指著錯的那一項,而沒有人會去比對編輯紀錄。 * * 勾選狀態與大小寫都不影響比對:`[ ]`、`[x]`、`[X]` 指的是同一行, * 已經勾過就交回 already,讓中斷後重跑是安靜的 no-op 而不是失敗。 * * 本函式不拋錯——它是純解析,錯誤碼由呼叫端決定。 * * @param {string} body 議題 body * @param {string} raw 抽取契約交出的原始 markdown 行,逐字包含縮排與行尾的 \r * @param {string} [section] 限定在這個段落內找;省略時找全文(圍欄照樣不算) * @returns {{status: 'ticked'|'already'|'not-found'|'ambiguous'|'no-checkbox'|'no-section', body?: string, line?: string, count: number}} */ export function tickLine(body, raw, section) { const item = LIST_ITEM.exec(raw); if (!item || item[2] === undefined) return { status: 'no-checkbox', count: 0 }; const rows = [...eachLine(body)]; const { start, end } = section === undefined ? { start: -1, end: rows.length } : sectionBounds(rows, section); if (section !== undefined && start === -1) return { status: 'no-section', count: 0 }; /** 同一行的三種寫法都指向它自己:比對時一律正規化成未勾的小寫版本 */ const normalize = (line) => line.replace(/\[[ xX]\]/, '[ ]'); const wanted = normalize(raw); const ticked = raw.replace(/\[[ xX]\]/, '[x]'); const hits = []; for (let i = start + 1; i < end; i += 1) { if (rows[i].inFence) continue; if (normalize(rows[i].line) === wanted) hits.push(i); } if (hits.length === 0) return { status: 'not-found', count: 0 }; if (hits.length > 1) return { status: 'ambiguous', count: hits.length }; const [at] = hits; // 已勾與否看方框本身,不比整行字串:`[X]` 是合法的 GFM,Gitea 也渲染成已勾, // 用字串相等判斷會把它當成還沒勾,於是重跑時硬把大寫改成小寫 if (LIST_ITEM.exec(rows[at].line)[2].toLowerCase() === '[x]') { return { status: 'already', line: rows[at].line, count: 1 }; } const lines = rows.map((row) => row.line); lines[at] = ticked; return { status: 'ticked', body: lines.join('\n'), line: ticked, count: 1 }; } /** * 換掉一個段落的內容,標題與其餘段落一字不動。 * * 整併留言裡的決策時用它。不整份重寫的理由跟 upsertLineInSection 一樣,只是代價更大: * 重寫會把別人在其他段落的編輯一起蓋掉,而議題的編輯紀錄沒有人會去比對。 * * 同名標題出現不只一次時交回 `ambiguous`,不賭第一個——理由與 tickLine 相同, * 而這裡蓋掉的是一整段而不是一行,猜錯的代價更高。 * * 段落不存在時交回 `not-found` 讓呼叫端報錯,不補在結尾:「找不到那一段」多半是段落名 * 打錯,這時把內容塞到議題末尾,比什麼都不做更難收拾。 * * @param {string} body 議題 body * @param {string} section 段落名稱,例如 '目標' * @param {string} content 新的段落內容(不含 `## 標題` 那一行) * @returns {{status: 'replaced'|'not-found'|'ambiguous', body?: string, count: number}} */ export function replaceSection(body, section, content) { const rows = [...eachLine(body)]; const headings = []; for (let i = 0; i < rows.length; i += 1) { if (rows[i].inFence) continue; if (rows[i].line.match(/^##\s+(.+?)\s*$/)?.[1] === section) headings.push(i); } if (headings.length === 0) return { status: 'not-found', count: 0 }; if (headings.length > 1) return { status: 'ambiguous', count: headings.length }; const [start] = headings; const lines = rows.map((row) => row.line); // 下一個段落的標題;沒有就是到結尾 let end = lines.length; for (let i = start + 1; i < lines.length; i += 1) { if (!rows[i].inFence && /^##\s+/.test(lines[i])) { end = i; break; } } // 段落與段落之間的空行屬於版面,不屬於內容:換內容時把它留著。 // 原本就沒有空行(兩個標題緊貼)時補一個,免得新內容黏在下一個標題上。 let tail = end; while (tail > start + 1 && lines[tail - 1].trim() === '') tail -= 1; const spacer = end === lines.length || end > tail ? lines.slice(tail, end) : ['']; // 換行沿用 body 原本的那一種:CRLF 的 body 裡混進 LF,會讓抽取契約交出的 raw // 對不上原文,之後就勾不動那幾行了 const eol = body.includes('\r\n') ? '\r\n' : '\n'; const normalized = content.trim().split(/\r?\n/); return { status: 'replaced', body: [...lines.slice(0, start + 1), '', ...normalized, ...spacer, ...lines.slice(end)] .map((line) => line.replace(/\r$/, '')) .join(eol), count: 1, }; } /** * 在指定段落裡就地更新(或補上)一行「前綴+值」。 * * 用於人天估算這種「議題上只該有一行、重跑要覆蓋而不是累積」的欄位。 * * 三件事都限定在目標段落之內,因為弄錯的代價是靜靜改壞別人的內容: * - 標題要完全相同才算數,`## 關聯度說明` 不是 `## 關聯`。 * - 圍欄裡的假標題不算標題,沿用本檔共同的圍欄判斷。 * - 找既有那一行只在段落範圍內找,別的段落剛好有同前綴時不會被改掉。 * * 段落不存在時補在 body 結尾——寧可放錯位置,也不要讓值靜靜消失。 * * @param {string} body 議題 body * @param {string} section 段落名稱,例如 '關聯' * @param {string} line 完整的一行,例如 '估算人天:3' * @returns {string} 更新後的 body;內容沒有變動時回傳原字串 */ export function upsertLineInSection(body, section, line) { const prefix = line.slice(0, line.indexOf(':') + 1); const rows = [...eachLine(body)]; const { start, end } = sectionBounds(rows, section); if (start === -1) { return `${body.replace(/\n*$/, '')}\n\n${line}\n`; } const text = rows.map((row) => row.line); for (let i = start + 1; i < end; i += 1) { if (rows[i].inFence || !text[i].startsWith(prefix)) continue; if (text[i] === line) return body; text[i] = line; return text.join('\n'); } // 插在段落內容的結尾,跳過段落與段落之間的空行 let insertAt = end; while (insertAt > start + 1 && text[insertAt - 1].trim() === '') insertAt -= 1; text.splice(insertAt, 0, line); return text.join('\n'); } /** 找出段落的起訖行號;start 為標題那一行,end 為下一個標題(或結尾) */ function sectionBounds(rows, section) { let start = -1; for (let i = 0; i < rows.length; i += 1) { if (rows[i].inFence) continue; const heading = rows[i].line.match(/^##\s+(.+?)\s*$/); if (!heading) continue; if (start === -1) { if (heading[1] === section) start = i; continue; } return { start, end: i }; } return { start, end: rows.length }; }