瀏覽器送出 textarea 一律用 CRLF,議題只要在 Gitea 網頁上被編輯過,body 逐行切開後 每一行行尾就掛著 \r。JS 的 . 不吃 \r,`(.*)$` 因此整行比不中——listSection 會把 目標、非目標、驗收標準這些段落一律回成空陣列,而且不報錯,下游拿到的是「這一段沒寫」 而不是「解析失敗」。 改用 [\s\S] 比對行尾,text 本來就有 trim 會把 \r 修掉。 Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
205 lines
7.0 KiB
JavaScript
205 lines
7.0 KiB
JavaScript
/**
|
||
* 議題 body 的 markdown 解析。
|
||
*
|
||
* 純函式,不碰網路也不碰檔案系統:抽取類腳本的解析全部走這裡,
|
||
* 解析規則只寫一次,需求議題與工作包議題共用同一套。
|
||
*
|
||
* 全檔的共同前提是「圍欄裡的東西不是內容」:議題裡放 mermaid 或程式碼是常態,
|
||
* 那裡面的井字號不是標題、減號不是清單項、直線不是表格。這件事只在 eachLine
|
||
* 處理一次,其餘函式都靠它。
|
||
*/
|
||
|
||
/**
|
||
* 逐行走過內容並標註圍欄狀態。
|
||
* 圍欄以 ``` 或 ~~~ 開啟,且要同一種標記才算關閉——混用時後者只是普通文字。
|
||
* 圍欄沒關就到結尾時,其後的內容一律算在圍欄內,這與 markdown 的實際渲染一致。
|
||
* @param {string} text
|
||
* @returns {Generator<{line: string, inFence: boolean, isFence: boolean}>}
|
||
*/
|
||
function* eachLine(text) {
|
||
let fence = null;
|
||
|
||
for (const line of (text ?? '').split('\n')) {
|
||
const marker = line.match(/^\s*(`{3,}|~{3,})/)?.[1]?.[0];
|
||
|
||
if (marker && fence === null) {
|
||
fence = marker;
|
||
yield { line, inFence: true, isFence: true };
|
||
continue;
|
||
}
|
||
if (marker && marker === fence) {
|
||
fence = null;
|
||
yield { line, inFence: true, isFence: true };
|
||
continue;
|
||
}
|
||
yield { line, inFence: fence !== null, isFence: false };
|
||
}
|
||
}
|
||
|
||
/**
|
||
* 依 `## 標題` 切出各段落。圍欄內的行原樣保留在段落內容裡
|
||
* ——流程圖那一段要的就是整個 mermaid 區塊。
|
||
* @param {string} body 議題 body
|
||
* @returns {Map<string, string>} 段落名稱 → 該段內容(前後空白已修掉)
|
||
*/
|
||
export function parseSections(body) {
|
||
const sections = new Map();
|
||
const buffer = [];
|
||
let current = null;
|
||
|
||
const flush = () => {
|
||
if (current !== null) sections.set(current, buffer.join('\n').trim());
|
||
buffer.length = 0;
|
||
};
|
||
|
||
for (const { line, inFence } of eachLine(body)) {
|
||
const heading = inFence ? null : line.match(/^##\s+(.+?)\s*$/);
|
||
if (heading) {
|
||
flush();
|
||
current = heading[1];
|
||
continue;
|
||
}
|
||
if (current !== null) buffer.push(line);
|
||
}
|
||
flush();
|
||
|
||
return sections;
|
||
}
|
||
|
||
/**
|
||
* 取出文字型段落的整段內容。段落不存在時回空字串,不報錯 ——
|
||
* 缺段落是模板的正常變體,不是解析失敗。
|
||
* @param {Map<string, string>} sections
|
||
* @param {string} name
|
||
* @returns {string}
|
||
*/
|
||
export function textSection(sections, name) {
|
||
return sections.get(name) ?? '';
|
||
}
|
||
|
||
/**
|
||
* 取出列表型段落的每一項。符號清單與編號清單一視同仁,checkbox 只留文字不留標記
|
||
* ——需求議題的驗收標準不追蹤勾選狀態。巢狀項目一律攤平:需求議題的模板沒有巢狀,
|
||
* 真的出現時寧可多帶一項,也不要無聲吃掉內容。
|
||
* @param {Map<string, string>} sections
|
||
* @param {string} name
|
||
* @returns {string[]}
|
||
*/
|
||
export function listSection(sections, name) {
|
||
const items = [];
|
||
|
||
for (const { line, inFence } of eachLine(sections.get(name))) {
|
||
if (inFence) continue;
|
||
// [\s\S] 而非 . :CRLF 的 body 逐行切開後行尾有 \r,而 . 不吃 \r,
|
||
// 用 . 會讓整行比不中,清單靜靜變成空的。後面的 trim 再把 \r 修掉。
|
||
const item = line.match(/^\s*(?:[-*+]|\d+\.)\s+([\s\S]*)$/);
|
||
if (!item) continue;
|
||
|
||
const text = item[1].replace(/^\[[ xX]\]\s*/, '').trim();
|
||
if (text !== '') items.push(text);
|
||
}
|
||
return items;
|
||
}
|
||
|
||
/**
|
||
* 取出兩欄表格型段落。以分隔列(|---|---|)為界,之後才是資料列;
|
||
* 沒有分隔列就當成沒有資料,避免把表頭當成一筆名詞。
|
||
*
|
||
* 只處理兩欄:多出來的欄會被丟掉。目前唯一的使用者是需求議題的領域名詞表。
|
||
* 工作包的介面契約是四欄(介面/產出者/消費者/形狀),wp-extract 需要另一個
|
||
* 保留全部欄位的版本,不能直接沿用這一支。
|
||
* @param {Map<string, string>} sections
|
||
* @param {string} name
|
||
* @returns {{term: string, def: string}[]}
|
||
*/
|
||
export function tableSection(sections, name) {
|
||
const rows = [];
|
||
|
||
for (const { line, inFence } of eachLine(sections.get(name))) {
|
||
if (inFence || !line.trim().startsWith('|')) continue;
|
||
rows.push(splitRow(line));
|
||
}
|
||
|
||
const separator = rows.findIndex(isSeparator);
|
||
if (separator === -1) return [];
|
||
|
||
return rows
|
||
.slice(separator + 1)
|
||
// 同一段落裡若不慎貼了第二張表,它的分隔列不該變成一筆 {term:'---'}
|
||
.filter((cells) => !isSeparator(cells))
|
||
.filter((cells) => cells.length >= 2 && cells.some((cell) => cell !== ''))
|
||
.map(([term, def]) => ({ term, def }));
|
||
}
|
||
|
||
/** 以未被逸脫的直線切欄,再把 `\|` 還原成內容裡的直線 */
|
||
function splitRow(line) {
|
||
return line
|
||
.trim()
|
||
.replace(/^\||\|$/g, '')
|
||
.split(/(?<!\\)\|/)
|
||
.map((cell) => cell.replace(/\\\|/g, '|').trim());
|
||
}
|
||
|
||
function isSeparator(cells) {
|
||
return cells.length > 0 && cells.every((cell) => /^:?-+:?$/.test(cell));
|
||
}
|
||
|
||
/**
|
||
* 在指定段落裡就地更新(或補上)一行「前綴+值」。
|
||
*
|
||
* 用於人天估算這種「議題上只該有一行、重跑要覆蓋而不是累積」的欄位。
|
||
*
|
||
* 三件事都限定在目標段落之內,因為弄錯的代價是靜靜改壞別人的內容:
|
||
* - 標題要完全相同才算數,`## 關聯度說明` 不是 `## 關聯`。
|
||
* - 圍欄裡的假標題不算標題,沿用本檔共同的圍欄判斷。
|
||
* - 找既有那一行只在段落範圍內找,別的段落剛好有同前綴時不會被改掉。
|
||
*
|
||
* 段落不存在時補在 body 結尾——寧可放錯位置,也不要讓值靜靜消失。
|
||
*
|
||
* @param {string} body 議題 body
|
||
* @param {string} section 段落名稱,例如 '關聯'
|
||
* @param {string} line 完整的一行,例如 '估算人天:3'
|
||
* @returns {string} 更新後的 body;內容沒有變動時回傳原字串
|
||
*/
|
||
export function upsertLineInSection(body, section, line) {
|
||
const prefix = line.slice(0, line.indexOf(':') + 1);
|
||
const rows = [...eachLine(body)];
|
||
const { start, end } = sectionBounds(rows, section);
|
||
|
||
if (start === -1) {
|
||
return `${body.replace(/\n*$/, '')}\n\n${line}\n`;
|
||
}
|
||
|
||
const text = rows.map((row) => row.line);
|
||
for (let i = start + 1; i < end; i += 1) {
|
||
if (rows[i].inFence || !text[i].startsWith(prefix)) continue;
|
||
if (text[i] === line) return body;
|
||
text[i] = line;
|
||
return text.join('\n');
|
||
}
|
||
|
||
// 插在段落內容的結尾,跳過段落與段落之間的空行
|
||
let insertAt = end;
|
||
while (insertAt > start + 1 && text[insertAt - 1].trim() === '') insertAt -= 1;
|
||
text.splice(insertAt, 0, line);
|
||
return text.join('\n');
|
||
}
|
||
|
||
/** 找出段落的起訖行號;start 為標題那一行,end 為下一個標題(或結尾) */
|
||
function sectionBounds(rows, section) {
|
||
let start = -1;
|
||
|
||
for (let i = 0; i < rows.length; i += 1) {
|
||
if (rows[i].inFence) continue;
|
||
const heading = rows[i].line.match(/^##\s+(.+?)\s*$/);
|
||
if (!heading) continue;
|
||
|
||
if (start === -1) {
|
||
if (heading[1] === section) start = i;
|
||
continue;
|
||
}
|
||
return { start, end: i };
|
||
}
|
||
return { start, end: rows.length };
|
||
}
|