From 2e8712126fb86621fdccd83b0fb6fdd95452c63b Mon Sep 17 00:00:00 2001 From: Jeffery Date: Wed, 2 Sep 2026 17:22:06 +0800 Subject: [PATCH] =?UTF-8?q?refactor(wiki-contents):=20=E7=9B=AE=E9=8C=84?= =?UTF-8?q?=E9=A0=81=E6=94=B9=E6=88=90=20H2=20=E5=8D=80=E5=A1=8A=20upsert?= =?UTF-8?q?=EF=BC=8C=E8=88=8A=E8=A1=A8=E6=A0=BC=E8=87=AA=E5=8B=95=E8=BD=89?= =?UTF-8?q?=E6=AA=94?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 目錄頁的一筆紀錄從 markdown 表格的一列,改成一個 H2 區塊:標題就是這一筆 對應內容頁的頁名,欄位變成標題底下一層的條列「- {欄位名}:{值}」。upsert 換掉標題相同那一塊,找不到就附加到頁尾。 表格一列擠著所有欄位,欄位一多就超出可讀寬度,得橫向捲才看得完;換行之後 也分不出哪幾格屬於同一筆。條列沒有寬度上限,一筆看得完整。 讀到的舊頁還是表格,就先整頁轉成區塊再在轉好的頁面上做 upsert,一頁同時 有表格與區塊也照樣接得起來。轉檔取標題只看持有身分那一欄:有連結取網址 最後一段路徑並解掉百分號編碼,沒連結取純文字——連結的顯示文字常常是計畫 或工作包名稱而不是頁名,拿它當標題會跟呼叫端傳進來的鍵對不上,同一筆長出 第二個區塊,舊的那塊從此再也更新不到。那一欄取不出鍵就整支擋下來,不猜 標題。轉檔那一次有給範本,就連 H1 與引言一起換成範本那一份,舊引言否則 會一直講「一列一筆」;頁面已是條列時只更新自己那一筆,不動引言。組頁邏輯 另外拆出 format 子命令,離線驗證才叫得到,不必打 API。 功能範圍:目錄頁版面改版,涵蓋所有以 _CONTENTS 結尾的頁面與每一支寫目錄 頁的技能。 --- tools/wiki-contents.sh | 365 ++++++++++++++++++++++++++++++----------- 1 file changed, 272 insertions(+), 93 deletions(-) diff --git a/tools/wiki-contents.sh b/tools/wiki-contents.sh index b352a7a..23abed9 100755 --- a/tools/wiki-contents.sh +++ b/tools/wiki-contents.sh @@ -1,25 +1,48 @@ #!/usr/bin/env sh -# wiki-contents.sh — 目錄頁(*_CONTENTS)的表格列 upsert。 +# wiki-contents.sh — 目錄頁(*_CONTENTS)的區塊 upsert。 # -# 為什麼要有這支腳本:目錄頁的「找同一列就取代、找不到就附加」原本靠模型照 +# 為什麼要有這支腳本:目錄頁的「找同一筆就取代、找不到就附加」原本靠模型照 # SKILL.md 手工做,十四個目錄頁只有一處寫成程式。同一段判斷做十四次,錯一次 # 就少一筆紀錄。抽成一支,讀舊頁、比對鍵、整頁寫回只有一種做法。 # +# 版面:一筆紀錄一個 H2 區塊。H2 標題就是這一筆的鍵,寫成對應內容頁的頁名;欄位是 +# 標題底下一層條列,一行一條「- {欄位名}:{值}」。目錄頁上不留 markdown 表格。 +# # 用法: -# wiki-contents.sh upsert [template-file] +# wiki-contents.sh upsert [template-file] # TYPE 頁型,決定頁名 {TYPE}_CONTENTS -# key-col 鍵在表格第幾欄,1 起算 -# key 鍵值,用來找既有列 -# row-file 整列 markdown 表格列的檔案 -# template-file 選用。頁不存在時用它建新頁 +# key-col 只有舊頁還是表格時才用得到:舊表格裡持有這一筆身分的欄位序號, +# 1 起算。轉檔時該欄格子有連結就取網址最後一段路徑當 H2 標題, +# 沒有連結才取格子純文字。頁面已經是條列格式時完全忽略這個參數 +# key 這一筆的 H2 標題文字,也就是內容頁頁名。用來找既有區塊 +# entry-file 整個 H2 區塊的 markdown:「## {key}」那一行、空行、各條條列 +# template-file 選用。頁不存在時用它建新頁;舊頁還是表格而要自動轉檔時, +# 也用它的 H1 與「>」引言取代舊頁那一份 +# +# wiki-contents.sh format [template-file|--fresh] +# 只做文字處理,不碰 API,把結果寫進 new-file 並印出 updated 或 added。 +# 轉檔與 upsert 的判斷只有這一份,離線驗證餵檔案給它就好,不必打 API。 +# 第六個參數給 --fresh 代表 old-file 是範本,要剝掉示範資料;給檔案路徑 +# 則等同 upsert 的 template-file,只在轉檔那一次用來換掉引言。 # # 規則: # 目錄頁一律住在 CONTENTS 專用存取庫,所以存取庫走 wiki-repo CONTENTS, # 不走各自的頁型。 +# 舊頁還是 markdown 表格時,先整頁轉成 H2 區塊再做 upsert;一頁同時有表格與區塊, +# 表格轉出來的區塊接在既有區塊後面。三種舊頁狀態都不得毀掉別人那一筆。 +# 轉檔取 H2 標題只看身分欄那一格:有連結就取網址最後一段路徑,網址經過百分號編碼 +# 就先解碼;沒有連結就取純文字,去掉反引號與頭尾空白。取到什麼就用什麼,不驗頁名樣式。 +# 找「## {key}」:標題文字去頭尾空白後完全相等才算命中。命中就換掉整塊,從那一行 +# 到下一個「## 」之前或檔尾;沒命中就附加到最後一個區塊之後。 # 只有 wiki-get 回 4 才准建新頁。回 7 或 8 一律中止:把金鑰失效讀成 # 「頁面不存在」,就會拿新範本蓋掉活著的頁,舊紀錄整份沒了。 # 這條規則的正本在 skills/wiki/SKILL.md 的 Rules 第 4 條。 -# 結束碼: 0=已更新或已新增 1=寫入失敗 2=用法錯誤 3=CONTENTS 存取庫未設定 +# 建新頁時剝掉範本的示範資料,只留 H1 與「>」引言。 +# 轉檔那一次若呼叫端給了範本,連引言一起換成範本那一份:轉檔只搬表格不動散文, +# 舊引言會一直講「每個存取庫一列」這種只對表格成立的話,誤導之後讀的人。 +# 頁面已經是條列格式、不需要轉檔時引言原樣不動,那時呼叫端只是更新自己那一筆, +# 沒有理由改別人寫的散文;沒給範本也保留舊引言,因為沒有正本可換。 +# 結束碼: 0=已更新或已新增 1=組不出頁面內容或寫入失敗 2=用法錯誤 3=CONTENTS 存取庫未設定 # 4=頁不存在且沒給範本 7=金鑰失效或權限不足 8=其他 API 失敗 set -eu @@ -28,18 +51,253 @@ gitea="$dir/gitea.sh" page_name="$dir/page-name.sh" usage() { - echo 'usage: wiki-contents.sh upsert [template-file]' >&2 + echo 'usage: wiki-contents.sh upsert [template-file]' >&2 + echo ' wiki-contents.sh format [template-file|--fresh]' >&2 exit 2 } -[ "${1-}" = upsert ] || usage +check_keycol() { + case "$1" in + ''|*[!0-9]*) echo "key-col must be a positive integer: $1" >&2; exit 2 ;; + esac + [ "$1" -ge 1 ] || { echo "key-col must be a positive integer: $1" >&2; exit 2; } +} + +# 轉檔與 upsert 只有這一份實作。upsert 與 format 共用它,離線驗證跑的就是正式路徑那一段。 +# 參數:舊頁檔 區塊檔 鍵欄序號 鍵 輸出檔 是否為新建(1 或 0) 範本檔(沒有就給空字串) +render() { + python3 - "$1" "$2" "$3" "$4" "$5" "$6" "$7" <<'PY' +import re +import sys +from urllib.parse import unquote + +old_path, entry_path, keycol, key, new_path, fresh, template_path = sys.argv[1:8] +keycol = int(keycol) +fresh = fresh == '1' +key = key.strip() + +lines = open(old_path, encoding='utf-8').read().split('\n') +entry = open(entry_path, encoding='utf-8').read().strip('\n') + + +def cells(line): + s = line.strip() + if not s.startswith('|'): + return None + s = s[1:] + if s.endswith('|'): + s = s[:-1] + return [c.strip() for c in s.split('|')] + + +def is_sep(cs): + return bool(cs) and all(c and set(c) <= set('-: ') for c in cs) + + +def plain(v): + # 欄名只留文字。留著連結語法或反引號,條列的欄位名就跟頁面上寫的不一樣。 + v = re.sub(r'\[\[([^\]|]*)\|([^\]]*)\]\]', r'\1', v) + v = re.sub(r'\[\[([^\]]*)\]\]', r'\1', v) + v = re.sub(r'\[([^\]]*)\]\([^)]*\)', r'\1', v) + return v.replace('`', '').strip() + + +def last_segment(url): + """取網址最後一段路徑,也就是 .../wiki/{頁名} 的頁名。""" + u = url.strip().split('#', 1)[0].split('?', 1)[0] + parts = [p for p in u.split('/') if p] + seg = parts[-1] if parts else '' + # 頁名有空白或中文時網址會被百分號編碼,解碼後才是頁面上看到的頁名。 + return unquote(seg).replace('`', '').strip() + + +def title_from_cell(v): + # 身分欄的連結文字常常不是頁名,是工作包名稱或計畫名稱;真正的頁名在網址最後一段。 + # 拿連結文字當 H2 標題,就跟呼叫端傳進來的鍵對不上,同一筆會長出第二個區塊。 + m = re.search(r'\[[^\]]*\]\(([^)]*)\)', v) + if m: + return last_segment(m.group(1)) + # wiki 連結 [[頁名|文字]] 的目標寫在前半段,那一段就是頁名。 + m = re.search(r'\[\[([^\]|]*)(?:\|[^\]]*)?\]\]', v) + if m: + return m.group(1).replace('`', '').strip() + # 沒有連結就是純文字身分欄,例如 {owner}/{repo}。取到什麼就用什麼,不判形狀。 + return v.replace('`', '').strip() + + +def split_tables(src): + """把每一段 markdown 表格從行清單裡拿掉。回傳剩下的行與各表格的列。""" + rest = [] + tables = [] + i = 0 + fence = False + while i < len(src): + line = src[i] + # 程式碼圍欄裡的「|」是內容不是表格。mermaid 圖與範例被當表格拆掉,引言就毀了。 + if line.lstrip().startswith('```'): + fence = not fence + rest.append(line) + i += 1 + continue + if not fence and cells(line) is not None: + j = i + while j < len(src) and cells(src[j]) is not None: + j += 1 + rows = [cells(x) for x in src[i:j]] + if len(rows) >= 2 and is_sep(rows[1]): + tables.append(rows) + else: + # 沒有分隔列就不是表格,原樣留著。 + rest.extend(src[i:j]) + i = j + continue + rest.append(line) + i += 1 + return rest, tables + + +def table_blocks(rows): + """一列一個 H2 區塊,欄位順序照表頭從左到右。""" + head = rows[0] + out = [] + for cs in rows[2:]: + if is_sep(cs): + continue + if not any(c for c in cs): + continue + title = title_from_cell(cs[keycol - 1]) if len(cs) >= keycol else '' + if not title: + # 取不出身分就不猜標題。猜錯的標題比不到任何鍵,之後每次 upsert 都在它旁邊 + # 再長一筆;停下來讓人看那一列,比留一筆對不上的紀錄安全。 + sys.stderr.write( + '[jsc][gitea][ERR]:表格有一列取不出第 %d 欄的鍵,轉不成區塊。\n' % keycol) + raise SystemExit(1) + body = [] + for n, name in enumerate(head): + label = plain(name) or ('欄位%d' % (n + 1)) + value = cs[n].strip() if n < len(cs) else '' + body.append('- %s:%s' % (label, value)) + out.append('## %s\n\n%s' % (title, '\n'.join(body))) + return out + + +def split_blocks(src): + """切成引言與各 H2 區塊。第一個「## 」之前的都是引言。""" + pre = [] + blocks = [] + cur = None + fence = False + for line in src: + if line.lstrip().startswith('```'): + fence = not fence + if not fence and line.startswith('## '): + cur = [line] + blocks.append(cur) + continue + (cur if cur is not None else pre).append(line) + return pre, ['\n'.join(b).strip('\n') for b in blocks] + + +def preamble(path): + """取一份檔案第一個「## 」之前的內容,也就是 H1 加「>」引言那一段。""" + src = open(path, encoding='utf-8').read().split('\n') + # 先拆掉表格:範本若在引言之前放了示範表格,照搬進去就等於在目錄頁上留下表格。 + head, _ = split_blocks(split_tables(src)[0]) + return head + + +rest, tables = split_tables(lines) +pre, blocks = split_blocks(rest) + +# 範本的示範區塊會被當成真的一筆。照抄進新頁,那一筆就永遠留著,之後每次 upsert 都 +# 比不到它的鍵而跳過,正式頁上多出一筆指向不存在的頁的死紀錄。所以建新頁只留引言。 +if fresh: + blocks = [] +else: + # 有表格就代表這一頁還是舊版面,這一次要轉檔。 + converting = bool(tables) + for rows in tables: + blocks.extend(table_blocks(rows)) + # 轉檔只搬表格、不動散文,舊引言就會一直講只對表格成立的話。範本的引言是正本, + # 轉檔正好是換掉它的時機。不轉檔就不動引言:那時呼叫端只是更新自己那一筆。 + if converting and template_path: + tpl_pre = preamble(template_path) + # 範本沒有引言時保留舊的,換成空白等於把 H1 也弄掉。 + if '\n'.join(tpl_pre).strip(): + pre = tpl_pre + +# 鍵就是標題,所以標題一律重寫成 key。兩者不一致的話,這一筆下一次就找不回來。 +body = entry.split('\n') +if body and body[0].startswith('## '): + body = body[1:] +while body and not body[0].strip(): + body = body[1:] +block = '## %s' % key +if body: + block += '\n\n' + '\n'.join(body).strip('\n') + +hit = -1 +for i, b in enumerate(blocks): + if b.split('\n', 1)[0][3:].strip() == key: + hit = i + break + +if hit >= 0: + blocks[hit] = block + action = 'updated' +else: + blocks.append(block) + action = 'added' + +head = '\n'.join(pre).strip('\n') +parts = ([head] if head else []) + blocks +text = '\n\n'.join(parts) +if not text.strip(): + sys.stderr.write('[jsc][gitea][ERR]:組不出頁面內容,不寫入。\n') + raise SystemExit(1) + +open(new_path, 'w', encoding='utf-8').write(text + '\n') +print(action) +PY +} + +cmd="${1-}" +[ "$cmd" = upsert ] || [ "$cmd" = format ] || usage shift + +if [ "$cmd" = format ]; then + [ "$#" -ge 5 ] && [ "$#" -le 6 ] || usage + keycol="$1" + key="$2" + entryfile="$3" + oldfile="$4" + newfile="$5" + fresh=0 + template='' + if [ "$#" -eq 6 ]; then + if [ "$6" = --fresh ]; then + fresh=1 + else + template="$6" + fi + fi + check_keycol "$keycol" + [ -n "$key" ] || { echo 'key required' >&2; exit 2; } + [ -f "$entryfile" ] || { echo "找不到區塊檔案: $entryfile" >&2; exit 2; } + [ -f "$oldfile" ] || { echo "找不到舊頁檔案: $oldfile" >&2; exit 2; } + [ -z "$template" ] || [ -f "$template" ] || { echo "找不到範本檔: $template" >&2; exit 2; } + rc=0 + render "$oldfile" "$entryfile" "$keycol" "$key" "$newfile" "$fresh" "$template" || rc=$? + [ "$rc" -eq 0 ] || exit 1 + exit 0 +fi + [ "$#" -ge 4 ] && [ "$#" -le 5 ] || usage type=$(printf '%s' "${1-}" | tr a-z A-Z) keycol="${2-}" key="${3-}" -rowfile="${4-}" +entryfile="${4-}" template="${5-}" [ -n "$type" ] || usage @@ -48,12 +306,9 @@ page="${type}_CONTENTS" # 它連 CONTENTS 一起擋掉——CONTENTS 只用來解存取庫,沒有 CONTENTS_CONTENTS 這一頁。 sh "$page_name" check "$page" >/dev/null 2>&1 || { echo "不能用來組目錄頁頁名的頁型: $type" >&2; usage; } -case "$keycol" in - ''|*[!0-9]*) echo "key-col must be a positive integer: $keycol" >&2; exit 2 ;; -esac -[ "$keycol" -ge 1 ] || { echo "key-col must be a positive integer: $keycol" >&2; exit 2; } +check_keycol "$keycol" [ -n "$key" ] || { echo 'key required' >&2; exit 2; } -[ -f "$rowfile" ] || { echo "找不到列檔案: $rowfile" >&2; exit 2; } +[ -f "$entryfile" ] || { echo "找不到區塊檔案: $entryfile" >&2; exit 2; } [ -z "$template" ] || [ -f "$template" ] || { echo "找不到範本檔: $template" >&2; exit 2; } # 存取庫解析失敗照原碼傳出去:3 是「沒設定」,2 是型別不認得,兩者處置不同。 @@ -88,83 +343,7 @@ case "$rc" in esac rc=0 -action=$(python3 - "$old" "$rowfile" "$keycol" "$key" "$new" "$fresh" <<'PY' -import sys - -old_path, row_path, keycol, key, new_path, fresh = sys.argv[1:7] -keycol = int(keycol) -fresh = fresh == '1' - -lines = open(old_path, encoding='utf-8').read().split('\n') -row = open(row_path, encoding='utf-8').read().strip('\n') - - -def cells(line): - s = line.strip() - if not s.startswith('|'): - return None - s = s[1:] - if s.endswith('|'): - s = s[:-1] - return [c.strip() for c in s.split('|')] - - -def is_sep(cs): - return bool(cs) and all(c and set(c) <= set('-: ') for c in cs) - - -# 範本表格的示範列落在分隔列之後,會被當成真的資料列。照抄進新頁,那一列就永遠留著, -# 之後每次 upsert 都比不到它的鍵而跳過,正式頁上多出一條指向不存在的頁的死連結。 -# 所以建新頁時剝掉分隔列之後的所有資料列,只留標題、說明、表頭與分隔列。 -if fresh: - kept = [] - passed_sep = False - for line in lines: - cs = cells(line) - if cs is None: - kept.append(line) - continue - if is_sep(cs): - passed_sep = True - kept.append(line) - continue - if passed_sep: - continue - kept.append(line) - lines = kept - -# 分隔列之前的都是表頭。從分隔列之後才開始比對鍵,表頭第一欄剛好等於鍵時才不會被改掉。 -seen_sep = False -hit = -1 -last_row = -1 -for i, line in enumerate(lines): - cs = cells(line) - if cs is None: - continue - if is_sep(cs): - seen_sep = True - last_row = i - continue - if not seen_sep: - continue - last_row = i - if hit < 0 and len(cs) >= keycol and cs[keycol - 1] == key: - hit = i - -if hit >= 0: - lines[hit] = row - action = 'updated' -elif last_row >= 0: - lines.insert(last_row + 1, row) - action = 'added' -else: - sys.stderr.write('[jsc][gitea][ERR]:頁面裡找不到 markdown 表格,無處可放這一列。\n') - raise SystemExit(1) - -open(new_path, 'w', encoding='utf-8').write('\n'.join(lines)) -print(action) -PY -) || rc=$? +action=$(render "$old" "$entryfile" "$keycol" "$key" "$new" "$fresh" "$template") || rc=$? # 整不出正確的頁就不要送出去。組不出內容跟送出失敗一樣寫不進去,共用結束碼 1。 [ "$rc" -eq 0 ] || exit 1