wiki 目錄頁改成大標題加條列,舊表格讀到就自動轉檔 #52

Merged
admin merged 5 commits from feat/contents-list/main into develop 2026-09-02 10:00:22 +00:00
Showing only changes of commit 2e8712126f - Show all commits
+272 -93
View File
@@ -1,25 +1,48 @@
#!/usr/bin/env sh
# wiki-contents.sh — 目錄頁(*_CONTENTS)的表格列 upsert。
# wiki-contents.sh — 目錄頁(*_CONTENTS)的區塊 upsert。
#
# 為什麼要有這支腳本:目錄頁的「找同一列就取代、找不到就附加」原本靠模型照
# 為什麼要有這支腳本:目錄頁的「找同一筆就取代、找不到就附加」原本靠模型照
# SKILL.md 手工做,十四個目錄頁只有一處寫成程式。同一段判斷做十四次,錯一次
# 就少一筆紀錄。抽成一支,讀舊頁、比對鍵、整頁寫回只有一種做法。
#
# 版面:一筆紀錄一個 H2 區塊。H2 標題就是這一筆的鍵,寫成對應內容頁的頁名;欄位是
# 標題底下一層條列,一行一條「- {欄位名}:{值}」。目錄頁上不留 markdown 表格。
#
# 用法:
# wiki-contents.sh upsert <TYPE> <key-col> <key> <row-file> [template-file]
# wiki-contents.sh upsert <TYPE> <key-col> <key> <entry-file> [template-file]
# TYPE 頁型,決定頁名 {TYPE}_CONTENTS
# key-col 鍵在表格第幾欄,1 起算
# key 鍵值,用來找既有列
# row-file 整列 markdown 表格列的檔案
# template-file 選用。頁不存在時用它建新頁
# key-col 只有舊頁還是表格時才用得到:舊表格裡持有這一筆身分的欄位序號,
# 1 起算。轉檔時該欄格子有連結就取網址最後一段路徑當 H2 標題,
# 沒有連結才取格子純文字。頁面已經是條列格式時完全忽略這個參數
# key 這一筆的 H2 標題文字,也就是內容頁頁名。用來找既有區塊
# entry-file 整個 H2 區塊的 markdown:「## {key}」那一行、空行、各條條列
# template-file 選用。頁不存在時用它建新頁;舊頁還是表格而要自動轉檔時,
# 也用它的 H1 與「>」引言取代舊頁那一份
#
# wiki-contents.sh format <key-col> <key> <entry-file> <old-file> <new-file> [template-file|--fresh]
# 只做文字處理,不碰 API,把結果寫進 new-file 並印出 updated 或 added。
# 轉檔與 upsert 的判斷只有這一份,離線驗證餵檔案給它就好,不必打 API。
# 第六個參數給 --fresh 代表 old-file 是範本,要剝掉示範資料;給檔案路徑
# 則等同 upsert 的 template-file,只在轉檔那一次用來換掉引言。
#
# 規則:
# 目錄頁一律住在 CONTENTS 專用存取庫,所以存取庫走 wiki-repo CONTENTS,
# 不走各自的頁型。
# 舊頁還是 markdown 表格時,先整頁轉成 H2 區塊再做 upsert;一頁同時有表格與區塊,
# 表格轉出來的區塊接在既有區塊後面。三種舊頁狀態都不得毀掉別人那一筆。
# 轉檔取 H2 標題只看身分欄那一格:有連結就取網址最後一段路徑,網址經過百分號編碼
# 就先解碼;沒有連結就取純文字,去掉反引號與頭尾空白。取到什麼就用什麼,不驗頁名樣式。
# 找「## {key}」:標題文字去頭尾空白後完全相等才算命中。命中就換掉整塊,從那一行
# 到下一個「## 」之前或檔尾;沒命中就附加到最後一個區塊之後。
# 只有 wiki-get 回 4 才准建新頁。回 7 或 8 一律中止:把金鑰失效讀成
# 「頁面不存在」,就會拿新範本蓋掉活著的頁,舊紀錄整份沒了。
# 這條規則的正本在 skills/wiki/SKILL.md 的 Rules 第 4 條。
# 結束碼: 0=已更新或已新增 1=寫入失敗 2=用法錯誤 3=CONTENTS 存取庫未設定
# 建新頁時剝掉範本的示範資料,只留 H1 與「>」引言。
# 轉檔那一次若呼叫端給了範本,連引言一起換成範本那一份:轉檔只搬表格不動散文,
# 舊引言會一直講「每個存取庫一列」這種只對表格成立的話,誤導之後讀的人。
# 頁面已經是條列格式、不需要轉檔時引言原樣不動,那時呼叫端只是更新自己那一筆,
# 沒有理由改別人寫的散文;沒給範本也保留舊引言,因為沒有正本可換。
# 結束碼: 0=已更新或已新增 1=組不出頁面內容或寫入失敗 2=用法錯誤 3=CONTENTS 存取庫未設定
# 4=頁不存在且沒給範本 7=金鑰失效或權限不足 8=其他 API 失敗
set -eu
@@ -28,18 +51,253 @@ gitea="$dir/gitea.sh"
page_name="$dir/page-name.sh"
usage() {
echo 'usage: wiki-contents.sh upsert <TYPE> <key-col> <key> <row-file> [template-file]' >&2
echo 'usage: wiki-contents.sh upsert <TYPE> <key-col> <key> <entry-file> [template-file]' >&2
echo ' wiki-contents.sh format <key-col> <key> <entry-file> <old-file> <new-file> [template-file|--fresh]' >&2
exit 2
}
[ "${1-}" = upsert ] || usage
check_keycol() {
case "$1" in
''|*[!0-9]*) echo "key-col must be a positive integer: $1" >&2; exit 2 ;;
esac
[ "$1" -ge 1 ] || { echo "key-col must be a positive integer: $1" >&2; exit 2; }
}
# 轉檔與 upsert 只有這一份實作。upsert 與 format 共用它,離線驗證跑的就是正式路徑那一段。
# 參數:舊頁檔 區塊檔 鍵欄序號 鍵 輸出檔 是否為新建(1 或 0) 範本檔(沒有就給空字串)
render() {
python3 - "$1" "$2" "$3" "$4" "$5" "$6" "$7" <<'PY'
import re
import sys
from urllib.parse import unquote
old_path, entry_path, keycol, key, new_path, fresh, template_path = sys.argv[1:8]
keycol = int(keycol)
fresh = fresh == '1'
key = key.strip()
lines = open(old_path, encoding='utf-8').read().split('\n')
entry = open(entry_path, encoding='utf-8').read().strip('\n')
def cells(line):
s = line.strip()
if not s.startswith('|'):
return None
s = s[1:]
if s.endswith('|'):
s = s[:-1]
return [c.strip() for c in s.split('|')]
def is_sep(cs):
return bool(cs) and all(c and set(c) <= set('-: ') for c in cs)
def plain(v):
# 欄名只留文字。留著連結語法或反引號,條列的欄位名就跟頁面上寫的不一樣。
v = re.sub(r'\[\[([^\]|]*)\|([^\]]*)\]\]', r'\1', v)
v = re.sub(r'\[\[([^\]]*)\]\]', r'\1', v)
v = re.sub(r'\[([^\]]*)\]\([^)]*\)', r'\1', v)
return v.replace('`', '').strip()
def last_segment(url):
"""取網址最後一段路徑,也就是 .../wiki/{頁名} 的頁名。"""
u = url.strip().split('#', 1)[0].split('?', 1)[0]
parts = [p for p in u.split('/') if p]
seg = parts[-1] if parts else ''
# 頁名有空白或中文時網址會被百分號編碼,解碼後才是頁面上看到的頁名。
return unquote(seg).replace('`', '').strip()
def title_from_cell(v):
# 身分欄的連結文字常常不是頁名,是工作包名稱或計畫名稱;真正的頁名在網址最後一段。
# 拿連結文字當 H2 標題,就跟呼叫端傳進來的鍵對不上,同一筆會長出第二個區塊。
m = re.search(r'\[[^\]]*\]\(([^)]*)\)', v)
if m:
return last_segment(m.group(1))
# wiki 連結 [[頁名|文字]] 的目標寫在前半段,那一段就是頁名。
m = re.search(r'\[\[([^\]|]*)(?:\|[^\]]*)?\]\]', v)
if m:
return m.group(1).replace('`', '').strip()
# 沒有連結就是純文字身分欄,例如 {owner}/{repo}。取到什麼就用什麼,不判形狀。
return v.replace('`', '').strip()
def split_tables(src):
"""把每一段 markdown 表格從行清單裡拿掉。回傳剩下的行與各表格的列。"""
rest = []
tables = []
i = 0
fence = False
while i < len(src):
line = src[i]
# 程式碼圍欄裡的「|」是內容不是表格。mermaid 圖與範例被當表格拆掉,引言就毀了。
if line.lstrip().startswith('```'):
fence = not fence
rest.append(line)
i += 1
continue
if not fence and cells(line) is not None:
j = i
while j < len(src) and cells(src[j]) is not None:
j += 1
rows = [cells(x) for x in src[i:j]]
if len(rows) >= 2 and is_sep(rows[1]):
tables.append(rows)
else:
# 沒有分隔列就不是表格,原樣留著。
rest.extend(src[i:j])
i = j
continue
rest.append(line)
i += 1
return rest, tables
def table_blocks(rows):
"""一列一個 H2 區塊,欄位順序照表頭從左到右。"""
head = rows[0]
out = []
for cs in rows[2:]:
if is_sep(cs):
continue
if not any(c for c in cs):
continue
title = title_from_cell(cs[keycol - 1]) if len(cs) >= keycol else ''
if not title:
# 取不出身分就不猜標題。猜錯的標題比不到任何鍵,之後每次 upsert 都在它旁邊
# 再長一筆;停下來讓人看那一列,比留一筆對不上的紀錄安全。
sys.stderr.write(
'[jsc][gitea][ERR]:表格有一列取不出第 %d 欄的鍵,轉不成區塊。\n' % keycol)
raise SystemExit(1)
body = []
for n, name in enumerate(head):
label = plain(name) or ('欄位%d' % (n + 1))
value = cs[n].strip() if n < len(cs) else ''
body.append('- %s:%s' % (label, value))
out.append('## %s\n\n%s' % (title, '\n'.join(body)))
return out
def split_blocks(src):
"""切成引言與各 H2 區塊。第一個「## 」之前的都是引言。"""
pre = []
blocks = []
cur = None
fence = False
for line in src:
if line.lstrip().startswith('```'):
fence = not fence
if not fence and line.startswith('## '):
cur = [line]
blocks.append(cur)
continue
(cur if cur is not None else pre).append(line)
return pre, ['\n'.join(b).strip('\n') for b in blocks]
def preamble(path):
"""取一份檔案第一個「## 」之前的內容,也就是 H1 加「>」引言那一段。"""
src = open(path, encoding='utf-8').read().split('\n')
# 先拆掉表格:範本若在引言之前放了示範表格,照搬進去就等於在目錄頁上留下表格。
head, _ = split_blocks(split_tables(src)[0])
return head
rest, tables = split_tables(lines)
pre, blocks = split_blocks(rest)
# 範本的示範區塊會被當成真的一筆。照抄進新頁,那一筆就永遠留著,之後每次 upsert 都
# 比不到它的鍵而跳過,正式頁上多出一筆指向不存在的頁的死紀錄。所以建新頁只留引言。
if fresh:
blocks = []
else:
# 有表格就代表這一頁還是舊版面,這一次要轉檔。
converting = bool(tables)
for rows in tables:
blocks.extend(table_blocks(rows))
# 轉檔只搬表格、不動散文,舊引言就會一直講只對表格成立的話。範本的引言是正本,
# 轉檔正好是換掉它的時機。不轉檔就不動引言:那時呼叫端只是更新自己那一筆。
if converting and template_path:
tpl_pre = preamble(template_path)
# 範本沒有引言時保留舊的,換成空白等於把 H1 也弄掉。
if '\n'.join(tpl_pre).strip():
pre = tpl_pre
# 鍵就是標題,所以標題一律重寫成 key。兩者不一致的話,這一筆下一次就找不回來。
body = entry.split('\n')
if body and body[0].startswith('## '):
body = body[1:]
while body and not body[0].strip():
body = body[1:]
block = '## %s' % key
if body:
block += '\n\n' + '\n'.join(body).strip('\n')
hit = -1
for i, b in enumerate(blocks):
if b.split('\n', 1)[0][3:].strip() == key:
hit = i
break
if hit >= 0:
blocks[hit] = block
action = 'updated'
else:
blocks.append(block)
action = 'added'
head = '\n'.join(pre).strip('\n')
parts = ([head] if head else []) + blocks
text = '\n\n'.join(parts)
if not text.strip():
sys.stderr.write('[jsc][gitea][ERR]:組不出頁面內容,不寫入。\n')
raise SystemExit(1)
open(new_path, 'w', encoding='utf-8').write(text + '\n')
print(action)
PY
}
cmd="${1-}"
[ "$cmd" = upsert ] || [ "$cmd" = format ] || usage
shift
if [ "$cmd" = format ]; then
[ "$#" -ge 5 ] && [ "$#" -le 6 ] || usage
keycol="$1"
key="$2"
entryfile="$3"
oldfile="$4"
newfile="$5"
fresh=0
template=''
if [ "$#" -eq 6 ]; then
if [ "$6" = --fresh ]; then
fresh=1
else
template="$6"
fi
fi
check_keycol "$keycol"
[ -n "$key" ] || { echo 'key required' >&2; exit 2; }
[ -f "$entryfile" ] || { echo "找不到區塊檔案: $entryfile" >&2; exit 2; }
[ -f "$oldfile" ] || { echo "找不到舊頁檔案: $oldfile" >&2; exit 2; }
[ -z "$template" ] || [ -f "$template" ] || { echo "找不到範本檔: $template" >&2; exit 2; }
rc=0
render "$oldfile" "$entryfile" "$keycol" "$key" "$newfile" "$fresh" "$template" || rc=$?
[ "$rc" -eq 0 ] || exit 1
exit 0
fi
[ "$#" -ge 4 ] && [ "$#" -le 5 ] || usage
type=$(printf '%s' "${1-}" | tr a-z A-Z)
keycol="${2-}"
key="${3-}"
rowfile="${4-}"
entryfile="${4-}"
template="${5-}"
[ -n "$type" ] || usage
@@ -48,12 +306,9 @@ page="${type}_CONTENTS"
# 它連 CONTENTS 一起擋掉——CONTENTS 只用來解存取庫,沒有 CONTENTS_CONTENTS 這一頁。
sh "$page_name" check "$page" >/dev/null 2>&1 || { echo "不能用來組目錄頁頁名的頁型: $type" >&2; usage; }
case "$keycol" in
''|*[!0-9]*) echo "key-col must be a positive integer: $keycol" >&2; exit 2 ;;
esac
[ "$keycol" -ge 1 ] || { echo "key-col must be a positive integer: $keycol" >&2; exit 2; }
check_keycol "$keycol"
[ -n "$key" ] || { echo 'key required' >&2; exit 2; }
[ -f "$rowfile" ] || { echo "找不到列檔案: $rowfile" >&2; exit 2; }
[ -f "$entryfile" ] || { echo "找不到區塊檔案: $entryfile" >&2; exit 2; }
[ -z "$template" ] || [ -f "$template" ] || { echo "找不到範本檔: $template" >&2; exit 2; }
# 存取庫解析失敗照原碼傳出去:3 是「沒設定」,2 是型別不認得,兩者處置不同。
@@ -88,83 +343,7 @@ case "$rc" in
esac
rc=0
action=$(python3 - "$old" "$rowfile" "$keycol" "$key" "$new" "$fresh" <<'PY'
import sys
old_path, row_path, keycol, key, new_path, fresh = sys.argv[1:7]
keycol = int(keycol)
fresh = fresh == '1'
lines = open(old_path, encoding='utf-8').read().split('\n')
row = open(row_path, encoding='utf-8').read().strip('\n')
def cells(line):
s = line.strip()
if not s.startswith('|'):
return None
s = s[1:]
if s.endswith('|'):
s = s[:-1]
return [c.strip() for c in s.split('|')]
def is_sep(cs):
return bool(cs) and all(c and set(c) <= set('-: ') for c in cs)
# 範本表格的示範列落在分隔列之後,會被當成真的資料列。照抄進新頁,那一列就永遠留著,
# 之後每次 upsert 都比不到它的鍵而跳過,正式頁上多出一條指向不存在的頁的死連結。
# 所以建新頁時剝掉分隔列之後的所有資料列,只留標題、說明、表頭與分隔列。
if fresh:
kept = []
passed_sep = False
for line in lines:
cs = cells(line)
if cs is None:
kept.append(line)
continue
if is_sep(cs):
passed_sep = True
kept.append(line)
continue
if passed_sep:
continue
kept.append(line)
lines = kept
# 分隔列之前的都是表頭。從分隔列之後才開始比對鍵,表頭第一欄剛好等於鍵時才不會被改掉。
seen_sep = False
hit = -1
last_row = -1
for i, line in enumerate(lines):
cs = cells(line)
if cs is None:
continue
if is_sep(cs):
seen_sep = True
last_row = i
continue
if not seen_sep:
continue
last_row = i
if hit < 0 and len(cs) >= keycol and cs[keycol - 1] == key:
hit = i
if hit >= 0:
lines[hit] = row
action = 'updated'
elif last_row >= 0:
lines.insert(last_row + 1, row)
action = 'added'
else:
sys.stderr.write('[jsc][gitea][ERR]:頁面裡找不到 markdown 表格,無處可放這一列。\n')
raise SystemExit(1)
open(new_path, 'w', encoding='utf-8').write('\n'.join(lines))
print(action)
PY
) || rc=$?
action=$(render "$old" "$entryfile" "$keycol" "$key" "$new" "$fresh" "$template") || rc=$?
# 整不出正確的頁就不要送出去。組不出內容跟送出失敗一樣寫不進去,共用結束碼 1。
[ "$rc" -eq 0 ] || exit 1