Files
wiwi-blog-txt/txt.sh
2026-08-19 01:37:55 +00:00

163 lines
5.8 KiB
Bash
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/bin/bash
# ==============================================================
# txt.sh
#
# 把 Docusaurus build 出來的 HTML轉換成純文字檔.txt
#
# 運作方式:
# 1. 掃描 build 底下所有的 index.html
# 2. 排除非文章頁面tags、archive、page分頁、以及網站首頁
# 3. 用 BeautifulSoup 抓出 <article> 內容,過濾掉側邊欄
# 4. 移除 Docusaurus 自己塞在 <article> 裡的介面元素
# (麵包屑、本頁導覽、編輯此頁⋯⋯只有 /docs 會有這些)
# 5. 幫 h1~h4 標題加上裝飾符號,保留標題層級感
# 6. 用 w3m -dump 轉成純文字,正確處理中英文寬度
# 7. 輸出檔名採用文章 slug資料夾名稱
#
# 使用 xargs -P 平行處理,加快多篇文章的轉換速度
#
# 前置需求python3、beautifulsoup4pip install beautifulsoup4、w3m
# ==============================================================
cd "$(dirname "$0")"
SRC="/path/to/build"
DST="/path/to/build"
BASE_URL="https://wiwi.blog" # 相對連結會補上這個網域
mkdir -p "$DST"
export SRC DST BASE_URL
process_one() {
file="$1"
rel="${file#$SRC/}"
slug=$(dirname "$rel")
out="$DST/$slug.txt"
tmp=$(mktemp)
# 注意python 程式碼用單引號包住,檔名透過 sys.argv 傳進去,
# 這樣路徑裡有引號或空白也不會出事
python3 -c '
import os, sys
from bs4 import BeautifulSoup
soup = BeautifulSoup(open(sys.argv[1], encoding="utf-8"), "html.parser")
article = soup.find("article")
if not article:
article = soup
# Docusaurus 會把這些介面元素塞在 <article> 裡面,轉成純文字後很礙眼。
# 主要是 /docs 頁面才有;不想拿掉哪一個,把該行前面加 # 註解掉即可。
NOISE = [
"nav.theme-doc-breadcrumbs", # 麵包屑(首頁 科學 文章名)
".breadcrumbs", # 麵包屑的備用選擇器
".theme-doc-toc-mobile", # 手機版的「本頁導覽」
".theme-doc-toc-desktop", # 桌面版目錄(通常在 article 外,保險起見)
".tableOfContents", # 目錄的備用選擇器
".theme-doc-version-badge", # 版本標籤
".theme-doc-version-banner", # 版本提示橫幅
"footer.theme-doc-footer", # 頁尾的「編輯此頁」與最後更新時間
".pagination-nav", # 上一頁/下一頁
".theme-edit-this-page", # 編輯此頁的備用選擇器
]
for selector in NOISE:
for el in article.select(selector):
el.decompose()
# 收集文章裡的連結與圖片,在原處標上 [n],網址統一列到文章最下方。
# 跳過純錨點、標題旁的 # 錨點、以及腳註區塊裡的來回連結。
BASE = os.environ.get("BASE_URL", "")
def absolute(u):
if not u:
return None
if u.startswith("data:"): # base64 的 placeholder沒意義
return None
if u.startswith("/"):
return BASE + u
if u.startswith(("http://", "https://", "mailto:")):
return u
return None
def to_txt(u):
# 站內連結改指向純文字版,讓 .txt 之間可以互相跳轉。
# 圖片、外部連結、已經有副檔名的都不動。
if any(s in u for s in ("/tags/", "/archive/", "/page/")):
return u
if not BASE or not u.startswith(BASE):
return u
path = u[len(BASE):]
# 切掉 query string 跟錨點,處理完再接回去
tail = ""
for sep in ("#", "?"):
i = path.find(sep)
if i != -1:
tail = path[i:] + tail
path = path[:i]
if not path or path == "/":
return u
path = path.rstrip("/")
if "." in path.rsplit("/", 1)[-1]: # 已經有副檔名(.png、.pdf⋯⋯
return u
return BASE + path + ".txt" + tail
links = []
seen = {}
def mark(el, url, label):
n = seen.get(url)
if n is None:
n = len(links) + 1
seen[url] = n
links.append((n, url, label))
el.insert_after(soup.new_string(" [" + str(n) + "]"))
for el in article.find_all(["a", "img"]):
if el.name == "a":
if not el.get_text(strip=True):
continue
href = el.get("href", "")
if href.startswith("#"):
continue
if el.find_parent(["h1", "h2", "h3", "h4", "h5", "h6"]):
continue
if el.find_parent(class_="footnotes"):
continue
url = absolute(href)
if url:
mark(el, to_txt(url), "")
else:
# 圖片srcset 裡通常有原始尺寸,優先取 src
url = absolute(el.get("src", ""))
if url:
# 沒有 alt 的圖片w3m 會印出檔名之類的神秘文字,
# 乾脆把 alt 換成統一的標記
el["alt"] = "圖片"
mark(el, url, "圖片")
if links:
pre = soup.new_tag("pre") # 用 pre 包住,長網址才不會被 w3m 斷行切壞
pre.string = "\n── 連結與圖片 ──\n\n" + "\n".join(
"[" + str(n) + "] " + ("(圖片)" if lab else "") + u
for n, u, lab in links)
article.append(pre)
# 幫不同層級的標題加上裝飾,這樣轉成純文字之後還是看得出層級
for tag in article.find_all("h1"):
tag.string = "\n═══ " + tag.get_text() + " ═══\n"
for tag in article.find_all("h2"):
tag.string = "\n── " + tag.get_text() + " ──\n"
for tag in article.find_all("h3"):
tag.string = "\n◆ " + tag.get_text() + "\n"
for tag in article.find_all("h4"):
tag.string = "\n・" + tag.get_text() + "\n"
print(str(article))
' "$file" > "$tmp"
w3m -dump -cols 80 -O UTF-8 -T text/html "$tmp" > "$out"
rm -f "$tmp"
echo "完成:$out"
}
export -f process_one
find "$SRC" -name "*.html" \
| grep -vE "/tags/|/archive/|/page/" \
| grep -v "^${SRC}/index.html$" \
| xargs -P 23 -I {} bash -c 'process_one "$@"' _ {}
echo ""
echo "全部轉換完成!輸出在 $DST/"
python3 txt_index.py