diff --git a/txt.sh b/txt.sh index d6147e5..84a3272 100644 --- a/txt.sh +++ b/txt.sh @@ -1,5 +1,4 @@ #!/bin/bash - # ============================================================== # txt.sh # @@ -19,31 +18,23 @@ # # 前置需求:python3、beautifulsoup4(pip install beautifulsoup4)、w3m # ============================================================== - cd "$(dirname "$0")" - - -# 換成你自己的網頁 build 資料夾 SRC="/path/to/build" DST="/path/to/build" +BASE_URL="https://wiwi.blog" # 相對連結會補上這個網域 mkdir -p "$DST" - -export SRC DST - +export SRC DST BASE_URL process_one() { file="$1" rel="${file#$SRC/}" slug=$(dirname "$rel") out="$DST/$slug.txt" - tmp=$(mktemp) - # 注意:python 程式碼用單引號包住,檔名透過 sys.argv 傳進去, # 這樣路徑裡有引號或空白也不會出事 python3 -c ' -import sys +import os, sys from bs4 import BeautifulSoup - soup = BeautifulSoup(open(sys.argv[1], encoding="utf-8"), "html.parser") article = soup.find("article") if not article: @@ -63,10 +54,88 @@ NOISE = [ ".pagination-nav", # 上一頁/下一頁 ".theme-edit-this-page", # 編輯此頁的備用選擇器 ] - for selector in NOISE: for el in article.select(selector): el.decompose() +# 收集文章裡的連結與圖片,在原處標上 [n],網址統一列到文章最下方。 +# 跳過純錨點、標題旁的 # 錨點、以及腳註區塊裡的來回連結。 +BASE = os.environ.get("BASE_URL", "") + +def absolute(u): + if not u: + return None + if u.startswith("data:"): # base64 的 placeholder,沒意義 + return None + if u.startswith("/"): + return BASE + u + if u.startswith(("http://", "https://", "mailto:")): + return u + return None + +def to_txt(u): + # 站內連結改指向純文字版,讓 .txt 之間可以互相跳轉。 + # 圖片、外部連結、已經有副檔名的都不動。 + if any(s in u for s in ("/tags/", "/archive/", "/page/")): + return u + if not BASE or not u.startswith(BASE): + return u + path = u[len(BASE):] + # 切掉 query string 跟錨點,處理完再接回去 + tail = "" + for sep in ("#", "?"): + i = path.find(sep) + if i != -1: + tail = path[i:] + tail + path = path[:i] + if not path or path == "/": + return u + path = path.rstrip("/") + if "." in path.rsplit("/", 1)[-1]: # 已經有副檔名(.png、.pdf⋯⋯) + return u + return BASE + path + ".txt" + tail + + +links = [] +seen = {} + +def mark(el, url, label): + n = seen.get(url) + if n is None: + n = len(links) + 1 + seen[url] = n + links.append((n, url, label)) + el.insert_after(soup.new_string(" [" + str(n) + "]")) + +for el in article.find_all(["a", "img"]): + if el.name == "a": + if not el.get_text(strip=True): + continue + href = el.get("href", "") + if href.startswith("#"): + continue + if el.find_parent(["h1", "h2", "h3", "h4", "h5", "h6"]): + continue + if el.find_parent(class_="footnotes"): + continue + url = absolute(href) + if url: + mark(el, to_txt(url), "") + else: + # 圖片:srcset 裡通常有原始尺寸,優先取 src + url = absolute(el.get("src", "")) + if url: + # 沒有 alt 的圖片,w3m 會印出檔名之類的神秘文字, + # 乾脆把 alt 換成統一的標記 + el["alt"] = "圖片" + mark(el, url, "圖片") + +if links: + pre = soup.new_tag("pre") # 用 pre 包住,長網址才不會被 w3m 斷行切壞 + pre.string = "\n── 連結與圖片 ──\n\n" + "\n".join( + "[" + str(n) + "] " + ("(圖片)" if lab else "") + u + for n, u, lab in links) + article.append(pre) + # 幫不同層級的標題加上裝飾,這樣轉成純文字之後還是看得出層級 for tag in article.find_all("h1"): @@ -77,23 +146,17 @@ for tag in article.find_all("h3"): tag.string = "\n◆ " + tag.get_text() + "\n" for tag in article.find_all("h4"): tag.string = "\n・" + tag.get_text() + "\n" - print(str(article)) ' "$file" > "$tmp" - w3m -dump -cols 80 -O UTF-8 -T text/html "$tmp" > "$out" rm -f "$tmp" - echo "完成:$out" } export -f process_one - find "$SRC" -name "*.html" \ | grep -vE "/tags/|/archive/|/page/" \ | grep -v "^${SRC}/index.html$" \ | xargs -P 23 -I {} bash -c 'process_one "$@"' _ {} - echo "" echo "全部轉換完成!輸出在 $DST/" - -python3 txt_index.py \ No newline at end of file +python3 txt_index.py