Files
wiwi-blog-txt/txt_index.py
2026-08-17 09:01:27 +00:00

342 lines
11 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# ==============================================================
# txt_index.py
#
# 產生 build/txt/index.html —— 網站所有 .txt 純文字檔的索引頁
#
# 執行順序:先跑 txt.sh再跑這支。
# 因為這支會檢查 .txt 是否真的存在,不存在的就不列出來。
#
# 運作方式:
# 1. 掃 build/blog 和 build/docs 底下的 index.html
# 2. 從 <meta property="og:title"> 抓標題、article:published_time 抓日期
# 3. 算出對應的 .txt 路徑,確認檔案存在才收進列表
# 4. 吐出一個沒有任何框架、只有幾行 CSS 的 HTML
#
# 前置需求python3、beautifulsoup4
# ==============================================================
import html
import os
import sys
import time
from concurrent.futures import ThreadPoolExecutor
from datetime import datetime
from bs4 import BeautifulSoup
# ---------- 以下是你可以改的地方 ----------
BUILD = "path/to/build"
# 首頁文字,想寫幾行就寫幾行,空行會自動變成分段
INTRO = """
這裡是 Wiwi.Blog 的純文字版本。
站台本體在 https://wiwi.blog
聯絡 Wiwib@wiwikuan.com
"""
# 除了 /blog 和 /docs 之外,還要列出哪些單獨的頁面
# 想加新的就加一行,寫網址路徑就好(不用寫 .txt
EXTRA_PAGES = [
"/now",
"/use",
]
# 標題後面的站名後綴,會從連結文字裡拿掉
# 例如「近況 | Wiwi.Blog」→「近況」
# 只比對結尾,所以標題裡自己有 | 不會被誤砍
SITE_TITLE = "Wiwi.Blog"
# 站名前面可能出現的分隔符號,通常不用改
TITLE_SEPARATORS = ["|", "", "-", "", "", "·", ""]
# 連結顏色
LINK_COLOR = "#0f766e" # 未讀,深青綠
LINK_VISITED_COLOR = "#6b5b95" # 讀過,霧紫
# 同時處理幾個檔案
# 設 1 就是單執行緒。實測在 SSD 上差別不大(真正的瓶頸是解析,
# 而下面的 HEAD_BYTES 已經解決了),但在慢一點的硬碟上多開會有幫助
WORKERS = 8
# 每個 HTML 只讀開頭這麼多位元組
# 我們只要 <head> 裡的 meta沒必要把整份 100KB 的 HTML 讀進來解析
HEAD_BYTES = 16384
# 要不要印出詳細過程(也可以在命令列加 -q 暫時關掉)
VERBOSE = True
# ---------- 以下通常不用改 ----------
# blog 底下這些路徑不是文章,要跳過
SKIP_PARTS = {"tags", "archive", "page", "authors"}
def log(msg):
"""VERBOSE 開著才印。"""
if VERBOSE:
print(msg)
def strip_suffix(title):
"""
把「近況 | Wiwi.Blog」變成「近況」。
只認結尾的「分隔符號 + 站名」,所以標題本身含有 | 不受影響。
"""
title = title.strip()
if not SITE_TITLE:
return title
for sep in TITLE_SEPARATORS:
tail = f"{sep} {SITE_TITLE}"
if title.endswith(tail):
stripped = title[: -len(tail)].strip()
return stripped or title # 整個被砍光就還是用原本的
# 也處理沒空格的寫法:「近況|Wiwi.Blog」
tail = f"{sep}{SITE_TITLE}"
if title.endswith(tail):
stripped = title[: -len(tail)].strip()
return stripped or title
return title
def read_page(html_path):
"""讀一個 index.html回傳 (標題, 日期字串)。抓不到就回 None。"""
if not os.path.isfile(html_path):
return None
# 只讀開頭一小段,而且在 </head> 就切斷
# 這是整支程式最重要的效能關鍵Docusaurus 的 HTML 可能有 100KB 以上,
# 但我們要的 meta 全都在最前面的幾 KB 裡
with open(html_path, encoding="utf-8", errors="replace") as f:
chunk = f.read(HEAD_BYTES)
cut = chunk.find("</head>")
if cut != -1:
chunk = chunk[:cut]
soup = BeautifulSoup(chunk, "html.parser")
# 標題優先用 og:title抓不到再退回 <title>
# 兩者都可能帶著「| Wiwi.Blog」統一用 strip_suffix() 清掉
title = None
og = soup.find("meta", property="og:title")
if og and og.get("content"):
title = strip_suffix(og["content"])
elif soup.title and soup.title.string:
title = strip_suffix(soup.title.string)
if not title:
return None
# 日期只有 blog 文章會有
date = ""
meta_date = soup.find("meta", property="article:published_time")
if meta_date and meta_date.get("content"):
try:
date = datetime.fromisoformat(
meta_date["content"].replace("Z", "+00:00")
).strftime("%Y-%m-%d")
except ValueError:
pass
return title, date
def collect(section):
"""
掃 build/<section> 底下所有文章,回傳 [(網址, 標題, 日期), ...]
section 例如 "blog""docs"
分兩階段:
1. 走訪資料夾,篩掉非文章頁與沒有 .txt 的(很快,單執行緒就好)
2. 剩下的用執行緒池平行讀取 HTML 抓標題
這樣訊息才不會因為平行處理而變成亂序。
"""
root = os.path.join(BUILD, section)
stats = {"跳過(非文章頁)": 0, "跳過(沒有 .txt": 0, "跳過(抓不到標題)": 0}
log(f"\n[{section}] 掃描 {root}")
if not os.path.isdir(root):
log(f" 找不到這個資料夾,跳過整個 {section}")
return []
# ---- 階段一:挑出候選 ----
candidates = [] # [(rel, html_path, txt_path), ...]
for dirpath, dirnames, filenames in os.walk(root):
if "index.html" not in filenames:
continue
# build/blog/foo/index.html → blog/foo
rel = os.path.relpath(dirpath, BUILD)
# 跳過 section 首頁本身(例如 build/blog/index.html
if rel == section:
log(f" {rel}/ ← section 首頁")
stats["跳過(非文章頁)"] += 1
continue
# 跳過 tags、archive、分頁那些非文章頁面
hit = SKIP_PARTS & set(rel.split(os.sep))
if hit:
log(f" {rel}/ ← 含有 {''.join(sorted(hit))}")
stats["跳過(非文章頁)"] += 1
continue
# 對應的純文字檔build/blog/foo/index.html → build/blog/foo.txt
txt_path = os.path.join(BUILD, rel + ".txt")
if not os.path.isfile(txt_path):
log(f" {rel}/ ← 找不到 {rel}.txttxt.sh 跑過了嗎?")
stats["跳過(沒有 .txt"] += 1
continue
candidates.append((rel, os.path.join(dirpath, "index.html"), txt_path))
candidates.sort() # 固定順序,訊息才不會每次跑都不一樣
# ---- 階段二:平行讀取 HTML ----
if WORKERS > 1 and len(candidates) > 1:
with ThreadPoolExecutor(max_workers=WORKERS) as pool:
pages = list(pool.map(lambda c: read_page(c[1]), candidates))
else:
pages = [read_page(c[1]) for c in candidates]
items = []
for (rel, _, txt_path), page in zip(candidates, pages):
if not page:
log(f" {rel}/ ← HTML 裡抓不到標題")
stats["跳過(抓不到標題)"] += 1
continue
title, date = page
size = os.path.getsize(txt_path)
log(f" {date:<10} {title} ({size:,} bytes)")
items.append(("/" + rel.replace(os.sep, "/") + ".txt", title, date))
log(f" 收錄 {len(items)}" + "".join(
f"{k} {v}" for k, v in stats.items() if v
))
return items
def collect_extra():
"""處理 EXTRA_PAGES 裡指定的單獨頁面。"""
items = []
log(f"\n[頁面] EXTRA_PAGES 共 {len(EXTRA_PAGES)}")
for url in EXTRA_PAGES:
rel = url.strip("/")
txt_path = os.path.join(BUILD, rel + ".txt")
if not os.path.isfile(txt_path):
log(f" /{rel} ← 找不到 {rel}.txt")
continue
page = read_page(os.path.join(BUILD, rel, "index.html"))
if page:
title = page[0]
else:
title = rel
log(f" /{rel} ← 抓不到標題,先用路徑名稱當連結文字")
size = os.path.getsize(txt_path)
log(f" {title} ({size:,} bytes)")
items.append((f"/{rel}.txt", title, ""))
log(f" 收錄 {len(items)}")
return items
def render_list(items):
"""把 [(網址, 標題, 日期)] 變成 <ul>。"""
lines = ["<ul>"]
for url, title, date in items:
prefix = f'<span class="d">{date}</span> ' if date else ""
lines.append(
f'<li>{prefix}<a href="{html.escape(url)}">{html.escape(title)}</a></li>'
)
lines.append("</ul>")
return "\n".join(lines)
def render_intro(text):
"""把 INTRO 的空行切成 <p>。"""
blocks = [b.strip() for b in text.strip().split("\n\n")]
return "\n".join(
"<p>" + html.escape(b).replace("\n", "<br>") + "</p>" for b in blocks if b
)
def main():
started = time.time()
log("=" * 60)
log(f"txt_index.py — 產生純文字索引")
log(f"build 目錄:{BUILD}")
log(f"執行緒:{WORKERS}")
log("=" * 60)
extra = collect_extra()
# blog 依日期新到舊;沒日期的排最後
blog = sorted(collect("blog"), key=lambda x: x[2], reverse=True)
# docs 依網址排,這樣同一個資料夾的會排在一起
docs = sorted(collect("docs"), key=lambda x: x[0])
parts = [
f"""<!DOCTYPE html>
<html lang="zh-Hant">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>Wiwi.Blog純文字版</title>
<style>
body {{ font-family: monospace; max-width: 80ch; margin: 2rem auto;
padding: 0 1rem; line-height: 1.6; }}
a {{ color: {LINK_COLOR}; }}
a:visited {{ color: {LINK_VISITED_COLOR}; }}
.d {{ opacity: 0.5; }}
</style>
</head>
<body>
{render_intro(INTRO)}
""",
]
if extra:
parts.append("<h2>頁面</h2>\n" + render_list(extra))
if blog:
parts.append("<h2>貼文</h2>\n" + render_list(blog))
if docs:
parts.append("<h2>筆記</h2>\n" + render_list(docs))
parts.append("</body>\n</html>\n")
out_dir = os.path.join(BUILD, "txt")
os.makedirs(out_dir, exist_ok=True)
out_path = os.path.join(out_dir, "index.html")
with open(out_path, "w", encoding="utf-8") as f:
f.write("\n".join(parts))
total = len(extra) + len(blog) + len(docs)
log("\n" + "=" * 60)
print(f"完成:{out_path}")
log(f" 檔案大小:{os.path.getsize(out_path):,} bytes")
log(f" 頁面 {len(extra)} 個、Blog {len(blog)} 篇、Docs {len(docs)}"
f",共 {total} 個連結")
if blog:
log(f" 最新一篇:{blog[0][2]} {blog[0][1]}")
log(f" 耗時 {time.time() - started:.2f}")
if __name__ == "__main__":
# 加 -q 就安靜執行,只印最後一行
if "-q" in sys.argv or "--quiet" in sys.argv:
VERBOSE = False
main()