新增 txt_index.py
This commit is contained in:
342
txt_index.py
Normal file
342
txt_index.py
Normal file
@@ -0,0 +1,342 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
# ==============================================================
|
||||||
|
# txt_index.py
|
||||||
|
#
|
||||||
|
# 產生 build/txt/index.html —— 網站所有 .txt 純文字檔的索引頁
|
||||||
|
#
|
||||||
|
# 執行順序:先跑 txt.sh,再跑這支。
|
||||||
|
# 因為這支會檢查 .txt 是否真的存在,不存在的就不列出來。
|
||||||
|
#
|
||||||
|
# 運作方式:
|
||||||
|
# 1. 掃 build/blog 和 build/docs 底下的 index.html
|
||||||
|
# 2. 從 <meta property="og:title"> 抓標題、article:published_time 抓日期
|
||||||
|
# 3. 算出對應的 .txt 路徑,確認檔案存在才收進列表
|
||||||
|
# 4. 吐出一個沒有任何框架、只有幾行 CSS 的 HTML
|
||||||
|
#
|
||||||
|
# 前置需求:python3、beautifulsoup4
|
||||||
|
# ==============================================================
|
||||||
|
|
||||||
|
import html
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
from concurrent.futures import ThreadPoolExecutor
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
|
||||||
|
# ---------- 以下是你可以改的地方 ----------
|
||||||
|
|
||||||
|
BUILD = "path/to/build"
|
||||||
|
|
||||||
|
# 首頁文字,想寫幾行就寫幾行,空行會自動變成分段
|
||||||
|
INTRO = """
|
||||||
|
這裡是 Wiwi.Blog 的純文字版本。
|
||||||
|
|
||||||
|
站台本體在 https://wiwi.blog
|
||||||
|
|
||||||
|
聯絡 Wiwi:b@wiwikuan.com
|
||||||
|
"""
|
||||||
|
|
||||||
|
# 除了 /blog 和 /docs 之外,還要列出哪些單獨的頁面
|
||||||
|
# 想加新的就加一行,寫網址路徑就好(不用寫 .txt)
|
||||||
|
EXTRA_PAGES = [
|
||||||
|
"/now",
|
||||||
|
"/use",
|
||||||
|
]
|
||||||
|
|
||||||
|
# 標題後面的站名後綴,會從連結文字裡拿掉
|
||||||
|
# 例如「近況 | Wiwi.Blog」→「近況」
|
||||||
|
# 只比對結尾,所以標題裡自己有 | 不會被誤砍
|
||||||
|
SITE_TITLE = "Wiwi.Blog"
|
||||||
|
|
||||||
|
# 站名前面可能出現的分隔符號,通常不用改
|
||||||
|
TITLE_SEPARATORS = ["|", "|", "-", "—", "–", "·", "•"]
|
||||||
|
|
||||||
|
# 連結顏色
|
||||||
|
LINK_COLOR = "#0f766e" # 未讀,深青綠
|
||||||
|
LINK_VISITED_COLOR = "#6b5b95" # 讀過,霧紫
|
||||||
|
|
||||||
|
# 同時處理幾個檔案
|
||||||
|
# 設 1 就是單執行緒。實測在 SSD 上差別不大(真正的瓶頸是解析,
|
||||||
|
# 而下面的 HEAD_BYTES 已經解決了),但在慢一點的硬碟上多開會有幫助
|
||||||
|
WORKERS = 8
|
||||||
|
|
||||||
|
# 每個 HTML 只讀開頭這麼多位元組
|
||||||
|
# 我們只要 <head> 裡的 meta,沒必要把整份 100KB 的 HTML 讀進來解析
|
||||||
|
HEAD_BYTES = 16384
|
||||||
|
|
||||||
|
# 要不要印出詳細過程(也可以在命令列加 -q 暫時關掉)
|
||||||
|
VERBOSE = True
|
||||||
|
|
||||||
|
# ---------- 以下通常不用改 ----------
|
||||||
|
|
||||||
|
# blog 底下這些路徑不是文章,要跳過
|
||||||
|
SKIP_PARTS = {"tags", "archive", "page", "authors"}
|
||||||
|
|
||||||
|
|
||||||
|
def log(msg):
|
||||||
|
"""VERBOSE 開著才印。"""
|
||||||
|
if VERBOSE:
|
||||||
|
print(msg)
|
||||||
|
|
||||||
|
|
||||||
|
def strip_suffix(title):
|
||||||
|
"""
|
||||||
|
把「近況 | Wiwi.Blog」變成「近況」。
|
||||||
|
只認結尾的「分隔符號 + 站名」,所以標題本身含有 | 不受影響。
|
||||||
|
"""
|
||||||
|
title = title.strip()
|
||||||
|
if not SITE_TITLE:
|
||||||
|
return title
|
||||||
|
|
||||||
|
for sep in TITLE_SEPARATORS:
|
||||||
|
tail = f"{sep} {SITE_TITLE}"
|
||||||
|
if title.endswith(tail):
|
||||||
|
stripped = title[: -len(tail)].strip()
|
||||||
|
return stripped or title # 整個被砍光就還是用原本的
|
||||||
|
# 也處理沒空格的寫法:「近況|Wiwi.Blog」
|
||||||
|
tail = f"{sep}{SITE_TITLE}"
|
||||||
|
if title.endswith(tail):
|
||||||
|
stripped = title[: -len(tail)].strip()
|
||||||
|
return stripped or title
|
||||||
|
|
||||||
|
return title
|
||||||
|
|
||||||
|
|
||||||
|
def read_page(html_path):
|
||||||
|
"""讀一個 index.html,回傳 (標題, 日期字串)。抓不到就回 None。"""
|
||||||
|
if not os.path.isfile(html_path):
|
||||||
|
return None
|
||||||
|
|
||||||
|
# 只讀開頭一小段,而且在 </head> 就切斷
|
||||||
|
# 這是整支程式最重要的效能關鍵:Docusaurus 的 HTML 可能有 100KB 以上,
|
||||||
|
# 但我們要的 meta 全都在最前面的幾 KB 裡
|
||||||
|
with open(html_path, encoding="utf-8", errors="replace") as f:
|
||||||
|
chunk = f.read(HEAD_BYTES)
|
||||||
|
|
||||||
|
cut = chunk.find("</head>")
|
||||||
|
if cut != -1:
|
||||||
|
chunk = chunk[:cut]
|
||||||
|
|
||||||
|
soup = BeautifulSoup(chunk, "html.parser")
|
||||||
|
|
||||||
|
# 標題優先用 og:title,抓不到再退回 <title>
|
||||||
|
# 兩者都可能帶著「| Wiwi.Blog」,統一用 strip_suffix() 清掉
|
||||||
|
title = None
|
||||||
|
og = soup.find("meta", property="og:title")
|
||||||
|
if og and og.get("content"):
|
||||||
|
title = strip_suffix(og["content"])
|
||||||
|
elif soup.title and soup.title.string:
|
||||||
|
title = strip_suffix(soup.title.string)
|
||||||
|
|
||||||
|
if not title:
|
||||||
|
return None
|
||||||
|
|
||||||
|
# 日期只有 blog 文章會有
|
||||||
|
date = ""
|
||||||
|
meta_date = soup.find("meta", property="article:published_time")
|
||||||
|
if meta_date and meta_date.get("content"):
|
||||||
|
try:
|
||||||
|
date = datetime.fromisoformat(
|
||||||
|
meta_date["content"].replace("Z", "+00:00")
|
||||||
|
).strftime("%Y-%m-%d")
|
||||||
|
except ValueError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
return title, date
|
||||||
|
|
||||||
|
|
||||||
|
def collect(section):
|
||||||
|
"""
|
||||||
|
掃 build/<section> 底下所有文章,回傳 [(網址, 標題, 日期), ...]
|
||||||
|
section 例如 "blog" 或 "docs"
|
||||||
|
|
||||||
|
分兩階段:
|
||||||
|
1. 走訪資料夾,篩掉非文章頁與沒有 .txt 的(很快,單執行緒就好)
|
||||||
|
2. 剩下的用執行緒池平行讀取 HTML 抓標題
|
||||||
|
這樣訊息才不會因為平行處理而變成亂序。
|
||||||
|
"""
|
||||||
|
root = os.path.join(BUILD, section)
|
||||||
|
stats = {"跳過(非文章頁)": 0, "跳過(沒有 .txt)": 0, "跳過(抓不到標題)": 0}
|
||||||
|
|
||||||
|
log(f"\n[{section}] 掃描 {root}")
|
||||||
|
|
||||||
|
if not os.path.isdir(root):
|
||||||
|
log(f" 找不到這個資料夾,跳過整個 {section}")
|
||||||
|
return []
|
||||||
|
|
||||||
|
# ---- 階段一:挑出候選 ----
|
||||||
|
candidates = [] # [(rel, html_path, txt_path), ...]
|
||||||
|
|
||||||
|
for dirpath, dirnames, filenames in os.walk(root):
|
||||||
|
if "index.html" not in filenames:
|
||||||
|
continue
|
||||||
|
|
||||||
|
# build/blog/foo/index.html → blog/foo
|
||||||
|
rel = os.path.relpath(dirpath, BUILD)
|
||||||
|
|
||||||
|
# 跳過 section 首頁本身(例如 build/blog/index.html)
|
||||||
|
if rel == section:
|
||||||
|
log(f" - {rel}/ ← section 首頁")
|
||||||
|
stats["跳過(非文章頁)"] += 1
|
||||||
|
continue
|
||||||
|
|
||||||
|
# 跳過 tags、archive、分頁那些非文章頁面
|
||||||
|
hit = SKIP_PARTS & set(rel.split(os.sep))
|
||||||
|
if hit:
|
||||||
|
log(f" - {rel}/ ← 含有 {'、'.join(sorted(hit))}")
|
||||||
|
stats["跳過(非文章頁)"] += 1
|
||||||
|
continue
|
||||||
|
|
||||||
|
# 對應的純文字檔:build/blog/foo/index.html → build/blog/foo.txt
|
||||||
|
txt_path = os.path.join(BUILD, rel + ".txt")
|
||||||
|
if not os.path.isfile(txt_path):
|
||||||
|
log(f" ! {rel}/ ← 找不到 {rel}.txt,txt.sh 跑過了嗎?")
|
||||||
|
stats["跳過(沒有 .txt)"] += 1
|
||||||
|
continue
|
||||||
|
|
||||||
|
candidates.append((rel, os.path.join(dirpath, "index.html"), txt_path))
|
||||||
|
|
||||||
|
candidates.sort() # 固定順序,訊息才不會每次跑都不一樣
|
||||||
|
|
||||||
|
# ---- 階段二:平行讀取 HTML ----
|
||||||
|
if WORKERS > 1 and len(candidates) > 1:
|
||||||
|
with ThreadPoolExecutor(max_workers=WORKERS) as pool:
|
||||||
|
pages = list(pool.map(lambda c: read_page(c[1]), candidates))
|
||||||
|
else:
|
||||||
|
pages = [read_page(c[1]) for c in candidates]
|
||||||
|
|
||||||
|
items = []
|
||||||
|
for (rel, _, txt_path), page in zip(candidates, pages):
|
||||||
|
if not page:
|
||||||
|
log(f" ! {rel}/ ← HTML 裡抓不到標題")
|
||||||
|
stats["跳過(抓不到標題)"] += 1
|
||||||
|
continue
|
||||||
|
|
||||||
|
title, date = page
|
||||||
|
size = os.path.getsize(txt_path)
|
||||||
|
log(f" + {date:<10} {title} ({size:,} bytes)")
|
||||||
|
items.append(("/" + rel.replace(os.sep, "/") + ".txt", title, date))
|
||||||
|
|
||||||
|
log(f" 收錄 {len(items)} 篇" + "".join(
|
||||||
|
f"、{k} {v}" for k, v in stats.items() if v
|
||||||
|
))
|
||||||
|
return items
|
||||||
|
|
||||||
|
|
||||||
|
def collect_extra():
|
||||||
|
"""處理 EXTRA_PAGES 裡指定的單獨頁面。"""
|
||||||
|
items = []
|
||||||
|
log(f"\n[頁面] EXTRA_PAGES 共 {len(EXTRA_PAGES)} 項")
|
||||||
|
|
||||||
|
for url in EXTRA_PAGES:
|
||||||
|
rel = url.strip("/")
|
||||||
|
txt_path = os.path.join(BUILD, rel + ".txt")
|
||||||
|
if not os.path.isfile(txt_path):
|
||||||
|
log(f" ! /{rel} ← 找不到 {rel}.txt")
|
||||||
|
continue
|
||||||
|
|
||||||
|
page = read_page(os.path.join(BUILD, rel, "index.html"))
|
||||||
|
if page:
|
||||||
|
title = page[0]
|
||||||
|
else:
|
||||||
|
title = rel
|
||||||
|
log(f" ? /{rel} ← 抓不到標題,先用路徑名稱當連結文字")
|
||||||
|
|
||||||
|
size = os.path.getsize(txt_path)
|
||||||
|
log(f" + {title} ({size:,} bytes)")
|
||||||
|
items.append((f"/{rel}.txt", title, ""))
|
||||||
|
|
||||||
|
log(f" 收錄 {len(items)} 頁")
|
||||||
|
return items
|
||||||
|
|
||||||
|
|
||||||
|
def render_list(items):
|
||||||
|
"""把 [(網址, 標題, 日期)] 變成 <ul>。"""
|
||||||
|
lines = ["<ul>"]
|
||||||
|
for url, title, date in items:
|
||||||
|
prefix = f'<span class="d">{date}</span> ' if date else ""
|
||||||
|
lines.append(
|
||||||
|
f'<li>{prefix}<a href="{html.escape(url)}">{html.escape(title)}</a></li>'
|
||||||
|
)
|
||||||
|
lines.append("</ul>")
|
||||||
|
return "\n".join(lines)
|
||||||
|
|
||||||
|
|
||||||
|
def render_intro(text):
|
||||||
|
"""把 INTRO 的空行切成 <p>。"""
|
||||||
|
blocks = [b.strip() for b in text.strip().split("\n\n")]
|
||||||
|
return "\n".join(
|
||||||
|
"<p>" + html.escape(b).replace("\n", "<br>") + "</p>" for b in blocks if b
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
started = time.time()
|
||||||
|
log("=" * 60)
|
||||||
|
log(f"txt_index.py — 產生純文字索引")
|
||||||
|
log(f"build 目錄:{BUILD}")
|
||||||
|
log(f"執行緒:{WORKERS}")
|
||||||
|
log("=" * 60)
|
||||||
|
|
||||||
|
extra = collect_extra()
|
||||||
|
|
||||||
|
# blog 依日期新到舊;沒日期的排最後
|
||||||
|
blog = sorted(collect("blog"), key=lambda x: x[2], reverse=True)
|
||||||
|
|
||||||
|
# docs 依網址排,這樣同一個資料夾的會排在一起
|
||||||
|
docs = sorted(collect("docs"), key=lambda x: x[0])
|
||||||
|
|
||||||
|
parts = [
|
||||||
|
f"""<!DOCTYPE html>
|
||||||
|
<html lang="zh-Hant">
|
||||||
|
<head>
|
||||||
|
<meta charset="utf-8">
|
||||||
|
<meta name="viewport" content="width=device-width, initial-scale=1">
|
||||||
|
<title>Wiwi.Blog|純文字版</title>
|
||||||
|
<style>
|
||||||
|
body {{ font-family: monospace; max-width: 80ch; margin: 2rem auto;
|
||||||
|
padding: 0 1rem; line-height: 1.6; }}
|
||||||
|
a {{ color: {LINK_COLOR}; }}
|
||||||
|
a:visited {{ color: {LINK_VISITED_COLOR}; }}
|
||||||
|
.d {{ opacity: 0.5; }}
|
||||||
|
</style>
|
||||||
|
</head>
|
||||||
|
<body>
|
||||||
|
{render_intro(INTRO)}
|
||||||
|
""",
|
||||||
|
]
|
||||||
|
|
||||||
|
if extra:
|
||||||
|
parts.append("<h2>頁面</h2>\n" + render_list(extra))
|
||||||
|
if blog:
|
||||||
|
parts.append("<h2>貼文</h2>\n" + render_list(blog))
|
||||||
|
if docs:
|
||||||
|
parts.append("<h2>筆記</h2>\n" + render_list(docs))
|
||||||
|
|
||||||
|
parts.append("</body>\n</html>\n")
|
||||||
|
|
||||||
|
out_dir = os.path.join(BUILD, "txt")
|
||||||
|
os.makedirs(out_dir, exist_ok=True)
|
||||||
|
out_path = os.path.join(out_dir, "index.html")
|
||||||
|
|
||||||
|
with open(out_path, "w", encoding="utf-8") as f:
|
||||||
|
f.write("\n".join(parts))
|
||||||
|
|
||||||
|
total = len(extra) + len(blog) + len(docs)
|
||||||
|
log("\n" + "=" * 60)
|
||||||
|
print(f"完成:{out_path}")
|
||||||
|
log(f" 檔案大小:{os.path.getsize(out_path):,} bytes")
|
||||||
|
log(f" 頁面 {len(extra)} 個、Blog {len(blog)} 篇、Docs {len(docs)} 篇"
|
||||||
|
f",共 {total} 個連結")
|
||||||
|
if blog:
|
||||||
|
log(f" 最新一篇:{blog[0][2]} {blog[0][1]}")
|
||||||
|
log(f" 耗時 {time.time() - started:.2f} 秒")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
# 加 -q 就安靜執行,只印最後一行
|
||||||
|
if "-q" in sys.argv or "--quiet" in sys.argv:
|
||||||
|
VERBOSE = False
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user