|
- import argparse
- import datetime as dt
- import json
- import pathlib
- import re
- import sqlite3
- import sys
-
- from pypdf import PdfReader
-
-
- TOC_MARKERS = ("目录", "目 录", "contents", "table of contents")
- TOC_LINE = re.compile(r"^(?P<title>.{2,160}?)(?:[..·…]{2,}|\s{2,})(?P<page>-?\d{1,4})\s*$")
-
-
- def compact(value: str) -> str:
- return re.sub(r"\s+", " ", value or "").strip()
-
-
- def text_lines(value: str) -> list[str]:
- return [compact(line) for line in (value or "").splitlines() if compact(line)]
-
-
- def is_toc_page(text: str) -> bool:
- lowered = text.lower()
- if any(marker in lowered for marker in TOC_MARKERS):
- return True
- return sum(1 for line in text_lines(text) if TOC_LINE.match(line)) >= 4
-
-
- def parse_toc_entries(text: str, pdf_page: int) -> list[dict[str, object]]:
- entries = []
- for line in text_lines(text):
- match = TOC_LINE.match(line)
- if not match:
- continue
- title = compact(match.group("title")).strip("..·… ")
- if len(title) < 2 or title in TOC_MARKERS:
- continue
- entries.append(
- {
- "title": title,
- "printed_page": int(match.group("page")),
- "toc_pdf_page": pdf_page,
- }
- )
- return entries
-
-
- def open_index(path: pathlib.Path) -> sqlite3.Connection:
- path.parent.mkdir(parents=True, exist_ok=True)
- if path.exists():
- path.unlink()
- connection = sqlite3.connect(path)
- connection.execute("PRAGMA journal_mode=OFF")
- connection.execute("PRAGMA synchronous=OFF")
- connection.execute("PRAGMA temp_store=MEMORY")
- connection.execute(
- "CREATE TABLE manuals ("
- "filename TEXT PRIMARY KEY, page_count INTEGER NOT NULL, size_bytes INTEGER NOT NULL, "
- "toc_pdf_pages TEXT NOT NULL)"
- )
- connection.execute(
- "CREATE VIRTUAL TABLE page_search USING fts5("
- "manual UNINDEXED, filename UNINDEXED, pdf_page UNINDEXED, text, tokenize='trigram')"
- )
- return connection
-
-
- def build_index(manual_dir: pathlib.Path, database_path: pathlib.Path, toc_scan_pages: int) -> dict:
- connection = open_index(database_path)
- manuals = []
- page_total = 0
- try:
- for pdf_path in sorted(manual_dir.glob("*.pdf"), key=lambda item: item.name.casefold()):
- reader = PdfReader(str(pdf_path))
- toc_pages = []
- toc_entries = []
- print(f"indexing {pdf_path.name}: {len(reader.pages)} pages", flush=True)
- for page_number, page in enumerate(reader.pages, start=1):
- text = page.extract_text() or ""
- normalized = compact(text)
- connection.execute(
- "INSERT INTO page_search(manual, filename, pdf_page, text) VALUES (?, ?, ?, ?)",
- (pdf_path.stem, pdf_path.name, page_number, normalized),
- )
- if page_number <= toc_scan_pages and is_toc_page(text):
- toc_pages.append(page_number)
- toc_entries.extend(parse_toc_entries(text, page_number))
- toc_entries = toc_entries[:300]
- connection.execute(
- "INSERT INTO manuals(filename, page_count, size_bytes, toc_pdf_pages) VALUES (?, ?, ?, ?)",
- (pdf_path.name, len(reader.pages), pdf_path.stat().st_size, json.dumps(toc_pages)),
- )
- manuals.append(
- {
- "filename": pdf_path.name,
- "page_count": len(reader.pages),
- "size_bytes": pdf_path.stat().st_size,
- "toc_pdf_pages": toc_pages,
- "toc_entries": toc_entries,
- }
- )
- page_total += len(reader.pages)
- connection.commit()
- connection.execute("VACUUM")
- finally:
- connection.close()
- return {
- "generated_at": dt.datetime.now(dt.timezone.utc).replace(microsecond=0).isoformat(),
- "manual_count": len(manuals),
- "page_count": page_total,
- "manuals": manuals,
- }
-
-
- def write_toc_markdown(index: dict, path: pathlib.Path) -> None:
- lines = [
- "# Manual Chapter Index",
- "",
- "This generated index covers all PDFs currently under `assets/manuals/`. `手册页` is the printed page number extracted from the table of contents; `目录 PDF 页` is the physical PDF page where the entry was found. Use `search_xinje_manuals.py` for exact PDF-page results.",
- "",
- f"Generated: `{index['generated_at']}`. Manuals: `{index['manual_count']}`. PDF pages: `{index['page_count']}`.",
- "",
- ]
- for manual in index["manuals"]:
- lines.append(f"## {manual['filename']}")
- toc_pages = ", ".join(str(item) for item in manual["toc_pdf_pages"]) or "not detected"
- lines.append("")
- lines.append(f"PDF pages: `{manual['page_count']}`. Detected TOC PDF pages: `{toc_pages}`.")
- entries = manual["toc_entries"]
- if not entries:
- lines.extend(["", "No reliable TOC lines were extracted. Use the full-text index or inspect the first 40 PDF pages.", ""])
- continue
- lines.extend(["", "| 手册页 | 章节/条目 | 目录 PDF 页 |", "|---:|---|---:|"])
- seen = set()
- for entry in entries:
- key = (entry["title"], entry["printed_page"])
- if key in seen:
- continue
- seen.add(key)
- title = str(entry["title"]).replace("|", "\\|")
- lines.append(f"| {entry['printed_page']} | {title} | {entry['toc_pdf_page']} |")
- lines.append("")
- path.parent.mkdir(parents=True, exist_ok=True)
- path.write_text("\n".join(lines), encoding="utf-8")
-
-
- def main() -> None:
- parser = argparse.ArgumentParser(description="Build a page-level full-text and table-of-contents index for Xinje manuals.")
- parser.add_argument("--manual-dir", required=True, type=pathlib.Path)
- parser.add_argument("--output-db", required=True, type=pathlib.Path)
- parser.add_argument("--output-toc-json", required=True, type=pathlib.Path)
- parser.add_argument("--output-toc-md", required=True, type=pathlib.Path)
- parser.add_argument("--toc-scan-pages", type=int, default=40)
- args = parser.parse_args()
- if args.toc_scan_pages < 1:
- raise SystemExit("--toc-scan-pages must be positive")
- index = build_index(args.manual_dir, args.output_db, args.toc_scan_pages)
- args.output_toc_json.parent.mkdir(parents=True, exist_ok=True)
- args.output_toc_json.write_text(json.dumps(index, ensure_ascii=False, indent=2), encoding="utf-8")
- write_toc_markdown(index, args.output_toc_md)
- print(
- f"indexed {index['manual_count']} manuals and {index['page_count']} pages; "
- f"database: {args.output_db}",
- flush=True,
- )
-
-
- if __name__ == "__main__":
- if hasattr(sys.stdout, "reconfigure"):
- sys.stdout.reconfigure(encoding="utf-8")
- main()
|