| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213 |
- """
- 药典 PDF 目录提取工具
- 提取 2025 年版《中国药典》四部 PDF 的目录结构
- 用法:
- python extract_catalog.py /path/to/药典四部.pdf
- 输出:
- data-pipeline/data/catalog_volume4.json — 结构化目录 JSON
- data-pipeline/data/catalog_volume4.md — Markdown 目录
- """
- import sys
- import json
- import time
- import re
- from pathlib import Path
- import fitz # PyMuPDF
- # ============================================
- # 配置
- # ============================================
- INTERVAL = 0.2 # 每页处理间隔(秒),控制提取频率
- TOC_START_PAGE = 0 # 目录起始页(0-indexed),通常在前几页
- TOC_SCAN_PAGES = 30 # 扫描前30页找目录
- OUTPUT_DIR = Path(__file__).resolve().parent / "data"
- # ============================================
- # 提取逻辑
- # ============================================
- def extract_pdf_text(pdf_path: str, interval: float = INTERVAL) -> list[dict]:
- """逐页提取 PDF 文本,控制处理频率"""
- doc = fitz.open(pdf_path)
- total = len(doc)
- pages = []
- print(f"📄 打开 PDF: {pdf_path}")
- print(f" 总页数: {total}")
- for i in range(min(TOC_SCAN_PAGES, total)):
- page = doc[i]
- text = page.get_text("text")
- blocks = page.get_text("blocks") # 按段落提取
- pages.append({
- "page_num": i + 1,
- "text": text,
- "blocks": [
- {
- "text": b[4].strip(),
- "x0": b[0], "y0": b[1], # 坐标
- "x1": b[2], "y1": b[3],
- }
- for b in blocks if b[6] == 0 # 0=文本块
- ],
- })
- if i > 0 and i % 10 == 0:
- print(f" 已处理 {i}/{min(TOC_SCAN_PAGES, total)} 页")
- time.sleep(interval) # 控制频率
- doc.close()
- return pages
- def find_toc_pages(pages: list[dict]) -> tuple[int, int]:
- """定位目录起始和结束页"""
- start = 0
- end = TOC_SCAN_PAGES
- for i, p in enumerate(pages):
- text = p["text"]
- if "目" in text and "录" in text and i < 10:
- # 确认是目录页
- lines = text.strip().split("\n")
- if any("录" in l for l in lines[:3]):
- start = i
- else:
- start = i
- break
- # 找目录结束(正文开始)
- for i in range(start + 1, len(pages)):
- text = pages[i]["text"]
- if "凡例" in text or "通则" in text or "正文" in text:
- end = i
- break
- print(f" 目录定位: 第{start+1}页 ~ 第{end+1}页")
- return start, end
- def parse_toc_structure(pages: list[dict], start: int, end: int) -> list[dict]:
- """解析目录层级结构"""
- toc_items = []
- current_part = "" # 部
- current_chapter = "" # 章
- for pi in range(start, min(end + 1, len(pages))):
- for block in pages[pi]["blocks"]:
- text = block["text"].strip()
- if len(text) < 3:
- continue
- x = block["x0"] # 缩进量判断层级
- # 部级(最左对齐)
- if x < 50 and ("凡例" in text or "通则" in text or "正文" in text):
- item = {
- "level": 1,
- "title": text,
- "page": pages[pi]["page_num"],
- "children": [],
- }
- toc_items.append(item)
- current_part = text
- current_chapter = ""
- print(f" 📘 {text}")
- # 章级(左缩进约 50-80)
- elif 40 <= x < 100 and ("章" in text or len(text) > 3):
- item = {
- "level": 2,
- "title": text,
- "page": pages[pi]["page_num"],
- "children": [],
- }
- if toc_items:
- toc_items[-1]["children"].append(item)
- else:
- toc_items.append(item)
- current_chapter = text
- print(f" 📗 {text}")
- # 节级(更右的缩进)
- elif x >= 80 and ("节" in text or "章" not in text):
- item = {
- "level": 3,
- "title": text,
- "page": pages[pi]["page_num"],
- }
- if toc_items and toc_items[-1]["children"]:
- toc_items[-1]["children"][-1].setdefault("children", []).append(item)
- print(f" 📙 {text}" if len(text) < 50 else f" 📙 {text[:50]}...")
- return toc_items
- def export_markdown(toc: list[dict], output_path: Path):
- """导出为 Markdown 格式"""
- with open(output_path, "w", encoding="utf-8") as f:
- f.write("# 《中国药典》2025年版 四部 目录\n\n")
- for part in toc:
- f.write(f"## {part['title']}\n\n")
- for ch in part.get("children", []):
- f.write(f"- **{ch['title']}** ")
- if ch.get("page"):
- f.write(f"(第{ch['page']}页)")
- f.write("\n")
- for sec in ch.get("children", []):
- f.write(f" - {sec['title']}")
- if sec.get("page"):
- f.write(f"(第{sec['page']}页)")
- f.write("\n")
- f.write("\n")
- print(f"\n📝 Markdown 已保存: {output_path}")
- # ============================================
- # 主入口
- # ============================================
- def main(pdf_path: str):
- print("=" * 55)
- print(" 药典 PDF 目录提取器")
- print("=" * 55)
- print()
- pages = extract_pdf_text(pdf_path)
- print(f"\n✅ 提取完成,共处理 {len(pages)} 页\n")
- start, end = find_toc_pages(pages)
- print(f"\n📋 解析目录结构...\n")
- toc = parse_toc_structure(pages, start, end)
- OUTPUT_DIR.mkdir(parents=True, exist_ok=True)
- json_path = OUTPUT_DIR / "catalog_volume4.json"
- with open(json_path, "w", encoding="utf-8") as f:
- json.dump(toc, f, ensure_ascii=False, indent=2)
- print(f"\n📦 JSON 已保存: {json_path}")
- md_path = OUTPUT_DIR / "catalog_volume4.md"
- export_markdown(toc, md_path)
- # 统计
- total = len(toc)
- chapters = sum(len(p.get("children", [])) for p in toc)
- sections = sum(
- len(ch.get("children", []))
- for p in toc
- for ch in p.get("children", [])
- )
- print(f"\n📊 统计: {total} 部, {chapters} 章, {sections} 节")
- print("=" * 55)
- if __name__ == "__main__":
- if len(sys.argv) < 2:
- print("用法: python extract_catalog.py <PDF文件路径>")
- print("示例: python extract_catalog.py ~/Downloads/CP2025_Volume4.pdf")
- sys.exit(1)
- main(sys.argv[1])
|