| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306 |
- #!/usr/bin/env python3
- """
- Standalone image extraction script for 2025 Pharmacopoeia DOCX files.
- ===== ======== ========= ====== === ===== ============== === ======
- Extracts all embedded images from DOCX files and generates a
- drug-image mapping manifest (drug_images.json).
- Key features:
- - Monkey-patches docx_ingest config so images go to /opt/static/images/
- - Does NOT write to PostgreSQL or call any embedding / vectorization API
- - Tracks which section (性状/鉴别/检查/…) each image appears in
- - Falls back to filesystem scan for images the inline-parser may have missed
- - Generates a comprehensive drug_images.json manifest
- Environment variables:
- DOCX_SOURCE_DIR – root of the DOCX tree (default: /opt/2025)
- EXTRACT_IMAGES_DIR – where to write images (default: /opt/static/images)
- EXTRACT_IMAGE_URL_PREFIX – public URL prefix (default: /images/)
- EXTRACT_MANIFEST_FILE – path for the JSON manifest (default: …)
- Author: auto-generated via Claude Code (2026-07-28)
- """
- from __future__ import annotations
- import json
- import os
- import re
- import sys
- import time
- from pathlib import Path
- # ---------------------------------------------------------------------------
- # 1. load .env (replicated from docx_ingest.py so we work before importing it)
- # ---------------------------------------------------------------------------
- def _load_env() -> None:
- env_file = Path(__file__).resolve().parent.parent / ".env"
- if not env_file.exists():
- return
- with open(env_file, encoding="utf-8") as fh:
- for line in fh:
- line = line.strip()
- if not line or line.startswith("#") or "=" not in line:
- continue
- key, _, val = line.partition("=")
- os.environ.setdefault(key.strip(), val.strip())
- _load_env()
- # ---------------------------------------------------------------------------
- # 2. Configuration
- # ---------------------------------------------------------------------------
- SOURCE_DIR = os.environ.get("DOCX_SOURCE_DIR", "/opt/2025")
- IMAGES_DIR = os.environ.get("EXTRACT_IMAGES_DIR", "/opt/static/images")
- IMAGE_URL_PREFIX = os.environ.get("EXTRACT_IMAGE_URL_PREFIX", "/images/")
- MANIFEST_FILE = os.environ.get(
- "EXTRACT_MANIFEST_FILE",
- os.path.join(IMAGES_DIR, "drug_images.json"),
- )
- # ---------------------------------------------------------------------------
- # 3. Monkey-patch *before* calling any docx_ingest function.
- # Python resolves module-level globals at call-time, so mutating the
- # module dict after import is enough to redirect _extract_images.
- # ---------------------------------------------------------------------------
- sys.path.insert(0, str(Path(__file__).resolve().parent))
- import docx_ingest # noqa: E402
- docx_ingest.IMAGES_DIR = IMAGES_DIR
- docx_ingest.IMAGE_URL_PREFIX = IMAGE_URL_PREFIX
- # Convenience aliases — these are the SAME function objects,
- # they just happen to use our overridden globals when called.
- parse_docx = docx_ingest.parse_docx
- find_docx_files = docx_ingest.find_docx_files
- # ---------------------------------------------------------------------------
- # 4. Image -> section helpers
- # ---------------------------------------------------------------------------
- IMG_TAG_RE = re.compile(r'<img\s+src="([^"]+)"[^>]*/?>', re.IGNORECASE)
- # Characters that need sanitising for filesystem-safe drug names
- _RE_UNSAFE = re.compile(r'[\s()() /\\:?*"<>|]+')
- def _safe_drug_name(name: str) -> str:
- """Return a filesystem-safe version of *name* (used as image filename prefix)."""
- return _RE_UNSAFE.sub("_", name).strip("_")
- def _collect_images_from_sections(
- sections: dict[str, str],
- ) -> list[dict[str, str]]:
- """Walk section text looking for ``<img src="…">`` tags.
- Returns a flat list of ``{section, filename, url}`` dicts.
- """
- records: list[dict[str, str]] = []
- for sec_key, sec_text in sections.items():
- for m in IMG_TAG_RE.finditer(sec_text):
- url = m.group(1)
- filename = url.rsplit("/", 1)[-1] if "/" in url else url
- records.append({"section": sec_key, "filename": filename, "url": url})
- return records
- def _find_orphan_images(
- safe_name: str,
- images_dir: str,
- already_found: set[str],
- ) -> list[dict[str, str]]:
- """Filesystem 兜底:找 <img> 标签中没有覆盖到的图片。
- Some images sit in the same paragraph as text (line 214 of docx_ingest.py
- has the ``if not p_text`` guard), so the inline parser does not emit an
- ``<img>`` tag for them. We catch those here by matching file names.
- """
- if not os.path.isdir(images_dir):
- return []
- prefix = safe_name + "_"
- orphans: list[dict[str, str]] = []
- try:
- for fname in os.listdir(images_dir):
- if fname.startswith(prefix) and fname not in already_found:
- orphans.append(
- {
- "section": "未分类",
- "filename": fname,
- "url": IMAGE_URL_PREFIX.rstrip("/") + "/" + fname,
- }
- )
- except OSError:
- pass
- return orphans
- def _group_by_section(
- flat: list[dict[str, str]],
- ) -> list[dict]:
- """Group a flat list of ``{section, filename, url}`` by section key."""
- groups: dict[str, list[dict[str, str]]] = {}
- for rec in flat:
- groups.setdefault(rec["section"], []).append(
- {"filename": rec["filename"], "url": rec["url"]}
- )
- return [
- {
- "section": sec,
- "images": imgs,
- "count": len(imgs),
- }
- for sec, imgs in groups.items()
- ]
- # ---------------------------------------------------------------------------
- # 5. Main orchestrator
- # ---------------------------------------------------------------------------
- def main() -> None:
- # re-read env in case caller uses different values after import (rare,
- # but harmless)
- global IMAGES_DIR, IMAGE_URL_PREFIX, SOURCE_DIR, MANIFEST_FILE
- IMAGES_DIR = os.environ.get("EXTRACT_IMAGES_DIR", "/opt/static/images")
- IMAGE_URL_PREFIX = os.environ.get("EXTRACT_IMAGE_URL_PREFIX", "/images/")
- SOURCE_DIR = os.environ.get("DOCX_SOURCE_DIR", "/opt/2025")
- MANIFEST_FILE = os.environ.get(
- "EXTRACT_MANIFEST_FILE",
- os.path.join(IMAGES_DIR, "drug_images.json"),
- )
- # Apply any env-driven change back to docx_ingest
- docx_ingest.IMAGES_DIR = IMAGES_DIR
- docx_ingest.IMAGE_URL_PREFIX = IMAGE_URL_PREFIX
- # Ensure directories exist
- os.makedirs(IMAGES_DIR, exist_ok=True)
- manifest_dir = os.path.dirname(MANIFEST_FILE) or "."
- os.makedirs(manifest_dir, exist_ok=True)
- print("=" * 60)
- print("IMAGE EXTRACTION: Scanning DOCX files …")
- print(f" Source: {SOURCE_DIR}")
- print(f" Images out: {IMAGES_DIR}")
- print(f" URL prefix: {IMAGE_URL_PREFIX}")
- print(f" Manifest: {MANIFEST_FILE}")
- print("=" * 60)
- files = find_docx_files(SOURCE_DIR)
- if not files:
- print("No DOCX files found – nothing to do.")
- return
- print(f"Found {len(files)} DOCX file(s)")
- # ── process every file ──────────────────────────────────────────────
- drug_records: list[dict] = []
- drugs_parsed = 0
- drugs_with_images = 0
- total_unique_images = 0
- skipped = 0
- t0 = time.time()
- for i, fp in enumerate(files, 1):
- entry = parse_docx(fp) # _extract_images runs as a side-effect
- if entry is None:
- skipped += 1
- continue
- drugs_parsed += 1
- # 4a – extract from <img> tags embedded in section text
- section_images = _collect_images_from_sections(entry["sections"])
- found_filenames = {r["filename"] for r in section_images}
- # 4b – filesystem fallback for images that the inline parser missed
- safe = _safe_drug_name(entry["name"])
- orphans = _find_orphan_images(safe, IMAGES_DIR, found_filenames)
- section_images.extend(orphans)
- if not section_images:
- continue
- grouped = _group_by_section(section_images)
- unique_count = len({r["filename"] for r in section_images})
- drug_records.append(
- {
- "drug_id": entry["drug_id"],
- "name": entry["name"],
- "pinyin": entry.get("pinyin", ""),
- "category": entry.get("category", ""),
- "subcategory": entry.get("subcategory", ""),
- "volume": entry["source"]["volume"],
- "source_file": fp,
- "total_image_count": unique_count,
- "sections_with_images": grouped,
- }
- )
- drugs_with_images += 1
- total_unique_images += unique_count
- if i % 500 == 0:
- elapsed = time.time() - t0
- rate = i / elapsed if elapsed > 0 else 0
- print(
- f" … {i:>5d}/{len(files)} | "
- f"{rate:5.1f} docs/s | "
- f"{drugs_with_images} drugs with images | "
- f"{total_unique_images} images"
- )
- # ── summary ─────────────────────────────────────────────────────────
- elapsed = time.time() - t0
- # rough disk usage
- du_bytes = 0
- try:
- for dirpath, _dirnames, filenames in os.walk(IMAGES_DIR):
- for fn in filenames:
- fp = os.path.join(dirpath, fn)
- try:
- du_bytes += os.path.getsize(fp)
- except OSError:
- pass
- except OSError:
- pass
- print()
- print("=" * 60)
- print("EXTRACTION COMPLETE")
- print(f" Files scanned: {len(files)}")
- print(f" Parsed successfully: {drugs_parsed}")
- print(f" Skipped (parse errors): {skipped}")
- print(f" Drugs with images: {drugs_with_images}")
- print(f" Total unique images: {total_unique_images}")
- print(f" Disk usage: {du_bytes / 1024 / 1024:.1f} MiB")
- print(f" Elapsed: {elapsed:.0f}s "
- f"({elapsed / 60:.1f} min)")
- print("=" * 60)
- # ── write manifest ──────────────────────────────────────────────────
- manifest = {
- "meta": {
- "generated_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
- "source_dir": SOURCE_DIR,
- "images_dir": IMAGES_DIR,
- "image_url_prefix": IMAGE_URL_PREFIX,
- "total_drugs_parsed": drugs_parsed,
- "total_drugs_with_images": drugs_with_images,
- "total_unique_images": total_unique_images,
- "skipped_files": skipped,
- },
- "drugs": drug_records,
- }
- with open(MANIFEST_FILE, "w", encoding="utf-8") as fh:
- json.dump(manifest, fh, ensure_ascii=False, indent=2)
- print(f"Manifest written to {MANIFEST_FILE}")
- if __name__ == "__main__":
- main()
|