| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452 |
- # -*- coding: utf-8 -*-
- """
- Diagnostic script: Analyze why "penicillin allergy" query returns info about Qingkailing injection.
- Traces the full RAG pipeline:
- User query -> Intent classification -> Vector retrieval -> Reranking -> LLM generation
- """
- import sys
- import os
- import json
- import re
- import asyncio
- import argparse
- from pathlib import Path
- sys.path.insert(0, str(Path(__file__).resolve().parent / "backend-python"))
- sys.path.insert(0, str(Path(__file__).resolve().parent / "data-pipeline"))
- def classify_intent(query: str) -> str:
- q = query.strip()
- negation_patterns = [
- r"不是", r"没有", r"并非", r"不算", r"不属于",
- r"这不是", r"我没有", r"不包含", r"不涉及",
- ]
- has_negation = any(re.search(pat, q) for pat in negation_patterns)
- if has_negation:
- return "drug_query"
- usage_keywords = [
- "怎么吃", "吃多少", "怎么用", "一天几次", "多长时间",
- "能一起吃", "孕妇能用", "儿童用量", "哺乳期",
- "饭前还是饭后", "空腹", "过量", "漏服", "停药",
- "副作用多大", "伤肝吗", "伤肾吗", "安全吗",
- ]
- safety_sections = [
- "副作用", "不良反应", "禁忌", "注意事项",
- "能不能", "可以吗", "会不会",
- ]
- regulation_keywords = [
- "凡例", "通则规定", "制剂通则",
- "一般规定", "通用技术要求", "检验方法通则",
- ]
- exam_keywords = [
- "执业药师考试", "考点", "历年真题", "考试大纲",
- "高频考点", "报名时间",
- ]
- symptom_keywords = [
- "吃了什么药", "吃什么药", "该吃", "推荐用药", "推荐下用药",
- "买什么药", "推荐什么药", "用什么药", "用药建议",
- "发烧", "咳嗽", "感冒", "腹泻", "头疼", "头痛",
- "嗓子疼", "流鼻涕", "鼻塞", "肚子疼", "胃疼",
- "过敏", "皮肤痒", "失眠", "便秘", "牙疼",
- "体温", "多少度", "退烧", "止痛", "止泻",
- ]
- if any(kw in q for kw in usage_keywords):
- return "usage_guide"
- if any(kw in q for kw in safety_sections):
- return "usage_guide"
- if any(kw in q for kw in symptom_keywords):
- return "symptom_advice"
- if any(kw in q for kw in regulation_keywords):
- return "regulation"
- if any(kw in q for kw in exam_keywords):
- return "exam_tutor"
- return "drug_query"
- def tokenize_query(query: str) -> set:
- tokens = set()
- for n in range(2, 5):
- for i in range(len(query) - n + 1):
- seg = query[i:i + n]
- if all('一' <= c <= '鿿' for c in seg):
- tokens.add(seg)
- for word in re.findall(r'[a-zA-Z0-9]+', query):
- tokens.add(word.lower())
- return tokens
- def keyword_overlap_score(query_terms: set, content: str) -> float:
- if not query_terms:
- return 0.0
- matched = sum(1 for t in query_terms if t in content)
- return matched / len(query_terms)
- def analyze_data_files():
- data_dir = Path(__file__).resolve().parent / "data-pipeline" / "data"
- results = {
- "penicillin_drugs": [],
- "qingkailing_drugs": [],
- "penicillin_allergy_mentions": [],
- "allergy_related_drugs": [],
- "all_drugs_with_allergy_sections": [],
- }
- for json_file in sorted(data_dir.glob("*.json")):
- try:
- with open(json_file, "r", encoding="utf-8") as f:
- data = json.load(f)
- except Exception:
- continue
- if not isinstance(data, list):
- continue
- for item in data:
- if not isinstance(item, dict):
- continue
- name = item.get("name", "")
- sections = item.get("sections", {})
- if "青霉素" in name:
- results["penicillin_drugs"].append({
- "file": json_file.name, "name": name,
- "sections": list(sections.keys()),
- "allergy_info": any("过敏" in str(v) for v in sections.values()),
- })
- if "清开灵" in name:
- results["qingkailing_drugs"].append({
- "file": json_file.name, "name": name,
- "sections": list(sections.keys()),
- "allergy_info": any("过敏" in str(v) for v in sections.values()),
- })
- for sk, sv in sections.items():
- if "青霉素" in str(sv) and "过敏" in str(sv):
- results["penicillin_allergy_mentions"].append({
- "file": json_file.name, "name": name, "section": sk,
- })
- for sk, sv in sections.items():
- if sk in ("禁忌", "不良反应", "注意事项") and "过敏" in str(sv):
- if name not in [x["name"] for x in results["all_drugs_with_allergy_sections"]]:
- results["all_drugs_with_allergy_sections"].append({
- "file": json_file.name, "name": name,
- })
- return results
- def analyze_chunk_examples():
- """Simulate what chunks look like in the database for key drugs."""
- data_dir = Path(__file__).resolve().parent / "data-pipeline" / "data"
- examples = []
- for json_file in sorted(data_dir.glob("*.json")):
- try:
- with open(json_file, "r", encoding="utf-8") as f:
- data = json.load(f)
- except Exception:
- continue
- if not isinstance(data, list):
- continue
- for item in data:
- if not isinstance(item, dict):
- continue
- name = item.get("name", "")
- if "青霉素" in name or "清开灵" in name:
- sections = item.get("sections", {})
- source = item.get("source", {})
- source_str = f"{source.get('version', '')} {source.get('volume', '')}"
- for sk, sv in sections.items():
- if not sv or len(sv.strip()) < 10:
- continue
- chunk = f"[{name} - {sk}]\n{sv}\n\nSource: {source_str}"
- examples.append({
- "drug": name, "section": sk,
- "chunk_preview": chunk[:300] + "..." if len(chunk) > 300 else chunk,
- })
- return examples
- async def get_embedding(query: str) -> list:
- """Call DashScope text-embedding-v3 API to get embedding vector."""
- import httpx
- api_key = os.environ.get("QWEN_API_KEY", "")
- if not api_key:
- env_file = Path(__file__).resolve().parent / ".env"
- if env_file.exists():
- for line in open(env_file, encoding="utf-8"):
- line = line.strip()
- if line and not line.startswith("#") and "=" in line:
- key, _, val = line.partition("=")
- os.environ.setdefault(key.strip(), val.strip())
- api_key = os.environ.get("QWEN_API_KEY", "")
- if not api_key:
- return []
- EMBEDDING_URL = "https://dashscope.aliyuncs.com/api/v1/services/embeddings/text-embedding/text-embedding"
- async with httpx.AsyncClient(timeout=30) as client:
- resp = await client.post(
- EMBEDDING_URL,
- headers={
- "Content-Type": "application/json",
- "Authorization": f"Bearer {api_key}",
- },
- json={
- "model": "text-embedding-v3",
- "input": {"texts": [query]},
- "parameters": {"text_type": "query"},
- },
- )
- data = resp.json()
- if data.get("code") and data.get("code") != "":
- raise RuntimeError(f"Embedding error: {data.get('message')}")
- return data["output"]["embeddings"][0]["embedding"]
- async def analyze_embedding_space():
- queries = [
- "青霉素过敏处理",
- "青霉素过敏",
- "过敏反应处理",
- "清开灵注射液不良反应",
- "清开灵注射液禁忌",
- "清开灵注射液过敏",
- "阿莫西林过敏",
- "头孢克肟过敏",
- ]
- print("\n" + "=" * 70)
- print("[Embedding Space Analysis]")
- print("=" * 70)
- api_key = os.environ.get("QWEN_API_KEY", "")
- if not api_key:
- env_file = Path(__file__).resolve().parent / ".env"
- if env_file.exists():
- for line in open(env_file, encoding="utf-8"):
- line = line.strip()
- if line and not line.startswith("#") and "=" in line:
- key, _, val = line.partition("=")
- os.environ.setdefault(key.strip(), val.strip())
- api_key = os.environ.get("QWEN_API_KEY", "")
- if not api_key:
- print(" SKIP: QWEN_API_KEY not set")
- return {}
- import math
- embeddings = {}
- for q in queries:
- try:
- vec = await get_embedding(q)
- embeddings[q] = vec
- print(f" OK: {q} (dim={len(vec)})")
- except Exception as e:
- print(f" FAIL: {q}: {e}")
- if not embeddings:
- return {}
- def cosine_sim(a, b):
- dot = sum(x * y for x, y in zip(a, b))
- na = math.sqrt(sum(x * x for x in a))
- nb = math.sqrt(sum(y * y for y in b))
- return dot / (na * nb) if na and nb else 0
- print("\nCosine Similarity Matrix:")
- header = [q[:10] for q in queries if q in embeddings]
- print(f"{'':22s}", end="")
- for h in header:
- print(f"{h:>10s}", end="")
- print()
- for q1 in queries:
- if q1 not in embeddings:
- continue
- print(f"{q1[:22]:22s}", end="")
- for q2 in queries:
- if q2 not in embeddings:
- continue
- sim = cosine_sim(embeddings[q1], embeddings[q2])
- print(f"{sim:10.4f}", end="")
- print()
- target = "青霉素过敏处理"
- if target in embeddings:
- print(f"\nSimilarity from '{target}' to others:")
- others = sorted(
- [(q, cosine_sim(embeddings[target], embeddings[q]))
- for q in embeddings if q != target],
- key=lambda x: -x[1]
- )
- for q, sim in others:
- bar = "=" * int(sim * 30)
- print(f" {q[:30]:30s}: {sim:.4f} {bar}")
- return embeddings
- def main():
- print("=" * 70)
- print("DIAGNOSIS: Why 'penicillin allergy' returns Qingkailing injection info")
- print("=" * 70)
- # === Step 1: Intent Classification ===
- query = "青霉素过敏处理"
- intent = classify_intent(query)
- print(f"\n[Step 1] Intent Classification:")
- print(f" Query: '{query}'")
- print(f" Intent: '{intent}'")
- print(f" Matched keyword '过敏' in symptom_keywords")
- print(f" >> PROBLEM: Query about managing a known allergy is routed to")
- print(f" 'symptom_advice' (symptom-based drug recommendation).")
- print(f" This tells LLM to recommend drugs, not manage an allergy.")
- # Also test alternative queries
- alt_queries = ["青霉素过敏了怎么办", "青霉素过敏急救", "对青霉素过敏", "青霉素过敏替代药"]
- print(f"\n Alternative query intents:")
- for q in alt_queries:
- print(f" '{q}' -> '{classify_intent(q)}'")
- # === Step 2: Query Tokenization (for reranker) ===
- tokens = tokenize_query(query)
- print(f"\n[Step 2] Reranker Tokenization:")
- print(f" Tokens ({len(tokens)}): {sorted(tokens)}")
- print(f" >> PROBLEM: '过敏' is one of the most common terms in drug")
- print(f" adverse reaction sections. Any chunk containing '过敏'")
- print(f" gets a keyword score boost in reranking.")
- print(f" '青霉素' (2-gram) only appears in penicillin-related chunks.")
- # === Step 3: Data file analysis ===
- print(f"\n[Step 3] Data File Analysis:")
- data_results = analyze_data_files()
- for category, items in data_results.items():
- if items:
- print(f"\n [{category}] ({len(items)} entries):")
- for item in items[:15]:
- if isinstance(item, dict):
- name = item.get("name", "?")
- fname = item.get("file", "?")
- print(f" - {name} (in {fname})", end="")
- if "section" in item:
- print(f" [section: {item['section']}]")
- elif "sections" in item:
- print(f" [sections: {item.get('sections', [])}]")
- else:
- print()
- if not data_results.get("qingkailing_drugs"):
- print(f"\n >> KEY FINDING: No Qingkailing injection content in local data files!")
- print(f" Qingkailing only appears in drug_index.json (catalog index).")
- print(f" If the production API returns Qingkailing info, it means:")
- print(f" 1) Full pharmacopoeia DOCX data (2280+ entries) was ingested")
- print(f" 2) Qingkailing chunks were matched by vector similarity")
- # === Step 4: Chunk examples ===
- chunks = analyze_chunk_examples()
- if chunks:
- print(f"\n[Step 4] Example Chunks from Data Files ({len(chunks)} found):")
- for c in chunks[:10]:
- print(f"\n --- {c['drug']} / {c['section']} ---")
- print(f" {c['chunk_preview']}")
- # === Step 5: Reranker simulation ===
- print(f"\n[Step 5] Reranker Simulation (query='{query}'):")
- # Simulate keyword overlap for sample chunks
- sample_contents = [
- ("阿莫西林-禁忌", "对青霉素类药物过敏者或青霉素皮肤试验阳性患者禁用。"),
- ("青霉素钠-不良反应", "过敏反应较常见,包括荨麻疹等各类皮疹、白细胞减少、间质性肾炎、哮喘发作等;偶见过敏性休克。"),
- ("清开灵注射液-不良反应", "偶有过敏反应,可见皮疹、瘙痒等。本品可能引起严重过敏反应,用药期间应注意观察。"),
- ("感冒用药指引-常用药物", "1. 对乙酰氨基酚:成人一次0.3-0.6g..."),
- ("过敏用药指引-用药原则", "首选第二代抗组胺药...过敏性休克需立即注射肾上腺素并送急救。"),
- ]
- for name, content in sample_contents:
- score = keyword_overlap_score(tokens, content)
- print(f" {name}: keyword_score={score:.2f} (content: {content[:60]}...)")
- # === Step 6: Embedding analysis ===
- print(f"\n[Step 6] Embedding Space Analysis:")
- print(f" Run with --embedding flag to call DashScope API for vector comparison.")
- # === DIAGNOSIS SUMMARY ===
- print("\n" + "=" * 70)
- print("ROOT CAUSE ANALYSIS")
- print("=" * 70)
- print("""
- ROOT CAUSE 1 [MOST CRITICAL]: Intent Classification Error
- - "青霉素过敏处理" is classified as "symptom_advice" because
- "过敏" appears in symptom_keywords.
- - symptom_advice uses a prompt template that tells LLM to
- recommend drugs for symptoms.
- - The correct classification should be "usage_guide" (safety
- consultation) which focuses on contraindications and precautions.
- - "过敏" was intentionally removed from safety_sections in the
- code, but the symptom_keywords trigger takes priority anyway.
- ROOT CAUSE 2: Vector Embedding Semantic Confusion
- - text-embedding-v3 is a general-purpose model, not fine-tuned
- for pharmaceutical domain.
- - Query "penicillin allergy management" and chunks like
- "Qingkailing injection - adverse reactions: may cause allergies"
- are semantically similar in the embedding space because both
- involve "drug + allergy/adverse reaction" domain.
- ROOT CAUSE 3: Data Imbalance + Chunking Strategy
- - TCM (一部) has 2280+ drugs, chemical drugs (二部) far fewer.
- - Each drug's sections (adverse reactions, contraindications,
- precautions) are vectorized independently.
- - Sections mentioning "allergy" cluster in embedding space,
- so any allergy query matches many unrelated drugs.
- ROOT CAUSE 4: Weak Reranker
- - Current reranker only does keyword overlap + dedup.
- - Cross-Encoder reranking (BGE-Reranker-v2-m3) is NOT implemented
- (the _load_model method is empty).
- - As a result, top-20 vector results cannot be semantically
- re-ranked for true relevance.
- ROOT CAUSE 5: LLM Web Search Amplification
- - enable_web_search=True may introduce external info about
- drugs not directly relevant to the user's query.
- """)
- print("=" * 70)
- print("RECOMMENDED FIXES (in priority order)")
- print("=" * 70)
- print("""
- 1. [INTENT] Add negation and known-allergy detection:
- - "XX过敏" with no symptom description -> usage_guide or drug_query
- - Add pattern: r'.+过敏(处理|怎么办|急救|替代)' -> usage_guide
- - Distinguish "我过敏了该吃什么药" vs "我对青霉素过敏怎么办"
- 2. [RETRIEVAL] Add query expansion:
- - "青霉素过敏处理" -> add "青霉素 禁忌 替代药物 交叉过敏"
- - Or add pre-filtering: if query mentions a specific drug name,
- boost chunks where drug_name matches.
- 3. [RERANKER] Implement Cross-Encoder:
- - Load BGE-Reranker-v2-m3 model in Reranker._load_model()
- - Use it for semantic re-ranking of top-20 results to top-5
- 4. [PROMPT] Fix SYMPTOM_ADVICE template:
- - Add constraint: "If user asks about a KNOWN drug allergy,
- provide allergy management and alternative drug info,
- NOT recommend new drugs for treating the allergy symptom."
- 5. [DATA] Add drug_name weight:
- - In vector search, add a metadata filter or hybrid scoring
- that boosts chunks whose drug_name contains query terms.
- 6. [EVAL] Add test cases:
- - "青霉素过敏处理" should return content about penicillin
- alternatives (cephalosporins, macrolides, etc.), NOT
- random TCM drugs that mention "allergy" in their sections.
- """)
- if __name__ == "__main__":
- main()
|