main1.py 140 KB

12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273747576777879808182838485868788899091929394959697989910010110210310410510610710810911011111211311411511611711811912012112212312412512612712812913013113213313413513613713813914014114214314414514614714814915015115215315415515615715815916016116216316416516616716816917017117217317417517617717817918018118218318418518618718818919019119219319419519619719819920020120220320420520620720820921021121221321421521621721821922022122222322422522622722822923023123223323423523623723823924024124224324424524624724824925025125225325425525625725825926026126226326426526626726826927027127227327427527627727827928028128228328428528628728828929029129229329429529629729829930030130230330430530630730830931031131231331431531631731831932032132232332432532632732832933033133233333433533633733833934034134234334434534634734834935035135235335435535635735835936036136236336436536636736836937037137237337437537637737837938038138238338438538638738838939039139239339439539639739839940040140240340440540640740840941041141241341441541641741841942042142242342442542642742842943043143243343443543643743843944044144244344444544644744844945045145245345445545645745845946046146246346446546646746846947047147247347447547647747847948048148248348448548648748848949049149249349449549649749849950050150250350450550650750850951051151251351451551651751851952052152252352452552652752852953053153253353453553653753853954054154254354454554654754854955055155255355455555655755855956056156256356456556656756856957057157257357457557657757857958058158258358458558658758858959059159259359459559659759859960060160260360460560660760860961061161261361461561661761861962062162262362462562662762862963063163263363463563663763863964064164264364464564664764864965065165265365465565665765865966066166266366466566666766866967067167267367467567667767867968068168268368468568668768868969069169269369469569669769869970070170270370470570670770870971071171271371471571671771871972072172272372472572672772872973073173273373473573673773873974074174274374474574674774874975075175275375475575675775875976076176276376476576676776876977077177277377477577677777877978078178278378478578678778878979079179279379479579679779879980080180280380480580680780880981081181281381481581681781881982082182282382482582682782882983083183283383483583683783883984084184284384484584684784884985085185285385485585685785885986086186286386486586686786886987087187287387487587687787887988088188288388488588688788888989089189289389489589689789889990090190290390490590690790890991091191291391491591691791891992092192292392492592692792892993093193293393493593693793893994094194294394494594694794894995095195295395495595695795895996096196296396496596696796896997097197297397497597697797897998098198298398498598698798898999099199299399499599699799899910001001100210031004100510061007100810091010101110121013101410151016101710181019102010211022102310241025102610271028102910301031103210331034103510361037103810391040104110421043104410451046104710481049105010511052105310541055105610571058105910601061106210631064106510661067106810691070107110721073107410751076107710781079108010811082108310841085108610871088108910901091109210931094109510961097109810991100110111021103110411051106110711081109111011111112111311141115111611171118111911201121112211231124112511261127112811291130113111321133113411351136113711381139114011411142114311441145114611471148114911501151115211531154115511561157115811591160116111621163116411651166116711681169117011711172117311741175117611771178117911801181118211831184118511861187118811891190119111921193119411951196119711981199120012011202120312041205120612071208120912101211121212131214121512161217121812191220122112221223122412251226122712281229123012311232123312341235123612371238123912401241124212431244124512461247124812491250125112521253125412551256125712581259126012611262126312641265126612671268126912701271127212731274127512761277127812791280128112821283128412851286128712881289129012911292129312941295129612971298129913001301130213031304130513061307130813091310131113121313131413151316131713181319132013211322132313241325132613271328132913301331133213331334133513361337133813391340134113421343134413451346134713481349135013511352135313541355135613571358135913601361136213631364136513661367136813691370137113721373137413751376137713781379138013811382138313841385138613871388138913901391139213931394139513961397139813991400140114021403140414051406140714081409141014111412141314141415141614171418141914201421142214231424142514261427142814291430143114321433143414351436143714381439144014411442144314441445144614471448144914501451145214531454145514561457145814591460146114621463146414651466146714681469147014711472147314741475147614771478147914801481148214831484148514861487148814891490149114921493149414951496149714981499150015011502150315041505150615071508150915101511151215131514151515161517151815191520152115221523152415251526152715281529153015311532153315341535153615371538153915401541154215431544154515461547154815491550155115521553155415551556155715581559156015611562156315641565156615671568156915701571157215731574157515761577157815791580158115821583158415851586158715881589159015911592159315941595159615971598159916001601160216031604160516061607160816091610161116121613161416151616161716181619162016211622162316241625162616271628162916301631163216331634163516361637163816391640164116421643164416451646164716481649165016511652165316541655165616571658165916601661166216631664166516661667166816691670167116721673167416751676167716781679168016811682168316841685168616871688168916901691169216931694169516961697169816991700170117021703170417051706170717081709171017111712171317141715171617171718171917201721172217231724172517261727172817291730173117321733173417351736173717381739174017411742174317441745174617471748174917501751175217531754175517561757175817591760176117621763176417651766176717681769177017711772177317741775177617771778177917801781178217831784178517861787178817891790179117921793179417951796179717981799180018011802180318041805180618071808180918101811181218131814181518161817181818191820182118221823182418251826182718281829183018311832183318341835183618371838183918401841184218431844184518461847184818491850185118521853185418551856185718581859186018611862186318641865186618671868186918701871187218731874187518761877187818791880188118821883188418851886188718881889189018911892189318941895189618971898189919001901190219031904190519061907190819091910191119121913191419151916191719181919192019211922192319241925192619271928192919301931193219331934193519361937193819391940194119421943194419451946194719481949195019511952195319541955195619571958195919601961196219631964196519661967196819691970197119721973197419751976197719781979198019811982198319841985198619871988198919901991199219931994199519961997199819992000200120022003200420052006200720082009201020112012201320142015201620172018201920202021202220232024202520262027202820292030203120322033203420352036203720382039204020412042204320442045204620472048204920502051205220532054205520562057205820592060206120622063206420652066206720682069207020712072207320742075207620772078207920802081208220832084208520862087208820892090209120922093209420952096209720982099210021012102210321042105210621072108210921102111211221132114211521162117211821192120212121222123212421252126212721282129213021312132213321342135213621372138213921402141214221432144214521462147214821492150215121522153215421552156215721582159216021612162216321642165216621672168216921702171217221732174217521762177217821792180218121822183218421852186218721882189219021912192219321942195219621972198219922002201220222032204220522062207220822092210221122122213221422152216221722182219222022212222222322242225222622272228222922302231223222332234223522362237223822392240224122422243224422452246224722482249225022512252225322542255225622572258225922602261226222632264226522662267226822692270227122722273227422752276227722782279228022812282228322842285228622872288228922902291229222932294229522962297229822992300230123022303230423052306230723082309231023112312231323142315231623172318231923202321232223232324232523262327232823292330233123322333233423352336233723382339234023412342234323442345234623472348234923502351235223532354235523562357235823592360236123622363236423652366236723682369237023712372237323742375237623772378237923802381238223832384238523862387238823892390239123922393239423952396239723982399240024012402240324042405240624072408240924102411241224132414241524162417241824192420242124222423242424252426242724282429243024312432243324342435243624372438243924402441244224432444244524462447244824492450245124522453245424552456245724582459246024612462246324642465246624672468246924702471247224732474247524762477247824792480248124822483248424852486248724882489249024912492249324942495249624972498249925002501250225032504250525062507250825092510251125122513251425152516251725182519252025212522252325242525252625272528252925302531253225332534253525362537253825392540254125422543254425452546254725482549255025512552255325542555255625572558255925602561256225632564256525662567256825692570257125722573257425752576257725782579258025812582258325842585258625872588258925902591259225932594259525962597259825992600260126022603260426052606260726082609261026112612261326142615261626172618261926202621262226232624262526262627262826292630263126322633263426352636263726382639264026412642264326442645264626472648264926502651265226532654265526562657265826592660266126622663266426652666266726682669267026712672267326742675267626772678267926802681268226832684268526862687268826892690269126922693269426952696269726982699270027012702270327042705270627072708270927102711271227132714271527162717271827192720272127222723272427252726272727282729273027312732273327342735273627372738273927402741274227432744274527462747274827492750275127522753275427552756275727582759276027612762276327642765276627672768276927702771277227732774277527762777277827792780278127822783278427852786278727882789279027912792279327942795279627972798279928002801280228032804280528062807280828092810281128122813281428152816281728182819282028212822282328242825282628272828282928302831283228332834283528362837283828392840284128422843284428452846284728482849285028512852285328542855285628572858285928602861286228632864286528662867286828692870287128722873287428752876287728782879288028812882288328842885288628872888288928902891289228932894289528962897289828992900290129022903290429052906290729082909291029112912291329142915291629172918291929202921292229232924292529262927292829292930293129322933293429352936293729382939294029412942294329442945294629472948294929502951295229532954295529562957295829592960296129622963296429652966296729682969297029712972297329742975297629772978297929802981298229832984298529862987298829892990299129922993299429952996299729982999300030013002300330043005300630073008300930103011301230133014301530163017301830193020302130223023302430253026302730283029303030313032303330343035303630373038303930403041304230433044304530463047304830493050305130523053305430553056305730583059306030613062306330643065306630673068306930703071307230733074307530763077307830793080308130823083308430853086308730883089309030913092
  1. """
  2. 饿了么闪购 — 主入口 V1
  3. 步骤驱动,按步执行
  4. """
  5. import sys
  6. import time
  7. import re
  8. import json
  9. import cv2
  10. import numpy as np
  11. from pathlib import Path
  12. from typing import Optional
  13. sys.path.insert(0, str(Path(__file__).parent / "steps"))
  14. from ocr import OCR
  15. from executor import SafeExecutor
  16. from ai_helper1 import AIParser, parse_instructions, extract_value_after, detect_popup
  17. from ai_helper_vision_cross import VisionParser # 交叉验证版(双次识别+仲裁); 稳定版= ai_helper_vision1
  18. from db import save_record, get_existing_license
  19. from snapshot import collect_snapshot
  20. from human_touch import wrist_arc_pts, wrist_arc_pts_h, motionevent_drag, touchpipe_drag
  21. # ── 配置 ────────────────────────────────────────────────
  22. APP_PACKAGE = "me.ele"
  23. SCREENSHOT_DIR = Path(__file__).parent / "screenshots"
  24. # 标题过滤开关:
  25. # True = 按品牌/药品名/规格过滤,只采目标商品(正常模式)
  26. # False = 不过滤,直接采集OCR识别的全部结果入库(测试OCR准确率用)
  27. ENABLE_TITLE_FILTER = False
  28. # 详情页AI提取的店铺名(step4进店确认时更新,供Excel对比列表页OCR店名用)
  29. AI_SHOP_NAME = ""
  30. # 验证码/休息策略:
  31. # 一天内 ≥CAPTCHA_DAILY_LIMIT 次 → 立即停止采集并回告(风控可能封号)
  32. # 每次验证码解决后 → 休息 CAPTCHA_REST_MINUTES~CAPTCHA_REST_MAX_MINUTES 分钟(随机)
  33. # 每爬 ROW_REST_EVERY 条数据 → 休息 ROW_REST_MIN~ROW_REST_MAX 分钟(随机,期间回告防假死)
  34. CAPTCHA_DAILY_LIMIT = 8
  35. CAPTCHA_REST_MINUTES = 1 # 验证码后休息最短分钟数
  36. CAPTCHA_REST_MAX_MINUTES = 5 # 验证码后休息最长分钟数
  37. ROW_REST_EVERY = 50 # 每爬多少条休息一次
  38. ROW_REST_MIN = 10 # 50条整点休息最短分钟数
  39. ROW_REST_MAX = 17 # 50条整点休息最长分钟数
  40. # 进入商品页unknown(页面没加载出来)时的处理:
  41. # 每确认一次unknown → back一层+重新点击进入;累计UNKNOWN_REENTER_LIMIT次加载全失败 → 跳过本商品
  42. # 连续2个商品都这样 → step3白屏恢复:关App休息60~120s → 重启App(最多MAX_WHITE_RESTART次)→ 超限回告调度重派
  43. UNKNOWN_REENTER_LIMIT = 5
  44. MAX_WHITE_RESTART = 3 # 白屏重启恢复上限(超过→置重派标志,回告调度换设备,不终止任务数据)
  45. WHITE_SUCCESS_RESET_N = 5 # 连续成功采N个商品后清零白屏重启计数(额度恢复)
  46. WHITE_EXCEPTION_TYPE = 5 # 白屏重派回告exception_type(复用通用异常值5,remark区分具体原因)
  47. SEARCH_ROUNDS = 5 # step2搜索失败重启重试轮数(共5轮,第2轮起每轮前重启App,全部失败才报错)
  48. CAPTCHA_LOG_DIR = Path(__file__).parent / "logs" # 验证码日志目录(每设备独立一个文件,独立计数)
  49. CAPTCHA_ABORTED = False # 全局标志:验证码频繁触发停止采集(调度上报用)
  50. CAPTCHA_ABORT_REASON = "" # 停止原因(调度回告用)
  51. ACCOUNT_ABORTED = False # 全局标志:账号被踢/封号,停止采集(调度上报用)
  52. WHITE_SCREEN_REASSIGN = False # 全局标志:商品页反复白屏,停止采集并回告调度重派(调度上报用)
  53. WHITE_SCREEN_REASSIGN_REASON = "" # 重派原因(调度回告remark用)
  54. _LAST_SKIP_WHITE = False # step4最后一次__SKIP__是否白屏(日志标注用)
  55. CURRENT_PAGE = 0 # 当前页码(逐页回告/终止回告用,调度侧存续采集进度)
  56. # 验证码判定截图存档目录(每次判定出现验证码时截图保存,人工确认是否误判)
  57. CAPTCHA_CHECK_DIR = Path(__file__).parent / "logs" / "captcha_check"
  58. # 休息期间进度回告用的全局上下文(step3 任务开始时设置,_captcha_rest 休息时每10分钟回告)
  59. _SCHEDULER = None
  60. _TASK_ID = None
  61. _CRAWLED_COUNT = 0
  62. _ROW_REST_MARK = 0
  63. OCR = OCR()
  64. def human_sleep(seconds: float, jitter: float = 0.3):
  65. """拟人等待:标称时长 ±30% 随机抖动(操作节奏不被风控建模成固定周期)。
  66. 仅用于 UI 操作间的节奏等待;功能性等待(回告间隔/截图重试)仍用 time.sleep 精确控制。"""
  67. import random as _random
  68. try:
  69. base = float(seconds)
  70. except (TypeError, ValueError):
  71. base = 1.0
  72. time.sleep(max(0.1, base * _random.uniform(1 - jitter, 1 + jitter)))
  73. # 说明书打叉坐标缓存(仅本任务内有效):任务中第一个商品模板匹配找到打叉后,
  74. # 本任务后续商品直接复用该坐标(说明书页布局固定,同设备位置不变);
  75. # 新任务开始时清空,重新匹配。
  76. _CLOSE_BTN_CACHE = None
  77. def _find_device(device_id: str = "") -> str:
  78. import subprocess
  79. r = subprocess.run(["adb", "devices"], capture_output=True, text=True, timeout=5)
  80. devices = []
  81. for line in r.stdout.strip().split("\n")[1:]:
  82. if line.strip() and "device" in line and "offline" not in line:
  83. s = line.split("\t")[0].strip()
  84. if s:
  85. devices.append(s)
  86. if not devices:
  87. raise RuntimeError("未找到设备")
  88. # 如果指定了设备ID,精确匹配
  89. if device_id:
  90. for d in devices:
  91. if d == device_id:
  92. return d
  93. raise RuntimeError(f"未找到指定设备: {device_id},可用设备: {devices}")
  94. # 只有一台直接返回
  95. if len(devices) == 1:
  96. return devices[0]
  97. # 多台设备:列出并让用户选择
  98. print(f"\n发现 {len(devices)} 台设备:")
  99. for i, d in enumerate(devices):
  100. print(f" [{i}] {d}")
  101. while True:
  102. try:
  103. choice = input(f"请选择设备 [0-{len(devices)-1}],回车默认第一台: ").strip()
  104. if choice == "":
  105. return devices[0]
  106. idx = int(choice)
  107. if 0 <= idx < len(devices):
  108. return devices[idx]
  109. except ValueError:
  110. pass
  111. print(f"输入无效,请输入 0-{len(devices)-1}")
  112. def _find_text_in_area(shot_path: str, target: str, max_y: int) -> Optional[dict]:
  113. results = OCR.recognize(shot_path, detail="all")
  114. for r in results:
  115. if target in r["text"]:
  116. y = r["bbox"][0][1]
  117. if y < max_y:
  118. cx = r["bbox"][0][0] + (r["bbox"][2][0] - r["bbox"][0][0]) // 2
  119. cy = y + (r["bbox"][2][1] - y) // 2
  120. return {"x": cx, "y": cy, "text": r["text"], "conf": r["confidence"]}
  121. return None
  122. def _shot_path(ex: SafeExecutor, name: str) -> str:
  123. """截图路径:按 设备ID/步骤 分类组织目录(screenshots/设备/step1/xxx.png)"""
  124. cat = "misc"
  125. for prefix, c in (("step1", "step1"), ("step2", "step2"), ("step3", "step3"),
  126. ("_s4_", "step4"), ("sort", "step2"), ("inst_", "instructions"),
  127. ("lic_", "license"), ("snap_", "snapshot"), ("ad_", "popup")):
  128. if name.startswith(prefix):
  129. cat = c
  130. break
  131. d = SCREENSHOT_DIR / ex.device_id / cat
  132. d.mkdir(parents=True, exist_ok=True)
  133. return str(d / name)
  134. def _save_evidence(shot: str, tag: str):
  135. """决策留档: 触发重启/放弃时刻的截图另存时间戳副本。
  136. page_check/detail_check 同一家店内会被下一轮覆盖,
  137. 留档(时间戳+原因标签)便于事后人工核对AI是否误判。"""
  138. try:
  139. if shot and os.path.exists(shot):
  140. ts = time.strftime("%Y%m%d_%H%M%S")
  141. shutil.copy2(shot, str(Path(shot).parent / f"{ts}_{tag}.png"))
  142. except Exception:
  143. pass
  144. def _screenshot(ex: SafeExecutor, name: str) -> str:
  145. import os
  146. # 按 设备ID/步骤 隔离截图(目录已含设备ID,文件名不再加设备前缀)
  147. path = _shot_path(ex, name)
  148. if os.path.exists(path):
  149. # 保留历史截图副本:固定文件名被 test/调试脚本引用,不能被后续运行覆盖丢失
  150. import shutil
  151. stem, ext = os.path.splitext(name)
  152. backup = str(Path(path).parent / f"{stem}_{int(time.time() * 1000)}{ext}")
  153. try:
  154. shutil.copy2(path, backup)
  155. except Exception:
  156. pass
  157. # 自动清理:每个固定文件只保留最近3份备份,超出删除最旧的
  158. try:
  159. olds = sorted(
  160. Path(path).parent.glob(f"{stem}_[0-9]*{ext}"),
  161. key=lambda p: p.stat().st_mtime, reverse=True,
  162. )
  163. for p in olds[3:]:
  164. p.unlink(missing_ok=True)
  165. except Exception:
  166. pass
  167. # 截图后验证完整性(adb 流式传输可能中断,导致 PNG 损坏),连续5次失败才抛异常
  168. last_err = None
  169. for _try in range(5):
  170. try:
  171. ex.driver.screenshot(path)
  172. except Exception as e:
  173. last_err = e
  174. time.sleep(1)
  175. continue
  176. if cv2.imread(path) is not None:
  177. return path
  178. time.sleep(0.5)
  179. raise RuntimeError(f"截图连续5次损坏/失败: {name} ({last_err})")
  180. def _where_am_i(ex: SafeExecutor) -> str:
  181. """判断当前位置:list=搜索结果列表页 home=首页 other=其他(全屏OCR,区分列表页与首页防止退过头)"""
  182. import os as _os
  183. tmp = _shot_path(ex, "check_pos.png")
  184. ex.driver.screenshot(tmp)
  185. if cv2.imread(tmp) is None:
  186. ex.driver.screenshot(tmp)
  187. texts = [r["text"] for r in OCR.recognize(tmp, detail="all")]
  188. if any("筛选" in t for t in texts):
  189. return "list"
  190. if any("看病买药" in t for t in texts) and any("我的" in t for t in texts):
  191. return "home"
  192. return "other"
  193. # ── 步骤 1:打开 App ────────────────────────────────────
  194. def step1_open_app(ex: SafeExecutor) -> bool:
  195. print("=" * 40)
  196. print(" 步骤 1:打开饿了么闪购")
  197. print("=" * 40)
  198. w, h = ex.driver.window_size()
  199. print(f"[step1] 屏幕尺寸: {w}x{h}")
  200. print(f"[step1] 关闭 {APP_PACKAGE}...")
  201. ex.driver.app_stop(APP_PACKAGE)
  202. human_sleep(2)
  203. print(f"[step1] 启动 {APP_PACKAGE}...")
  204. ex.driver.app_start(APP_PACKAGE)
  205. human_sleep(5)
  206. # 等待首页加载:底部导航「我的」出现(元素没加载完就多等重试,不一次定生死)
  207. for attempt in range(3):
  208. shot = _screenshot(ex, "step1_home.png")
  209. texts = OCR.recognize(shot, rect=[0, int(h * 0.88), w, h], detail="text")
  210. print(f"[step1] 底部识别(第{attempt+1}次): {texts}")
  211. if any("我的" in t for t in texts):
  212. print("[step1] OK - 成功进入 App")
  213. _close_ad_popup(ex) # 每天首启的广告弹窗
  214. return True
  215. human_sleep(3)
  216. print("[step1] FAIL - 多次重试未检测到「我的」")
  217. # 识别不到"我的"可能因为账号被踢/封号——检测登录页xpath确认(存在才是封号)
  218. _check_account_kicked(ex)
  219. return False
  220. # ── 步骤 2:搜索商品 ────────────────────────────────────
  221. def _click_low_price_sort(ex: SafeExecutor) -> bool:
  222. """
  223. 搜索结果排序:点「综合」→ 选「低价优先」(失败不阻塞,找不到就跳过)。
  224. 找不到「低价优先」时再点一次「综合」重试,仍找不到则关闭排序弹窗。
  225. 备用函数:默认不调用(排序已在 step2 快递分支内联),需要按低价优先采集时手动启用。
  226. """
  227. try:
  228. # 1. 找「综合」并点击(排序入口,结果页顶部)
  229. shot = _shot_path(ex, "sort1.png")
  230. ex.driver.screenshot(shot)
  231. btn = None
  232. for r in OCR.recognize(shot, detail="all"):
  233. if "综合" in r["text"]:
  234. box = r["bbox"]
  235. btn = {"x": (box[0][0] + box[2][0]) // 2, "y": (box[0][1] + box[2][1]) // 2}
  236. break
  237. if not btn:
  238. print(" ⚠ 未找到「综合」排序入口,跳过排序")
  239. return False
  240. print(f" 排序: 点击「综合」({btn['x']}, {btn['y']})")
  241. ex.tap(btn["x"], btn["y"])
  242. human_sleep(2)
  243. # 2. 找「低价优先」
  244. shot2 = _shot_path(ex, "sort2.png")
  245. ex.driver.screenshot(shot2)
  246. low = None
  247. for r in OCR.recognize(shot2, detail="all"):
  248. if "低价优先" in r["text"]:
  249. box = r["bbox"]
  250. low = {"x": (box[0][0] + box[2][0]) // 2, "y": (box[0][1] + box[2][1]) // 2}
  251. break
  252. if low:
  253. print(f" 排序: 点击「低价优先」({low['x']}, {low['y']})")
  254. ex.tap(low["x"], low["y"])
  255. human_sleep(3)
  256. return True
  257. # 3. 没找到:再点一次「综合」关闭排序弹窗(点击综合弹出面板,再点一次收起)
  258. print(" ⚠ 未找到「低价优先」,再点一次「综合」关闭排序弹窗")
  259. ex.tap(btn["x"], btn["y"])
  260. human_sleep(1.5)
  261. return False
  262. except Exception as e:
  263. print(f" ⚠ 排序异常: {e}")
  264. return False
  265. def _search_once(ex: SafeExecutor, keyword: str) -> bool:
  266. """单轮搜索(原step2_search主体): 看病买药→搜索→输入→结果确认, 失败返回False"""
  267. print("\n" + "=" * 40)
  268. print(f" 步骤 2:搜索「{keyword}」")
  269. print("=" * 40)
  270. global _POPUP_RESTART_COUNT
  271. _POPUP_RESTART_COUNT = 0 # 每次搜索重置弹窗重启计数(仅本步骤内最多3次)
  272. w, h = ex.driver.window_size()
  273. top_th = int(h * 0.3)
  274. # ── 阶段0+1:点击「看病买药」→ 找「搜索」(失败重试3次,3次失败重启App重试,最多2轮)──
  275. btn = None
  276. for round_idx in range(2):
  277. if round_idx > 0:
  278. # 第2轮:重启 App 重新进入(弹窗/页面异常时清状态)
  279. print("[step2] 重启App重新进入...")
  280. if not step1_open_app(ex):
  281. print("[step2] FAIL - 重启App失败")
  282. return False
  283. # 阶段0:点击「看病买药」
  284. shot0 = _screenshot(ex, "step2_phase0.png")
  285. btn_med = _find_text_in_area(shot0, "看病买药", h)
  286. if not btn_med:
  287. print(f"[step2] 未找到「看病买药」(第{round_idx+1}轮)")
  288. continue
  289. print(f"[step2] 找到「看病买药」: ({btn_med['x']}, {btn_med['y']})")
  290. ex.tap(btn_med["x"], btn_med["y"])
  291. human_sleep(5)
  292. _close_ad_popup(ex) # 点击买药后可能出现的广告弹窗
  293. # 阶段1:找「搜索」(重试3次,每次关弹窗+等待)
  294. for attempt in range(3):
  295. shot = _screenshot(ex, "step2_phase1.png")
  296. btn = _find_text_in_area(shot, "搜索", top_th)
  297. if btn:
  298. break
  299. print(f"[step2] 未找到「搜索」(第{attempt+1}次),关闭弹窗后重试")
  300. _close_ad_popup(ex)
  301. human_sleep(2)
  302. if btn:
  303. break # 找到搜索,继续
  304. print("[step2] 3次未找到「搜索」,准备重启App重试")
  305. if not btn:
  306. print("[step2] FAIL - 多次重试+重启后仍未找到「搜索」")
  307. return False
  308. print(f"[step2] 找到「搜索」: ({btn['x']}, {btn['y']})")
  309. cx = btn["x"] - 300 # 搜索左边约120px
  310. cy = btn["y"]
  311. print(f"[step2] 点击搜索栏: ({cx}, {cy})")
  312. ex.tap(cx, cy)
  313. human_sleep(5)
  314. # 搜索页确认:「搜索」位置变化才认为进入(加载慢/弹窗遮挡/点击未生效就重试)
  315. moved = False
  316. btn2 = None
  317. for attempt in range(3):
  318. try:
  319. shot2 = _screenshot(ex, "step2_phase2.png")
  320. btn2 = _find_text_in_area(shot2, "搜索", top_th)
  321. except Exception as e:
  322. print(f"[step2] 截图/识别异常(第{attempt+1}次): {e},重试")
  323. human_sleep(2)
  324. continue
  325. if btn2 and (abs(btn2["x"] - btn["x"]) > 50 or abs(btn2["y"] - btn["y"]) > 50):
  326. moved = True
  327. print(f"[step2] 搜索页搜索: ({btn2['x']}, {btn2['y']})")
  328. break
  329. # 位置没变:先关广告弹窗 + AI验证码检测,再重新点击搜索栏(可能点击没生效或被验证码拦截)
  330. if _ai_check_captcha(ex):
  331. print(f"[step2] AI检测到验证码(第{attempt+1}次),处理完成")
  332. if _close_ad_popup(ex):
  333. print(f"[step2] 已关闭广告弹窗(第{attempt+1}次)")
  334. print(f"[step2] 搜索位置未变化(第{attempt+1}次),重新点击搜索栏")
  335. ex.tap(cx, cy)
  336. human_sleep(4)
  337. if not btn2:
  338. print("[step2] FAIL - 进入搜索页后找不到「搜索」")
  339. return False
  340. if not moved:
  341. print("[step2] FAIL - 搜索位置未改变")
  342. return False
  343. cx2 = btn2["x"] - 180 # 搜索页输入框在搜索左边约180px
  344. cy2 = btn2["y"]
  345. ex.tap(cx2, cy2)
  346. human_sleep(2)
  347. print(f"[step2] 聚焦输入框,等待2s")
  348. print(f"[step2] 输入关键词: {keyword}")
  349. ex.driver.set_input_ime(True)
  350. time.sleep(0.3)
  351. ex.driver.send_keys(keyword)
  352. human_sleep(1)
  353. ex.tap(btn2["x"], btn2["y"])
  354. human_sleep(3)
  355. # 搜索后检测列表页:先处理验证码(可能挡住列表页),再检测列表页特征(加载慢就重试)
  356. for attempt in range(5):
  357. try:
  358. shot3 = _screenshot(ex, "step2_result.png")
  359. raw3 = OCR.recognize(shot3, detail="all")
  360. except Exception as e:
  361. print(f"[step2] 截图/识别异常(第{attempt+1}次): {e},重试")
  362. human_sleep(2)
  363. continue
  364. all_texts = [r["text"] for r in raw3]
  365. # 1. 验证码优先:验证码弹窗会挡住列表页特征,先处理再重新检测
  366. # 关键词没命中时用AI再判断一次(九宫格/点击式验证码文字不在关键词里)
  367. if any(("拖动滑块" in t) or ("请按住滑块" in t) or ("安全验证" in t) for t in all_texts):
  368. print(f"[step2] 检测到列表页验证码(第{attempt+1}次),尝试处理...")
  369. if _handle_captcha(ex, all_texts):
  370. print("[step2] 验证码已解决")
  371. else:
  372. print("[step2] 验证码处理失败")
  373. return False
  374. # 验证码通过后可能出现「出错了/检修中」页,点重新加载
  375. if _click_reload(ex):
  376. print("[step2] 已点击重新加载")
  377. human_sleep(2)
  378. continue
  379. if _ai_check_captcha(ex):
  380. print(f"[step2] AI检测到验证码(第{attempt+1}次),处理完成")
  381. continue
  382. # 1.2 广告/活动弹窗(全屏活动页——无遮罩无列表标记时才检测; 列表页上有"红包"字样是正常的)
  383. if not any(("筛选" in t or "快递" in t) for t in all_texts) and _close_ad_popup(ex, force=True):
  384. # back可能只是导航回了上一页(搜索建议页),页面状态已不可信——
  385. # 必须重发搜索(重新点搜索按钮),下一轮再确认结果页
  386. print(f"[step2] 已关闭活动弹窗(第{attempt+1}次),重发搜索")
  387. ex.tap(btn2["x"], btn2["y"])
  388. human_sleep(3)
  389. continue
  390. # 1.5 出错/检修页(无验证码时也可能出现):点「重新加载」后重新检测
  391. if any("重新加载" in t for t in all_texts):
  392. print(f"[step2] 检测到「出错了/检修中」页面,点击重新加载")
  393. _click_reload(ex)
  394. continue
  395. # 2. 列表页特征检测
  396. has_filter = "筛选" in all_texts
  397. has_express = "快递" in all_texts
  398. kw_found = any(keyword in t for t in all_texts)
  399. print(f"[step2] 有筛选: {has_filter}, 有快递: {has_express}, 关键词存在: {kw_found}")
  400. if has_filter or has_express:
  401. # 如果有「快递」则点击它
  402. if has_express:
  403. for r in raw3:
  404. if "快递" in r["text"]:
  405. bx = r["bbox"]
  406. cx = (bx[0][0] + bx[2][0]) // 2
  407. cy = (bx[0][1] + bx[2][1]) // 2
  408. print(f"[step2] 点击「快递」: ({cx}, {cy})")
  409. ex.tap(cx, cy)
  410. human_sleep(3)
  411. break
  412. # 点击「快递」后设置排序:综合 → 低价优先
  413. # _click_low_price_sort(ex) ← 需要按低价优先采集时,去掉行首#即可启用
  414. print("[step2] OK - 搜索成功")
  415. return True
  416. print(f"[step2] 列表页特征未出现(第{attempt+1}次),等3秒重试")
  417. human_sleep(3)
  418. print("[step2] FAIL - 搜索未成功")
  419. return False
  420. def step2_search(ex: SafeExecutor, keyword: str) -> bool:
  421. """搜索总入口:单轮搜索失败/异常 → 重启App重试,共SEARCH_ROUNDS轮,全部失败才报错"""
  422. for round_idx in range(SEARCH_ROUNDS):
  423. if round_idx > 0:
  424. print(f"\n[step2] 第{round_idx}轮搜索失败,重启App重试(重启{round_idx}/{SEARCH_ROUNDS - 1}次)")
  425. if not step1_open_app(ex):
  426. print("[step2] 重启App失败,直接报错")
  427. return False
  428. try:
  429. if _search_once(ex, keyword):
  430. return True
  431. print(f"[step2] 第{round_idx + 1}/{SEARCH_ROUNDS}轮搜索未成功")
  432. except Exception as e:
  433. # 截图损坏/识别异常/网络抖动等都按本轮失败处理,重启后再试
  434. print(f"[step2] 第{round_idx + 1}轮搜索异常({e}),重启后重试")
  435. print(f"[step2] FAIL - {SEARCH_ROUNDS}轮搜索(含重启)均未成功")
  436. return False
  437. def normalize_match_text(value):
  438. """归一化:统一全角/半角、去除空白和零宽字符(美团同款,避免'看起来一样但匹配失败')"""
  439. import unicodedata
  440. text = "" if value is None else str(value)
  441. text = unicodedata.normalize("NFKC", text)
  442. text = re.sub(r"[\s ​-‍]+", "", text)
  443. return text
  444. def _edit_distance(a: str, b: str) -> int:
  445. """编辑距离(Levenshtein),用于OCR错字容错"""
  446. if len(a) < len(b):
  447. a, b = b, a
  448. if not b:
  449. return len(a)
  450. prev = list(range(len(b) + 1))
  451. for i, ca in enumerate(a, 1):
  452. cur = [i]
  453. for j, cb in enumerate(b, 1):
  454. cur.append(min(prev[j] + 1, cur[j - 1] + 1, prev[j - 1] + (ca != cb)))
  455. prev = cur
  456. return prev[-1]
  457. def _match_fuzzy(target: str, text: str) -> bool:
  458. """
  459. 容错匹配:目标词是否在文本中(处理OCR漏字/错字1个)。
  460. 用于判断「疑似生僻字」——本地识别差1个字时触发云端验证。
  461. """
  462. if not target:
  463. return True
  464. if target in text:
  465. return True
  466. if len(target) >= 3:
  467. for i in range(len(target)): # 漏字:目标删任意1字后匹配
  468. if target[:i] + target[i + 1:] in text:
  469. return True
  470. for L in (len(target), len(target) + 1, len(target) - 1): # 错字:编辑距离<=1
  471. if L < 2:
  472. continue
  473. for i in range(len(text) - L + 1):
  474. if _edit_distance(target, text[i:i + L]) <= 1:
  475. return True
  476. return False
  477. def _correct_title_with_cloud(title: str, n_brand: str, n_key: str, cloud_texts: list) -> str:
  478. """
  479. 用云端文本修正标题中的生僻字错误(精确子串替换,不做整标题替换):
  480. 标题里 fuzzy 匹配到目标词的子串 → 替换为云端文本确认的准确形式。
  481. 只处理"漏字"场景(云端形式更长,如"理王"→"理洫王");
  482. 同长度的形近替换(如"温胃舒"→"养胃舒")【不修正】——
  483. 云端文本里的目标词可能来自其他商品,误用来改标题会造成错误采集。
  484. """
  485. c_joined = "".join(cloud_texts)
  486. nt = normalize_match_text(title)
  487. for target in (n_brand, n_key):
  488. if not target:
  489. continue
  490. if target in nt:
  491. continue # 已精确匹配,无需修正
  492. if target not in c_joined:
  493. continue # 云端也没有准确形式 → 无法修正,跳过
  494. # 在标题里找漏字子串并替换为准确形式。
  495. # 只处理"目标词比子串长且编辑距离≤1"(漏字:理王→理洫王)——
  496. # 同长度形近替换(温胃舒→养胃舒)不触发:云端文本里的目标词可能来自其他商品,
  497. # 误用来改标题会造成错误采集。
  498. for i in range(len(nt)):
  499. for L in (len(target) - 1, len(target), len(target) + 1):
  500. if L < 2 or i + L > len(nt):
  501. continue
  502. sub = nt[i:i + L]
  503. if len(target) > len(sub) and _edit_distance(target, sub) <= 1:
  504. nt = nt[:i] + target + nt[i + L:]
  505. break
  506. else:
  507. continue
  508. break
  509. return nt
  510. def _is_cjk(ch: str) -> bool:
  511. """是否中文字符"""
  512. return '一' <= ch <= '鿿'
  513. def _is_missing_char(target: str, title_text: str) -> bool:
  514. """
  515. 判断目标词是否在标题里以"漏1字"形式出现(生僻字特征):
  516. 目标词删掉1个字后的子串,在标题里是【独立词】(前后不是汉字)。
  517. 例:理洫王删"洫"="理王",标题"[理王]"里独立 → 漏字 ✓
  518. 例:养胃舒删"舒"="养胃",标题"滋阴养胃"里嵌在词中(前有"滋")→ 非独立 → 不算漏字 ✗
  519. """
  520. for k in range(len(target)):
  521. sub = target[:k] + target[k + 1:]
  522. if len(sub) < 2:
  523. continue
  524. idx = title_text.find(sub)
  525. while idx != -1:
  526. before_ok = idx == 0 or not _is_cjk(title_text[idx - 1])
  527. after_ok = idx + len(sub) >= len(title_text) or not _is_cjk(title_text[idx + len(sub)])
  528. if before_ok and after_ok:
  529. return True
  530. idx = title_text.find(sub, idx + 1)
  531. return False
  532. def _ensure_cloud(cloud_cache: dict, shot_path: str) -> None:
  533. """触发一次云端OCR(每批只调一次),保存 (文本, y) 带坐标的块列表"""
  534. if cloud_cache["done"]:
  535. return
  536. cloud_cache["done"] = True
  537. try:
  538. raw_c = OCR.recognize(shot_path, detail="all", engine="cloud")
  539. if not raw_c:
  540. # 云端返回空(配额用尽/限流时百度返回错误码但不抛异常)→ 视为云端不可用
  541. print("[step3] 云端返回空结果(可能配额不足/限流),按容错处理")
  542. cloud_cache["blocks"] = None
  543. cloud_cache["texts"] = None
  544. return
  545. cloud_cache["blocks"] = [
  546. (normalize_match_text(r["text"]), (r["bbox"][0][1] + r["bbox"][2][1]) // 2)
  547. for r in raw_c
  548. ]
  549. cloud_cache["texts"] = [t for t, _ in cloud_cache["blocks"]]
  550. print(f"[step3] 云端二次确认({len(cloud_cache['blocks'])}块)")
  551. except Exception as e:
  552. print(f"[step3] 云端识别失败({e})")
  553. cloud_cache["blocks"] = None
  554. cloud_cache["texts"] = None
  555. def _cloud_crop_confirm(shot_path: str, card_y: int) -> Optional[str]:
  556. """
  557. 用本地OCR的标题块坐标裁剪标题区域(标题y-20 ~ y+80,右列),放大2倍后云端识别。
  558. 裁剪区只含当前商品的标题 → 云端不需要返回坐标,标准版(无配额问题)即可用,且字放大识别更准。
  559. """
  560. try:
  561. img = cv2.imread(shot_path)
  562. if img is None:
  563. return None
  564. h, w = img.shape[:2]
  565. y1, y2 = max(0, card_y - 20), min(h, card_y + 80)
  566. x1 = int(w * 0.15)
  567. crop = img[y1:y2, x1:w]
  568. if crop.size == 0:
  569. return None
  570. big = cv2.resize(crop, None, fx=2, fy=2, interpolation=cv2.INTER_LANCZOS4)
  571. raw = OCR.recognize(big, detail="all", engine="cloud")
  572. texts = [normalize_match_text(r["text"]) for r in raw
  573. if any('一' <= c <= '鿿' for c in r["text"])]
  574. return "".join(texts) if texts else None
  575. except Exception as e:
  576. print(f" [调试] 裁剪云端确认异常: {e}")
  577. return None
  578. def _match_verify(title: str, n_brand: str, n_key: str, shot_path: str, cloud_cache: dict,
  579. card_y: int = None) -> tuple:
  580. """
  581. 品牌/药品名匹配(只判断品牌+药品名核心词,规格/功效文字不参与)。
  582. 返回 (判定, 不匹配原因):判定 "ok"/"fuzzy"/"fail",原因如 "品牌"/"药品名"/"品牌、药品名"
  583. - 本地精确匹配 → "ok"(免费)
  584. - 完全不像 → "fail"(免费,不花云端)
  585. - 差1字/漏字 → 用本地坐标裁剪该商品标题区域,云端识别确认:
  586. 裁剪区含目标词(本地认错字,如理王→理洫王/甲疏咪唑)→ "ok"
  587. 裁剪区不含目标词(确实不是该商品,如温胃舒vs养胃舒)→ "fail"
  588. - 裁剪确认不可用 → 漏字场景容错放行 "fuzzy",换字场景 "fail"
  589. """
  590. nt = normalize_match_text(title)
  591. brand_match = (not n_brand) or (n_brand in nt)
  592. key_match = (not n_key) or (n_key in nt)
  593. if brand_match and key_match:
  594. return "ok", ""
  595. def _unmatched_reason():
  596. parts = []
  597. if n_brand and not brand_match:
  598. parts.append("品牌")
  599. if n_key and not key_match:
  600. parts.append("药品名")
  601. return "、".join(parts) or "品牌/药品名"
  602. # 便宜判断:完全不像(非差1字也非漏字)→ 直接过滤,不花云端
  603. near_brand = (not n_brand) or _match_fuzzy(n_brand, nt) or _is_missing_char(n_brand, nt)
  604. near_key = (not n_key) or _match_fuzzy(n_key, nt) or _is_missing_char(n_key, nt)
  605. if not (near_brand and near_key):
  606. return "fail", _unmatched_reason()
  607. # 差1字/漏字 → 用本地坐标裁剪该商品标题区域,云端确认(标准版即可,无需坐标)
  608. if card_y is not None:
  609. crop_text = _cloud_crop_confirm(shot_path, card_y)
  610. if crop_text is not None:
  611. if (not n_brand or n_brand in crop_text) and (not n_key or n_key in crop_text):
  612. # 把裁剪确认文本交给标题修正逻辑(如"理王"→"理洫王")
  613. cloud_cache["texts"] = [crop_text]
  614. return "ok", ""
  615. print(f" [调试] 卡片y={card_y} 裁剪云端=[{crop_text[:40]}] 不含目标词")
  616. return "fail", _unmatched_reason()
  617. print(f" [调试] 卡片y={card_y} 裁剪云端识别失败")
  618. # 裁剪确认不可用 → 原逻辑:漏字(独立词)→ 整页云端兜底;换字 → 过滤
  619. missing = []
  620. if n_brand and not brand_match and _is_missing_char(n_brand, nt):
  621. missing.append("品牌")
  622. if n_key and not key_match and _is_missing_char(n_key, nt):
  623. missing.append("药品名")
  624. if not missing:
  625. return "fail", _unmatched_reason() # 换字 → 过滤(安全默认)
  626. # 漏字 → 整页云端兜底(cloud_cache 保证每批只调一次)
  627. _ensure_cloud(cloud_cache, shot_path)
  628. if cloud_cache["texts"] is None:
  629. return "fuzzy", ""
  630. c_joined = "".join(cloud_cache["texts"])
  631. if (not n_brand or n_brand in c_joined) and (not n_key or n_key in c_joined):
  632. return "ok", ""
  633. return "fail", _unmatched_reason()
  634. def _screen_state(texts: list) -> str:
  635. """
  636. 根据 OCR 文本判断屏幕状态(统一的状态识别,各步骤共用):
  637. detail=商品详情页 shop=店铺页 list=搜索结果列表页 unknown=其他
  638. """
  639. joined = "".join(texts)
  640. if any(("加入购物车" in t) or ("立即购买" in t) or ("选规格" in t) or ("加入购物袋" in t) for t in texts):
  641. return "detail"
  642. if ("刚刚搜过" in joined) and ("评价" in joined):
  643. return "shop"
  644. if "筛选" in joined:
  645. return "list"
  646. return "unknown"
  647. def _is_list_page(ex: SafeExecutor) -> bool:
  648. """当前是否在搜索结果列表页(全屏检测「筛选」,店铺页/详情页不含此词)"""
  649. import os as _os
  650. tmp = _shot_path(ex, "check_list.png")
  651. ex.driver.screenshot(tmp)
  652. if cv2.imread(tmp) is None:
  653. ex.driver.screenshot(tmp)
  654. texts = [r["text"] for r in OCR.recognize(tmp, detail="all")]
  655. return any("筛选" in t for t in texts)
  656. def _is_white_screen(shot_path: str) -> bool:
  657. """白屏检测:截图缩至90x200灰度后均值>250判白(详情页没加载出来的纯白页)。
  658. 单指标风格同 _page_moved;状态栏/加载圈等少量深色像素拉低不了均值。"""
  659. img = cv2.imread(shot_path, cv2.IMREAD_GRAYSCALE)
  660. if img is None:
  661. time.sleep(0.5)
  662. img = cv2.imread(shot_path, cv2.IMREAD_GRAYSCALE)
  663. if img is None:
  664. return False
  665. return float(np.mean(cv2.resize(img, (90, 200)))) > 250.0
  666. def _look_like_shop_page(raw: list) -> bool:
  667. """OCR复核是否店铺页(防AI把店铺页误判成列表页):
  668. ① 「刚刚搜过」+「评价」同时在(店铺页特有元素)
  669. ② 店铺名特征块只有1个且在屏幕上部25%(列表页会有多个店铺名散布全屏)
  670. 店铺名特征 = 含"快递发"后缀或 药房/药店/医药/连锁/自营/旗舰 等商家关键词
  671. (海王星辰等不含"药房"的店名靠"快递发"后缀兜住)"""
  672. texts = [r.get("text", "") for r in raw]
  673. joined = "".join(texts)
  674. if ("刚刚搜过" in joined) and ("评价" in joined):
  675. return True
  676. kw = re.compile(r'快递发|药房|药店|医药|商城|超市|自营|连锁|旗舰')
  677. ys = [(r.get("box") or [0, 0, 0, 0])[1] for r in raw if kw.search(r.get("text", ""))]
  678. if len(ys) == 1:
  679. max_y = max(((r.get("box") or [0, 0, 0, 0])[3] for r in raw), default=0)
  680. return ys[0] < max_y * 0.25
  681. return False
  682. def _find_ad_close(shot_path: str) -> Optional[dict]:
  683. """
  684. 广告弹窗打叉按钮:二值化模板匹配,形状匹配不受颜色/背景干扰。
  685. 双模板双二值化(固定阈值100 + OTSU自适应,各自同阈值组合),不同弹窗打叉深浅不同,取最佳。
  686. 命中返回 {"x","y"}。
  687. """
  688. base_dir = Path(__file__).parent / "files"
  689. screen = cv2.imread(shot_path, cv2.IMREAD_GRAYSCALE)
  690. if screen is None:
  691. return None
  692. h_s, w_s = screen.shape[:2]
  693. base_w = 1220.0 # 模板裁自1220宽屏,其他分辨率按比例缩放
  694. scale_ratio = w_s / base_w
  695. scales = [round(scale_ratio * s, 2) for s in [0.8, 0.9, 1.0, 1.1, 1.2]]
  696. best_val, best_loc, best_sw, best_sh = 0, None, 0, 0
  697. # (模板文件, 截图二值化方式):同阈值组合(固定×固定、OTSU×OTSU)
  698. pairs = (
  699. ("ad_close_bin.png", lambda g: cv2.threshold(g, 100, 255, cv2.THRESH_BINARY_INV)[1]),
  700. ("ad_close_otsu.png", lambda g: cv2.threshold(g, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)[1]),
  701. )
  702. for tpl_name, binarize in pairs:
  703. template = cv2.imread(str(base_dir / tpl_name), cv2.IMREAD_GRAYSCALE)
  704. if template is None:
  705. continue
  706. screen_bin = binarize(screen)
  707. for scale in scales:
  708. scaled = cv2.resize(template, None, fx=scale, fy=scale)
  709. sw, sh = scaled.shape[1], scaled.shape[0]
  710. if sh > h_s or sw > w_s:
  711. continue
  712. res = cv2.matchTemplate(screen_bin, scaled, cv2.TM_CCOEFF_NORMED)
  713. _, mv, _, ml = cv2.minMaxLoc(res)
  714. if mv > best_val:
  715. best_val, best_loc, best_sw, best_sh = mv, ml, sw, sh
  716. # 阈值0.4:二值化形状匹配值高(实测0.65+),误匹配低
  717. if best_val >= 0.4 and best_loc is not None:
  718. sx = best_loc[0] + best_sw // 2
  719. sy = best_loc[1] + best_sh // 2
  720. # 位置约束:打叉在弹窗正下方(下半屏 y>0.4h),上半屏匹配视为误报
  721. if sy < int(h_s * 0.4):
  722. return None
  723. return {"x": sx, "y": sy}
  724. return None
  725. def _click_reload(ex: SafeExecutor) -> bool:
  726. """检测「出错了/正在检修中」页面并点击「重新加载」(验证码通过后可能出现)"""
  727. try:
  728. shot = _shot_path(ex, "reload_check.png")
  729. ex.driver.screenshot(shot)
  730. if cv2.imread(shot) is None:
  731. ex.driver.screenshot(shot)
  732. btn = None
  733. for r in OCR.recognize(shot, detail="all"):
  734. if "重新加载" in r["text"]:
  735. box = r["bbox"]
  736. btn = {"x": (box[0][0] + box[2][0]) // 2, "y": (box[0][1] + box[2][1]) // 2}
  737. break
  738. if not btn:
  739. return False
  740. print(f" 点击「重新加载」: ({btn['x']}, {btn['y']})")
  741. ex.tap(btn["x"], btn["y"])
  742. human_sleep(3)
  743. return True
  744. except Exception as e:
  745. print(f" ⚠ 重新加载处理异常: {e}")
  746. return False
  747. # 弹窗找不到关闭按钮时的重启恢复:仅限 step2 搜索流程内,最多重启3次
  748. _POPUP_RESTART_COUNT = 0
  749. MAX_POPUP_RESTART = 3
  750. def _has_popup_mask(shot_path: str) -> bool:
  751. """
  752. 弹窗遮罩检测:弹窗出现时周围被半透明遮罩变暗(暗区比例大幅上升)。
  753. 实测:有弹窗暗区0.18-0.33,无弹窗<0.1,阈值0.12安全区分。
  754. """
  755. img = cv2.imread(shot_path, cv2.IMREAD_GRAYSCALE)
  756. if img is None:
  757. return False
  758. dark_ratio = float((img < 80).mean())
  759. return dark_ratio > 0.12
  760. # 全屏活动页特征词(红包/签到类弹窗无遮罩变暗,遮罩检测漏检;命中即back关闭)
  761. ACTIVITY_KW = ("红包", "领取", "金币", "翻牌", "开宝箱", "签到", "逛一逛")
  762. def _close_ad_popup(ex: SafeExecutor, force: bool = False) -> bool:
  763. """
  764. 检测并关闭广告弹窗:
  765. ① force=True: 跳过遮罩门槛,OCR找活动页特征词(红包/领取/金币等全屏活动页——
  766. 这类弹窗背景不变暗,遮罩检测会漏检),命中则 back 关闭
  767. ② 遮罩检测(弹窗周围变暗,本地计算零成本,最快最准)
  768. ③ AI 找关闭按钮(云端OCR → AI判断)
  769. ④ 模板匹配打叉兜底
  770. ⑤ back键关闭(activity对话框弹窗back即可关)
  771. ⑥ 都失败 → 重启App恢复,仅限搜索流程内3次
  772. """
  773. try:
  774. shot = _shot_path(ex, "ad_check.png")
  775. ex.driver.screenshot(shot)
  776. if cv2.imread(shot) is None:
  777. ex.driver.screenshot(shot)
  778. # 0. force: 活动页特征词检测(无遮罩门槛)
  779. if force:
  780. texts = [r.get("text", "") for r in OCR.recognize(shot, detail="all")]
  781. # 列表页标记在场(筛选/快递) → 我们就在搜索列表页上, 页面里的"红包"字样
  782. # 是页签/促销文案, 不是弹窗 —— 绝不能back(back会退出列表页)
  783. on_list = any(("筛选" in t or "快递" in t) for t in texts)
  784. hit = [w for w in ACTIVITY_KW if any(w in t for t in texts)]
  785. if hit and not on_list:
  786. print(f" 检测到活动页弹窗(特征词:{hit[0]},无列表标记),back关闭")
  787. ex.driver.press("back")
  788. human_sleep(1.2)
  789. return True
  790. if hit and on_list:
  791. print(f" 页面含活动词({hit[0]})但列表标记在场 → 是列表页不是弹窗,不处理")
  792. # 1. 遮罩检测:无弹窗直接返回(不浪费AI调用)
  793. if not _has_popup_mask(shot):
  794. return False
  795. print(" 检测到弹窗(周围遮罩变暗)")
  796. # 2. AI 找关闭按钮(云端OCR → AI判断坐标)
  797. raw = OCR.recognize(shot, detail="all", engine="cloud")
  798. info = detect_popup(raw)
  799. if info["has_popup"] and info.get("close_xy") and len(info["close_xy"]) == 2:
  800. print(f" 弹窗检测(AI): {info['reason']}")
  801. print(f" 关闭广告弹窗(AI坐标): ({info['close_xy'][0]}, {info['close_xy'][1]})")
  802. ex.tap(int(info["close_xy"][0]), int(info["close_xy"][1]))
  803. human_sleep(1.5)
  804. return True
  805. # 3. 模板匹配打叉兜底(弹窗正下方居中的 ×)
  806. btn = _find_ad_close(shot)
  807. if btn:
  808. print(f" 关闭广告弹窗(模板): ({btn['x']}, {btn['y']})")
  809. ex.tap(btn["x"], btn["y"])
  810. human_sleep(1.5)
  811. return True
  812. # 3.5 back键关闭(红包/活动类弹窗多是activity对话框,back即可关;关不掉无损继续)
  813. ex.driver.press("back")
  814. human_sleep(1.2)
  815. ex.driver.screenshot(shot)
  816. if cv2.imread(shot) is not None and not _has_popup_mask(shot):
  817. print(" 弹窗已通过back键关闭")
  818. return True
  819. # 4. AI/模板/back都关不掉(图片型弹窗)→ 重启App恢复,仅限搜索流程内3次
  820. global _POPUP_RESTART_COUNT
  821. _POPUP_RESTART_COUNT += 1
  822. if _POPUP_RESTART_COUNT > MAX_POPUP_RESTART:
  823. print(f" ⚠ 弹窗重启恢复已达{MAX_POPUP_RESTART}次上限,不再重启")
  824. return True
  825. print(f" ⚠ 弹窗关不掉(红包类图片弹窗),重启App恢复(第{_POPUP_RESTART_COUNT}/{MAX_POPUP_RESTART}次)")
  826. try:
  827. step1_open_app(ex)
  828. except Exception as e:
  829. print(f" ⚠ 重启App异常: {e}")
  830. return True
  831. except Exception as e:
  832. print(f" ⚠ 广告弹窗处理异常: {e}")
  833. return False
  834. def _adb_swipe_up(ex: SafeExecutor, distance: int):
  835. """拟人上滑, 三级降级(每级滑动后像素验证, 未生效自动降下一级):
  836. 1) TouchPipe弧线(60点, 100Hz级, 丝滑+弧线; 同验证码滑动通道)
  837. 2) motionevent弧线(10点, ~45ms/点, 略步进)
  838. 3) 三段式adb swipe(兜底)"""
  839. import subprocess as _sp
  840. w, h = ex.driver.window_size()
  841. chk = _shot_path(ex, "swipe_check.png")
  842. ex.driver.screenshot(chk)
  843. img = cv2.imread(chk)
  844. before = cv2.resize(img, (90, 200)) if img is not None else None
  845. def _page_moved():
  846. ex.driver.screenshot(chk)
  847. img = cv2.imread(chk)
  848. after = cv2.resize(img, (90, 200)) if img is not None else None
  849. return (before is not None and after is not None
  850. and float(np.mean(cv2.absdiff(before, after))) > 2.0)
  851. # 1) TouchPipe 弧线(最丝滑)
  852. try:
  853. pts = wrist_arc_pts(w, h, distance, n=60)
  854. if touchpipe_drag(ex.driver, pts):
  855. time.sleep(1.0)
  856. if _page_moved():
  857. return
  858. print(" [滑动] TouchPipe未生效,降级motionevent")
  859. except Exception as e:
  860. print(f" [滑动] TouchPipe异常({e}),降级motionevent")
  861. # 2) motionevent 弧线
  862. try:
  863. pts10 = wrist_arc_pts(w, h, distance, n=10)
  864. motionevent_drag(ex.device_id, pts10)
  865. time.sleep(1.0)
  866. if _page_moved():
  867. return
  868. print(" [滑动] motionevent未生效,退回adb swipe")
  869. except Exception as e:
  870. print(f" [滑动] motionevent异常({e}),退回adb swipe")
  871. # 3) 兜底: 三段式 adb swipe
  872. swipe_x = w // 2
  873. seg_px = distance // 3
  874. for i in range(3):
  875. s = int(h * 0.8) - i * 80
  876. e = max(50, s - seg_px)
  877. _sp.run(["adb", "-s", ex.device_id, "shell", "input", "swipe",
  878. str(swipe_x), str(s), str(swipe_x), str(e), "400"],
  879. capture_output=True, timeout=10)
  880. time.sleep(0.35)
  881. time.sleep(0.6)
  882. def _adb_swipe_left(ex: SafeExecutor, y_ratio: float = 0.3):
  883. """从右往左滑(拟人弧线):切换商品图片轮播; TouchPipe→motionevent→adb 三级降级"""
  884. import random as _random
  885. import subprocess
  886. w, h = ex.driver.window_size()
  887. distance = int(w * _random.uniform(0.55, 0.68))
  888. try:
  889. pts = wrist_arc_pts_h(w, h, distance, y_ratio=y_ratio, n=30, drift_range=(30, 70))
  890. # 真人轻扫~0.2s: 点数减半 + 步进/停顿收紧, 整笔含down/up往返~0.25s完成
  891. if touchpipe_drag(ex.driver, pts, step=_random.uniform(0.003, 0.005),
  892. hold=_random.uniform(0.02, 0.05),
  893. tail=_random.uniform(0.015, 0.04)):
  894. human_sleep(1.2)
  895. return
  896. print(" [图片左滑] TouchPipe未生效,降级motionevent")
  897. pts10 = wrist_arc_pts_h(w, h, distance, y_ratio=y_ratio, n=10, drift_range=(30, 70))
  898. motionevent_drag(ex.device_id, pts10, step=_random.uniform(0.015, 0.022))
  899. human_sleep(1.2)
  900. return
  901. except Exception as e:
  902. print(f" [图片左滑] 弧线异常({e}),退回adb swipe")
  903. subprocess.run(
  904. ["adb", "-s", ex.device_id, "shell", "input", "swipe",
  905. str(int(w * 0.85)), str(int(h * y_ratio)), str(int(w * 0.15)), str(int(h * y_ratio)), "300"],
  906. capture_output=True, timeout=10
  907. )
  908. human_sleep(1.2)
  909. def _adb_swipe_up_short(ex: SafeExecutor):
  910. """上滑半屏(拟人弧线):说明书页内容可能需滑动; TouchPipe→motionevent→adb 三级降级"""
  911. import subprocess
  912. w, h = ex.driver.window_size()
  913. try:
  914. pts = wrist_arc_pts(w, h, int(h * 0.4), n=40, drift_range=(40, 90))
  915. if touchpipe_drag(ex.driver, pts):
  916. human_sleep(0.9)
  917. return
  918. print(" [说明书上滑] TouchPipe未生效,降级motionevent")
  919. pts10 = wrist_arc_pts(w, h, int(h * 0.4), n=8, drift_range=(40, 90))
  920. motionevent_drag(ex.device_id, pts10)
  921. human_sleep(0.9)
  922. return
  923. except Exception as e:
  924. print(f" [说明书上滑] 弧线异常({e}),退回adb swipe")
  925. subprocess.run(
  926. ["adb", "-s", ex.device_id, "shell", "input", "swipe",
  927. str(w // 2), str(int(h * 0.7)), str(w // 2), str(int(h * 0.3)), "300"],
  928. capture_output=True, timeout=10
  929. )
  930. human_sleep(1)
  931. # ── 店铺反馈弹窗(滑动被误判长按店铺名)检测与关闭 ──
  932. FEEDBACK_POPUP_KW = ("店铺问题", "不喜欢该店铺", "与搜索词无关",
  933. "图片不够真实", "商品品质不好", "价格太贵", "疑似情色")
  934. def _close_feedback_popup(ex: SafeExecutor) -> bool:
  935. """检测并关闭"店铺问题/商品问题"反馈弹窗(滑动被误判长按所致)。
  936. 关闭方式: 点弹窗上方暗区,或右上角×(back 会退回搜索页,不能用)。
  937. 返回 True=检测到弹窗(无论是否成功关闭), False=没弹窗。"""
  938. try:
  939. shot = _shot_path(ex, "ad_feedback_popup.png")
  940. ex.driver.screenshot(shot)
  941. blocks = OCR.recognize(shot, detail="all")
  942. texts = [b.get("text", "") for b in blocks]
  943. if not any(kw in t for t in texts for kw in FEEDBACK_POPUP_KW):
  944. return False
  945. print("[step3] 检测到店铺反馈弹窗(滑动误触长按),关闭中")
  946. w, h = ex.driver.window_size()
  947. anchor_y = 0
  948. for b in blocks:
  949. if "店铺问题" in b.get("text", ""):
  950. anchor_y = (b["box"][1] + b["box"][3]) // 2
  951. break
  952. # 1) 点弹窗上方暗区(大目标); 2) 点×(与"店铺问题"同行右侧); 3) ×实测兜底位
  953. taps = [(int(w * 0.5), int(h * 0.25))]
  954. if anchor_y:
  955. taps.append((int(w * 0.84), anchor_y))
  956. taps.append((int(w * 0.84), int(h * 0.52)))
  957. for tx, ty in taps:
  958. ex.tap(tx, ty)
  959. time.sleep(1.2)
  960. ex.driver.screenshot(shot)
  961. blocks = OCR.recognize(shot, detail="all")
  962. texts = [b.get("text", "") for b in blocks]
  963. if not any(kw in t for t in texts for kw in FEEDBACK_POPUP_KW):
  964. print("[step3] 反馈弹窗已关闭")
  965. return True
  966. print("[step3] 反馈弹窗关闭未生效,后续批次可能空转")
  967. return True
  968. except Exception as e:
  969. print(f"[step3] 反馈弹窗检测异常: {e}")
  970. return False
  971. def _swipe_next_batch(ex: SafeExecutor, distance: int):
  972. """列表上滑统一入口: 滑动后检查长按误触的反馈弹窗。
  973. 弹窗会骗过滑动的像素diff验证且挡住列表(后续批次全是旧卡片) → 关掉后补滑一次。"""
  974. _adb_swipe_up(ex, distance)
  975. if _close_feedback_popup(ex):
  976. _adb_swipe_up(ex, distance)
  977. def _find_close_btn(shot_path: str) -> Optional[dict]:
  978. """
  979. 说明书页右上角找打叉关闭按钮(多尺度模板匹配,适配多分辨率)。
  980. 模板 files/close.png,命中返回 {"x","y"},失败返回 None。
  981. """
  982. import os as _os
  983. tpl_path = str(Path(__file__).parent / "files" / "close.png")
  984. screen = cv2.imread(shot_path)
  985. template = cv2.imread(tpl_path)
  986. if screen is None or template is None:
  987. return None
  988. h_s, w_s = screen.shape[:2]
  989. # 右上角区域(打叉永远在右上角)
  990. roi_x1, roi_y1 = w_s * 2 // 3, 0
  991. roi = screen[roi_y1:h_s // 4, roi_x1:w_s]
  992. # 多尺度模板匹配(分辨率适配:以720p为基准按屏宽比例缩放)
  993. base_w = 720.0
  994. scale_ratio = w_s / base_w
  995. scales = [round(scale_ratio * s, 2) for s in [0.7, 0.8, 0.9, 1.0, 1.1, 1.2, 1.3, 1.4, 1.5]]
  996. best_val, best_loc, best_sw, best_sh = 0, None, 0, 0
  997. for scale in scales:
  998. scaled = cv2.resize(template, None, fx=scale, fy=scale)
  999. sw, sh = scaled.shape[1], scaled.shape[0]
  1000. if sh > roi.shape[0] or sw > roi.shape[1]:
  1001. continue
  1002. res = cv2.matchTemplate(roi, scaled, cv2.TM_CCOEFF_NORMED)
  1003. _, mv, _, ml = cv2.minMaxLoc(res)
  1004. if mv > best_val:
  1005. best_val, best_loc, best_sw, best_sh = mv, ml, sw, sh
  1006. t_edge = cv2.Canny(scaled, 30, 100)
  1007. r_edge = cv2.Canny(roi, 30, 100)
  1008. if t_edge.shape[0] <= r_edge.shape[0] and t_edge.shape[1] <= r_edge.shape[1]:
  1009. res2 = cv2.matchTemplate(r_edge, t_edge, cv2.TM_CCOEFF_NORMED)
  1010. _, mv2, _, ml2 = cv2.minMaxLoc(res2)
  1011. if mv2 > best_val:
  1012. best_val, best_loc, best_sw, best_sh = mv2, ml2, sw, sh
  1013. if best_val >= 0.26 and best_loc is not None:
  1014. sx = roi_x1 + best_loc[0] + best_sw // 2
  1015. sy = best_loc[1] + best_sh // 2
  1016. return {"x": sx, "y": sy}
  1017. return None
  1018. def _collect_instructions(ex: SafeExecutor) -> dict:
  1019. """
  1020. 商品详情页采集说明书(采完链接后调用):
  1021. back关分享弹窗 → 左滑商品图片 → 点「查看详细说明」
  1022. → 说明书页截图OCR → 提取批准文号/有效期(最多上滑3次兜底)
  1023. 返回 {"approval_no": "", "validity": ""},任何失败留空不抛异常
  1024. """
  1025. w, h = ex.driver.window_size()
  1026. pfx = f"{ex.device_id}_inst"
  1027. # in_detail: 是否确认停留在商品详情页——主流程据此决定后续资质采集是否安全执行
  1028. result = {"approval_no": "", "validity": "", "in_detail": False}
  1029. # 1. 确认屏幕状态:分享弹窗→back关闭;在详情页→开始;在列表页→跳过(不再back,避免乱退)
  1030. for _try in range(3):
  1031. shot = _shot_path(ex, "inst_check.png")
  1032. ex.driver.screenshot(shot)
  1033. if cv2.imread(shot) is None:
  1034. ex.driver.screenshot(shot)
  1035. texts = [r["text"] for r in OCR.recognize(shot, detail="all")]
  1036. if any("分享到" in t for t in texts):
  1037. print(" 关闭分享弹窗")
  1038. ex.driver.press("back")
  1039. human_sleep(1.2)
  1040. continue
  1041. state = _screen_state(texts)
  1042. if state == "detail":
  1043. result["in_detail"] = True
  1044. break
  1045. if state == "list":
  1046. print(" ⚠ 说明书采集跳过:已在列表页(step4未成功进店),不再back")
  1047. return result
  1048. if _try < 2:
  1049. print(f" 屏幕状态[{state}],back一次重新确认")
  1050. ex.driver.press("back")
  1051. human_sleep(1.2)
  1052. else:
  1053. print(" ⚠ 说明书采集跳过:多次确认仍不在商品详情页")
  1054. return result
  1055. # 2. 左滑切换商品图片,找「查看详细说明」按钮
  1056. _adb_swipe_left(ex, 0.3)
  1057. shot2 = _shot_path(ex, "inst_btn.png")
  1058. ex.driver.screenshot(shot2)
  1059. btn = _find_text_in_area(shot2, "查看详细说明", h)
  1060. if not btn:
  1061. print(" ⚠ 未找到「查看详细说明」,跳过说明书采集")
  1062. return result
  1063. print(f" 说明书按钮: ({btn['x']}, {btn['y']})")
  1064. ex.tap(btn["x"], btn["y"])
  1065. human_sleep(2.5)
  1066. _ai_check_captcha(ex) # 说明书页可能出现验证码(无关键词检查环节)
  1067. # 3. 说明书页截图 + OCR 提取(模仿美团:批准文号和有效期都找到才停,最多滑3次)
  1068. for attempt in range(4):
  1069. shot3 = _shot_path(ex, "inst_page.png")
  1070. ex.driver.screenshot(shot3)
  1071. inst = parse_instructions(OCR.recognize(shot3, detail="all", engine="cloud"))
  1072. if inst["approval_no"] and inst["validity"]:
  1073. print(f" 说明书: 批准文号={inst['approval_no']} 有效期={inst['validity']}")
  1074. result = inst
  1075. break
  1076. if attempt < 3:
  1077. missing = [k for k in ("approval_no", "validity") if not inst[k]]
  1078. print(f" 说明书字段不全(第{attempt+1}次,缺{missing}),上滑重试")
  1079. _adb_swipe_up_short(ex)
  1080. else:
  1081. print(" ⚠ 说明书页4次均未解析到批准文号/有效期")
  1082. result = inst
  1083. # 4. 说明书采集完成:点右上角打叉关闭说明书页(不能back——back会直接回列表页)
  1084. # 本任务内缓存:第一个商品找到后,后续商品直接复用坐标
  1085. global _CLOSE_BTN_CACHE
  1086. close_btn = _CLOSE_BTN_CACHE
  1087. if close_btn is None:
  1088. close_shot = _shot_path(ex, "inst_close.png")
  1089. ex.driver.screenshot(close_shot)
  1090. close_btn = _find_close_btn(close_shot)
  1091. if close_btn:
  1092. _CLOSE_BTN_CACHE = close_btn
  1093. print(f" 关闭说明书页: ({close_btn['x']}, {close_btn['y']})(本任务已缓存,后续商品复用)")
  1094. else:
  1095. print(" ⚠ 未找到打叉按钮(后续步骤会按屏幕状态自行处理)")
  1096. else:
  1097. print(f" 关闭说明书页: ({close_btn['x']}, {close_btn['y']})(复用本任务缓存坐标)")
  1098. if close_btn:
  1099. ex.tap(close_btn["x"], close_btn["y"])
  1100. human_sleep(1.2)
  1101. return result
  1102. def _collect_snapshot(ex: SafeExecutor, title: str) -> str:
  1103. """
  1104. 网页快照(采集说明书之后调用):
  1105. 先识别屏幕状态——已在详情页直接拍;说明书页等未知页则back一次回详情页再拍;
  1106. 在列表页/店铺页等明确非详情页位置直接跳过(不再back,避免把列表页退到首页)
  1107. """
  1108. try:
  1109. # 1. 先截图识别状态,决定是否需要 back
  1110. shot = _shot_path(ex, "snap_check.png")
  1111. for _try in range(2):
  1112. ex.driver.screenshot(shot)
  1113. if cv2.imread(shot) is None:
  1114. ex.driver.screenshot(shot)
  1115. texts = [r["text"] for r in OCR.recognize(shot, detail="all")]
  1116. state = _screen_state(texts)
  1117. if state == "detail":
  1118. break
  1119. if state in ("list", "shop"):
  1120. print(f" ⚠ 快照跳过:屏幕状态[{state}],不在详情页也不再back")
  1121. return ""
  1122. if _try == 0:
  1123. # 说明书页/其他未知页:back 一次回详情页再确认
  1124. print(f" 快照:屏幕状态[{state}],back回详情页")
  1125. ex.driver.press("back")
  1126. human_sleep(1.2)
  1127. else:
  1128. print(f" ⚠ 快照跳过:back后屏幕状态[{state}],不在详情页")
  1129. return ""
  1130. else:
  1131. print(" ⚠ 快照跳过:无法确认在详情页")
  1132. return ""
  1133. # 2. 滚动截图 + 上传OSS(美团同款逻辑在 snapshot 模块)
  1134. url, snap_reason = collect_snapshot(ex.driver, title, ex.device_id)
  1135. print(f" 📷 快照: {url if url else f'失败({snap_reason})'}")
  1136. if url:
  1137. # 快照滚动改变了页面位置:back 回店铺页,再开始资质采集
  1138. ex.driver.press("back")
  1139. human_sleep(1.2)
  1140. return url
  1141. except Exception as e:
  1142. print(f" ⚠ 快照采集异常: {e}")
  1143. return ""
  1144. def _wait_for_any_text(ex: SafeExecutor, targets: list, max_s: float, prefix: str = "wait_text") -> bool:
  1145. """轮询截图(本地OCR)等待任一目标文字出现——等页面加载完成再进行下一步。
  1146. 返回 True=等到了, False=超时未出现(调用方自行决定是否继续)"""
  1147. import time as _t
  1148. deadline = _t.time() + max_s
  1149. while _t.time() < deadline:
  1150. shot = _shot_path(ex, f"{prefix}.png")
  1151. ex.driver.screenshot(shot)
  1152. if cv2.imread(shot) is not None:
  1153. texts = [r.get("text", "") for r in OCR.recognize(shot, detail="all")]
  1154. if any(tg in t for tg in targets for t in texts):
  1155. return True
  1156. _t.sleep(1.2)
  1157. return False
  1158. def _collect_license(ex: SafeExecutor, shop_name: str) -> dict:
  1159. """
  1160. 采集商家资质(采完说明书后调用,屏幕在说明书页):
  1161. back回店铺内 → OCR上半区找店铺名点击 → 下半区找「查看营业资质」点击
  1162. → 云端OCR找「资质编号」取值 → 点编号下方约3cm打开营业执照 → 百度营业执照专用接口OCR
  1163. 返回 {"license_no": "", "license": {}},任何失败留空不抛异常
  1164. """
  1165. w, h = ex.driver.window_size()
  1166. pfx = f"{ex.device_id}_lic"
  1167. result = {"license_no": "", "license": {}}
  1168. # 0. 已采集过的店铺:直接从数据库获取资质,不重复采集(美团/PDD同款)
  1169. try:
  1170. exist = get_existing_license(shop_name)
  1171. if exist.get("license_no") or exist.get("company"):
  1172. print(f" 资质已存在,从数据库获取: 编号={exist['license_no'][:24]} 公司={exist['company'][:20]}")
  1173. result["license_no"] = exist["license_no"]
  1174. lic = {}
  1175. if exist.get("company"):
  1176. lic["单位名称"] = exist["company"]
  1177. if exist.get("address"):
  1178. lic["地址"] = exist["address"]
  1179. result["license"] = lic
  1180. return result
  1181. except Exception as e:
  1182. print(f" ⚠ 查询已有资质失败: {e}")
  1183. # 1+2. 回到店铺页并采资质(最多2轮):每轮先找店铺名,找不到/点后无资质入口就 back 回退一层再看
  1184. # 注意不能无条件back:屏幕已在店铺页时再退会到列表页
  1185. def _find_license_btn():
  1186. # 先搜下半区,找不到整张图分析(「查看营业资质」位置不固定,有时在上半区)
  1187. for rect in ([0, int(h * 0.45), w, h], None):
  1188. for r in OCR.recognize(shot2, rect=rect, detail="all"):
  1189. if "查看营业资质" in r["text"]:
  1190. box = r["bbox"]
  1191. return {"x": (box[0][0] + box[2][0]) // 2, "y": (box[0][1] + box[2][1]) // 2}
  1192. return None
  1193. lic_btn = None
  1194. for attempt in range(2):
  1195. if _is_list_page(ex):
  1196. print(" ⚠ 资质采集跳过:已在列表页(不点列表卡片)")
  1197. return result
  1198. shot = _shot_path(ex, "lic_shopname.png")
  1199. ex.driver.screenshot(shot)
  1200. shop_btn = _find_text_in_area(shot, shop_name, h // 2)
  1201. if shop_btn:
  1202. print(f" 店铺名: ({shop_btn['x']}, {shop_btn['y']}) (第{attempt+1}轮)")
  1203. ex.tap(shop_btn["x"], shop_btn["y"])
  1204. human_sleep(1.2)
  1205. # 等店铺信息页加载出「查看营业资质」(最多6s,防止没等加载就找)
  1206. _wait_for_any_text(ex, ["查看营业资质"], 6, "lic_wait")
  1207. # 下半区找「查看营业资质」,找不到先上滑一次再看
  1208. shot2 = _shot_path(ex, "inst_btn.png")
  1209. ex.driver.screenshot(shot2)
  1210. lic_btn = _find_license_btn()
  1211. if not lic_btn:
  1212. print(f" 未找到「查看营业资质」(第{attempt+1}轮),上滑再看")
  1213. _adb_swipe_up_short(ex)
  1214. shot2 = _shot_path(ex, "inst_btn.png")
  1215. ex.driver.screenshot(shot2)
  1216. lic_btn = _find_license_btn()
  1217. if lic_btn:
  1218. break
  1219. if attempt == 0:
  1220. # 屏幕可能在详情页/说明书页:back 回退一层再看
  1221. print(" back一次回退后再试")
  1222. ex.driver.press("back")
  1223. human_sleep(1.5)
  1224. else:
  1225. print(" ⚠ 资质采集跳过:未找到店铺名或「查看营业资质」")
  1226. return result
  1227. print(f" 查看营业资质: ({lic_btn['x']}, {lic_btn['y']})")
  1228. ex.tap(lic_btn["x"], lic_btn["y"])
  1229. human_sleep(1.5)
  1230. _ai_check_captcha(ex) # 资质页可能出现验证码(无关键词检查环节)
  1231. # 4. 等资质页加载出「资质编号」(最多8s;刚点完就截图经常是加载中的页面)
  1232. _wait_for_any_text(ex, ["资质编号", "营业执照"], 8, "lic_no_wait")
  1233. shot3 = _shot_path(ex, "lic_no.png")
  1234. ex.driver.screenshot(shot3)
  1235. raw3 = OCR.recognize(shot3, detail="all", engine="cloud")
  1236. license_no = extract_value_after(raw3, "资质编号")
  1237. if license_no:
  1238. # OCR可能把长编号读散(如 "91450800MA5 KEHAHXT"),编号不该有空格,清洗掉
  1239. license_no = license_no.replace(" ", "")
  1240. if not license_no:
  1241. # 页面可能仍在加载/云端OCR波动:等2秒重拍再试一次
  1242. time.sleep(2)
  1243. ex.driver.screenshot(shot3)
  1244. raw3 = OCR.recognize(shot3, detail="all", engine="cloud")
  1245. license_no = extract_value_after(raw3, "资质编号")
  1246. if license_no:
  1247. license_no = license_no.replace(" ", "")
  1248. if not license_no:
  1249. print(" ⚠ 资质采集跳过:未找到资质编号")
  1250. return result
  1251. print(f" 资质编号: {license_no}")
  1252. result["license_no"] = license_no
  1253. # 5. 点资质编号下方约3cm(≈0.2屏高)打开营业执照大图
  1254. # 注意:云端OCR(标准版)无坐标,必须用本地OCR找资质编号的真实位置
  1255. no_x, no_y = None, None
  1256. for r in OCR.recognize(shot3, detail="all"): # 本地引擎,有真实bbox
  1257. t = r["text"].strip().rstrip(":: \t")
  1258. if t.startswith("资质编号"):
  1259. box = r["bbox"]
  1260. no_x = (box[0][0] + box[2][0]) // 2
  1261. no_y = (box[0][1] + box[2][1]) // 2
  1262. break
  1263. if no_y is None:
  1264. print(" ⚠ 资质采集跳过:本地OCR未找到资质编号位置")
  1265. return result
  1266. print(f" 资质编号位置: ({no_x}, {no_y}),点击下方打开执照")
  1267. ex.tap(no_x, min(no_y + int(h * 0.2), h - 50))
  1268. human_sleep(2.5)
  1269. # 6. 营业执照截图 + 百度营业执照专用接口(美团同款;执照在屏幕中下部,先裁剪再识别,逐级兜底)
  1270. shot4 = _shot_path(ex, "lic_license.png")
  1271. ex.driver.screenshot(shot4)
  1272. lic = {}
  1273. for t_ratio, b_ratio in [(0.35, 0.8), (0.4, 1.0), (0.0, 1.0)]:
  1274. lic = OCR.recognize_license(shot4, rect=[0, int(h * t_ratio), w, int(h * b_ratio)])
  1275. if lic:
  1276. break
  1277. if lic:
  1278. print(f" 营业执照: {lic}")
  1279. result["license"] = lic
  1280. else:
  1281. print(" ⚠ 营业执照OCR为空")
  1282. return result
  1283. def _detect_right_col_split(raw: list, w: int) -> Optional[int]:
  1284. """
  1285. 定位列表页"图片|文字"分界x:每张卡片都有价格,¥全在右列且x一致。
  1286. 取¥块左边缘x的中位数 − 15 作为分界(图片区在左、标题/价格在右)。
  1287. 实测3台设备:分界499时图片文字最右仅474,过滤干净。返回None=检测不到(用整图)。
  1288. """
  1289. price_x = sorted(r["box"][0] for r in raw if "¥" in r["text"] or "¥" in r["text"])
  1290. if len(price_x) < 3:
  1291. return None
  1292. return max(price_x[len(price_x) // 2] - 15, int(w * 0.3))
  1293. def _get_named_shops(ex: SafeExecutor, shot_name: str, keyword: str = "") -> tuple:
  1294. """截图 + OCR + AI → 返回 (有店铺名的列表, 本地OCR原始结果带坐标, 页面状态)
  1295. 页面状态: "ok"=正常(含空批次) | "page_wrong"=AI判定不在药品列表页"""
  1296. shot = _screenshot(ex, shot_name)
  1297. raw = OCR.recognize(shot, detail="all")
  1298. w, h = ex.driver.window_size()
  1299. # 只保留右列(商品描述列):用¥定位分界,过滤左侧图片文字(包装字/英文/乱码)
  1300. split_x = _detect_right_col_split(raw, w)
  1301. if split_x:
  1302. filtered = [r for r in raw if (r["box"][0] + r["box"][2]) // 2 >= split_x]
  1303. if filtered:
  1304. print(f"[step3] 右列识别: 分界x={split_x},过滤掉{len(raw) - len(filtered)}块图片文字")
  1305. raw = filtered
  1306. # 在截图副本上画红线保存(出错时核对分界是否偏左/偏右)
  1307. try:
  1308. img = cv2.imread(shot)
  1309. if img is not None:
  1310. cv2.line(img, (split_x, 0), (split_x, img.shape[0]), (0, 0, 255), 3)
  1311. cv2.putText(img, f"split={split_x}", (split_x + 5, 50),
  1312. cv2.FONT_HERSHEY_SIMPLEX, 0.9, (0, 0, 255), 2)
  1313. cv2.imwrite(shot.replace(".png", "_split.png"), img)
  1314. except Exception:
  1315. pass
  1316. # 视觉方案(唯一路径):PP-OCR识别 → GLM分卡 → 几何校验;GLM失败时内部有文本AI兜底
  1317. shops, vstatus = VisionParser().parse_shops(shot, screen_size=(w, h), keyword=keyword, crop_x=split_x or 0)
  1318. if vstatus == "page_wrong":
  1319. # GLM判定当前页面不是药品列表(back退过头/走错频道)→ 不提取,交给上层走恢复流程
  1320. return [], raw, "page_wrong"
  1321. if vstatus == "failed":
  1322. # GLM真失败才fallback到文本AI提取
  1323. print("[step3] GLM失败,fallback 到 OCR+文本AI")
  1324. shops = AIParser().parse_shops(raw, screen_size=(w, h), keyword=keyword)
  1325. if not shops:
  1326. # fallback也没提到 → 让文本AI判断当前页面再决定
  1327. ptype = AIParser().check_page(raw).get("type", "unknown")
  1328. if ptype == "list":
  1329. print("[step3] 文本AI判定在列表页(本轮无有效卡片,走空批次滑动)")
  1330. return [], raw, "ok"
  1331. print(f"[step3] 文本AI判定页面类型: {ptype}(非列表页)")
  1332. return [], raw, "page_wrong"
  1333. # 只保留有效店铺名+价格:店铺名必须含中文或字母(排除纯数字/标点/空格)
  1334. import re as _re
  1335. valid = []
  1336. for s in shops:
  1337. name = (s[0] or "").strip()
  1338. price = (s[2] or "").strip()
  1339. if name and _re.search(r'[一-鿿＀-￯a-zA-Z]', name) and price:
  1340. valid.append(s)
  1341. return valid, raw, "ok"
  1342. def _find_title_y(raw: list, title: str) -> Optional[int]:
  1343. """在本地OCR结果里找与标题开头重合最多的块的y坐标(标题行位置,用于裁剪)"""
  1344. nt = normalize_match_text(title)
  1345. best_len, best_cy = 0, None
  1346. for r in raw:
  1347. t = normalize_match_text(r["text"])
  1348. if not t:
  1349. continue
  1350. n = 0
  1351. for a, b in zip(t, nt):
  1352. if a == b:
  1353. n += 1
  1354. else:
  1355. break
  1356. if n > best_len:
  1357. best_len = n
  1358. best_cy = (r["bbox"][0][1] + r["bbox"][2][1]) // 2
  1359. return best_cy if best_len >= 2 else None
  1360. def _shop_key(shop: list) -> str:
  1361. """用店铺名+价格去重(去括号内分店名、去尾部点号)"""
  1362. import re
  1363. name = shop[0]
  1364. price = shop[2] if len(shop) > 2 else ""
  1365. name = name.replace("(", "(").replace(")", ")")
  1366. name = re.sub(r'(.*', '', name)
  1367. name = re.sub(r'[..…]+$', '', name)
  1368. return f"{name.strip()}|{price.strip()}"
  1369. def _visit_shop(ex: SafeExecutor, shop: list, visited: set, keyword: str = "", task: dict = None,
  1370. batch_no: int = None) -> dict:
  1371. """
  1372. 点击进入店铺 → step4 → 说明书 → 快照 → 资质 → 返回完整数据 dict
  1373. task: 调度任务 dict(task_id/enterprise_id/collect_round等),手动模式传 None
  1374. batch_no: 当前列表页批次号(Excel来源截图追溯用)
  1375. """
  1376. """点击进入店铺 → step4 → 返回完整数据 dict"""
  1377. key = _shop_key(shop)
  1378. if key in visited:
  1379. return None
  1380. visited.add(key)
  1381. shop_name = shop[0]
  1382. product_title = shop[1]
  1383. price = shop[2]
  1384. sales = str(shop[5]) if len(shop) > 5 and shop[5] else ""
  1385. click_x, click_y = shop[3]
  1386. print(f" → 进入 [{shop_name}] 商品: {product_title[:30]} 价格: {price} 月售: {sales}")
  1387. ex.tap(click_x, click_y)
  1388. try:
  1389. qr_url = step4_parse_qr(ex, product_title, shop_name, shop_xy=[click_x, click_y])
  1390. except Exception as e:
  1391. print(f" ⚠ step4异常: {e},跳过此店铺")
  1392. qr_url = ""
  1393. if qr_url == "__TERMINATE__":
  1394. print(f" ⚠ 遇到终止信号,停止遍历")
  1395. return {"__terminate__": True}
  1396. if qr_url == "__RESTART__":
  1397. print(f" ⚠ 页面异常,触发重启恢复")
  1398. return {"__restart__": True}
  1399. if qr_url == "__SKIP__":
  1400. # 连续5次进入页面均未加载(网络/加载异常):退回列表页,跳过本商品
  1401. # step3会对"连续2个商品都跳过"升级为白屏分级恢复(冷却→重启→超限重派)
  1402. print(f" ⛔ 页面多次未加载{'(白屏)' if _LAST_SKIP_WHITE else ''},跳过本商品: {shop_name} | {product_title[:30]}")
  1403. for _ in range(5):
  1404. try:
  1405. pos = _where_am_i(ex)
  1406. except Exception:
  1407. break
  1408. if pos in ("list", "home"):
  1409. break
  1410. ex.driver.press("back")
  1411. human_sleep(1.4)
  1412. return {"__skip__": True}
  1413. if qr_url:
  1414. print(f" ✅ QR: {qr_url[:80]}")
  1415. print(f" 📦 采集完成: {shop_name} | {product_title[:30]} | {price} | 月售{sales} | {qr_url[:60]}")
  1416. else:
  1417. # 未获取到链接(放弃本店/解析失败):数据不入库,退回列表页后直接跳过本店
  1418. print(f" ⛔ 未获取到链接,本店数据不入库: {shop_name} | {product_title[:30]}")
  1419. for _ in range(5):
  1420. try:
  1421. pos = _where_am_i(ex)
  1422. except Exception:
  1423. break
  1424. if pos in ("list", "home"):
  1425. break
  1426. ex.driver.press("back")
  1427. human_sleep(1.4)
  1428. return {}
  1429. # 内存熔断:本地OCR已OOM → 说明书/快照/资质全部跳过(状态检测不可靠,避免乱back乱点)
  1430. if getattr(OCR, "oom", False):
  1431. print(" ⚠ 内存不足熔断:跳过说明书/快照/资质,直接返回列表页(请关闭部分程序释放内存)")
  1432. inst = {"approval_no": "", "validity": "", "in_detail": False}
  1433. snapshot_url = ""
  1434. lic = {"license_no": "", "license": {}}
  1435. else:
  1436. # 采完链接后采集说明书(批准文号/有效期),失败不阻塞
  1437. try:
  1438. inst = _collect_instructions(ex)
  1439. except Exception as e:
  1440. print(f" ⚠ 说明书采集异常: {e}(疑似内存不足,请关闭部分程序)")
  1441. inst = {"approval_no": "", "validity": "", "in_detail": False}
  1442. print(f" 📄 说明书: 批准文号={inst['approval_no']} 有效期={inst['validity']}")
  1443. # 采完说明书后采集网页快照(美团同款顺序:说明书→快照→资质),失败不阻塞
  1444. snapshot_url = _collect_snapshot(ex, product_title)
  1445. # 采完快照后采集商家资质(_collect_license 会先 back 回店铺页,再按屏幕状态自校验:列表页/无店铺名都跳过)
  1446. try:
  1447. lic = _collect_license(ex, shop_name)
  1448. except Exception as e:
  1449. print(f" ⚠ 资质采集异常: {e}")
  1450. lic = {"license_no": "", "license": {}}
  1451. print(f" 📋 资质编号: {lic['license_no']} 单位名称: {lic['license'].get('单位名称', '')} 信用代码: {lic['license'].get('社会信用代码', '')}")
  1452. # 返回搜索页:先检测再退(已是列表页则一步不退;退到首页立即停,防止退过头退出app)
  1453. for _ in range(5):
  1454. try:
  1455. pos = _where_am_i(ex)
  1456. except Exception as e:
  1457. print(f" ⚠ 位置检测异常({e}),停止返回")
  1458. break
  1459. if pos == "list":
  1460. break
  1461. if pos == "home":
  1462. print(" ⚠ 已退到首页(可能退过头),停止返回")
  1463. break
  1464. ex.driver.press("back")
  1465. human_sleep(1.4)
  1466. # Excel 备份:一个药品一个sheet,两个店铺名对比(列表页OCR vs 详情页AI提取)
  1467. # 来源截图 = 设备/批次,供 audit_ocr.py 事后核对OCR识别准确率
  1468. try:
  1469. import excel_log
  1470. source = f"{ex.device_id}/step3_b{batch_no}" if batch_no is not None else ""
  1471. excel_log.append_row(product_title, price, shop_name, AI_SHOP_NAME,
  1472. snapshot_url, qr_url or "", sheet_name=keyword, source=source)
  1473. except Exception as _e:
  1474. print(f" ⚠ Excel备份失败: {_e}")
  1475. task = task or {}
  1476. return {
  1477. "shop": shop_name,
  1478. "title": product_title,
  1479. "price": price,
  1480. "sales": sales,
  1481. "approval_no": inst["approval_no"],
  1482. "validity": inst["validity"],
  1483. "license_no": lic["license_no"],
  1484. "license": lic["license"],
  1485. "snapshot_url": snapshot_url,
  1486. "search_name": keyword,
  1487. "link": qr_url or "",
  1488. "task_id": task.get("task_id"),
  1489. "enterprise_id": task.get("enterprise_id"),
  1490. "collect_round": task.get("collect_round"),
  1491. "collect_equipment_account_id": task.get("collect_equipment_account_id"),
  1492. "collect_region_id": task.get("collect_region_id"),
  1493. "collect_config_info": task.get("collect_config_info", ""),
  1494. }
  1495. def _captcha_log_path(device_id: str) -> Path:
  1496. """每台设备独立的验证码日志文件(独立计数,互不影响)"""
  1497. return CAPTCHA_LOG_DIR / f"captcha_log_{device_id}.txt"
  1498. def _log_captcha(ex: SafeExecutor) -> bool:
  1499. """
  1500. 记录验证码出现时间到日志,检查一天≥8次停止。
  1501. 返回 True=已触发停止(上层不再休息),False=正常(处理完成后按频率休息)。
  1502. """
  1503. global CAPTCHA_ABORTED, CAPTCHA_ABORT_REASON
  1504. import os as _os
  1505. ts = time.strftime("%Y-%m-%d %H:%M:%S")
  1506. line = f"{ts} 验证码出现"
  1507. try:
  1508. _os.makedirs(_os.path.dirname(_captcha_log_path(ex.device_id)), exist_ok=True)
  1509. with open(_captcha_log_path(ex.device_id), "a", encoding="utf-8") as f:
  1510. f.write(line + "\n")
  1511. except Exception as e:
  1512. print(f" [验证码记录] 写日志失败: {e}")
  1513. print(f" [验证码记录] {line}")
  1514. # 一天内 ≥8次 → 立即停止采集回告(风控可能封号)
  1515. today_count = _captcha_count_today(ex.device_id)
  1516. print(f" [验证码记录] 设备{ex.device_id}今天已出现{today_count}次")
  1517. if today_count >= CAPTCHA_DAILY_LIMIT:
  1518. CAPTCHA_ABORTED = True
  1519. CAPTCHA_ABORT_REASON = f"一天内验证码达{today_count}次,进入风控可能封号"
  1520. print(f" ⚠ {CAPTCHA_ABORT_REASON},停止采集")
  1521. return True
  1522. return False
  1523. def _captcha_count_today(device_id: str) -> int:
  1524. """统计该设备日志中今天(按日期)的验证码次数"""
  1525. try:
  1526. today = time.strftime("%Y-%m-%d")
  1527. count = 0
  1528. with open(_captcha_log_path(device_id), encoding="utf-8") as f:
  1529. for line in f:
  1530. if line.startswith(today):
  1531. count += 1
  1532. return count
  1533. except Exception:
  1534. return 0
  1535. def _save_captcha_check(ex: SafeExecutor) -> None:
  1536. """每次判定出现验证码时截图保存到 logs/captcha_check/{日期}/,人工确认是否真验证码"""
  1537. import os as _os
  1538. try:
  1539. day = time.strftime("%Y-%m-%d")
  1540. d = CAPTCHA_CHECK_DIR / day
  1541. d.mkdir(parents=True, exist_ok=True)
  1542. ts = time.strftime("%H%M%S")
  1543. path = str(d / f"{ts}_{ex.device_id}_captcha.png")
  1544. ex.driver.screenshot(path)
  1545. print(f" [验证码截图] 已保存: {path}")
  1546. except Exception as e:
  1547. print(f" [验证码截图] 保存失败: {e}")
  1548. def _handle_captcha(ex: SafeExecutor, ocr_texts: list) -> bool:
  1549. """处理验证码, 重试5次, 失败等人工, 返回True=已解决;
  1550. 只有自动解决成功才计数(失败走人工的不算),成功后按频率休息"""
  1551. _save_captcha_check(ex) # 判定出现验证码时截图存档(人工确认是否误判)
  1552. import sys as _sys
  1553. _sys.path.insert(0, r"D:\drug\sg\yzm")
  1554. solved = False
  1555. for attempt in range(1, 6):
  1556. print(f" [验证码] 第{attempt}次尝试...")
  1557. nine_kw = any("提交" in t or "没有新图片" in t for t in ocr_texts)
  1558. if nine_kw:
  1559. from yzm.nine_grid import solve as solve_nine
  1560. ok = solve_nine(ex.driver)
  1561. else:
  1562. from yzm.tmp_captcha_test6 import solve_slider
  1563. ok = solve_slider(ex.driver)
  1564. if ok:
  1565. print(f" ✅ 验证码已解决")
  1566. solved = True
  1567. break
  1568. print(f" ❌ 第{attempt}次失败")
  1569. human_sleep(1)
  1570. if solved:
  1571. # 只有解决成功才计数 + 一天≥8次停止检查 + 每2次休息30分钟
  1572. aborted = _log_captcha(ex)
  1573. if not aborted:
  1574. _captcha_rest(ex)
  1575. else:
  1576. print(f" ⚠ 5次自动处理失败, 请人工处理...(不计数)")
  1577. input(" 处理完成后按回车继续...")
  1578. return True
  1579. def _chunked_sleep_with_report(total_seconds: float, label: str) -> bool:
  1580. """分段睡眠并每10分钟回告一次进度(防后台判假死)。
  1581. 返回 False = 休息期间收到调度限额/错误(应停止采集)"""
  1582. rest_left = total_seconds
  1583. while rest_left > 0:
  1584. if _SCHEDULER is not None and getattr(_SCHEDULER, "limit_reached", False):
  1585. print(f" ⏸ {label}休息中收到调度限额/错误,提前结束休息并停止采集")
  1586. return False
  1587. chunk = min(rest_left, 600) # 每10分钟一段
  1588. time.sleep(chunk)
  1589. rest_left -= chunk
  1590. if rest_left > 0 and _SCHEDULER is not None:
  1591. try:
  1592. _SCHEDULER.post_report({
  1593. "task_id": _TASK_ID,
  1594. "platform": _SCHEDULER.platform,
  1595. "username": _SCHEDULER.username,
  1596. "is_finished": 0,
  1597. "need_reassign": 0,
  1598. "current_page": CURRENT_PAGE,
  1599. "crawled_count": _CRAWLED_COUNT,
  1600. })
  1601. print(f" [休息中回告] 当前页{CURRENT_PAGE},已采{_CRAWLED_COUNT}条")
  1602. except Exception as e:
  1603. print(f" [休息中回告] 失败: {e}")
  1604. return True
  1605. def _captcha_rest(ex: SafeExecutor) -> None:
  1606. """每次验证码解决后:休息 CAPTCHA_REST_MINUTES~MAX 分钟(随机)"""
  1607. import random as _random
  1608. today_count = _captcha_count_today(ex.device_id)
  1609. if today_count >= CAPTCHA_DAILY_LIMIT:
  1610. return
  1611. rest_minutes = _random.randint(CAPTCHA_REST_MINUTES, CAPTCHA_REST_MAX_MINUTES)
  1612. print(f" ⏸ 第{today_count}次验证码,休息{rest_minutes}分钟({CAPTCHA_REST_MINUTES}~{CAPTCHA_REST_MAX_MINUTES}随机)...")
  1613. _chunked_sleep_with_report(rest_minutes * 60, "验证码")
  1614. # 验证码强特征词(店铺页/列表页文字多但无这些词——AI判risk时用OCR文字二次确认防误判)
  1615. CAPTCHA_KW = ("拖动滑块", "请按住滑块", "请按照说明", "点我反馈", "进行验证",
  1616. "滑块验证", "拼图", "安全验证", "图形验证", "点击完成验证", "没有新图片", "操作频繁")
  1617. def _has_captcha_kw(texts: list) -> bool:
  1618. """OCR文字是否含验证码强特征词"""
  1619. joined = "".join(texts)
  1620. return any(k in joined for k in CAPTCHA_KW)
  1621. def _ai_check_captcha(ex: SafeExecutor) -> bool:
  1622. """
  1623. AI检测当前屏幕是否出现验证码(用于没有关键词检查的环节):
  1624. 截图 → 本地OCR → 强特征词预筛 → AI判断页面类型(risk=验证码)→ 有则自动处理。
  1625. 返回 True=检测到验证码(已处理或处理中),False=无验证码。
  1626. """
  1627. try:
  1628. shot = _shot_path(ex, "captcha_ai.png")
  1629. ex.driver.screenshot(shot)
  1630. if cv2.imread(shot) is None:
  1631. ex.driver.screenshot(shot)
  1632. raw = OCR.recognize(shot, detail="all") # 本地OCR即可(验证码文字是大字,不用百度)
  1633. # 预筛:OCR文本里连强特征词都没有 → 不是验证码,不调AI(省调用+防误判)
  1634. if not _has_captcha_kw([r["text"] for r in raw]):
  1635. return False
  1636. page = AIParser().check_page(raw)
  1637. if page.get("type") == "risk":
  1638. print(f" ⚠ AI检测到验证码页面: {page.get('detail', '')}")
  1639. _handle_captcha(ex, [r["text"] for r in raw])
  1640. human_sleep(2)
  1641. return True
  1642. return False
  1643. except Exception as e:
  1644. print(f" ⚠ AI验证码检测异常: {e}")
  1645. return False
  1646. def _check_account_kicked(ex: SafeExecutor) -> bool:
  1647. """
  1648. 检测账号是否被踢/封号:登录页元素(me.ele:id/login_onkey_login_ll)出现即判定。
  1649. 检测到 → 置 ACCOUNT_ABORTED 标志(停止采集 + 调度回告)。
  1650. """
  1651. global ACCOUNT_ABORTED
  1652. try:
  1653. if ex.driver.xpath('//*[@resource-id="me.ele:id/login_onkey_login_ll"]').exists:
  1654. print(" ⚠ 检测到账号被踢/封号(登录页),停止采集")
  1655. ACCOUNT_ABORTED = True
  1656. return True
  1657. except Exception as e:
  1658. print(f" ⚠ 封号检测异常: {e}")
  1659. return False
  1660. def step4_parse_qr(ex: SafeExecutor, product_title: str, shop_name: str = "",
  1661. shop_xy: Optional[list] = None) -> str:
  1662. """
  1663. 1. 等待加载 → OCR → AI找商品标题坐标
  1664. 2. 点击商品标题 → 进入商品详情
  1665. 3. 找右上角"分享" → 点击 → 二维码弹窗
  1666. 4. 截图 → pyzbar 解析二维码
  1667. 返回 URL 或空字符串
  1668. """
  1669. global AI_SHOP_NAME, _LAST_SKIP_WHITE
  1670. AI_SHOP_NAME = "" # 每次进店重置,避免上一家的AI店名残留
  1671. _LAST_SKIP_WHITE = False
  1672. # 安全的文件名前缀(用hash避免中文路径cv2兼容问题)
  1673. # 多设备隔离:加入设备ID,防止并发时两台设备写同一个文件
  1674. import hashlib
  1675. _hash = hashlib.md5(shop_name.encode()).hexdigest()[:8] if shop_name else "unknown"
  1676. _pfx = lambda name: _shot_path(ex, f"_s4_{_hash}_{name}")
  1677. human_sleep(6)
  1678. # ── 检测页面类型:验证码/风控/正常 ──
  1679. # unknown恢复流程:等1秒二次截图确认 → 每确认一次unknown就back一层+重新点击进入(网络没加载就刷新)
  1680. # → 最多试UNKNOWN_REENTER_LIMIT次加载,全失败返回__SKIP__(跳过本商品,由step3决定是否升级重启)
  1681. unknown_round = 0
  1682. for page_retry in range(16):
  1683. shot_check = _pfx("page_check.png")
  1684. ex.driver.screenshot(shot_check)
  1685. check_raw = OCR.recognize(shot_check, detail="all")
  1686. # 方法A: 模板匹配检测验证码
  1687. import os as _os
  1688. captcha_tpl = str(Path(__file__).parent / "files" / "captcha1.png")
  1689. if _os.path.exists(captcha_tpl):
  1690. si = cv2.imread(shot_check)
  1691. ti = cv2.imread(captcha_tpl)
  1692. if si is not None and ti is not None:
  1693. gs = cv2.cvtColor(si, cv2.COLOR_BGR2GRAY)
  1694. gt = cv2.cvtColor(ti, cv2.COLOR_BGR2GRAY)
  1695. h_s, w_s = gs.shape
  1696. crop_y1, crop_y2 = int(h_s * 0.25), int(h_s * 0.75)
  1697. crop_x1, crop_x2 = 0, 400
  1698. gs_crop = gs[crop_y1:crop_y2, crop_x1:crop_x2]
  1699. scores = []
  1700. for fn, ss, tt in [
  1701. ("gray", gs_crop, gt),
  1702. ("edge", cv2.Canny(gs_crop,30,100), cv2.Canny(gt,30,100)),
  1703. ("hist", cv2.equalizeHist(gs_crop), cv2.equalizeHist(gt)),
  1704. ("blur", cv2.GaussianBlur(gs_crop,(3,3),0), cv2.GaussianBlur(gt,(3,3),0)),
  1705. ("otsu", cv2.threshold(gs_crop,0,255,cv2.THRESH_BINARY+cv2.THRESH_OTSU)[1],
  1706. cv2.threshold(gt,0,255,cv2.THRESH_BINARY+cv2.THRESH_OTSU)[1]),
  1707. ]:
  1708. if ss.ndim == 2 and tt.ndim == 2 and ss.shape[0] >= tt.shape[0] and ss.shape[1] >= tt.shape[1]:
  1709. r = cv2.matchTemplate(ss, tt, cv2.TM_CCOEFF_NORMED)
  1710. _, mv, _, _ = cv2.minMaxLoc(r)
  1711. scores.append((mv, fn))
  1712. if scores:
  1713. best_v = max(s[0] for s in scores)
  1714. best_m = max(scores, key=lambda s: s[0])[1]
  1715. print(f" 验证码模板匹配: {best_m}={best_v:.3f}")
  1716. captcha_kw = any("拖动滑块" in r["text"] or "请按住滑块" in r["text"] or "安全验证" in r["text"] for r in check_raw)
  1717. nine_kw = any("提交" in r["text"] or "没有新图片" in r["text"] for r in check_raw)
  1718. if best_v >= 0.30 and (captcha_kw or nine_kw):
  1719. print(f" ⚠ 检测到验证码,尝试自动处理...")
  1720. if _handle_captcha(ex, [r["text"] for r in check_raw]):
  1721. continue
  1722. return "__TERMINATE__"
  1723. elif best_v >= 0.30 and not captcha_kw:
  1724. print(f" ⚠ 模板匹配命中但OCR无验证码关键词,忽略")
  1725. page_type = AIParser().check_page(check_raw)
  1726. ptype = page_type.get("type", "unknown")
  1727. if ptype == "risk":
  1728. # OCR文字二次确认:店铺页/列表页文字多但无验证码特征词 → 误判,忽略
  1729. if not _has_captcha_kw([r["text"] for r in check_raw]):
  1730. print(" AI判risk但OCR无验证码特征词,忽略(店铺页/列表页误判)")
  1731. continue
  1732. print(f" ⚠ AI检测到验证码,尝试自动处理...")
  1733. if _handle_captcha(ex, [r["text"] for r in check_raw]):
  1734. continue
  1735. return "__TERMINATE__"
  1736. if ptype == "home":
  1737. # 二次确认:截图太早页面没加载完时AI可能误判首页(底部导航"我的"在任何页面都可见)
  1738. human_sleep(1)
  1739. ex.driver.screenshot(shot_check)
  1740. confirm_raw = OCR.recognize(shot_check, detail="all")
  1741. ptype2 = AIParser().check_page(confirm_raw).get("type", "unknown")
  1742. if ptype2 == "home":
  1743. _save_evidence(shot_check, "restart_home1")
  1744. print(" ⚠ 二次确认仍为首页(被踢回/退过头),触发重启恢复")
  1745. return "__RESTART__"
  1746. if ptype2 == "normal":
  1747. break # 页面加载完成,恢复正常
  1748. print(f" AI先判home,二次确认为{ptype2},继续检测")
  1749. continue
  1750. if ptype == "list":
  1751. # 第一层:OCR复核防AI误判(店铺页只有1个店铺名且在屏幕上方,AI有时把店内商品列表看成列表页)
  1752. if _look_like_shop_page(check_raw):
  1753. print(" AI判list但OCR复核为店铺页(单店名且在顶部),按进店继续")
  1754. break
  1755. # 第二层:可能是点击后跳转未完成(残影还是列表页),等1秒重拍让AI再判一次
  1756. human_sleep(1)
  1757. ex.driver.screenshot(shot_check)
  1758. confirm_raw = OCR.recognize(shot_check, detail="all")
  1759. ptype2 = AIParser().check_page(confirm_raw).get("type", "unknown")
  1760. if ptype2 != "list":
  1761. print(f" 二次确认为{ptype2}(跳转完成),继续检测")
  1762. continue
  1763. print(" ⚠ 二次确认仍在列表页(未成功进店),放弃本店")
  1764. return ""
  1765. if ptype == "login":
  1766. global ACCOUNT_ABORTED
  1767. ACCOUNT_ABORTED = True
  1768. print(" ⚠ AI检测到登录页(账号被踢/封号),停止采集")
  1769. return "__TERMINATE__"
  1770. if ptype == "normal":
  1771. _ai_shop = str(page_type.get("shop") or "").strip()
  1772. if _ai_shop:
  1773. AI_SHOP_NAME = _ai_shop # 详情页AI提取的店铺名(Excel对比用)
  1774. break # 正常,跳出重试循环
  1775. # qrcode/unknown:等1秒二次截图确认(点击后立即截图可能页面没加载完,避免误判)
  1776. human_sleep(1)
  1777. ex.driver.screenshot(shot_check)
  1778. confirm_raw = OCR.recognize(shot_check, detail="all")
  1779. confirm_page = AIParser().check_page(confirm_raw)
  1780. ptype2 = confirm_page.get("type", "unknown")
  1781. if ptype2 == "normal":
  1782. _ai_shop = str(confirm_page.get("shop") or "").strip()
  1783. if _ai_shop:
  1784. AI_SHOP_NAME = _ai_shop # 详情页AI提取的店铺名(Excel对比用)
  1785. break
  1786. if ptype2 == "risk":
  1787. if not _has_captcha_kw([r["text"] for r in confirm_raw]):
  1788. print(" 二次确认risk但OCR无验证码特征词,忽略")
  1789. continue
  1790. if _handle_captcha(ex, [r["text"] for r in confirm_raw]):
  1791. continue
  1792. return "__TERMINATE__"
  1793. if ptype2 == "home":
  1794. _save_evidence(shot_check, "restart_home2")
  1795. print(" ⚠ 二次确认检测到首页,触发重启恢复")
  1796. return "__RESTART__"
  1797. if ptype2 == "list":
  1798. # OCR复核防AI误判(同第一处list分支)
  1799. if _look_like_shop_page(confirm_raw):
  1800. print(" AI二次确认判list但OCR复核为店铺页,按进店继续")
  1801. break
  1802. print(" ⚠ 二次确认检测到列表页(未成功进店),放弃本店")
  1803. return ""
  1804. if ptype2 == "login":
  1805. ACCOUNT_ABORTED = True # 本函数已声明global
  1806. print(" ⚠ 二次确认检测到登录页(账号被踢/封号),停止采集")
  1807. return "__TERMINATE__"
  1808. # 两次都是unknown → 立即back一层+重新点击进入(每确认一次就刷新,不攒次数)
  1809. unknown_round += 1
  1810. _ws = _is_white_screen(shot_check)
  1811. if unknown_round >= UNKNOWN_REENTER_LIMIT:
  1812. # 5次加载尝试全失败:保存现场截图,跳过本商品(不重启,由step3处理连续跳过)
  1813. import shutil
  1814. err_dir = SCREENSHOT_DIR / ex.device_id / "step4" / "unrecognized"
  1815. err_dir.mkdir(exist_ok=True)
  1816. shutil.copy(shot_check, str(err_dir / f"unknown_{int(time.time())}.png"))
  1817. print(f" ⚠ 连续{UNKNOWN_REENTER_LIMIT}次进入页面均未加载({'白屏' if _ws else '其他异常'}),跳过本商品")
  1818. _LAST_SKIP_WHITE = _ws
  1819. return "__SKIP__"
  1820. print(f" 确认unknown(第{unknown_round}次,{'白屏' if _ws else '非白屏'}),back一层并重新点击进入")
  1821. ex.driver.press("back")
  1822. human_sleep(1.5)
  1823. if shop_xy:
  1824. ex.tap(shop_xy[0], shop_xy[1])
  1825. human_sleep(2.5)
  1826. human_sleep(2)
  1827. else:
  1828. # 检测循环耗尽仍未正常 → 同样按跳过处理(不重启)
  1829. import shutil
  1830. _ws = _is_white_screen(shot_check)
  1831. err_dir = SCREENSHOT_DIR / ex.device_id / "step4" / "unrecognized"
  1832. err_dir.mkdir(exist_ok=True)
  1833. shutil.copy(shot_check, str(err_dir / f"unknown_{int(time.time())}.png"))
  1834. print(f" ⚠ 检测循环耗尽({ptype},{'白屏' if _ws else '其他异常'}),跳过本商品")
  1835. _LAST_SKIP_WHITE = _ws
  1836. return "__SKIP__"
  1837. # normal → 继续
  1838. # ── 店铺页判断:OCR同时检测到「刚刚搜过」和「评价」说明在店铺页 ──
  1839. in_shop = False
  1840. for _ in range(10):
  1841. shop_check = _pfx("shop_check.png")
  1842. ex.driver.screenshot(shop_check)
  1843. shop_raw = OCR.recognize(shop_check, detail="text")
  1844. has_ganggang = any("刚刚搜过" in t for t in shop_raw)
  1845. has_pingjia = any("评价" in t for t in shop_raw)
  1846. if has_ganggang and has_pingjia:
  1847. in_shop = True
  1848. print(f" 已确认在店铺页")
  1849. break
  1850. human_sleep(1)
  1851. if not in_shop:
  1852. print(f" ⚠ 未检测到店铺页,继续尝试...")
  1853. # ── 第1步:截图 + AI找商品标题坐标(AI失败或返回None时重试3次,页面可能未加载完)──
  1854. system_prompt = """你收到店铺页的OCR文字。商品标题文字坐标已知(从OCR中有x,y)。
  1855. 请找到和以下商品标题匹配的文字块,返回其点击坐标。
  1856. 【重要规则】
  1857. - 坐标必须从OCR数据中选取,不得编造或估算
  1858. - 如果找不到完全匹配的,找最相似的
  1859. - 如果完全找不到,返回null
  1860. 只返回JSON:
  1861. {"title_xy": [x, y] 或 null, "shop": "店铺名"}"""
  1862. parser = AIParser()
  1863. title_xy = None
  1864. for ai_try in range(3):
  1865. shot = _pfx("shop.png")
  1866. ex.driver.screenshot(shot)
  1867. raw = OCR.recognize(shot, detail="all")
  1868. sorted_r = sorted(raw, key=lambda r: r["bbox"][0][1])
  1869. lines = []
  1870. for r in sorted_r:
  1871. cx = (r["bbox"][0][0] + r["bbox"][2][0]) // 2
  1872. cy = (r["bbox"][0][1] + r["bbox"][2][1]) // 2
  1873. lines.append(f"[x={cx:4d}, y={cy:4d}] {r['text']}")
  1874. ocr_text = "\n".join(lines)
  1875. resp = parser._call(system_prompt, f"商品标题: {product_title}\n\nOCR文字:\n{ocr_text}\n\n请返回商品标题坐标。")
  1876. import json
  1877. cleaned = resp.strip()
  1878. if cleaned.startswith("```"):
  1879. cl = cleaned.split("\n")
  1880. if cl[0].startswith("```"): cl = cl[1:]
  1881. if cl and cl[-1].strip() == "```": cl = cl[:-1]
  1882. cleaned = "\n".join(cl).strip()
  1883. try:
  1884. data = json.loads(cleaned)
  1885. title_xy = data.get("title_xy")
  1886. except json.JSONDecodeError:
  1887. title_xy = None
  1888. if title_xy and isinstance(title_xy, list) and len(title_xy) == 2:
  1889. break
  1890. print(f" ⚠ AI未返回有效坐标(第{ai_try+1}次): {title_xy},2秒后重试")
  1891. human_sleep(2)
  1892. if not title_xy or not isinstance(title_xy, list) or len(title_xy) != 2:
  1893. print(f" ⚠ AI 3次均未返回有效坐标: {title_xy}")
  1894. return ""
  1895. tx, ty = title_xy
  1896. if tx is None or ty is None:
  1897. print(f" ⚠ AI返回空坐标")
  1898. return ""
  1899. tx, ty = int(tx), int(ty)
  1900. w, h = ex.driver.window_size()
  1901. if not (0 <= tx <= w and 0 <= ty <= h):
  1902. print(f" ⚠ 坐标越界: ({tx},{ty}) 超出屏幕 {w}x{h}")
  1903. return ""
  1904. # ── 第2步:点击商品标题 → 进入商品详情(最多重试3次)──
  1905. entered_detail = False
  1906. for attempt in range(3):
  1907. print(f" 点击商品标题: ({tx},{ty}) (第{attempt+1}次)")
  1908. ex.tap(tx, ty)
  1909. human_sleep(3) # 点击后先等页面响应(验证码处理后/网络慢时切换慢)
  1910. # 检测是否进入商品详情页(验证码检测优先——验证码页文字块少,不能被"加载中"分支挡掉)
  1911. notready_streak = 0 # 连续"未就绪"轮次(白屏时OCR只剩状态栏1~3块文字)
  1912. for _ in range(8):
  1913. human_sleep(2)
  1914. detail_check = _pfx("detail_check.png")
  1915. ex.driver.screenshot(detail_check)
  1916. detail_raw = OCR.recognize(detail_check, detail="all")
  1917. detail_texts = [r["text"] for r in detail_raw]
  1918. # 1. 验证码优先检测:验证码页文字块少(可能≤8),必须先于"加载中"判断
  1919. captcha_kw = any("拖动滑块" in t or "请按住滑块" in t or "安全验证" in t for t in detail_texts)
  1920. if captcha_kw:
  1921. print(f" ⚠ 检测到验证码页面,尝试自动处理...")
  1922. if _handle_captcha(ex, detail_texts):
  1923. continue
  1924. return "__TERMINATE__"
  1925. # 1.5 「重新加载」页(出错了/检修中,验证码通过后常见):点重新加载后继续等
  1926. if any("重新加载" in t for t in detail_texts):
  1927. for r in detail_raw:
  1928. if "重新加载" in r["text"]:
  1929. box = r["bbox"]
  1930. bx = (box[0][0] + box[2][0]) // 2
  1931. by = (box[0][1] + box[2][1]) // 2
  1932. print(f" 检测到「重新加载」页,点击重新加载 ({bx},{by})")
  1933. ex.tap(bx, by)
  1934. human_sleep(2)
  1935. break
  1936. continue
  1937. # 2. 页面加载中/切换中(文字少且无验证码)→ 继续等待,不误判
  1938. if len(detail_texts) <= 8:
  1939. notready_streak += 1
  1940. # 连续3轮仅状态栏级文字 → 大概率白屏(有图无字),立即跳过别盲等
  1941. # (原逻辑只会"继续等待",3次尝试×8轮能空转1分半)
  1942. if notready_streak >= 3 and _is_white_screen(detail_check):
  1943. _save_evidence(detail_check, "white_skip_detail")
  1944. _LAST_SKIP_WHITE = True
  1945. print(" ⚠ 详情页白屏(连续多轮仅状态栏文字),跳过本商品")
  1946. return "__SKIP__"
  1947. print(f" 页面未就绪(仅{len(detail_texts)}块文字),继续等待...")
  1948. continue
  1949. # 2.5 检测是否退回首页/列表页(点标题失败/back过头时常见)→ 立即处理,不盲等
  1950. joined_texts = "".join(detail_texts)
  1951. if ("看病买药" in joined_texts) and ("我的" in joined_texts):
  1952. _save_evidence(detail_check, "restart_home_kw")
  1953. print(" ⚠ 检测到已退回首页,触发重启恢复")
  1954. return "__RESTART__"
  1955. if "筛选" in joined_texts:
  1956. print(" ⚠ 检测到已回列表页,放弃本店(点标题未成功进店)")
  1957. return ""
  1958. # 3. 检测商品详情页关键词
  1959. if any("加入购物车" in t or "立即购买" in t or "选规格" in t or "商品详情页" in t for t in detail_texts):
  1960. print(f" 已进入商品详情页")
  1961. entered_detail = True
  1962. break
  1963. # 4. 不在详情页,检测是否还在店铺页
  1964. has_ganggang = any("刚刚搜过" in t for t in detail_texts)
  1965. has_pingjia = any("评价" in t for t in detail_texts)
  1966. if has_ganggang and has_pingjia:
  1967. print(f" 仍在店铺页,重试...")
  1968. break # 跳出内层循环
  1969. if entered_detail:
  1970. break
  1971. # 内层循环结束仍未进入详情页:AI判断当前实际页面类型(诊断 + 验证码/首页/列表页兜底)
  1972. try:
  1973. last_shot = _pfx("detail_check.png")
  1974. if cv2.imread(last_shot) is not None:
  1975. ai_raw = OCR.recognize(last_shot, detail="all")
  1976. ai_page = AIParser().check_page(ai_raw)
  1977. print(f" AI页面类型: {ai_page.get('type')} - {ai_page.get('detail', '')[:40]}")
  1978. if ai_page.get("type") == "risk":
  1979. if not _has_captcha_kw([r["text"] for r in ai_raw]):
  1980. print(" AI判risk但OCR无验证码特征词,忽略")
  1981. else:
  1982. print(" ⚠ AI检测到验证码页,尝试自动处理...")
  1983. if _handle_captcha(ex, [r["text"] for r in ai_raw]):
  1984. continue
  1985. return "__TERMINATE__"
  1986. if ai_page.get("type") == "home":
  1987. _save_evidence(last_shot, "restart_home_ai")
  1988. print(" ⚠ AI检测到首页,触发重启恢复")
  1989. return "__RESTART__"
  1990. if ai_page.get("type") == "list":
  1991. print(" ⚠ AI检测到列表页,放弃本店")
  1992. return ""
  1993. if ai_page.get("type") == "login":
  1994. ACCOUNT_ABORTED = True # 本函数已声明global
  1995. print(" ⚠ AI检测到登录页(账号被踢/封号),停止采集")
  1996. return "__TERMINATE__"
  1997. except Exception as e:
  1998. print(f" ⚠ AI页面类型检测异常: {e}")
  1999. # 第1次失败后,重新OCR+AI获取坐标(可能是页面滚动导致坐标偏移)
  2000. if attempt < 2:
  2001. print(f" 重新OCR获取坐标...")
  2002. re_shot = _pfx("shop.png")
  2003. ex.driver.screenshot(re_shot)
  2004. re_raw = OCR.recognize(re_shot, detail="all")
  2005. re_sorted = sorted(re_raw, key=lambda r: r["bbox"][0][1])
  2006. re_lines = []
  2007. for r in re_sorted:
  2008. r_cx = (r["bbox"][0][0] + r["bbox"][2][0]) // 2
  2009. r_cy = (r["bbox"][0][1] + r["bbox"][2][1]) // 2
  2010. re_lines.append(f"[x={r_cx:4d}, y={r_cy:4d}] {r['text']}")
  2011. re_ocr_text = "\n".join(re_lines)
  2012. re_resp = parser._call(system_prompt, f"商品标题: {product_title}\n\nOCR文字:\n{re_ocr_text}\n\n请返回商品标题坐标。")
  2013. re_cleaned = re_resp.strip()
  2014. if re_cleaned.startswith("```"):
  2015. rl = re_cleaned.split("\n")
  2016. if rl[0].startswith("```"): rl = rl[1:]
  2017. if rl and rl[-1].strip() == "```": rl = rl[:-1]
  2018. re_cleaned = "\n".join(rl).strip()
  2019. try:
  2020. re_data = json.loads(re_cleaned)
  2021. re_xy = re_data.get("title_xy")
  2022. if re_xy and isinstance(re_xy, list) and len(re_xy) == 2 and re_xy[0] is not None:
  2023. tx, ty = int(re_xy[0]), int(re_xy[1])
  2024. ww, hh = ex.driver.window_size()
  2025. if not (0 <= tx <= ww and 0 <= ty <= hh):
  2026. print(f" ⚠ 新坐标越界: ({tx},{ty}),保持原坐标")
  2027. else:
  2028. print(f" 新坐标: ({tx},{ty})")
  2029. except Exception:
  2030. pass
  2031. else:
  2032. pass # 3次重试结束
  2033. if not entered_detail:
  2034. print(f" ⚠ 3次点击未进入商品详情页,跳过")
  2035. return ""
  2036. # ── 第3步:ORB特征匹配找分享图标 ──
  2037. share_shot = _pfx("find_share.png")
  2038. ex.driver.screenshot(share_shot)
  2039. screen = cv2.imread(share_shot)
  2040. template_path = str(Path(__file__).parent / "files" / "share.png")
  2041. template = cv2.imread(template_path)
  2042. sx, sy = None, None
  2043. if screen is not None and template is not None:
  2044. h_s, w_s = screen.shape[:2]
  2045. # 右上角区域(分享图标永远在右上)
  2046. roi_x1, roi_y1 = w_s * 2 // 3, 0
  2047. roi = screen[roi_y1:h_s // 4, roi_x1:w_s]
  2048. # 方法A: SIFT 特征匹配(限制右上角区域,减少干扰)
  2049. sx, sy = None, None
  2050. sift = cv2.SIFT_create(nfeatures=1500)
  2051. kp1, des1 = sift.detectAndCompute(template, None)
  2052. kp2, des2 = sift.detectAndCompute(roi, None)
  2053. if des1 is not None and des2 is not None and len(kp1) >= 2 and len(kp2) >= 2:
  2054. bf = cv2.BFMatcher()
  2055. matches = bf.knnMatch(des1, des2, k=2)
  2056. good = []
  2057. for m, n in matches:
  2058. if m.distance < 0.75 * n.distance:
  2059. good.append(m)
  2060. print(f" 分享SIFT(右上区域): 模板{len(kp1)}特征 ROI{len(kp2)}特征 优质{len(good)}")
  2061. if len(good) >= 4:
  2062. src_pts = np.float32([kp1[m.queryIdx].pt for m in good]).reshape(-1, 1, 2)
  2063. dst_pts = np.float32([kp2[m.trainIdx].pt for m in good]).reshape(-1, 1, 2)
  2064. matrix, _ = cv2.findHomography(src_pts, dst_pts, cv2.RANSAC, 5.0)
  2065. if matrix is not None:
  2066. h_t, w_t = template.shape[:2]
  2067. corners = np.float32([[0, 0], [w_t, 0], [w_t, h_t], [0, h_t]]).reshape(-1, 1, 2)
  2068. transformed = cv2.perspectiveTransform(corners, matrix)
  2069. sx = roi_x1 + int(np.mean(transformed[:, 0, 0]))
  2070. sy = int(np.mean(transformed[:, 0, 1]))
  2071. print(f" 分享SIFT匹配: ({sx},{sy})")
  2072. # 方法B: 多尺度模板匹配(右上角区域)
  2073. # 分辨率适配:以 720p 为基准,按屏幕宽度比例调整搜索尺度,覆盖 480p~1080p
  2074. base_w = 720.0
  2075. scale_ratio = w_s / base_w
  2076. # 模板在 720p 下约占 11% 屏宽,目标尺度应使模板覆盖相同比例
  2077. scales = [round(scale_ratio * s, 2) for s in [0.7, 0.8, 0.9, 1.0, 1.1, 1.2, 1.3, 1.4, 1.5]]
  2078. if sx is None:
  2079. best_val, best_loc, best_sw, best_sh = 0, None, 0, 0
  2080. for scale in scales:
  2081. scaled = cv2.resize(template, None, fx=scale, fy=scale)
  2082. sw, sh = scaled.shape[1], scaled.shape[0]
  2083. if sh > roi.shape[0] or sw > roi.shape[1]:
  2084. continue
  2085. res = cv2.matchTemplate(roi, scaled, cv2.TM_CCOEFF_NORMED)
  2086. _, mv, _, ml = cv2.minMaxLoc(res)
  2087. if mv > best_val:
  2088. best_val, best_loc, best_sw, best_sh = mv, ml, sw, sh
  2089. t_edge = cv2.Canny(scaled, 30, 100)
  2090. r_edge = cv2.Canny(roi, 30, 100)
  2091. if t_edge.shape[0] <= r_edge.shape[0] and t_edge.shape[1] <= r_edge.shape[1]:
  2092. res2 = cv2.matchTemplate(r_edge, t_edge, cv2.TM_CCOEFF_NORMED)
  2093. _, mv2, _, ml2 = cv2.minMaxLoc(res2)
  2094. if mv2 > best_val:
  2095. best_val, best_loc, best_sw, best_sh = mv2, ml2, sw, sh
  2096. print(f" 分享模板匹配(右上): 最佳={best_val:.3f}")
  2097. if best_val >= 0.26 and best_loc is not None:
  2098. sx = roi_x1 + best_loc[0] + best_sw // 2
  2099. sy = best_loc[1] + best_sh // 2
  2100. if sx is not None and sy is not None:
  2101. print(f" 分享图标: ({sx},{sy})")
  2102. ex.tap(sx, sy)
  2103. else:
  2104. print(f" ⚠ 未找到分享图标")
  2105. return ""
  2106. # 等弹窗出现,同时记录"分享到"y坐标用于QR裁剪
  2107. share_y = None
  2108. waimai_y = None
  2109. for _ in range(8):
  2110. human_sleep(1)
  2111. ck = _pfx("share_popup.png")
  2112. ex.driver.screenshot(ck)
  2113. detail = OCR.recognize(ck, detail="all")
  2114. texts = [r["text"] for r in detail]
  2115. if any("分享到" in t for t in texts):
  2116. print(f" 分享弹窗出现")
  2117. # 记录"分享到"和"外卖"的y坐标
  2118. for r in detail:
  2119. cy = (r["bbox"][0][1] + r["bbox"][2][1]) // 2
  2120. if "分享到" in r["text"] and share_y is None:
  2121. share_y = cy
  2122. if "外卖" in r["text"] and waimai_y is None:
  2123. waimai_y = cy
  2124. break
  2125. if share_y is None:
  2126. share_y = int(ex.driver.window_size()[1] * 0.74) # fallback
  2127. # ── 第4步:截图 → 多方法解析二维码(多次重试) ──
  2128. import os as _os_debug
  2129. _debug_dir = str(SCREENSHOT_DIR / ex.device_id / "step4" / "debug_qr")
  2130. _os_debug.makedirs(_debug_dir, exist_ok=True)
  2131. # 微信QR解码器(对ECI编码的饿了么QR鲁棒,实测成功率94%)
  2132. _wx_detector = cv2.wechat_qrcode.WeChatQRCode() if hasattr(cv2, "wechat_qrcode") else None
  2133. def _decode_qr(img, share_y, waimai_y=None):
  2134. """基于OCR定位的share_y裁剪QR区域解析"""
  2135. if img is None: return ""
  2136. h, w = img.shape[:2]
  2137. gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)
  2138. detector = cv2.QRCodeDetector()
  2139. # 裁剪区域:y从"分享到"上方推算(不依赖"外卖"文字,避免OCR误判)
  2140. # QR 通常在"分享到"上方 15%~35% 屏高范围,取中间偏下
  2141. y_top = max(0, share_y - int(h * 0.30))
  2142. y_bot = share_y
  2143. x_l, x_r = int(w * 0.58), int(w * 0.94)
  2144. crop_save = img[y_top:y_bot, x_l:x_r]
  2145. cv2.imwrite(_pfx("qr_crop.png"), crop_save)
  2146. # 保存调试截图:标注裁剪区域 + QR 边界(按设备ID区分,方便对比不同分辨率)
  2147. debug_img = img.copy()
  2148. cv2.rectangle(debug_img, (x_l, y_top), (x_r, y_bot), (0, 255, 0), 3)
  2149. cv2.putText(debug_img, f"crop:({x_l},{y_top})~({x_r},{y_bot})", (x_l, y_top - 10),
  2150. cv2.FONT_HERSHEY_SIMPLEX, 0.8, (0, 255, 0), 2)
  2151. _ts = int(time.time())
  2152. cv2.imwrite(f"{_debug_dir}/{ex.device_id}_qr_region_{_ts}.png", debug_img)
  2153. # 方法0: 微信QR解码器(对ECI编码的饿了么QR鲁棒,实测94%成功率,全图直接解析)
  2154. if _wx_detector is not None:
  2155. try:
  2156. wx_texts, wx_points = _wx_detector.detectAndDecode(img)
  2157. if wx_texts and wx_texts[0]:
  2158. data = wx_texts[0]
  2159. if wx_points is not None and len(wx_points) > 0:
  2160. wp = wx_points[0].astype(int)
  2161. x1, y1 = wp[:, 0].min(), wp[:, 1].min()
  2162. x2, y2 = wp[:, 0].max(), wp[:, 1].max()
  2163. cv2.rectangle(debug_img, (x1, y1), (x2, y2), (0, 0, 255), 3)
  2164. cv2.putText(debug_img, "WX-QR", (x1, y1 - 10),
  2165. cv2.FONT_HERSHEY_SIMPLEX, 0.8, (0, 0, 255), 2)
  2166. cv2.imwrite(f"{_debug_dir}/{ex.device_id}_qr_found_{_ts}.png", debug_img)
  2167. return data
  2168. except Exception:
  2169. pass
  2170. def _try_decode(roi_gray, zooms=(1,)):
  2171. """在灰度图上尝试多种方式解码"""
  2172. if roi_gray is None or roi_gray.size == 0 or roi_gray.shape[0] == 0 or roi_gray.shape[1] == 0:
  2173. return ""
  2174. # 高分辨率下 QR 可能太大导致 detect 失败,先缩到合理尺寸再解析
  2175. h_roi, w_roi = roi_gray.shape[:2]
  2176. if max(h_roi, w_roi) > 500:
  2177. scale_down = 500 / max(h_roi, w_roi)
  2178. roi_small = cv2.resize(roi_gray, None, fx=scale_down, fy=scale_down, interpolation=cv2.INTER_AREA)
  2179. else:
  2180. roi_small = roi_gray
  2181. for z in zooms:
  2182. if z > 1:
  2183. big = cv2.resize(roi_small, None, fx=z, fy=z, interpolation=cv2.INTER_NEAREST)
  2184. else:
  2185. big = roi_small
  2186. data, _, _ = detector.detectAndDecode(big)
  2187. if data: return data
  2188. # OTSU + zoom
  2189. for z in zooms:
  2190. _, th = cv2.threshold(roi_small, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)
  2191. big = cv2.resize(th, None, fx=z, fy=z, interpolation=cv2.INTER_NEAREST) if z > 1 else th
  2192. data, _, _ = detector.detectAndDecode(big)
  2193. if data: return data
  2194. return ""
  2195. # 方法A: 全图detect定位QR → 中心 ±qr_half 精确裁剪解析(qr_half 随分辨率缩放)
  2196. # QR 在全图里 detect 可能失败(干扰太多),但先试试
  2197. ok, points = detector.detect(gray)
  2198. if ok and points is not None and len(points) > 0:
  2199. pts = points[0].astype(int)
  2200. cx = int(np.mean(pts[:, 0]))
  2201. cy = int(np.mean(pts[:, 1]))
  2202. # 根据 detect 到的 QR 边界估算大小,裁剪 QR 中心 ± qr_half
  2203. qr_half = max(int(max(np.linalg.norm(pts[0] - pts[1]), np.linalg.norm(pts[1] - pts[2])) / 2) + 20, 60)
  2204. x1, y1 = max(0, cx - qr_half), max(0, cy - qr_half)
  2205. x2, y2 = min(w, cx + qr_half), min(h, cy + qr_half)
  2206. if x2 > x1 and y2 > y1:
  2207. data = _try_decode(gray[y1:y2, x1:x2], (1, 2, 3))
  2208. if data:
  2209. # 在调试截图上标注 QR 检测位置
  2210. cv2.rectangle(debug_img, (x1, y1), (x2, y2), (0, 0, 255), 3)
  2211. cv2.putText(debug_img, f"QR:({cx},{cy})", (x1, y1 - 10),
  2212. cv2.FONT_HERSHEY_SIMPLEX, 0.8, (0, 0, 255), 2)
  2213. cv2.imwrite(f"{_debug_dir}/{ex.device_id}_qr_found_{_ts}.png", debug_img)
  2214. return data
  2215. # 方法B: 用 detector.detect 在扫描区域内定位 QR → 中心 ± qr_half 精确裁剪解析
  2216. scan_area = gray[y_top:y_bot, x_l:x_r]
  2217. sh, sw = scan_area.shape
  2218. ok2, pts2 = detector.detect(scan_area)
  2219. if ok2 and pts2 is not None and len(pts2) > 0:
  2220. qr_pts = pts2[0].astype(int)
  2221. qx = int(np.mean(qr_pts[:, 0]))
  2222. qy = int(np.mean(qr_pts[:, 1]))
  2223. qr_half = max(int(max(np.linalg.norm(qr_pts[0] - qr_pts[1]), np.linalg.norm(qr_pts[1] - qr_pts[2])) / 2) + 20, 60)
  2224. qx1, qy1 = max(0, qx - qr_half), max(0, qy - qr_half)
  2225. qx2, qy2 = min(sw, qx + qr_half), min(sh, qy + qr_half)
  2226. if qx2 > qx1 and qy2 > qy1:
  2227. data = _try_decode(scan_area[qy1:qy2, qx1:qx2], (1, 2, 3))
  2228. if data:
  2229. cv2.rectangle(debug_img, (x_l + qx1, y_top + qy1), (x_l + qx2, y_top + qy2), (0, 0, 255), 3)
  2230. cv2.putText(debug_img, f"QR:({x_l + qx},{y_top + qy})", (x_l + qx1, y_top + qy1 - 10),
  2231. cv2.FONT_HERSHEY_SIMPLEX, 0.8, (0, 0, 255), 2)
  2232. cv2.imwrite(f"{_debug_dir}/{ex.device_id}_qr_found_{_ts}.png", debug_img)
  2233. return data
  2234. # 方法C: 滑动窗口扫描(基于OCR定位区域,尺寸随分辨率缩放)
  2235. win = max(40, min(200, sh // 2, sw // 2))
  2236. step = max(40, win // 3)
  2237. for y in range(0, max(1, sh - win), step):
  2238. for x in range(0, max(1, sw - win), step):
  2239. patch = scan_area[y:y+win, x:x+win]
  2240. data = _try_decode(patch, (1, 2))
  2241. if data:
  2242. # 在调试截图上标注命中的窗口位置
  2243. cv2.rectangle(debug_img, (x_l + x, y_top + y), (x_l + x + win, y_top + y + win), (0, 0, 255), 3)
  2244. cv2.imwrite(f"{_debug_dir}/{ex.device_id}_qr_found_{_ts}.png", debug_img)
  2245. return data
  2246. # 方法D: 全图 OTSU + detect → 中心 ± qr_half 精确裁剪解析
  2247. _, full_th = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)
  2248. ok3, pts3 = detector.detect(full_th)
  2249. if ok3 and pts3 is not None and len(pts3) > 0:
  2250. qr_pts = pts3[0].astype(int)
  2251. qx = int(np.mean(qr_pts[:, 0]))
  2252. qy = int(np.mean(qr_pts[:, 1]))
  2253. qr_half = max(int(max(np.linalg.norm(qr_pts[0] - qr_pts[1]), np.linalg.norm(qr_pts[1] - qr_pts[2])) / 2) + 20, 60)
  2254. qx1, qy1 = max(0, qx - qr_half), max(0, qy - qr_half)
  2255. qx2, qy2 = min(w, qx + qr_half), min(h, qy + qr_half)
  2256. if qx2 > qx1 and qy2 > qy1:
  2257. data = _try_decode(gray[qy1:qy2, qx1:qx2], (1, 2, 3))
  2258. if data:
  2259. cv2.rectangle(debug_img, (qx1, qy1), (qx2, qy2), (0, 0, 255), 3)
  2260. cv2.putText(debug_img, f"QR:({qx},{qy})", (qx1, qy1 - 10),
  2261. cv2.FONT_HERSHEY_SIMPLEX, 0.8, (0, 0, 255), 2)
  2262. cv2.imwrite(f"{_debug_dir}/{ex.device_id}_qr_found_{_ts}.png", debug_img)
  2263. return data
  2264. return ""
  2265. for retry in range(6): # 最多等 5+2*5=15秒
  2266. time.sleep(5 if retry == 0 else 2)
  2267. qr_shot = _pfx("qr.png")
  2268. ex.driver.screenshot(qr_shot)
  2269. data = _decode_qr(cv2.imread(qr_shot), share_y, waimai_y)
  2270. if data:
  2271. print(f" QR链接: {data[:100]}")
  2272. return data
  2273. return ""
  2274. # ── 步骤 3:滑动 + 逐个点击店铺 ──────────────────────
  2275. def _progress_file_path(device_id: str, keyword: str) -> str:
  2276. """进度文件路径(美团同款:ycwj/{设备}_{药品}.txt)"""
  2277. import hashlib
  2278. safe = hashlib.md5(keyword.encode()).hexdigest()[:8]
  2279. return str(Path(__file__).parent / "ycwj" / f"{device_id}_{safe}.txt")
  2280. def _save_progress(device_id: str, keyword: str, visited: set, scroll_px: int, batch_no: int) -> None:
  2281. """保存采集进度(每批滑动后调用,异常退出时进度已在)"""
  2282. import os
  2283. try:
  2284. path = _progress_file_path(device_id, keyword)
  2285. os.makedirs(os.path.dirname(path), exist_ok=True)
  2286. data = {
  2287. "visited": sorted(visited),
  2288. "scroll_px": scroll_px,
  2289. "batch_no": batch_no,
  2290. "time": time.strftime("%Y-%m-%d %H:%M:%S"),
  2291. }
  2292. with open(path, "w", encoding="utf-8") as f:
  2293. json.dump(data, f, ensure_ascii=False, indent=2)
  2294. except Exception as e:
  2295. print(f"[step3] 保存进度失败: {e}")
  2296. def _load_progress(device_id: str, keyword: str):
  2297. """读取采集进度,无进度文件返回 None"""
  2298. import os
  2299. try:
  2300. path = _progress_file_path(device_id, keyword)
  2301. if not os.path.exists(path):
  2302. return None
  2303. with open(path, "r", encoding="utf-8") as f:
  2304. return json.load(f)
  2305. except Exception as e:
  2306. print(f"[step3] 读取进度失败: {e}")
  2307. return None
  2308. def _delete_progress(device_id: str, keyword: str) -> None:
  2309. """任务正常完成后删除进度文件"""
  2310. import os
  2311. try:
  2312. path = _progress_file_path(device_id, keyword)
  2313. if os.path.exists(path):
  2314. os.remove(path)
  2315. print(f"[step3] 进度文件已删除: {path}")
  2316. except Exception as e:
  2317. print(f"[step3] 删除进度失败: {e}")
  2318. def _spec_ok(title: str, spec_list: list) -> bool:
  2319. """
  2320. 标题是否包含任一目标规格(美团 is_link_spec_useful 同款)。
  2321. 规格是数字+单位,OCR对数字错误率极低,用精确匹配(模糊匹配会把"10袋"误配到"10g")。
  2322. """
  2323. if not spec_list:
  2324. return True
  2325. nt = normalize_match_text(title)
  2326. return any(normalize_match_text(s) in nt for s in spec_list)
  2327. def _recover_to_list(ex: SafeExecutor, brand_keyword: str,
  2328. h: int, total_scroll_px: int) -> str:
  2329. """
  2330. step3页面异常分级恢复:识别当前页 → 用最小代价回到搜索结果列表页。
  2331. 返回: list=本来就在列表(误报) back=back退回(列表位置不变,无需滑回)
  2332. research=重新搜索 restart=重启App fail=仍无法确认在列表页
  2333. """
  2334. def _classify() -> str:
  2335. """当前页分类: list/home/page/risk/login/unknown(OCR先查列表页特征「筛选」,其余交给AI)"""
  2336. shot = _shot_path(ex, "recover_pos.png")
  2337. ex.driver.screenshot(shot)
  2338. if cv2.imread(shot) is None:
  2339. ex.driver.screenshot(shot)
  2340. raw = OCR.recognize(shot, detail="all")
  2341. if any("筛选" in r["text"] for r in raw):
  2342. return "list"
  2343. ptype = str(AIParser().check_page(raw).get("type") or "unknown")
  2344. if ptype in ("shop", "detail", "normal", "qrcode"):
  2345. return "page"
  2346. if ptype in ("home", "risk", "login"):
  2347. return ptype
  2348. return "unknown" # 含AI判list但OCR无「筛选」的可疑情况 → 走重启兜底最稳
  2349. def _scroll_back():
  2350. # 从列表顶部滑动恢复到上次采集位置
  2351. remain = total_scroll_px
  2352. while remain > 0:
  2353. step = min(remain, int(h * 0.5))
  2354. _swipe_next_batch(ex, step)
  2355. remain -= step
  2356. human_sleep(2)
  2357. try:
  2358. page = _classify()
  2359. if page == "list":
  2360. return "list"
  2361. if page == "login":
  2362. # 登录页:账号可能被踢,循环顶部 _check_account_kicked 会统一处理
  2363. return "fail"
  2364. if page == "risk":
  2365. # 风控弹窗 → 先走验证码处理再看
  2366. _ai_check_captcha(ex)
  2367. page = _classify()
  2368. if page == "list":
  2369. return "back"
  2370. if page == "page":
  2371. # 认得出是店铺/详情/二维码页 → back退回(列表滚动位置不丢)
  2372. for _ in range(3):
  2373. ex.driver.press("back")
  2374. human_sleep(1.4)
  2375. page = _classify()
  2376. if page == "list":
  2377. return "back"
  2378. if page != "page":
  2379. break
  2380. if page == "home":
  2381. # 在App首页 → 只重新搜索,不重启App
  2382. if not step2_search(ex, brand_keyword):
  2383. return "fail"
  2384. _scroll_back()
  2385. return "research" if _classify() == "list" else "fail"
  2386. if page == "login":
  2387. return "fail"
  2388. # unknown → 认不出在哪 → 重启App
  2389. step1_open_app(ex)
  2390. if not step2_search(ex, brand_keyword):
  2391. return "fail"
  2392. _scroll_back()
  2393. return "restart" if _classify() == "list" else "fail"
  2394. except Exception as e:
  2395. print(f"[step3] 页面恢复异常: {e}")
  2396. return "fail"
  2397. def step3_swipe_and_enter(ex: SafeExecutor, keyword: str, brand: str = "", task: dict = None,
  2398. scheduler=None) -> list:
  2399. """
  2400. 截图 → AI分析 → 逐个点击全部可见店铺 → 下滑加载更多 → 继续点击 → 直到全部遍历
  2401. 品牌+药品名过滤(美团 is_link_useful 同款):标题必须同时包含品牌名和药品名,
  2402. 否则过滤;连续30个无关商品则任务结束停止采集。
  2403. task: 调度任务 dict(task_id/enterprise_id/collect_round/current_page等,手动模式传 None)
  2404. scheduler: 调度器(逐页回告进度,手动模式传 None)
  2405. """
  2406. print("\n" + "=" * 40)
  2407. print(" 步骤 3:遍历店铺")
  2408. print("=" * 40)
  2409. # 新任务开始:重置所有停止标志和缓存(上个任务的验证码/封号停止不能污染本任务)
  2410. global _CLOSE_BTN_CACHE, CAPTCHA_ABORTED, CAPTCHA_ABORT_REASON, ACCOUNT_ABORTED, CURRENT_PAGE
  2411. global WHITE_SCREEN_REASSIGN, WHITE_SCREEN_REASSIGN_REASON, _LAST_SKIP_WHITE
  2412. global _SCHEDULER, _TASK_ID, _CRAWLED_COUNT, _ROW_REST_MARK
  2413. _CLOSE_BTN_CACHE = None
  2414. CAPTCHA_ABORTED = False
  2415. CAPTCHA_ABORT_REASON = ""
  2416. ACCOUNT_ABORTED = False
  2417. WHITE_SCREEN_REASSIGN = False
  2418. WHITE_SCREEN_REASSIGN_REASON = ""
  2419. _LAST_SKIP_WHITE = False
  2420. CURRENT_PAGE = 0
  2421. _SCHEDULER = scheduler
  2422. _TASK_ID = (task or {}).get("task_id")
  2423. _CRAWLED_COUNT = 0
  2424. _ROW_REST_MARK = 0
  2425. w, h = ex.driver.window_size()
  2426. batch_no = 0
  2427. empty_streak = 0 # 连续没有新店铺的批次数
  2428. all_results = []
  2429. unrelated = 0 # 连续无关商品计数(美团同款;>=30 停止采集)
  2430. n_brand = normalize_match_text(brand)
  2431. # 目标规格(调度任务/手动 --spec 传入,如 "120粒|60粒"),用于规格过滤+入库
  2432. spec_raw = str((task or {}).get("product_specs") or "")
  2433. spec_list = (task or {}).get("spec_list") or []
  2434. if isinstance(spec_list, str): # 兼容直接传字符串
  2435. spec_list = [s.strip() for s in re.split(r'[|、,,\n\r]+', spec_list) if s.strip()]
  2436. n_key = normalize_match_text(keyword)
  2437. stopped = False
  2438. # 恢复进度:
  2439. # 1. 跨设备接力:task 带 current_page(调度重派时给)→ 按页码滑动恢复(每页=0.7屏高,跨分辨率一致)
  2440. # 2. 同设备异常恢复:本地进度文件(visited + 滑动px,精确恢复)
  2441. start_page = int((task or {}).get("current_page") or 0)
  2442. progress = _load_progress(ex.device_id, keyword)
  2443. visited = set()
  2444. total_scroll_px = 0
  2445. if start_page > 0:
  2446. print(f"[step3] 跨设备接力恢复: 调度页码={start_page},滑动{start_page}页...")
  2447. for i in range(start_page):
  2448. _swipe_next_batch(ex, int(h * 0.7))
  2449. human_sleep(2)
  2450. elif progress:
  2451. visited = set(progress.get("visited") or [])
  2452. total_scroll_px = int(progress.get("scroll_px") or 0)
  2453. print(f"[step3] 恢复进度: 已访问{len(visited)}个店铺,需滑动恢复{total_scroll_px}px")
  2454. remain = total_scroll_px
  2455. while remain > 0:
  2456. step = min(remain, int(h * 0.5))
  2457. _swipe_next_batch(ex, step)
  2458. remain -= step
  2459. human_sleep(2)
  2460. else:
  2461. visited = set()
  2462. white_restart_count = 0 # 白屏重启恢复累计(>=MAX_WHITE_RESTART → 置重派标志)
  2463. ws_success_count = 0 # 连续成功采集计数(>=WHITE_SUCCESS_RESET_N → 白屏重启额度清零)
  2464. unknown_skip_streak = 0 # 连续"进页失败跳过"计数(跨批次,采到正常商品清零;连续2→分级恢复)
  2465. ab_recover_fail_streak = 0 # a/b页面恢复连续失败计数(只拉长恢复前冷却,不终止任务)
  2466. parse_fail_streak = 0 # 连续"批次截图/识别失败"计数(连续3→终止任务)
  2467. while True:
  2468. if CAPTCHA_ABORTED:
  2469. print(f"[step3] {CAPTCHA_ABORT_REASON or chr(39)+chr(39)}停止采集")
  2470. break
  2471. if scheduler is not None and getattr(scheduler, "limit_reached", False):
  2472. # 回告接口返回 code=error(平台限额/任务已释放)→ 立即停止采集
  2473. print(f"[step3] 调度返回限额/错误({getattr(scheduler, 'limit_msg', '')}),停止采集")
  2474. break
  2475. _check_account_kicked(ex) # 每批检测账号是否被踢/封号(xpath检查,开销小)
  2476. if ACCOUNT_ABORTED:
  2477. print("[step3] 账号被踢/封号,停止采集")
  2478. break
  2479. try:
  2480. named, raw_local, page_status = _get_named_shops(ex, f"step3_b{batch_no}.png", keyword)
  2481. except Exception as e:
  2482. parse_fail_streak += 1
  2483. print(f"[step3] 批次{batch_no} 截图/识别异常({e}),连续{parse_fail_streak}/3")
  2484. if parse_fail_streak >= 3:
  2485. print("[step3] 连续3批截图/识别失败,终止任务")
  2486. stopped = True
  2487. break
  2488. human_sleep(3)
  2489. continue
  2490. parse_fail_streak = 0
  2491. ab_recover_fail_streak = 0 # 本批正常识别(在列表页),页面恢复失败计数清零
  2492. if page_status != "ok":
  2493. # AI判定不在药品列表页(back退过头/走错频道):定向恢复(认得出页面→back/重搜,认不出→重启)
  2494. # a/b类异常永不终止任务;连续失败只拉长恢复前的冷却时间
  2495. if ab_recover_fail_streak >= 3:
  2496. _cd = min(300 * (ab_recover_fail_streak - 2), 900)
  2497. print(f"[step3] 页面恢复连续失败{ab_recover_fail_streak}次,先冷却{_cd}秒再试...")
  2498. if not _chunked_sleep_with_report(_cd, "页面恢复冷却"):
  2499. stopped = True
  2500. break
  2501. print(f"[step3] 当前不在列表页({page_status}),执行页面恢复")
  2502. _save_progress(ex.device_id, keyword, visited, total_scroll_px, batch_no)
  2503. _act = _recover_to_list(ex, (brand + keyword).strip() or keyword, h, total_scroll_px)
  2504. print(f"[step3] 页面恢复动作={_act}")
  2505. if _act == "fail":
  2506. ab_recover_fail_streak += 1
  2507. else:
  2508. ab_recover_fail_streak = 0
  2509. continue
  2510. if _ai_check_captcha(ex): # 列表页每批检测验证码(风控弹窗可能出现在列表)
  2511. human_sleep(1)
  2512. named, raw_local, page_status = _get_named_shops(ex, f"step3_b{batch_no}.png", keyword) # 验证码处理后重新识别
  2513. if page_status != "ok":
  2514. # 验证码处理后仍不在列表页:同样定向恢复(不计数、不终止任务)
  2515. print(f"[step3] 验证码处理后仍不在列表页({page_status}),执行页面恢复")
  2516. _save_progress(ex.device_id, keyword, visited, total_scroll_px, batch_no)
  2517. _act = _recover_to_list(ex, (brand + keyword).strip() or keyword, h, total_scroll_px)
  2518. print(f"[step3] 页面恢复动作={_act}")
  2519. continue
  2520. raw_new = [s for s in named if _shop_key(s) not in visited]
  2521. # 品牌+药品名+规格过滤(测试OCR准确率时可关闭:ENABLE_TITLE_FILTER=False 直接采集全部结果)
  2522. if not ENABLE_TITLE_FILTER:
  2523. # 测试模式:不过滤,直接采集OCR识别到的全部结果
  2524. new_ones = list(raw_new)
  2525. else:
  2526. cloud_cache = {"done": False, "texts": None}
  2527. shot_path = _shot_path(ex, f"step3_b{batch_no}.png")
  2528. new_ones = []
  2529. for s in raw_new:
  2530. # 裁剪确认用标题块自己的坐标(卡片中心裁剪会漏掉标题行)
  2531. title_y = _find_title_y(raw_local, str(s[1] or ""))
  2532. if title_y is None and len(s) > 3 and s[3]:
  2533. title_y = int(s[3][1])
  2534. v, v_reason = _match_verify(str(s[1] or ""), n_brand, n_key, shot_path, cloud_cache,
  2535. card_y=title_y)
  2536. if v in ("ok", "fuzzy"):
  2537. # 规格过滤(美团 is_link_spec_useful 同款):标题需包含任一目标规格
  2538. if spec_list and not _spec_ok(str(s[1] or ""), spec_list):
  2539. unrelated += 1
  2540. print(f"[step3] 过滤: {str(s[1])[:30]} 不含目标规格{spec_list} (连续{unrelated}个无关)")
  2541. if unrelated >= 30:
  2542. print(f"[step3] 连续{unrelated}个非目标商品,任务结束停止采集")
  2543. stopped = True
  2544. break
  2545. continue
  2546. unrelated = 0
  2547. # 标题保持OCR/回贴原样(云端改字已废弃——曾有"温胃舒"被误改成"养胃舒"的错误采集)
  2548. new_ones.append(s)
  2549. else:
  2550. unrelated += 1
  2551. print(f"[step3] 过滤: {str(s[1])[:30]} 不含目标{v_reason} (连续{unrelated}个无关)")
  2552. if unrelated >= 30:
  2553. print(f"[step3] 连续{unrelated}个非目标商品,任务结束停止采集")
  2554. stopped = True
  2555. break
  2556. if stopped:
  2557. print("[step3] 停止采集,返回已采结果")
  2558. break
  2559. print(f"[step3] 批次{batch_no}: 共{len(named)}个, 新{len(raw_new)}个, 通过过滤{len(new_ones)}个")
  2560. # 逐页回告调度:页码 = 起点页码 + 批次数(每滑一屏算一页;global已在任务开头声明)
  2561. CURRENT_PAGE = start_page + batch_no
  2562. _CRAWLED_COUNT = len(all_results)
  2563. if scheduler is not None:
  2564. try:
  2565. scheduler.post_report({
  2566. "task_id": (task or {}).get("task_id"),
  2567. "platform": scheduler.platform,
  2568. "username": scheduler.username,
  2569. "is_finished": 0,
  2570. "need_reassign": 0,
  2571. "current_page": start_page + batch_no,
  2572. "crawled_count": len(all_results),
  2573. })
  2574. except Exception as e:
  2575. print(f"[step3] 逐页回告失败: {e}")
  2576. if getattr(scheduler, "limit_reached", False):
  2577. # 回告返回 code=error(平台限额/任务已释放)→ 立即停止采集
  2578. print(f"[step3] 回告返回限额/错误({getattr(scheduler, 'limit_msg', '')}),停止采集")
  2579. break
  2580. if not new_ones:
  2581. empty_streak += 1
  2582. print(f"[step3] 无新店铺 (连续{empty_streak}/3)")
  2583. if empty_streak >= 3:
  2584. print(f"[step3] 连续3批无新店铺,结束")
  2585. break
  2586. # 滑动后再试
  2587. print(f"[step3] 滑动查看下一批")
  2588. if len(named) >= 2:
  2589. target_y = named[-2][4]
  2590. swipe_dist = max(target_y - int(h * 0.15), int(h * 0.15))
  2591. else:
  2592. swipe_dist = int(h * 0.3)
  2593. _swipe_next_batch(ex, swipe_dist)
  2594. total_scroll_px += swipe_dist
  2595. human_sleep(1.1) # 滑动后停顿 0.8~1.4s(±30%抖动)
  2596. batch_no += 1
  2597. _save_progress(ex.device_id, keyword, visited, total_scroll_px, batch_no)
  2598. continue
  2599. empty_streak = 0 # 有新店铺,重置计数
  2600. ws_recover_requested = False
  2601. home_restart_requested = False
  2602. for shop in new_ones:
  2603. result = _visit_shop(ex, shop, visited, keyword, task, batch_no=batch_no)
  2604. if result and result.get("__terminate__"):
  2605. print("[step3] 收到终止信号,停止遍历")
  2606. all_results = [r for r in all_results if not r.get("__terminate__")]
  2607. stopped = True
  2608. break
  2609. if result and result.get("__restart__"):
  2610. home_restart_requested = True
  2611. break
  2612. if result and result.get("__skip__"):
  2613. # 页面未加载跳过:单次只跳过换下一个商品;连续2个都跳过才升级分级恢复
  2614. unknown_skip_streak += 1
  2615. print(f"[step3] 商品进页失败已跳过(连续{unknown_skip_streak}/2)")
  2616. if unknown_skip_streak >= 2:
  2617. print("[step3] 连续2个商品进页失败,触发白屏分级恢复")
  2618. ws_recover_requested = True
  2619. break
  2620. continue
  2621. if result:
  2622. unknown_skip_streak = 0 # 采到一个正常商品就清零
  2623. ws_success_count += 1
  2624. if ws_success_count >= WHITE_SUCCESS_RESET_N:
  2625. if white_restart_count > 0:
  2626. print(f"[step3] 连续成功{ws_success_count}个商品,白屏重启计数清零(额度恢复)")
  2627. white_restart_count = 0
  2628. ws_success_count = 0
  2629. result["brand"] = brand
  2630. result["product_specs"] = spec_raw
  2631. all_results.append(result)
  2632. save_record(result) # 每采完一个立即入库,中断不丢数据
  2633. _CRAWLED_COUNT = len(all_results)
  2634. # 每爬 ROW_REST_EVERY 条 → 休息 ROW_REST_MIN~MAX 分钟(随机),期间回告防假死
  2635. if _CRAWLED_COUNT // ROW_REST_EVERY > _ROW_REST_MARK:
  2636. _ROW_REST_MARK = _CRAWLED_COUNT // ROW_REST_EVERY
  2637. import random as _random
  2638. rest_min = _random.uniform(ROW_REST_MIN, ROW_REST_MAX)
  2639. print(f" ⏸ 已爬{_CRAWLED_COUNT}条({ROW_REST_EVERY}条整点),休息{rest_min:.1f}分钟...")
  2640. if not _chunked_sleep_with_report(rest_min * 60, "50条"):
  2641. stopped = True
  2642. break
  2643. if ws_recover_requested:
  2644. # 白屏恢复:冷却等待无效 → 直接关App → 休息60~120s → 重启重搜滑回(超限回告调度重派)
  2645. white_restart_count += 1
  2646. if white_restart_count >= MAX_WHITE_RESTART:
  2647. WHITE_SCREEN_REASSIGN = True
  2648. WHITE_SCREEN_REASSIGN_REASON = f"商品页反复白屏,关App重启恢复{white_restart_count}次无效"
  2649. print(f"[step3] 白屏重启恢复超限({MAX_WHITE_RESTART}次),置重派标志,停止采集(进度保留)")
  2650. stopped = True
  2651. break
  2652. print(f"[step3] 检测到白屏,关App休息后重启(第{white_restart_count}/{MAX_WHITE_RESTART}次)")
  2653. _save_progress(ex.device_id, keyword, visited, total_scroll_px, batch_no)
  2654. try:
  2655. ex.driver.app_stop(APP_PACKAGE) # 先把App彻底关掉再休息
  2656. except Exception as _e:
  2657. print(f" 关App失败(不影响流程): {_e}")
  2658. import random as _random
  2659. _rest = _random.uniform(60, 120)
  2660. print(f" 休息{_rest:.0f}秒...")
  2661. if not _chunked_sleep_with_report(_rest, "白屏重启"):
  2662. stopped = True
  2663. break
  2664. step1_open_app(ex)
  2665. step2_search(ex, (brand + keyword).strip() or keyword)
  2666. # 从列表顶部滑动恢复到上次位置
  2667. remain = total_scroll_px
  2668. while remain > 0:
  2669. step = min(remain, int(h * 0.5))
  2670. _swipe_next_batch(ex, step)
  2671. remain -= step
  2672. human_sleep(2)
  2673. unknown_skip_streak = 0
  2674. continue
  2675. if home_restart_requested:
  2676. # 进店被踢回首页等:重启App恢复(不计数、不终止任务)
  2677. print("[step3] 进店被踢回首页/页面异常,重启App恢复")
  2678. _save_progress(ex.device_id, keyword, visited, total_scroll_px, batch_no)
  2679. step1_open_app(ex)
  2680. step2_search(ex, (brand + keyword).strip() or keyword)
  2681. # 从列表顶部滑动恢复到上次位置
  2682. remain = total_scroll_px
  2683. while remain > 0:
  2684. step = min(remain, int(h * 0.5))
  2685. _swipe_next_batch(ex, step)
  2686. remain -= step
  2687. human_sleep(2)
  2688. continue
  2689. if stopped:
  2690. break
  2691. # 正常:滑动到倒数第二个卡片的配送距离位置,进入下一批
  2692. print(f"[step3] 已访问 {len(visited)} 个,滑动查看下一批")
  2693. if len(named) >= 2:
  2694. target_y = named[-2][4] # 倒数第二个卡片的配送距离y坐标
  2695. swipe_dist = max(target_y - int(h * 0.15), int(h * 0.15))
  2696. else:
  2697. swipe_dist = int(h * 0.3)
  2698. _swipe_next_batch(ex, swipe_dist)
  2699. total_scroll_px += swipe_dist
  2700. human_sleep(2)
  2701. batch_no += 1
  2702. _save_progress(ex.device_id, keyword, visited, total_scroll_px, batch_no) # 每页保存一次进度
  2703. continue
  2704. # 任务结束:只有正常完成才删除进度文件;验证码/封号/白屏重派等异常停止时保留(续采位置不丢)
  2705. if not CAPTCHA_ABORTED and not ACCOUNT_ABORTED and not WHITE_SCREEN_REASSIGN:
  2706. _delete_progress(ex.device_id, keyword)
  2707. else:
  2708. print(f"[step3] 异常停止,保留进度文件以便恢复: {_progress_file_path(ex.device_id, keyword)}")
  2709. # ── 输出最终结果表 ──
  2710. print("\n" + "=" * 70)
  2711. print(f" 最终结果 ({len(all_results)} 个店铺)")
  2712. print("=" * 70)
  2713. for i, r in enumerate(all_results, 1):
  2714. link_short = r["link"][:55] + "..." if len(r["link"]) > 55 else r["link"]
  2715. print(f" [{i}] {r['shop']}")
  2716. print(f" 商品: {r['title'][:40]}")
  2717. print(f" 价格: {r['price']}")
  2718. print(f" 月售: {r.get('sales', '')}")
  2719. print(f" 批准文号: {r.get('approval_no', '')}")
  2720. print(f" 有效期: {r.get('validity', '')}")
  2721. print(f" 资质编号: {r.get('license_no', '')}")
  2722. lic = r.get("license") or {}
  2723. print(f" 执照: {lic.get('单位名称', '')} 信用代码:{lic.get('社会信用代码', '')} 法人:{lic.get('法人', '')}")
  2724. print(f" 执照地址: {lic.get('地址', '')[:40]}")
  2725. print(f" 链接: {link_short}")
  2726. print()
  2727. return all_results
  2728. # ── 主入口 ──────────────────────────────────────────────
  2729. if __name__ == "__main__":
  2730. # 启动时自动清理:debug_qr 调试图只保留1天
  2731. try:
  2732. for p in SCREENSHOT_DIR.glob("*/step4/debug_qr/*.png"):
  2733. if time.time() - p.stat().st_mtime > 86400:
  2734. p.unlink(missing_ok=True)
  2735. except Exception:
  2736. pass
  2737. args = sys.argv[1:]
  2738. device_id = "RG5LFYT8UKK7BI95"
  2739. brand = "三九胃泰"
  2740. keyword = "养胃舒颗粒"
  2741. spec_raw = ""
  2742. spec_list = []
  2743. # 解析 --device / --brand / --keyword / --spec 参数(品牌和药品名分开传,对接调度系统)
  2744. filtered = []
  2745. i = 0
  2746. while i < len(args):
  2747. if args[i] == "--device" and i + 1 < len(args):
  2748. device_id = args[i + 1]
  2749. i += 2
  2750. elif args[i] == "--brand" and i + 1 < len(args):
  2751. brand = args[i + 1]
  2752. i += 2
  2753. elif args[i] == "--keyword" and i + 1 < len(args):
  2754. keyword = args[i + 1]
  2755. i += 2
  2756. elif args[i] == "--spec" and i + 1 < len(args):
  2757. spec_raw = args[i + 1]
  2758. spec_list = [s.strip() for s in re.split(r'[|、,,\n\r]+', spec_raw) if s.strip()]
  2759. i += 2
  2760. else:
  2761. filtered.append(args[i])
  2762. i += 1
  2763. cmd = filtered[0] if filtered else "all"
  2764. if not keyword:
  2765. keyword = filtered[1] if len(filtered) > 1 else "矿泉水"
  2766. # 搜索词 = 品牌+药品名+规格 合起来(美团同款:分开配置,搜索时合并)
  2767. search_key = (brand + keyword + spec_raw).strip() or keyword
  2768. print(f"品牌: {brand or '(无)'} | 药品名: {keyword} | 规格: {spec_list or '(不限)'} | 搜索词: {search_key}")
  2769. print("设备连接中...")
  2770. device_id = _find_device(device_id)
  2771. print(f"设备: {device_id}")
  2772. ex = SafeExecutor(device_id)
  2773. if cmd in ("all", "step1"):
  2774. ok = step1_open_app(ex)
  2775. if not ok:
  2776. sys.exit(1)
  2777. if cmd in ("all", "step2"):
  2778. ok = step2_search(ex, search_key)
  2779. if not ok:
  2780. sys.exit(1)
  2781. if cmd in ("all", "step3"):
  2782. task = {"spec_list": spec_list, "product_specs": spec_raw}
  2783. visited = step3_swipe_and_enter(ex, keyword, brand, task)
  2784. print(f"\n最终访问: {visited}")
  2785. sys.exit(0)
  2786. if cmd in ("all", "step4"):
  2787. # step4 需要先跑完 step3 获取所有商品标题,单独跑时需要手动传标题
  2788. title = keyword
  2789. link = step4_parse_qr(ex, title)
  2790. print(f"\n链接: {link}")
  2791. sys.exit(0)
  2792. sys.exit(0)