main1.py 154 KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162636465666768697071727374757677787980818283848586878889909192939495969798991001011021031041051061071081091101111121131141151161171181191201211221231241251261271281291301311321331341351361371381391401411421431441451461471481491501511521531541551561571581591601611621631641651661671681691701711721731741751761771781791801811821831841851861871881891901911921931941951961971981992002012022032042052062072082092102112122132142152162172182192202212222232242252262272282292302312322332342352362372382392402412422432442452462472482492502512522532542552562572582592602612622632642652662672682692702712722732742752762772782792802812822832842852862872882892902912922932942952962972982993003013023033043053063073083093103113123133143153163173183193203213223233243253263273283293303313323333343353363373383393403413423433443453463473483493503513523533543553563573583593603613623633643653663673683693703713723733743753763773783793803813823833843853863873883893903913923933943953963973983994004014024034044054064074084094104114124134144154164174184194204214224234244254264274284294304314324334344354364374384394404414424434444454464474484494504514524534544554564574584594604614624634644654664674684694704714724734744754764774784794804814824834844854864874884894904914924934944954964974984995005015025035045055065075085095105115125135145155165175185195205215225235245255265275285295305315325335345355365375385395405415425435445455465475485495505515525535545555565575585595605615625635645655665675685695705715725735745755765775785795805815825835845855865875885895905915925935945955965975985996006016026036046056066076086096106116126136146156166176186196206216226236246256266276286296306316326336346356366376386396406416426436446456466476486496506516526536546556566576586596606616626636646656666676686696706716726736746756766776786796806816826836846856866876886896906916926936946956966976986997007017027037047057067077087097107117127137147157167177187197207217227237247257267277287297307317327337347357367377387397407417427437447457467477487497507517527537547557567577587597607617627637647657667677687697707717727737747757767777787797807817827837847857867877887897907917927937947957967977987998008018028038048058068078088098108118128138148158168178188198208218228238248258268278288298308318328338348358368378388398408418428438448458468478488498508518528538548558568578588598608618628638648658668678688698708718728738748758768778788798808818828838848858868878888898908918928938948958968978988999009019029039049059069079089099109119129139149159169179189199209219229239249259269279289299309319329339349359369379389399409419429439449459469479489499509519529539549559569579589599609619629639649659669679689699709719729739749759769779789799809819829839849859869879889899909919929939949959969979989991000100110021003100410051006100710081009101010111012101310141015101610171018101910201021102210231024102510261027102810291030103110321033103410351036103710381039104010411042104310441045104610471048104910501051105210531054105510561057105810591060106110621063106410651066106710681069107010711072107310741075107610771078107910801081108210831084108510861087108810891090109110921093109410951096109710981099110011011102110311041105110611071108110911101111111211131114111511161117111811191120112111221123112411251126112711281129113011311132113311341135113611371138113911401141114211431144114511461147114811491150115111521153115411551156115711581159116011611162116311641165116611671168116911701171117211731174117511761177117811791180118111821183118411851186118711881189119011911192119311941195119611971198119912001201120212031204120512061207120812091210121112121213121412151216121712181219122012211222122312241225122612271228122912301231123212331234123512361237123812391240124112421243124412451246124712481249125012511252125312541255125612571258125912601261126212631264126512661267126812691270127112721273127412751276127712781279128012811282128312841285128612871288128912901291129212931294129512961297129812991300130113021303130413051306130713081309131013111312131313141315131613171318131913201321132213231324132513261327132813291330133113321333133413351336133713381339134013411342134313441345134613471348134913501351135213531354135513561357135813591360136113621363136413651366136713681369137013711372137313741375137613771378137913801381138213831384138513861387138813891390139113921393139413951396139713981399140014011402140314041405140614071408140914101411141214131414141514161417141814191420142114221423142414251426142714281429143014311432143314341435143614371438143914401441144214431444144514461447144814491450145114521453145414551456145714581459146014611462146314641465146614671468146914701471147214731474147514761477147814791480148114821483148414851486148714881489149014911492149314941495149614971498149915001501150215031504150515061507150815091510151115121513151415151516151715181519152015211522152315241525152615271528152915301531153215331534153515361537153815391540154115421543154415451546154715481549155015511552155315541555155615571558155915601561156215631564156515661567156815691570157115721573157415751576157715781579158015811582158315841585158615871588158915901591159215931594159515961597159815991600160116021603160416051606160716081609161016111612161316141615161616171618161916201621162216231624162516261627162816291630163116321633163416351636163716381639164016411642164316441645164616471648164916501651165216531654165516561657165816591660166116621663166416651666166716681669167016711672167316741675167616771678167916801681168216831684168516861687168816891690169116921693169416951696169716981699170017011702170317041705170617071708170917101711171217131714171517161717171817191720172117221723172417251726172717281729173017311732173317341735173617371738173917401741174217431744174517461747174817491750175117521753175417551756175717581759176017611762176317641765176617671768176917701771177217731774177517761777177817791780178117821783178417851786178717881789179017911792179317941795179617971798179918001801180218031804180518061807180818091810181118121813181418151816181718181819182018211822182318241825182618271828182918301831183218331834183518361837183818391840184118421843184418451846184718481849185018511852185318541855185618571858185918601861186218631864186518661867186818691870187118721873187418751876187718781879188018811882188318841885188618871888188918901891189218931894189518961897189818991900190119021903190419051906190719081909191019111912191319141915191619171918191919201921192219231924192519261927192819291930193119321933193419351936193719381939194019411942194319441945194619471948194919501951195219531954195519561957195819591960196119621963196419651966196719681969197019711972197319741975197619771978197919801981198219831984198519861987198819891990199119921993199419951996199719981999200020012002200320042005200620072008200920102011201220132014201520162017201820192020202120222023202420252026202720282029203020312032203320342035203620372038203920402041204220432044204520462047204820492050205120522053205420552056205720582059206020612062206320642065206620672068206920702071207220732074207520762077207820792080208120822083208420852086208720882089209020912092209320942095209620972098209921002101210221032104210521062107210821092110211121122113211421152116211721182119212021212122212321242125212621272128212921302131213221332134213521362137213821392140214121422143214421452146214721482149215021512152215321542155215621572158215921602161216221632164216521662167216821692170217121722173217421752176217721782179218021812182218321842185218621872188218921902191219221932194219521962197219821992200220122022203220422052206220722082209221022112212221322142215221622172218221922202221222222232224222522262227222822292230223122322233223422352236223722382239224022412242224322442245224622472248224922502251225222532254225522562257225822592260226122622263226422652266226722682269227022712272227322742275227622772278227922802281228222832284228522862287228822892290229122922293229422952296229722982299230023012302230323042305230623072308230923102311231223132314231523162317231823192320232123222323232423252326232723282329233023312332233323342335233623372338233923402341234223432344234523462347234823492350235123522353235423552356235723582359236023612362236323642365236623672368236923702371237223732374237523762377237823792380238123822383238423852386238723882389239023912392239323942395239623972398239924002401240224032404240524062407240824092410241124122413241424152416241724182419242024212422242324242425242624272428242924302431243224332434243524362437243824392440244124422443244424452446244724482449245024512452245324542455245624572458245924602461246224632464246524662467246824692470247124722473247424752476247724782479248024812482248324842485248624872488248924902491249224932494249524962497249824992500250125022503250425052506250725082509251025112512251325142515251625172518251925202521252225232524252525262527252825292530253125322533253425352536253725382539254025412542254325442545254625472548254925502551255225532554255525562557255825592560256125622563256425652566256725682569257025712572257325742575257625772578257925802581258225832584258525862587258825892590259125922593259425952596259725982599260026012602260326042605260626072608260926102611261226132614261526162617261826192620262126222623262426252626262726282629263026312632263326342635263626372638263926402641264226432644264526462647264826492650265126522653265426552656265726582659266026612662266326642665266626672668266926702671267226732674267526762677267826792680268126822683268426852686268726882689269026912692269326942695269626972698269927002701270227032704270527062707270827092710271127122713271427152716271727182719272027212722272327242725272627272728272927302731273227332734273527362737273827392740274127422743274427452746274727482749275027512752275327542755275627572758275927602761276227632764276527662767276827692770277127722773277427752776277727782779278027812782278327842785278627872788278927902791279227932794279527962797279827992800280128022803280428052806280728082809281028112812281328142815281628172818281928202821282228232824282528262827282828292830283128322833283428352836283728382839284028412842284328442845284628472848284928502851285228532854285528562857285828592860286128622863286428652866286728682869287028712872287328742875287628772878287928802881288228832884288528862887288828892890289128922893289428952896289728982899290029012902290329042905290629072908290929102911291229132914291529162917291829192920292129222923292429252926292729282929293029312932293329342935293629372938293929402941294229432944294529462947294829492950295129522953295429552956295729582959296029612962296329642965296629672968296929702971297229732974297529762977297829792980298129822983298429852986298729882989299029912992299329942995299629972998299930003001300230033004300530063007300830093010301130123013301430153016301730183019302030213022302330243025302630273028302930303031303230333034303530363037303830393040304130423043304430453046304730483049305030513052305330543055305630573058305930603061306230633064306530663067306830693070307130723073307430753076307730783079308030813082308330843085308630873088308930903091309230933094309530963097309830993100310131023103310431053106310731083109311031113112311331143115311631173118311931203121312231233124312531263127312831293130313131323133313431353136313731383139314031413142314331443145314631473148314931503151315231533154315531563157315831593160316131623163316431653166316731683169317031713172317331743175317631773178317931803181318231833184318531863187318831893190319131923193319431953196319731983199320032013202320332043205320632073208320932103211321232133214321532163217321832193220322132223223322432253226322732283229323032313232323332343235323632373238323932403241324232433244324532463247324832493250325132523253325432553256325732583259326032613262326332643265326632673268326932703271327232733274327532763277327832793280328132823283328432853286328732883289329032913292329332943295329632973298329933003301330233033304330533063307330833093310331133123313331433153316331733183319332033213322332333243325
  1. """
  2. 饿了么闪购 — 主入口 V1
  3. 步骤驱动,按步执行
  4. """
  5. import sys
  6. import os
  7. import time
  8. import re
  9. import json
  10. import shutil
  11. import cv2
  12. import numpy as np
  13. from pathlib import Path
  14. from typing import Optional
  15. sys.path.insert(0, str(Path(__file__).parent / "steps"))
  16. sys.path.insert(0, str(Path(__file__).parent))
  17. from ocr import OCR
  18. from executor import SafeExecutor
  19. from ai_helper1 import AIParser, parse_instructions, extract_value_after, detect_popup
  20. from ai_helper_vision2 import VisionParser # 起送行锚定版(纯规则零AI, 2026-09-20联调通过; 回滚=改回 ai_helper_vision_cross)
  21. from db import save_record, get_existing_license
  22. from snapshot import collect_snapshot
  23. from human_touch import (wrist_arc_pts, wrist_arc_pts_h, motionevent_drag, touchpipe_drag,
  24. wrist_arc_pts_side, RIGHT_SIDE_ZONE, LEFT_SIDE_ZONE)
  25. from commons import err_log
  26. # ── 配置 ────────────────────────────────────────────────
  27. APP_PACKAGE = "me.ele"
  28. SCREENSHOT_DIR = Path(__file__).parent / "screenshots"
  29. # 标题过滤开关:
  30. # True = 按品牌/药品名/规格过滤,只采目标商品(正常模式)
  31. # False = 不过滤,直接采集OCR识别的全部结果入库(测试OCR准确率用)
  32. ENABLE_TITLE_FILTER = False
  33. # 详情页AI提取的店铺名(step4进店确认时更新,供Excel对比列表页OCR店名用)
  34. AI_SHOP_NAME = ""
  35. # 验证码/休息策略:
  36. # 一天内 ≥CAPTCHA_DAILY_LIMIT 次 → 立即停止采集并回告(风控可能封号)
  37. # 每次验证码解决后 → 休息 CAPTCHA_REST_MINUTES~CAPTCHA_REST_MAX_MINUTES 分钟(随机)
  38. # 每爬 ROW_REST_EVERY 条数据 → 休息 ROW_REST_MIN~ROW_REST_MAX 分钟(随机,期间回告防假死)
  39. CAPTCHA_DAILY_LIMIT = 8
  40. CAPTCHA_REST_MINUTES = 1 # 验证码后休息最短分钟数
  41. CAPTCHA_REST_MAX_MINUTES = 5 # 验证码后休息最长分钟数
  42. ROW_REST_EVERY = 50 # 每爬多少条休息一次
  43. ROW_REST_MIN = 10 # 50条整点休息最短分钟数
  44. ROW_REST_MAX = 17 # 50条整点休息最长分钟数
  45. # 进入商品页unknown(页面没加载出来)时的处理:
  46. # 每确认一次unknown → back一层+重新点击进入;累计UNKNOWN_REENTER_LIMIT次加载全失败 → 跳过本商品
  47. # 连续2个商品都这样 → step3白屏恢复:关App休息60~120s → 重启App(最多MAX_WHITE_RESTART次)→ 超限回告调度重派
  48. UNKNOWN_REENTER_LIMIT = 5
  49. MAX_WHITE_RESTART = 3 # 白屏重启恢复上限(超过→置重派标志,回告调度换设备,不终止任务数据)
  50. WHITE_SUCCESS_RESET_N = 5 # 连续成功采N个商品后清零白屏重启计数(额度恢复)
  51. WHITE_EXCEPTION_TYPE = 5 # 白屏重派回告exception_type(复用通用异常值5,remark区分具体原因)
  52. SEARCH_ROUNDS = 5 # step2搜索失败重启重试轮数(共5轮,第2轮起每轮前重启App,全部失败才报错)
  53. # 推荐流检测:卡片标题与「品牌和药品名都不相关」连续达此数 → 判定搜索结果已耗尽、
  54. # 列表进入推荐区 → 结束任务(独立于ENABLE_TITLE_FILTER,生产关闭过滤时也生效)。
  55. # 注意"品牌或药名任一命中"即算相关:同品牌兄弟产品(如999皮炎平/999感冒灵混排)不触发
  56. RECOMMEND_STREAK_END = 20
  57. EMPTY_BATCH_SAFETY = 8 # 无新店铺批次的保底停止数(防纯弹跳底部无限循环;推荐流检测正常时到不了这里)
  58. # 列表上滑策略概率(每次滑动随机): 右手两段式 / 左手两段式 / 原单笔(三者合计1.0)
  59. SWIPE_RIGHT_TWO_P = 0.80
  60. SWIPE_LEFT_TWO_P = 0.10
  61. CAPTCHA_LOG_DIR = Path(__file__).parent / "logs" # 验证码日志目录(每设备独立一个文件,独立计数)
  62. CAPTCHA_ABORTED = False # 全局标志:验证码频繁触发停止采集(调度上报用)
  63. CAPTCHA_ABORT_REASON = "" # 停止原因(调度回告用)
  64. ACCOUNT_ABORTED = False # 全局标志:账号被踢/封号,停止采集(调度上报用)
  65. WHITE_SCREEN_REASSIGN = False # 全局标志:商品页反复白屏,停止采集并回告调度重派(调度上报用)
  66. WHITE_SCREEN_REASSIGN_REASON = "" # 重派原因(调度回告remark用)
  67. _LAST_SKIP_WHITE = False # step4最后一次__SKIP__是否白屏(日志标注用)
  68. CURRENT_PAGE = 0 # 当前页码(逐页回告/终止回告用,调度侧存续采集进度)
  69. # 验证码判定截图存档目录(每次判定出现验证码时截图保存,人工确认是否误判)
  70. CAPTCHA_CHECK_DIR = Path(__file__).parent / "logs" / "captcha_check"
  71. # 休息期间进度回告用的全局上下文(step3 任务开始时设置,_captcha_rest 休息时每10分钟回告)
  72. _SCHEDULER = None
  73. _TASK_ID = None
  74. _CRAWLED_COUNT = 0
  75. _ROW_REST_MARK = 0
  76. OCR = OCR()
  77. def human_sleep(seconds: float, jitter: float = 0.3):
  78. """拟人等待:标称时长 ±30% 随机抖动(操作节奏不被风控建模成固定周期)。
  79. 仅用于 UI 操作间的节奏等待;功能性等待(回告间隔/截图重试)仍用 time.sleep 精确控制。"""
  80. import random as _random
  81. try:
  82. base = float(seconds)
  83. except (TypeError, ValueError):
  84. base = 1.0
  85. time.sleep(max(0.1, base * _random.uniform(1 - jitter, 1 + jitter)))
  86. # 说明书打叉坐标缓存(仅本任务内有效):任务中第一个商品模板匹配找到打叉后,
  87. # 本任务后续商品直接复用该坐标(说明书页布局固定,同设备位置不变);
  88. # 新任务开始时清空,重新匹配。
  89. _CLOSE_BTN_CACHE = None
  90. def _find_device(device_id: str = "") -> str:
  91. import subprocess
  92. r = subprocess.run(["adb", "devices"], capture_output=True, text=True, timeout=5)
  93. devices = []
  94. for line in r.stdout.strip().split("\n")[1:]:
  95. if line.strip() and "device" in line and "offline" not in line:
  96. s = line.split("\t")[0].strip()
  97. if s:
  98. devices.append(s)
  99. if not devices:
  100. raise RuntimeError("未找到设备")
  101. # 如果指定了设备ID,精确匹配
  102. if device_id:
  103. for d in devices:
  104. if d == device_id:
  105. return d
  106. raise RuntimeError(f"未找到指定设备: {device_id},可用设备: {devices}")
  107. # 只有一台直接返回
  108. if len(devices) == 1:
  109. return devices[0]
  110. # 多台设备:列出并让用户选择
  111. print(f"\n发现 {len(devices)} 台设备:")
  112. for i, d in enumerate(devices):
  113. print(f" [{i}] {d}")
  114. while True:
  115. try:
  116. choice = input(f"请选择设备 [0-{len(devices)-1}],回车默认第一台: ").strip()
  117. if choice == "":
  118. return devices[0]
  119. idx = int(choice)
  120. if 0 <= idx < len(devices):
  121. return devices[idx]
  122. except ValueError:
  123. pass
  124. print(f"输入无效,请输入 0-{len(devices)-1}")
  125. def _find_text_in_area(shot_path: str, target: str, max_y: int) -> Optional[dict]:
  126. results = OCR.recognize(shot_path, detail="all")
  127. for r in results:
  128. if target in r["text"]:
  129. y = r["bbox"][0][1]
  130. if y < max_y:
  131. cx = r["bbox"][0][0] + (r["bbox"][2][0] - r["bbox"][0][0]) // 2
  132. cy = y + (r["bbox"][2][1] - y) // 2
  133. return {"x": cx, "y": cy, "text": r["text"], "conf": r["confidence"]}
  134. return None
  135. def _shot_path(ex: SafeExecutor, name: str) -> str:
  136. """截图路径:按 设备ID/步骤 分类组织目录(screenshots/设备/step1/xxx.png)"""
  137. cat = "misc"
  138. for prefix, c in (("step1", "step1"), ("step2", "step2"), ("step3", "step3"),
  139. ("_s4_", "step4"), ("sort", "step2"), ("inst_", "instructions"),
  140. ("lic_", "license"), ("snap_", "snapshot"), ("ad_", "popup")):
  141. if name.startswith(prefix):
  142. cat = c
  143. break
  144. d = SCREENSHOT_DIR / ex.device_id / cat
  145. d.mkdir(parents=True, exist_ok=True)
  146. return str(d / name)
  147. def _save_evidence(shot: str, tag: str):
  148. """决策留档: 触发重启/放弃时刻的截图另存时间戳副本。
  149. page_check/detail_check 同一家店内会被下一轮覆盖,
  150. 留档(时间戳+原因标签)便于事后人工核对AI是否误判。"""
  151. try:
  152. if shot and os.path.exists(shot):
  153. ts = time.strftime("%Y%m%d_%H%M%S")
  154. shutil.copy2(shot, str(Path(shot).parent / f"{ts}_{tag}.png"))
  155. except Exception:
  156. pass
  157. def _screenshot(ex: SafeExecutor, name: str) -> str:
  158. import os
  159. # 按 设备ID/步骤 隔离截图(目录已含设备ID,文件名不再加设备前缀)
  160. path = _shot_path(ex, name)
  161. if os.path.exists(path):
  162. # 保留历史截图副本:固定文件名被 test/调试脚本引用,不能被后续运行覆盖丢失
  163. import shutil
  164. stem, ext = os.path.splitext(name)
  165. backup = str(Path(path).parent / f"{stem}_{int(time.time() * 1000)}{ext}")
  166. try:
  167. shutil.copy2(path, backup)
  168. except Exception:
  169. pass
  170. # 自动清理:每个固定文件只保留最近3份备份,超出删除最旧的
  171. try:
  172. olds = sorted(
  173. Path(path).parent.glob(f"{stem}_[0-9]*{ext}"),
  174. key=lambda p: p.stat().st_mtime, reverse=True,
  175. )
  176. for p in olds[3:]:
  177. p.unlink(missing_ok=True)
  178. except Exception:
  179. pass
  180. # 截图后验证完整性(adb 流式传输可能中断,导致 PNG 损坏),连续5次失败才抛异常
  181. last_err = None
  182. for _try in range(5):
  183. try:
  184. ex.driver.screenshot(path)
  185. except Exception as e:
  186. last_err = e
  187. time.sleep(1)
  188. continue
  189. if cv2.imread(path) is not None:
  190. return path
  191. time.sleep(0.5)
  192. raise RuntimeError(f"截图连续5次损坏/失败: {name} ({last_err})")
  193. def _where_am_i(ex: SafeExecutor) -> str:
  194. """判断当前位置:list=搜索结果列表页 home=首页 other=其他(全屏OCR,区分列表页与首页防止退过头)"""
  195. import os as _os
  196. tmp = _shot_path(ex, "check_pos.png")
  197. ex.driver.screenshot(tmp)
  198. if cv2.imread(tmp) is None:
  199. ex.driver.screenshot(tmp)
  200. texts = [r["text"] for r in OCR.recognize(tmp, detail="all")]
  201. if any("筛选" in t for t in texts):
  202. return "list"
  203. if any("看病买药" in t for t in texts) and any("我的" in t for t in texts):
  204. return "home"
  205. return "other"
  206. # ── 步骤 1:打开 App ────────────────────────────────────
  207. def step1_open_app(ex: SafeExecutor) -> bool:
  208. print("=" * 40)
  209. print(" 步骤 1:打开饿了么闪购")
  210. print("=" * 40)
  211. w, h = ex.driver.window_size()
  212. print(f"[step1] 屏幕尺寸: {w}x{h}")
  213. print(f"[step1] 关闭 {APP_PACKAGE}...")
  214. ex.driver.app_stop(APP_PACKAGE)
  215. human_sleep(2)
  216. print(f"[step1] 启动 {APP_PACKAGE}...")
  217. ex.driver.app_start(APP_PACKAGE)
  218. human_sleep(5)
  219. # 等待首页加载:底部导航「我的」出现(元素没加载完就多等重试,不一次定生死)
  220. for attempt in range(3):
  221. shot = _screenshot(ex, "step1_home.png")
  222. texts = OCR.recognize(shot, rect=[0, int(h * 0.88), w, h], detail="text")
  223. print(f"[step1] 底部识别(第{attempt+1}次): {texts}")
  224. if any("我的" in t for t in texts):
  225. print("[step1] OK - 成功进入 App")
  226. _close_ad_popup(ex) # 每天首启的广告弹窗
  227. return True
  228. human_sleep(3)
  229. print("[step1] FAIL - 多次重试未检测到「我的」")
  230. # 识别不到"我的"可能因为账号被踢/封号——检测登录页xpath确认(存在才是封号)
  231. _check_account_kicked(ex)
  232. return False
  233. # ── 步骤 2:搜索商品 ────────────────────────────────────
  234. def _click_low_price_sort(ex: SafeExecutor) -> bool:
  235. """
  236. 搜索结果排序:点「综合」→ 选「低价优先」(失败不阻塞,找不到就跳过)。
  237. 找不到「低价优先」时再点一次「综合」重试,仍找不到则关闭排序弹窗。
  238. 备用函数:默认不调用(排序已在 step2 快递分支内联),需要按低价优先采集时手动启用。
  239. """
  240. try:
  241. # 1. 找「综合」并点击(排序入口,结果页顶部)
  242. shot = _shot_path(ex, "sort1.png")
  243. ex.driver.screenshot(shot)
  244. btn = None
  245. for r in OCR.recognize(shot, detail="all"):
  246. if "综合" in r["text"]:
  247. box = r["bbox"]
  248. btn = {"x": (box[0][0] + box[2][0]) // 2, "y": (box[0][1] + box[2][1]) // 2}
  249. break
  250. if not btn:
  251. print(" ⚠ 未找到「综合」排序入口,跳过排序")
  252. return False
  253. print(f" 排序: 点击「综合」({btn['x']}, {btn['y']})")
  254. ex.tap(btn["x"], btn["y"])
  255. human_sleep(2)
  256. # 2. 找「低价优先」
  257. shot2 = _shot_path(ex, "sort2.png")
  258. ex.driver.screenshot(shot2)
  259. low = None
  260. for r in OCR.recognize(shot2, detail="all"):
  261. if "低价优先" in r["text"]:
  262. box = r["bbox"]
  263. low = {"x": (box[0][0] + box[2][0]) // 2, "y": (box[0][1] + box[2][1]) // 2}
  264. break
  265. if low:
  266. print(f" 排序: 点击「低价优先」({low['x']}, {low['y']})")
  267. ex.tap(low["x"], low["y"])
  268. human_sleep(3)
  269. return True
  270. # 3. 没找到:再点一次「综合」关闭排序弹窗(点击综合弹出面板,再点一次收起)
  271. print(" ⚠ 未找到「低价优先」,再点一次「综合」关闭排序弹窗")
  272. ex.tap(btn["x"], btn["y"])
  273. human_sleep(1.5)
  274. return False
  275. except Exception as e:
  276. print(f" ⚠ 排序异常: {e}")
  277. return False
  278. def _search_once(ex: SafeExecutor, keyword: str) -> bool:
  279. """单轮搜索(原step2_search主体): 看病买药→搜索→输入→结果确认, 失败返回False"""
  280. print("\n" + "=" * 40)
  281. print(f" 步骤 2:搜索「{keyword}」")
  282. print("=" * 40)
  283. global _POPUP_RESTART_COUNT
  284. _POPUP_RESTART_COUNT = 0 # 每次搜索重置弹窗重启计数(仅本步骤内最多3次)
  285. w, h = ex.driver.window_size()
  286. top_th = int(h * 0.3)
  287. # ── 阶段0+1:点击「看病买药」→ 找「搜索」(失败重试3次,3次失败重启App重试,最多2轮)──
  288. btn = None
  289. for round_idx in range(2):
  290. if round_idx > 0:
  291. # 第2轮:重启 App 重新进入(弹窗/页面异常时清状态)
  292. print("[step2] 重启App重新进入...")
  293. if not step1_open_app(ex):
  294. print("[step2] FAIL - 重启App失败")
  295. return False
  296. # 阶段0:点击「看病买药」(首页频道宫格加载慢 → 轮询等待最多~25s,不急着重启;
  297. # 重启=冷启动反而更慢)
  298. btn_med = None
  299. for wait_i in range(10):
  300. shot0 = _screenshot(ex, "step2_phase0.png")
  301. btn_med = _find_text_in_area(shot0, "看病买药", h)
  302. if btn_med:
  303. break
  304. if wait_i == 0:
  305. print("[step2] 未找到「看病买药」,等待首页频道宫格加载...")
  306. human_sleep(2.5)
  307. if not btn_med:
  308. print(f"[step2] 等待25s仍未找到「看病买药」(第{round_idx+1}轮)")
  309. continue
  310. print(f"[step2] 找到「看病买药」: ({btn_med['x']}, {btn_med['y']})")
  311. ex.tap(btn_med["x"], btn_med["y"])
  312. human_sleep(5)
  313. _close_ad_popup(ex) # 点击买药后可能出现的广告弹窗
  314. # 阶段1:找「搜索」(重试3次,每次关弹窗+等待)
  315. for attempt in range(3):
  316. shot = _screenshot(ex, "step2_phase1.png")
  317. btn = _find_text_in_area(shot, "搜索", top_th)
  318. if btn:
  319. break
  320. print(f"[step2] 未找到「搜索」(第{attempt+1}次),关闭弹窗后重试")
  321. _close_ad_popup(ex)
  322. human_sleep(2)
  323. if btn:
  324. break # 找到搜索,继续
  325. print("[step2] 3次未找到「搜索」,准备重启App重试")
  326. if not btn:
  327. print("[step2] FAIL - 多次重试+重启后仍未找到「搜索」")
  328. return False
  329. print(f"[step2] 找到「搜索」: ({btn['x']}, {btn['y']})")
  330. cx = btn["x"] - 300 # 搜索左边约120px
  331. cy = btn["y"]
  332. print(f"[step2] 点击搜索栏: ({cx}, {cy})")
  333. ex.tap(cx, cy)
  334. human_sleep(5)
  335. # 搜索页确认:「搜索」位置变化才认为进入(加载慢/弹窗遮挡/点击未生效就重试)
  336. moved = False
  337. btn2 = None
  338. for attempt in range(3):
  339. try:
  340. shot2 = _screenshot(ex, "step2_phase2.png")
  341. btn2 = _find_text_in_area(shot2, "搜索", top_th)
  342. except Exception as e:
  343. print(f"[step2] 截图/识别异常(第{attempt+1}次): {e},重试")
  344. human_sleep(2)
  345. continue
  346. if btn2 and (abs(btn2["x"] - btn["x"]) > 50 or abs(btn2["y"] - btn["y"]) > 50):
  347. moved = True
  348. print(f"[step2] 搜索页搜索: ({btn2['x']}, {btn2['y']})")
  349. break
  350. # 位置没变:先关广告弹窗 + AI验证码检测,再重新点击搜索栏(可能点击没生效或被验证码拦截)
  351. if _ai_check_captcha(ex):
  352. print(f"[step2] AI检测到验证码(第{attempt+1}次),处理完成")
  353. if _close_ad_popup(ex):
  354. print(f"[step2] 已关闭广告弹窗(第{attempt+1}次)")
  355. print(f"[step2] 搜索位置未变化(第{attempt+1}次),重新点击搜索栏")
  356. ex.tap(cx, cy)
  357. human_sleep(4)
  358. if not btn2:
  359. print("[step2] FAIL - 进入搜索页后找不到「搜索」")
  360. return False
  361. if not moved:
  362. print("[step2] FAIL - 搜索位置未改变")
  363. return False
  364. cx2 = btn2["x"] - 180 # 搜索页输入框在搜索左边约180px
  365. cy2 = btn2["y"]
  366. ex.tap(cx2, cy2)
  367. human_sleep(2)
  368. print(f"[step2] 聚焦输入框,等待2s")
  369. print(f"[step2] 输入关键词: {keyword}")
  370. ex.driver.set_input_ime(True)
  371. time.sleep(0.3)
  372. ex.driver.send_keys(keyword)
  373. human_sleep(1)
  374. ex.tap(btn2["x"], btn2["y"])
  375. human_sleep(3)
  376. # 搜索后检测列表页:先处理验证码(可能挡住列表页),再检测列表页特征(加载慢就重试)
  377. for attempt in range(5):
  378. try:
  379. shot3 = _screenshot(ex, "step2_result.png")
  380. raw3 = OCR.recognize(shot3, detail="all")
  381. except Exception as e:
  382. print(f"[step2] 截图/识别异常(第{attempt+1}次): {e},重试")
  383. human_sleep(2)
  384. continue
  385. all_texts = [r["text"] for r in raw3]
  386. # 1. 验证码优先:验证码弹窗会挡住列表页特征,先处理再重新检测
  387. # 关键词没命中时用AI再判断一次(九宫格/点击式验证码文字不在关键词里)
  388. if any(("拖动滑块" in t) or ("请按住滑块" in t) or ("安全验证" in t) for t in all_texts):
  389. print(f"[step2] 检测到列表页验证码(第{attempt+1}次),尝试处理...")
  390. if _handle_captcha(ex, all_texts):
  391. print("[step2] 验证码已解决")
  392. else:
  393. print("[step2] 验证码处理失败")
  394. return False
  395. # 验证码通过后可能出现「出错了/检修中」页,点重新加载
  396. if _click_reload(ex):
  397. print("[step2] 已点击重新加载")
  398. human_sleep(2)
  399. continue
  400. if _ai_check_captcha(ex):
  401. print(f"[step2] AI检测到验证码(第{attempt+1}次),处理完成")
  402. continue
  403. # 1.2 广告/活动弹窗(全屏活动页——无遮罩无列表标记时才检测; 列表页上有"红包"字样是正常的)
  404. if not any(("筛选" in t or "快递" in t) for t in all_texts) and _close_ad_popup(ex, force=True):
  405. # back可能只是导航回了上一页(搜索建议页),页面状态已不可信——
  406. # 必须重发搜索(重新点搜索按钮),下一轮再确认结果页
  407. print(f"[step2] 已关闭活动弹窗(第{attempt+1}次),重发搜索")
  408. ex.tap(btn2["x"], btn2["y"])
  409. human_sleep(3)
  410. continue
  411. # 1.5 出错/检修页(无验证码时也可能出现):点「重新加载」后重新检测
  412. if any("重新加载" in t for t in all_texts):
  413. print(f"[step2] 检测到「出错了/检修中」页面,点击重新加载")
  414. _click_reload(ex)
  415. continue
  416. # 2. 列表页特征检测
  417. has_filter = "筛选" in all_texts
  418. has_express = "快递" in all_texts
  419. kw_found = any(keyword in t for t in all_texts)
  420. print(f"[step2] 有筛选: {has_filter}, 有快递: {has_express}, 关键词存在: {kw_found}")
  421. if has_filter or has_express:
  422. # 如果有「快递」则点击它
  423. if has_express:
  424. for r in raw3:
  425. if "快递" in r["text"]:
  426. bx = r["bbox"]
  427. cx = (bx[0][0] + bx[2][0]) // 2
  428. cy = (bx[0][1] + bx[2][1]) // 2
  429. print(f"[step2] 点击「快递」: ({cx}, {cy})")
  430. ex.tap(cx, cy)
  431. human_sleep(3)
  432. break
  433. # 点击「快递」后设置排序:综合 → 低价优先
  434. # _click_low_price_sort(ex) ← 需要按低价优先采集时,去掉行首#即可启用
  435. print("[step2] OK - 搜索成功")
  436. return True
  437. print(f"[step2] 列表页特征未出现(第{attempt+1}次),等3秒重试")
  438. human_sleep(3)
  439. print("[step2] FAIL - 搜索未成功")
  440. return False
  441. def step2_search(ex: SafeExecutor, keyword: str) -> bool:
  442. """搜索总入口:单轮搜索失败/异常 → 重启App重试,共SEARCH_ROUNDS轮,全部失败才报错"""
  443. for round_idx in range(SEARCH_ROUNDS):
  444. if round_idx > 0:
  445. print(f"\n[step2] 第{round_idx}轮搜索失败,重启App重试(重启{round_idx}/{SEARCH_ROUNDS - 1}次)")
  446. if not step1_open_app(ex):
  447. print("[step2] 重启App失败,直接报错")
  448. return False
  449. try:
  450. if _search_once(ex, keyword):
  451. return True
  452. print(f"[step2] 第{round_idx + 1}/{SEARCH_ROUNDS}轮搜索未成功")
  453. except Exception as e:
  454. # 截图损坏/识别异常/网络抖动等都按本轮失败处理,重启后再试
  455. print(f"[step2] 第{round_idx + 1}轮搜索异常({e}),重启后重试")
  456. print(f"[step2] FAIL - {SEARCH_ROUNDS}轮搜索(含重启)均未成功")
  457. return False
  458. def normalize_match_text(value):
  459. """归一化:统一全角/半角、去除空白和零宽字符(美团同款,避免'看起来一样但匹配失败')"""
  460. import unicodedata
  461. text = "" if value is None else str(value)
  462. text = unicodedata.normalize("NFKC", text)
  463. text = re.sub(r"[\s ​-‍]+", "", text)
  464. return text
  465. def _edit_distance(a: str, b: str) -> int:
  466. """编辑距离(Levenshtein),用于OCR错字容错"""
  467. if len(a) < len(b):
  468. a, b = b, a
  469. if not b:
  470. return len(a)
  471. prev = list(range(len(b) + 1))
  472. for i, ca in enumerate(a, 1):
  473. cur = [i]
  474. for j, cb in enumerate(b, 1):
  475. cur.append(min(prev[j] + 1, cur[j - 1] + 1, prev[j - 1] + (ca != cb)))
  476. prev = cur
  477. return prev[-1]
  478. def _match_fuzzy(target: str, text: str) -> bool:
  479. """
  480. 容错匹配:目标词是否在文本中(处理OCR漏字/错字1个)。
  481. 用于判断「疑似生僻字」——本地识别差1个字时触发云端验证。
  482. """
  483. if not target:
  484. return True
  485. if target in text:
  486. return True
  487. if len(target) >= 3:
  488. for i in range(len(target)): # 漏字:目标删任意1字后匹配
  489. if target[:i] + target[i + 1:] in text:
  490. return True
  491. for L in (len(target), len(target) + 1, len(target) - 1): # 错字:编辑距离<=1
  492. if L < 2:
  493. continue
  494. for i in range(len(text) - L + 1):
  495. if _edit_distance(target, text[i:i + L]) <= 1:
  496. return True
  497. return False
  498. def _correct_title_with_cloud(title: str, n_brand: str, n_key: str, cloud_texts: list) -> str:
  499. """
  500. 用云端文本修正标题中的生僻字错误(精确子串替换,不做整标题替换):
  501. 标题里 fuzzy 匹配到目标词的子串 → 替换为云端文本确认的准确形式。
  502. 只处理"漏字"场景(云端形式更长,如"理王"→"理洫王");
  503. 同长度的形近替换(如"温胃舒"→"养胃舒")【不修正】——
  504. 云端文本里的目标词可能来自其他商品,误用来改标题会造成错误采集。
  505. """
  506. c_joined = "".join(cloud_texts)
  507. nt = normalize_match_text(title)
  508. for target in (n_brand, n_key):
  509. if not target:
  510. continue
  511. if target in nt:
  512. continue # 已精确匹配,无需修正
  513. if target not in c_joined:
  514. continue # 云端也没有准确形式 → 无法修正,跳过
  515. # 在标题里找漏字子串并替换为准确形式。
  516. # 只处理"目标词比子串长且编辑距离≤1"(漏字:理王→理洫王)——
  517. # 同长度形近替换(温胃舒→养胃舒)不触发:云端文本里的目标词可能来自其他商品,
  518. # 误用来改标题会造成错误采集。
  519. for i in range(len(nt)):
  520. for L in (len(target) - 1, len(target), len(target) + 1):
  521. if L < 2 or i + L > len(nt):
  522. continue
  523. sub = nt[i:i + L]
  524. if len(target) > len(sub) and _edit_distance(target, sub) <= 1:
  525. nt = nt[:i] + target + nt[i + L:]
  526. break
  527. else:
  528. continue
  529. break
  530. return nt
  531. def _is_cjk(ch: str) -> bool:
  532. """是否中文字符"""
  533. return '一' <= ch <= '鿿'
  534. def _is_missing_char(target: str, title_text: str) -> bool:
  535. """
  536. 判断目标词是否在标题里以"漏1字"形式出现(生僻字特征):
  537. 目标词删掉1个字后的子串,在标题里是【独立词】(前后不是汉字)。
  538. 例:理洫王删"洫"="理王",标题"[理王]"里独立 → 漏字 ✓
  539. 例:养胃舒删"舒"="养胃",标题"滋阴养胃"里嵌在词中(前有"滋")→ 非独立 → 不算漏字 ✗
  540. """
  541. for k in range(len(target)):
  542. sub = target[:k] + target[k + 1:]
  543. if len(sub) < 2:
  544. continue
  545. idx = title_text.find(sub)
  546. while idx != -1:
  547. before_ok = idx == 0 or not _is_cjk(title_text[idx - 1])
  548. after_ok = idx + len(sub) >= len(title_text) or not _is_cjk(title_text[idx + len(sub)])
  549. if before_ok and after_ok:
  550. return True
  551. idx = title_text.find(sub, idx + 1)
  552. return False
  553. def _ensure_cloud(cloud_cache: dict, shot_path: str) -> None:
  554. """触发一次云端OCR(每批只调一次),保存 (文本, y) 带坐标的块列表"""
  555. if cloud_cache["done"]:
  556. return
  557. cloud_cache["done"] = True
  558. try:
  559. raw_c = OCR.recognize(shot_path, detail="all", engine="cloud")
  560. if not raw_c:
  561. # 云端返回空(配额用尽/限流时百度返回错误码但不抛异常)→ 视为云端不可用
  562. print("[step3] 云端返回空结果(可能配额不足/限流),按容错处理")
  563. cloud_cache["blocks"] = None
  564. cloud_cache["texts"] = None
  565. return
  566. cloud_cache["blocks"] = [
  567. (normalize_match_text(r["text"]), (r["bbox"][0][1] + r["bbox"][2][1]) // 2)
  568. for r in raw_c
  569. ]
  570. cloud_cache["texts"] = [t for t, _ in cloud_cache["blocks"]]
  571. print(f"[step3] 云端二次确认({len(cloud_cache['blocks'])}块)")
  572. except Exception as e:
  573. print(f"[step3] 云端识别失败({e})")
  574. cloud_cache["blocks"] = None
  575. cloud_cache["texts"] = None
  576. def _cloud_crop_confirm(shot_path: str, card_y: int) -> Optional[str]:
  577. """
  578. 用本地OCR的标题块坐标裁剪标题区域(标题y-20 ~ y+80,右列),放大2倍后云端识别。
  579. 裁剪区只含当前商品的标题 → 云端不需要返回坐标,标准版(无配额问题)即可用,且字放大识别更准。
  580. """
  581. try:
  582. img = cv2.imread(shot_path)
  583. if img is None:
  584. return None
  585. h, w = img.shape[:2]
  586. y1, y2 = max(0, card_y - 20), min(h, card_y + 80)
  587. x1 = int(w * 0.15)
  588. crop = img[y1:y2, x1:w]
  589. if crop.size == 0:
  590. return None
  591. big = cv2.resize(crop, None, fx=2, fy=2, interpolation=cv2.INTER_LANCZOS4)
  592. raw = OCR.recognize(big, detail="all", engine="cloud")
  593. texts = [normalize_match_text(r["text"]) for r in raw
  594. if any('一' <= c <= '鿿' for c in r["text"])]
  595. return "".join(texts) if texts else None
  596. except Exception as e:
  597. print(f" [调试] 裁剪云端确认异常: {e}")
  598. return None
  599. def _match_verify(title: str, n_brand: str, n_key: str, shot_path: str, cloud_cache: dict,
  600. card_y: int = None) -> tuple:
  601. """
  602. 品牌/药品名匹配(只判断品牌+药品名核心词,规格/功效文字不参与)。
  603. 返回 (判定, 不匹配原因):判定 "ok"/"fuzzy"/"fail",原因如 "品牌"/"药品名"/"品牌、药品名"
  604. - 本地精确匹配 → "ok"(免费)
  605. - 完全不像 → "fail"(免费,不花云端)
  606. - 差1字/漏字 → 用本地坐标裁剪该商品标题区域,云端识别确认:
  607. 裁剪区含目标词(本地认错字,如理王→理洫王/甲疏咪唑)→ "ok"
  608. 裁剪区不含目标词(确实不是该商品,如温胃舒vs养胃舒)→ "fail"
  609. - 裁剪确认不可用 → 漏字场景容错放行 "fuzzy",换字场景 "fail"
  610. """
  611. nt = normalize_match_text(title)
  612. brand_match = (not n_brand) or (n_brand in nt)
  613. key_match = (not n_key) or (n_key in nt)
  614. if brand_match and key_match:
  615. return "ok", ""
  616. def _unmatched_reason():
  617. parts = []
  618. if n_brand and not brand_match:
  619. parts.append("品牌")
  620. if n_key and not key_match:
  621. parts.append("药品名")
  622. return "、".join(parts) or "品牌/药品名"
  623. # 便宜判断:完全不像(非差1字也非漏字)→ 直接过滤,不花云端
  624. near_brand = (not n_brand) or _match_fuzzy(n_brand, nt) or _is_missing_char(n_brand, nt)
  625. near_key = (not n_key) or _match_fuzzy(n_key, nt) or _is_missing_char(n_key, nt)
  626. if not (near_brand and near_key):
  627. return "fail", _unmatched_reason()
  628. # 差1字/漏字 → 用本地坐标裁剪该商品标题区域,云端确认(标准版即可,无需坐标)
  629. if card_y is not None:
  630. crop_text = _cloud_crop_confirm(shot_path, card_y)
  631. if crop_text is not None:
  632. if (not n_brand or n_brand in crop_text) and (not n_key or n_key in crop_text):
  633. # 把裁剪确认文本交给标题修正逻辑(如"理王"→"理洫王")
  634. cloud_cache["texts"] = [crop_text]
  635. return "ok", ""
  636. print(f" [调试] 卡片y={card_y} 裁剪云端=[{crop_text[:40]}] 不含目标词")
  637. return "fail", _unmatched_reason()
  638. print(f" [调试] 卡片y={card_y} 裁剪云端识别失败")
  639. # 裁剪确认不可用 → 原逻辑:漏字(独立词)→ 整页云端兜底;换字 → 过滤
  640. missing = []
  641. if n_brand and not brand_match and _is_missing_char(n_brand, nt):
  642. missing.append("品牌")
  643. if n_key and not key_match and _is_missing_char(n_key, nt):
  644. missing.append("药品名")
  645. if not missing:
  646. return "fail", _unmatched_reason() # 换字 → 过滤(安全默认)
  647. # 漏字 → 整页云端兜底(cloud_cache 保证每批只调一次)
  648. _ensure_cloud(cloud_cache, shot_path)
  649. if cloud_cache["texts"] is None:
  650. return "fuzzy", ""
  651. c_joined = "".join(cloud_cache["texts"])
  652. if (not n_brand or n_brand in c_joined) and (not n_key or n_key in c_joined):
  653. return "ok", ""
  654. return "fail", _unmatched_reason()
  655. def _screen_state(texts: list) -> str:
  656. """
  657. 根据 OCR 文本判断屏幕状态(统一的状态识别,各步骤共用):
  658. detail=商品详情页 shop=店铺页 list=搜索结果列表页 unknown=其他
  659. """
  660. joined = "".join(texts)
  661. if any(("加入购物车" in t) or ("立即购买" in t) or ("选规格" in t) or ("加入购物袋" in t) for t in texts):
  662. return "detail"
  663. if ("刚刚搜过" in joined) and ("评价" in joined):
  664. return "shop"
  665. if "筛选" in joined:
  666. return "list"
  667. return "unknown"
  668. def _is_list_page(ex: SafeExecutor) -> bool:
  669. """当前是否在搜索结果列表页(全屏检测「筛选」,店铺页/详情页不含此词)"""
  670. import os as _os
  671. tmp = _shot_path(ex, "check_list.png")
  672. ex.driver.screenshot(tmp)
  673. if cv2.imread(tmp) is None:
  674. ex.driver.screenshot(tmp)
  675. texts = [r["text"] for r in OCR.recognize(tmp, detail="all")]
  676. return any("筛选" in t for t in texts)
  677. def _is_white_screen(shot_path: str) -> bool:
  678. """白屏检测:截图缩至90x200灰度后均值>250判白(详情页没加载出来的纯白页)。
  679. 单指标风格同 _page_moved;状态栏/加载圈等少量深色像素拉低不了均值。"""
  680. img = cv2.imread(shot_path, cv2.IMREAD_GRAYSCALE)
  681. if img is None:
  682. time.sleep(0.5)
  683. img = cv2.imread(shot_path, cv2.IMREAD_GRAYSCALE)
  684. if img is None:
  685. return False
  686. return float(np.mean(cv2.resize(img, (90, 200)))) > 250.0
  687. def _looks_like_app_home(raw: list) -> bool:
  688. """真App首页判定(多特征组合,特征证据@screenshots/*/step1/step1_home.png):
  689. 强锚 = 底部导航「AI点外卖」(全App唯此一处;店铺页/列表页/详情页都没有。
  690. OCR可能把AI读成Al/a1,统一小写后用「点外卖」宽容匹配)
  691. 组合 = 频道宫格名(看病买药/盒马鲜生/超市便利/美食外卖/到店团购/天天红包/赚吃货豆/数码潮玩)
  692. 命中≥3个(宫格可能被滑出屏幕,按在屏内容计;盒马鲜生兼容"盒马生鲜"误读)
  693. 排除 = 首页不可能有「刚刚搜过」(店铺页专属元素)
  694. 背景:店铺页自带「首页/商品/评价」页签,AI仅凭"首页"字样会把店铺页误判成home
  695. (实测@阿里健康自营大药房 2026-09-18 10:00),故AI判home后必须经此复核。"""
  696. if not raw:
  697. return False
  698. joined = "".join(normalize_match_text(r.get("text", "")) for r in raw).lower()
  699. if "刚刚搜过" in joined:
  700. return False
  701. if "点外卖" in joined: # AI点外卖(AI/Al/a1变体一并覆盖)
  702. return True
  703. hits = sum(1 for m in ("看病买药", "盒马鲜生", "盒马生鲜", "超市便利", "美食外卖",
  704. "到店团购", "天天红包", "赚吃货豆", "数码潮玩") if m in joined)
  705. return hits >= 3
  706. def _look_like_shop_page(raw: list) -> bool:
  707. """OCR复核是否店铺页(防AI把店铺页误判成列表页):
  708. ① 「刚刚搜过」+「评价」同时在(店铺页特有元素)
  709. ② 店铺名特征块只有1个且在屏幕上部25%(列表页会有多个店铺名散布全屏)
  710. 店铺名特征 = 含"快递发"后缀或 药房/药店/医药/连锁/自营/旗舰 等商家关键词
  711. (海王星辰等不含"药房"的店名靠"快递发"后缀兜住)"""
  712. texts = [r.get("text", "") for r in raw]
  713. joined = "".join(texts)
  714. if ("刚刚搜过" in joined) and ("评价" in joined):
  715. return True
  716. kw = re.compile(r'快递发|药房|药店|医药|商城|超市|自营|连锁|旗舰')
  717. ys = [(r.get("box") or [0, 0, 0, 0])[1] for r in raw if kw.search(r.get("text", ""))]
  718. if len(ys) == 1:
  719. max_y = max(((r.get("box") or [0, 0, 0, 0])[3] for r in raw), default=0)
  720. return ys[0] < max_y * 0.25
  721. return False
  722. def _find_ad_close(shot_path: str) -> Optional[dict]:
  723. """
  724. 广告弹窗打叉按钮:二值化模板匹配,形状匹配不受颜色/背景干扰。
  725. 双模板双二值化(固定阈值100 + OTSU自适应,各自同阈值组合),不同弹窗打叉深浅不同,取最佳。
  726. 命中返回 {"x","y"}。
  727. """
  728. base_dir = Path(__file__).parent / "files"
  729. screen = cv2.imread(shot_path, cv2.IMREAD_GRAYSCALE)
  730. if screen is None:
  731. return None
  732. h_s, w_s = screen.shape[:2]
  733. base_w = 1220.0 # 模板裁自1220宽屏,其他分辨率按比例缩放
  734. scale_ratio = w_s / base_w
  735. scales = [round(scale_ratio * s, 2) for s in [0.8, 0.9, 1.0, 1.1, 1.2]]
  736. best_val, best_loc, best_sw, best_sh = 0, None, 0, 0
  737. # (模板文件, 截图二值化方式):同阈值组合(固定×固定、OTSU×OTSU)
  738. pairs = (
  739. ("ad_close_bin.png", lambda g: cv2.threshold(g, 100, 255, cv2.THRESH_BINARY_INV)[1]),
  740. ("ad_close_otsu.png", lambda g: cv2.threshold(g, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)[1]),
  741. )
  742. for tpl_name, binarize in pairs:
  743. template = cv2.imread(str(base_dir / tpl_name), cv2.IMREAD_GRAYSCALE)
  744. if template is None:
  745. continue
  746. screen_bin = binarize(screen)
  747. for scale in scales:
  748. scaled = cv2.resize(template, None, fx=scale, fy=scale)
  749. sw, sh = scaled.shape[1], scaled.shape[0]
  750. if sh > h_s or sw > w_s:
  751. continue
  752. res = cv2.matchTemplate(screen_bin, scaled, cv2.TM_CCOEFF_NORMED)
  753. _, mv, _, ml = cv2.minMaxLoc(res)
  754. if mv > best_val:
  755. best_val, best_loc, best_sw, best_sh = mv, ml, sw, sh
  756. # 阈值0.4:二值化形状匹配值高(实测0.65+),误匹配低
  757. if best_val >= 0.4 and best_loc is not None:
  758. sx = best_loc[0] + best_sw // 2
  759. sy = best_loc[1] + best_sh // 2
  760. # 位置约束:打叉在弹窗正下方(下半屏 y>0.4h),上半屏匹配视为误报
  761. if sy < int(h_s * 0.4):
  762. return None
  763. return {"x": sx, "y": sy}
  764. return None
  765. def _click_reload(ex: SafeExecutor) -> bool:
  766. """检测「出错了/正在检修中」页面并点击「重新加载」(验证码通过后可能出现)"""
  767. try:
  768. shot = _shot_path(ex, "reload_check.png")
  769. ex.driver.screenshot(shot)
  770. if cv2.imread(shot) is None:
  771. ex.driver.screenshot(shot)
  772. btn = None
  773. for r in OCR.recognize(shot, detail="all"):
  774. if "重新加载" in r["text"]:
  775. box = r["bbox"]
  776. btn = {"x": (box[0][0] + box[2][0]) // 2, "y": (box[0][1] + box[2][1]) // 2}
  777. break
  778. if not btn:
  779. return False
  780. print(f" 点击「重新加载」: ({btn['x']}, {btn['y']})")
  781. ex.tap(btn["x"], btn["y"])
  782. human_sleep(3)
  783. return True
  784. except Exception as e:
  785. print(f" ⚠ 重新加载处理异常: {e}")
  786. return False
  787. # 弹窗找不到关闭按钮时的重启恢复:仅限 step2 搜索流程内,最多重启3次
  788. _POPUP_RESTART_COUNT = 0
  789. MAX_POPUP_RESTART = 3
  790. def _has_popup_mask(shot_path: str) -> bool:
  791. """
  792. 弹窗遮罩检测:弹窗出现时周围被半透明遮罩变暗(暗区比例大幅上升)。
  793. 实测:有弹窗暗区0.18-0.33,无弹窗<0.1,阈值0.12安全区分。
  794. """
  795. img = cv2.imread(shot_path, cv2.IMREAD_GRAYSCALE)
  796. if img is None:
  797. return False
  798. dark_ratio = float((img < 80).mean())
  799. return dark_ratio > 0.12
  800. # 全屏活动页特征词(红包/签到类弹窗无遮罩变暗,遮罩检测漏检;命中即back关闭)
  801. ACTIVITY_KW = ("红包", "领取", "金币", "翻牌", "开宝箱", "签到", "逛一逛")
  802. def _close_ad_popup(ex: SafeExecutor, force: bool = False) -> bool:
  803. """
  804. 检测并关闭广告弹窗:
  805. ① force=True: 跳过遮罩门槛,OCR找活动页特征词(红包/领取/金币等全屏活动页——
  806. 这类弹窗背景不变暗,遮罩检测会漏检),命中则 back 关闭
  807. ② 遮罩检测(弹窗周围变暗,本地计算零成本,最快最准)
  808. ③ AI 找关闭按钮(云端OCR → AI判断)
  809. ④ 模板匹配打叉兜底
  810. ⑤ back键关闭(activity对话框弹窗back即可关)
  811. ⑥ 都失败 → 重启App恢复,仅限搜索流程内3次
  812. """
  813. try:
  814. shot = _shot_path(ex, "ad_check.png")
  815. ex.driver.screenshot(shot)
  816. if cv2.imread(shot) is None:
  817. ex.driver.screenshot(shot)
  818. # 0. force: 活动页特征词检测(无遮罩门槛)
  819. if force:
  820. texts = [r.get("text", "") for r in OCR.recognize(shot, detail="all")]
  821. # 列表页标记在场(筛选/快递) → 我们就在搜索列表页上, 页面里的"红包"字样
  822. # 是页签/促销文案, 不是弹窗 —— 绝不能back(back会退出列表页)
  823. on_list = any(("筛选" in t or "快递" in t) for t in texts)
  824. hit = [w for w in ACTIVITY_KW if any(w in t for t in texts)]
  825. if hit and not on_list:
  826. print(f" 检测到活动页弹窗(特征词:{hit[0]},无列表标记),back关闭")
  827. ex.driver.press("back")
  828. human_sleep(1.2)
  829. return True
  830. if hit and on_list:
  831. print(f" 页面含活动词({hit[0]})但列表标记在场 → 是列表页不是弹窗,不处理")
  832. # 1. 遮罩检测:无弹窗直接返回(不浪费AI调用)
  833. if not _has_popup_mask(shot):
  834. return False
  835. print(" 检测到弹窗(周围遮罩变暗)")
  836. # 2. AI 找关闭按钮(云端OCR → AI判断坐标)
  837. raw = OCR.recognize(shot, detail="all", engine="cloud")
  838. info = detect_popup(raw)
  839. if info["has_popup"] and info.get("close_xy") and len(info["close_xy"]) == 2:
  840. print(f" 弹窗检测(AI): {info['reason']}")
  841. print(f" 关闭广告弹窗(AI坐标): ({info['close_xy'][0]}, {info['close_xy'][1]})")
  842. ex.tap(int(info["close_xy"][0]), int(info["close_xy"][1]))
  843. human_sleep(1.5)
  844. return True
  845. # 3. 模板匹配打叉兜底(弹窗正下方居中的 ×)
  846. btn = _find_ad_close(shot)
  847. if btn:
  848. print(f" 关闭广告弹窗(模板): ({btn['x']}, {btn['y']})")
  849. ex.tap(btn["x"], btn["y"])
  850. human_sleep(1.5)
  851. return True
  852. # 3.5 back键关闭(红包/活动类弹窗多是activity对话框,back即可关;关不掉无损继续)
  853. ex.driver.press("back")
  854. human_sleep(1.2)
  855. ex.driver.screenshot(shot)
  856. if cv2.imread(shot) is not None and not _has_popup_mask(shot):
  857. print(" 弹窗已通过back键关闭")
  858. return True
  859. # 4. AI/模板/back都关不掉(图片型弹窗)→ 重启App恢复,仅限搜索流程内3次
  860. global _POPUP_RESTART_COUNT
  861. _POPUP_RESTART_COUNT += 1
  862. if _POPUP_RESTART_COUNT > MAX_POPUP_RESTART:
  863. print(f" ⚠ 弹窗重启恢复已达{MAX_POPUP_RESTART}次上限,不再重启")
  864. return True
  865. print(f" ⚠ 弹窗关不掉(红包类图片弹窗),重启App恢复(第{_POPUP_RESTART_COUNT}/{MAX_POPUP_RESTART}次)")
  866. try:
  867. step1_open_app(ex)
  868. except Exception as e:
  869. print(f" ⚠ 重启App异常: {e}")
  870. return True
  871. except Exception as e:
  872. print(f" ⚠ 广告弹窗处理异常: {e}")
  873. return False
  874. def _swipe_side_two_segment(ex: SafeExecutor, w: int, h: int, distance: int, zone: dict):
  875. """两段式侧滑: 拇指锚点±30px起指, 拆两笔独立手势(55~65%+剩余), 笔间自然停顿。
  876. 每笔TouchPipe→motionevent降级; 全部完成后像素验证, 未生效退adb三段式兜底"""
  877. import subprocess as _sp
  878. import random as _random
  879. d1 = int(distance * _random.uniform(0.55, 0.65))
  880. d2 = distance - d1
  881. anchor = (int(w * _random.uniform(*zone["x"])), int(h * _random.uniform(*zone["y"])))
  882. dmax = max(d1, d2)
  883. zs = -1 if zone["drift"][0] > 0 else 1 # 右手凸左"(" / 左手凸右")"
  884. chk = _shot_path(ex, "swipe_check.png")
  885. ex.driver.screenshot(chk)
  886. img = cv2.imread(chk)
  887. before = cv2.resize(img, (90, 200)) if img is not None else None
  888. segs = ((d1, 0.15 + 0.10 * (1 - d1 / dmax)), (d2, 0.15 + 0.10 * (1 - d2 / dmax)))
  889. for i, (d, ratio) in enumerate(segs):
  890. pts = wrist_arc_pts_side(w, h, d, zone, anchor=anchor, bow_ratio=zs * ratio, n=40)
  891. if not touchpipe_drag(ex.driver, pts, step=_random.uniform(0.006, 0.010),
  892. hold=_random.uniform(0.06, 0.12), tail=_random.uniform(0.02, 0.05)):
  893. print(f" [滑动] 两段式第{i+1}笔TouchPipe未生效,降级motionevent")
  894. pts10 = wrist_arc_pts_side(w, h, d, zone, anchor=anchor, bow_ratio=zs * ratio, n=10)
  895. motionevent_drag(ex.device_id, pts10)
  896. if i == 0:
  897. human_sleep(_random.uniform(0.35, 0.6)) # 抬指自然停顿
  898. time.sleep(1.0)
  899. ex.driver.screenshot(chk)
  900. img = cv2.imread(chk)
  901. after = cv2.resize(img, (90, 200)) if img is not None else None
  902. if (before is not None and after is not None
  903. and float(np.mean(cv2.absdiff(before, after))) <= 2.0):
  904. print(" [滑动] 两段式未生效,退回adb三段式兜底")
  905. swipe_x = w // 2
  906. seg_px = distance // 3
  907. for k in range(3):
  908. s = int(h * 0.8) - k * 80
  909. e = max(50, s - seg_px)
  910. _sp.run(["adb", "-s", ex.device_id, "shell", "input", "swipe",
  911. str(swipe_x), str(s), str(swipe_x), str(e), "400"],
  912. capture_output=True, timeout=10)
  913. time.sleep(0.35)
  914. time.sleep(0.6)
  915. def _adb_swipe_up(ex: SafeExecutor, distance: int):
  916. """列表上滑策略分发(每次滑动随机): 80%右手两段式 / 10%左手两段式 / 10%原单笔。
  917. 两段式 = 拇指锚点起指 + 两笔独立"("弧线 + 笔间停顿(用户实机认可);
  918. 单笔 = 旧逻辑保留(三级降级+逐级像素验证)"""
  919. import random as _random
  920. w, h = ex.driver.window_size()
  921. r = _random.random()
  922. if r < SWIPE_RIGHT_TWO_P:
  923. _swipe_side_two_segment(ex, w, h, distance, RIGHT_SIDE_ZONE)
  924. elif r < SWIPE_RIGHT_TWO_P + SWIPE_LEFT_TWO_P:
  925. _swipe_side_two_segment(ex, w, h, distance, LEFT_SIDE_ZONE)
  926. else:
  927. _swipe_single_old(ex, distance)
  928. def _swipe_single_old(ex: SafeExecutor, distance: int):
  929. """原单笔滑动(旧起点带), 三级降级(每级滑动后像素验证, 未生效自动降下一级):
  930. 1) TouchPipe弧线(60点, 100Hz级, 丝滑+弧线; 同验证码滑动通道)
  931. 2) motionevent弧线(10点, ~45ms/点, 略步进)
  932. 3) 三段式adb swipe(兜底)"""
  933. import subprocess as _sp
  934. w, h = ex.driver.window_size()
  935. chk = _shot_path(ex, "swipe_check.png")
  936. ex.driver.screenshot(chk)
  937. img = cv2.imread(chk)
  938. before = cv2.resize(img, (90, 200)) if img is not None else None
  939. def _page_moved():
  940. ex.driver.screenshot(chk)
  941. img = cv2.imread(chk)
  942. after = cv2.resize(img, (90, 200)) if img is not None else None
  943. return (before is not None and after is not None
  944. and float(np.mean(cv2.absdiff(before, after))) > 2.0)
  945. # 1) TouchPipe 弧线(最丝滑)
  946. try:
  947. pts = wrist_arc_pts(w, h, distance, n=60)
  948. if touchpipe_drag(ex.driver, pts):
  949. time.sleep(1.0)
  950. if _page_moved():
  951. return
  952. print(" [滑动] TouchPipe未生效,降级motionevent")
  953. except Exception as e:
  954. print(f" [滑动] TouchPipe异常({e}),降级motionevent")
  955. # 2) motionevent 弧线
  956. try:
  957. pts10 = wrist_arc_pts(w, h, distance, n=10)
  958. motionevent_drag(ex.device_id, pts10)
  959. time.sleep(1.0)
  960. if _page_moved():
  961. return
  962. print(" [滑动] motionevent未生效,退回adb swipe")
  963. except Exception as e:
  964. print(f" [滑动] motionevent异常({e}),退回adb swipe")
  965. # 3) 兜底: 三段式 adb swipe
  966. swipe_x = w // 2
  967. seg_px = distance // 3
  968. for i in range(3):
  969. s = int(h * 0.8) - i * 80
  970. e = max(50, s - seg_px)
  971. _sp.run(["adb", "-s", ex.device_id, "shell", "input", "swipe",
  972. str(swipe_x), str(s), str(swipe_x), str(e), "400"],
  973. capture_output=True, timeout=10)
  974. time.sleep(0.35)
  975. time.sleep(0.6)
  976. def _adb_swipe_left(ex: SafeExecutor, y_ratio: float = 0.3):
  977. """从右往左滑(拟人弧线):切换商品图片轮播; TouchPipe→motionevent→adb。
  978. 带滑动验证: 轮播没切换(顶部区域像素无变化)→换触点高度/加大距离重试,最多2次。
  979. 手势极快(20点/2~4ms步进, 整笔~0.15s), 快甩才带得动轮播。
  980. 返回 True=确认切换过, False=两次都未生效(可能本商品无轮播)"""
  981. import random as _random
  982. import subprocess
  983. w, h = ex.driver.window_size()
  984. chk = _shot_path(ex, "left_check.png")
  985. def _region_moved(before):
  986. """顶部42%区域(商品图区)与基线对比是否变化"""
  987. ex.driver.screenshot(chk)
  988. img = cv2.imread(chk)
  989. if img is None or before is None:
  990. return True # 截图不可用时按生效处理,避免无谓重试
  991. a = cv2.resize(img[0:int(h * 0.42), :], (90, 60))
  992. b = cv2.resize(before[0:int(h * 0.42), :], (90, 60))
  993. return float(np.mean(cv2.absdiff(b, a))) > 3.0
  994. ex.driver.screenshot(chk)
  995. base = cv2.imread(chk)
  996. tried = []
  997. for i in range(2):
  998. yr = y_ratio + (0.0, -0.07)[i] # 轮播图高度因商品而异,换高度重试
  999. dist = int(w * (0.65 + 0.08 * i) * _random.uniform(0.95, 1.05))
  1000. try:
  1001. pts = wrist_arc_pts_h(w, h, dist, y_ratio=yr, n=20, drift_range=(30, 70))
  1002. if touchpipe_drag(ex.driver, pts, step=_random.uniform(0.002, 0.004),
  1003. hold=_random.uniform(0.015, 0.035),
  1004. tail=_random.uniform(0.01, 0.025)):
  1005. tried.append(f"tp@{yr:.2f}")
  1006. else:
  1007. print(" [图片左滑] TouchPipe未生效,降级motionevent")
  1008. pts10 = wrist_arc_pts_h(w, h, dist, y_ratio=yr, n=10, drift_range=(30, 70))
  1009. motionevent_drag(ex.device_id, pts10, step=_random.uniform(0.015, 0.022))
  1010. tried.append(f"me@{yr:.2f}")
  1011. except Exception as e:
  1012. print(f" [图片左滑] 弧线异常({e}),退回adb swipe")
  1013. subprocess.run(
  1014. ["adb", "-s", ex.device_id, "shell", "input", "swipe",
  1015. str(int(w * 0.85)), str(int(h * yr)), str(int(w * 0.15)), str(int(h * yr)), "300"],
  1016. capture_output=True, timeout=10
  1017. )
  1018. tried.append(f"adb@{yr:.2f}")
  1019. human_sleep(1.0)
  1020. if _region_moved(base):
  1021. print(f" [图片左滑] 轮播已切换 ({', '.join(tried)})")
  1022. human_sleep(0.4)
  1023. return True
  1024. print(f" [图片左滑] 2次均未切换,可能本商品无轮播 ({', '.join(tried)})")
  1025. return False
  1026. def _adb_swipe_up_short(ex: SafeExecutor):
  1027. """上滑半屏(拟人弧线):说明书页内容可能需滑动; TouchPipe→motionevent→adb 三级降级"""
  1028. import subprocess
  1029. w, h = ex.driver.window_size()
  1030. try:
  1031. pts = wrist_arc_pts(w, h, int(h * 0.4), n=40, drift_range=(40, 90))
  1032. if touchpipe_drag(ex.driver, pts):
  1033. human_sleep(0.9)
  1034. return
  1035. print(" [说明书上滑] TouchPipe未生效,降级motionevent")
  1036. pts10 = wrist_arc_pts(w, h, int(h * 0.4), n=8, drift_range=(40, 90))
  1037. motionevent_drag(ex.device_id, pts10)
  1038. human_sleep(0.9)
  1039. return
  1040. except Exception as e:
  1041. print(f" [说明书上滑] 弧线异常({e}),退回adb swipe")
  1042. subprocess.run(
  1043. ["adb", "-s", ex.device_id, "shell", "input", "swipe",
  1044. str(w // 2), str(int(h * 0.7)), str(w // 2), str(int(h * 0.3)), "300"],
  1045. capture_output=True, timeout=10
  1046. )
  1047. human_sleep(1)
  1048. # ── 店铺反馈弹窗(滑动被误判长按店铺名)检测与关闭 ──
  1049. FEEDBACK_POPUP_KW = ("店铺问题", "不喜欢该店铺", "与搜索词无关",
  1050. "图片不够真实", "商品品质不好", "价格太贵", "疑似情色")
  1051. def _close_feedback_popup(ex: SafeExecutor) -> bool:
  1052. """检测并关闭"店铺问题/商品问题"反馈弹窗(滑动被误判长按所致)。
  1053. 关闭方式: 点弹窗上方暗区,或右上角×(back 会退回搜索页,不能用)。
  1054. 返回 True=检测到弹窗(无论是否成功关闭), False=没弹窗。"""
  1055. try:
  1056. shot = _shot_path(ex, "ad_feedback_popup.png")
  1057. ex.driver.screenshot(shot)
  1058. blocks = OCR.recognize(shot, detail="all")
  1059. texts = [b.get("text", "") for b in blocks]
  1060. if not any(kw in t for t in texts for kw in FEEDBACK_POPUP_KW):
  1061. return False
  1062. print("[step3] 检测到店铺反馈弹窗(滑动误触长按),关闭中")
  1063. w, h = ex.driver.window_size()
  1064. anchor_y = 0
  1065. for b in blocks:
  1066. if "店铺问题" in b.get("text", ""):
  1067. anchor_y = (b["box"][1] + b["box"][3]) // 2
  1068. break
  1069. # 1) 点弹窗上方暗区(大目标); 2) 点×(与"店铺问题"同行右侧); 3) ×实测兜底位
  1070. taps = [(int(w * 0.5), int(h * 0.25))]
  1071. if anchor_y:
  1072. taps.append((int(w * 0.84), anchor_y))
  1073. taps.append((int(w * 0.84), int(h * 0.52)))
  1074. for tx, ty in taps:
  1075. ex.tap(tx, ty)
  1076. time.sleep(1.2)
  1077. ex.driver.screenshot(shot)
  1078. blocks = OCR.recognize(shot, detail="all")
  1079. texts = [b.get("text", "") for b in blocks]
  1080. if not any(kw in t for t in texts for kw in FEEDBACK_POPUP_KW):
  1081. print("[step3] 反馈弹窗已关闭")
  1082. return True
  1083. print("[step3] 反馈弹窗关闭未生效,后续批次可能空转")
  1084. return True
  1085. except Exception as e:
  1086. print(f"[step3] 反馈弹窗检测异常: {e}")
  1087. return False
  1088. def _swipe_next_batch(ex: SafeExecutor, distance: int):
  1089. """列表上滑统一入口: 滑动后检查长按误触的反馈弹窗。
  1090. 弹窗会骗过滑动的像素diff验证且挡住列表(后续批次全是旧卡片) → 关掉后补滑一次。"""
  1091. _adb_swipe_up(ex, distance)
  1092. if _close_feedback_popup(ex):
  1093. _adb_swipe_up(ex, distance)
  1094. def _find_close_btn(shot_path: str) -> Optional[dict]:
  1095. """
  1096. 说明书页右上角找打叉关闭按钮(多尺度模板匹配,适配多分辨率)。
  1097. 模板 files/close.png,命中返回 {"x","y"},失败返回 None。
  1098. """
  1099. import os as _os
  1100. tpl_path = str(Path(__file__).parent / "files" / "close.png")
  1101. screen = cv2.imread(shot_path)
  1102. template = cv2.imread(tpl_path)
  1103. if screen is None or template is None:
  1104. return None
  1105. h_s, w_s = screen.shape[:2]
  1106. # 右上角区域(打叉永远在右上角)
  1107. roi_x1, roi_y1 = w_s * 2 // 3, 0
  1108. roi = screen[roi_y1:h_s // 4, roi_x1:w_s]
  1109. # 多尺度模板匹配(分辨率适配:以720p为基准按屏宽比例缩放)
  1110. base_w = 720.0
  1111. scale_ratio = w_s / base_w
  1112. scales = [round(scale_ratio * s, 2) for s in [0.7, 0.8, 0.9, 1.0, 1.1, 1.2, 1.3, 1.4, 1.5]]
  1113. best_val, best_loc, best_sw, best_sh = 0, None, 0, 0
  1114. for scale in scales:
  1115. scaled = cv2.resize(template, None, fx=scale, fy=scale)
  1116. sw, sh = scaled.shape[1], scaled.shape[0]
  1117. if sh > roi.shape[0] or sw > roi.shape[1]:
  1118. continue
  1119. res = cv2.matchTemplate(roi, scaled, cv2.TM_CCOEFF_NORMED)
  1120. _, mv, _, ml = cv2.minMaxLoc(res)
  1121. if mv > best_val:
  1122. best_val, best_loc, best_sw, best_sh = mv, ml, sw, sh
  1123. t_edge = cv2.Canny(scaled, 30, 100)
  1124. r_edge = cv2.Canny(roi, 30, 100)
  1125. if t_edge.shape[0] <= r_edge.shape[0] and t_edge.shape[1] <= r_edge.shape[1]:
  1126. res2 = cv2.matchTemplate(r_edge, t_edge, cv2.TM_CCOEFF_NORMED)
  1127. _, mv2, _, ml2 = cv2.minMaxLoc(res2)
  1128. if mv2 > best_val:
  1129. best_val, best_loc, best_sw, best_sh = mv2, ml2, sw, sh
  1130. if best_val >= 0.26 and best_loc is not None:
  1131. sx = roi_x1 + best_loc[0] + best_sw // 2
  1132. sy = best_loc[1] + best_sh // 2
  1133. return {"x": sx, "y": sy}
  1134. return None
  1135. def _collect_instructions(ex: SafeExecutor) -> dict:
  1136. """
  1137. 商品详情页采集说明书(采完链接后调用):
  1138. back关分享弹窗 → 左滑商品图片 → 点「查看详细说明」
  1139. → 说明书页截图OCR → 提取批准文号/有效期(最多上滑3次兜底)
  1140. 返回 {"approval_no": "", "validity": ""},任何失败留空不抛异常
  1141. """
  1142. w, h = ex.driver.window_size()
  1143. pfx = f"{ex.device_id}_inst"
  1144. # in_detail: 是否确认停留在商品详情页——主流程据此决定后续资质采集是否安全执行
  1145. result = {"approval_no": "", "validity": "", "in_detail": False}
  1146. # 1. 确认屏幕状态:分享弹窗→back关闭;在详情页→开始;在列表页→跳过(不再back,避免乱退)
  1147. for _try in range(3):
  1148. shot = _shot_path(ex, "inst_check.png")
  1149. ex.driver.screenshot(shot)
  1150. if cv2.imread(shot) is None:
  1151. ex.driver.screenshot(shot)
  1152. texts = [r["text"] for r in OCR.recognize(shot, detail="all")]
  1153. if any("分享到" in t for t in texts):
  1154. print(" 关闭分享弹窗")
  1155. ex.driver.press("back")
  1156. human_sleep(1.2)
  1157. continue
  1158. state = _screen_state(texts)
  1159. if state == "detail":
  1160. result["in_detail"] = True
  1161. break
  1162. if state == "list":
  1163. print(" ⚠ 说明书采集跳过:已在列表页(step4未成功进店),不再back")
  1164. return result
  1165. if _try < 2:
  1166. print(f" 屏幕状态[{state}],back一次重新确认")
  1167. ex.driver.press("back")
  1168. human_sleep(1.2)
  1169. else:
  1170. print(" ⚠ 说明书采集跳过:多次确认仍不在商品详情页")
  1171. return result
  1172. # 2. 左滑切换商品图片,找「查看详细说明」按钮
  1173. _adb_swipe_left(ex, 0.3)
  1174. shot2 = _shot_path(ex, "inst_btn.png")
  1175. ex.driver.screenshot(shot2)
  1176. btn = _find_text_in_area(shot2, "查看详细说明", h)
  1177. if not btn:
  1178. print(" ⚠ 未找到「查看详细说明」,跳过说明书采集")
  1179. return result
  1180. print(f" 说明书按钮: ({btn['x']}, {btn['y']})")
  1181. ex.tap(btn["x"], btn["y"])
  1182. human_sleep(2.5)
  1183. _ai_check_captcha(ex) # 说明书页可能出现验证码(无关键词检查环节)
  1184. # 3. 说明书页截图 + OCR 提取(模仿美团:批准文号和有效期都找到才停,最多滑3次)
  1185. for attempt in range(4):
  1186. shot3 = _shot_path(ex, "inst_page.png")
  1187. ex.driver.screenshot(shot3)
  1188. inst = parse_instructions(OCR.recognize(shot3, detail="all", engine="cloud"))
  1189. if inst["approval_no"] and inst["validity"]:
  1190. print(f" 说明书: 批准文号={inst['approval_no']} 有效期={inst['validity']}")
  1191. result = inst
  1192. break
  1193. if attempt < 3:
  1194. missing = [k for k in ("approval_no", "validity") if not inst[k]]
  1195. print(f" 说明书字段不全(第{attempt+1}次,缺{missing}),上滑重试")
  1196. _adb_swipe_up_short(ex)
  1197. else:
  1198. print(" ⚠ 说明书页4次均未解析到批准文号/有效期")
  1199. result = inst
  1200. # 4. 说明书采集完成:点右上角打叉关闭说明书页(不能back——back会直接回列表页)
  1201. # 本任务内缓存:第一个商品找到后,后续商品直接复用坐标
  1202. global _CLOSE_BTN_CACHE
  1203. close_btn = _CLOSE_BTN_CACHE
  1204. if close_btn is None:
  1205. close_shot = _shot_path(ex, "inst_close.png")
  1206. ex.driver.screenshot(close_shot)
  1207. close_btn = _find_close_btn(close_shot)
  1208. if close_btn:
  1209. _CLOSE_BTN_CACHE = close_btn
  1210. print(f" 关闭说明书页: ({close_btn['x']}, {close_btn['y']})(本任务已缓存,后续商品复用)")
  1211. else:
  1212. print(" ⚠ 未找到打叉按钮(后续步骤会按屏幕状态自行处理)")
  1213. else:
  1214. print(f" 关闭说明书页: ({close_btn['x']}, {close_btn['y']})(复用本任务缓存坐标)")
  1215. if close_btn:
  1216. ex.tap(close_btn["x"], close_btn["y"])
  1217. human_sleep(1.2)
  1218. return result
  1219. def _collect_snapshot(ex: SafeExecutor, title: str) -> str:
  1220. """
  1221. 网页快照(采集说明书之后调用):
  1222. 先识别屏幕状态——已在详情页直接拍;说明书页等未知页则back一次回详情页再拍;
  1223. 在列表页/店铺页等明确非详情页位置直接跳过(不再back,避免把列表页退到首页)
  1224. """
  1225. try:
  1226. # 1. 先截图识别状态,决定是否需要 back
  1227. shot = _shot_path(ex, "snap_check.png")
  1228. for _try in range(2):
  1229. ex.driver.screenshot(shot)
  1230. if cv2.imread(shot) is None:
  1231. ex.driver.screenshot(shot)
  1232. texts = [r["text"] for r in OCR.recognize(shot, detail="all")]
  1233. state = _screen_state(texts)
  1234. if state == "detail":
  1235. break
  1236. if state in ("list", "shop"):
  1237. print(f" ⚠ 快照跳过:屏幕状态[{state}],不在详情页也不再back")
  1238. return ""
  1239. if _try == 0:
  1240. # 说明书页/其他未知页:back 一次回详情页再确认
  1241. print(f" 快照:屏幕状态[{state}],back回详情页")
  1242. ex.driver.press("back")
  1243. human_sleep(1.2)
  1244. else:
  1245. print(f" ⚠ 快照跳过:back后屏幕状态[{state}],不在详情页")
  1246. return ""
  1247. else:
  1248. print(" ⚠ 快照跳过:无法确认在详情页")
  1249. return ""
  1250. # 2. 滚动截图 + 上传OSS(美团同款逻辑在 snapshot 模块)
  1251. url, snap_reason = collect_snapshot(ex.driver, title, ex.device_id)
  1252. print(f" 📷 快照: {url if url else f'失败({snap_reason})'}")
  1253. if url:
  1254. # 快照滚动改变了页面位置:back 回店铺页,再开始资质采集
  1255. ex.driver.press("back")
  1256. human_sleep(1.2)
  1257. return url
  1258. except Exception as e:
  1259. print(f" ⚠ 快照采集异常: {e}")
  1260. return ""
  1261. def _wait_for_any_text(ex: SafeExecutor, targets: list, max_s: float, prefix: str = "wait_text") -> bool:
  1262. """轮询截图(本地OCR)等待任一目标文字出现——等页面加载完成再进行下一步。
  1263. 返回 True=等到了, False=超时未出现(调用方自行决定是否继续)"""
  1264. import time as _t
  1265. deadline = _t.time() + max_s
  1266. while _t.time() < deadline:
  1267. shot = _shot_path(ex, f"{prefix}.png")
  1268. ex.driver.screenshot(shot)
  1269. if cv2.imread(shot) is not None:
  1270. texts = [r.get("text", "") for r in OCR.recognize(shot, detail="all")]
  1271. if any(tg in t for tg in targets for t in texts):
  1272. return True
  1273. _t.sleep(1.2)
  1274. return False
  1275. def _collect_license(ex: SafeExecutor, shop_name: str) -> dict:
  1276. """
  1277. 采集商家资质(采完说明书后调用,屏幕在说明书页):
  1278. back回店铺内 → OCR上半区找店铺名点击 → 下半区找「查看营业资质」点击
  1279. → 云端OCR找「资质编号」取值 → 点编号下方约3cm打开营业执照 → 百度营业执照专用接口OCR
  1280. 返回 {"license_no": "", "license": {}},任何失败留空不抛异常
  1281. """
  1282. w, h = ex.driver.window_size()
  1283. pfx = f"{ex.device_id}_lic"
  1284. result = {"license_no": "", "license": {}}
  1285. # 0. 已采集过的店铺:直接从数据库获取资质,不重复采集(美团/PDD同款)
  1286. try:
  1287. exist = get_existing_license(shop_name)
  1288. if exist.get("license_no") or exist.get("company"):
  1289. print(f" 资质已存在,从数据库获取: 编号={exist['license_no'][:24]} 公司={exist['company'][:20]}")
  1290. result["license_no"] = exist["license_no"]
  1291. lic = {}
  1292. if exist.get("company"):
  1293. lic["单位名称"] = exist["company"]
  1294. if exist.get("address"):
  1295. lic["地址"] = exist["address"]
  1296. result["license"] = lic
  1297. return result
  1298. except Exception as e:
  1299. print(f" ⚠ 查询已有资质失败: {e}")
  1300. # 1+2. 回到店铺页并采资质(最多2轮):每轮先找店铺名,找不到/点后无资质入口就 back 回退一层再看
  1301. # 注意不能无条件back:屏幕已在店铺页时再退会到列表页
  1302. def _find_license_btn():
  1303. # 先搜下半区,找不到整张图分析(「查看营业资质」位置不固定,有时在上半区)
  1304. for rect in ([0, int(h * 0.45), w, h], None):
  1305. for r in OCR.recognize(shot2, rect=rect, detail="all"):
  1306. if "查看营业资质" in r["text"]:
  1307. box = r["bbox"]
  1308. return {"x": (box[0][0] + box[2][0]) // 2, "y": (box[0][1] + box[2][1]) // 2}
  1309. return None
  1310. lic_btn = None
  1311. for attempt in range(2):
  1312. if _is_list_page(ex):
  1313. print(" ⚠ 资质采集跳过:已在列表页(不点列表卡片)")
  1314. return result
  1315. shot = _shot_path(ex, "lic_shopname.png")
  1316. ex.driver.screenshot(shot)
  1317. shop_btn = _find_text_in_area(shot, shop_name, h // 2)
  1318. if shop_btn:
  1319. print(f" 店铺名: ({shop_btn['x']}, {shop_btn['y']}) (第{attempt+1}轮)")
  1320. ex.tap(shop_btn["x"], shop_btn["y"])
  1321. human_sleep(1.2)
  1322. # 等店铺信息页加载出「查看营业资质」(最多6s,防止没等加载就找)
  1323. _wait_for_any_text(ex, ["查看营业资质"], 6, "lic_wait")
  1324. # 下半区找「查看营业资质」,找不到先上滑一次再看
  1325. shot2 = _shot_path(ex, "inst_btn.png")
  1326. ex.driver.screenshot(shot2)
  1327. lic_btn = _find_license_btn()
  1328. if not lic_btn:
  1329. print(f" 未找到「查看营业资质」(第{attempt+1}轮),上滑再看")
  1330. _adb_swipe_up_short(ex)
  1331. shot2 = _shot_path(ex, "inst_btn.png")
  1332. ex.driver.screenshot(shot2)
  1333. lic_btn = _find_license_btn()
  1334. if lic_btn:
  1335. break
  1336. if attempt == 0:
  1337. # 屏幕可能在详情页/说明书页:back 回退一层再看
  1338. print(" back一次回退后再试")
  1339. ex.driver.press("back")
  1340. human_sleep(1.5)
  1341. else:
  1342. print(" ⚠ 资质采集跳过:未找到店铺名或「查看营业资质」")
  1343. return result
  1344. print(f" 查看营业资质: ({lic_btn['x']}, {lic_btn['y']})")
  1345. ex.tap(lic_btn["x"], lic_btn["y"])
  1346. human_sleep(1.5)
  1347. _ai_check_captcha(ex) # 资质页可能出现验证码(无关键词检查环节)
  1348. # 4. 等资质页加载出「资质编号」(最多8s;刚点完就截图经常是加载中的页面)
  1349. _wait_for_any_text(ex, ["资质编号", "营业执照"], 8, "lic_no_wait")
  1350. shot3 = _shot_path(ex, "lic_no.png")
  1351. ex.driver.screenshot(shot3)
  1352. raw3 = OCR.recognize(shot3, detail="all", engine="cloud")
  1353. license_no = extract_value_after(raw3, "资质编号")
  1354. if license_no:
  1355. # OCR可能把长编号读散(如 "91450800MA5 KEHAHXT"),编号不该有空格,清洗掉
  1356. license_no = license_no.replace(" ", "")
  1357. if not license_no:
  1358. # 页面可能仍在加载/云端OCR波动:等2秒重拍再试一次
  1359. time.sleep(2)
  1360. ex.driver.screenshot(shot3)
  1361. raw3 = OCR.recognize(shot3, detail="all", engine="cloud")
  1362. license_no = extract_value_after(raw3, "资质编号")
  1363. if license_no:
  1364. license_no = license_no.replace(" ", "")
  1365. if not license_no:
  1366. print(" ⚠ 资质采集跳过:未找到资质编号")
  1367. return result
  1368. print(f" 资质编号: {license_no}")
  1369. result["license_no"] = license_no
  1370. # 5. 点资质编号下方约3cm(≈0.2屏高)打开营业执照大图
  1371. # 注意:云端OCR(标准版)无坐标,必须用本地OCR找资质编号的真实位置
  1372. no_x, no_y = None, None
  1373. for r in OCR.recognize(shot3, detail="all"): # 本地引擎,有真实bbox
  1374. t = r["text"].strip().rstrip(":: \t")
  1375. if t.startswith("资质编号"):
  1376. box = r["bbox"]
  1377. no_x = (box[0][0] + box[2][0]) // 2
  1378. no_y = (box[0][1] + box[2][1]) // 2
  1379. break
  1380. if no_y is None:
  1381. print(" ⚠ 资质采集跳过:本地OCR未找到资质编号位置")
  1382. return result
  1383. print(f" 资质编号位置: ({no_x}, {no_y}),点击下方打开执照")
  1384. ex.tap(no_x, min(no_y + int(h * 0.2), h - 50))
  1385. human_sleep(2.5)
  1386. # 6. 营业执照截图 + 百度营业执照专用接口(美团同款;执照在屏幕中下部,先裁剪再识别,逐级兜底)
  1387. shot4 = _shot_path(ex, "lic_license.png")
  1388. ex.driver.screenshot(shot4)
  1389. lic = {}
  1390. for t_ratio, b_ratio in [(0.35, 0.8), (0.4, 1.0), (0.0, 1.0)]:
  1391. lic = OCR.recognize_license(shot4, rect=[0, int(h * t_ratio), w, int(h * b_ratio)])
  1392. if lic:
  1393. break
  1394. if lic:
  1395. print(f" 营业执照: {lic}")
  1396. result["license"] = lic
  1397. else:
  1398. print(" ⚠ 营业执照OCR为空")
  1399. return result
  1400. def _detect_right_col_split(raw: list, w: int) -> Optional[int]:
  1401. """
  1402. 定位列表页"图片|文字"分界x:每张卡片都有价格,¥全在右列且x一致。
  1403. 取¥块左边缘x的中位数 − 15 作为分界(图片区在左、标题/价格在右)。
  1404. 实测3台设备:分界499时图片文字最右仅474,过滤干净。返回None=检测不到(用整图)。
  1405. """
  1406. price_x = sorted(r["box"][0] for r in raw if "¥" in r["text"] or "¥" in r["text"])
  1407. if len(price_x) < 3:
  1408. return None
  1409. return max(price_x[len(price_x) // 2] - 15, int(w * 0.3))
  1410. def _get_named_shops(ex: SafeExecutor, shot_name: str, keyword: str = "") -> tuple:
  1411. """截图 + OCR + AI → 返回 (有店铺名的列表, 本地OCR原始结果带坐标, 页面状态)
  1412. 页面状态: "ok"=正常(含空批次) | "page_wrong"=AI判定不在药品列表页
  1413. | "no_more"=检出「搜索结果较少」底部标志(结果耗尽,本屏真卡仍会返回)"""
  1414. shot = _screenshot(ex, shot_name)
  1415. raw_full = OCR.recognize(shot, detail="all")
  1416. raw = raw_full
  1417. w, h = ex.driver.window_size()
  1418. # 底部耗尽标志(必须用裁剪前的全屏OCR检测,防右列过滤把横跨全屏的标志行滤掉):
  1419. # 「搜索结果较少,为你推荐相关店铺」= 本药品搜索结果已耗尽
  1420. no_more = any(("搜索结果较少" in r.get("text", "")) or ("为你推荐相关店铺" in r.get("text", ""))
  1421. for r in raw_full)
  1422. # 只保留右列(商品描述列):用¥定位分界,过滤左侧图片文字(包装字/英文/乱码)
  1423. split_x = _detect_right_col_split(raw, w)
  1424. if split_x:
  1425. filtered = [r for r in raw if (r["box"][0] + r["box"][2]) // 2 >= split_x]
  1426. if filtered:
  1427. print(f"[step3] 右列识别: 分界x={split_x},过滤掉{len(raw) - len(filtered)}块图片文字")
  1428. raw = filtered
  1429. # 在截图副本上画红线保存(出错时核对分界是否偏左/偏右)
  1430. try:
  1431. img = cv2.imread(shot)
  1432. if img is not None:
  1433. cv2.line(img, (split_x, 0), (split_x, img.shape[0]), (0, 0, 255), 3)
  1434. cv2.putText(img, f"split={split_x}", (split_x + 5, 50),
  1435. cv2.FONT_HERSHEY_SIMPLEX, 0.9, (0, 0, 255), 2)
  1436. cv2.imwrite(shot.replace(".png", "_split.png"), img)
  1437. except Exception:
  1438. pass
  1439. # 视觉方案(唯一路径):PP-OCR识别 → GLM分卡 → 几何校验;GLM失败时内部有文本AI兜底
  1440. shops, vstatus = VisionParser().parse_shops(shot, screen_size=(w, h), keyword=keyword, crop_x=split_x or 0)
  1441. if vstatus == "page_wrong":
  1442. if no_more:
  1443. # 底部耗尽标志在场说明仍在列表页(下方只是店铺推荐流),忽略page_wrong误判
  1444. print("[step3] 底部「搜索结果较少」标志在场,忽略page_wrong")
  1445. else:
  1446. # GLM判定当前页面不是药品列表(back退过头/走错频道)→ 不提取,交给上层走恢复流程
  1447. return [], raw, "page_wrong"
  1448. if vstatus == "failed":
  1449. if no_more:
  1450. # 耗尽标志在场但GLM失败: 不走文本AI兜底(防把底部推荐店铺采进库),按耗尽收尾
  1451. print("[step3] 底部耗尽标志在场且GLM失败,跳过文本AI兜底")
  1452. return [], raw, "no_more"
  1453. # GLM真失败才fallback到文本AI提取
  1454. print("[step3] GLM失败,fallback 到 OCR+文本AI")
  1455. shops = AIParser().parse_shops(raw, screen_size=(w, h), keyword=keyword)
  1456. if not shops:
  1457. # fallback也没提到 → 让文本AI判断当前页面再决定
  1458. ptype = AIParser().check_page(raw).get("type", "unknown")
  1459. if ptype == "list":
  1460. print("[step3] 文本AI判定在列表页(本轮无有效卡片,走空批次滑动)")
  1461. return [], raw, ("no_more" if no_more else "ok")
  1462. print(f"[step3] 文本AI判定页面类型: {ptype}(非列表页)")
  1463. return [], raw, "page_wrong"
  1464. # 只保留有效店铺名+价格:店铺名必须含中文或字母(排除纯数字/标点/空格)
  1465. import re as _re
  1466. valid = []
  1467. for s in shops:
  1468. name = (s[0] or "").strip()
  1469. price = (s[2] or "").strip()
  1470. if name and _re.search(r'[一-鿿＀-￯a-zA-Z]', name) and price:
  1471. valid.append(s)
  1472. return valid, raw, ("no_more" if no_more else "ok")
  1473. def _find_title_y(raw: list, title: str) -> Optional[int]:
  1474. """在本地OCR结果里找与标题开头重合最多的块的y坐标(标题行位置,用于裁剪)"""
  1475. nt = normalize_match_text(title)
  1476. best_len, best_cy = 0, None
  1477. for r in raw:
  1478. t = normalize_match_text(r["text"])
  1479. if not t:
  1480. continue
  1481. n = 0
  1482. for a, b in zip(t, nt):
  1483. if a == b:
  1484. n += 1
  1485. else:
  1486. break
  1487. if n > best_len:
  1488. best_len = n
  1489. best_cy = (r["bbox"][0][1] + r["bbox"][2][1]) // 2
  1490. return best_cy if best_len >= 2 else None
  1491. def _shop_key(shop: list) -> str:
  1492. """用店铺名+价格去重(去括号内分店名、去尾部点号)"""
  1493. import re
  1494. name = shop[0]
  1495. price = shop[2] if len(shop) > 2 else ""
  1496. name = name.replace("(", "(").replace(")", ")")
  1497. name = re.sub(r'(.*', '', name)
  1498. name = re.sub(r'[..…]+$', '', name)
  1499. return f"{name.strip()}|{price.strip()}"
  1500. def _visit_shop(ex: SafeExecutor, shop: list, visited: set, keyword: str = "", task: dict = None,
  1501. batch_no: int = None) -> dict:
  1502. """
  1503. 点击进入店铺 → step4 → 说明书 → 快照 → 资质 → 返回完整数据 dict
  1504. task: 调度任务 dict(task_id/enterprise_id/collect_round等),手动模式传 None
  1505. batch_no: 当前列表页批次号(Excel来源截图追溯用)
  1506. """
  1507. """点击进入店铺 → step4 → 返回完整数据 dict"""
  1508. key = _shop_key(shop)
  1509. if key in visited:
  1510. return None
  1511. visited.add(key)
  1512. shop_name = shop[0]
  1513. product_title = shop[1]
  1514. price = shop[2]
  1515. sales = str(shop[5]) if len(shop) > 5 and shop[5] else ""
  1516. click_x, click_y = shop[3]
  1517. print(f" → 进入 [{shop_name}] 商品: {product_title[:30]} 价格: {price} 月售: {sales}")
  1518. ex.tap(click_x, click_y)
  1519. try:
  1520. qr_url = step4_parse_qr(ex, product_title, shop_name, shop_xy=[click_x, click_y])
  1521. except Exception as e:
  1522. print(f" ⚠ step4异常: {e},跳过此店铺")
  1523. qr_url = ""
  1524. if qr_url == "__TERMINATE__":
  1525. print(f" ⚠ 遇到终止信号,停止遍历")
  1526. return {"__terminate__": True}
  1527. if qr_url == "__RESTART__":
  1528. print(f" ⚠ 页面异常,触发重启恢复")
  1529. return {"__restart__": True}
  1530. if qr_url == "__SKIP__":
  1531. # 连续5次进入页面均未加载(网络/加载异常):退回列表页,跳过本商品
  1532. # step3会对"连续2个商品都跳过"升级为白屏分级恢复(冷却→重启→超限重派)
  1533. print(f" ⛔ 页面多次未加载{'(白屏)' if _LAST_SKIP_WHITE else ''},跳过本商品: {shop_name} | {product_title[:30]}")
  1534. for _ in range(5):
  1535. try:
  1536. pos = _where_am_i(ex)
  1537. except Exception:
  1538. break
  1539. if pos in ("list", "home"):
  1540. break
  1541. ex.driver.press("back")
  1542. human_sleep(1.4)
  1543. return {"__skip__": True}
  1544. if qr_url:
  1545. print(f" ✅ QR: {qr_url[:80]}")
  1546. print(f" 📦 采集完成: {shop_name} | {product_title[:30]} | {price} | 月售{sales} | {qr_url[:60]}")
  1547. else:
  1548. # 未获取到链接(放弃本店/解析失败):数据不入库,退回列表页后直接跳过本店
  1549. print(f" ⛔ 未获取到链接,本店数据不入库: {shop_name} | {product_title[:30]}")
  1550. for _ in range(5):
  1551. try:
  1552. pos = _where_am_i(ex)
  1553. except Exception:
  1554. break
  1555. if pos in ("list", "home"):
  1556. break
  1557. ex.driver.press("back")
  1558. human_sleep(1.4)
  1559. return {}
  1560. # 内存熔断:本地OCR已OOM → 说明书/快照/资质全部跳过(状态检测不可靠,避免乱back乱点)
  1561. if getattr(OCR, "oom", False):
  1562. print(" ⚠ 内存不足熔断:跳过说明书/快照/资质,直接返回列表页(请关闭部分程序释放内存)")
  1563. inst = {"approval_no": "", "validity": "", "in_detail": False}
  1564. snapshot_url = ""
  1565. lic = {"license_no": "", "license": {}}
  1566. else:
  1567. # 采完链接后采集说明书(批准文号/有效期),失败不阻塞
  1568. try:
  1569. inst = _collect_instructions(ex)
  1570. except Exception as e:
  1571. print(f" ⚠ 说明书采集异常: {e}(疑似内存不足,请关闭部分程序)")
  1572. inst = {"approval_no": "", "validity": "", "in_detail": False}
  1573. print(f" 📄 说明书: 批准文号={inst['approval_no']} 有效期={inst['validity']}")
  1574. # 采完说明书后采集网页快照(美团同款顺序:说明书→快照→资质),失败不阻塞
  1575. snapshot_url = _collect_snapshot(ex, product_title)
  1576. # 采完快照后采集商家资质(_collect_license 会先 back 回店铺页,再按屏幕状态自校验:列表页/无店铺名都跳过)
  1577. try:
  1578. lic = _collect_license(ex, shop_name)
  1579. except Exception as e:
  1580. print(f" ⚠ 资质采集异常: {e}")
  1581. lic = {"license_no": "", "license": {}}
  1582. print(f" 📋 资质编号: {lic['license_no']} 单位名称: {lic['license'].get('单位名称', '')} 信用代码: {lic['license'].get('社会信用代码', '')}")
  1583. # 返回搜索页:先检测再退(已是列表页则一步不退;退到首页立即停,防止退过头退出app)
  1584. for _ in range(5):
  1585. try:
  1586. pos = _where_am_i(ex)
  1587. except Exception as e:
  1588. print(f" ⚠ 位置检测异常({e}),停止返回")
  1589. break
  1590. if pos == "list":
  1591. break
  1592. if pos == "home":
  1593. print(" ⚠ 已退到首页(可能退过头),停止返回")
  1594. break
  1595. ex.driver.press("back")
  1596. human_sleep(1.4)
  1597. # Excel 备份:一个药品一个sheet,两个店铺名对比(列表页OCR vs 详情页AI提取)
  1598. # 来源截图 = 设备/批次,供 audit_ocr.py 事后核对OCR识别准确率
  1599. try:
  1600. import excel_log
  1601. source = f"{ex.device_id}/step3_b{batch_no}" if batch_no is not None else ""
  1602. excel_log.append_row(product_title, price, shop_name, AI_SHOP_NAME,
  1603. snapshot_url, qr_url or "", sheet_name=keyword, source=source)
  1604. except Exception as _e:
  1605. print(f" ⚠ Excel备份失败: {_e}")
  1606. task = task or {}
  1607. return {
  1608. "shop": shop_name,
  1609. "title": product_title,
  1610. "price": price,
  1611. "sales": sales,
  1612. "approval_no": inst["approval_no"],
  1613. "validity": inst["validity"],
  1614. "license_no": lic["license_no"],
  1615. "license": lic["license"],
  1616. "snapshot_url": snapshot_url,
  1617. "search_name": keyword,
  1618. "link": qr_url or "",
  1619. "task_id": task.get("task_id"),
  1620. "enterprise_id": task.get("enterprise_id"),
  1621. "collect_round": task.get("collect_round"),
  1622. "collect_equipment_account_id": task.get("collect_equipment_account_id"),
  1623. "collect_region_id": task.get("collect_region_id"),
  1624. "collect_config_info": task.get("collect_config_info", ""),
  1625. }
  1626. def _captcha_log_path(device_id: str) -> Path:
  1627. """每台设备独立的验证码日志文件(独立计数,互不影响)"""
  1628. return CAPTCHA_LOG_DIR / f"captcha_log_{device_id}.txt"
  1629. def _log_captcha(ex: SafeExecutor) -> bool:
  1630. """
  1631. 记录验证码出现时间到日志,检查一天≥8次停止。
  1632. 返回 True=已触发停止(上层不再休息),False=正常(处理完成后按频率休息)。
  1633. """
  1634. global CAPTCHA_ABORTED, CAPTCHA_ABORT_REASON
  1635. import os as _os
  1636. ts = time.strftime("%Y-%m-%d %H:%M:%S")
  1637. line = f"{ts} 验证码出现"
  1638. try:
  1639. _os.makedirs(_os.path.dirname(_captcha_log_path(ex.device_id)), exist_ok=True)
  1640. with open(_captcha_log_path(ex.device_id), "a", encoding="utf-8") as f:
  1641. f.write(line + "\n")
  1642. except Exception as e:
  1643. print(f" [验证码记录] 写日志失败: {e}")
  1644. print(f" [验证码记录] {line}")
  1645. # 一天内 ≥8次 → 立即停止采集回告(风控可能封号)
  1646. today_count = _captcha_count_today(ex.device_id)
  1647. print(f" [验证码记录] 设备{ex.device_id}今天已出现{today_count}次")
  1648. if today_count >= CAPTCHA_DAILY_LIMIT:
  1649. CAPTCHA_ABORTED = True
  1650. CAPTCHA_ABORT_REASON = f"一天内验证码达{today_count}次,进入风控可能封号"
  1651. err_log.log_error(ex.device_id, "captcha_daily_limit", message=CAPTCHA_ABORT_REASON,
  1652. extra={"today_count": today_count, "task_id": _TASK_ID})
  1653. print(f" ⚠ {CAPTCHA_ABORT_REASON},停止采集")
  1654. return True
  1655. return False
  1656. def _captcha_count_today(device_id: str) -> int:
  1657. """统计该设备日志中今天(按日期)的验证码次数"""
  1658. try:
  1659. today = time.strftime("%Y-%m-%d")
  1660. count = 0
  1661. with open(_captcha_log_path(device_id), encoding="utf-8") as f:
  1662. for line in f:
  1663. if line.startswith(today):
  1664. count += 1
  1665. return count
  1666. except Exception:
  1667. return 0
  1668. def _save_captcha_check(ex: SafeExecutor) -> None:
  1669. """每次判定出现验证码时截图保存到 logs/captcha_check/{日期}/,人工确认是否真验证码"""
  1670. import os as _os
  1671. try:
  1672. day = time.strftime("%Y-%m-%d")
  1673. d = CAPTCHA_CHECK_DIR / day
  1674. d.mkdir(parents=True, exist_ok=True)
  1675. ts = time.strftime("%H%M%S")
  1676. path = str(d / f"{ts}_{ex.device_id}_captcha.png")
  1677. ex.driver.screenshot(path)
  1678. print(f" [验证码截图] 已保存: {path}")
  1679. except Exception as e:
  1680. print(f" [验证码截图] 保存失败: {e}")
  1681. def _handle_captcha(ex: SafeExecutor, ocr_texts: list) -> bool:
  1682. """处理验证码, 重试5次, 失败等人工, 返回True=已解决;
  1683. 只有自动解决成功才计数(失败走人工的不算),成功后按频率休息"""
  1684. _save_captcha_check(ex) # 判定出现验证码时截图存档(人工确认是否误判)
  1685. import sys as _sys
  1686. _sys.path.insert(0, r"D:\drug\sg\yzm")
  1687. solved = False
  1688. for attempt in range(1, 6):
  1689. print(f" [验证码] 第{attempt}次尝试...")
  1690. nine_kw = any("提交" in t or "没有新图片" in t for t in ocr_texts)
  1691. if nine_kw:
  1692. from yzm.nine_grid import solve as solve_nine
  1693. ok = solve_nine(ex.driver)
  1694. else:
  1695. try:
  1696. from yzm.tmp_captcha_test7 import solve_slider # 2026-09-23 全面真人化版(轨迹/速度/停顿/微调)
  1697. except ImportError:
  1698. from yzm.tmp_captcha_test6 import solve_slider # 旧版兜底(生产机未同步test7时)
  1699. ok = solve_slider(ex.driver)
  1700. if ok:
  1701. print(f" ✅ 验证码已解决")
  1702. solved = True
  1703. break
  1704. print(f" ❌ 第{attempt}次失败")
  1705. human_sleep(1)
  1706. if solved:
  1707. # 只有解决成功才计数 + 一天≥8次停止检查 + 每2次休息30分钟
  1708. aborted = _log_captcha(ex)
  1709. if not aborted:
  1710. _captcha_rest(ex)
  1711. else:
  1712. print(f" ⚠ 5次自动处理失败, 请人工处理...(不计数)")
  1713. err_log.log_error(ex.device_id, "captcha_solve_failed",
  1714. message="验证码连续5次自动处理失败,转人工(input阻塞前留档)")
  1715. input(" 处理完成后按回车继续...")
  1716. return True
  1717. def _chunked_sleep_with_report(total_seconds: float, label: str) -> bool:
  1718. """分段睡眠并每10分钟回告一次进度(防后台判假死)。
  1719. 返回 False = 休息期间收到调度限额/错误(应停止采集)"""
  1720. rest_left = total_seconds
  1721. while rest_left > 0:
  1722. if _SCHEDULER is not None and getattr(_SCHEDULER, "limit_reached", False):
  1723. print(f" ⏸ {label}休息中收到调度限额/错误,提前结束休息并停止采集")
  1724. return False
  1725. chunk = min(rest_left, 600) # 每10分钟一段
  1726. time.sleep(chunk)
  1727. rest_left -= chunk
  1728. if rest_left > 0 and _SCHEDULER is not None:
  1729. try:
  1730. _SCHEDULER.post_report({
  1731. "task_id": _TASK_ID,
  1732. "platform": _SCHEDULER.platform,
  1733. "username": _SCHEDULER.username,
  1734. "is_finished": 0,
  1735. "need_reassign": 0,
  1736. "current_page": CURRENT_PAGE,
  1737. "crawled_count": _CRAWLED_COUNT,
  1738. })
  1739. print(f" [休息中回告] 当前页{CURRENT_PAGE},已采{_CRAWLED_COUNT}条")
  1740. except Exception as e:
  1741. print(f" [休息中回告] 失败: {e}")
  1742. return True
  1743. def _captcha_rest(ex: SafeExecutor) -> None:
  1744. """每次验证码解决后:休息 CAPTCHA_REST_MINUTES~MAX 分钟(随机)"""
  1745. import random as _random
  1746. today_count = _captcha_count_today(ex.device_id)
  1747. if today_count >= CAPTCHA_DAILY_LIMIT:
  1748. return
  1749. rest_minutes = _random.randint(CAPTCHA_REST_MINUTES, CAPTCHA_REST_MAX_MINUTES)
  1750. print(f" ⏸ 第{today_count}次验证码,休息{rest_minutes}分钟({CAPTCHA_REST_MINUTES}~{CAPTCHA_REST_MAX_MINUTES}随机)...")
  1751. _chunked_sleep_with_report(rest_minutes * 60, "验证码")
  1752. # 验证码强特征词(店铺页/列表页文字多但无这些词——AI判risk时用OCR文字二次确认防误判)
  1753. CAPTCHA_KW = ("拖动滑块", "请按住滑块", "请按照说明", "点我反馈", "进行验证",
  1754. "滑块验证", "拼图", "安全验证", "图形验证", "点击完成验证", "没有新图片", "操作频繁")
  1755. def _has_captcha_kw(texts: list) -> bool:
  1756. """OCR文字是否含验证码强特征词"""
  1757. joined = "".join(texts)
  1758. return any(k in joined for k in CAPTCHA_KW)
  1759. def _ai_check_captcha(ex: SafeExecutor) -> bool:
  1760. """
  1761. AI检测当前屏幕是否出现验证码(用于没有关键词检查的环节):
  1762. 截图 → 本地OCR → 强特征词预筛 → AI判断页面类型(risk=验证码)→ 有则自动处理。
  1763. 返回 True=检测到验证码(已处理或处理中),False=无验证码。
  1764. """
  1765. try:
  1766. shot = _shot_path(ex, "captcha_ai.png")
  1767. ex.driver.screenshot(shot)
  1768. if cv2.imread(shot) is None:
  1769. ex.driver.screenshot(shot)
  1770. raw = OCR.recognize(shot, detail="all") # 本地OCR即可(验证码文字是大字,不用百度)
  1771. # 预筛:OCR文本里连强特征词都没有 → 不是验证码,不调AI(省调用+防误判)
  1772. if not _has_captcha_kw([r["text"] for r in raw]):
  1773. return False
  1774. page = AIParser().check_page(raw)
  1775. if page.get("type") == "risk":
  1776. print(f" ⚠ AI检测到验证码页面: {page.get('detail', '')}")
  1777. _handle_captcha(ex, [r["text"] for r in raw])
  1778. human_sleep(2)
  1779. return True
  1780. return False
  1781. except Exception as e:
  1782. print(f" ⚠ AI验证码检测异常: {e}")
  1783. return False
  1784. def _check_account_kicked(ex: SafeExecutor) -> bool:
  1785. """
  1786. 检测账号是否被踢/封号:登录页元素(me.ele:id/login_onkey_login_ll)出现即判定。
  1787. 检测到 → 置 ACCOUNT_ABORTED 标志(停止采集 + 调度回告)。
  1788. """
  1789. global ACCOUNT_ABORTED
  1790. try:
  1791. if ex.driver.xpath('//*[@resource-id="me.ele:id/login_onkey_login_ll"]').exists:
  1792. print(" ⚠ 检测到账号被踢/封号(登录页),停止采集")
  1793. err_log.log_error(ex.device_id, "account_kicked", message="检测到登录页元素,账号被踢/封号")
  1794. ACCOUNT_ABORTED = True
  1795. return True
  1796. except Exception as e:
  1797. print(f" ⚠ 封号检测异常: {e}")
  1798. return False
  1799. def step4_parse_qr(ex: SafeExecutor, product_title: str, shop_name: str = "",
  1800. shop_xy: Optional[list] = None) -> str:
  1801. """
  1802. 1. 等待加载 → OCR → AI找商品标题坐标
  1803. 2. 点击商品标题 → 进入商品详情
  1804. 3. 找右上角"分享" → 点击 → 二维码弹窗
  1805. 4. 截图 → pyzbar 解析二维码
  1806. 返回 URL 或空字符串
  1807. """
  1808. global AI_SHOP_NAME, _LAST_SKIP_WHITE
  1809. AI_SHOP_NAME = "" # 每次进店重置,避免上一家的AI店名残留
  1810. _LAST_SKIP_WHITE = False
  1811. # 安全的文件名前缀(用hash避免中文路径cv2兼容问题)
  1812. # 多设备隔离:加入设备ID,防止并发时两台设备写同一个文件
  1813. import hashlib
  1814. _hash = hashlib.md5(shop_name.encode()).hexdigest()[:8] if shop_name else "unknown"
  1815. _pfx = lambda name: _shot_path(ex, f"_s4_{_hash}_{name}")
  1816. human_sleep(6)
  1817. # ── 检测页面类型:验证码/风控/正常 ──
  1818. # unknown恢复流程:等1秒二次截图确认 → 每确认一次unknown就back一层+重新点击进入(网络没加载就刷新)
  1819. # → 最多试UNKNOWN_REENTER_LIMIT次加载,全失败返回__SKIP__(跳过本商品,由step3决定是否升级重启)
  1820. unknown_round = 0
  1821. for page_retry in range(16):
  1822. shot_check = _pfx("page_check.png")
  1823. ex.driver.screenshot(shot_check)
  1824. check_raw = OCR.recognize(shot_check, detail="all")
  1825. # 方法A: 模板匹配检测验证码
  1826. import os as _os
  1827. captcha_tpl = str(Path(__file__).parent / "files" / "captcha1.png")
  1828. if _os.path.exists(captcha_tpl):
  1829. si = cv2.imread(shot_check)
  1830. ti = cv2.imread(captcha_tpl)
  1831. if si is not None and ti is not None:
  1832. gs = cv2.cvtColor(si, cv2.COLOR_BGR2GRAY)
  1833. gt = cv2.cvtColor(ti, cv2.COLOR_BGR2GRAY)
  1834. h_s, w_s = gs.shape
  1835. crop_y1, crop_y2 = int(h_s * 0.25), int(h_s * 0.75)
  1836. crop_x1, crop_x2 = 0, 400
  1837. gs_crop = gs[crop_y1:crop_y2, crop_x1:crop_x2]
  1838. scores = []
  1839. for fn, ss, tt in [
  1840. ("gray", gs_crop, gt),
  1841. ("edge", cv2.Canny(gs_crop,30,100), cv2.Canny(gt,30,100)),
  1842. ("hist", cv2.equalizeHist(gs_crop), cv2.equalizeHist(gt)),
  1843. ("blur", cv2.GaussianBlur(gs_crop,(3,3),0), cv2.GaussianBlur(gt,(3,3),0)),
  1844. ("otsu", cv2.threshold(gs_crop,0,255,cv2.THRESH_BINARY+cv2.THRESH_OTSU)[1],
  1845. cv2.threshold(gt,0,255,cv2.THRESH_BINARY+cv2.THRESH_OTSU)[1]),
  1846. ]:
  1847. if ss.ndim == 2 and tt.ndim == 2 and ss.shape[0] >= tt.shape[0] and ss.shape[1] >= tt.shape[1]:
  1848. r = cv2.matchTemplate(ss, tt, cv2.TM_CCOEFF_NORMED)
  1849. _, mv, _, _ = cv2.minMaxLoc(r)
  1850. scores.append((mv, fn))
  1851. if scores:
  1852. best_v = max(s[0] for s in scores)
  1853. best_m = max(scores, key=lambda s: s[0])[1]
  1854. print(f" 验证码模板匹配: {best_m}={best_v:.3f}")
  1855. captcha_kw = any("拖动滑块" in r["text"] or "请按住滑块" in r["text"] or "安全验证" in r["text"] for r in check_raw)
  1856. nine_kw = any("提交" in r["text"] or "没有新图片" in r["text"] for r in check_raw)
  1857. if best_v >= 0.30 and (captcha_kw or nine_kw):
  1858. print(f" ⚠ 检测到验证码,尝试自动处理...")
  1859. if _handle_captcha(ex, [r["text"] for r in check_raw]):
  1860. continue
  1861. return "__TERMINATE__"
  1862. elif best_v >= 0.30 and not captcha_kw:
  1863. print(f" ⚠ 模板匹配命中但OCR无验证码关键词,忽略")
  1864. page_type = AIParser().check_page(check_raw)
  1865. ptype = page_type.get("type", "unknown")
  1866. if ptype == "risk":
  1867. # OCR文字二次确认:店铺页/列表页文字多但无验证码特征词 → 误判,忽略
  1868. if not _has_captcha_kw([r["text"] for r in check_raw]):
  1869. print(" AI判risk但OCR无验证码特征词,忽略(店铺页/列表页误判)")
  1870. continue
  1871. print(f" ⚠ AI检测到验证码,尝试自动处理...")
  1872. if _handle_captcha(ex, [r["text"] for r in check_raw]):
  1873. continue
  1874. return "__TERMINATE__"
  1875. if ptype == "home":
  1876. # OCR多特征复核:店铺页自带「首页/商品/评价」页签会骗过AI(实测阿里健康误判)。
  1877. # 真首页特征 = 「AI点外卖」导航 或 频道宫格≥3个;「刚刚搜过」在场必非首页
  1878. if not _looks_like_app_home(check_raw):
  1879. if _look_like_shop_page(check_raw):
  1880. print(" AI判home但OCR复核为店铺页('首页'是页签字样),按进店继续")
  1881. break
  1882. print(" AI判home但OCR复核非首页(无AI点外卖/频道宫格≥3),继续检测")
  1883. continue
  1884. # 二次确认:截图太早页面没加载完时AI可能误判首页(底部导航"我的"在任何页面都可见)
  1885. human_sleep(1)
  1886. ex.driver.screenshot(shot_check)
  1887. confirm_raw = OCR.recognize(shot_check, detail="all")
  1888. ptype2 = AIParser().check_page(confirm_raw).get("type", "unknown")
  1889. if ptype2 == "home":
  1890. _save_evidence(shot_check, "restart_home1")
  1891. print(" ⚠ 二次确认仍为首页(被踢回/退过头),触发重启恢复")
  1892. return "__RESTART__"
  1893. if ptype2 == "normal":
  1894. break # 页面加载完成,恢复正常
  1895. print(f" AI先判home,二次确认为{ptype2},继续检测")
  1896. continue
  1897. if ptype == "list":
  1898. # 第一层:OCR复核防AI误判(店铺页只有1个店铺名且在屏幕上方,AI有时把店内商品列表看成列表页)
  1899. if _look_like_shop_page(check_raw):
  1900. print(" AI判list但OCR复核为店铺页(单店名且在顶部),按进店继续")
  1901. break
  1902. # 第二层:可能是点击后跳转未完成(残影还是列表页),等1秒重拍让AI再判一次
  1903. human_sleep(1)
  1904. ex.driver.screenshot(shot_check)
  1905. confirm_raw = OCR.recognize(shot_check, detail="all")
  1906. ptype2 = AIParser().check_page(confirm_raw).get("type", "unknown")
  1907. if ptype2 != "list":
  1908. print(f" 二次确认为{ptype2}(跳转完成),继续检测")
  1909. continue
  1910. print(" ⚠ 二次确认仍在列表页(未成功进店),放弃本店")
  1911. return ""
  1912. if ptype == "login":
  1913. global ACCOUNT_ABORTED
  1914. ACCOUNT_ABORTED = True
  1915. err_log.log_error(ex.device_id, "account_kicked", message="AI检测到登录页,账号被踢/封号")
  1916. print(" ⚠ AI检测到登录页(账号被踢/封号),停止采集")
  1917. return "__TERMINATE__"
  1918. if ptype == "normal":
  1919. _ai_shop = str(page_type.get("shop") or "").strip()
  1920. if _ai_shop:
  1921. AI_SHOP_NAME = _ai_shop # 详情页AI提取的店铺名(Excel对比用)
  1922. break # 正常,跳出重试循环
  1923. # qrcode/unknown:等1秒二次截图确认(点击后立即截图可能页面没加载完,避免误判)
  1924. human_sleep(1)
  1925. ex.driver.screenshot(shot_check)
  1926. confirm_raw = OCR.recognize(shot_check, detail="all")
  1927. confirm_page = AIParser().check_page(confirm_raw)
  1928. ptype2 = confirm_page.get("type", "unknown")
  1929. if ptype2 == "normal":
  1930. _ai_shop = str(confirm_page.get("shop") or "").strip()
  1931. if _ai_shop:
  1932. AI_SHOP_NAME = _ai_shop # 详情页AI提取的店铺名(Excel对比用)
  1933. break
  1934. if ptype2 == "risk":
  1935. if not _has_captcha_kw([r["text"] for r in confirm_raw]):
  1936. print(" 二次确认risk但OCR无验证码特征词,忽略")
  1937. continue
  1938. if _handle_captcha(ex, [r["text"] for r in confirm_raw]):
  1939. continue
  1940. return "__TERMINATE__"
  1941. if ptype2 == "home":
  1942. # 同上:AI判home先OCR多特征复核,防店铺页「首页」页签误触发重启
  1943. if not _looks_like_app_home(confirm_raw):
  1944. if _look_like_shop_page(confirm_raw):
  1945. print(" AI二次判home但OCR复核为店铺页,按进店继续")
  1946. break
  1947. print(" AI二次判home但OCR复核非首页,继续检测")
  1948. continue
  1949. _save_evidence(shot_check, "restart_home2")
  1950. print(" ⚠ 二次确认检测到首页,触发重启恢复")
  1951. return "__RESTART__"
  1952. if ptype2 == "list":
  1953. # OCR复核防AI误判(同第一处list分支)
  1954. if _look_like_shop_page(confirm_raw):
  1955. print(" AI二次确认判list但OCR复核为店铺页,按进店继续")
  1956. break
  1957. print(" ⚠ 二次确认检测到列表页(未成功进店),放弃本店")
  1958. return ""
  1959. if ptype2 == "login":
  1960. ACCOUNT_ABORTED = True # 本函数已声明global
  1961. err_log.log_error(ex.device_id, "account_kicked", message="二次确认检测到登录页,账号被踢/封号")
  1962. print(" ⚠ 二次确认检测到登录页(账号被踢/封号),停止采集")
  1963. return "__TERMINATE__"
  1964. # 两次都是unknown → 立即back一层+重新点击进入(每确认一次就刷新,不攒次数)
  1965. unknown_round += 1
  1966. _ws = _is_white_screen(shot_check)
  1967. if unknown_round >= UNKNOWN_REENTER_LIMIT:
  1968. # 5次加载尝试全失败:保存现场截图,跳过本商品(不重启,由step3处理连续跳过)
  1969. import shutil
  1970. err_dir = SCREENSHOT_DIR / ex.device_id / "step4" / "unrecognized"
  1971. err_dir.mkdir(exist_ok=True)
  1972. shutil.copy(shot_check, str(err_dir / f"unknown_{int(time.time())}.png"))
  1973. print(f" ⚠ 连续{UNKNOWN_REENTER_LIMIT}次进入页面均未加载({'白屏' if _ws else '其他异常'}),跳过本商品")
  1974. _LAST_SKIP_WHITE = _ws
  1975. return "__SKIP__"
  1976. print(f" 确认unknown(第{unknown_round}次,{'白屏' if _ws else '非白屏'}),back一层并重新点击进入")
  1977. ex.driver.press("back")
  1978. human_sleep(1.5)
  1979. if shop_xy:
  1980. ex.tap(shop_xy[0], shop_xy[1])
  1981. human_sleep(2.5)
  1982. human_sleep(2)
  1983. else:
  1984. # 检测循环耗尽仍未正常 → 同样按跳过处理(不重启)
  1985. import shutil
  1986. _ws = _is_white_screen(shot_check)
  1987. err_dir = SCREENSHOT_DIR / ex.device_id / "step4" / "unrecognized"
  1988. err_dir.mkdir(exist_ok=True)
  1989. shutil.copy(shot_check, str(err_dir / f"unknown_{int(time.time())}.png"))
  1990. print(f" ⚠ 检测循环耗尽({ptype},{'白屏' if _ws else '其他异常'}),跳过本商品")
  1991. _LAST_SKIP_WHITE = _ws
  1992. return "__SKIP__"
  1993. # normal → 继续
  1994. # ── 店铺页判断:OCR同时检测到「刚刚搜过」和「评价」说明在店铺页 ──
  1995. in_shop = False
  1996. for _ in range(10):
  1997. shop_check = _pfx("shop_check.png")
  1998. ex.driver.screenshot(shop_check)
  1999. shop_raw = OCR.recognize(shop_check, detail="text")
  2000. has_ganggang = any("刚刚搜过" in t for t in shop_raw)
  2001. has_pingjia = any("评价" in t for t in shop_raw)
  2002. if has_ganggang and has_pingjia:
  2003. in_shop = True
  2004. print(f" 已确认在店铺页")
  2005. break
  2006. human_sleep(1)
  2007. if not in_shop:
  2008. print(f" ⚠ 未检测到店铺页,继续尝试...")
  2009. # ── 第1步:截图 + AI找商品标题坐标(AI失败或返回None时重试3次,页面可能未加载完)──
  2010. system_prompt = """你收到店铺页的OCR文字。商品标题文字坐标已知(从OCR中有x,y)。
  2011. 请找到和以下商品标题匹配的文字块,返回其点击坐标。
  2012. 【重要规则】
  2013. - 坐标必须从OCR数据中选取,不得编造或估算
  2014. - 如果找不到完全匹配的,找最相似的
  2015. - 如果完全找不到,返回null
  2016. 只返回JSON:
  2017. {"title_xy": [x, y] 或 null, "shop": "店铺名"}"""
  2018. parser = AIParser()
  2019. title_xy = None
  2020. for ai_try in range(3):
  2021. shot = _pfx("shop.png")
  2022. ex.driver.screenshot(shot)
  2023. raw = OCR.recognize(shot, detail="all")
  2024. sorted_r = sorted(raw, key=lambda r: r["bbox"][0][1])
  2025. lines = []
  2026. for r in sorted_r:
  2027. cx = (r["bbox"][0][0] + r["bbox"][2][0]) // 2
  2028. cy = (r["bbox"][0][1] + r["bbox"][2][1]) // 2
  2029. lines.append(f"[x={cx:4d}, y={cy:4d}] {r['text']}")
  2030. ocr_text = "\n".join(lines)
  2031. resp = parser._call(system_prompt, f"商品标题: {product_title}\n\nOCR文字:\n{ocr_text}\n\n请返回商品标题坐标。")
  2032. import json
  2033. cleaned = resp.strip()
  2034. if cleaned.startswith("```"):
  2035. cl = cleaned.split("\n")
  2036. if cl[0].startswith("```"): cl = cl[1:]
  2037. if cl and cl[-1].strip() == "```": cl = cl[:-1]
  2038. cleaned = "\n".join(cl).strip()
  2039. try:
  2040. data = json.loads(cleaned)
  2041. title_xy = data.get("title_xy")
  2042. except json.JSONDecodeError:
  2043. title_xy = None
  2044. if title_xy and isinstance(title_xy, list) and len(title_xy) == 2:
  2045. break
  2046. print(f" ⚠ AI未返回有效坐标(第{ai_try+1}次): {title_xy},2秒后重试")
  2047. human_sleep(2)
  2048. if not title_xy or not isinstance(title_xy, list) or len(title_xy) != 2:
  2049. print(f" ⚠ AI 3次均未返回有效坐标: {title_xy}")
  2050. return ""
  2051. tx, ty = title_xy
  2052. if tx is None or ty is None:
  2053. print(f" ⚠ AI返回空坐标")
  2054. return ""
  2055. tx, ty = int(tx), int(ty)
  2056. w, h = ex.driver.window_size()
  2057. if not (0 <= tx <= w and 0 <= ty <= h):
  2058. print(f" ⚠ 坐标越界: ({tx},{ty}) 超出屏幕 {w}x{h}")
  2059. return ""
  2060. # ── 第2步:点击商品标题 → 进入商品详情(最多重试3次)──
  2061. entered_detail = False
  2062. for attempt in range(3):
  2063. print(f" 点击商品标题: ({tx},{ty}) (第{attempt+1}次)")
  2064. ex.tap(tx, ty)
  2065. human_sleep(3) # 点击后先等页面响应(验证码处理后/网络慢时切换慢)
  2066. # 检测是否进入商品详情页(验证码检测优先——验证码页文字块少,不能被"加载中"分支挡掉)
  2067. notready_streak = 0 # 连续"未就绪"轮次(白屏时OCR只剩状态栏1~3块文字)
  2068. for _ in range(8):
  2069. human_sleep(2)
  2070. detail_check = _pfx("detail_check.png")
  2071. ex.driver.screenshot(detail_check)
  2072. detail_raw = OCR.recognize(detail_check, detail="all")
  2073. detail_texts = [r["text"] for r in detail_raw]
  2074. # 1. 验证码优先检测:验证码页文字块少(可能≤8),必须先于"加载中"判断
  2075. captcha_kw = any("拖动滑块" in t or "请按住滑块" in t or "安全验证" in t for t in detail_texts)
  2076. if captcha_kw:
  2077. print(f" ⚠ 检测到验证码页面,尝试自动处理...")
  2078. if _handle_captcha(ex, detail_texts):
  2079. continue
  2080. return "__TERMINATE__"
  2081. # 1.5 「重新加载」页(出错了/检修中,验证码通过后常见):点重新加载后继续等
  2082. if any("重新加载" in t for t in detail_texts):
  2083. for r in detail_raw:
  2084. if "重新加载" in r["text"]:
  2085. box = r["bbox"]
  2086. bx = (box[0][0] + box[2][0]) // 2
  2087. by = (box[0][1] + box[2][1]) // 2
  2088. print(f" 检测到「重新加载」页,点击重新加载 ({bx},{by})")
  2089. ex.tap(bx, by)
  2090. human_sleep(2)
  2091. break
  2092. continue
  2093. # 2. 页面加载中/切换中(文字少且无验证码)→ 继续等待,不误判
  2094. if len(detail_texts) <= 8:
  2095. notready_streak += 1
  2096. # 连续3轮仅状态栏级文字 → 大概率白屏(有图无字),立即跳过别盲等
  2097. # (原逻辑只会"继续等待",3次尝试×8轮能空转1分半)
  2098. if notready_streak >= 3 and _is_white_screen(detail_check):
  2099. _save_evidence(detail_check, "white_skip_detail")
  2100. _LAST_SKIP_WHITE = True
  2101. print(" ⚠ 详情页白屏(连续多轮仅状态栏文字),跳过本商品")
  2102. return "__SKIP__"
  2103. print(f" 页面未就绪(仅{len(detail_texts)}块文字),继续等待...")
  2104. continue
  2105. # 2.5 检测是否退回首页/列表页(点标题失败/back过头时常见)→ 立即处理,不盲等
  2106. joined_texts = "".join(detail_texts)
  2107. if ("看病买药" in joined_texts) and ("我的" in joined_texts):
  2108. _save_evidence(detail_check, "restart_home_kw")
  2109. print(" ⚠ 检测到已退回首页,触发重启恢复")
  2110. return "__RESTART__"
  2111. if "筛选" in joined_texts:
  2112. print(" ⚠ 检测到已回列表页,放弃本店(点标题未成功进店)")
  2113. return ""
  2114. # 3. 检测商品详情页关键词
  2115. if any("加入购物车" in t or "立即购买" in t or "选规格" in t or "商品详情页" in t for t in detail_texts):
  2116. print(f" 已进入商品详情页")
  2117. entered_detail = True
  2118. break
  2119. # 4. 不在详情页,检测是否还在店铺页
  2120. has_ganggang = any("刚刚搜过" in t for t in detail_texts)
  2121. has_pingjia = any("评价" in t for t in detail_texts)
  2122. if has_ganggang and has_pingjia:
  2123. print(f" 仍在店铺页,重试...")
  2124. break # 跳出内层循环
  2125. if entered_detail:
  2126. break
  2127. # 内层循环结束仍未进入详情页:AI判断当前实际页面类型(诊断 + 验证码/首页/列表页兜底)
  2128. try:
  2129. last_shot = _pfx("detail_check.png")
  2130. if cv2.imread(last_shot) is not None:
  2131. ai_raw = OCR.recognize(last_shot, detail="all")
  2132. ai_page = AIParser().check_page(ai_raw)
  2133. print(f" AI页面类型: {ai_page.get('type')} - {ai_page.get('detail', '')[:40]}")
  2134. if ai_page.get("type") == "risk":
  2135. if not _has_captcha_kw([r["text"] for r in ai_raw]):
  2136. print(" AI判risk但OCR无验证码特征词,忽略")
  2137. else:
  2138. print(" ⚠ AI检测到验证码页,尝试自动处理...")
  2139. if _handle_captcha(ex, [r["text"] for r in ai_raw]):
  2140. continue
  2141. return "__TERMINATE__"
  2142. if ai_page.get("type") == "home":
  2143. # OCR多特征复核(与前两处守卫一致):店铺页「首页」页签/左侧分类栏会骗过AI
  2144. if not _looks_like_app_home(ai_raw):
  2145. print(" AI判home但OCR复核非首页(无AI点外卖/频道宫格≥3),放弃本店")
  2146. return ""
  2147. _save_evidence(last_shot, "restart_home_ai")
  2148. print(" ⚠ AI检测到首页,触发重启恢复")
  2149. return "__RESTART__"
  2150. if ai_page.get("type") == "list":
  2151. print(" ⚠ AI检测到列表页,放弃本店")
  2152. return ""
  2153. if ai_page.get("type") == "login":
  2154. ACCOUNT_ABORTED = True # 本函数已声明global
  2155. err_log.log_error(ex.device_id, "account_kicked", message="进详情失败后AI检测到登录页,账号被踢/封号")
  2156. print(" ⚠ AI检测到登录页(账号被踢/封号),停止采集")
  2157. return "__TERMINATE__"
  2158. except Exception as e:
  2159. print(f" ⚠ AI页面类型检测异常: {e}")
  2160. # 第1次失败后,重新OCR+AI获取坐标(可能是页面滚动导致坐标偏移)
  2161. if attempt < 2:
  2162. print(f" 重新OCR获取坐标...")
  2163. re_shot = _pfx("shop.png")
  2164. ex.driver.screenshot(re_shot)
  2165. re_raw = OCR.recognize(re_shot, detail="all")
  2166. re_sorted = sorted(re_raw, key=lambda r: r["bbox"][0][1])
  2167. re_lines = []
  2168. for r in re_sorted:
  2169. r_cx = (r["bbox"][0][0] + r["bbox"][2][0]) // 2
  2170. r_cy = (r["bbox"][0][1] + r["bbox"][2][1]) // 2
  2171. re_lines.append(f"[x={r_cx:4d}, y={r_cy:4d}] {r['text']}")
  2172. re_ocr_text = "\n".join(re_lines)
  2173. re_resp = parser._call(system_prompt, f"商品标题: {product_title}\n\nOCR文字:\n{re_ocr_text}\n\n请返回商品标题坐标。")
  2174. re_cleaned = re_resp.strip()
  2175. if re_cleaned.startswith("```"):
  2176. rl = re_cleaned.split("\n")
  2177. if rl[0].startswith("```"): rl = rl[1:]
  2178. if rl and rl[-1].strip() == "```": rl = rl[:-1]
  2179. re_cleaned = "\n".join(rl).strip()
  2180. try:
  2181. re_data = json.loads(re_cleaned)
  2182. re_xy = re_data.get("title_xy")
  2183. if re_xy and isinstance(re_xy, list) and len(re_xy) == 2 and re_xy[0] is not None:
  2184. tx, ty = int(re_xy[0]), int(re_xy[1])
  2185. ww, hh = ex.driver.window_size()
  2186. if not (0 <= tx <= ww and 0 <= ty <= hh):
  2187. print(f" ⚠ 新坐标越界: ({tx},{ty}),保持原坐标")
  2188. else:
  2189. print(f" 新坐标: ({tx},{ty})")
  2190. except Exception:
  2191. pass
  2192. else:
  2193. pass # 3次重试结束
  2194. if not entered_detail:
  2195. print(f" ⚠ 3次点击未进入商品详情页,跳过")
  2196. return ""
  2197. # ── 第3步:ORB特征匹配找分享图标 ──
  2198. share_shot = _pfx("find_share.png")
  2199. ex.driver.screenshot(share_shot)
  2200. screen = cv2.imread(share_shot)
  2201. template_path = str(Path(__file__).parent / "files" / "share.png")
  2202. template = cv2.imread(template_path)
  2203. sx, sy = None, None
  2204. if screen is not None and template is not None:
  2205. h_s, w_s = screen.shape[:2]
  2206. # 右上角区域(分享图标永远在右上)
  2207. roi_x1, roi_y1 = w_s * 2 // 3, 0
  2208. roi = screen[roi_y1:h_s // 4, roi_x1:w_s]
  2209. # 方法A: SIFT 特征匹配(限制右上角区域,减少干扰)
  2210. sx, sy = None, None
  2211. sift = cv2.SIFT_create(nfeatures=1500)
  2212. kp1, des1 = sift.detectAndCompute(template, None)
  2213. kp2, des2 = sift.detectAndCompute(roi, None)
  2214. if des1 is not None and des2 is not None and len(kp1) >= 2 and len(kp2) >= 2:
  2215. bf = cv2.BFMatcher()
  2216. matches = bf.knnMatch(des1, des2, k=2)
  2217. good = []
  2218. for m, n in matches:
  2219. if m.distance < 0.75 * n.distance:
  2220. good.append(m)
  2221. print(f" 分享SIFT(右上区域): 模板{len(kp1)}特征 ROI{len(kp2)}特征 优质{len(good)}")
  2222. if len(good) >= 4:
  2223. src_pts = np.float32([kp1[m.queryIdx].pt for m in good]).reshape(-1, 1, 2)
  2224. dst_pts = np.float32([kp2[m.trainIdx].pt for m in good]).reshape(-1, 1, 2)
  2225. matrix, _ = cv2.findHomography(src_pts, dst_pts, cv2.RANSAC, 5.0)
  2226. if matrix is not None:
  2227. h_t, w_t = template.shape[:2]
  2228. corners = np.float32([[0, 0], [w_t, 0], [w_t, h_t], [0, h_t]]).reshape(-1, 1, 2)
  2229. transformed = cv2.perspectiveTransform(corners, matrix)
  2230. sx = roi_x1 + int(np.mean(transformed[:, 0, 0]))
  2231. sy = int(np.mean(transformed[:, 0, 1]))
  2232. print(f" 分享SIFT匹配: ({sx},{sy})")
  2233. # 方法B: 多尺度模板匹配(右上角区域)
  2234. # 分辨率适配:以 720p 为基准,按屏幕宽度比例调整搜索尺度,覆盖 480p~1080p
  2235. base_w = 720.0
  2236. scale_ratio = w_s / base_w
  2237. # 模板在 720p 下约占 11% 屏宽,目标尺度应使模板覆盖相同比例
  2238. scales = [round(scale_ratio * s, 2) for s in [0.7, 0.8, 0.9, 1.0, 1.1, 1.2, 1.3, 1.4, 1.5]]
  2239. if sx is None:
  2240. best_val, best_loc, best_sw, best_sh = 0, None, 0, 0
  2241. for scale in scales:
  2242. scaled = cv2.resize(template, None, fx=scale, fy=scale)
  2243. sw, sh = scaled.shape[1], scaled.shape[0]
  2244. if sh > roi.shape[0] or sw > roi.shape[1]:
  2245. continue
  2246. res = cv2.matchTemplate(roi, scaled, cv2.TM_CCOEFF_NORMED)
  2247. _, mv, _, ml = cv2.minMaxLoc(res)
  2248. if mv > best_val:
  2249. best_val, best_loc, best_sw, best_sh = mv, ml, sw, sh
  2250. t_edge = cv2.Canny(scaled, 30, 100)
  2251. r_edge = cv2.Canny(roi, 30, 100)
  2252. if t_edge.shape[0] <= r_edge.shape[0] and t_edge.shape[1] <= r_edge.shape[1]:
  2253. res2 = cv2.matchTemplate(r_edge, t_edge, cv2.TM_CCOEFF_NORMED)
  2254. _, mv2, _, ml2 = cv2.minMaxLoc(res2)
  2255. if mv2 > best_val:
  2256. best_val, best_loc, best_sw, best_sh = mv2, ml2, sw, sh
  2257. print(f" 分享模板匹配(右上): 最佳={best_val:.3f}")
  2258. if best_val >= 0.26 and best_loc is not None:
  2259. sx = roi_x1 + best_loc[0] + best_sw // 2
  2260. sy = best_loc[1] + best_sh // 2
  2261. if sx is not None and sy is not None:
  2262. print(f" 分享图标: ({sx},{sy})")
  2263. ex.tap(sx, sy)
  2264. else:
  2265. print(f" ⚠ 未找到分享图标")
  2266. return ""
  2267. # 等弹窗出现,同时记录"分享到"y坐标用于QR裁剪
  2268. share_y = None
  2269. waimai_y = None
  2270. for _ in range(8):
  2271. human_sleep(1)
  2272. ck = _pfx("share_popup.png")
  2273. ex.driver.screenshot(ck)
  2274. detail = OCR.recognize(ck, detail="all")
  2275. texts = [r["text"] for r in detail]
  2276. if any("分享到" in t for t in texts):
  2277. print(f" 分享弹窗出现")
  2278. # 记录"分享到"和"外卖"的y坐标
  2279. for r in detail:
  2280. cy = (r["bbox"][0][1] + r["bbox"][2][1]) // 2
  2281. if "分享到" in r["text"] and share_y is None:
  2282. share_y = cy
  2283. if "外卖" in r["text"] and waimai_y is None:
  2284. waimai_y = cy
  2285. break
  2286. if share_y is None:
  2287. share_y = int(ex.driver.window_size()[1] * 0.74) # fallback
  2288. # ── 第4步:截图 → 多方法解析二维码(多次重试) ──
  2289. import os as _os_debug
  2290. _debug_dir = str(SCREENSHOT_DIR / ex.device_id / "step4" / "debug_qr")
  2291. _os_debug.makedirs(_debug_dir, exist_ok=True)
  2292. # 微信QR解码器(对ECI编码的饿了么QR鲁棒,实测成功率94%)
  2293. _wx_detector = cv2.wechat_qrcode.WeChatQRCode() if hasattr(cv2, "wechat_qrcode") else None
  2294. def _decode_qr(img, share_y, waimai_y=None):
  2295. """基于OCR定位的share_y裁剪QR区域解析"""
  2296. if img is None: return ""
  2297. h, w = img.shape[:2]
  2298. gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)
  2299. detector = cv2.QRCodeDetector()
  2300. # 裁剪区域:y从"分享到"上方推算(不依赖"外卖"文字,避免OCR误判)
  2301. # QR 通常在"分享到"上方 15%~35% 屏高范围,取中间偏下
  2302. y_top = max(0, share_y - int(h * 0.30))
  2303. y_bot = share_y
  2304. x_l, x_r = int(w * 0.58), int(w * 0.94)
  2305. crop_save = img[y_top:y_bot, x_l:x_r]
  2306. cv2.imwrite(_pfx("qr_crop.png"), crop_save)
  2307. # 保存调试截图:标注裁剪区域 + QR 边界(按设备ID区分,方便对比不同分辨率)
  2308. debug_img = img.copy()
  2309. cv2.rectangle(debug_img, (x_l, y_top), (x_r, y_bot), (0, 255, 0), 3)
  2310. cv2.putText(debug_img, f"crop:({x_l},{y_top})~({x_r},{y_bot})", (x_l, y_top - 10),
  2311. cv2.FONT_HERSHEY_SIMPLEX, 0.8, (0, 255, 0), 2)
  2312. _ts = int(time.time())
  2313. cv2.imwrite(f"{_debug_dir}/{ex.device_id}_qr_region_{_ts}.png", debug_img)
  2314. # 方法0: 微信QR解码器(对ECI编码的饿了么QR鲁棒,实测94%成功率,全图直接解析)
  2315. if _wx_detector is not None:
  2316. try:
  2317. wx_texts, wx_points = _wx_detector.detectAndDecode(img)
  2318. if wx_texts and wx_texts[0]:
  2319. data = wx_texts[0]
  2320. if wx_points is not None and len(wx_points) > 0:
  2321. wp = wx_points[0].astype(int)
  2322. x1, y1 = wp[:, 0].min(), wp[:, 1].min()
  2323. x2, y2 = wp[:, 0].max(), wp[:, 1].max()
  2324. cv2.rectangle(debug_img, (x1, y1), (x2, y2), (0, 0, 255), 3)
  2325. cv2.putText(debug_img, "WX-QR", (x1, y1 - 10),
  2326. cv2.FONT_HERSHEY_SIMPLEX, 0.8, (0, 0, 255), 2)
  2327. cv2.imwrite(f"{_debug_dir}/{ex.device_id}_qr_found_{_ts}.png", debug_img)
  2328. return data
  2329. except Exception:
  2330. pass
  2331. def _try_decode(roi_gray, zooms=(1,)):
  2332. """在灰度图上尝试多种方式解码"""
  2333. if roi_gray is None or roi_gray.size == 0 or roi_gray.shape[0] == 0 or roi_gray.shape[1] == 0:
  2334. return ""
  2335. # 高分辨率下 QR 可能太大导致 detect 失败,先缩到合理尺寸再解析
  2336. h_roi, w_roi = roi_gray.shape[:2]
  2337. if max(h_roi, w_roi) > 500:
  2338. scale_down = 500 / max(h_roi, w_roi)
  2339. roi_small = cv2.resize(roi_gray, None, fx=scale_down, fy=scale_down, interpolation=cv2.INTER_AREA)
  2340. else:
  2341. roi_small = roi_gray
  2342. for z in zooms:
  2343. if z > 1:
  2344. big = cv2.resize(roi_small, None, fx=z, fy=z, interpolation=cv2.INTER_NEAREST)
  2345. else:
  2346. big = roi_small
  2347. data, _, _ = detector.detectAndDecode(big)
  2348. if data: return data
  2349. # OTSU + zoom
  2350. for z in zooms:
  2351. _, th = cv2.threshold(roi_small, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)
  2352. big = cv2.resize(th, None, fx=z, fy=z, interpolation=cv2.INTER_NEAREST) if z > 1 else th
  2353. data, _, _ = detector.detectAndDecode(big)
  2354. if data: return data
  2355. return ""
  2356. # 方法A: 全图detect定位QR → 中心 ±qr_half 精确裁剪解析(qr_half 随分辨率缩放)
  2357. # QR 在全图里 detect 可能失败(干扰太多),但先试试
  2358. ok, points = detector.detect(gray)
  2359. if ok and points is not None and len(points) > 0:
  2360. pts = points[0].astype(int)
  2361. cx = int(np.mean(pts[:, 0]))
  2362. cy = int(np.mean(pts[:, 1]))
  2363. # 根据 detect 到的 QR 边界估算大小,裁剪 QR 中心 ± qr_half
  2364. qr_half = max(int(max(np.linalg.norm(pts[0] - pts[1]), np.linalg.norm(pts[1] - pts[2])) / 2) + 20, 60)
  2365. x1, y1 = max(0, cx - qr_half), max(0, cy - qr_half)
  2366. x2, y2 = min(w, cx + qr_half), min(h, cy + qr_half)
  2367. if x2 > x1 and y2 > y1:
  2368. data = _try_decode(gray[y1:y2, x1:x2], (1, 2, 3))
  2369. if data:
  2370. # 在调试截图上标注 QR 检测位置
  2371. cv2.rectangle(debug_img, (x1, y1), (x2, y2), (0, 0, 255), 3)
  2372. cv2.putText(debug_img, f"QR:({cx},{cy})", (x1, y1 - 10),
  2373. cv2.FONT_HERSHEY_SIMPLEX, 0.8, (0, 0, 255), 2)
  2374. cv2.imwrite(f"{_debug_dir}/{ex.device_id}_qr_found_{_ts}.png", debug_img)
  2375. return data
  2376. # 方法B: 用 detector.detect 在扫描区域内定位 QR → 中心 ± qr_half 精确裁剪解析
  2377. scan_area = gray[y_top:y_bot, x_l:x_r]
  2378. sh, sw = scan_area.shape
  2379. ok2, pts2 = detector.detect(scan_area)
  2380. if ok2 and pts2 is not None and len(pts2) > 0:
  2381. qr_pts = pts2[0].astype(int)
  2382. qx = int(np.mean(qr_pts[:, 0]))
  2383. qy = int(np.mean(qr_pts[:, 1]))
  2384. qr_half = max(int(max(np.linalg.norm(qr_pts[0] - qr_pts[1]), np.linalg.norm(qr_pts[1] - qr_pts[2])) / 2) + 20, 60)
  2385. qx1, qy1 = max(0, qx - qr_half), max(0, qy - qr_half)
  2386. qx2, qy2 = min(sw, qx + qr_half), min(sh, qy + qr_half)
  2387. if qx2 > qx1 and qy2 > qy1:
  2388. data = _try_decode(scan_area[qy1:qy2, qx1:qx2], (1, 2, 3))
  2389. if data:
  2390. cv2.rectangle(debug_img, (x_l + qx1, y_top + qy1), (x_l + qx2, y_top + qy2), (0, 0, 255), 3)
  2391. cv2.putText(debug_img, f"QR:({x_l + qx},{y_top + qy})", (x_l + qx1, y_top + qy1 - 10),
  2392. cv2.FONT_HERSHEY_SIMPLEX, 0.8, (0, 0, 255), 2)
  2393. cv2.imwrite(f"{_debug_dir}/{ex.device_id}_qr_found_{_ts}.png", debug_img)
  2394. return data
  2395. # 方法C: 滑动窗口扫描(基于OCR定位区域,尺寸随分辨率缩放)
  2396. win = max(40, min(200, sh // 2, sw // 2))
  2397. step = max(40, win // 3)
  2398. for y in range(0, max(1, sh - win), step):
  2399. for x in range(0, max(1, sw - win), step):
  2400. patch = scan_area[y:y+win, x:x+win]
  2401. data = _try_decode(patch, (1, 2))
  2402. if data:
  2403. # 在调试截图上标注命中的窗口位置
  2404. cv2.rectangle(debug_img, (x_l + x, y_top + y), (x_l + x + win, y_top + y + win), (0, 0, 255), 3)
  2405. cv2.imwrite(f"{_debug_dir}/{ex.device_id}_qr_found_{_ts}.png", debug_img)
  2406. return data
  2407. # 方法D: 全图 OTSU + detect → 中心 ± qr_half 精确裁剪解析
  2408. _, full_th = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)
  2409. ok3, pts3 = detector.detect(full_th)
  2410. if ok3 and pts3 is not None and len(pts3) > 0:
  2411. qr_pts = pts3[0].astype(int)
  2412. qx = int(np.mean(qr_pts[:, 0]))
  2413. qy = int(np.mean(qr_pts[:, 1]))
  2414. qr_half = max(int(max(np.linalg.norm(qr_pts[0] - qr_pts[1]), np.linalg.norm(qr_pts[1] - qr_pts[2])) / 2) + 20, 60)
  2415. qx1, qy1 = max(0, qx - qr_half), max(0, qy - qr_half)
  2416. qx2, qy2 = min(w, qx + qr_half), min(h, qy + qr_half)
  2417. if qx2 > qx1 and qy2 > qy1:
  2418. data = _try_decode(gray[qy1:qy2, qx1:qx2], (1, 2, 3))
  2419. if data:
  2420. cv2.rectangle(debug_img, (qx1, qy1), (qx2, qy2), (0, 0, 255), 3)
  2421. cv2.putText(debug_img, f"QR:({qx},{qy})", (qx1, qy1 - 10),
  2422. cv2.FONT_HERSHEY_SIMPLEX, 0.8, (0, 0, 255), 2)
  2423. cv2.imwrite(f"{_debug_dir}/{ex.device_id}_qr_found_{_ts}.png", debug_img)
  2424. return data
  2425. return ""
  2426. for retry in range(6): # 最多等 5+2*5=15秒
  2427. time.sleep(5 if retry == 0 else 2)
  2428. qr_shot = _pfx("qr.png")
  2429. ex.driver.screenshot(qr_shot)
  2430. data = _decode_qr(cv2.imread(qr_shot), share_y, waimai_y)
  2431. if data:
  2432. print(f" QR链接: {data[:100]}")
  2433. return data
  2434. return ""
  2435. # ── 步骤 3:滑动 + 逐个点击店铺 ──────────────────────
  2436. def _progress_file_path(device_id: str, keyword: str) -> str:
  2437. """进度文件路径(美团同款:ycwj/{设备}_{药品}.txt)"""
  2438. import hashlib
  2439. safe = hashlib.md5(keyword.encode()).hexdigest()[:8]
  2440. return str(Path(__file__).parent / "ycwj" / f"{device_id}_{safe}.txt")
  2441. def _save_progress(device_id: str, keyword: str, visited: set, scroll_px: int, batch_no: int) -> None:
  2442. """保存采集进度(每批滑动后调用,异常退出时进度已在)"""
  2443. import os
  2444. try:
  2445. path = _progress_file_path(device_id, keyword)
  2446. os.makedirs(os.path.dirname(path), exist_ok=True)
  2447. data = {
  2448. "visited": sorted(visited),
  2449. "scroll_px": scroll_px,
  2450. "batch_no": batch_no,
  2451. "time": time.strftime("%Y-%m-%d %H:%M:%S"),
  2452. }
  2453. with open(path, "w", encoding="utf-8") as f:
  2454. json.dump(data, f, ensure_ascii=False, indent=2)
  2455. except Exception as e:
  2456. print(f"[step3] 保存进度失败: {e}")
  2457. def _load_progress(device_id: str, keyword: str):
  2458. """读取采集进度,无进度文件返回 None"""
  2459. import os
  2460. try:
  2461. path = _progress_file_path(device_id, keyword)
  2462. if not os.path.exists(path):
  2463. return None
  2464. with open(path, "r", encoding="utf-8") as f:
  2465. return json.load(f)
  2466. except Exception as e:
  2467. print(f"[step3] 读取进度失败: {e}")
  2468. return None
  2469. def _delete_progress(device_id: str, keyword: str) -> None:
  2470. """任务正常完成后删除进度文件"""
  2471. import os
  2472. try:
  2473. path = _progress_file_path(device_id, keyword)
  2474. if os.path.exists(path):
  2475. os.remove(path)
  2476. print(f"[step3] 进度文件已删除: {path}")
  2477. except Exception as e:
  2478. print(f"[step3] 删除进度失败: {e}")
  2479. def _spec_ok(title: str, spec_list: list) -> bool:
  2480. """
  2481. 标题是否包含任一目标规格(美团 is_link_spec_useful 同款)。
  2482. 规格是数字+单位,OCR对数字错误率极低,用精确匹配(模糊匹配会把"10袋"误配到"10g")。
  2483. """
  2484. if not spec_list:
  2485. return True
  2486. nt = normalize_match_text(title)
  2487. return any(normalize_match_text(s) in nt for s in spec_list)
  2488. def _recover_to_list(ex: SafeExecutor, brand_keyword: str,
  2489. h: int, total_scroll_px: int) -> str:
  2490. """
  2491. step3页面异常分级恢复:识别当前页 → 用最小代价回到搜索结果列表页。
  2492. 返回: list=本来就在列表(误报) back=back退回(列表位置不变,无需滑回)
  2493. research=重新搜索 restart=重启App fail=仍无法确认在列表页
  2494. """
  2495. def _classify() -> str:
  2496. """当前页分类: list/home/page/risk/login/unknown(OCR先查列表页特征「筛选」,其余交给AI)"""
  2497. shot = _shot_path(ex, "recover_pos.png")
  2498. ex.driver.screenshot(shot)
  2499. if cv2.imread(shot) is None:
  2500. ex.driver.screenshot(shot)
  2501. raw = OCR.recognize(shot, detail="all")
  2502. # 药品列表页特征 = 「筛选」+「快递」并排(与step2搜索成功判据一致)。
  2503. # 外卖首页也有「筛选」,单看筛选会把首页误判成列表 → 恢复动作=list → 死循环
  2504. _joined = "".join(r["text"] for r in raw)
  2505. if ("筛选" in _joined) and ("快递" in _joined):
  2506. return "list"
  2507. ptype = str(AIParser().check_page(raw).get("type") or "unknown")
  2508. if ptype in ("shop", "detail", "normal", "qrcode"):
  2509. return "page"
  2510. if ptype in ("home", "risk", "login"):
  2511. return ptype
  2512. return "unknown" # 含AI判list但OCR无「筛选」的可疑情况 → 走重启兜底最稳
  2513. def _scroll_back():
  2514. # 从列表顶部滑动恢复到上次采集位置
  2515. remain = total_scroll_px
  2516. while remain > 0:
  2517. step = min(remain, int(h * 0.5))
  2518. _swipe_next_batch(ex, step)
  2519. remain -= step
  2520. human_sleep(2)
  2521. try:
  2522. page = _classify()
  2523. if page == "list":
  2524. return "list"
  2525. if page == "login":
  2526. # 登录页:账号可能被踢,循环顶部 _check_account_kicked 会统一处理
  2527. return "fail"
  2528. if page == "risk":
  2529. # 风控弹窗 → 先走验证码处理再看
  2530. _ai_check_captcha(ex)
  2531. page = _classify()
  2532. if page == "list":
  2533. return "back"
  2534. if page == "page":
  2535. # 认得出是店铺/详情/二维码页 → back退回(列表滚动位置不丢)
  2536. for _ in range(3):
  2537. ex.driver.press("back")
  2538. human_sleep(1.4)
  2539. page = _classify()
  2540. if page == "list":
  2541. return "back"
  2542. if page != "page":
  2543. break
  2544. if page == "home":
  2545. # 在App首页 → 只重新搜索,不重启App
  2546. if not step2_search(ex, brand_keyword):
  2547. return "fail"
  2548. _scroll_back()
  2549. return "research" if _classify() == "list" else "fail"
  2550. if page == "login":
  2551. return "fail"
  2552. # unknown → 认不出在哪 → 重启App
  2553. step1_open_app(ex)
  2554. if not step2_search(ex, brand_keyword):
  2555. return "fail"
  2556. _scroll_back()
  2557. return "restart" if _classify() == "list" else "fail"
  2558. except Exception as e:
  2559. print(f"[step3] 页面恢复异常: {e}")
  2560. return "fail"
  2561. def step3_swipe_and_enter(ex: SafeExecutor, keyword: str, brand: str = "", task: dict = None,
  2562. scheduler=None) -> list:
  2563. """
  2564. 截图 → AI分析 → 逐个点击全部可见店铺 → 下滑加载更多 → 继续点击 → 直到全部遍历
  2565. 品牌+药品名过滤(美团 is_link_useful 同款):标题必须同时包含品牌名和药品名,
  2566. 否则过滤;连续30个无关商品则任务结束停止采集。
  2567. task: 调度任务 dict(task_id/enterprise_id/collect_round/current_page等,手动模式传 None)
  2568. scheduler: 调度器(逐页回告进度,手动模式传 None)
  2569. """
  2570. print("\n" + "=" * 40)
  2571. print(" 步骤 3:遍历店铺")
  2572. print("=" * 40)
  2573. # 新任务开始:重置所有停止标志和缓存(上个任务的验证码/封号停止不能污染本任务)
  2574. global _CLOSE_BTN_CACHE, CAPTCHA_ABORTED, CAPTCHA_ABORT_REASON, ACCOUNT_ABORTED, CURRENT_PAGE
  2575. global WHITE_SCREEN_REASSIGN, WHITE_SCREEN_REASSIGN_REASON, _LAST_SKIP_WHITE
  2576. global _SCHEDULER, _TASK_ID, _CRAWLED_COUNT, _ROW_REST_MARK
  2577. _CLOSE_BTN_CACHE = None
  2578. CAPTCHA_ABORTED = False
  2579. CAPTCHA_ABORT_REASON = ""
  2580. ACCOUNT_ABORTED = False
  2581. WHITE_SCREEN_REASSIGN = False
  2582. WHITE_SCREEN_REASSIGN_REASON = ""
  2583. _LAST_SKIP_WHITE = False
  2584. CURRENT_PAGE = 0
  2585. _SCHEDULER = scheduler
  2586. _TASK_ID = (task or {}).get("task_id")
  2587. _CRAWLED_COUNT = 0
  2588. _ROW_REST_MARK = 0
  2589. w, h = ex.driver.window_size()
  2590. batch_no = 0
  2591. empty_streak = 0 # 连续没有新店铺的批次数
  2592. all_results = []
  2593. unrelated = 0 # 连续无关商品计数(美团同款;>=30 停止采集)
  2594. n_brand = normalize_match_text(brand)
  2595. # 目标规格(调度任务/手动 --spec 传入,如 "120粒|60粒"),用于规格过滤+入库
  2596. spec_raw = str((task or {}).get("product_specs") or "")
  2597. spec_list = (task or {}).get("spec_list") or []
  2598. if isinstance(spec_list, str): # 兼容直接传字符串
  2599. spec_list = [s.strip() for s in re.split(r'[|、,,\n\r]+', spec_list) if s.strip()]
  2600. n_key = normalize_match_text(keyword)
  2601. stopped = False
  2602. # 恢复进度:
  2603. # 1. 跨设备接力:task 带 current_page(调度重派时给)→ 按页码滑动恢复(每页=0.7屏高,跨分辨率一致)
  2604. # 2. 同设备异常恢复:本地进度文件(visited + 滑动px,精确恢复)
  2605. start_page = int((task or {}).get("current_page") or 0)
  2606. progress = _load_progress(ex.device_id, keyword)
  2607. visited = set()
  2608. total_scroll_px = 0
  2609. if start_page > 0:
  2610. print(f"[step3] 跨设备接力恢复: 调度页码={start_page},滑动{start_page}页...")
  2611. for i in range(start_page):
  2612. _swipe_next_batch(ex, int(h * 0.7))
  2613. human_sleep(2)
  2614. elif progress:
  2615. visited = set(progress.get("visited") or [])
  2616. total_scroll_px = int(progress.get("scroll_px") or 0)
  2617. print(f"[step3] 恢复进度: 已访问{len(visited)}个店铺,需滑动恢复{total_scroll_px}px")
  2618. remain = total_scroll_px
  2619. while remain > 0:
  2620. step = min(remain, int(h * 0.5))
  2621. _swipe_next_batch(ex, step)
  2622. remain -= step
  2623. human_sleep(2)
  2624. else:
  2625. visited = set()
  2626. white_restart_count = 0 # 白屏重启恢复累计(>=MAX_WHITE_RESTART → 置重派标志)
  2627. ws_success_count = 0 # 连续成功采集计数(>=WHITE_SUCCESS_RESET_N → 白屏重启额度清零)
  2628. unknown_skip_streak = 0 # 连续"进页失败跳过"计数(跨批次,采到正常商品清零;连续2→分级恢复)
  2629. ab_recover_fail_streak = 0 # a/b页面恢复连续失败计数(只拉长恢复前冷却,不终止任务)
  2630. _list_noop_streak = 0 # 恢复连续返回"list"但页面始终page_wrong的计数(分类误判保护)
  2631. parse_fail_streak = 0 # 连续"批次截图/识别失败"计数(连续3→终止任务)
  2632. while True:
  2633. if CAPTCHA_ABORTED:
  2634. print(f"[step3] {CAPTCHA_ABORT_REASON or chr(39)+chr(39)}停止采集")
  2635. break
  2636. if scheduler is not None and getattr(scheduler, "limit_reached", False):
  2637. # 回告接口返回 code=error(平台限额/任务已释放)→ 立即停止采集
  2638. print(f"[step3] 调度返回限额/错误({getattr(scheduler, 'limit_msg', '')}),停止采集")
  2639. break
  2640. _check_account_kicked(ex) # 每批检测账号是否被踢/封号(xpath检查,开销小)
  2641. if ACCOUNT_ABORTED:
  2642. print("[step3] 账号被踢/封号,停止采集")
  2643. break
  2644. try:
  2645. named, raw_local, page_status = _get_named_shops(ex, f"step3_b{batch_no}.png", keyword)
  2646. except Exception as e:
  2647. parse_fail_streak += 1
  2648. print(f"[step3] 批次{batch_no} 截图/识别异常({e}),连续{parse_fail_streak}/3")
  2649. if parse_fail_streak >= 3:
  2650. print("[step3] 连续3批截图/识别失败,终止任务")
  2651. err_log.log_error(ex.device_id, "parse_fail_terminate",
  2652. message="连续3批截图/识别失败,终止任务",
  2653. extra={"task_id": _TASK_ID, "keyword": keyword})
  2654. stopped = True
  2655. break
  2656. human_sleep(3)
  2657. continue
  2658. parse_fail_streak = 0
  2659. ab_recover_fail_streak = 0 # 本批正常识别(在列表页),页面恢复失败计数清零
  2660. feed_end = (page_status == "no_more") # 底部「搜索结果较少」标志 → 采完本屏真卡后正常结束
  2661. if page_status == "page_wrong":
  2662. if ab_recover_fail_streak >= 3:
  2663. _cd = min(300 * (ab_recover_fail_streak - 2), 900)
  2664. print(f"[step3] 页面恢复连续失败{ab_recover_fail_streak}次,先冷却{_cd}秒再试...")
  2665. if not _chunked_sleep_with_report(_cd, "页面恢复冷却"):
  2666. stopped = True
  2667. break
  2668. print(f"[step3] 当前不在列表页({page_status}),执行页面恢复")
  2669. _save_progress(ex.device_id, keyword, visited, total_scroll_px, batch_no)
  2670. _act = _recover_to_list(ex, (brand + keyword).strip() or keyword, h, total_scroll_px)
  2671. print(f"[step3] 页面恢复动作={_act}")
  2672. if _act == "fail":
  2673. ab_recover_fail_streak += 1
  2674. else:
  2675. ab_recover_fail_streak = 0
  2676. if _act == "list":
  2677. _list_noop_streak += 1
  2678. else:
  2679. _list_noop_streak = 0
  2680. if _list_noop_streak >= 2:
  2681. # 恢复连续2次声称"本来就在列表页"但页面始终page_wrong
  2682. # → 分类误判(如外卖首页也带「筛选」),强制重启App兜底,防死循环
  2683. print(f"[step3] 恢复连续{_list_noop_streak}次返回list仍不在列表页,强制重启App恢复")
  2684. _save_progress(ex.device_id, keyword, visited, total_scroll_px, batch_no)
  2685. step1_open_app(ex)
  2686. step2_search(ex, (brand + keyword).strip() or keyword)
  2687. remain = total_scroll_px
  2688. while remain > 0:
  2689. _st = min(remain, int(h * 0.5))
  2690. _swipe_next_batch(ex, _st)
  2691. remain -= _st
  2692. human_sleep(2)
  2693. _list_noop_streak = 0
  2694. continue
  2695. if _ai_check_captcha(ex): # 列表页每批检测验证码(风控弹窗可能出现在列表)
  2696. human_sleep(1)
  2697. named, raw_local, page_status = _get_named_shops(ex, f"step3_b{batch_no}.png", keyword) # 验证码处理后重新识别
  2698. if page_status == "no_more":
  2699. feed_end = True
  2700. elif page_status != "ok":
  2701. # 验证码处理后仍不在列表页:同样定向恢复(不计数、不终止任务)
  2702. print(f"[step3] 验证码处理后仍不在列表页({page_status}),执行页面恢复")
  2703. _save_progress(ex.device_id, keyword, visited, total_scroll_px, batch_no)
  2704. _act = _recover_to_list(ex, (brand + keyword).strip() or keyword, h, total_scroll_px)
  2705. print(f"[step3] 页面恢复动作={_act}")
  2706. continue
  2707. raw_new = [s for s in named if _shop_key(s) not in visited]
  2708. # 品牌+药品名+规格过滤(测试OCR准确率时可关闭:ENABLE_TITLE_FILTER=False 直接采集全部结果)
  2709. if not ENABLE_TITLE_FILTER:
  2710. # 不过滤,直接采集全部结果;但推荐流检测始终生效:
  2711. # 标题与「品牌和药品名都不相关」的卡片连续出现 = 搜索结果耗尽、列表进入推荐区 → 结束
  2712. # (品牌或药名任一命中即算相关,避免同品牌兄弟产品误触发)
  2713. new_ones = []
  2714. for s in raw_new:
  2715. t = normalize_match_text(str(s[1] or ""))
  2716. hit = (_match_fuzzy(n_key, t) if n_key else False) or \
  2717. (_match_fuzzy(n_brand, t) if n_brand else False)
  2718. if hit:
  2719. unrelated = 0
  2720. new_ones.append(s)
  2721. else:
  2722. unrelated += 1
  2723. print(f"[step3] 推荐: {str(s[1])[:30]} 与品牌/药品名均无关 (连续{unrelated}个)")
  2724. if unrelated >= RECOMMEND_STREAK_END:
  2725. print(f"[step3] 连续{unrelated}个无关商品,判定搜索结果已耗尽(推荐流),任务结束")
  2726. err_log.log_error(ex.device_id, "recommend_feed_end",
  2727. message=f"连续{unrelated}个无关商品,判定搜索结果耗尽",
  2728. extra={"task_id": _TASK_ID, "keyword": keyword})
  2729. stopped = True
  2730. break
  2731. else:
  2732. cloud_cache = {"done": False, "texts": None}
  2733. shot_path = _shot_path(ex, f"step3_b{batch_no}.png")
  2734. new_ones = []
  2735. for s in raw_new:
  2736. # 裁剪确认用标题块自己的坐标(卡片中心裁剪会漏掉标题行)
  2737. title_y = _find_title_y(raw_local, str(s[1] or ""))
  2738. if title_y is None and len(s) > 3 and s[3]:
  2739. title_y = int(s[3][1])
  2740. v, v_reason = _match_verify(str(s[1] or ""), n_brand, n_key, shot_path, cloud_cache,
  2741. card_y=title_y)
  2742. if v in ("ok", "fuzzy"):
  2743. # 规格过滤(美团 is_link_spec_useful 同款):标题需包含任一目标规格
  2744. if spec_list and not _spec_ok(str(s[1] or ""), spec_list):
  2745. unrelated += 1
  2746. print(f"[step3] 过滤: {str(s[1])[:30]} 不含目标规格{spec_list} (连续{unrelated}个无关)")
  2747. if unrelated >= 30:
  2748. print(f"[step3] 连续{unrelated}个非目标商品,任务结束停止采集")
  2749. stopped = True
  2750. break
  2751. continue
  2752. unrelated = 0
  2753. # 标题保持OCR/回贴原样(云端改字已废弃——曾有"温胃舒"被误改成"养胃舒"的错误采集)
  2754. new_ones.append(s)
  2755. else:
  2756. unrelated += 1
  2757. print(f"[step3] 过滤: {str(s[1])[:30]} 不含目标{v_reason} (连续{unrelated}个无关)")
  2758. if unrelated >= 30:
  2759. print(f"[step3] 连续{unrelated}个非目标商品,任务结束停止采集")
  2760. stopped = True
  2761. break
  2762. if stopped:
  2763. print("[step3] 停止采集,返回已采结果")
  2764. break
  2765. print(f"[step3] 批次{batch_no}: 共{len(named)}个, 新{len(raw_new)}个, 通过过滤{len(new_ones)}个")
  2766. # 逐页回告调度:页码 = 起点页码 + 批次数(每滑一屏算一页;global已在任务开头声明)
  2767. CURRENT_PAGE = start_page + batch_no
  2768. _CRAWLED_COUNT = len(all_results)
  2769. if scheduler is not None:
  2770. try:
  2771. scheduler.post_report({
  2772. "task_id": (task or {}).get("task_id"),
  2773. "platform": scheduler.platform,
  2774. "username": scheduler.username,
  2775. "is_finished": 0,
  2776. "need_reassign": 0,
  2777. "current_page": start_page + batch_no,
  2778. "crawled_count": len(all_results),
  2779. })
  2780. except Exception as e:
  2781. print(f"[step3] 逐页回告失败: {e}")
  2782. if getattr(scheduler, "limit_reached", False):
  2783. # 回告返回 code=error(平台限额/任务已释放)→ 立即停止采集
  2784. print(f"[step3] 回告返回限额/错误({getattr(scheduler, 'limit_msg', '')}),停止采集")
  2785. break
  2786. if not new_ones:
  2787. if feed_end:
  2788. print("[step3] 底部耗尽标志在场且本批无新卡 → 搜索结果已耗尽,正常结束")
  2789. break
  2790. empty_streak += 1
  2791. # 连续无新店铺不再直接结束(终止由推荐流检测负责,这里只留防死循环保底)
  2792. if empty_streak >= EMPTY_BATCH_SAFETY:
  2793. print(f"[step3] 连续{empty_streak}批无新店铺且未进推荐流(疑似弹跳到底),保底结束")
  2794. err_log.log_error(ex.device_id, "empty_batch_safety_stop",
  2795. message=f"连续{empty_streak}批无新店铺且推荐流检测未触发,保底结束",
  2796. extra={"task_id": _TASK_ID, "keyword": keyword})
  2797. stopped = True
  2798. break
  2799. # 滑动后再试
  2800. print(f"[step3] 滑动查看下一批(无新店铺连续{empty_streak}批)")
  2801. if len(named) >= 1:
  2802. target_y = named[-1][4]
  2803. swipe_dist = max(target_y - int(h * 0.05), int(h * 0.3))
  2804. else:
  2805. swipe_dist = int(h * 0.3)
  2806. _swipe_next_batch(ex, swipe_dist)
  2807. total_scroll_px += swipe_dist
  2808. human_sleep(1.1) # 滑动后停顿 0.8~1.4s(±30%抖动)
  2809. batch_no += 1
  2810. _save_progress(ex.device_id, keyword, visited, total_scroll_px, batch_no)
  2811. continue
  2812. empty_streak = 0 # 有新店铺,重置计数
  2813. ws_recover_requested = False
  2814. home_restart_requested = False
  2815. for shop in new_ones:
  2816. result = _visit_shop(ex, shop, visited, keyword, task, batch_no=batch_no)
  2817. if result and result.get("__terminate__"):
  2818. print("[step3] 收到终止信号,停止遍历")
  2819. all_results = [r for r in all_results if not r.get("__terminate__")]
  2820. stopped = True
  2821. break
  2822. if result and result.get("__restart__"):
  2823. home_restart_requested = True
  2824. break
  2825. if result and result.get("__skip__"):
  2826. # 页面未加载跳过:单次只跳过换下一个商品;连续2个都跳过才升级分级恢复
  2827. unknown_skip_streak += 1
  2828. print(f"[step3] 商品进页失败已跳过(连续{unknown_skip_streak}/2)")
  2829. if unknown_skip_streak >= 2:
  2830. print("[step3] 连续2个商品进页失败,触发白屏分级恢复")
  2831. ws_recover_requested = True
  2832. break
  2833. continue
  2834. if result:
  2835. unknown_skip_streak = 0 # 采到一个正常商品就清零
  2836. ws_success_count += 1
  2837. if ws_success_count >= WHITE_SUCCESS_RESET_N:
  2838. if white_restart_count > 0:
  2839. print(f"[step3] 连续成功{ws_success_count}个商品,白屏重启计数清零(额度恢复)")
  2840. white_restart_count = 0
  2841. ws_success_count = 0
  2842. result["brand"] = brand
  2843. result["product_specs"] = spec_raw
  2844. all_results.append(result)
  2845. save_record(result) # 每采完一个立即入库,中断不丢数据
  2846. _CRAWLED_COUNT = len(all_results)
  2847. # 每爬 ROW_REST_EVERY 条 → 休息 ROW_REST_MIN~MAX 分钟(随机),期间回告防假死
  2848. if _CRAWLED_COUNT // ROW_REST_EVERY > _ROW_REST_MARK:
  2849. _ROW_REST_MARK = _CRAWLED_COUNT // ROW_REST_EVERY
  2850. import random as _random
  2851. rest_min = _random.uniform(ROW_REST_MIN, ROW_REST_MAX)
  2852. print(f" ⏸ 已爬{_CRAWLED_COUNT}条({ROW_REST_EVERY}条整点),休息{rest_min:.1f}分钟...")
  2853. if not _chunked_sleep_with_report(rest_min * 60, "50条"):
  2854. stopped = True
  2855. break
  2856. if ws_recover_requested:
  2857. # 白屏恢复:冷却等待无效 → 直接关App → 休息60~120s → 重启重搜滑回(超限回告调度重派)
  2858. white_restart_count += 1
  2859. if white_restart_count >= MAX_WHITE_RESTART:
  2860. WHITE_SCREEN_REASSIGN = True
  2861. WHITE_SCREEN_REASSIGN_REASON = f"商品页反复白屏,关App重启恢复{white_restart_count}次无效"
  2862. err_log.log_error(ex.device_id, "white_screen_reassign", message=WHITE_SCREEN_REASSIGN_REASON,
  2863. extra={"task_id": _TASK_ID, "restarts": white_restart_count, "keyword": keyword})
  2864. print(f"[step3] 白屏重启恢复超限({MAX_WHITE_RESTART}次),置重派标志,停止采集(进度保留)")
  2865. stopped = True
  2866. break
  2867. print(f"[step3] 检测到白屏,关App休息后重启(第{white_restart_count}/{MAX_WHITE_RESTART}次)")
  2868. _save_progress(ex.device_id, keyword, visited, total_scroll_px, batch_no)
  2869. try:
  2870. ex.driver.app_stop(APP_PACKAGE) # 先把App彻底关掉再休息
  2871. except Exception as _e:
  2872. print(f" 关App失败(不影响流程): {_e}")
  2873. import random as _random
  2874. _rest = _random.uniform(60, 120)
  2875. print(f" 休息{_rest:.0f}秒...")
  2876. if not _chunked_sleep_with_report(_rest, "白屏重启"):
  2877. stopped = True
  2878. break
  2879. step1_open_app(ex)
  2880. step2_search(ex, (brand + keyword).strip() or keyword)
  2881. # 从列表顶部滑动恢复到上次位置
  2882. remain = total_scroll_px
  2883. while remain > 0:
  2884. step = min(remain, int(h * 0.5))
  2885. _swipe_next_batch(ex, step)
  2886. remain -= step
  2887. human_sleep(2)
  2888. unknown_skip_streak = 0
  2889. continue
  2890. if home_restart_requested:
  2891. # 进店被踢回首页等:重启App恢复(不计数、不终止任务)
  2892. print("[step3] 进店被踢回首页/页面异常,重启App恢复")
  2893. _save_progress(ex.device_id, keyword, visited, total_scroll_px, batch_no)
  2894. step1_open_app(ex)
  2895. step2_search(ex, (brand + keyword).strip() or keyword)
  2896. # 从列表顶部滑动恢复到上次位置
  2897. remain = total_scroll_px
  2898. while remain > 0:
  2899. step = min(remain, int(h * 0.5))
  2900. _swipe_next_batch(ex, step)
  2901. remain -= step
  2902. human_sleep(2)
  2903. continue
  2904. if stopped:
  2905. break
  2906. if feed_end:
  2907. # 底部「搜索结果较少,为你推荐相关店铺」→ 本屏真卡已采完,搜索结果耗尽,正常结束
  2908. print("[step3] 底部耗尽标志在场,本屏真卡已采完 → 搜索结果耗尽,正常结束")
  2909. break
  2910. # 正常:滑动到本批最后一张卡片的店铺行落到屏幕5%处 → 该卡标题已离屏(下批全是新卡),
  2911. # 下一张未访问卡完整落在 y≈9%h 起,带安全边距
  2912. print(f"[step3] 已访问 {len(visited)} 个,滑动查看下一批")
  2913. if len(named) >= 1:
  2914. target_y = named[-1][4] # 本批最后一张卡片的配送距离y坐标
  2915. swipe_dist = max(target_y - int(h * 0.05), int(h * 0.3))
  2916. else:
  2917. swipe_dist = int(h * 0.3)
  2918. _swipe_next_batch(ex, swipe_dist)
  2919. total_scroll_px += swipe_dist
  2920. human_sleep(2)
  2921. batch_no += 1
  2922. _save_progress(ex.device_id, keyword, visited, total_scroll_px, batch_no) # 每页保存一次进度
  2923. continue
  2924. # 任务结束:只有正常完成才删除进度文件;验证码/封号/白屏重派等异常停止时保留(续采位置不丢)
  2925. if not CAPTCHA_ABORTED and not ACCOUNT_ABORTED and not WHITE_SCREEN_REASSIGN:
  2926. _delete_progress(ex.device_id, keyword)
  2927. else:
  2928. print(f"[step3] 异常停止,保留进度文件以便恢复: {_progress_file_path(ex.device_id, keyword)}")
  2929. # ── 输出最终结果表 ──
  2930. print("\n" + "=" * 70)
  2931. print(f" 最终结果 ({len(all_results)} 个店铺)")
  2932. print("=" * 70)
  2933. for i, r in enumerate(all_results, 1):
  2934. link_short = r["link"][:55] + "..." if len(r["link"]) > 55 else r["link"]
  2935. print(f" [{i}] {r['shop']}")
  2936. print(f" 商品: {r['title'][:40]}")
  2937. print(f" 价格: {r['price']}")
  2938. print(f" 月售: {r.get('sales', '')}")
  2939. print(f" 批准文号: {r.get('approval_no', '')}")
  2940. print(f" 有效期: {r.get('validity', '')}")
  2941. print(f" 资质编号: {r.get('license_no', '')}")
  2942. lic = r.get("license") or {}
  2943. print(f" 执照: {lic.get('单位名称', '')} 信用代码:{lic.get('社会信用代码', '')} 法人:{lic.get('法人', '')}")
  2944. print(f" 执照地址: {lic.get('地址', '')[:40]}")
  2945. print(f" 链接: {link_short}")
  2946. print()
  2947. return all_results
  2948. # ── 主入口 ──────────────────────────────────────────────
  2949. if __name__ == "__main__":
  2950. # 启动时自动清理:debug_qr 调试图只保留1天
  2951. try:
  2952. for p in SCREENSHOT_DIR.glob("*/step4/debug_qr/*.png"):
  2953. if time.time() - p.stat().st_mtime > 86400:
  2954. p.unlink(missing_ok=True)
  2955. except Exception:
  2956. pass
  2957. args = sys.argv[1:]
  2958. device_id = "RG5LFYT8UKK7BI95"
  2959. brand = "三九胃泰"
  2960. keyword = "养胃舒颗粒"
  2961. spec_raw = ""
  2962. spec_list = []
  2963. # 解析 --device / --brand / --keyword / --spec 参数(品牌和药品名分开传,对接调度系统)
  2964. filtered = []
  2965. i = 0
  2966. while i < len(args):
  2967. if args[i] == "--device" and i + 1 < len(args):
  2968. device_id = args[i + 1]
  2969. i += 2
  2970. elif args[i] == "--brand" and i + 1 < len(args):
  2971. brand = args[i + 1]
  2972. i += 2
  2973. elif args[i] == "--keyword" and i + 1 < len(args):
  2974. keyword = args[i + 1]
  2975. i += 2
  2976. elif args[i] == "--spec" and i + 1 < len(args):
  2977. spec_raw = args[i + 1]
  2978. spec_list = [s.strip() for s in re.split(r'[|、,,\n\r]+', spec_raw) if s.strip()]
  2979. i += 2
  2980. else:
  2981. filtered.append(args[i])
  2982. i += 1
  2983. cmd = filtered[0] if filtered else "all"
  2984. if not keyword:
  2985. keyword = filtered[1] if len(filtered) > 1 else "矿泉水"
  2986. # 搜索词 = 品牌+药品名+规格 合起来(美团同款:分开配置,搜索时合并)
  2987. search_key = (brand + keyword + spec_raw).strip() or keyword
  2988. print(f"品牌: {brand or '(无)'} | 药品名: {keyword} | 规格: {spec_list or '(不限)'} | 搜索词: {search_key}")
  2989. print("设备连接中...")
  2990. device_id = _find_device(device_id)
  2991. print(f"设备: {device_id}")
  2992. ex = SafeExecutor(device_id)
  2993. if cmd in ("all", "step1"):
  2994. ok = step1_open_app(ex)
  2995. if not ok:
  2996. sys.exit(1)
  2997. if cmd in ("all", "step2"):
  2998. ok = step2_search(ex, search_key)
  2999. if not ok:
  3000. sys.exit(1)
  3001. if cmd in ("all", "step3"):
  3002. task = {"spec_list": spec_list, "product_specs": spec_raw}
  3003. visited = step3_swipe_and_enter(ex, keyword, brand, task)
  3004. print(f"\n最终访问: {visited}")
  3005. sys.exit(0)
  3006. if cmd in ("all", "step4"):
  3007. # step4 需要先跑完 step3 获取所有商品标题,单独跑时需要手动传标题
  3008. title = keyword
  3009. link = step4_parse_qr(ex, title)
  3010. print(f"\n链接: {link}")
  3011. sys.exit(0)
  3012. sys.exit(0)