search_auto_websites.py 34 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795796797798799800801802803804805806807808809810811812813814815816817818819820821822823824825826827828829830831832833834835836837838839840841842843844845846847848849850851852853854855856857858859860861862863864865866867868869870871872873874875876877878879880881882883884885886887888889890891892893894895896897898899900901902
  1. """
  2. Collect Morocco auto-dealer leads from vertical auto sites and business directories.
  3. This script uses an already-open AdsPower browser profile, searches six approved
  4. source websites, deep-scrapes candidate/company pages, follows public merchant
  5. website contact pages for emails, and writes results to a dedicated workbook
  6. sheet. It intentionally does not use Moteur.
  7. """
  8. import argparse
  9. import json
  10. import random
  11. import re
  12. import shutil
  13. import sys
  14. import time
  15. from dataclasses import dataclass, field
  16. from datetime import datetime, timezone
  17. from pathlib import Path
  18. import sys
  19. sys.path.append(str(Path(__file__).resolve().parents[1]))
  20. from common.artifact_manager import resolve_artifact_path, create_backup_once
  21. from typing import Any, Dict, Iterable, List, Optional, Set, Tuple
  22. from urllib.parse import quote_plus, urljoin, urlparse
  23. import requests
  24. from openpyxl import Workbook, load_workbook
  25. from playwright.sync_api import BrowserContext, Page, sync_playwright
  26. try:
  27. from . import discovery_common as dc
  28. from ..common import resolve_workbook_path
  29. except ImportError:
  30. sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
  31. from scraper import discovery_common as dc
  32. from common import resolve_workbook_path
  33. if hasattr(sys.stdout, "reconfigure"):
  34. sys.stdout.reconfigure(encoding="utf-8")
  35. if hasattr(sys.stderr, "reconfigure"):
  36. sys.stderr.reconfigure(encoding="utf-8")
  37. DEFAULT_EXCEL_DIR = Path.cwd()
  38. DEFAULT_SHEET = "汽车网站精选线索"
  39. DEFAULT_COUNTRY = "摩洛哥"
  40. DEFAULT_CITY = "摩洛哥全国"
  41. DEFAULT_ADSPOWER_URL = "http://127.0.0.1:50325"
  42. HEADERS = [
  43. "序号",
  44. "客户姓名/公司",
  45. "国家",
  46. "城市",
  47. "客户类型",
  48. "主页/链接",
  49. "来源网站",
  50. "联系人",
  51. "职位",
  52. "电话/WhatsApp",
  53. "邮箱",
  54. "主营业务",
  55. "建联状态",
  56. "下次跟进",
  57. "备注",
  58. ]
  59. CONTACT_LINK_HINTS = {
  60. "contact",
  61. "contactez",
  62. "nous contacter",
  63. "a propos",
  64. "à propos",
  65. "apropos",
  66. "qui sommes",
  67. "mentions",
  68. "legal",
  69. "devis",
  70. }
  71. HIGH_VALUE_TERMS = {
  72. "importation",
  73. "importateur",
  74. "importateurs",
  75. "véhicules neufs",
  76. "vehicules neufs",
  77. "voitures neuves",
  78. "professionnel",
  79. "professionnels",
  80. "concessionnaire",
  81. "concessionnaires",
  82. "distributeur",
  83. "distributeurs",
  84. "showroom",
  85. "stock",
  86. "parc auto",
  87. "location",
  88. "lld",
  89. "leasing",
  90. "flotte",
  91. "fleet",
  92. "utilitaire",
  93. "utilitaires",
  94. "camionnette",
  95. "fourgon",
  96. "pick-up",
  97. "pickup",
  98. "mpv",
  99. "minibus",
  100. "économique",
  101. "economique",
  102. }
  103. LOW_VALUE_TERMS = {
  104. "garage réparation",
  105. "garage reparation",
  106. "diagnostic",
  107. "pièces détachées",
  108. "pieces detachees",
  109. "assurance",
  110. "lavage",
  111. "car wash",
  112. "pare-brise",
  113. "immobilier",
  114. "emploi",
  115. }
  116. @dataclass
  117. class SourceConfig:
  118. name: str
  119. domains: Tuple[str, ...]
  120. queries: List[str]
  121. start_urls: List[str] = field(default_factory=list)
  122. SOURCES = [
  123. SourceConfig(
  124. name="OtoMoto.ma",
  125. domains=("otomoto.ma",),
  126. start_urls=["https://otomoto.ma/guide/professionnels", "https://otomoto.ma/"],
  127. queries=[
  128. "site:otomoto.ma Maroc professionnel concessionnaire automobile",
  129. "site:otomoto.ma Maroc revendeur automobile professionnel",
  130. "site:otomoto.ma Maroc importateur voiture professionnel",
  131. ],
  132. ),
  133. SourceConfig(
  134. name="Wandaloo",
  135. domains=("wandaloo.com",),
  136. start_urls=["https://www.wandaloo.com/neuf/maroc/concessionnaire.html"],
  137. queries=[
  138. "site:wandaloo.com/neuf/maroc concessionnaire distributeur automobile Maroc",
  139. "site:wandaloo.com/neuf/maroc importateur showroom Maroc",
  140. "site:wandaloo.com/neuf/maroc utilitaire concessionnaire Maroc",
  141. ],
  142. ),
  143. SourceConfig(
  144. name="Kerix",
  145. domains=("kerix.net",),
  146. start_urls=["https://www.kerix.net/fr/annuaire-entreprise/automobiles.html"],
  147. queries=[
  148. "site:kerix.net Maroc automobiles importation concessionnaire",
  149. "site:kerix.net Maroc concessionnaires régionaux automobiles",
  150. "site:kerix.net Maroc vehicules utilitaires importateur",
  151. ],
  152. ),
  153. SourceConfig(
  154. name="Kompass",
  155. domains=("kompass.com",),
  156. start_urls=["https://ma.kompass.com/y/importer/a/vehicules-utilitaires/66340/"],
  157. queries=[
  158. "site:ma.kompass.com Maroc importateur véhicules utilitaires",
  159. "site:ma.kompass.com Maroc concessionnaire automobile",
  160. "site:ma.kompass.com Maroc distributeur véhicules automobiles",
  161. ],
  162. ),
  163. SourceConfig(
  164. name="Maroc Annuaire",
  165. domains=("marocannuaire.org",),
  166. start_urls=["https://marocannuaire.org/Annuaire/Activite.php?activite=Automobile+%28Concessionnaires%29"],
  167. queries=[
  168. "site:marocannuaire.org Automobile Concessionnaires Maroc email",
  169. "site:marocannuaire.org importateur automobile Maroc email",
  170. "site:marocannuaire.org location voitures Maroc entreprise",
  171. ],
  172. ),
  173. SourceConfig(
  174. name="Telecontact",
  175. domains=("telecontact.ma",),
  176. start_urls=["https://www.telecontact.ma/villes/automobiles-agents-concessionnaires.php"],
  177. queries=[
  178. "site:telecontact.ma automobiles agents concessionnaires Maroc",
  179. "site:telecontact.ma concessionnaire automobile Casablanca Maroc",
  180. "site:telecontact.ma véhicules utilitaires concessionnaire Maroc",
  181. ],
  182. ),
  183. ]
  184. def clean(value: Any) -> str:
  185. return re.sub(r"\s+", " ", str(value or "")).strip()
  186. def is_source_url(url: str, source: SourceConfig) -> bool:
  187. host = urlparse(str(url or "")).netloc.casefold()
  188. return any(domain in host for domain in source.domains)
  189. def is_html_candidate(url: str) -> bool:
  190. if not url.startswith(("http://", "https://")):
  191. return False
  192. lowered = url.casefold()
  193. blocked_parts = [
  194. "facebook.com",
  195. "instagram.com",
  196. "linkedin.com",
  197. "youtube.com",
  198. "wa.me",
  199. "whatsapp",
  200. "google.",
  201. "bing.com",
  202. "/login",
  203. "/signup",
  204. "/privacy",
  205. ]
  206. if any(part in lowered for part in blocked_parts):
  207. return False
  208. if re.search(r"\.(pdf|jpg|jpeg|png|gif|webp|zip|rar)(?:$|\?)", lowered):
  209. return False
  210. return True
  211. def find_workbook(base_dir: Path = DEFAULT_EXCEL_DIR) -> Path:
  212. candidates: List[Path] = []
  213. for path in base_dir.glob("*.xlsx"):
  214. if path.name.startswith("~$") or "_backup_" in path.name or "_with_" in path.name:
  215. continue
  216. try:
  217. workbook = load_workbook(path, read_only=True, data_only=False)
  218. if any(sheet in workbook.sheetnames for sheet in ["Facebook", "Google Maps", "LinkedIn", "本地汽车网站"]):
  219. candidates.append(path)
  220. workbook.close()
  221. except Exception:
  222. continue
  223. if not candidates:
  224. raise FileNotFoundError("未找到摩洛哥客户建联表")
  225. return sorted(candidates, key=lambda p: p.stat().st_mtime, reverse=True)[0]
  226. def get_active_profile(base_url: str) -> Tuple[str, str]:
  227. resp = requests.get(f"{base_url.rstrip('/')}/api/v1/browser/local-active", timeout=10)
  228. resp.raise_for_status()
  229. data = resp.json()
  230. active = ((data.get("data") or {}).get("list") or [])
  231. if not active:
  232. raise RuntimeError("未发现已打开的 AdsPower 浏览器配置")
  233. item = active[0]
  234. ws = (item.get("ws") or {}).get("puppeteer") or (item.get("ws") or {}).get("selenium")
  235. if not ws:
  236. raise RuntimeError(f"已打开配置没有 ws endpoint: {item}")
  237. return item.get("user_id", ""), ws
  238. def dismiss_dialogs(page: Page) -> None:
  239. labels = [
  240. "Accept all",
  241. "Tout accepter",
  242. "J'accepte",
  243. "Accepter",
  244. "Reject all",
  245. "Plus tard",
  246. "Fermer",
  247. "Close",
  248. "OK",
  249. ]
  250. for label in labels:
  251. try:
  252. page.get_by_text(label, exact=False).first.click(timeout=800)
  253. time.sleep(0.3)
  254. except Exception:
  255. pass
  256. def safe_body_text(page: Page, timeout: int = 12000) -> str:
  257. try:
  258. return page.locator("body").inner_text(timeout=timeout)
  259. except Exception:
  260. return ""
  261. def page_links(page: Page) -> List[Dict[str, str]]:
  262. try:
  263. return page.evaluate(
  264. """
  265. () => Array.from(document.querySelectorAll('a[href]')).map((a) => ({
  266. href: a.href || '',
  267. raw: a.getAttribute('href') || '',
  268. text: (a.innerText || '').trim(),
  269. aria: (a.getAttribute('aria-label') || '').trim()
  270. }))
  271. """
  272. )
  273. except Exception:
  274. return []
  275. def collect_bing_results(page: Page, source: SourceConfig, query: str, limit: int) -> List[Dict[str, Any]]:
  276. url = f"https://www.bing.com/search?q={quote_plus(query)}"
  277. print(f"搜索 {source.name}: {query}", flush=True)
  278. page.goto(url, wait_until="domcontentloaded", timeout=60000)
  279. time.sleep(random.uniform(2.5, 4.0))
  280. dismiss_dialogs(page)
  281. for _ in range(2):
  282. page.mouse.wheel(0, 1000)
  283. time.sleep(random.uniform(0.7, 1.2))
  284. links = page.evaluate(
  285. """
  286. () => Array.from(document.querySelectorAll('li.b_algo h2 a, a[href]')).map((a) => ({
  287. href: a.href || '',
  288. text: (a.innerText || a.getAttribute('aria-label') || '').trim()
  289. }))
  290. """
  291. )
  292. results: List[Dict[str, Any]] = []
  293. seen: Set[str] = set()
  294. for item in links:
  295. href = item.get("href", "")
  296. if not is_source_url(href, source) or not is_html_candidate(href):
  297. continue
  298. key = dc.normalize_url(href)
  299. if not key or key in seen:
  300. continue
  301. seen.add(key)
  302. results.append({
  303. "url": href,
  304. "title": clean(item.get("text")),
  305. "source": source.name,
  306. "source_queries": [query],
  307. "collection_method": "bing",
  308. })
  309. if len(results) >= limit:
  310. break
  311. return results
  312. def collect_start_url_links(page: Page, source: SourceConfig, start_url: str, limit: int) -> List[Dict[str, Any]]:
  313. print(f"打开来源页 {source.name}: {start_url}", flush=True)
  314. page.goto(start_url, wait_until="domcontentloaded", timeout=60000)
  315. time.sleep(random.uniform(2.5, 4.0))
  316. dismiss_dialogs(page)
  317. for _ in range(3):
  318. page.mouse.wheel(0, 1200)
  319. time.sleep(random.uniform(0.7, 1.2))
  320. body = safe_body_text(page)[:3000]
  321. results: List[Dict[str, Any]] = [{
  322. "url": page.url,
  323. "title": clean(page.title()),
  324. "text_hint": body,
  325. "source": source.name,
  326. "source_queries": [start_url],
  327. "collection_method": "start_url",
  328. }]
  329. seen = {dc.normalize_url(page.url)}
  330. for item in page_links(page):
  331. href = urljoin(page.url, item.get("href", ""))
  332. if not is_source_url(href, source) or not is_html_candidate(href):
  333. continue
  334. label = clean(" ".join([item.get("text", ""), item.get("aria", "")]))
  335. if not label or len(label) > 120:
  336. continue
  337. key = dc.normalize_url(href)
  338. if not key or key in seen:
  339. continue
  340. if not any(term in f"{label} {href}".casefold() for term in ["concession", "auto", "voiture", "garage", "import", "vehicule", "véhicule", "dealer", "annuaire", "societe", "entreprise"]):
  341. continue
  342. seen.add(key)
  343. results.append({
  344. "url": href,
  345. "title": label,
  346. "source": source.name,
  347. "source_queries": [start_url],
  348. "collection_method": "start_url_link",
  349. })
  350. if len(results) >= limit:
  351. break
  352. return results
  353. def merge_candidates(candidates: Iterable[Dict[str, Any]]) -> List[Dict[str, Any]]:
  354. merged: Dict[str, Dict[str, Any]] = {}
  355. for candidate in candidates:
  356. key = dc.normalize_url(candidate.get("url", "")) or dc.normalized_company_key(candidate.get("title"))
  357. if not key:
  358. continue
  359. if key not in merged:
  360. merged[key] = candidate
  361. continue
  362. existing = merged[key]
  363. existing["source_queries"] = sorted(set(existing.get("source_queries", []) + candidate.get("source_queries", [])))
  364. if candidate.get("text_hint"):
  365. existing["text_hint"] = clean(" ".join([existing.get("text_hint", ""), candidate.get("text_hint", "")]))[:4000]
  366. return list(merged.values())
  367. def extract_company_name(page: Page, candidate: Dict[str, Any], body: str) -> str:
  368. for selector in ["h1", "h2"]:
  369. try:
  370. value = clean(page.locator(selector).first.inner_text(timeout=1500))
  371. if value and len(value) <= 100:
  372. return re.sub(r"\s*[-|].*$", "", value).strip()
  373. except Exception:
  374. pass
  375. title = clean(page.title() or candidate.get("title", ""))
  376. title = re.sub(r"\s*[-|]\s*(Kerix|Kompass|Telecontact|Wandaloo|OtoMoto|Maroc Annuaire).*$", "", title, flags=re.I)
  377. return title[:100] or clean(candidate.get("title", ""))[:100]
  378. def extract_external_websites(current_url: str, links: List[Dict[str, str]], source: SourceConfig) -> List[str]:
  379. websites: List[str] = []
  380. source_hosts = set(source.domains)
  381. for item in links:
  382. href = urljoin(current_url, item.get("href") or item.get("raw") or "")
  383. if not dc.is_external_business_url(href):
  384. continue
  385. host = urlparse(href).netloc.casefold().removeprefix("www.")
  386. if any(domain in host for domain in source_hosts):
  387. continue
  388. label = clean(" ".join([item.get("text", ""), item.get("aria", ""), item.get("raw", "")])).casefold()
  389. if any(term in label for term in ["site web", "website", "web", "www", "visiter", "voir le site", "site internet"]) or len(websites) < 2:
  390. normalized = href.split("#")[0]
  391. if normalized not in websites:
  392. websites.append(normalized)
  393. return websites[:3]
  394. def same_host(url: str, base_url: str) -> bool:
  395. return urlparse(url).netloc.casefold().removeprefix("www.") == urlparse(base_url).netloc.casefold().removeprefix("www.")
  396. def scrape_website_email(context: BrowserContext, website_url: str, max_pages: int = 4) -> Dict[str, Any]:
  397. if not website_url or not dc.is_external_business_url(website_url):
  398. return {"email": "", "emails": [], "sources": [], "checked_urls": []}
  399. checked: List[str] = []
  400. queue: List[str] = [website_url]
  401. emails: List[str] = []
  402. sources: List[str] = []
  403. page = context.new_page()
  404. page.set_viewport_size({"width": 1280, "height": 850})
  405. try:
  406. while queue and len(checked) < max_pages:
  407. url = dc.normalize_website_url(queue.pop(0))
  408. if url in checked or (checked and not same_host(url, website_url)):
  409. continue
  410. checked.append(url)
  411. try:
  412. print(f" 检查官网邮箱: {url}", flush=True)
  413. page.goto(url, wait_until="domcontentloaded", timeout=35000)
  414. time.sleep(random.uniform(1.2, 2.2))
  415. dismiss_dialogs(page)
  416. body = safe_body_text(page)
  417. links = page_links(page)
  418. mailto_text = " ".join(item.get("raw", "") for item in links if item.get("raw", "").casefold().startswith("mailto:"))
  419. for email in dc.extract_emails(body + " " + mailto_text):
  420. if email not in emails:
  421. emails.append(email)
  422. sources.append(url)
  423. if emails:
  424. break
  425. for item in links:
  426. label = clean(" ".join([item.get("raw", ""), item.get("text", ""), item.get("aria", "")])).casefold()
  427. if not any(hint in label for hint in CONTACT_LINK_HINTS):
  428. continue
  429. href = dc.normalize_website_url(item.get("href") or item.get("raw") or "", page.url)
  430. if href and href not in checked and href not in queue and same_host(href, website_url):
  431. queue.append(href)
  432. except Exception:
  433. continue
  434. finally:
  435. try:
  436. page.close()
  437. except Exception:
  438. pass
  439. return {"email": emails[0] if emails else "", "emails": emails, "sources": sources, "checked_urls": checked}
  440. def score_record(name: str, url: str, body: str, source: str) -> Tuple[int, List[str], List[str]]:
  441. combined = f"{name} {url} {body}".casefold()
  442. score = 0
  443. reasons: List[str] = []
  444. risks: List[str] = []
  445. sales_terms = [term for term in ["véhicules neufs", "vehicules neufs", "voitures neuves", "concessionnaire", "showroom", "vente automobile", "stock", "parc auto", "professionnel"] if term in combined]
  446. import_terms = [term for term in ["importation", "importateur", "importateurs", "distributeur", "distributeurs", "réseau", "reseau", "points de vente", "succursales", "agences"] if term in combined]
  447. multibrand_terms = [term for term in ["multimarque", "multi-brand", "multi brand", "plusieurs marques", "marques multiples"] if term in combined]
  448. rental_terms = [term for term in ["location", "lld", "leasing", "flotte", "fleet"] if term in combined]
  449. service_terms = [term for term in ["garage réparation", "garage reparation", "diagnostic", "pièces détachées", "pieces detachees", "pneus", "tires", "lavage", "car wash", "pare-brise", "assurance", "immobilier", "emploi"] if term in combined]
  450. has_actual_channel = bool(sales_terms or import_terms or multibrand_terms)
  451. if sales_terms:
  452. score += 3
  453. reasons.append("实际新整车销售/showroom/库存信号: " + ", ".join(sales_terms[:5]))
  454. if import_terms:
  455. score += 3
  456. reasons.append("进口/分销/网络能力信号: " + ", ".join(import_terms[:5]))
  457. if multibrand_terms:
  458. score += 2
  459. reasons.append("多品牌经营信号: " + ", ".join(multibrand_terms[:4]))
  460. if source in {"Kerix", "Kompass", "Maroc Annuaire", "Telecontact"}:
  461. score += 1
  462. reasons.append("annuaire professionnel")
  463. if re.search(r"(?:\+212|0)\s?\d[\d\s.-]{6,}\d", body):
  464. score += 2
  465. reasons.append("可建联入口: 电话/WhatsApp")
  466. if dc.extract_emails(body):
  467. score += 2
  468. reasons.append("可建联入口: email public")
  469. if any(term in combined for term in ["facebook.com", "linkedin.com", "whatsapp", "wa.me", "contact"]):
  470. score += 1
  471. reasons.append("可建联入口: social/contact link")
  472. if any(term in combined for term in ["stock", "annonces", "véhicules disponibles", "vehicules disponibles", "parc"]):
  473. score += 1
  474. reasons.append("stock véhicules")
  475. if service_terms:
  476. score -= 6
  477. risks.append("纯维修/配件/轮胎/服务类,不纳入汽车渠道合作伙伴: " + ", ".join(service_terms[:5]))
  478. if rental_terms and not has_actual_channel:
  479. score += 1
  480. risks.append("纯租赁/车队线索,转入批量采购与运营客户/汽车租赁公司: " + ", ".join(rental_terms[:4]))
  481. new_vehicle_needs_review = False
  482. if not has_actual_channel and not rental_terms:
  483. score -= 3
  484. risks.append("未发现新整车销售、进口、分销或 showroom 证据")
  485. new_vehicle_needs_review = True
  486. if new_vehicle_needs_review and dc.should_request_manual_review(score, "preview_only", reasons):
  487. dc.add_manual_review_flag(risks, "new_vehicle")
  488. return score, reasons, risks
  489. def summarize_record(
  490. source: str,
  491. score: int,
  492. reasons: List[str],
  493. risks: List[str],
  494. page_url: str,
  495. website: str,
  496. email_source: str,
  497. queries: List[str],
  498. ) -> str:
  499. capability = "批量采购能力判断:"
  500. reason_text = "、".join(reasons[:8]) if reasons else "证据不足,低优先级;如无更多渠道价值证据应跳过"
  501. if any(term in reason_text for term in ["multi-site", "stock", "importation", "distributeur", "utilitaire", "flotte", "location"]):
  502. capability += "有批量/车队/分销潜力"
  503. else:
  504. capability += "待确认"
  505. parts = [
  506. f"来源线索:{source};详情页 {page_url}",
  507. f"主营业务判断:{reason_text}",
  508. capability,
  509. f"库存/门店/租赁/进口证据:{reason_text}",
  510. f"联系方式证据:官网 {website}" if website else "联系方式证据:未发现独立官网",
  511. f"邮箱来源:{email_source}" if email_source else "邮箱来源:未发现公开邮箱",
  512. f"来源搜索词:{', '.join(queries[:3])}",
  513. f"评分:{score}",
  514. ]
  515. if risks:
  516. parts.append("风险/待确认项:" + "、".join(risks[:5]))
  517. return " | ".join(parts)[:1400]
  518. def deep_scrape_candidate(page: Page, context: BrowserContext, candidate: Dict[str, Any], source: SourceConfig) -> Optional[Dict[str, Any]]:
  519. url = candidate["url"]
  520. print(f"深采 {source.name}: {url}", flush=True)
  521. try:
  522. page.goto(url, wait_until="domcontentloaded", timeout=60000)
  523. time.sleep(random.uniform(2.0, 3.4))
  524. dismiss_dialogs(page)
  525. body = safe_body_text(page)
  526. except Exception as exc:
  527. return {
  528. "_skip": True,
  529. "_skip_reason": f"打开失败: {exc}",
  530. "url": url,
  531. "source": source.name,
  532. }
  533. if not body or len(body) < 80:
  534. return {"_skip": True, "_skip_reason": "页面正文过短", "url": url, "source": source.name}
  535. name = extract_company_name(page, candidate, body)
  536. if not name or len(name) < 2:
  537. return {"_skip": True, "_skip_reason": "未识别公司名", "url": url, "source": source.name}
  538. is_oem, brand = dc.looks_like_oem_local_branch(name, url, body)
  539. if is_oem:
  540. return {"_skip": True, "_skip_reason": f"疑似品牌官方国家页: {brand}", "url": url, "source": source.name}
  541. score, reasons, risks = score_record(name, url, body, source.name)
  542. if score < 3:
  543. return {"_skip": True, "_skip_reason": f"评分过低: {score}", "url": url, "source": source.name}
  544. links = page_links(page)
  545. websites = extract_external_websites(page.url, links, source)
  546. website = websites[0] if websites else ""
  547. page_emails = dc.extract_emails(body + " " + " ".join(item.get("raw", "") for item in links))
  548. website_email_result = scrape_website_email(context, website) if website else {"email": "", "emails": [], "sources": [], "checked_urls": []}
  549. email = website_email_result.get("email") or (page_emails[0] if page_emails else "")
  550. email_source = ""
  551. if website_email_result.get("email"):
  552. email_source = "官网公开页面 " + ", ".join(website_email_result.get("sources", [])[:2])
  553. elif email:
  554. email_source = "来源网站页面公开文本"
  555. combined_text = f"{name}\n{body}\n{' '.join(candidate.get('source_queries', []))}"
  556. phone = dc.extract_phone(body)
  557. city = dc.extract_city(combined_text, fallback=DEFAULT_CITY)
  558. business = dc.summarize_business(combined_text)
  559. customer_type = dc.classify_customer_type(combined_text)
  560. note = summarize_record(
  561. source=source.name,
  562. score=score,
  563. reasons=reasons,
  564. risks=risks,
  565. page_url=page.url,
  566. website=website,
  567. email_source=email_source,
  568. queries=candidate.get("source_queries", []),
  569. )
  570. return {
  571. "序号": "",
  572. "客户姓名/公司": name,
  573. "国家": DEFAULT_COUNTRY,
  574. "城市": city,
  575. "客户类型": customer_type,
  576. "主页/链接": page.url,
  577. "来源网站": source.name,
  578. "联系人": "",
  579. "职位": "",
  580. "电话/WhatsApp": phone,
  581. "邮箱": email,
  582. "主营业务": business,
  583. "建联状态": "未联系",
  584. "下次跟进": "",
  585. "备注": note,
  586. "_score": score,
  587. "_score_reasons": reasons,
  588. "_risk_flags": risks,
  589. "_website": website,
  590. "_website_email_result": website_email_result,
  591. }
  592. def split_multi(value: str) -> List[str]:
  593. return [clean(part) for part in re.split(r"[;;|]+", str(value or "")) if clean(part)]
  594. def append_unique_text(old: str, addition: str, separator: str = " | ") -> str:
  595. old = clean(old)
  596. addition = clean(addition)
  597. if not addition:
  598. return old
  599. if not old:
  600. return addition
  601. if addition in old:
  602. return old
  603. return (old + separator + addition)[:3000]
  604. def merge_sources(old: str, addition: str) -> str:
  605. values = []
  606. for item in split_multi(old) + split_multi(addition):
  607. if item and item not in values:
  608. values.append(item)
  609. return ";".join(values)
  610. def duplicate_key(record: Dict[str, Any]) -> List[str]:
  611. keys = []
  612. website = record.get("_website") or ""
  613. if website:
  614. keys.append("website:" + dc.normalize_url(website))
  615. name = record.get("客户姓名/公司")
  616. if name:
  617. keys.append("name:" + dc.normalized_company_key(name))
  618. for field in ["电话/WhatsApp", "邮箱", "主页/链接"]:
  619. value = clean(record.get(field))
  620. if value:
  621. normalized = dc.normalize_url(value) if field == "主页/链接" else value.casefold()
  622. keys.append(f"{field}:{normalized}")
  623. return [key for key in keys if key and not key.endswith(":")]
  624. def prepare_sheet(workbook: Workbook, sheet_name: str):
  625. if sheet_name in workbook.sheetnames:
  626. ws = workbook[sheet_name]
  627. for idx, header in enumerate(HEADERS, start=1):
  628. ws.cell(row=1, column=idx).value = header
  629. return ws
  630. ws = workbook.create_sheet(sheet_name)
  631. ws.append(HEADERS)
  632. return ws
  633. def read_existing_sheet_index(ws) -> Dict[str, int]:
  634. index: Dict[str, int] = {}
  635. header_map = {clean(ws.cell(1, col).value): col for col in range(1, ws.max_column + 1)}
  636. for row in range(2, ws.max_row + 1):
  637. record = {header: ws.cell(row, col).value for header, col in header_map.items()}
  638. for key in duplicate_key(record):
  639. index[key] = row
  640. return index
  641. def write_records_to_workbook(excel_path: Path, sheet_name: str, records: List[Dict[str, Any]]) -> Dict[str, Any]:
  642. locks = sorted(p.name for p in excel_path.parent.glob("~$*.xlsx"))
  643. if locks:
  644. raise PermissionError("检测到 Excel 临时锁文件: " + ", ".join(locks))
  645. backup_path = create_backup_once(excel_path, purpose="auto_websites", run_id="auto_websites")
  646. wb = load_workbook(excel_path)
  647. ws = prepare_sheet(wb, sheet_name)
  648. header_map = {header: idx for idx, header in enumerate(HEADERS, start=1)}
  649. existing_index = read_existing_sheet_index(ws)
  650. appended = 0
  651. merged = 0
  652. no_email = 0
  653. for record in records:
  654. if not record.get("邮箱"):
  655. no_email += 1
  656. row = None
  657. for key in duplicate_key(record):
  658. if key in existing_index:
  659. row = existing_index[key]
  660. break
  661. if row:
  662. merged += 1
  663. for field in ["来源网站", "备注"]:
  664. col = header_map[field]
  665. if field == "来源网站":
  666. ws.cell(row, col).value = merge_sources(ws.cell(row, col).value, record.get(field, ""))
  667. else:
  668. ws.cell(row, col).value = append_unique_text(ws.cell(row, col).value, record.get(field, ""))
  669. for field in ["电话/WhatsApp", "邮箱", "主页/链接", "主营业务", "客户类型", "城市"]:
  670. col = header_map[field]
  671. if not clean(ws.cell(row, col).value) and clean(record.get(field)):
  672. ws.cell(row, col).value = record.get(field)
  673. continue
  674. appended += 1
  675. row = ws.max_row + 1
  676. record["序号"] = row - 1
  677. for header, col in header_map.items():
  678. ws.cell(row, col).value = record.get(header, "")
  679. for key in duplicate_key(record):
  680. existing_index[key] = row
  681. wb.save(excel_path)
  682. wb.close()
  683. return {
  684. "backup_path": str(backup_path),
  685. "appended": appended,
  686. "merged": merged,
  687. "no_email": no_email,
  688. "sheet": sheet_name,
  689. "workbook": str(excel_path),
  690. }
  691. def collect_records(
  692. profile_id: str,
  693. ws_endpoint: str,
  694. max_per_source: int,
  695. max_candidates_per_source: int,
  696. queries_per_source: int,
  697. start_urls_per_source: int,
  698. ) -> Dict[str, Any]:
  699. playwright = sync_playwright().start()
  700. browser = playwright.chromium.connect_over_cdp(ws_endpoint)
  701. context = browser.contexts[0] if browser.contexts else browser.new_context()
  702. page = context.new_page()
  703. page.set_viewport_size({"width": 1366, "height": 850})
  704. source_logs: List[Dict[str, Any]] = []
  705. records: List[Dict[str, Any]] = []
  706. skipped: List[Dict[str, Any]] = []
  707. try:
  708. for source in SOURCES:
  709. raw: List[Dict[str, Any]] = []
  710. errors: List[str] = []
  711. try:
  712. for start_url in source.start_urls[:start_urls_per_source]:
  713. raw.extend(collect_start_url_links(page, source, start_url, max_candidates_per_source))
  714. time.sleep(random.uniform(1.2, 2.0))
  715. for query in source.queries[:queries_per_source]:
  716. raw.extend(collect_bing_results(page, source, query, max_candidates_per_source))
  717. time.sleep(random.uniform(1.5, 2.5))
  718. except Exception as exc:
  719. errors.append(str(exc))
  720. candidates = merge_candidates(raw)[:max_candidates_per_source]
  721. source_records: List[Dict[str, Any]] = []
  722. for candidate in candidates:
  723. if len(source_records) >= max_per_source:
  724. break
  725. result = deep_scrape_candidate(page, context, candidate, source)
  726. if not result:
  727. continue
  728. if result.get("_skip"):
  729. skipped.append(result)
  730. continue
  731. source_records.append(result)
  732. records.append(result)
  733. time.sleep(random.uniform(1.4, 2.4))
  734. source_logs.append({
  735. "source": source.name,
  736. "raw_candidates": len(raw),
  737. "merged_candidates": len(candidates),
  738. "records": len(source_records),
  739. "errors": errors,
  740. })
  741. finally:
  742. try:
  743. page.close()
  744. except Exception:
  745. pass
  746. playwright.stop()
  747. return {
  748. "generated_at": datetime.now(timezone.utc).isoformat(),
  749. "profile_id": profile_id,
  750. "records": records,
  751. "skipped": skipped,
  752. "source_logs": source_logs,
  753. }
  754. def main() -> None:
  755. parser = argparse.ArgumentParser(description="Collect Morocco auto website leads via AdsPower")
  756. parser.add_argument("--profile-id", default="", help="AdsPower profile ID; defaults to current local active browser")
  757. parser.add_argument("--ads-power-url", default=DEFAULT_ADSPOWER_URL)
  758. parser.add_argument("--excel", default="", help="Workbook path; defaults to detected Morocco outreach workbook")
  759. parser.add_argument("--sheet", default=DEFAULT_SHEET)
  760. parser.add_argument("--max-per-source", type=int, default=5)
  761. parser.add_argument("--max-candidates-per-source", type=int, default=12)
  762. parser.add_argument("--queries-per-source", type=int, default=3)
  763. parser.add_argument("--start-urls-per-source", type=int, default=1)
  764. parser.add_argument("--run-id", default="", help="Run ID used for runs/YYYYMMDD/<run_id>/ artifacts.")
  765. parser.add_argument("--output", default="auto_website_leads.json")
  766. parser.add_argument("--write-excel", action="store_true")
  767. args = parser.parse_args()
  768. if args.profile_id:
  769. ws_endpoint = dc.get_active_ws_endpoint(args.ads_power_url, args.profile_id)
  770. profile_id = args.profile_id
  771. else:
  772. profile_id, ws_endpoint = get_active_profile(args.ads_power_url)
  773. workbook_info = resolve_workbook_path(args.excel, create_from_template=args.write_excel)
  774. excel_path = workbook_info.get("path")
  775. if excel_path and (args.write_excel or workbook_info.get("source") != "missing"):
  776. print(f"Workbook resolved: {excel_path} ({workbook_info['source']})", flush=True)
  777. result = collect_records(
  778. profile_id=profile_id,
  779. ws_endpoint=ws_endpoint,
  780. max_per_source=args.max_per_source,
  781. max_candidates_per_source=args.max_candidates_per_source,
  782. queries_per_source=max(0, args.queries_per_source),
  783. start_urls_per_source=max(0, args.start_urls_per_source),
  784. )
  785. result["summary"] = {
  786. "record_count": len(result["records"]),
  787. "records_with_email": len([r for r in result["records"] if r.get("邮箱")]),
  788. "sources": {log["source"]: log["records"] for log in result["source_logs"]},
  789. "skipped_count": len(result["skipped"]),
  790. "excel": str(excel_path) if excel_path else "",
  791. "sheet": args.sheet,
  792. }
  793. output_path = resolve_artifact_path(args.output, kind="scraper_preview", default_name=Path(args.output).name, run_id=args.run_id or None)
  794. output_path.parent.mkdir(parents=True, exist_ok=True)
  795. output_path.write_text(json.dumps(result, ensure_ascii=False, indent=2), encoding="utf-8")
  796. print(f"已保存采集结果: {output_path}", flush=True)
  797. if args.write_excel:
  798. if not excel_path:
  799. raise SystemExit("使用 --write-excel 时必须提供 --excel 工作簿路径,或在当前目录放置可识别的建联表。")
  800. write_result = write_records_to_workbook(excel_path, args.sheet, result["records"])
  801. result["write_result"] = write_result
  802. output_path.write_text(json.dumps(result, ensure_ascii=False, indent=2), encoding="utf-8")
  803. print("Excel 写入结果: " + json.dumps(write_result, ensure_ascii=False), flush=True)
  804. else:
  805. print("预览模式:未写入 Excel。", flush=True)
  806. if __name__ == "__main__":
  807. main()