search_linkedin.py 26 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650
  1. """
  2. LinkedIn Morocco dealer discovery.
  3. Preview-first workflow:
  4. - Connect to an already-open AdsPower browser by profile ID.
  5. - Search LinkedIn company results with Morocco-wide dealer/importer keywords.
  6. - Score and de-duplicate candidates before opening company pages.
  7. - Reject OEM local brand-country pages such as BYD Maroc or BMW Maroc.
  8. - Optionally deep-scrape company About pages and write records to the LinkedIn sheet.
  9. """
  10. import argparse
  11. import json
  12. import random
  13. import re
  14. import sys
  15. import time
  16. from datetime import datetime, timezone
  17. from pathlib import Path
  18. import sys
  19. sys.path.append(str(Path(__file__).resolve().parents[1]))
  20. from common.artifact_manager import resolve_artifact_path, create_backup_once
  21. from typing import Any, Dict, List, Optional, Set, Tuple
  22. from urllib.parse import quote_plus, urlparse
  23. if hasattr(sys.stdout, "reconfigure"):
  24. sys.stdout.reconfigure(encoding="utf-8")
  25. if hasattr(sys.stderr, "reconfigure"):
  26. sys.stderr.reconfigure(encoding="utf-8")
  27. import pandas as pd
  28. import requests
  29. from playwright.sync_api import Page, sync_playwright
  30. try:
  31. from ..common import append_records, resolve_workbook_path
  32. from . import discovery_common as dc
  33. except ImportError:
  34. sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
  35. from common import append_records, resolve_workbook_path
  36. from scraper import discovery_common as dc
  37. DEFAULT_EXCEL = ""
  38. DEFAULT_SHEET = "LinkedIn"
  39. DEFAULT_CITY_SCOPE = "摩洛哥全国"
  40. DEFAULT_KEYWORDS = [
  41. "showroom auto Maroc",
  42. "concessionnaire automobile Maroc",
  43. "concessionnaire multimarque Maroc",
  44. "distributeur automobile Maroc",
  45. "importateur automobile Maroc",
  46. "groupe automobile Maroc",
  47. "voiture occasion Maroc",
  48. "concessionnaire utilitaire Maroc",
  49. "La Continentale Auto Maroc",
  50. "Prince Auto Maroc",
  51. "Kifal Auto Maroc",
  52. "Bamotors Maroc",
  53. "SMEIA Maroc",
  54. "Sopriam Maroc",
  55. "Auto Hall Maroc",
  56. "Auto Nejma Maroc",
  57. "Centrale Automobile Chérifienne Maroc",
  58. ]
  59. MOROCCO_TERMS = {
  60. "maroc", "morocco", "casablanca", "rabat", "marrakech", "tanger",
  61. "fes", "fès", "agadir", "meknes", "oujda", "kenitra", "tetouan",
  62. "nador", "safi", "settat", "dar bouazza", "el jadida",
  63. }
  64. CHINA_BRAND_TERMS = {
  65. "baic", "byd", "changan", "chery", "dfsk", "dongfeng", "foton", "forland",
  66. "gac", "geely", "great wall", "haval", "jac", "jetour", "maxus", "mg",
  67. "omoda", "saic", "sitrak", "sinotruk", "shacman", "faw", "yutong",
  68. "king long", "golden dragon", "higer", "wuling",
  69. }
  70. NON_CHINA_BRAND_TERMS = {
  71. "audi", "bmw", "chevrolet", "citroen", "citroën", "dacia", "daf", "fiat",
  72. "ford", "hino", "honda", "hyundai", "isuzu", "iveco", "jeep", "kia", "man",
  73. "mazda", "mercedes", "mercedes-benz", "mitsubishi", "nissan", "opel", "peugeot",
  74. "renault", "scania", "seat", "skoda", "suzuki", "toyota", "volkswagen", "volvo",
  75. }
  76. OEM_BRANCH_BRANDS = CHINA_BRAND_TERMS | NON_CHINA_BRAND_TERMS
  77. INDEPENDENT_CHANNEL_CLUES = {
  78. "auto hall", "smaa", "smeia", "auto nejma", "la continentale", "prince auto",
  79. "kifal", "autochek", "bugshan", "bamotors", "sopriam", "cac",
  80. "centrale automobile", "univers motors", "m-automotiv", "m automotiv",
  81. }
  82. HIGH_INTENT_TERMS = {
  83. "concessionnaire", "dealer", "distributeur", "distribution", "importateur",
  84. "importation", "showroom", "groupe", "group", "automobile", "auto", "retail",
  85. "vente", "voiture", "occasion", "reprise", "multimarque", "service après-vente",
  86. "après-vente", "pieces", "pièces", "fleet", "flotte", "utilitaire", "camion",
  87. }
  88. LOW_VALUE_TERMS = {
  89. "aerospace", "aéronautique", "assurance", "insurance", "software", "marketing",
  90. "emailing", "real estate", "immobilier", "building materials", "construction materials",
  91. "diagnostic", "spare parts only", "pièces uniquement",
  92. }
  93. CITY_PATTERNS = [
  94. "Casablanca", "Rabat", "Marrakech", "Tanger", "Fes", "Fès", "Agadir", "Meknes",
  95. "Oujda", "Kenitra", "Tetouan", "Tétouan", "Nador", "Safi", "Settat",
  96. "Dar Bouazza", "El Jadida",
  97. ]
  98. def compact_text(value: str) -> str:
  99. return re.sub(r"[^a-z0-9]+", "", value.casefold())
  100. def normalize_linkedin_url(url: str) -> str:
  101. return dc.normalize_linkedin_url(url)
  102. def load_existing_links(excel_path: str, sheet_name: str) -> Set[str]:
  103. path = Path(excel_path)
  104. if not path.exists():
  105. return set()
  106. try:
  107. df = pd.read_excel(path, sheet_name=sheet_name)
  108. except Exception:
  109. return set()
  110. link_col = "linkin链接" if "linkin链接" in df.columns else "主页/链接"
  111. if link_col not in df.columns:
  112. return set()
  113. return {
  114. normalized
  115. for normalized in (normalize_linkedin_url(str(value).strip()) for value in df[link_col].tolist())
  116. if normalized
  117. }
  118. def load_blocklist_json(blocklist_path: Optional[str]) -> Set[str]:
  119. return dc.load_blocklist_json(blocklist_path, normalizer=dc.normalize_linkedin_url)
  120. def matched_terms(text: str, terms: Set[str]) -> List[str]:
  121. text_lower = text.casefold()
  122. return sorted(term for term in terms if term.casefold() in text_lower)
  123. def looks_like_oem_local_branch(name: str, url: str, text: str = "") -> Tuple[bool, str]:
  124. return dc.looks_like_oem_local_branch(name, url, text)
  125. def score_candidate(candidate: Dict[str, Any], existing_links: Set[str], existing_names: Optional[Set[str]] = None) -> Dict[str, Any]:
  126. name = candidate.get("name", "")
  127. href = candidate.get("href", "")
  128. text = candidate.get("text", "")
  129. queries = candidate.get("source_queries", [])
  130. entity_text = " ".join([name, href, text]).casefold()
  131. query_text = " ".join(queries).casefold()
  132. combined = " ".join([entity_text, query_text]).casefold()
  133. score = 0
  134. reasons: List[str] = []
  135. risks: List[str] = []
  136. recommended_action = "preview_only"
  137. normalized = normalize_linkedin_url(href)
  138. existing_names = existing_names or set()
  139. name_key = dc.normalized_company_key(name)
  140. if normalized in existing_links:
  141. return {
  142. **candidate,
  143. "score": -10,
  144. "score_reasons": ["Already exists in workbook"],
  145. "risk_flags": ["duplicate_existing_link"],
  146. "recommended_action": "skip_existing",
  147. }
  148. if name_key and name_key in existing_names:
  149. return {
  150. **candidate,
  151. "score": -9,
  152. "score_reasons": ["Company name already exists in workbook"],
  153. "risk_flags": ["duplicate_existing_name"],
  154. "recommended_action": "skip_existing",
  155. }
  156. is_oem, brand = looks_like_oem_local_branch(name, href, text)
  157. brand_terms = matched_terms(entity_text, OEM_BRANCH_BRANDS)
  158. if is_oem:
  159. score -= 8
  160. risks.append(f"疑似品牌当地分公司/官方页,排除: {brand}")
  161. recommended_action = "skip_brand_branch"
  162. ownership_needs_review = (not is_oem) and brand_terms and not any(clue in entity_text for clue in INDEPENDENT_CHANNEL_CLUES)
  163. morocco_terms = matched_terms(entity_text, MOROCCO_TERMS)
  164. if morocco_terms:
  165. score += 2
  166. reasons.append("Morocco signal: " + ", ".join(morocco_terms[:4]))
  167. intent_terms = matched_terms(entity_text, HIGH_INTENT_TERMS)
  168. if intent_terms:
  169. score += min(5, len(intent_terms))
  170. reasons.append("dealer/import/channel signal: " + ", ".join(intent_terms[:6]))
  171. china_terms = matched_terms(entity_text, CHINA_BRAND_TERMS)
  172. if china_terms and not is_oem:
  173. score += 2
  174. reasons.append("China-brand context inside dealer channel: " + ", ".join(china_terms[:4]))
  175. canonical_name = dc.canonical_dealer_name(name, text)
  176. if canonical_name and canonical_name != name:
  177. candidate = {**candidate, "canonical_name": canonical_name}
  178. reasons.append(f"mapped to independent dealer group: {canonical_name}")
  179. if any(clue in entity_text for clue in INDEPENDENT_CHANNEL_CLUES):
  180. score += 4
  181. reasons.append("known independent Moroccan auto channel")
  182. low_terms = matched_terms(entity_text, LOW_VALUE_TERMS)
  183. if low_terms:
  184. score -= min(4, len(low_terms) * 2)
  185. risks.append("possible low-value/non-dealer result: " + ", ".join(low_terms[:4]))
  186. channel_terms = matched_terms(entity_text, {"concessionnaire", "dealer", "showroom", "new vehicle", "new cars", "vehicules neufs", "v?hicules neufs", "vente", "importateur", "distributeur", "distribution", "groupe", "group", "multimarque", "multi-brand"})
  187. new_vehicle_needs_review = not channel_terms
  188. if ownership_needs_review and dc.should_request_manual_review(score, recommended_action, reasons):
  189. dc.add_manual_review_flag(risks, "ownership")
  190. if new_vehicle_needs_review and dc.should_request_manual_review(score, recommended_action, reasons):
  191. dc.add_manual_review_flag(risks, "new_vehicle")
  192. if not reasons:
  193. reasons.append("weak LinkedIn search signal; low priority unless deeper evidence confirms channel value")
  194. if recommended_action == "preview_only" and score >= 5:
  195. recommended_action = "deep_scrape"
  196. return {
  197. **candidate,
  198. "score": score,
  199. "score_reasons": reasons,
  200. "risk_flags": risks,
  201. "recommended_action": recommended_action,
  202. }
  203. def get_active_ws_endpoint(base_url: str, profile_id: str) -> str:
  204. return dc.get_active_ws_endpoint(base_url, profile_id)
  205. def dismiss_linkedin_dialogs(page: Page) -> None:
  206. labels = [
  207. "继续前往公司主页",
  208. "Continue to company page",
  209. "Continuer vers la page de l’entreprise",
  210. "关闭",
  211. "Close",
  212. ]
  213. for label in labels:
  214. try:
  215. page.get_by_text(label, exact=False).first.click(timeout=1500)
  216. time.sleep(0.5)
  217. except Exception:
  218. pass
  219. def collect_company_results(page: Page, query: str, max_results: int) -> List[Dict[str, Any]]:
  220. url = f"https://www.linkedin.com/search/results/companies/?keywords={quote_plus(query)}&origin=GLOBAL_SEARCH_HEADER"
  221. print(f"搜索 LinkedIn: {query}", flush=True)
  222. page.goto(url, wait_until="domcontentloaded", timeout=60000)
  223. time.sleep(random.uniform(3, 5))
  224. dismiss_linkedin_dialogs(page)
  225. for _ in range(2):
  226. page.mouse.wheel(0, 900)
  227. time.sleep(random.uniform(1, 1.8))
  228. raw = page.evaluate(
  229. """
  230. () => {
  231. const anchors = Array.from(document.querySelectorAll('a[href*="/company/"]'));
  232. return anchors.map((a) => {
  233. let node = a.closest('li') || a.closest('.reusable-search__result-container') || a.parentElement;
  234. let text = '';
  235. let cur = node;
  236. for (let i = 0; i < 7 && cur; i++) {
  237. const value = (cur.innerText || '').trim();
  238. if (value.length > text.length) text = value;
  239. if (value.length > 80) break;
  240. cur = cur.parentElement;
  241. }
  242. return { href: a.href || '', anchorText: (a.innerText || '').trim(), text };
  243. });
  244. }
  245. """
  246. )
  247. results: List[Dict[str, Any]] = []
  248. seen: Set[str] = set()
  249. for item in raw:
  250. href = normalize_linkedin_url(item.get("href", ""))
  251. if not href or href in seen:
  252. continue
  253. text = re.sub(r"\n{2,}", "\n", item.get("text", "")).strip()
  254. lines = [line.strip() for line in text.splitlines() if line.strip()]
  255. anchor = re.sub(r"\s+", " ", item.get("anchorText", "")).strip()
  256. if lines:
  257. name = lines[0]
  258. elif anchor and len(anchor) <= 80:
  259. name = anchor
  260. else:
  261. name = anchor.split(" ")[0].strip()
  262. name = re.sub(r"\s+\d+(?:\.\d+)?\s*万?\s*位关注者.*$", "", name).strip()
  263. if not name or name.lower() in {"linkedin", "home", "search"}:
  264. continue
  265. seen.add(href)
  266. results.append({
  267. "name": name,
  268. "href": href,
  269. "text": text[:1000],
  270. "source_queries": [query],
  271. })
  272. if len(results) >= max_results:
  273. break
  274. print(f" 收集到 {len(results)} 个公司候选", flush=True)
  275. return results
  276. def merge_candidates(items: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
  277. by_href: Dict[str, Dict[str, Any]] = {}
  278. for item in items:
  279. href = item["href"]
  280. if href not in by_href:
  281. by_href[href] = item
  282. continue
  283. existing = by_href[href]
  284. existing["source_queries"] = sorted(set(existing.get("source_queries", []) + item.get("source_queries", [])))
  285. snippets = existing.setdefault("snippets", [existing.get("text", "")])
  286. if item.get("text") and item["text"] not in snippets:
  287. snippets.append(item["text"])
  288. return list(by_href.values())
  289. def extract_lines(text: str) -> List[str]:
  290. return [line.strip() for line in text.splitlines() if line.strip()]
  291. def value_after_label(lines: List[str], label: str) -> str:
  292. for idx, line in enumerate(lines):
  293. if line.strip() == label and idx + 1 < len(lines):
  294. return lines[idx + 1].strip()
  295. return ""
  296. def extract_urls(text: str) -> List[str]:
  297. urls = re.findall(r"https?://[^\s<>()]+", text)
  298. cleaned: List[str] = []
  299. for url in urls:
  300. url = url.rstrip(".,,。;;")
  301. if "linkedin.com" not in url and url not in cleaned:
  302. cleaned.append(url)
  303. return cleaned
  304. def extract_email(text: str) -> str:
  305. match = re.search(r"[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}", text)
  306. return match.group(0) if match else ""
  307. def extract_phone(text: str) -> str:
  308. explicit = re.search(r"(?:Téléphone|电话|Tél|Tel|WhatsApp|电话号码)[::]?\s*([+()\d][+()\d\s.-]{6,}\d)", text, re.I)
  309. if explicit:
  310. return re.sub(r"\s+", " ", explicit.group(1)).strip()
  311. match = re.search(r"(?:\+212|0)\s?\d[\d\s.-]{6,}\d", text)
  312. return re.sub(r"\s+", " ", match.group(0)).strip() if match else ""
  313. def extract_city(text: str, fallback: str = DEFAULT_CITY_SCOPE) -> str:
  314. hits = []
  315. for city in CITY_PATTERNS:
  316. if re.search(rf"\b{re.escape(city)}\b", text, re.I) and city not in hits:
  317. hits.append(city)
  318. return " / ".join(hits[:4]) if hits else fallback
  319. def extract_followers(text: str) -> str:
  320. match = re.search(r"([\d,.]+\s*万?|[\d,.]+)\s*位关注者", text)
  321. return match.group(1).strip() + "位关注者" if match else ""
  322. def classify_customer_type(text: str) -> str:
  323. return dc.classify_customer_type(str(text or ""))
  324. def summarize_business(text: str) -> str:
  325. lower = text.casefold()
  326. parts: List[str] = []
  327. if any(term in lower for term in ["occasion", "reprise", "voiture d'occasion"]):
  328. parts.append("二手车买卖/置换")
  329. if any(term in lower for term in ["concessionnaire", "showroom", "vente automobile", "汽车零售"]):
  330. parts.append("汽车销售/showroom")
  331. if any(term in lower for term in ["importateur", "distribution", "distributeur"]):
  332. parts.append("进口/分销")
  333. if any(term in lower for term in ["utilitaire", "camion", "truck", "fleet", "flotte"]):
  334. parts.append("商用车/车队")
  335. china = matched_terms(text, CHINA_BRAND_TERMS)
  336. non_china = matched_terms(text, NON_CHINA_BRAND_TERMS)
  337. if china:
  338. parts.append("涉及中国品牌: " + ", ".join(china[:4]))
  339. elif non_china:
  340. parts.append("主要品牌信号: " + ", ".join(non_china[:5]))
  341. return ";".join(parts) if parts else "LinkedIn 汽车渠道线索,主营业务证据不足,需深搜确认"
  342. def deep_scrape_about(page: Page, candidate: Dict[str, Any], country: str) -> Dict[str, Any]:
  343. about_url = candidate["href"].rstrip("/") + "/about/"
  344. print(f"深采 LinkedIn About: {candidate['name']}", flush=True)
  345. page.goto(about_url, wait_until="domcontentloaded", timeout=60000)
  346. time.sleep(random.uniform(3, 5))
  347. dismiss_linkedin_dialogs(page)
  348. time.sleep(random.uniform(1, 2))
  349. title = page.title()
  350. text = page.locator("body").inner_text(timeout=10000)
  351. if not text.strip():
  352. candidate.setdefault("risk_flags", [])
  353. dc.add_manual_review_flag(candidate["risk_flags"], "detail")
  354. lines = extract_lines(text)
  355. urls = extract_urls(text)
  356. website = value_after_label(lines, "网站") or (urls[0] if urls else "")
  357. phone = value_after_label(lines, "电话") or extract_phone(text)
  358. email = extract_email(text)
  359. industry = value_after_label(lines, "行业")
  360. city = extract_city(text)
  361. followers = extract_followers(text)
  362. company_type = classify_customer_type(text + "\n" + candidate.get("text", ""))
  363. business = summarize_business(text + "\n" + candidate.get("text", ""))
  364. is_oem, brand = looks_like_oem_local_branch(candidate.get("name", ""), candidate.get("href", ""), text)
  365. if is_oem:
  366. candidate["recommended_action"] = "skip_brand_branch"
  367. candidate.setdefault("risk_flags", []).append(f"深采确认疑似品牌官方页: {brand}")
  368. dc.add_manual_review_flag(candidate["risk_flags"], "ownership")
  369. note_parts = [
  370. "LinkedIn深采",
  371. f"行业:{industry}" if industry else "",
  372. f"关注者:{followers}" if followers else "",
  373. f"网站:{website}" if website else "",
  374. f"来源搜索词:{', '.join(candidate.get('source_queries', []))}",
  375. f"评分:{candidate.get('score')};原因:{'; '.join(candidate.get('score_reasons', []))}",
  376. ]
  377. if candidate.get("risk_flags"):
  378. note_parts.append("风险:" + "; ".join(candidate["risk_flags"]))
  379. note = " | ".join(part for part in note_parts if part)
  380. return {
  381. "公司名称": candidate.get("canonical_name") or candidate.get("name", ""),
  382. "国家": country,
  383. "城市": city,
  384. "客户类型": company_type,
  385. "linkin链接": candidate.get("href", ""),
  386. "联系人": "",
  387. "职位": "",
  388. "公司公共电话": phone,
  389. "公司公共邮箱(任一有效即可)": email,
  390. "个人邮箱": "",
  391. "公司主营业务": business,
  392. "建联状态": "未联系",
  393. "备注": note[:1200],
  394. "_debug_title": title,
  395. "_about_head": lines[:60],
  396. }
  397. def search_linkedin_dealers(
  398. profile_id: str,
  399. keywords: Optional[List[str]],
  400. ads_power_url: str,
  401. excel_path: str,
  402. sheet_name: str,
  403. max_results: int,
  404. max_results_per_query: int,
  405. min_score: int,
  406. deep_scrape: bool,
  407. country: str,
  408. blocklist_json: Optional[str] = None,
  409. ) -> Dict[str, Any]:
  410. existing_identity = dc.load_existing_identity(
  411. excel_path=excel_path,
  412. sheet_name=sheet_name,
  413. link_columns=["linkin链接", "主页/链接"],
  414. name_columns=["公司名称", "客户姓名/公司"],
  415. )
  416. existing_links = existing_identity["links"]
  417. existing_names = existing_identity["names"]
  418. blocklist_links = load_blocklist_json(blocklist_json)
  419. existing_links = existing_links | blocklist_links
  420. print(f"已从 {sheet_name} Sheet 加载 {len(existing_links)} 个 LinkedIn 现有链接、{len(existing_names)} 个公司名用于去重", flush=True)
  421. if blocklist_json:
  422. print(f" 其中来自黑名单 JSON: {len(blocklist_links)} 个", flush=True)
  423. ws_endpoint = get_active_ws_endpoint(ads_power_url, profile_id)
  424. playwright = sync_playwright().start()
  425. browser = playwright.chromium.connect_over_cdp(ws_endpoint)
  426. context = browser.contexts[0] if browser.contexts else browser.new_context()
  427. page = context.new_page()
  428. page.set_viewport_size({"width": 1366, "height": 850})
  429. queries = keywords or DEFAULT_KEYWORDS
  430. raw_candidates: List[Dict[str, Any]] = []
  431. search_log: List[Dict[str, Any]] = []
  432. records: List[Dict[str, Any]] = []
  433. try:
  434. for query in queries:
  435. items = collect_company_results(page, query, max_results_per_query)
  436. raw_candidates.extend(items)
  437. search_log.append({"query": query, "found": len(items), "url": page.url, "title": page.title()})
  438. time.sleep(random.uniform(2, 4))
  439. merged = merge_candidates(raw_candidates)
  440. scored_all = [score_candidate(item, existing_links, existing_names) for item in merged]
  441. skipped_existing = [c for c in scored_all if c.get("recommended_action") == "skip_existing"]
  442. active_scored = [c for c in scored_all if c.get("recommended_action") != "skip_existing"]
  443. scored = dc.dedupe_dealer_groups(active_scored)
  444. scored.sort(key=lambda c: (-int(c.get("score", 0)), c.get("name", "")))
  445. selected = [
  446. c for c in scored
  447. if c.get("recommended_action") == "deep_scrape" and int(c.get("score", 0)) >= min_score
  448. ][:max_results]
  449. if deep_scrape:
  450. for candidate in selected:
  451. record = deep_scrape_about(page, candidate, country=country)
  452. if candidate.get("recommended_action") == "skip_brand_branch":
  453. continue
  454. records.append(record)
  455. time.sleep(random.uniform(2, 4))
  456. finally:
  457. try:
  458. page.close()
  459. except Exception:
  460. pass
  461. playwright.stop()
  462. return {
  463. "generated_at": datetime.now(timezone.utc).isoformat(),
  464. "source": "LinkedIn company search via AdsPower active profile",
  465. "summary": {
  466. "profile_id": profile_id,
  467. "queries": queries,
  468. "raw_candidates": len(raw_candidates),
  469. "candidate_count": len(scored),
  470. "selected_for_deep_scrape": len(selected),
  471. "deep_scrape": deep_scrape,
  472. "record_count": len(records),
  473. "min_score": min_score,
  474. "max_results": max_results,
  475. "existing_links": len(existing_links),
  476. "existing_names": len(existing_names),
  477. "skipped_existing": len(skipped_existing),
  478. },
  479. "search_log": search_log,
  480. "candidates": scored,
  481. "skipped_existing": skipped_existing,
  482. "selected_candidates": selected,
  483. "records": records,
  484. }
  485. def parse_keywords(value: str) -> Optional[List[str]]:
  486. if not value.strip():
  487. return None
  488. return [item.strip() for item in value.split(",") if item.strip()]
  489. def main() -> None:
  490. parser = argparse.ArgumentParser(description="LinkedIn Morocco dealer discovery, preview-first")
  491. parser.add_argument("--profile-id", required=True, help="AdsPower profile ID; must already be open and logged in")
  492. parser.add_argument("--ads-power-url", default="http://127.0.0.1:50325", help="AdsPower local API URL")
  493. parser.add_argument("--keywords", default="", help="Comma-separated LinkedIn company search keywords")
  494. parser.add_argument("--excel", default=DEFAULT_EXCEL, help="Workbook for duplicate checking and optional write-back")
  495. parser.add_argument("--sheet", default=DEFAULT_SHEET, help="Target sheet, normally LinkedIn")
  496. parser.add_argument("--country", default="摩洛哥", help="Country value for records")
  497. parser.add_argument("--max-results", type=int, default=10, help="Maximum deep-scraped records")
  498. parser.add_argument("--max-results-per-query", type=int, default=5, help="Company links collected per query")
  499. parser.add_argument("--min-score", type=int, default=5, help="Minimum score for deep-scrape selection")
  500. parser.add_argument("--deep-scrape", action="store_true", help="Open selected company About pages and build records")
  501. parser.add_argument("--write-excel", action="store_true", help="Write deep-scraped records to the workbook")
  502. parser.add_argument("--blocklist-json", default="", help="Optional JSON file with additional LinkedIn URLs to skip")
  503. parser.add_argument("--run-id", default="", help="Run ID used for runs/YYYYMMDD/<run_id>/ artifacts.")
  504. parser.add_argument("--output", default="linkedin_candidate_preview.json", help="Output JSON path")
  505. args = parser.parse_args()
  506. read_workbook_info = resolve_workbook_path(args.excel, create_from_template=False)
  507. if read_workbook_info.get("path"):
  508. args.excel = str(read_workbook_info["path"])
  509. print(f"Workbook for duplicate checking: {args.excel} ({read_workbook_info['source']})", flush=True)
  510. result = search_linkedin_dealers(
  511. profile_id=args.profile_id,
  512. keywords=parse_keywords(args.keywords),
  513. ads_power_url=args.ads_power_url,
  514. excel_path=args.excel,
  515. sheet_name=args.sheet,
  516. max_results=args.max_results,
  517. max_results_per_query=args.max_results_per_query,
  518. min_score=args.min_score,
  519. deep_scrape=args.deep_scrape,
  520. country=args.country,
  521. blocklist_json=args.blocklist_json or None,
  522. )
  523. output_path = resolve_artifact_path(args.output, kind="scraper_preview", default_name=Path(args.output).name, run_id=args.run_id or None)
  524. output_path.parent.mkdir(parents=True, exist_ok=True)
  525. output_path.write_text(json.dumps(result, ensure_ascii=False, indent=2), encoding="utf-8")
  526. print(f"已保存 LinkedIn 候选结果: {output_path}", flush=True)
  527. if args.write_excel:
  528. write_workbook_info = resolve_workbook_path(args.excel, create_from_template=True)
  529. if write_workbook_info.get("path"):
  530. args.excel = str(write_workbook_info["path"])
  531. print(f"Workbook for writing: {args.excel} ({write_workbook_info['source']})", flush=True)
  532. if not args.deep_scrape:
  533. print("未写入 Excel:需要先使用 --deep-scrape 生成 records。", flush=True)
  534. elif not result.get("records"):
  535. print("未写入 Excel:没有可写入记录。", flush=True)
  536. else:
  537. write_result = append_records(
  538. excel_path=args.excel,
  539. sheet_name=args.sheet,
  540. records=result["records"],
  541. dedup_keys=["公司名称", "linkin链接"],
  542. )
  543. print(f"Excel 回写结果: {write_result}", flush=True)
  544. else:
  545. print("默认预览模式:未写入 Excel。确认要入表时再使用 --deep-scrape --write-excel。", flush=True)
  546. if __name__ == "__main__":
  547. main()