import re import urllib.parse from loguru import logger from pydantic import HttpUrl from src.config import config from src.linkedin_scraper import LinkedInScraper from src.schemas import Profile _VANITY_RE = re.compile(r"/in/([^/?#]+)/?") class ScraperService: @staticmethod def _parse_vanity(url: HttpUrl) -> str: """Extract the LinkedIn vanity name from a profile URL. The vanity is the path segment following `/in/` in a LinkedIn profile URL (e.g. `https://www.linkedin.com/in//`). Query strings, fragments, and trailing sub-paths (`/in//details`) are ignored. Args: url: A validated LinkedIn profile URL. Returns: The vanity name extracted from the URL path. Raises: ValueError: If the URL is not a LinkedIn profile URL or the vanity segment is empty. """ path = urllib.parse.urlparse(str(url)).path m = _VANITY_RE.search(path) if not m: raise ValueError(f"not a LinkedIn profile URL: {url}") vanity = m.group(1) if not vanity: raise ValueError(f"empty vanity in URL: {url}") return vanity async def scrape(self, url: HttpUrl) -> Profile: """Scrape a LinkedIn profile from its profile URL. Parses the vanity name out of `/in//` and delegates to `LinkedInScraper.scrape`. Any exception (bad URL, network error, missing profile id, parse failure) is logged with full context and re-raised so the caller can decide how to handle it. Args: url: A validated LinkedIn profile URL. Returns: The parsed :class:`Profile` produced by the scraper. Raises: ValueError: If `url` is not a valid LinkedIn profile URL. Exception: Any error raised by the scraper (network, parsing, missing profile id, etc.) is logged and re-raised unchanged. """ try: vanity = self._parse_vanity(url) except ValueError: logger.error("invalid LinkedIn profile URL: {}", url) raise async with LinkedInScraper(cookies=config.linkedin.cookies) as scraper: try: profile = await scraper.scrape(vanity) except Exception: logger.exception("scrape failed for vanity={!r} url={}", vanity, url) raise logger.debug("scraped profile {!r} from {}", profile.name, url) return profile