Something went wrong. Try again.
This repository has no description
Something went wrong. Try again.
2.6 kB · 75 lines
Python
at main
12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273747576import reimport urllib.parse
from loguru import loggerfrom pydantic import HttpUrl
from src.config import configfrom src.linkedin_scraper import LinkedInScraperfrom src.schemas import Profile
_VANITY_RE = re.compile(r"/in/([^/?#]+)/?")
class ScraperService: @staticmethod def _parse_vanity(url: HttpUrl) -> str: """Extract the LinkedIn vanity name from a profile URL.
The vanity is the path segment following `/in/` in a LinkedIn profile URL (e.g. `https://www.linkedin.com/in/<vanity>/`). Query strings, fragments, and trailing sub-paths (`/in/<vanity>/details`) are ignored.
Args: url: A validated LinkedIn profile URL.
Returns: The vanity name extracted from the URL path.
Raises: ValueError: If the URL is not a LinkedIn profile URL or the vanity segment is empty. """ path = urllib.parse.urlparse(str(url)).path m = _VANITY_RE.search(path) if not m: raise ValueError(f"not a LinkedIn profile URL: {url}") vanity = m.group(1) if not vanity: raise ValueError(f"empty vanity in URL: {url}") return vanity
async def scrape(self, url: HttpUrl) -> Profile: """Scrape a LinkedIn profile from its profile URL.
Parses the vanity name out of `/in/<vanity>/` and delegates to `LinkedInScraper.scrape`. Any exception (bad URL, network error, missing profile id, parse failure) is logged with full context and re-raised so the caller can decide how to handle it.
Args: url: A validated LinkedIn profile URL.
Returns: The parsed :class:`Profile` produced by the scraper.
Raises: ValueError: If `url` is not a valid LinkedIn profile URL. Exception: Any error raised by the scraper (network, parsing, missing profile id, etc.) is logged and re-raised unchanged. """ try: vanity = self._parse_vanity(url) except ValueError: logger.error("invalid LinkedIn profile URL: {}", url) raise
async with LinkedInScraper(cookies=config.linkedin.cookies) as scraper: try: profile = await scraper.scrape(vanity) except Exception: logger.exception("scrape failed for vanity={!r} url={}", vanity, url) raise logger.debug("scraped profile {!r} from {}", profile.name, url) return profile