"""LinkedIn profile-link extraction and normalization. One shared vocabulary for "the same person" across the two places a LinkedIn identity appears: sourced talent profiles (a normalized URL from the Apify actor) and CV text (a link the candidate wrote, often mangled by PDF extraction). The match key is the lowercase public slug from /in/. Top-level module on purpose: talent/, inbox/ and job/ all need it, and any package-local home would invite an import cycle. """ import re from urllib.parse import unquote # CV text arrives from PDF extraction: URLs may carry percent-escapes, no # scheme ("linkedin.com/in/jane-doe"), trailing sentence punctuation glued on by # layout, or line-wraps inside the path ("linkedin.com/in/\njane-doe"). # /pub/ is the legacy public-profile path; /mwlite/in/ is the mobile web path. _SLUG_RE = re.compile( r"linkedin\.com/(?:in|pub|mwlite/in)/([A-Za-z0-9\-_.%]+)", re.IGNORECASE, ) # pypdf wraps URLs across lines / glyph gaps. Flatten those runs before matching # so "linkedin.com/in/\n jane-doe" still yields a slug. _LINKEDIN_RUN_RE = re.compile( r"(?:https?://)?(?:(?:[a-z0-9-]+\.)*)linkedin\.com(?:\s*/\s*[A-Za-z0-9\-_.%]*)+", re.IGNORECASE, ) # Clickable CV icons often store the URL only in an HTML href or a PDF # annotation, not in the visible text layer. _HREF_RE = re.compile( r"""href\s*=\s*["']([^"'>\s]*(?:linkedin\.com|lnkd\.in)[^"']*)["']""", re.IGNORECASE, ) # Short links from LinkedIn's own share button. Not a match key (no /in/) # but enough to open a profile from the inbox button. _LNKD_RE = re.compile(r"lnkd\.in/([A-Za-z0-9_-]+)", re.IGNORECASE) # Sentinel stored on application rows: NULL means "never scanned", the empty # string means "scanned, no link found". The distinction is what lets the lazy # backfill converge instead of rescanning every CV on every request. NO_SLUG = "" def normalize_slug(raw) -> str | None: """Lowercase, percent-decoded, stripped of trailing sentence punctuation.""" if not raw: return None slug = unquote(str(raw)).strip().lower().rstrip(".") return slug or None def slug_from_url(url) -> str | None: """Slug from an already-normalized profile URL (talent_profiles.linkedin_url).""" if not url: return None match = _SLUG_RE.search(_flatten_linkedin_runs(str(url))) return normalize_slug(match.group(1)) if match else None def _flatten_linkedin_runs(text: str) -> str: """Remove whitespace inside linkedin.com/... runs so wrapped PDFs still match.""" if not text: return "" return _LINKEDIN_RUN_RE.sub(lambda m: re.sub(r"\s+", "", m.group(0)), text) def _haystack(text) -> str: """Flatten wrapped LinkedIn URLs and splice href= targets into the scan text.""" raw = text or "" hrefs = "\n".join(_HREF_RE.findall(raw)) blob = f"{raw}\n{hrefs}" if hrefs else raw return _flatten_linkedin_runs(blob) def slugs_from_text(text) -> list[str]: """Every distinct slug mentioned in a CV, in order of first appearance.""" found: list[str] = [] for match in _SLUG_RE.finditer(_haystack(text)): slug = normalize_slug(match.group(1)) if slug and slug not in found: found.append(slug) return found def primary_slug_from_text(text) -> str: """The slug to persist on an application row; NO_SLUG when the CV has none.""" slugs = slugs_from_text(text) return slugs[0] if slugs else NO_SLUG def profile_url_from_text(text) -> str | None: """Public profile URL for the inbox LinkedIn button, or None. Prefers /in/ (and /pub/, /mwlite/in/). Falls back to lnkd.in short links which open the profile but are not a Find Talent match key. """ slug = primary_slug_from_text(text) if slug: return f"https://www.linkedin.com/in/{slug}" short = _LNKD_RE.search(_haystack(text)) if short: return f"https://lnkd.in/{short.group(1)}" return None