"""Precision rules (connectors/_precision): the navigation / CTA / cookie-consent / marketing noise observed in production (M2U64, 2026-09-13) is rejected while every real item in the fixtures is still extracted — precision must not cost recall.""" from __future__ import annotations from datetime import UTC, datetime import pytest from conftest import fixture_path from companyatlas.connectors import _precision as P from companyatlas.fetch import file_result from companyatlas.sdk import connector as C from companyatlas.sdk.models import ( ExtractedJob, ExtractedLocation, ExtractedNewsItem, ExtractedPerson, ExtractedPlan, ExtractedProduct, Extraction, ) from companyatlas.services.pipeline import _drop_corrupt_entities BASE = "https://www.ex.example" CFG = {"canonical_domain": "ex.example"} def _run(surface: str, name: str, path: str): # type: ignore[no-untyped-def] url = BASE + path return C.get("generic-html-v1").extract({"url": url, "surface": surface, "config": CFG}, file_result(fixture_path("generic_html", name), url=url)) # ------------------------------------------------------------------------------------------------------------ fixtures end to end def test_careers_cta_anchors_rejected_real_listings_kept() -> None: ex = _run("careers", "careers_cta_noise.html", "/nl/jobs") titles = [j.title for j in ex.jobs] assert titles == ["Senior Backend Software Engineer - Infrastructure", "Verpleegkundige spoedgevallen", "Data Analyst", "Accountmanager KMO", "Technicien réseau"] assert not {"Voor starters die willen gáán", "Early careers bij Telenet group", "Meer info", "Sales", "Alle vacatures", "Waarom werken bij ons?"} & set(titles) by = {j.title: j for j in ex.jobs} assert by["Senior Backend Software Engineer - Infrastructure"].location_text == "London, United Kingdom" assert by["Senior Backend Software Engineer - Infrastructure"].department == "Engineering" and by["Senior Backend Software Engineer - Infrastructure"].country == "GB" assert by["Data Analyst"].external_id == "12933" and by["Data Analyst"].country == "DE" # id stripped from display, kept for identity assert by["Verpleegkundige spoedgevallen"].city == "Gent" and by["Verpleegkundige spoedgevallen"].country == "BE" assert [b.kind for b in ex.blocks].count("job_listing") == 5 def test_careers_german_gender_markers_and_cta() -> None: ex = _run("careers", "careers_de.html", "/karriere") assert [j.title for j in ex.jobs] == ["Softwareentwickler Backend", "Projektleiter Anlagenbau", "Ausbildung zum Industriemechaniker 2027", "Werkstudent Marketing", "Pflegefachkraft Intensivstation"] assert [j.location_text for j in ex.jobs] == ["München", "Stuttgart", "Hamburg", "Berlin", "Köln"] # bare known cities are locations assert ex.jobs[0].country is None # …but no country is guessed def test_leadership_swapped_cards_and_contact_links() -> None: ex = _run("leadership", "leadership_swapped.html", "/leadership") people = {p.name: p for p in ex.people} assert set(people) == {"Jane Doe", "Brian Jacobson", "Priya Natarajan", "Marc van der Berg", "Samuel Adebayo"} assert people["Jane Doe"].title == "Chief Executive Officer" and people["Jane Doe"].role_category == "ceo" and people["Jane Doe"].is_executive assert people["Samuel Adebayo"].title == "Chair of the Board" and people["Samuel Adebayo"].role_category == "chair" assert people["Brian Jacobson"].title == "Chief Financial Officer" and people["Brian Jacobson"].role_category == "cfo" assert people["Priya Natarajan"].title is None and people["Priya Natarajan"].role_category == "other" # "Contact" is not a title assert people["Marc van der Berg"].role_category == "coo" def test_leadership_japanese() -> None: ex = _run("leadership", "leadership_ja.html", "/company/officers") people = {p.name: p for p in ex.people} assert set(people) == {"山田 太郎", "佐藤 花子", "鈴木 一郎", "高橋 美咲"} assert people["山田 太郎"].role_category == "ceo" and people["山田 太郎"].is_executive assert people["佐藤 花子"].role_category == "cfo" and people["佐藤 花子"].is_executive assert people["鈴木 一郎"].role_category == "board" and not people["鈴木 一郎"].is_executive assert people["高橋 美咲"].role_category == "vp" def test_locations_cookie_categories_and_nav_cards_rejected() -> None: ex = _run("locations", "locations_cookie_noise.html", "/locations") locs = {loc.name: loc for loc in ex.locations} assert set(locs) == {"Hormuz Grand Hotel", "Dubai Office", "London", "Singapore", "Rotterdam Warehouse"} assert locs["Hormuz Grand Hotel"].country == "OM" and locs["Hormuz Grand Hotel"].city is None # "Oman | Hormuz Grand Hotel" repaired assert locs["Dubai Office"].city == "Dubai" and locs["Dubai Office"].country == "AE" # not "Level 12" assert locs["Rotterdam Warehouse"].city == "Rotterdam" and locs["Rotterdam Warehouse"].kind == "warehouse" and locs["Rotterdam Warehouse"].country == "NL" assert locs["London"].address_text.startswith("1 Finsbury Avenue") and locs["Singapore"].city == "Singapore" assert all(loc.kind != "store" for loc in ex.locations) # "…providers store data" is not a shop def test_pricing_marketing_headings_rejected_real_tiers_kept() -> None: ex = _run("pricing", "pricing_marketing_noise.html", "/pricing") plans = {p.plan_name: p for p in ex.plans} assert {"Free", "Starter", "Team", "Business", "Enterprise"} <= set(plans) assert not {"Win your market with Similar Example for businesses", "Worry-free roaming.", "Unlock the full potential of your data", "Most popular", "Plans"} & set(plans) assert plans["Free"].price == 0 and plans["Free"].price_text == "$0 forever" assert plans["Starter"].price == 125 and plans["Starter"].billing_period == "month" and plans["Starter"].price_text == "$125 per month, billed annually" assert plans["Team"].price == 333 and plans["Team"].currency == "EUR" and plans["Team"].price_text == "Starting at €333 / month" assert plans["Business"].price == 1199 and plans["Business"].unit == "user" and plans["Business"].price_text == "US$ 1,199 per user / month" assert plans["Enterprise"].contact_sales and plans["Enterprise"].price is None and plans["Enterprise"].price_text == "Talk to sales" assert all("destina" not in (p.price_text or "") for p in ex.plans) # never a truncated sentence def test_products_nav_words_and_slogans_rejected() -> None: ex = _run("products", "products_news_noise.html", "/products") assert [p.name for p in ex.products] == ["Atlas Metrics™", "Atlas Logs®", "Atlas Traces"] # ® / ™ kept def test_news_pagination_and_category_labels_rejected() -> None: ex = _run("newsroom", "products_news_noise.html", "/news") titles = [n.title for n in ex.news] assert titles == ["Acme launches Atlas AI, an assistant for cloud operations", "Q2 results", "Acme and BigCo announce strategic partnership"] assert ex.news[1].published_at.date().isoformat() == "2026-08-28" # short title kept because a date confirms it def test_existing_fixtures_recall_unchanged() -> None: assert {p.plan_name for p in _run("pricing", "pricing.html", "/pricing").plans} == {"Starter", "Pro", "Enterprise"} assert len(_run("leadership", "leadership.html", "/about/leadership").people) == 7 assert len(_run("locations", "locations.html", "/company/locations").locations) == 8 assert len(_run("careers", "careers.html", "/careers").jobs) == 5 assert len(_run("newsroom", "newsroom.html", "/news").news) == 4 # ------------------------------------------------------------------------------------------------------------ unit rules @pytest.mark.parametrize("title,url,location,ok", [ ("Meer info", "https://x.example/nl/jobs/search?page=2", None, False), ("Voor starters die willen gáán", "https://x.example/nl/jobs/starters", None, False), ("Early careers bij Telenet group", "https://x.example/nl/jobs/early-careers", None, False), ("Mehr erfahren →", "https://x.example/karriere/stellen", None, False), ("Sales", "https://x.example/nl/jobs/12990-sales", None, False), # single word ("Wij zoeken mensen die het verschil willen maken…", "https://x.example/jobs/13001", "Mechelen", False), ("Senior Backend Software Engineer - Infrastructure", None, "London, United Kingdom", True), ("Regional Coordinator", "https://x.example/careers/x", None, True), # role vocabulary ("Something Unusual", "https://x.example/jobs/12345-something-unusual", None, True), # job-like URL ("Something Unusual", None, "Toronto, ON, Canada", True), # explicit location cell ("Something Unusual", None, None, False), # no signal at all ("Infirmier(ère) de nuit", None, None, True), ("Ingeniero de datos", None, None, True), ("ソフトウェアエンジニア", None, None, True), ]) def test_job_verdict(title: str, url: str | None, location: str | None, ok: bool) -> None: assert P.job_verdict(title, url=url, location=location).ok is ok @pytest.mark.parametrize("raw,clean,ident", [ ("Data Analyst (m/w/d) [12933]", "Data Analyst", "12933"), ("Accountmanager KMO - Apply now", "Accountmanager KMO", None), ("Technicien réseau (h/f)", "Technicien réseau", None), ("Werkstudent Marketing (all genders)", "Werkstudent Marketing", None), ("Product Manager (Job ID: 44812) →", "Product Manager", "44812"), ("Senior Engineer (REQ-501)", "Senior Engineer", "REQ-501"), ]) def test_clean_job_title(raw: str, clean: str, ident: str | None) -> None: assert P.clean_job_title(raw) == (clean, ident) @pytest.mark.parametrize("name,title,expected", [ ("Chief Executive Officer", "Jane Doe", ("Jane Doe", "Chief Executive Officer")), # swapped card ("Chair Emeritus", "Warner Bros. Discovery", None), # role as name, company as title ("Brian Jacobson", "Contact", ("Brian Jacobson", None)), ("Board of Directors", "Meet the people who govern the company", None), ("Leadership", "Read more", None), ("Contact", "Media relations", None), ("Our team", None, None), ("Marc van der Berg", "Chief Operating Officer", ("Marc van der Berg", "Chief Operating Officer")), ("Jane Doe", "Jane leads the company since 2019 and previously ran BigCo.", ("Jane Doe", None)), # bio sentence is not a title ("Dr. Aiko Tanaka", "Head of People", ("Dr. Aiko Tanaka", "Head of People")), ("Jane Doe 2", "CEO", None), ("Edgar S. Woolard, Jr.", "Key person", ("Edgar S. Woolard, Jr.", "Key person")), # abbreviation dot kept ]) def test_normalize_person(name: str, title: str | None, expected: tuple[str, str | None] | None) -> None: assert P.normalize_person(name, title) == expected @pytest.mark.parametrize("name", [ "Judy McGrath", "F. William McNabb III", "Michael G. McCaffery", "Catherine MacGregor", "Francis deSouza", "Calvin McDonald", "José Vicente de los Mozos", "Alexander Trotman, Baron Trotman", "Stephen Green, Baron Green of Hurstpierpoint", "Thomas John Watson, Sr.", "Leonardo DiCaprio", "Marc van der Berg", "María García-López", "Tom O'Neill", "山田 太郎", ]) def test_real_world_names_are_names(name: str) -> None: assert P.looks_like_person_name(name) @pytest.mark.parametrize("name", ["Meet the team", "Chief Executive Officer", "Read more", "Warner Bros. Discovery", "Key person", "Doe, Jane", "株式会社サンプル"]) def test_non_names_are_rejected(name: str) -> None: assert not P.looks_like_person_name(name) def test_role_category_after_swap_is_consistent() -> None: ex = Extraction(text="", blocks=[], people=[ExtractedPerson(name="Chief Executive Officer", title="Jane Doe", role_category="other", is_executive=False)]) P.apply_precision(ex, html_jobs=False) assert ex.people[0].name == "Jane Doe" and ex.people[0].role_category == "ceo" and ex.people[0].is_executive @pytest.mark.parametrize("loc,ok", [ (ExtractedLocation(name="Performance & Analytics", kind="store", city="Allows use of behavioural data to optimise performance"), False), (ExtractedLocation(name="Contact Us", city="Contact Us", country="KR"), False), (ExtractedLocation(name="Careers", country="KR"), False), (ExtractedLocation(name="Sign up for our newsletter", city="Paris", country="FR"), False), (ExtractedLocation(name="Acme Regional Hub"), False), # no evidence of a place (ExtractedLocation(name="Acme Regional Hub", kind="factory"), True), # explicit kind label (ExtractedLocation(name="Berlin"), True), # known city (ExtractedLocation(name="548 Market Street, Suite 200"), True), # street address (ExtractedLocation(name="Oman", city="Hormuz Grand Hotel", country="OM"), True), (ExtractedLocation(name="We are present in twelve countries across three continents.", country="US"), False), ]) def test_location_verdict(loc: ExtractedLocation, ok: bool) -> None: assert P.location_verdict(loc).ok is ok def test_normalize_location_repairs_country_as_name() -> None: fixed = P.normalize_location(ExtractedLocation(name="Oman", city="Hormuz Grand Hotel", country="OM")) assert fixed is not None and (fixed.name, fixed.city, fixed.country) == ("Hormuz Grand Hotel", None, "OM") kept = P.normalize_location(ExtractedLocation(name="Oman", city="Muscat", country="OM")) assert kept is not None and (kept.name, kept.city) == ("Oman", "Muscat") # a real city stays a city cleaned = P.normalize_location(ExtractedLocation(name="Dubai Office", city="Level 12", region="Emirates Towers", country="AE")) assert cleaned is not None and cleaned.city is None # "Level 12" is not a city @pytest.mark.parametrize("text,expected", [ ("Stay connected in the U.S. ($13/day) and over 200 international destinations", "$13/day"), ("$99 per user / month, billed annually", "$99 per user / month, billed annually"), ("Starting at €333 / month", "Starting at €333 / month"), ("US$ 1,199 per user / month", "US$ 1,199 per user / month"), ("Talk to sales", "Talk to sales"), ("Free", "Free"), ("Up to 5 users and email support", None), ]) def test_price_text_from(text: str, expected: str | None) -> None: assert P.price_text_from(text) == expected @pytest.mark.parametrize("name,ok", [ ("Win your market with Similarweb for businesses", False), ("Worry-free roaming.", False), ("Most popular", False), ("Plans", False), ("Unlock the full potential of your data", False), ("Talk to sales", False), ("Pro", True), ("Business Plus", True), ("Free", True), ("Enterprise", True), ("Team (annual)", True), ]) def test_plan_name_ok(name: str, ok: bool) -> None: assert P.plan_name_ok(name) is ok def test_plan_verdict_requires_price_contact_or_free() -> None: assert P.plan_verdict(ExtractedPlan(plan_name="Pro", price=29.0, price_text="$29 per month")).ok assert P.plan_verdict(ExtractedPlan(plan_name="Enterprise", contact_sales=True, price_text="Contact sales")).ok assert P.plan_verdict(ExtractedPlan(plan_name="Free", price=0.0, price_text="Free")).ok assert not P.plan_verdict(ExtractedPlan(plan_name="Pro", price=None, price_text="Everything you need")).ok assert not P.plan_verdict(ExtractedPlan(plan_name="Worry-free roaming.", price=13.0, price_text="n the U.S. ($13/day) and over 200 international destina")).ok @pytest.mark.parametrize("name,ok", [ ("Overview", False), ("Learn more", False), ("All products", False), ("Solutions", False), ("Discover how Atlas helps teams ship faster.", False), ("Atlas Metrics™", True), ("Atlas Logs®", True), ("Microsoft 365", True), ("Discover", True), ("Produits", False), ("製品一覧", False), ]) def test_product_verdict(name: str, ok: bool) -> None: assert P.product_verdict(name).ok is ok def test_news_verdict() -> None: assert not P.news_verdict("Read more").ok and not P.news_verdict("Older posts »").ok and not P.news_verdict("Press releases").ok assert not P.news_verdict("3").ok and not P.news_verdict("Page 2").ok and not P.news_verdict("Actualités").ok assert not P.news_verdict("Q2 results").ok assert P.news_verdict("Q2 results", published_at=datetime(2026, 8, 28, tzinfo=UTC)).ok assert P.news_verdict("Q2 results", url="https://x.example/news/2026/08/q2").ok assert P.news_verdict("Acme launches Atlas AI, an assistant for cloud operations").ok # ------------------------------------------------------------------------------------------------------------ pipeline last line of defence def _noisy_extraction() -> Extraction: return Extraction( text="ok", blocks=[], jobs=[ExtractedJob(title="ML Engineer"), ExtractedJob(title="Meer info", url="https://x.example/jobs"), ExtractedJob(title="Sales")], people=[ExtractedPerson(name="Jane Doe", title="CEO"), ExtractedPerson(name="Chair Emeritus", title="Warner Bros. Discovery"), ExtractedPerson(name="Brian Jacobson", title="Contact")], products=[ExtractedProduct(name="Atlas Metrics"), ExtractedProduct(name="Overview")], plans=[ExtractedPlan(plan_name="Pro", price=29.0, price_text="$29 per month"), ExtractedPlan(plan_name="Win your market with Similarweb for businesses", contact_sales=True, price_text="Talk to sales")], locations=[ExtractedLocation(name="Berlin", country="DE"), ExtractedLocation(name="Performance & Analytics", kind="store")], news=[ExtractedNewsItem(title="Quarterly results published", url="https://x.example/a"), ExtractedNewsItem(title="Read more", url="https://x.example/b")], ) def test_pipeline_filter_applies_precision_for_html_connector() -> None: ex = _noisy_extraction() _drop_corrupt_entities(ex, connector_id="generic-html-v1") assert [j.title for j in ex.jobs] == ["ML Engineer"] assert [(p.name, p.title) for p in ex.people] == [("Jane Doe", "CEO"), ("Brian Jacobson", None)] assert [p.name for p in ex.products] == ["Atlas Metrics"] assert [p.plan_name for p in ex.plans] == ["Pro"] assert [loc.name for loc in ex.locations] == ["Berlin"] assert [n.title for n in ex.news] == ["Quarterly results published"] def test_pipeline_filter_trusts_structured_job_boards() -> None: ex = _noisy_extraction() _drop_corrupt_entities(ex, connector_id="greenhouse-v1") assert [j.title for j in ex.jobs] == ["ML Engineer", "Meer info", "Sales"] # ATS jobs are never filtered by title rules assert [p.name for p in ex.products] == ["Atlas Metrics"] # other rules still apply