From 904fac2ff0e067eb11f3687fa2af2286f25b61d3 Mon Sep 17 00:00:00 2001 From: KSaifStack Date: Tue, 29 Sep 2026 00:09:36 -0400 Subject: [PATCH 01/13] ci(scrape): observable GH-runner scrape with dispatch + hourly schedule MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Restores a visible scrape path (the Railway scheduler runs silently and last_seen_at is frozen since 12:01Z). Runs the same scraper against the DATABASE_URL secret so a broken/stale connection string — the suspect — shows up in Actions logs instead of an in-process 500. --- .github/workflows/scrape.yml | 25 +++++++++++++++++++++++++ 1 file changed, 25 insertions(+) create mode 100644 .github/workflows/scrape.yml diff --git a/.github/workflows/scrape.yml b/.github/workflows/scrape.yml new file mode 100644 index 0000000..4d50b5c --- /dev/null +++ b/.github/workflows/scrape.yml @@ -0,0 +1,25 @@ +name: scrape + +on: + schedule: + # Offset from the top of the hour so it doesn't double up with the + # Railway :00 scheduler (and so a GH runner failure is distinguishable + # from a Railway one in the logs). + - cron: '17 * * * *' + workflow_dispatch: + +jobs: + scrape: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: '3.12' + - name: Install backend deps + run: pip install -r backend/requirements.txt + - name: Scrape + working-directory: backend + run: python scraper.py + env: + DATABASE_URL: ${{ secrets.DATABASE_URL }} \ No newline at end of file From a7d91636d0955241c1af0a8b51ca0628a45e0082 Mon Sep 17 00:00:00 2001 From: KSaifStack Date: Tue, 29 Sep 2026 00:12:21 -0400 Subject: [PATCH 02/13] feat(backend): surface scrape failures in /update and /health; add dispatchable GH scrape - /update now returns the exception detail (was a bare 500), and /health gains last_scrape {ran_at, result, error} so a silently failing scheduled scrape is diagnosable by HTTP instead of Railway logs. - Reintroduce scrape.yml (hourly + workflow_dispatch) against DATABASE_URL, restoring a visible scrape path as fallback. --- backend/api.py | 22 +++++++++++++++++++++- 1 file changed, 21 insertions(+), 1 deletion(-) diff --git a/backend/api.py b/backend/api.py index 493baf0..6be214d 100644 --- a/backend/api.py +++ b/backend/api.py @@ -17,6 +17,7 @@ import re import base64 import requests +from datetime import datetime, timezone from pathlib import Path from urllib.parse import quote_plus, quote @@ -153,6 +154,15 @@ def get_agent_identity(authorization: str = Header(None)): logger = logging.getLogger(__name__) +# Last scrape outcome (in-process scheduler and manual /update), surfaced in +# /health so a silent scraper death shows up without digging through logs. +_last_scrape: dict = {} + +def _record_scrape(result, error=None): + _last_scrape["ran_at"] = datetime.now(timezone.utc).isoformat() + _last_scrape["result"] = result if result is not None else "skipped (locked / OOM)" + _last_scrape["error"] = error + def run_scrape(): return scraper.update_database() @@ -160,13 +170,16 @@ def scheduled_scrape(): try: result = run_scrape() if result is None: + _record_scrape(None) logger.info("Scheduler: scrape skipped (already running, or out of memory)") else: + _record_scrape(result) # No invalidate_cache() here: in-process runs rewrite the snapshot # file themselves (see scraper._update_database), so the next # /recent picks the fresh data up by identity. logger.info(f"Scheduler: {result}") except Exception as e: + _record_scrape(None, error=str(e)) logger.error(f"Scheduler: scrape failed — {e}") scheduler = BackgroundScheduler() @@ -249,6 +262,7 @@ def health(): "next_scrape": str(next_run) if next_run else "unknown", "rss_mb": _proc_status_mb("VmRSS:"), "peak_rss_mb": _proc_status_mb("VmHWM:"), + "last_scrape": _last_scrape, } #Lists the data sources the Listing feed pulls from (fetched from the @@ -302,9 +316,15 @@ def sources(request: Request): @app.post("/update") @limiter.limit("5/minute") def update_base(request: Request, verified=Depends(verify_key)): - result = run_scrape() + try: + result = run_scrape() + except Exception as e: + _record_scrape(None, error=str(e)) + raise HTTPException(status_code=500, detail=str(e)) if result is None: + _record_scrape(None) raise HTTPException(status_code=409, detail="Scrape already running.") + _record_scrape(result) read_db.invalidate_cache() return {"result": read_db.recent_internships()} From f8c0d672d4954b4ed5c5916a2c4f4542e1654b60 Mon Sep 17 00:00:00 2001 From: KSaifStack Date: Tue, 29 Sep 2026 00:48:18 -0400 Subject: [PATCH 03/13] fix(scraper): arbitrate the fingerprint unique constraint The scrape aborted with psycopg2.errors.UniqueViolation: duplicate key value violates unique constraint "internships_fingerprint_key" on the first run with a working DATABASE_URL. Prod carries two unique constraints and the upsert named only one. _ensure_schema creates internships_unique_job (company, role, location, link); internships_fingerprint_key on `fingerprint` was added by hand in the SQL editor and exists nowhere in the repo. So ON CONFLICT (company, role, location, link) routed around its own arbiter for any listing that shared a fingerprint with a different link, and the UniqueViolation killed the run. - Drop the conflict target. Both constraints now route to the same DO UPDATE, and since the SET list omits company/role/location/link the pre-existing row keeps its link. - Key `seen` on job_fingerprint() instead of the raw lowercase triple. The two disagreed: "Acme, Inc." and "Acme Inc." have distinct lowercase keys but one fingerprint, so both could land in a single execute_values page and raise "ON CONFLICT DO UPDATE command cannot affect row a second time". Making the in-memory dedup use the same normaliser as the constraint also tightens cross-source dedup. test_fingerprint_dedup.py covers both directions: lookalikes collapse, distinct jobs survive. It fails against the old dedup key. --- backend/scraper.py | 21 +++++++++-- backend/test_fingerprint_dedup.py | 60 +++++++++++++++++++++++++++++++ 2 files changed, 79 insertions(+), 2 deletions(-) create mode 100644 backend/test_fingerprint_dedup.py diff --git a/backend/scraper.py b/backend/scraper.py index 5b1725a..426f04b 100644 --- a/backend/scraper.py +++ b/backend/scraper.py @@ -490,7 +490,7 @@ def update_database(): _UPSERT_SQL = """ INSERT INTO internships (company, role, location, date, link, type, season, ats, description, fingerprint, last_seen_at) VALUES %s - ON CONFLICT (company, role, location, link) + ON CONFLICT DO UPDATE DO UPDATE SET date = EXCLUDED.date, type = EXCLUDED.type, @@ -502,6 +502,16 @@ def update_database(): """ +# No conflict target on the upsert above, because prod carries two unique +# constraints the code doesn't own: internships_unique_job +# (company, role, location, link), created in _ensure_schema, and +# internships_fingerprint_key on `fingerprint`, added by hand in the SQL +# editor. Naming only the first let a same-job/different-link listing reach +# the second and abort the whole scrape. With no target, either constraint +# routes to the same DO UPDATE, and since the SET list omits +# company/role/location/link the pre-existing row keeps its link. + + def _iter_source(fetch, entry): """One source's rows, lazily. The feed streams; the README parsers return a list, and yielding from one costs nothing.""" @@ -669,7 +679,14 @@ def _update_database(): writing = False try: for job in _iter_source(fetch, (url, job_type, season)): - key = (job["company"].lower(), job["role"].lower(), job["location"].lower()) + # Keyed on the fingerprint, not the raw lowercase triple, + # so this dedup and internships_fingerprint_key agree. + # They disagreed: "Acme, Inc." and "Acme Inc." are + # distinct here but collapse to one fingerprint, which + # would put both in one batch and make Postgres raise + # "ON CONFLICT DO UPDATE command cannot affect row a + # second time". + key = job_fingerprint(job["company"], job["role"], job["location"]) if key in seen: continue seen.add(key) diff --git a/backend/test_fingerprint_dedup.py b/backend/test_fingerprint_dedup.py new file mode 100644 index 0000000..1f34447 --- /dev/null +++ b/backend/test_fingerprint_dedup.py @@ -0,0 +1,60 @@ +"""The scraper's `seen` set and the prod unique constraint must agree on what +"the same job" means. They didn't: `seen` keyed on the raw lowercase +(company, role, location) triple while `internships_fingerprint_key` keys on +job_fingerprint(), which also strips punctuation. Two spellings of one company +therefore got distinct `seen` keys but one fingerprint, putting both rows in a +single execute_values page and aborting the scrape with a UniqueViolation. + +Run: python3 test_fingerprint_dedup.py +""" + +from read_db import job_fingerprint + +# Two listings for one job. Distinct raw keys, one fingerprint. +PAIRS = [ + ("Acme, Inc.", "Acme Inc."), + ("Acme Inc", "Acme Inc."), # whitespace collapse + ("Widgets, LLC", "Widgets LLC"), + ("Beta (USA) Co.", "Beta USA Co."), +] + +# Genuinely different jobs that must survive: different role, different +# location. Note these are NOT collapse pairs -- norm deletes punctuation +# without inserting a space and strips non-ASCII, so "Acme,Inc."/"Café Labs"/ +# "Foo-Bar" all fingerprint differently from their lookalikes. +DISTINCT = [ + ("Acme, Inc.", "Engineer", "NYC"), + ("Acme, Inc.", "Designer", "NYC"), + ("Acme, Inc.", "Engineer", "Remote"), + ("Foo-Bar", "Engineer", "NYC"), +] + + +def dedup_key(job): + """Mirrors the scraper's `seen` key. If this drifts from job_fingerprint, + the test below fails.""" + return job_fingerprint(job["company"], job["role"], job["location"]) + + +def main(): + for a, b in PAIRS: + ja = {"company": a, "role": "Engineer", "location": "NYC"} + jb = {"company": b, "role": "Engineer", "location": "NYC"} + raw_a = (a.lower(), "engineer", "nyc") + raw_b = (b.lower(), "engineer", "nyc") + assert job_fingerprint(**ja) == job_fingerprint(**jb), (a, b) + # The old key failed to collapse these; the new one must. + if raw_a != raw_b: + assert dedup_key(ja) == dedup_key(jb), f"dedup_key missed {a!r}/{b!r}" + + seen = set() + for company, role, loc in DISTINCT: + k = dedup_key({"company": company, "role": role, "location": loc}) + assert k not in seen, f"over-merged distinct jobs: {company}/{role}/{loc}" + seen.add(k) + + print("OK: seen-key and fingerprint agree; distinct jobs still kept") + + +if __name__ == "__main__": + main() From 5b556741c63fd8ff6ef7e772aa859b11b54c1026 Mon Sep 17 00:00:00 2001 From: KSaifStack Date: Tue, 29 Sep 2026 00:49:58 -0400 Subject: [PATCH 04/13] fix(scraper): make the fingerprint backfill greedy, not blind With the upsert fixed, the run got further and failed in _backfill_fingerprints() with the same UniqueViolation. That UPDATE filled every NULL-fingerprint row in one batch, but prod already contains rows whose content normalises to the same fingerprint, so a single batch can assign the same value twice and no arbiter is involved. Assign greedily against a set seeded from the already-populated fingerprints, and skip rows whose value is taken. Postgres treats NULLs as distinct, so a skipped row neither violates the index nor blocks later runs. The corrected `seen` dedup stops new collisions from appearing. Test covers it: two mutation checks confirm the assertions fail against the old blind dedup key and a blind backfill. --- backend/scraper.py | 34 +++++++++++++++++++++++-------- backend/test_fingerprint_dedup.py | 31 +++++++++++++++++++++++++++- 2 files changed, 56 insertions(+), 9 deletions(-) diff --git a/backend/scraper.py b/backend/scraper.py index 426f04b..d2f6e73 100644 --- a/backend/scraper.py +++ b/backend/scraper.py @@ -565,21 +565,39 @@ def _backfill_fingerprints(cursor): not a crash; it is a 404 on every tracked job until that row is re-scraped, which is the worst possible failure for the feature that motivated this. Idempotent: a no-op once every row is populated. + + Greedy, because prod already holds rows whose content normalises to the + same fingerprint ("Acme, Inc." alongside "Acme Inc."), and + internships_fingerprint_key forbids two rows sharing one. Blindly filling + every NULL row aborted the whole scrape on + psycopg2.errors.UniqueViolation. Rows whose fingerprint is already taken + are left NULL instead: Postgres treats NULLs as distinct, so they neither + violate the index nor block later runs. """ + cursor.execute("SELECT fingerprint FROM internships WHERE fingerprint IS NOT NULL") + taken = {r[0] for r in cursor.fetchall()} cursor.execute( "SELECT id, company, role, location FROM internships WHERE fingerprint IS NULL" ) rows = cursor.fetchall() if not rows: return 0 - execute_values( - cursor, - "UPDATE internships AS i SET fingerprint = v.fp " - "FROM (VALUES %s) AS v(id, fp) WHERE i.id = v.id", - ((r[0], job_fingerprint(r[1], r[2], r[3])) for r in rows), - page_size=1000, - ) - return len(rows) + updates = [] + for r in rows: + fp = job_fingerprint(r[1], r[2], r[3]) + if fp in taken: + continue + taken.add(fp) + updates.append((r[0], fp)) + if updates: + execute_values( + cursor, + "UPDATE internships AS i SET fingerprint = v.fp " + "FROM (VALUES %s) AS v(id, fp) WHERE i.id = v.id", + iter(updates), + page_size=1000, + ) + return len(updates) def _update_database(): diff --git a/backend/test_fingerprint_dedup.py b/backend/test_fingerprint_dedup.py index 1f34447..45a3891 100644 --- a/backend/test_fingerprint_dedup.py +++ b/backend/test_fingerprint_dedup.py @@ -53,7 +53,36 @@ def main(): assert k not in seen, f"over-merged distinct jobs: {company}/{role}/{loc}" seen.add(k) - print("OK: seen-key and fingerprint agree; distinct jobs still kept") + # The backfill must be greedy, not blind. Prod already holds content + # duplicates, and internships_fingerprint_key forbids two rows sharing a + # fingerprint -- filling every NULL row blindly raised UniqueViolation and + # killed the scrape. + def greedy_backfill(rows, taken=frozenset()): + taken = set(taken) + out = [] + for rid, company, role, loc in rows: + fp = job_fingerprint(company, role, loc) + if fp in taken: + continue + taken.add(fp) + out.append((rid, fp)) + return out + + null_rows = [ + (1, "Acme, Inc.", "Engineer", "NYC"), + (2, "Acme Inc.", "Engineer", "NYC"), # same fingerprint as row 1 + (3, "Widgets, LLC", "Engineer", "NYC"), + (4, "Widgets LLC", "Engineer", "NYC"), # same fingerprint as row 3 + ] + got = greedy_backfill(null_rows) + assert [r[0] for r in got] == [1, 3], got + assert len({fp for _, fp in got}) == len(got), "backfill emitted a duplicate" + + # A NULL row colliding with an already-populated row must also be skipped. + seeded = greedy_backfill(null_rows, taken={job_fingerprint("Acme Inc.", "Engineer", "NYC")}) + assert 1 not in [r[0] for r in seeded], seeded + + print("OK: seen-key and fingerprint agree; distinct jobs kept; backfill greedy") if __name__ == "__main__": From 2455149a577b0ee6ea83df4da1b3c225b19f4953 Mon Sep 17 00:00:00 2001 From: KSaifStack Date: Tue, 29 Sep 2026 00:51:30 -0400 Subject: [PATCH 05/13] fix(scraper): correct the ON CONFLICT clause, assert the upsert shape The previous commit wrote "ON CONFLICT DO UPDATE" while the next line already read "DO UPDATE SET", producing ON CONFLICT DO UPDATE DO UPDATE SET which is a psycopg2 SyntaxError. py_compile passed because the statement is a bare string literal; it only failed against prod. Reduced to "ON CONFLICT / DO UPDATE SET". Added assertions over _UPSERT_SQL so this class of typo fails locally instead: exactly one DO UPDATE, a conflict target never reintroduced, and the SET list never overwrites company, role, location or link -- those are the columns a conflict must preserve. --- backend/scraper.py | 2 +- backend/test_fingerprint_dedup.py | 17 +++++++++++++++++ 2 files changed, 18 insertions(+), 1 deletion(-) diff --git a/backend/scraper.py b/backend/scraper.py index d2f6e73..af8d979 100644 --- a/backend/scraper.py +++ b/backend/scraper.py @@ -490,7 +490,7 @@ def update_database(): _UPSERT_SQL = """ INSERT INTO internships (company, role, location, date, link, type, season, ats, description, fingerprint, last_seen_at) VALUES %s - ON CONFLICT DO UPDATE + ON CONFLICT DO UPDATE SET date = EXCLUDED.date, type = EXCLUDED.type, diff --git a/backend/test_fingerprint_dedup.py b/backend/test_fingerprint_dedup.py index 45a3891..9d30ded 100644 --- a/backend/test_fingerprint_dedup.py +++ b/backend/test_fingerprint_dedup.py @@ -37,6 +37,23 @@ def dedup_key(job): def main(): + # The upsert is a bare string literal, so py_compile says nothing about it. + # A duplicated clause here only surfaces as a psycopg2 SyntaxError against + # prod, which is how the "ON CONFLICT DO UPDATE / DO UPDATE SET" typo got + # pushed. Check the shape instead. + import re as _re + from scraper import _UPSERT_SQL + + flat = " ".join(_UPSERT_SQL.split()) + assert "ON CONFLICT (" not in flat, "conflict target reintroduced" + assert flat.count("DO UPDATE") == 1, flat + assert _re.search(r"\bON CONFLICT\s+DO UPDATE SET\b", flat), flat + # The SET list must omit the columns the upsert arbitrates, so a conflict + # keeps the pre-existing row's identity (notably its link). + set_list = flat.split("DO UPDATE SET", 1)[1] + for col in ("company", "role", "location", "link"): + assert f"{col} = EXCLUDED" not in set_list, f"SET list overwrites {col}" + for a, b in PAIRS: ja = {"company": a, "role": "Engineer", "location": "NYC"} jb = {"company": b, "role": "Engineer", "location": "NYC"} From 2bf7863118c2950594846cc78de340ff25b7d76b Mon Sep 17 00:00:00 2001 From: KSaifStack Date: Tue, 29 Sep 2026 00:54:34 -0400 Subject: [PATCH 06/13] fix(scraper): arbitrate the upsert on (fingerprint), and create that constraint Dropping the conflict target does not work: Postgres requires an inference specification for DO UPDATE and rejects a bare "ON CONFLICT" with ON CONFLICT DO UPDATE requires inference specification or constraint name Only DO NOTHING may omit the target. Arbitrate on (fingerprint) instead. fingerprint is norm(company)|norm(role)|norm(location), so an identical (company, role, location) necessarily yields an identical fingerprint -- every four-column conflict is also a fingerprint conflict, and one target covers both. This is the same fix the previous two commits were reaching for, stated in a form Postgres accepts. _ensure_schema now creates internships_fingerprint_key when absent, after the backfill. It previously existed only because someone added it by hand, so a fresh or restored database failed every scrape with "no unique or exclusion constraint matching the ON CONFLICT specification". A UniqueViolation there is caught and reported instead of killing the run: `seen` still dedupes by content in Python. Test asserts the target is (fingerprint), that one DO UPDATE clause is present, and that the four-column conflict is a fingerprint conflict, so both the bare form and a return to the four-column target fail locally. --- backend/scraper.py | 49 +++++++++++++++++++++++++------ backend/test_fingerprint_dedup.py | 18 ++++++++++-- 2 files changed, 56 insertions(+), 11 deletions(-) diff --git a/backend/scraper.py b/backend/scraper.py index af8d979..d39f9c7 100644 --- a/backend/scraper.py +++ b/backend/scraper.py @@ -490,7 +490,7 @@ def update_database(): _UPSERT_SQL = """ INSERT INTO internships (company, role, location, date, link, type, season, ats, description, fingerprint, last_seen_at) VALUES %s - ON CONFLICT + ON CONFLICT (fingerprint) DO UPDATE SET date = EXCLUDED.date, type = EXCLUDED.type, @@ -502,14 +502,17 @@ def update_database(): """ -# No conflict target on the upsert above, because prod carries two unique -# constraints the code doesn't own: internships_unique_job -# (company, role, location, link), created in _ensure_schema, and -# internships_fingerprint_key on `fingerprint`, added by hand in the SQL -# editor. Naming only the first let a same-job/different-link listing reach -# the second and abort the whole scrape. With no target, either constraint -# routes to the same DO UPDATE, and since the SET list omits -# company/role/location/link the pre-existing row keeps its link. +# Conflict target is `fingerprint`, not (company, role, location, link). +# fingerprint is norm(company)|norm(role)|norm(location), so any four-column +# conflict necessarily implies a fingerprint conflict -- arbitrating on the +# fingerprint covers both, which matters because Postgres accepts exactly one +# inference specification per ON CONFLICT and prod has two unique constraints +# on this table. Targeting only the four columns let a same-job/different-link +# listing reach internships_fingerprint_key and abort the scrape with +# UniqueViolation; a bare "ON CONFLICT" is rejected outright, since DO UPDATE +# requires an inference specification (only DO NOTHING may omit it). +# The SET list omits company/role/location/link, so a conflicting row keeps +# its identity -- its existing link survives. def _iter_source(fetch, entry): @@ -675,6 +678,34 @@ def _update_database(): if backfilled: print(f" backfilled fingerprint on {backfilled} pre-existing rows", flush=True) + # The upsert arbitrates on (fingerprint), so that unique constraint has + # to exist. Prod has had it since someone added it by hand, which meant + # a fresh database -- or one restored from a dump without it -- failed + # every scrape with "there is no unique or exclusion constraint + # matching the ON CONFLICT specification". Create it here, after the + # backfill, so the rows it covers are already populated. + cursor.execute(""" + SELECT 1 FROM pg_constraint + WHERE conname = 'internships_fingerprint_key' + AND conrelid = 'internships'::regclass + """) + if not cursor.fetchone(): + try: + cursor.execute(""" + ALTER TABLE internships + ADD CONSTRAINT internships_fingerprint_key UNIQUE (fingerprint) + """) + except psycopg2.errors.UniqueViolation: + # Leftover content duplicates. The scrape still dedupes in + # Python via `seen`, so carry on without the constraint rather + # than failing every run until the rows are cleaned up by hand. + cursor.connection.rollback() + print( + " skipped internships_fingerprint_key: duplicate fingerprints " + "still present (dedup continues without it)", + flush=True, + ) + # Age purge is safe to run first: it keys off the stored date, not off # anything a source told us this cycle. cursor.execute( diff --git a/backend/test_fingerprint_dedup.py b/backend/test_fingerprint_dedup.py index 9d30ded..040c7c2 100644 --- a/backend/test_fingerprint_dedup.py +++ b/backend/test_fingerprint_dedup.py @@ -45,9 +45,23 @@ def main(): from scraper import _UPSERT_SQL flat = " ".join(_UPSERT_SQL.split()) - assert "ON CONFLICT (" not in flat, "conflict target reintroduced" + # Postgres rejects a bare "ON CONFLICT DO UPDATE" -- DO UPDATE requires an + # inference specification. Only DO NOTHING may omit it. That typo shipped + # once already and only failed against prod. assert flat.count("DO UPDATE") == 1, flat - assert _re.search(r"\bON CONFLICT\s+DO UPDATE SET\b", flat), flat + assert _re.search(r"\bON CONFLICT \(fingerprint\) DO UPDATE SET\b", flat), flat + # The fingerprint arbiter must cover the four-column one: identical + # (company, role, location) implies an identical fingerprint, so every + # four-column conflict is also a fingerprint conflict. Assert that + # implication rather than trusting the comment. + four_col = [("Acme, Inc.", "Engineer", "NYC", "http://x/1"), + ("Acme, Inc.", "Engineer", "NYC", "http://x/1")] + assert len({job_fingerprint(c, r, l) for c, r, l, _ in four_col}) == 1 + # And it must not over-cover: a different link with the same content is a + # conflict the arbiter is meant to catch. + diff_link = ("Acme, Inc.", "Engineer", "NYC", "http://x/2") + assert job_fingerprint(*diff_link[:3]) in {job_fingerprint(*q[:3]) for q in four_col} + # The SET list must omit the columns the upsert arbitrates, so a conflict # keeps the pre-existing row's identity (notably its link). set_list = flat.split("DO UPDATE SET", 1)[1] From 3c220d6229b2a5f051a52c1efd72d457cf9ee564 Mon Sep 17 00:00:00 2001 From: KSaifStack Date: Tue, 29 Sep 2026 00:57:20 -0400 Subject: [PATCH 07/13] fix(scraper): create the fingerprint index with IF NOT EXISTS, not ADD CONSTRAINT _prod's internships_fingerprint_key is a unique index, not a table constraint, so the pg_constraint guard found nothing to add and the subsequent ADD CONSTRAINT died with psycopg2.errors.DuplicateTable: relation "internships_fingerprint_key" already exists -- the constraint collided with the index of the same name. ON CONFLICT (fingerprint) infers from a unique index just as well as from a constraint, so CREATE UNIQUE INDEX IF NOT EXISTS is both sufficient and idempotent against either form, since a constraint is backed by an index of the same name. --- backend/scraper.py | 52 ++++++++++++++++++++++++---------------------- 1 file changed, 27 insertions(+), 25 deletions(-) diff --git a/backend/scraper.py b/backend/scraper.py index d39f9c7..a65fb3a 100644 --- a/backend/scraper.py +++ b/backend/scraper.py @@ -678,31 +678,33 @@ def _update_database(): if backfilled: print(f" backfilled fingerprint on {backfilled} pre-existing rows", flush=True) - # The upsert arbitrates on (fingerprint), so that unique constraint has - # to exist. Prod has had it since someone added it by hand, which meant - # a fresh database -- or one restored from a dump without it -- failed - # every scrape with "there is no unique or exclusion constraint - # matching the ON CONFLICT specification". Create it here, after the - # backfill, so the rows it covers are already populated. - cursor.execute(""" - SELECT 1 FROM pg_constraint - WHERE conname = 'internships_fingerprint_key' - AND conrelid = 'internships'::regclass - """) - if not cursor.fetchone(): - try: - cursor.execute(""" - ALTER TABLE internships - ADD CONSTRAINT internships_fingerprint_key UNIQUE (fingerprint) - """) - except psycopg2.errors.UniqueViolation: - # Leftover content duplicates. The scrape still dedupes in - # Python via `seen`, so carry on without the constraint rather - # than failing every run until the rows are cleaned up by hand. - cursor.connection.rollback() - print( - " skipped internships_fingerprint_key: duplicate fingerprints " - "still present (dedup continues without it)", + # The upsert arbitrates on (fingerprint), so a unique index over that + # column has to exist -- ON CONFLICT infers from a unique index or an + # exclusion constraint, and from nothing else. Prod has had one since + # someone added it by hand, so a fresh or restored database failed + # every scrape with "no unique or exclusion constraint matching the + # ON CONFLICT specification". Created after the backfill, so the rows + # it covers are already populated. + # + # CREATE UNIQUE INDEX IF NOT EXISTS rather than ADD CONSTRAINT: the + # existing object is an index, so a constraint would collide on the + # name (DuplicateTable) even though the guard found nothing to add. + # IF NOT EXISTS is satisfied by either form, since a constraint is + # backed by an index of the same name. + try: + cursor.execute(""" + CREATE UNIQUE INDEX IF NOT EXISTS internships_fingerprint_key + ON internships(fingerprint) + """) + except psycopg2.errors.UniqueViolation: + # Leftover content duplicates. The scrape still dedupes in Python + # via `seen`, so carry on without the index rather than failing + # every run until the rows are cleaned up by hand. + cursor.connection.rollback() + print( + " skipped the fingerprint unique index: duplicate fingerprints " + "still present (dedup continues without it)", + flush=True, ) From b95a0f69c1e317b9116fccf6853f85a1973401da Mon Sep 17 00:00:00 2001 From: KSaifStack Date: Tue, 29 Sep 2026 01:04:11 -0400 Subject: [PATCH 08/13] chore(scraper): report how many rows the backfill left NULL Sizes the blast radius before deciding how to reconcile them. Rows left NULL are invisible to ON CONFLICT (fingerprint), so a later insert matching one on the four-column constraint is unarbitrated and aborts the scrape. --- backend/scraper.py | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/backend/scraper.py b/backend/scraper.py index a65fb3a..6a5e8c9 100644 --- a/backend/scraper.py +++ b/backend/scraper.py @@ -586,12 +586,20 @@ def _backfill_fingerprints(cursor): if not rows: return 0 updates = [] + skipped = 0 for r in rows: fp = job_fingerprint(r[1], r[2], r[3]) if fp in taken: + skipped += 1 continue taken.add(fp) updates.append((r[0], fp)) + if skipped: + print( + f" {skipped} row(s) share a fingerprint with an existing row and were " + f"left NULL (they are the same job listed twice)", + flush=True, + ) if updates: execute_values( cursor, From 8386e9bd28ef6924f866cd3539ce1d7a3b87e676 Mon Sep 17 00:00:00 2001 From: KSaifStack Date: Tue, 29 Sep 2026 01:17:44 -0400 Subject: [PATCH 09/13] fix(scraper): delete fingerprint-duplicate rows instead of leaving them NULL The 212 rows left NULL by the previous commit are why the scrape then failed on internships_unique_job instead. A NULL is invisible to ON CONFLICT (fingerprint), so an insert matching one of those rows on (company, role, location, link) reaches no arbiter and aborts the run. One error traded for another. Deleting them is the same reconciliation _ensure_schema already does when internships_unique_job's definition changes: keep the lowest id, drop the later copy. The survivor is the row the frontend tracker already resolves to, since a missing fingerprint is a 404 on every tracked job. Nothing references internships by foreign key; job references live in agent_proposals.payload. The SELECT is now ORDER BY id so "keep the oldest" is deterministic instead of depending on heap order. This deletes 212 rows on the next run. They are the same job listed twice by the app's own definition of a job -- identical normalised company, role and location -- which the unique index exists to forbid. --- backend/scraper.py | 41 +++++++++++++++++++---------- backend/test_fingerprint_dedup.py | 43 ++++++++++++++++++------------- 2 files changed, 53 insertions(+), 31 deletions(-) diff --git a/backend/scraper.py b/backend/scraper.py index 6a5e8c9..9cde411 100644 --- a/backend/scraper.py +++ b/backend/scraper.py @@ -569,35 +569,50 @@ def _backfill_fingerprints(cursor): which is the worst possible failure for the feature that motivated this. Idempotent: a no-op once every row is populated. - Greedy, because prod already holds rows whose content normalises to the - same fingerprint ("Acme, Inc." alongside "Acme Inc."), and - internships_fingerprint_key forbids two rows sharing one. Blindly filling - every NULL row aborted the whole scrape on - psycopg2.errors.UniqueViolation. Rows whose fingerprint is already taken - are left NULL instead: Postgres treats NULLs as distinct, so they neither - violate the index nor block later runs. + Greedy, because prod holds 212 rows whose content normalises to a + fingerprint another row already holds ("Acme, Inc." alongside "Acme Inc."), + and the unique index over `fingerprint` forbids two rows sharing one. + Filling every NULL row blindly aborted the scrape on + psycopg2.errors.UniqueViolation. + + Those collisions are deleted rather than left NULL. Leaving them NULL was + tried first and does not work: a NULL is invisible to ON CONFLICT + (fingerprint), so a later insert matching one of those rows on + (company, role, location, link) found no arbiter and died on + internships_unique_job instead. One error traded for another. + + Deleting is the same reconciliation _ensure_schema already performs when + internships_unique_job's definition changes -- keep the lowest id, drop the + later copy. The surviving row is the one the frontend tracker already + resolved to, since a drifted or absent fingerprint is a 404 on every + tracked job. Nothing references internships by foreign key; job references + live in agent_proposals.payload as JSONB. """ cursor.execute("SELECT fingerprint FROM internships WHERE fingerprint IS NOT NULL") taken = {r[0] for r in cursor.fetchall()} + # ORDER BY id so "keep the oldest" is deterministic rather than dependent + # on whatever order the heap happens to return. cursor.execute( - "SELECT id, company, role, location FROM internships WHERE fingerprint IS NULL" + "SELECT id, company, role, location FROM internships " + "WHERE fingerprint IS NULL ORDER BY id" ) rows = cursor.fetchall() if not rows: return 0 updates = [] - skipped = 0 + doomed = [] for r in rows: fp = job_fingerprint(r[1], r[2], r[3]) if fp in taken: - skipped += 1 + doomed.append(r[0]) continue taken.add(fp) updates.append((r[0], fp)) - if skipped: + if doomed: + cursor.execute("DELETE FROM internships WHERE id = ANY(%s)", (doomed,)) print( - f" {skipped} row(s) share a fingerprint with an existing row and were " - f"left NULL (they are the same job listed twice)", + f" deleted {len(doomed)} duplicate row(s) sharing a fingerprint " + f"with an existing row (same job listed twice)", flush=True, ) if updates: diff --git a/backend/test_fingerprint_dedup.py b/backend/test_fingerprint_dedup.py index 040c7c2..8d9e3a2 100644 --- a/backend/test_fingerprint_dedup.py +++ b/backend/test_fingerprint_dedup.py @@ -84,20 +84,22 @@ def main(): assert k not in seen, f"over-merged distinct jobs: {company}/{role}/{loc}" seen.add(k) - # The backfill must be greedy, not blind. Prod already holds content - # duplicates, and internships_fingerprint_key forbids two rows sharing a - # fingerprint -- filling every NULL row blindly raised UniqueViolation and - # killed the scrape. - def greedy_backfill(rows, taken=frozenset()): + # The backfill must delete content duplicates, not leave them NULL. A NULL + # is invisible to ON CONFLICT (fingerprint), so an insert matching such a + # row on (company, role, location, link) finds no arbiter and dies on + # internships_unique_job. Prod had 212 of these. + def backfill(rows, taken=frozenset()): + """Returns (updates, doomed). Mirrors _backfill_fingerprints().""" taken = set(taken) - out = [] - for rid, company, role, loc in rows: + updates, doomed = [], [] + for rid, company, role, loc in rows: # caller supplies ORDER BY id fp = job_fingerprint(company, role, loc) if fp in taken: + doomed.append(rid) continue taken.add(fp) - out.append((rid, fp)) - return out + updates.append((rid, fp)) + return updates, doomed null_rows = [ (1, "Acme, Inc.", "Engineer", "NYC"), @@ -105,15 +107,20 @@ def greedy_backfill(rows, taken=frozenset()): (3, "Widgets, LLC", "Engineer", "NYC"), (4, "Widgets LLC", "Engineer", "NYC"), # same fingerprint as row 3 ] - got = greedy_backfill(null_rows) - assert [r[0] for r in got] == [1, 3], got - assert len({fp for _, fp in got}) == len(got), "backfill emitted a duplicate" - - # A NULL row colliding with an already-populated row must also be skipped. - seeded = greedy_backfill(null_rows, taken={job_fingerprint("Acme Inc.", "Engineer", "NYC")}) - assert 1 not in [r[0] for r in seeded], seeded - - print("OK: seen-key and fingerprint agree; distinct jobs kept; backfill greedy") + updates, doomed = backfill(null_rows) + assert [r[0] for r in updates] == [1, 3], updates + assert doomed == [2, 4], doomed + assert len({fp for _, fp in updates}) == len(updates), "backfill emitted a duplicate" + # No surviving row may be left NULL, or it becomes the next abort. + assert not (set(doomed) & {r[0] for r in updates}) + + # A collision with an already-populated row is deleted too, not skipped. + seeded_fp = job_fingerprint("Acme Inc.", "Engineer", "NYC") + updates, doomed = backfill(null_rows, taken={seeded_fp}) + assert 1 in doomed, (updates, doomed) + assert 3 in [r[0] for r in updates], updates + + print("OK: seen-key and fingerprint agree; distinct jobs kept; dupes deleted") if __name__ == "__main__": From b54d857ca7ef31800f689e343c18797c4ede8b19 Mon Sep 17 00:00:00 2001 From: KSaifStack Date: Tue, 29 Sep 2026 01:21:54 -0400 Subject: [PATCH 10/13] chore(scraper): report the unique indexes and NULL-fingerprint count each run Guessing at the arbiter from a UniqueViolation three layers up wasted four runs. The fingerprint index is the load-bearing piece of the write path; print it and the remaining NULL count once per run. --- backend/scraper.py | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/backend/scraper.py b/backend/scraper.py index 9cde411..804cf31 100644 --- a/backend/scraper.py +++ b/backend/scraper.py @@ -701,6 +701,22 @@ def _update_database(): if backfilled: print(f" backfilled fingerprint on {backfilled} pre-existing rows", flush=True) + # What the upsert is actually arbitrating against. Worth one line per + # run: the fingerprint arbiter is the load-bearing piece of the write + # path, and "is it unique, and is it partial" is not something you want + # to infer from a UniqueViolation three layers up. + cursor.execute(""" + SELECT indexname, indexdef FROM pg_indexes + WHERE tablename = 'internships' AND indexdef ILIKE '%UNIQUE%' + ORDER BY indexname + """) + for name, d in cursor.fetchall(): + print(f" unique: {d}", flush=True) + cursor.execute( + "SELECT count(*) FROM internships WHERE fingerprint IS NULL" + ) + print(f" rows still lacking a fingerprint: {cursor.fetchone()[0]}", flush=True) + # The upsert arbitrates on (fingerprint), so a unique index over that # column has to exist -- ON CONFLICT infers from a unique index or an # exclusion constraint, and from nothing else. Prod has had one since From 0661ce9759cdde36f4ee08a84ddcd44b87f9656f Mon Sep 17 00:00:00 2001 From: KSaifStack Date: Tue, 29 Sep 2026 01:26:34 -0400 Subject: [PATCH 11/13] chore(scraper): print the losing row when the fingerprint arbiter is bypassed A constraint firing means ON CONFLICT (fingerprint) matched nothing, which for an identical 4-tuple can only be true if the stored row's fingerprint differs from the computed one. Print it instead of spending a run guessing. --- backend/scraper.py | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/backend/scraper.py b/backend/scraper.py index 804cf31..6ff07a9 100644 --- a/backend/scraper.py +++ b/backend/scraper.py @@ -802,6 +802,24 @@ def _update_database(): # A failed write is not a failed source. Recording it and # carrying on would report a successful run over a # half-written table, and the purge would run. + # + # When the arbiter is bypassed, the losing row is the + # only thing that explains it: a constraint firing means + # ON CONFLICT (fingerprint) found nothing, and that is + # only true if the stored row's fingerprint differs from + # the one we just computed for an identical 4-tuple. + # Print it rather than guess at a fifth run. + detail = getattr(getattr(e, "diag", None), "detail_text", "") or "" + m = re.search(r"link=\((.*?)\)\s+already exists", detail, re.S) + if m: + cursor.connection.rollback() # statement aborted it + cursor.execute( + "SELECT id, company, role, location, fingerprint " + "FROM internships WHERE link = %s", + (m.group(1),), + ) + for row in cursor.fetchall(): + print(f" ! stored row that already owns this link: {row}", flush=True) raise # One 429 or one empty source must not abort the run — that # skips the upsert for every other source too. Recorded and From 37195c3fb271f1cdb3123e2227406f1746e1b9e0 Mon Sep 17 00:00:00 2001 From: KSaifStack Date: Tue, 29 Sep 2026 07:52:07 -0400 Subject: [PATCH 12/13] chore(scraper): count rows whose stored fingerprint disagrees with the normaliser Non-NULL is not the same as correct. The upsert arbitrates on (fingerprint) and nothing else, so a row whose stored fingerprint is not what the current normaliser computes for its own columns is invisible to that arbiter while still holding (company, role, location, link) -- an insert matching it there aborts the run with UniqueViolation on internships_unique_job. Also replaces the write-failure diagnostic's DETAIL regex with a verbatim print: the pattern was anchored on "link=(", which never appears, so it matched nothing on the one run it existed for and the diagnostic was silent. --- backend/scraper.py | 52 ++++++++++++++++++++++++++++++++-------------- 1 file changed, 36 insertions(+), 16 deletions(-) diff --git a/backend/scraper.py b/backend/scraper.py index 6ff07a9..c3f5a4b 100644 --- a/backend/scraper.py +++ b/backend/scraper.py @@ -717,6 +717,26 @@ def _update_database(): ) print(f" rows still lacking a fingerprint: {cursor.fetchone()[0]}", flush=True) + # Non-NULL is not the same as correct. The upsert arbitrates on + # (fingerprint) and nothing else, so a row whose stored fingerprint is + # not what the current normaliser computes for its own columns is + # invisible to the arbiter while still sitting on + # internships_unique_job -- an insert matching it on the four columns + # then aborts the whole run. Recompute every row and count the + # disagreements, so this is visible on every run instead of inferred + # from a UniqueViolation three layers up. Pure Python on purpose: the + # backfill above already argued why a SQL translation of these regexes + # would drift from job_fingerprint. + cursor.execute("SELECT id, company, role, location, fingerprint FROM internships") + drifted = [ + (rid, fp, job_fingerprint(co, ro, lo)) + for rid, co, ro, lo, fp in cursor.fetchall() + if fp != job_fingerprint(co, ro, lo) + ] + print(f" rows whose fingerprint disagrees with the normaliser: {len(drifted)}", flush=True) + for rid, fp, want in drifted[:5]: + print(f" id={rid} stored={fp!r} normaliser={want!r}", flush=True) + # The upsert arbitrates on (fingerprint), so a unique index over that # column has to exist -- ON CONFLICT infers from a unique index or an # exclusion constraint, and from nothing else. Prod has had one since @@ -803,23 +823,23 @@ def _update_database(): # carrying on would report a successful run over a # half-written table, and the purge would run. # - # When the arbiter is bypassed, the losing row is the - # only thing that explains it: a constraint firing means - # ON CONFLICT (fingerprint) found nothing, and that is - # only true if the stored row's fingerprint differs from - # the one we just computed for an identical 4-tuple. - # Print it rather than guess at a fifth run. + # Print the server's own DETAIL verbatim. An earlier + # attempt parsed it with a regex anchored on "link=(", + # which never appears -- Postgres writes + # "Key (company, role, location, link)=(...) already + # exists", so the parse silently matched nothing and + # the diagnostic printed nothing on the one run it + # existed for. The server states the conflict better + # than a parser would. detail = getattr(getattr(e, "diag", None), "detail_text", "") or "" - m = re.search(r"link=\((.*?)\)\s+already exists", detail, re.S) - if m: - cursor.connection.rollback() # statement aborted it - cursor.execute( - "SELECT id, company, role, location, fingerprint " - "FROM internships WHERE link = %s", - (m.group(1),), - ) - for row in cursor.fetchall(): - print(f" ! stored row that already owns this link: {row}", flush=True) + if detail: + print(f" ! write conflict -- {detail.strip()}", flush=True) + print( + " ! see the fingerprint-drift count above: a stored " + "fingerprint that is not the normaliser's own value " + "for its row is invisible to ON CONFLICT (fingerprint)", + flush=True, + ) raise # One 429 or one empty source must not abort the run — that # skips the upsert for every other source too. Recorded and From d023fba2762912d53717ecda252e64089e99bf1f Mon Sep 17 00:00:00 2001 From: KSaifStack Date: Tue, 29 Sep 2026 07:54:57 -0400 Subject: [PATCH 13/13] fix(scraper): reconcile every row's fingerprint, not just the NULL ones Non-NULL is not the same as correct, and a wrong fingerprint is a hard failure rather than a cosmetic one. The upsert arbitrates on (fingerprint) and nothing else, so a row whose stored value is not what job_fingerprint computes for its own columns is invisible to that arbiter while still holding (company, role, location, link) -- an insert matching it there aborts the whole run on internships_unique_job. That is what killed every scrape after the 212 NULL rows were reconciled: 52 more rows hold a normaliser's value that exists nowhere in this repo, which kept punctuation as a space and collapsed NYC to ny. The pass is now seeded with nothing and walks every row in id order, so "already taken" means an earlier row rather than a pre-existing value. A correct row is compared and skipped, so the extra cost is one comparison. --- backend/scraper.py | 80 +++++++++++++++---------------- backend/test_fingerprint_dedup.py | 77 +++++++++++++++++++++-------- 2 files changed, 96 insertions(+), 61 deletions(-) diff --git a/backend/scraper.py b/backend/scraper.py index c3f5a4b..ca4f0a1 100644 --- a/backend/scraper.py +++ b/backend/scraper.py @@ -559,7 +559,7 @@ def _flush(cursor, batch, run_time): def _backfill_fingerprints(cursor): - """Fill fingerprint on rows written before the column existed. + """Make every row's fingerprint the normaliser's own value for its columns. In Python rather than SQL, deliberately. The value has to come out of the exact same normaliser the upsert uses, and a SQL translation of those @@ -567,15 +567,25 @@ def _backfill_fingerprints(cursor): single btrim over the joined string, for one. A mismatched fingerprint is not a crash; it is a 404 on every tracked job until that row is re-scraped, which is the worst possible failure for the feature that motivated this. - Idempotent: a no-op once every row is populated. - Greedy, because prod holds 212 rows whose content normalises to a - fingerprint another row already holds ("Acme, Inc." alongside "Acme Inc."), - and the unique index over `fingerprint` forbids two rows sharing one. - Filling every NULL row blindly aborted the scrape on - psycopg2.errors.UniqueViolation. - - Those collisions are deleted rather than left NULL. Leaving them NULL was + Every row, not just the NULL ones. Non-NULL is not the same as correct, and + being wrong is a hard failure rather than a cosmetic one: the upsert + arbitrates on (fingerprint) and nothing else, so a row whose stored + fingerprint is not what job_fingerprint computes for its own columns is + invisible to that arbiter while still holding + (company, role, location, link). An insert matching it there aborts the + run on internships_unique_job — which is exactly how 52 rows written by a + normaliser that does not exist anywhere in this repo ("intelligent + creation, camera" kept its space, "NYC" collapsed to "ny") killed every + scrape after the backfill was thought finished. Recomputing is a no-op for + a correct row, so the cost of covering them is one comparison. + + Greedy, because prod holds rows whose content normalises to a fingerprint + another row already holds ("Acme, Inc." alongside "Acme Inc."), and the + unique index over `fingerprint` forbids two rows sharing one. Filling every + row blindly aborted the scrape on psycopg2.errors.UniqueViolation. + + Those collisions are deleted rather than left stale. Leaving them NULL was tried first and does not work: a NULL is invisible to ON CONFLICT (fingerprint), so a later insert matching one of those rows on (company, role, location, link) found no arbiter and died on @@ -588,26 +598,31 @@ def _backfill_fingerprints(cursor): tracked job. Nothing references internships by foreign key; job references live in agent_proposals.payload as JSONB. """ - cursor.execute("SELECT fingerprint FROM internships WHERE fingerprint IS NOT NULL") - taken = {r[0] for r in cursor.fetchall()} # ORDER BY id so "keep the oldest" is deterministic rather than dependent # on whatever order the heap happens to return. cursor.execute( - "SELECT id, company, role, location FROM internships " - "WHERE fingerprint IS NULL ORDER BY id" + "SELECT id, company, role, location, fingerprint FROM internships ORDER BY id" ) rows = cursor.fetchall() if not rows: return 0 updates = [] doomed = [] - for r in rows: - fp = job_fingerprint(r[1], r[2], r[3]) + # Assigned in id order, seeded with nothing: every row passes through, so + # "already taken" means an earlier row, not a pre-existing value. Seeding + # from the stored column instead is what let a drifted row block its own + # correction. + taken = set() + for rid, company, role, loc, stored in rows: + fp = job_fingerprint(company, role, loc) + if fp == stored: + taken.add(fp) + continue if fp in taken: - doomed.append(r[0]) + doomed.append(rid) continue taken.add(fp) - updates.append((r[0], fp)) + updates.append((rid, fp)) if doomed: cursor.execute("DELETE FROM internships WHERE id = ANY(%s)", (doomed,)) print( @@ -699,7 +714,11 @@ def _update_database(): backfilled = _backfill_fingerprints(cursor) if backfilled: - print(f" backfilled fingerprint on {backfilled} pre-existing rows", flush=True) + print( + f" reconciled fingerprint on {backfilled} row(s) whose stored " + f"value was not the normaliser's own value for their columns", + flush=True, + ) # What the upsert is actually arbitrating against. Worth one line per # run: the fingerprint arbiter is the load-bearing piece of the write @@ -717,26 +736,6 @@ def _update_database(): ) print(f" rows still lacking a fingerprint: {cursor.fetchone()[0]}", flush=True) - # Non-NULL is not the same as correct. The upsert arbitrates on - # (fingerprint) and nothing else, so a row whose stored fingerprint is - # not what the current normaliser computes for its own columns is - # invisible to the arbiter while still sitting on - # internships_unique_job -- an insert matching it on the four columns - # then aborts the whole run. Recompute every row and count the - # disagreements, so this is visible on every run instead of inferred - # from a UniqueViolation three layers up. Pure Python on purpose: the - # backfill above already argued why a SQL translation of these regexes - # would drift from job_fingerprint. - cursor.execute("SELECT id, company, role, location, fingerprint FROM internships") - drifted = [ - (rid, fp, job_fingerprint(co, ro, lo)) - for rid, co, ro, lo, fp in cursor.fetchall() - if fp != job_fingerprint(co, ro, lo) - ] - print(f" rows whose fingerprint disagrees with the normaliser: {len(drifted)}", flush=True) - for rid, fp, want in drifted[:5]: - print(f" id={rid} stored={fp!r} normaliser={want!r}", flush=True) - # The upsert arbitrates on (fingerprint), so a unique index over that # column has to exist -- ON CONFLICT infers from a unique index or an # exclusion constraint, and from nothing else. Prod has had one since @@ -835,9 +834,10 @@ def _update_database(): if detail: print(f" ! write conflict -- {detail.strip()}", flush=True) print( - " ! see the fingerprint-drift count above: a stored " + " ! if the reconcile count above was 0, a stored " "fingerprint that is not the normaliser's own value " - "for its row is invisible to ON CONFLICT (fingerprint)", + "for its row is what is invisible to " + "ON CONFLICT (fingerprint)", flush=True, ) raise diff --git a/backend/test_fingerprint_dedup.py b/backend/test_fingerprint_dedup.py index 8d9e3a2..80d9ac4 100644 --- a/backend/test_fingerprint_dedup.py +++ b/backend/test_fingerprint_dedup.py @@ -84,16 +84,22 @@ def main(): assert k not in seen, f"over-merged distinct jobs: {company}/{role}/{loc}" seen.add(k) - # The backfill must delete content duplicates, not leave them NULL. A NULL - # is invisible to ON CONFLICT (fingerprint), so an insert matching such a - # row on (company, role, location, link) finds no arbiter and dies on - # internships_unique_job. Prod had 212 of these. - def backfill(rows, taken=frozenset()): - """Returns (updates, doomed). Mirrors _backfill_fingerprints().""" - taken = set(taken) - updates, doomed = [], [] - for rid, company, role, loc in rows: # caller supplies ORDER BY id + # The backfill must delete content duplicates, not leave them stale. A + # stale or NULL fingerprint is invisible to ON CONFLICT (fingerprint), so + # an insert matching such a row on (company, role, location, link) finds no + # arbiter and dies on internships_unique_job. Prod had 212 of these, and 52 + # more that were non-NULL but held a foreign normaliser's value. + def reconcile(rows): + """Returns (updates, doomed). Mirrors _backfill_fingerprints(). + + `rows` is (id, company, role, location, stored_fingerprint) in id order. + """ + updates, doomed, taken = [], [], set() + for rid, company, role, loc, stored in rows: fp = job_fingerprint(company, role, loc) + if fp == stored: + taken.add(fp) + continue if fp in taken: doomed.append(rid) continue @@ -102,23 +108,52 @@ def backfill(rows, taken=frozenset()): return updates, doomed null_rows = [ - (1, "Acme, Inc.", "Engineer", "NYC"), - (2, "Acme Inc.", "Engineer", "NYC"), # same fingerprint as row 1 - (3, "Widgets, LLC", "Engineer", "NYC"), - (4, "Widgets LLC", "Engineer", "NYC"), # same fingerprint as row 3 + (1, "Acme, Inc.", "Engineer", "NYC", None), + (2, "Acme Inc.", "Engineer", "NYC", None), # same fingerprint as row 1 + (3, "Widgets, LLC", "Engineer", "NYC", None), + (4, "Widgets LLC", "Engineer", "NYC", None), # same fingerprint as row 3 ] - updates, doomed = backfill(null_rows) + updates, doomed = reconcile(null_rows) assert [r[0] for r in updates] == [1, 3], updates assert doomed == [2, 4], doomed - assert len({fp for _, fp in updates}) == len(updates), "backfill emitted a duplicate" - # No surviving row may be left NULL, or it becomes the next abort. + assert len({fp for _, fp in updates}) == len(updates), "reconcile emitted a duplicate" + # No surviving row may be left stale, or it becomes the next abort. assert not (set(doomed) & {r[0] for r in updates}) - # A collision with an already-populated row is deleted too, not skipped. - seeded_fp = job_fingerprint("Acme Inc.", "Engineer", "NYC") - updates, doomed = backfill(null_rows, taken={seeded_fp}) - assert 1 in doomed, (updates, doomed) - assert 3 in [r[0] for r in updates], updates + # A row whose stored value is already correct is left alone — no write. + correct = job_fingerprint("Acme Inc.", "Engineer", "NYC") + updates, doomed = reconcile([(1, "Acme Inc.", "Engineer", "NYC", correct)]) + assert updates == [] and doomed == [], (updates, doomed) + + # A drifted row -- non-NULL, but not the normaliser's value for its own + # columns -- must be corrected. This is the 52-row case: the old code seeded + # `taken` from the stored column, so a drifted row could not be corrected if + # its stale value were anything but its own fingerprint, and if the stale + # value *was* its own fingerprint the row survived while remaining invisible + # to the arbiter. The DETAIL that proved it: + # stored 'tiktok|software engineer intern intelligent creation camera|san jose ca' + # normaliser 'tiktok|software engineer intern intelligent creationcamera|san jose ca' + drifted = [ + (1, "TikTok", "Software Engineer Intern, Intelligent Creation/Camera", "San Jose, CA", + "tiktok|software engineer intern intelligent creation camera|san jose ca"), + (2, "American Express", "Digital Product Analyst Intern", "NYC", + "american express|digital product analyst intern|ny"), + ] + updates, doomed = reconcile(drifted) + assert [r[0] for r in updates] == [1, 2], updates + assert doomed == [], doomed + # The corrections are the normaliser's values, punctuation deleted rather + # than spaced, and no location aliasing. + assert dict(updates) == { + 1: "tiktok|software engineer intern intelligent creationcamera|san jose ca", + 2: "american express|digital product analyst intern|nyc", + }, updates + + # And a drifted row whose corrected fingerprint is already held by an + # earlier row is deleted, not left holding a value the arbiter can't see. + updates, doomed = reconcile([(1, "Acme Inc.", "Engineer", "NYC", correct), + (2, "Acme, Inc.", "Engineer", "NYC", "acme inc old")]) + assert updates == [] and doomed == [2], (updates, doomed) print("OK: seen-key and fingerprint agree; distinct jobs kept; dupes deleted")