extract() no longer duplicates NULL-containing rows in shared lookups

The lookup-table insert relied on INSERT OR IGNORE and the unique index
to dedupe against existing rows, but SQLite unique indexes treat NULLs
as distinct - extracting a second table into the same lookup table
re-inserted every NULL-containing value, growing orphan rows on each
extract. The insert now also has an IS-based NOT EXISTS guard, matching
how the foreign keys themselves are resolved.

Refs https://github.com/simonw/sqlite-utils/issues/769#issuecomment-4900034150

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Simon Willison 2026-07-06 21:55:21 -07:00
commit 8572d1e39c
3 changed files with 49 additions and 1 deletions

View file

@ -271,3 +271,34 @@ def test_extract_null_values_existing_lookup_table_with_null_row(fresh_db):
{"id": 10, "name": "Terriana", "species_id": 2},
{"id": 11, "name": "Spenidorm", "species_id": None},
]
def test_extract_repeated_into_shared_lookup_with_nulls(fresh_db):
# Unique indexes treat NULLs as distinct, so INSERT OR IGNORE alone
# cannot dedupe NULL-containing rows against the existing lookup
# table - extracting a second table into the same lookup previously
# inserted duplicate rows that nothing pointed to
fresh_db["t1"].insert_all(
[
{"id": 1, "species": None, "common": "X"},
{"id": 2, "species": "Oak", "common": "Oak"},
],
pk="id",
)
fresh_db["t2"].insert_all([{"id": 1, "species": None, "common": "X"}], pk="id")
fresh_db["t1"].extract(["species", "common"], table="lk")
fresh_db["t2"].extract(["species", "common"], table="lk")
assert fresh_db["lk"].count == 2
# Both tables point at the same lookup row
t1_fk = fresh_db.execute("select lk_id from t1 where id = 1").fetchone()[0]
t2_fk = fresh_db.execute("select lk_id from t2 where id = 1").fetchone()[0]
assert t1_fk == t2_fk
def test_extract_repeated_into_shared_lookup_no_nulls(fresh_db):
# Non-NULL rows were already deduped by the unique index - keep it so
fresh_db["t1"].insert_all([{"id": 1, "species": "Oak"}], pk="id")
fresh_db["t2"].insert_all([{"id": 1, "species": "Oak"}], pk="id")
fresh_db["t1"].extract(["species"], table="lk")
fresh_db["t2"].extract(["species"], table="lk")
assert fresh_db["lk"].count == 1