Merge pull request #2087 from KeilerHirsch/fix/repair-sparse-embedding-metadata

fix(repair): stop dropping drawers with zero embedding_metadata rows
This commit is contained in:
Igor Lins e Silva 2026-08-02 05:24:27 -03:00 committed by GitHub
commit cf2ebce7ef
No known key found for this signature in database
GPG Key ID: B5690EEEBB952194
2 changed files with 61 additions and 3 deletions

View File

@ -1310,6 +1310,13 @@ def extract_via_sqlite(palace_path: str, collection_name: str) -> Iterator[tuple
returned as the document; this matches how chromadb itself stores
``add(documents=...)``.
Driven from ``embeddings`` (LEFT JOIN ``embedding_metadata``), not
the other way around: an embedding with zero ``embedding_metadata``
rows a sparse historical write with no ``chroma:document`` and no
other key, the same condition ``_extract_drawers`` already sanitizes
for the collection-layer path, see #1458 — must still be yielded
with an empty metadata dict, not silently excluded by the join.
Silent on missing palace, missing ``chroma.sqlite3``, or unknown
collection name yields nothing. Callers that need to distinguish
"empty collection" from "collection not present" should query
@ -1339,15 +1346,21 @@ def extract_via_sqlite(palace_path: str, collection_name: str) -> Iterator[tuple
"""
SELECT e.embedding_id, em.key, em.string_value, em.int_value,
em.float_value, em.bool_value
FROM embedding_metadata em
JOIN embeddings e ON em.id = e.id
FROM embeddings e
LEFT JOIN embedding_metadata em ON em.id = e.id
WHERE e.segment_id = ?
ORDER BY em.id
ORDER BY e.id
""",
(segment_id,),
):
if emb_id not in per_id:
order.append(emb_id)
if key is None:
# LEFT JOIN unmatched row: this embedding has zero
# embedding_metadata rows. `order`/`per_id` already
# account for it via the defaultdict below; nothing to
# merge for this row.
continue
if sv is not None:
per_id[emb_id][key] = sv
elif iv is not None:

View File

@ -1841,6 +1841,51 @@ def test_extract_via_sqlite_missing_palace_yields_nothing(tmp_path):
assert list(repair.extract_via_sqlite(str(empty), "mempalace_drawers")) == []
def test_extract_via_sqlite_yields_rows_with_zero_metadata_rows(tmp_path):
"""A drawer whose embedding has ZERO rows in ``embedding_metadata``
(no ``chroma:document``, no other key) must still be yielded, not
silently dropped.
This is the same "sparse historical write" condition ``_extract_drawers``
already sanitizes for the collection-layer rebuild path (see the
``sanitized_metas`` comment above, and #1458) — current chromadb
validates against empty metadata on write, so this only happens to
rows already sitting in an older palace. Modeled here by seeding a
real chromadb collection, then stripping one drawer's metadata rows
directly via SQLite, mirroring how such a row actually looks on disk.
An INNER JOIN driven from ``embedding_metadata`` can never see a row
with zero metadata rows: it must be driven from ``embeddings`` instead.
"""
rows = [
("drawer_ok", "normal document", {"wing": "w"}),
("drawer_sparse", "will lose all its metadata rows", {"wing": "w"}),
]
_seed_palace(tmp_path, "mempalace_drawers", rows)
sqlite_path = os.path.join(str(tmp_path), "chroma.sqlite3")
conn = sqlite3.connect(sqlite_path)
try:
emb_row = conn.execute(
"SELECT id FROM embeddings WHERE embedding_id = ?", ("drawer_sparse",)
).fetchone()
assert emb_row is not None, "seed helper didn't create the expected embedding row"
conn.execute("DELETE FROM embedding_metadata WHERE id = ?", (emb_row[0],))
conn.commit()
finally:
conn.close()
extracted = list(repair.extract_via_sqlite(str(tmp_path), "mempalace_drawers"))
ids = {emb_id for emb_id, _doc, _meta in extracted}
assert "drawer_sparse" in ids, (
"extract_via_sqlite silently dropped a drawer with zero "
"embedding_metadata rows — the INNER JOIN starting at "
"embedding_metadata structurally excludes it"
)
assert "drawer_ok" in ids
def test_rebuild_from_sqlite_roundtrips_via_real_chromadb(tmp_path):
"""End-to-end: seed source palace, rebuild into a fresh dest, then
open dest with a fresh ChromaBackend and verify ``count()`` and