fix(graphrag): return a chunk once from entity_pages

The subject flag was in the GROUP BY, so a chunk two matching entities link --
one the page is about, one merely mentioned in it -- came back as two
identical pages and spent the caller's page budget twice on the same text.
It is aggregated with bool_or now, which is what the ordering wanted anyway.

Covered by a live test against a pgvector-shaped documents table: the graph
tables alone cannot answer this query, so nothing exercised it before.
This commit is contained in:
Alex committed 2026-09-20 10:45:04 +01:00
1 parent ccd8eb612f
commit 2b6d4d509e
2 files changed
+69 -2

No files matched your search

+6 -2
View File
@@ -1265,13 +1265,17 @@ class GraphStore:
sql.SQL(
"""
SELECT d.{metadata}, d.{text},
(lower(n.name) = %s OR lower(n.name) LIKE %s) AS is_subject
bool_or(lower(n.name) = %s OR lower(n.name) LIKE %s) AS is_subject
FROM graph_node_chunks gc
JOIN graph_nodes n ON n.id = gc.node_id
JOIN {table} d ON d.id::text = gc.chunk_id
WHERE gc.source_id = %s AND d.{source} = %s
AND (lower(n.name) = %s OR lower(n.name) LIKE %s OR n.name ILIKE %s)
GROUP BY d.{metadata}, d.{text}, is_subject
-- One page per chunk. Grouping on the subject flag as well
-- split a chunk two entities link -- one naming it, one
-- merely mentioned -- into two identical pages, spending
-- the caller's page budget twice on the same text.
GROUP BY d.{metadata}, d.{text}
ORDER BY is_subject DESC, (d.{text} ILIKE %s) DESC
LIMIT %s;
"""
+63
View File
@@ -19,6 +19,7 @@ import uuid
from unittest.mock import MagicMock, patch
import pytest
from psycopg.types.json import Jsonb
import docsgpt.graphrag.store as store_module
from docsgpt.vectorstore import pgconn
@@ -661,6 +662,68 @@ class TestGraphStoreParameterization:
assert params[-1] == embedding
@pytest.mark.integration
class TestEntityPagesLive:
"""``entity_pages`` against a real pgvector-shaped table.
The graph tables alone cannot answer it: the rows it returns live in the
documents table the sources were ingested into, so the test creates a
minimal one with the same column names ``PGVectorStore`` uses.
"""
@pytest.fixture
def store(self, postgresql):
store = GraphStore(connection_string=_ephemeral_dsn(postgresql.info))
try:
store._ensure_tables()
except Exception as exc:
pytest.skip(f"pgvector extension unavailable: {exc}")
conn = store._get_connection()
cursor = conn.cursor()
cursor.execute(
"""
CREATE TABLE IF NOT EXISTS documents (
id SERIAL PRIMARY KEY,
text TEXT,
metadata JSONB,
source_id TEXT
);
"""
)
conn.commit()
cursor.close()
yield store
store.close()
def test_a_page_linked_by_two_entities_is_returned_once(self, store):
"""One chunk, two nodes whose names both match: an exact hit and a
mention. They differ only in whether the page is *about* the entity, so
grouping on that flag returned the same page twice and spent a quarter
of the page budget on it."""
source_id = str(uuid.uuid4())
conn = store._get_connection()
cursor = conn.cursor()
cursor.execute(
"INSERT INTO documents (text, metadata, source_id) VALUES (%s, %s, %s) RETURNING id;",
("Quill is a write-ahead store.", Jsonb({"title": "quill.md"}), source_id),
)
chunk_id = str(cursor.fetchone()[0])
conn.commit()
cursor.close()
try:
subject = store.upsert_node(source_id, "Quill", "quill")
mention = store.upsert_node(source_id, "Legacy Quill", "legacy quill")
store.link_node_chunk(source_id, subject, chunk_id)
store.link_node_chunk(source_id, mention, chunk_id)
pages = store.entity_pages(source_id, "Quill", limit=4)
assert [page["text"] for page in pages] == ["Quill is a write-ahead store."]
assert pages[0]["metadata"] == {"title": "quill.md"}
finally:
store.delete_by_source(source_id)
@pytest.mark.unit
class TestGraphReadQueries:
"""The reads behind fact seeding and the agent's graph tool, without a DB.