mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-11 03:12:55 +00:00
fix(graphrag): return a chunk once from entity_pages
The subject flag was in the GROUP BY, so a chunk two matching entities link -- one the page is about, one merely mentioned in it -- came back as two identical pages and spent the caller's page budget twice on the same text. It is aggregated with bool_or now, which is what the ordering wanted anyway. Covered by a live test against a pgvector-shaped documents table: the graph tables alone cannot answer this query, so nothing exercised it before.
This commit is contained in:
1 parent
ccd8eb612f
commit
2b6d4d509e
2 files changed
+69
-2
No files matched your search
@@ -1265,13 +1265,17 @@ class GraphStore:
|
||||
sql.SQL(
|
||||
"""
|
||||
SELECT d.{metadata}, d.{text},
|
||||
(lower(n.name) = %s OR lower(n.name) LIKE %s) AS is_subject
|
||||
bool_or(lower(n.name) = %s OR lower(n.name) LIKE %s) AS is_subject
|
||||
FROM graph_node_chunks gc
|
||||
JOIN graph_nodes n ON n.id = gc.node_id
|
||||
JOIN {table} d ON d.id::text = gc.chunk_id
|
||||
WHERE gc.source_id = %s AND d.{source} = %s
|
||||
AND (lower(n.name) = %s OR lower(n.name) LIKE %s OR n.name ILIKE %s)
|
||||
GROUP BY d.{metadata}, d.{text}, is_subject
|
||||
-- One page per chunk. Grouping on the subject flag as well
|
||||
-- split a chunk two entities link -- one naming it, one
|
||||
-- merely mentioned -- into two identical pages, spending
|
||||
-- the caller's page budget twice on the same text.
|
||||
GROUP BY d.{metadata}, d.{text}
|
||||
ORDER BY is_subject DESC, (d.{text} ILIKE %s) DESC
|
||||
LIMIT %s;
|
||||
"""
|
||||
|
||||
@@ -19,6 +19,7 @@ import uuid
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
from psycopg.types.json import Jsonb
|
||||
|
||||
import docsgpt.graphrag.store as store_module
|
||||
from docsgpt.vectorstore import pgconn
|
||||
@@ -661,6 +662,68 @@ class TestGraphStoreParameterization:
|
||||
assert params[-1] == embedding
|
||||
|
||||
|
||||
@pytest.mark.integration
|
||||
class TestEntityPagesLive:
|
||||
"""``entity_pages`` against a real pgvector-shaped table.
|
||||
|
||||
The graph tables alone cannot answer it: the rows it returns live in the
|
||||
documents table the sources were ingested into, so the test creates a
|
||||
minimal one with the same column names ``PGVectorStore`` uses.
|
||||
"""
|
||||
|
||||
@pytest.fixture
|
||||
def store(self, postgresql):
|
||||
store = GraphStore(connection_string=_ephemeral_dsn(postgresql.info))
|
||||
try:
|
||||
store._ensure_tables()
|
||||
except Exception as exc:
|
||||
pytest.skip(f"pgvector extension unavailable: {exc}")
|
||||
conn = store._get_connection()
|
||||
cursor = conn.cursor()
|
||||
cursor.execute(
|
||||
"""
|
||||
CREATE TABLE IF NOT EXISTS documents (
|
||||
id SERIAL PRIMARY KEY,
|
||||
text TEXT,
|
||||
metadata JSONB,
|
||||
source_id TEXT
|
||||
);
|
||||
"""
|
||||
)
|
||||
conn.commit()
|
||||
cursor.close()
|
||||
yield store
|
||||
store.close()
|
||||
|
||||
def test_a_page_linked_by_two_entities_is_returned_once(self, store):
|
||||
"""One chunk, two nodes whose names both match: an exact hit and a
|
||||
mention. They differ only in whether the page is *about* the entity, so
|
||||
grouping on that flag returned the same page twice and spent a quarter
|
||||
of the page budget on it."""
|
||||
source_id = str(uuid.uuid4())
|
||||
conn = store._get_connection()
|
||||
cursor = conn.cursor()
|
||||
cursor.execute(
|
||||
"INSERT INTO documents (text, metadata, source_id) VALUES (%s, %s, %s) RETURNING id;",
|
||||
("Quill is a write-ahead store.", Jsonb({"title": "quill.md"}), source_id),
|
||||
)
|
||||
chunk_id = str(cursor.fetchone()[0])
|
||||
conn.commit()
|
||||
cursor.close()
|
||||
try:
|
||||
subject = store.upsert_node(source_id, "Quill", "quill")
|
||||
mention = store.upsert_node(source_id, "Legacy Quill", "legacy quill")
|
||||
store.link_node_chunk(source_id, subject, chunk_id)
|
||||
store.link_node_chunk(source_id, mention, chunk_id)
|
||||
|
||||
pages = store.entity_pages(source_id, "Quill", limit=4)
|
||||
|
||||
assert [page["text"] for page in pages] == ["Quill is a write-ahead store."]
|
||||
assert pages[0]["metadata"] == {"title": "quill.md"}
|
||||
finally:
|
||||
store.delete_by_source(source_id)
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
class TestGraphReadQueries:
|
||||
"""The reads behind fact seeding and the agent's graph tool, without a DB.
|
||||
|
||||
Reference in new issue
Block a user