mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-11 03:12:55 +00:00
fix(graphrag): fold an entity's singular and plural onto one key
canonical_name dropped "es" from every -ches/-ses/-zes plural, so "caches" became "cach" while "cache" stayed "cache" -- the singular and plural landed on two nodes, which is the split the function exists to prevent. The same rule split the words documentation uses most: databases/database, responses/ response, releases/release, sizes/size. An "-es" plural cannot say whether it is cache + "s" or batch + "es", so instead of guessing, both sides now meet at the stem: the singular endings -che/-she/-se/-ze/-xe drop their "e" the way the plurals drop "es", and a singular's -ie folds to -y as -ies already did (cookie/cookies). The key is a merge key that is never shown, so it only has to agree, not be a word. alias, canvas, atlas and bias join the words that only look plural. Found by the naming tests CI was missing: the module had no direct tests.
This commit is contained in:
1 parent
5285bb4115
commit
022cf69b49
2 files changed
+118
-9
No files matched your search
@@ -13,6 +13,9 @@ plural. Cautious matters: this corpus contains ``postgres``, ``kubernetes``,
|
||||
``https`` and ``aws``, none of which are plurals, so a naive "strip trailing s"
|
||||
would corrupt them into new entities rather than merge anything.
|
||||
|
||||
The result is a merge key, never shown to anyone, so it only has to be the
|
||||
same for a word's singular and plural — not to be a word itself.
|
||||
|
||||
Always on: every graph is built with canonical names.
|
||||
"""
|
||||
|
||||
@@ -33,18 +36,30 @@ _NOT_PLURAL = frozenset(
|
||||
"analysis", "basis", "axis", "https", "rss", "less", "express",
|
||||
"redis", "nats", "kibana", "elasticsearch", "os", "ios", "macos",
|
||||
"always", "sometimes", "series", "docs", "ops", "devops", "sse",
|
||||
"alias", "canvas", "atlas", "bias", "pandas",
|
||||
}
|
||||
)
|
||||
|
||||
#: Plural endings that drop ``es``, and the singular endings that meet them.
|
||||
#: ``caches`` cannot say whether it is ``cache`` + "s" or ``cach`` + "es"
|
||||
#: (as ``batches`` is ``batch`` + "es"), so rather than guess, both
|
||||
#: ``caches`` and ``cache`` fold to ``cach`` — as ``databases``/``database``
|
||||
#: fold to ``databas``. Hardly any real word differs from one of these singulars
|
||||
#: by its final "e" alone, so the fold merges next to nothing it should not.
|
||||
_ES_PLURAL = ("ches", "shes", "ses", "zes", "xes")
|
||||
_E_SINGULAR = ("che", "she", "se", "ze", "xe")
|
||||
|
||||
|
||||
def _singular(word: str) -> str:
|
||||
"""Best-effort singular of one word, biased hard towards leaving it alone.
|
||||
"""Fold one word so its singular and plural share a key, else leave it alone.
|
||||
|
||||
Only the endings that are unambiguous in this domain are touched:
|
||||
``-ies`` -> ``-y`` (``policies``), ``-ses``/``-xes``/``-zes``/``-ches``/
|
||||
``-shes`` -> drop ``es`` (``indexes``, ``batches``), and a bare trailing
|
||||
``s`` on a word long enough to be safe. Everything in :data:`_NOT_PLURAL`,
|
||||
and anything ending in ``ss``/``us``/``is``, is returned unchanged.
|
||||
``-ies`` and a singular's ``-ie`` both fold to ``-y`` (``policies``,
|
||||
``cookies``/``cookie``). The ``-es`` endings in :data:`_ES_PLURAL` drop
|
||||
``es`` and the singular endings in :data:`_E_SINGULAR` drop their ``e``, so
|
||||
both sides of an ambiguous plural meet (``caches``/``cache`` -> ``cach``).
|
||||
Otherwise a bare trailing ``s`` is dropped on a word long enough to be
|
||||
safe. Everything in :data:`_NOT_PLURAL`, and anything ending in
|
||||
``ss``/``us``/``is``, is returned unchanged.
|
||||
"""
|
||||
if len(word) < 4 or word in _NOT_PLURAL:
|
||||
return word
|
||||
@@ -52,8 +67,12 @@ def _singular(word: str) -> str:
|
||||
return word
|
||||
if word.endswith("ies") and len(word) > 4:
|
||||
return word[:-3] + "y"
|
||||
if word.endswith(("ses", "xes", "zes", "ches", "shes")):
|
||||
if word.endswith(_ES_PLURAL):
|
||||
return word[:-2]
|
||||
if word.endswith("ie") and len(word) > 4:
|
||||
return word[:-2] + "y"
|
||||
if word.endswith(_E_SINGULAR):
|
||||
return word[:-1]
|
||||
if word.endswith("s"):
|
||||
return word[:-1]
|
||||
return word
|
||||
@@ -66,12 +85,14 @@ def canonical_name(name: str) -> str:
|
||||
name: The entity name as the model wrote it.
|
||||
|
||||
Returns:
|
||||
A lowercase, punctuation-free, singularised key. Returns ``""`` for an
|
||||
empty or punctuation-only name, which callers treat as "no entity".
|
||||
A lowercase, punctuation-free key shared by a name's singular and
|
||||
plural. Returns ``""`` for an empty or punctuation-only name, which
|
||||
callers treat as "no entity".
|
||||
|
||||
Examples:
|
||||
``VECTOR_STORE`` and ``Vector stores`` -> ``vector store``;
|
||||
``.env file`` and ``env_file`` -> ``env file``;
|
||||
``cache`` and ``caches`` -> ``cach``;
|
||||
``postgres`` stays ``postgres``.
|
||||
"""
|
||||
if not name:
|
||||
|
||||
@@ -0,0 +1,88 @@
|
||||
"""Tests for canonical entity naming (the key graph nodes are merged on).
|
||||
|
||||
Two failure directions matter. Too little folding splits one entity across
|
||||
nodes ("agent" / "agents", "VECTOR_STORE" / "vector stores"), so the walk never
|
||||
connects what the text connects. Too much folding invents entities: stripping
|
||||
the "s" off ``postgres`` or ``redis`` would merge nothing and create a node no
|
||||
chunk ever named.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from docsgpt.graphrag.naming import canonical_name, normalize_entity_name
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
class TestCanonicalName:
|
||||
@pytest.mark.parametrize(
|
||||
"variants, key",
|
||||
[
|
||||
(["VECTOR_STORE", "Vector store", "vector stores", "vector-stores"], "vector store"),
|
||||
([".env file", "env_file", "ENV FILE"], "env file"),
|
||||
(["Celery worker", "Celery workers"], "celery worker"),
|
||||
(["agent", "Agents", "agents!"], "agent"),
|
||||
],
|
||||
)
|
||||
def test_orthographic_variants_share_one_key(self, variants, key):
|
||||
assert {canonical_name(v) for v in variants} == {key}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"singular, plural",
|
||||
[
|
||||
("policy", "policies"),
|
||||
("index", "indexes"),
|
||||
("batch", "batches"),
|
||||
("hash", "hashes"),
|
||||
("class", "classes"),
|
||||
("process", "processes"),
|
||||
("status", "statuses"),
|
||||
("bus", "buses"),
|
||||
("alias", "aliases"),
|
||||
("document", "documents"),
|
||||
("service", "services"),
|
||||
# Singulars ending in "e" whose plural also ends in "-es": the
|
||||
# plural alone cannot say whether to drop "s" or "es".
|
||||
("cache", "caches"),
|
||||
("database", "databases"),
|
||||
("response", "responses"),
|
||||
("release", "releases"),
|
||||
("case", "cases"),
|
||||
("size", "sizes"),
|
||||
("cookie", "cookies"),
|
||||
],
|
||||
)
|
||||
def test_singular_and_plural_share_one_key(self, singular, plural):
|
||||
assert canonical_name(singular) == canonical_name(plural)
|
||||
|
||||
def test_the_key_need_not_be_a_word(self):
|
||||
# It is a merge key, never shown: "cache" and "caches" meet at the
|
||||
# stem an "-es" plural cannot see past, rather than guessing a form.
|
||||
assert canonical_name("caches") == "cach"
|
||||
assert canonical_name("batches") == "batch"
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"word",
|
||||
["postgres", "kubernetes", "redis", "https", "status", "analysis", "access", "docs", "series"],
|
||||
)
|
||||
def test_words_that_only_look_plural_are_left_alone(self, word):
|
||||
assert canonical_name(word) == word
|
||||
|
||||
@pytest.mark.parametrize("word", ["class", "corpus", "thesis"])
|
||||
def test_ss_us_is_endings_are_never_stripped(self, word):
|
||||
assert canonical_name(word) == word
|
||||
|
||||
@pytest.mark.parametrize("word", ["aws", "ids", "ops"])
|
||||
def test_short_words_are_left_alone(self, word):
|
||||
assert canonical_name(word) == word
|
||||
|
||||
def test_each_word_of_a_phrase_is_folded(self):
|
||||
assert canonical_name("Postgres Replicas") == "postgres replica"
|
||||
|
||||
@pytest.mark.parametrize("name", [None, "", " ", "!!!", "--_--"])
|
||||
def test_a_name_with_nothing_left_is_no_entity(self, name):
|
||||
assert canonical_name(name) == ""
|
||||
|
||||
def test_normalize_entity_name_is_the_canonical_key(self):
|
||||
assert normalize_entity_name("Vector Stores") == canonical_name("Vector Stores")
|
||||
Reference in new issue
Block a user