fix(graphrag): fold an entity's singular and plural onto one key

canonical_name dropped "es" from every -ches/-ses/-zes plural, so "caches"
became "cach" while "cache" stayed "cache" -- the singular and plural landed on
two nodes, which is the split the function exists to prevent. The same rule
split the words documentation uses most: databases/database, responses/
response, releases/release, sizes/size.

An "-es" plural cannot say whether it is cache + "s" or batch + "es", so
instead of guessing, both sides now meet at the stem: the singular endings
-che/-she/-se/-ze/-xe drop their "e" the way the plurals drop "es", and a
singular's -ie folds to -y as -ies already did (cookie/cookies). The key is a
merge key that is never shown, so it only has to agree, not be a word. alias,
canvas, atlas and bias join the words that only look plural.

Found by the naming tests CI was missing: the module had no direct tests.
This commit is contained in:
Alex committed 2026-09-19 14:33:14 +01:00
1 parent 5285bb4115
commit 022cf69b49
2 files changed
+118 -9

No files matched your search

+30 -9
View File
@@ -13,6 +13,9 @@ plural. Cautious matters: this corpus contains ``postgres``, ``kubernetes``,
``https`` and ``aws``, none of which are plurals, so a naive "strip trailing s"
would corrupt them into new entities rather than merge anything.
The result is a merge key, never shown to anyone, so it only has to be the
same for a word's singular and plural — not to be a word itself.
Always on: every graph is built with canonical names.
"""
@@ -33,18 +36,30 @@ _NOT_PLURAL = frozenset(
"analysis", "basis", "axis", "https", "rss", "less", "express",
"redis", "nats", "kibana", "elasticsearch", "os", "ios", "macos",
"always", "sometimes", "series", "docs", "ops", "devops", "sse",
"alias", "canvas", "atlas", "bias", "pandas",
}
)
#: Plural endings that drop ``es``, and the singular endings that meet them.
#: ``caches`` cannot say whether it is ``cache`` + "s" or ``cach`` + "es"
#: (as ``batches`` is ``batch`` + "es"), so rather than guess, both
#: ``caches`` and ``cache`` fold to ``cach`` — as ``databases``/``database``
#: fold to ``databas``. Hardly any real word differs from one of these singulars
#: by its final "e" alone, so the fold merges next to nothing it should not.
_ES_PLURAL = ("ches", "shes", "ses", "zes", "xes")
_E_SINGULAR = ("che", "she", "se", "ze", "xe")
def _singular(word: str) -> str:
"""Best-effort singular of one word, biased hard towards leaving it alone.
"""Fold one word so its singular and plural share a key, else leave it alone.
Only the endings that are unambiguous in this domain are touched:
``-ies`` -> ``-y`` (``policies``), ``-ses``/``-xes``/``-zes``/``-ches``/
``-shes`` -> drop ``es`` (``indexes``, ``batches``), and a bare trailing
``s`` on a word long enough to be safe. Everything in :data:`_NOT_PLURAL`,
and anything ending in ``ss``/``us``/``is``, is returned unchanged.
``-ies`` and a singular's ``-ie`` both fold to ``-y`` (``policies``,
``cookies``/``cookie``). The ``-es`` endings in :data:`_ES_PLURAL` drop
``es`` and the singular endings in :data:`_E_SINGULAR` drop their ``e``, so
both sides of an ambiguous plural meet (``caches``/``cache`` -> ``cach``).
Otherwise a bare trailing ``s`` is dropped on a word long enough to be
safe. Everything in :data:`_NOT_PLURAL`, and anything ending in
``ss``/``us``/``is``, is returned unchanged.
"""
if len(word) < 4 or word in _NOT_PLURAL:
return word
@@ -52,8 +67,12 @@ def _singular(word: str) -> str:
return word
if word.endswith("ies") and len(word) > 4:
return word[:-3] + "y"
if word.endswith(("ses", "xes", "zes", "ches", "shes")):
if word.endswith(_ES_PLURAL):
return word[:-2]
if word.endswith("ie") and len(word) > 4:
return word[:-2] + "y"
if word.endswith(_E_SINGULAR):
return word[:-1]
if word.endswith("s"):
return word[:-1]
return word
@@ -66,12 +85,14 @@ def canonical_name(name: str) -> str:
name: The entity name as the model wrote it.
Returns:
A lowercase, punctuation-free, singularised key. Returns ``""`` for an
empty or punctuation-only name, which callers treat as "no entity".
A lowercase, punctuation-free key shared by a name's singular and
plural. Returns ``""`` for an empty or punctuation-only name, which
callers treat as "no entity".
Examples:
``VECTOR_STORE`` and ``Vector stores`` -> ``vector store``;
``.env file`` and ``env_file`` -> ``env file``;
``cache`` and ``caches`` -> ``cach``;
``postgres`` stays ``postgres``.
"""
if not name:
+88
View File
@@ -0,0 +1,88 @@
"""Tests for canonical entity naming (the key graph nodes are merged on).
Two failure directions matter. Too little folding splits one entity across
nodes ("agent" / "agents", "VECTOR_STORE" / "vector stores"), so the walk never
connects what the text connects. Too much folding invents entities: stripping
the "s" off ``postgres`` or ``redis`` would merge nothing and create a node no
chunk ever named.
"""
from __future__ import annotations
import pytest
from docsgpt.graphrag.naming import canonical_name, normalize_entity_name
@pytest.mark.unit
class TestCanonicalName:
@pytest.mark.parametrize(
"variants, key",
[
(["VECTOR_STORE", "Vector store", "vector stores", "vector-stores"], "vector store"),
([".env file", "env_file", "ENV FILE"], "env file"),
(["Celery worker", "Celery workers"], "celery worker"),
(["agent", "Agents", "agents!"], "agent"),
],
)
def test_orthographic_variants_share_one_key(self, variants, key):
assert {canonical_name(v) for v in variants} == {key}
@pytest.mark.parametrize(
"singular, plural",
[
("policy", "policies"),
("index", "indexes"),
("batch", "batches"),
("hash", "hashes"),
("class", "classes"),
("process", "processes"),
("status", "statuses"),
("bus", "buses"),
("alias", "aliases"),
("document", "documents"),
("service", "services"),
# Singulars ending in "e" whose plural also ends in "-es": the
# plural alone cannot say whether to drop "s" or "es".
("cache", "caches"),
("database", "databases"),
("response", "responses"),
("release", "releases"),
("case", "cases"),
("size", "sizes"),
("cookie", "cookies"),
],
)
def test_singular_and_plural_share_one_key(self, singular, plural):
assert canonical_name(singular) == canonical_name(plural)
def test_the_key_need_not_be_a_word(self):
# It is a merge key, never shown: "cache" and "caches" meet at the
# stem an "-es" plural cannot see past, rather than guessing a form.
assert canonical_name("caches") == "cach"
assert canonical_name("batches") == "batch"
@pytest.mark.parametrize(
"word",
["postgres", "kubernetes", "redis", "https", "status", "analysis", "access", "docs", "series"],
)
def test_words_that_only_look_plural_are_left_alone(self, word):
assert canonical_name(word) == word
@pytest.mark.parametrize("word", ["class", "corpus", "thesis"])
def test_ss_us_is_endings_are_never_stripped(self, word):
assert canonical_name(word) == word
@pytest.mark.parametrize("word", ["aws", "ids", "ops"])
def test_short_words_are_left_alone(self, word):
assert canonical_name(word) == word
def test_each_word_of_a_phrase_is_folded(self):
assert canonical_name("Postgres Replicas") == "postgres replica"
@pytest.mark.parametrize("name", [None, "", " ", "!!!", "--_--"])
def test_a_name_with_nothing_left_is_no_entity(self, name):
assert canonical_name(name) == ""
def test_normalize_entity_name_is_the_canonical_key(self):
assert normalize_entity_name("Vector Stores") == canonical_name("Vector Stores")