mirror of
https://github.com/tiennm99/goclaw.git
synced 2026-10-11 12:18:59 +00:00
vault_search only indexed title + path + the auto-summary, and the summary is written from the first 3000 runes of the file. Anything the summary left out (room names, prices, codes) could never be found. On top of that the FTS query used plainto_tsquery, which ANDs every word, so a normal question like "what is the price of the Deluxe Ocean Suite?" matched nothing even when the key words were indexed. - Add vault_document_chunks (migration 98): the file body split into chunks with their own tsvector and embedding. body_indexed_hash on vault_documents records which content_hash the chunks came from, so unchanged files are not re-chunked or re-embedded. - The enrich worker rebuilds chunks when a file changes. This needs no LLM, so it runs even when no provider is configured. - Rescan backfills chunks for docs indexed before this change, since they already have a summary and never go back through the worker. - FTS matches any query word; ts_rank still ranks docs with more matching words first. Both FTS and vector search look at the doc and its chunks and score each doc by its best hit. SQLite is unchanged: its vault search is LIKE on title/path only.
67 lines
1.9 KiB
Go
67 lines
1.9 KiB
Go
package vault
|
|
|
|
import (
|
|
"context"
|
|
"os"
|
|
"path/filepath"
|
|
"testing"
|
|
|
|
"github.com/nextlevelbuilder/goclaw/internal/store"
|
|
)
|
|
|
|
type fakeVaultStoreBody struct {
|
|
store.VaultStore
|
|
stale []store.VaultDocument
|
|
replaced map[string][]store.VaultChunk // docID -> chunks
|
|
hashes map[string]string // docID -> contentHash passed in
|
|
}
|
|
|
|
func (f *fakeVaultStoreBody) ListDocsNeedingBodyIndex(ctx context.Context, tenantID string, limit int) ([]store.VaultDocument, error) {
|
|
return f.stale, nil
|
|
}
|
|
|
|
func (f *fakeVaultStoreBody) ReplaceDocumentChunks(ctx context.Context, tenantID, docID, contentHash string, chunks []store.VaultChunk) error {
|
|
f.replaced[docID] = chunks
|
|
f.hashes[docID] = contentHash
|
|
return nil
|
|
}
|
|
|
|
func TestIndexStaleBodies_ChunksWorkspaceFiles(t *testing.T) {
|
|
ws := t.TempDir()
|
|
body := "# Prices\n| Deluxe Ocean Suite | 4,200,000 |\x00\nDeposit 30%\xff\n"
|
|
if err := os.WriteFile(filepath.Join(ws, "prices.md"), []byte(body), 0o644); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
fs := &fakeVaultStoreBody{
|
|
stale: []store.VaultDocument{
|
|
{ID: "d1", Path: "prices.md", ContentHash: "h1"},
|
|
{ID: "d2", Path: "gone.md", ContentHash: "h2"},
|
|
},
|
|
replaced: map[string][]store.VaultChunk{},
|
|
hashes: map[string]string{},
|
|
}
|
|
|
|
if !IndexStaleBodies(context.Background(), fs, "t1", ws) {
|
|
t.Fatal("first pass should run")
|
|
}
|
|
|
|
chunks := fs.replaced["d1"]
|
|
if len(chunks) != 1 {
|
|
t.Fatalf("chunks = %+v, want 1", chunks)
|
|
}
|
|
want := "# Prices\n| Deluxe Ocean Suite | 4,200,000 |\nDeposit 30%"
|
|
if chunks[0].Text != want {
|
|
t.Fatalf("chunk text = %q, want %q", chunks[0].Text, want)
|
|
}
|
|
if chunks[0].StartLine != 1 || chunks[0].EndLine != 4 {
|
|
t.Fatalf("lines = %d-%d, want 1-4", chunks[0].StartLine, chunks[0].EndLine)
|
|
}
|
|
if fs.hashes["d1"] != "h1" {
|
|
t.Fatalf("hash = %q, want h1", fs.hashes["d1"])
|
|
}
|
|
// A missing file must not mark the doc as indexed.
|
|
if _, ok := fs.replaced["d2"]; ok {
|
|
t.Fatal("missing file should be skipped")
|
|
}
|
|
}
|