import { type ReactNode, useMemo, useState } from 'react'; import { useTranslation } from 'react-i18next'; import { cn } from '@/lib/utils'; import { Button } from '../../components/ui/button'; import { Input } from '../../components/ui/input'; import { Label } from '../../components/ui/label'; import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue, } from '../../components/ui/select'; import { Switch } from '../../components/ui/switch'; import type { ChunkingStrategy, GraphSeedStrategy, RetrievalExposure, SourceConfig, } from '../../models/misc'; import type { Model } from '../../models/types'; import ChevronRight from '../../assets/chevron-right.svg'; // Defaults mirror the backend SourceConfig // (application/storage/db/source_config.py). A form seeded with these and sent // verbatim reproduces today's behavior, so the backend's "absent == default" // contract is preserved either way. export const DEFAULT_PRESCREEN = { candidate_k: 40, model: null as string | null, batch_size: 10, max_keep: 8, }; // Fully-populated form shape so every control is controlled. Prescreen is // flattened into the form (with an `enabled` flag) and re-nested on serialize. // `kind` is carried so serialization preserves a source's behavior selector // (classic/wiki/graphrag) instead of silently downgrading it. export type RetrievalOptionsValue = { kind: string; chunking: { strategy: ChunkingStrategy; max_tokens: number; min_tokens: number; duplicate_headers: boolean; }; retrieval: { retriever: string; exposure: RetrievalExposure; chunks: number; score_threshold: number | null; rephrase_query: boolean; prescreen: { enabled: boolean; candidate_k: number; model: string | null; batch_size: number; max_keep: number; }; graph: { seed_strategy: GraphSeedStrategy; passage_nodes: boolean; blend_vector: boolean; }; }; graph: { extraction_model: string | null; max_chunks: number | null; gleanings: number; }; }; export const DEFAULT_RETRIEVAL_OPTIONS: RetrievalOptionsValue = { kind: 'classic', chunking: { strategy: 'classic_chunk', max_tokens: 1250, min_tokens: 150, duplicate_headers: false, }, retrieval: { retriever: 'classic', exposure: 'prefetch', chunks: 2, score_threshold: null, rephrase_query: true, prescreen: { enabled: false, ...DEFAULT_PRESCREEN, }, // The configuration that measured best across the corpora tested. graph: { seed_strategy: 'entities', passage_nodes: true, blend_vector: true, }, }, graph: { extraction_model: null, max_chunks: null, gleanings: 0, }, }; const CHUNKING_STRATEGIES: ChunkingStrategy[] = [ 'classic_chunk', 'recursive', 'markdown', 'parent_child', 'semantic', ]; const CLASSIC_RETRIEVER = 'classic'; const HYBRID_RETRIEVER = 'hybrid'; const GRAPHRAG_RETRIEVER = 'graphrag'; // Sentinel select value for "use the instance default model" (Select can't take // an empty/null item value); maps to extraction_model = null on change. const GRAPH_DEFAULT_MODEL = '__default__'; /** * Score threshold is a cosine-similarity cutoff; the hybrid (RRF) and graphrag * (PPR) retrievers don't produce comparable scores, so the control is hidden * for them. */ export function scoreThresholdHidden(retriever: string): boolean { return retriever === HYBRID_RETRIEVER || retriever === GRAPHRAG_RETRIEVER; } /** * Retriever options to offer. Classic is always available; hybrid and graphrag * are gated on instance support, but the currently-selected retriever is always * included so an existing source never renders an out-of-range select. */ export function availableRetrievers( current: string, hybridAvailable: boolean, graphRAGAvailable: boolean, ): string[] { const options = [CLASSIC_RETRIEVER]; if (hybridAvailable || current === HYBRID_RETRIEVER) { options.push(HYBRID_RETRIEVER); } if (graphRAGAvailable || current === GRAPHRAG_RETRIEVER) { options.push(GRAPHRAG_RETRIEVER); } return options; } /** * Group eyebrow: an uppercase, tracked title with a normal-weight muted ` · tag` * suffix. Reads as more prominent than the individual field labels below it. */ function GroupHeader({ title, tag }: { title: string; tag: string }) { return (

{title} {' · '} {tag}

); } /** * One settings row: a left block (medium-weight label plus optional muted * description) and a right block holding the control. `alignStart` top-aligns * the row for controls paired with a multi-line description; otherwise both * sides are vertically centered. */ function SettingRow({ label, htmlFor, description, alignStart = false, children, }: { label: string; htmlFor?: string; description?: ReactNode; alignStart?: boolean; children: ReactNode; }) { return (
{description ? (

{description}

) : null}
{children}
); } /** * Hydrate the form value from a stored (possibly partial/absent) SourceConfig, * filling every missing field with the documented default. Lenient on read so a * legacy `{}` row produces an all-defaults form. */ export function configToOptions(config?: SourceConfig): RetrievalOptionsValue { const chunking = config?.chunking ?? {}; const retrieval = config?.retrieval ?? {}; const prescreen = retrieval.prescreen ?? null; const retrievalGraph = retrieval.graph ?? {}; const graph = config?.graph ?? {}; const d = DEFAULT_RETRIEVAL_OPTIONS; return { kind: config?.kind ?? d.kind, chunking: { strategy: chunking.strategy ?? d.chunking.strategy, max_tokens: chunking.max_tokens ?? d.chunking.max_tokens, min_tokens: chunking.min_tokens ?? d.chunking.min_tokens, duplicate_headers: chunking.duplicate_headers ?? d.chunking.duplicate_headers, }, retrieval: { retriever: retrieval.retriever ?? d.retrieval.retriever, exposure: retrieval.exposure ?? d.retrieval.exposure, chunks: retrieval.chunks ?? d.retrieval.chunks, score_threshold: retrieval.score_threshold ?? d.retrieval.score_threshold, rephrase_query: retrieval.rephrase_query ?? d.retrieval.rephrase_query, prescreen: { enabled: prescreen != null, candidate_k: prescreen?.candidate_k ?? DEFAULT_PRESCREEN.candidate_k, model: prescreen?.model ?? DEFAULT_PRESCREEN.model, batch_size: prescreen?.batch_size ?? DEFAULT_PRESCREEN.batch_size, max_keep: prescreen?.max_keep ?? DEFAULT_PRESCREEN.max_keep, }, graph: { seed_strategy: retrievalGraph.seed_strategy ?? d.retrieval.graph.seed_strategy, passage_nodes: retrievalGraph.passage_nodes ?? d.retrieval.graph.passage_nodes, blend_vector: retrievalGraph.blend_vector ?? d.retrieval.graph.blend_vector, }, }, graph: { extraction_model: graph.extraction_model ?? d.graph.extraction_model, max_chunks: graph.max_chunks ?? d.graph.max_chunks, gleanings: graph.gleanings ?? d.graph.gleanings, }, }; } /** * Serialize the form value into a full SourceConfig object the backend accepts. * The backend uses `extra="forbid"` and re-validates the whole object, so we * always send the complete (kind + chunking + retrieval + graph) shape. * Prescreen is re-nested to `null` when disabled. `kind` is preserved from the * value and forced to `graphrag` when the graphrag retriever is chosen (so the * create-flow ingest auto-extracts), never silently downgrading wiki/graphrag. */ export function optionsToConfig(value: RetrievalOptionsValue): SourceConfig { const ps = value.retrieval.prescreen; const kind = value.retrieval.retriever === GRAPHRAG_RETRIEVER ? 'graphrag' : value.kind; return { kind, chunking: { strategy: value.chunking.strategy, max_tokens: value.chunking.max_tokens, min_tokens: value.chunking.min_tokens, duplicate_headers: value.chunking.duplicate_headers, }, retrieval: { retriever: value.retrieval.retriever, exposure: value.retrieval.exposure, chunks: value.retrieval.chunks, score_threshold: value.retrieval.score_threshold, rephrase_query: value.retrieval.rephrase_query, prescreen: ps.enabled ? { candidate_k: ps.candidate_k, model: ps.model?.trim() ? ps.model.trim() : null, batch_size: ps.batch_size, max_keep: ps.max_keep, } : null, graph: { seed_strategy: value.retrieval.graph.seed_strategy, passage_nodes: value.retrieval.graph.passage_nodes, blend_vector: value.retrieval.graph.blend_vector, }, }, graph: { extraction_model: value.graph.extraction_model?.trim() ? value.graph.extraction_model.trim() : null, max_chunks: value.graph.max_chunks, gleanings: value.graph.gleanings, }, }; } /** * True when the prescreen config is consistent with the backend's rules. * Disabled prescreen is always valid; enabled requires `candidate_k >= chunks` * and `max_keep <= candidate_k` (mirrors SourceConfig/PreScreenConfig). */ export function isPrescreenConfigValid(value: RetrievalOptionsValue): boolean { const ps = value.retrieval.prescreen; if (!ps.enabled) return true; return ( ps.candidate_k >= value.retrieval.chunks && ps.max_keep <= ps.candidate_k ); } /** True when the form differs from the stored config's chunking group only. */ export function chunkingChanged( before: RetrievalOptionsValue, after: RetrievalOptionsValue, ): boolean { return ( before.chunking.strategy !== after.chunking.strategy || before.chunking.max_tokens !== after.chunking.max_tokens || before.chunking.min_tokens !== after.chunking.min_tokens || before.chunking.duplicate_headers !== after.chunking.duplicate_headers ); } type RetrievalOptionsProps = { value: RetrievalOptionsValue; onChange: (value: RetrievalOptionsValue) => void; // When true the section is always expanded (modal use); when false it renders // its own collapsible toggle (create-flow use). Defaults to collapsible. alwaysOpen?: boolean; disabled?: boolean; // Adds the hybrid retriever option when the instance supports it (pgvector, // from /api/config). hybridAvailable?: boolean; // Adds the graphrag retriever option + extraction config when the instance // supports it (pgvector + GRAPHRAG_ENABLED, from /api/config). graphRAGAvailable?: boolean; // Models for the graph extraction-model picker, same shape as the agent form. availableModels?: Model[]; // Shows only the knobs that change what a query retrieves, hiding the // ingest-time groups (chunking, graph extraction) and `exposure` (which picks // *when* a source is searched at answer time, not what comes back). Used by // the retrieval tester, where those knobs cannot affect the result and // showing them would imply they do. queryOnly?: boolean; }; /** * Shared "Retrieval options" panel reused by the create flow (Upload) and the * edit modal (SourceConfigModal). Groups live "Retrieval" knobs from * re-ingest-gated "Chunking" knobs, with a visible note on the chunking group. */ export default function RetrievalOptions({ value, onChange, alwaysOpen = false, disabled = false, hybridAvailable = false, graphRAGAvailable = false, availableModels = [], queryOnly = false, }: RetrievalOptionsProps) { const { t } = useTranslation(); const [open, setOpen] = useState(false); const expanded = alwaysOpen || open; const strategyOptions = useMemo( () => CHUNKING_STRATEGIES.map((s) => ({ value: s, label: t(`settings.sources.retrievalOptions.chunking.strategies.${s}`), })), [t], ); const isGraphRAG = value.retrieval.retriever === GRAPHRAG_RETRIEVER; const retrievers = useMemo( () => availableRetrievers( value.retrieval.retriever, hybridAvailable, graphRAGAvailable, ), [value.retrieval.retriever, hybridAvailable, graphRAGAvailable], ); const setRetrieval = (patch: Partial) => { onChange({ ...value, retrieval: { ...value.retrieval, ...patch }, }); }; const setChunking = (patch: Partial) => { onChange({ ...value, chunking: { ...value.chunking, ...patch }, }); }; const setPrescreen = ( patch: Partial, ) => { setRetrieval({ prescreen: { ...value.retrieval.prescreen, ...patch }, }); }; const setGraph = (patch: Partial) => { onChange({ ...value, graph: { ...value.graph, ...patch }, }); }; const setGraphRetrieval = ( patch: Partial, ) => { setRetrieval({ graph: { ...value.retrieval.graph, ...patch } }); }; const modelOptions = useMemo(() => { const builtin: Model[] = []; const user: Model[] = []; availableModels.forEach((m) => (m.source === 'user' ? user : builtin).push(m), ); return { builtin, user }; }, [availableModels]); const tr = (key: string) => t(`settings.sources.retrievalOptions.${key}`); const body = (
{/* Retrieval group (live) */}
setRetrieval({ chunks: Math.max(1, Number(e.target.value) || 1), }) } /> {!scoreThresholdHidden(value.retrieval.retriever) && ( { const raw = e.target.value; setRetrieval({ score_threshold: raw === '' ? null : Math.min(1, Math.max(0, Number(raw) || 0)), }); }} /> )} {/* Rephrasing only fires when there is chat history to rephrase against, and a test has none — so the switch would do nothing. */} {!queryOnly && ( setRetrieval({ rephrase_query: checked }) } /> )} {!queryOnly && ( )} setPrescreen({ enabled: checked })} />
{/* Prescreen expanded inputs (kept as floating-label cards) */} {value.retrieval.prescreen.enabled && (
setPrescreen({ candidate_k: Math.max( value.retrieval.chunks, Number(e.target.value) || value.retrieval.chunks, ), }) } /> setPrescreen({ max_keep: Math.min( value.retrieval.prescreen.candidate_k, Math.max(1, Number(e.target.value) || 1), ), }) } /> setPrescreen({ batch_size: Math.max(1, Number(e.target.value) || 1), }) } />
)}
{/* Graph retrieval group (graphrag only; live, so shown when testing too) */} {isGraphRAG && (

{tr('graphRetrieval.agentToolHint')}

setGraphRetrieval({ passage_nodes: checked }) } /> setGraphRetrieval({ blend_vector: checked }) } />
)} {/* Graph extraction group (graphrag only; re-ingest required to apply) */} {isGraphRAG && !queryOnly && (
{ const raw = e.target.value; setGraph({ max_chunks: raw === '' ? null : Math.max(1, Number(raw) || 1), }); }} />
)} {/* Chunking group (re-ingest required) */}
setChunking({ max_tokens: Math.max(1, Number(e.target.value) || 1), }) } /> setChunking({ min_tokens: Math.max(0, Number(e.target.value) || 0), }) } /> setChunking({ duplicate_headers: checked }) } />
); if (alwaysOpen) { return body; } return (
{body}
); }