import { type ReactNode, useMemo, useState } from 'react';
import { useTranslation } from 'react-i18next';
import { cn } from '@/lib/utils';
import { Button } from '../../components/ui/button';
import { Input } from '../../components/ui/input';
import { Label } from '../../components/ui/label';
import {
Select,
SelectContent,
SelectItem,
SelectTrigger,
SelectValue,
} from '../../components/ui/select';
import { Switch } from '../../components/ui/switch';
import type {
ChunkingStrategy,
GraphSeedStrategy,
RetrievalExposure,
SourceConfig,
} from '../../models/misc';
import type { Model } from '../../models/types';
import ChevronRight from '../../assets/chevron-right.svg';
// Defaults mirror the backend SourceConfig
// (application/storage/db/source_config.py). A form seeded with these and sent
// verbatim reproduces today's behavior, so the backend's "absent == default"
// contract is preserved either way.
export const DEFAULT_PRESCREEN = {
candidate_k: 40,
model: null as string | null,
batch_size: 10,
max_keep: 8,
};
// Fully-populated form shape so every control is controlled. Prescreen is
// flattened into the form (with an `enabled` flag) and re-nested on serialize.
// `kind` is carried so serialization preserves a source's behavior selector
// (classic/wiki/graphrag) instead of silently downgrading it.
export type RetrievalOptionsValue = {
kind: string;
chunking: {
strategy: ChunkingStrategy;
max_tokens: number;
min_tokens: number;
duplicate_headers: boolean;
};
retrieval: {
retriever: string;
exposure: RetrievalExposure;
chunks: number;
score_threshold: number | null;
rephrase_query: boolean;
prescreen: {
enabled: boolean;
candidate_k: number;
model: string | null;
batch_size: number;
max_keep: number;
};
graph: {
seed_strategy: GraphSeedStrategy;
passage_nodes: boolean;
blend_vector: boolean;
};
};
graph: {
extraction_model: string | null;
max_chunks: number | null;
gleanings: number;
};
};
export const DEFAULT_RETRIEVAL_OPTIONS: RetrievalOptionsValue = {
kind: 'classic',
chunking: {
strategy: 'classic_chunk',
max_tokens: 1250,
min_tokens: 150,
duplicate_headers: false,
},
retrieval: {
retriever: 'classic',
exposure: 'prefetch',
chunks: 2,
score_threshold: null,
rephrase_query: true,
prescreen: {
enabled: false,
...DEFAULT_PRESCREEN,
},
// The configuration that measured best across the corpora tested.
graph: {
seed_strategy: 'entities',
passage_nodes: true,
blend_vector: true,
},
},
graph: {
extraction_model: null,
max_chunks: null,
gleanings: 0,
},
};
const CHUNKING_STRATEGIES: ChunkingStrategy[] = [
'classic_chunk',
'recursive',
'markdown',
'parent_child',
'semantic',
];
const CLASSIC_RETRIEVER = 'classic';
const HYBRID_RETRIEVER = 'hybrid';
const GRAPHRAG_RETRIEVER = 'graphrag';
// Sentinel select value for "use the instance default model" (Select can't take
// an empty/null item value); maps to extraction_model = null on change.
const GRAPH_DEFAULT_MODEL = '__default__';
/**
* Score threshold is a cosine-similarity cutoff; the hybrid (RRF) and graphrag
* (PPR) retrievers don't produce comparable scores, so the control is hidden
* for them.
*/
export function scoreThresholdHidden(retriever: string): boolean {
return retriever === HYBRID_RETRIEVER || retriever === GRAPHRAG_RETRIEVER;
}
/**
* Retriever options to offer. Classic is always available; hybrid and graphrag
* are gated on instance support, but the currently-selected retriever is always
* included so an existing source never renders an out-of-range select.
*/
export function availableRetrievers(
current: string,
hybridAvailable: boolean,
graphRAGAvailable: boolean,
): string[] {
const options = [CLASSIC_RETRIEVER];
if (hybridAvailable || current === HYBRID_RETRIEVER) {
options.push(HYBRID_RETRIEVER);
}
if (graphRAGAvailable || current === GRAPHRAG_RETRIEVER) {
options.push(GRAPHRAG_RETRIEVER);
}
return options;
}
/**
* Group eyebrow: an uppercase, tracked title with a normal-weight muted ` · tag`
* suffix. Reads as more prominent than the individual field labels below it.
*/
function GroupHeader({ title, tag }: { title: string; tag: string }) {
return (
{title}
{' · '}
{tag}
);
}
/**
* One settings row: a left block (medium-weight label plus optional muted
* description) and a right block holding the control. `alignStart` top-aligns
* the row for controls paired with a multi-line description; otherwise both
* sides are vertically centered.
*/
function SettingRow({
label,
htmlFor,
description,
alignStart = false,
children,
}: {
label: string;
htmlFor?: string;
description?: ReactNode;
alignStart?: boolean;
children: ReactNode;
}) {
return (
{description ? (
{description}
) : null}
{children}
);
}
/**
* Hydrate the form value from a stored (possibly partial/absent) SourceConfig,
* filling every missing field with the documented default. Lenient on read so a
* legacy `{}` row produces an all-defaults form.
*/
export function configToOptions(config?: SourceConfig): RetrievalOptionsValue {
const chunking = config?.chunking ?? {};
const retrieval = config?.retrieval ?? {};
const prescreen = retrieval.prescreen ?? null;
const retrievalGraph = retrieval.graph ?? {};
const graph = config?.graph ?? {};
const d = DEFAULT_RETRIEVAL_OPTIONS;
return {
kind: config?.kind ?? d.kind,
chunking: {
strategy: chunking.strategy ?? d.chunking.strategy,
max_tokens: chunking.max_tokens ?? d.chunking.max_tokens,
min_tokens: chunking.min_tokens ?? d.chunking.min_tokens,
duplicate_headers:
chunking.duplicate_headers ?? d.chunking.duplicate_headers,
},
retrieval: {
retriever: retrieval.retriever ?? d.retrieval.retriever,
exposure: retrieval.exposure ?? d.retrieval.exposure,
chunks: retrieval.chunks ?? d.retrieval.chunks,
score_threshold: retrieval.score_threshold ?? d.retrieval.score_threshold,
rephrase_query: retrieval.rephrase_query ?? d.retrieval.rephrase_query,
prescreen: {
enabled: prescreen != null,
candidate_k: prescreen?.candidate_k ?? DEFAULT_PRESCREEN.candidate_k,
model: prescreen?.model ?? DEFAULT_PRESCREEN.model,
batch_size: prescreen?.batch_size ?? DEFAULT_PRESCREEN.batch_size,
max_keep: prescreen?.max_keep ?? DEFAULT_PRESCREEN.max_keep,
},
graph: {
seed_strategy:
retrievalGraph.seed_strategy ?? d.retrieval.graph.seed_strategy,
passage_nodes:
retrievalGraph.passage_nodes ?? d.retrieval.graph.passage_nodes,
blend_vector:
retrievalGraph.blend_vector ?? d.retrieval.graph.blend_vector,
},
},
graph: {
extraction_model: graph.extraction_model ?? d.graph.extraction_model,
max_chunks: graph.max_chunks ?? d.graph.max_chunks,
gleanings: graph.gleanings ?? d.graph.gleanings,
},
};
}
/**
* Serialize the form value into a full SourceConfig object the backend accepts.
* The backend uses `extra="forbid"` and re-validates the whole object, so we
* always send the complete (kind + chunking + retrieval + graph) shape.
* Prescreen is re-nested to `null` when disabled. `kind` is preserved from the
* value and forced to `graphrag` when the graphrag retriever is chosen (so the
* create-flow ingest auto-extracts), never silently downgrading wiki/graphrag.
*/
export function optionsToConfig(value: RetrievalOptionsValue): SourceConfig {
const ps = value.retrieval.prescreen;
const kind =
value.retrieval.retriever === GRAPHRAG_RETRIEVER ? 'graphrag' : value.kind;
return {
kind,
chunking: {
strategy: value.chunking.strategy,
max_tokens: value.chunking.max_tokens,
min_tokens: value.chunking.min_tokens,
duplicate_headers: value.chunking.duplicate_headers,
},
retrieval: {
retriever: value.retrieval.retriever,
exposure: value.retrieval.exposure,
chunks: value.retrieval.chunks,
score_threshold: value.retrieval.score_threshold,
rephrase_query: value.retrieval.rephrase_query,
prescreen: ps.enabled
? {
candidate_k: ps.candidate_k,
model: ps.model?.trim() ? ps.model.trim() : null,
batch_size: ps.batch_size,
max_keep: ps.max_keep,
}
: null,
graph: {
seed_strategy: value.retrieval.graph.seed_strategy,
passage_nodes: value.retrieval.graph.passage_nodes,
blend_vector: value.retrieval.graph.blend_vector,
},
},
graph: {
extraction_model: value.graph.extraction_model?.trim()
? value.graph.extraction_model.trim()
: null,
max_chunks: value.graph.max_chunks,
gleanings: value.graph.gleanings,
},
};
}
/**
* True when the prescreen config is consistent with the backend's rules.
* Disabled prescreen is always valid; enabled requires `candidate_k >= chunks`
* and `max_keep <= candidate_k` (mirrors SourceConfig/PreScreenConfig).
*/
export function isPrescreenConfigValid(value: RetrievalOptionsValue): boolean {
const ps = value.retrieval.prescreen;
if (!ps.enabled) return true;
return (
ps.candidate_k >= value.retrieval.chunks && ps.max_keep <= ps.candidate_k
);
}
/** True when the form differs from the stored config's chunking group only. */
export function chunkingChanged(
before: RetrievalOptionsValue,
after: RetrievalOptionsValue,
): boolean {
return (
before.chunking.strategy !== after.chunking.strategy ||
before.chunking.max_tokens !== after.chunking.max_tokens ||
before.chunking.min_tokens !== after.chunking.min_tokens ||
before.chunking.duplicate_headers !== after.chunking.duplicate_headers
);
}
type RetrievalOptionsProps = {
value: RetrievalOptionsValue;
onChange: (value: RetrievalOptionsValue) => void;
// When true the section is always expanded (modal use); when false it renders
// its own collapsible toggle (create-flow use). Defaults to collapsible.
alwaysOpen?: boolean;
disabled?: boolean;
// Adds the hybrid retriever option when the instance supports it (pgvector,
// from /api/config).
hybridAvailable?: boolean;
// Adds the graphrag retriever option + extraction config when the instance
// supports it (pgvector + GRAPHRAG_ENABLED, from /api/config).
graphRAGAvailable?: boolean;
// Models for the graph extraction-model picker, same shape as the agent form.
availableModels?: Model[];
// Shows only the knobs that change what a query retrieves, hiding the
// ingest-time groups (chunking, graph extraction) and `exposure` (which picks
// *when* a source is searched at answer time, not what comes back). Used by
// the retrieval tester, where those knobs cannot affect the result and
// showing them would imply they do.
queryOnly?: boolean;
};
/**
* Shared "Retrieval options" panel reused by the create flow (Upload) and the
* edit modal (SourceConfigModal). Groups live "Retrieval" knobs from
* re-ingest-gated "Chunking" knobs, with a visible note on the chunking group.
*/
export default function RetrievalOptions({
value,
onChange,
alwaysOpen = false,
disabled = false,
hybridAvailable = false,
graphRAGAvailable = false,
availableModels = [],
queryOnly = false,
}: RetrievalOptionsProps) {
const { t } = useTranslation();
const [open, setOpen] = useState(false);
const expanded = alwaysOpen || open;
const strategyOptions = useMemo(
() =>
CHUNKING_STRATEGIES.map((s) => ({
value: s,
label: t(`settings.sources.retrievalOptions.chunking.strategies.${s}`),
})),
[t],
);
const isGraphRAG = value.retrieval.retriever === GRAPHRAG_RETRIEVER;
const retrievers = useMemo(
() =>
availableRetrievers(
value.retrieval.retriever,
hybridAvailable,
graphRAGAvailable,
),
[value.retrieval.retriever, hybridAvailable, graphRAGAvailable],
);
const setRetrieval = (patch: Partial) => {
onChange({
...value,
retrieval: { ...value.retrieval, ...patch },
});
};
const setChunking = (patch: Partial) => {
onChange({
...value,
chunking: { ...value.chunking, ...patch },
});
};
const setPrescreen = (
patch: Partial,
) => {
setRetrieval({
prescreen: { ...value.retrieval.prescreen, ...patch },
});
};
const setGraph = (patch: Partial) => {
onChange({
...value,
graph: { ...value.graph, ...patch },
});
};
const setGraphRetrieval = (
patch: Partial,
) => {
setRetrieval({ graph: { ...value.retrieval.graph, ...patch } });
};
const modelOptions = useMemo(() => {
const builtin: Model[] = [];
const user: Model[] = [];
availableModels.forEach((m) =>
(m.source === 'user' ? user : builtin).push(m),
);
return { builtin, user };
}, [availableModels]);
const tr = (key: string) => t(`settings.sources.retrievalOptions.${key}`);
const body = (
{/* Retrieval group (live) */}
setRetrieval({
chunks: Math.max(1, Number(e.target.value) || 1),
})
}
/>
{!scoreThresholdHidden(value.retrieval.retriever) && (
{
const raw = e.target.value;
setRetrieval({
score_threshold:
raw === ''
? null
: Math.min(1, Math.max(0, Number(raw) || 0)),
});
}}
/>
)}
{/* Rephrasing only fires when there is chat history to rephrase
against, and a test has none — so the switch would do nothing. */}
{!queryOnly && (
setRetrieval({ rephrase_query: checked })
}
/>
)}
{!queryOnly && (
)}
setPrescreen({ enabled: checked })}
/>