refactor(embed): candle provider/crate 제거 — fastembed+ollama로 충분
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_012Mc6W1fgsrbFKTsqA6P8La
This commit is contained in:
@@ -255,26 +255,24 @@ impl NliCfg {
|
||||
|
||||
#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
|
||||
pub struct EmbeddingModelCfg {
|
||||
/// `fastembed` (default, onnxruntime), `candle` (pure-Rust, NUMA-safe),
|
||||
/// or `ollama` (remote HTTP embedding endpoint). `none` disables
|
||||
/// embeddings (lexical-only). Unknown values error at embedder
|
||||
/// construction.
|
||||
/// `fastembed` (default, onnxruntime) or `ollama` (remote HTTP embedding
|
||||
/// endpoint). `none` disables embeddings (lexical-only). Unknown values
|
||||
/// error at embedder construction.
|
||||
pub provider: String,
|
||||
pub model: String,
|
||||
pub version: String,
|
||||
pub dimensions: usize,
|
||||
pub batch_size: usize,
|
||||
/// Cap on the CPU worker threads the `candle` provider spins up
|
||||
/// (sizes the global rayon pool; env `KEBAB_EMBED_THREADS` overrides).
|
||||
/// `0` = auto (rayon default = #cores). Lever to sidestep the
|
||||
/// onnxruntime 48-thread NUMA double-free; ignored by the `fastembed`
|
||||
/// provider. Defaulted on load so pre-0.22 config files still parse.
|
||||
/// Legacy field — previously used by the removed `candle` provider to cap
|
||||
/// CPU worker threads. Retained for backward-compatible TOML parsing
|
||||
/// (`num_threads = 0` in old config files must not error). Ignored by all
|
||||
/// current providers.
|
||||
#[serde(default)]
|
||||
pub num_threads: u32,
|
||||
/// HTTP endpoint for the `ollama` embedding provider (e.g.
|
||||
/// `"http://127.0.0.1:11434"`). `None` (or a missing key in TOML) means
|
||||
/// "fall back to `models.llm.endpoint`" — same convention as the OCR /
|
||||
/// vision endpoints. Ignored by the `fastembed` / `candle` providers.
|
||||
/// vision endpoints. Ignored by the `fastembed` provider.
|
||||
/// Defaulted on load so pre-0.26 config files still parse.
|
||||
#[serde(default)]
|
||||
pub endpoint: Option<String>,
|
||||
|
||||
@@ -99,9 +99,9 @@ fn key_comment(path: &str) -> Option<&'static str> {
|
||||
"workspace.root" => "색인 루트. 절대/~/${VAR}/상대(=이 파일 기준).",
|
||||
"workspace.exclude" => "denylist glob.",
|
||||
"storage.copy_threshold_mb" => "이 크기(MB) 초과 파일은 사본 대신 참조.",
|
||||
"models.embedding.provider" => "fastembed | candle | ollama | none.",
|
||||
"models.embedding.provider" => "fastembed | ollama | none.",
|
||||
"models.embedding.dimensions" => "모델 출력 차원. 틀리면 검색 0건.",
|
||||
"models.embedding.num_threads" => "candle 전용 CPU 스레드 cap(0=auto).",
|
||||
"models.embedding.num_threads" => "레거시 필드 (deprecated, 무시됨).",
|
||||
"models.embedding.endpoint" => "ollama provider 시 HTTP. 비우면 llm.endpoint fallback.",
|
||||
"models.llm.request_timeout_secs" => "단일 HTTP 상한. 0=즉시실패(비활성화 아님).",
|
||||
"ingest.max_parallel_extractors" => "동시 extractor 수.",
|
||||
|
||||
Reference in New Issue
Block a user