style: cargo fmt --all (round 4 ingest log feature follow-up)

Phase C4 executor 의 마지막 `fix(test): clippy + fmt fixes` commit 이
test file 부분만 fmt 적용. workspace 전체 fmt 누락 발견 → cargo fmt --all
적용. 모든 import alphabetical reorder + line wrapping 정합.

추가 untracked artifact 동시 commit:
- docs/superpowers/specs/2026-05-28-v0.20-ingest-log-spec.md (491 line, ACCEPT)
- docs/superpowers/plans/2026-05-28-v0.20-ingest-log-plan.md (616 line, ACCEPT)

workspace test: 1370 passed / 0 failed / 50 ignored, ingest_log_smoke green.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-05-28 04:18:40 +00:00
parent 445b096215
commit 685007789a
235 changed files with 6520 additions and 3955 deletions

1
Cargo.lock generated
View File

@@ -4166,6 +4166,7 @@ dependencies = [
"tracing-appender", "tracing-appender",
"tracing-subscriber", "tracing-subscriber",
"unicode-normalization", "unicode-normalization",
"uuid",
"wiremock", "wiremock",
] ]

View File

@@ -46,9 +46,8 @@ use kebab_core::{
use kebab_embed_local::FastembedEmbedder; use kebab_embed_local::FastembedEmbedder;
use kebab_llm_local::OllamaLanguageModel; use kebab_llm_local::OllamaLanguageModel;
use kebab_parse_code::{ use kebab_parse_code::{
CAstExtractor, CppAstExtractor, GoAstExtractor, JavaAstExtractor, CAstExtractor, CppAstExtractor, GoAstExtractor, JavaAstExtractor, JavascriptAstExtractor,
JavascriptAstExtractor, KotlinAstExtractor, PythonAstExtractor, RustAstExtractor, KotlinAstExtractor, PythonAstExtractor, RustAstExtractor, TypescriptAstExtractor,
TypescriptAstExtractor,
}; };
use kebab_parse_image::ImageExtractor; use kebab_parse_image::ImageExtractor;
use kebab_parse_pdf::PdfTextExtractor; use kebab_parse_pdf::PdfTextExtractor;
@@ -242,15 +241,15 @@ impl App {
// kebab-nli construction. Failure (`?`) surfaces as a user- // kebab-nli construction. Failure (`?`) surfaces as a user-
// facing error at App boot — never a panic in the pipeline's // facing error at App boot — never a panic in the pipeline's
// `expect("verifier must be Some when nli_threshold > 0.0")`. // `expect("verifier must be Some when nli_threshold > 0.0")`.
let pipeline_verifier: Option<Arc<dyn kebab_nli::NliVerifier>> = let pipeline_verifier: Option<Arc<dyn kebab_nli::NliVerifier>> = if config.rag.nli_threshold
if config.rag.nli_threshold > 0.0 { > 0.0
let v = kebab_nli::OnnxNliVerifier::new(&config).context( {
"kebab-app: construct OnnxNliVerifier (config.rag.nli_threshold > 0)", let v = kebab_nli::OnnxNliVerifier::new(&config)
)?; .context("kebab-app: construct OnnxNliVerifier (config.rag.nli_threshold > 0)")?;
Some(Arc::new(v)) Some(Arc::new(v))
} else { } else {
None None
}; };
Ok(Self { Ok(Self {
config, config,
sqlite: Arc::new(sqlite), sqlite: Arc::new(sqlite),
@@ -350,7 +349,9 @@ impl App {
// so other in-flight searches can use the cache concurrently. // so other in-flight searches can use the cache concurrently.
drop(guard); drop(guard);
let hits = self.search_uncached(query)?; let hits = self.search_uncached(query)?;
let mut guard = cache.lock().unwrap_or_else(std::sync::PoisonError::into_inner); let mut guard = cache
.lock()
.unwrap_or_else(std::sync::PoisonError::into_inner);
guard.put(key, hits.clone()); guard.put(key, hits.clone());
Ok(hits) Ok(hits)
} }
@@ -430,11 +431,7 @@ impl App {
/// ///
/// `SearchResponse.next_cursor` and `truncated` are independent /// `SearchResponse.next_cursor` and `truncated` are independent
/// signals — see `SearchResponse` doc for details. /// signals — see `SearchResponse` doc for details.
pub fn search_with_opts( pub fn search_with_opts(&self, query: SearchQuery, opts: SearchOpts) -> Result<SearchResponse> {
&self,
query: SearchQuery,
opts: SearchOpts,
) -> Result<SearchResponse> {
use crate::cursor; use crate::cursor;
let corpus_revision = self.sqlite.corpus_revision().to_string(); let corpus_revision = self.sqlite.corpus_revision().to_string();
@@ -519,8 +516,7 @@ impl App {
// Apply offset + k_effective truncation (mirrors non-trace path). // Apply offset + k_effective truncation (mirrors non-trace path).
let drop_n = offset.min(traced_hits.len()); let drop_n = offset.min(traced_hits.len());
traced_hits.drain(..drop_n); traced_hits.drain(..drop_n);
let mut hits: Vec<SearchHit> = let mut hits: Vec<SearchHit> = traced_hits.into_iter().take(k_effective).collect();
traced_hits.into_iter().take(k_effective).collect();
// Snippet truncation if opts.snippet_chars set (mirror non-trace path). // Snippet truncation if opts.snippet_chars set (mirror non-trace path).
if opts.snippet_chars.is_some() { if opts.snippet_chars.is_some() {
@@ -551,8 +547,7 @@ impl App {
// Skip offset. // Skip offset.
let drop_n = offset.min(all_hits.len()); let drop_n = offset.min(all_hits.len());
all_hits.drain(..drop_n); all_hits.drain(..drop_n);
let mut hits: Vec<SearchHit> = let mut hits: Vec<SearchHit> = all_hits.into_iter().take(k_effective).collect();
all_hits.into_iter().take(k_effective).collect();
// Apply snippet_chars override if shorter than what the // Apply snippet_chars override if shorter than what the
// retriever returned (retriever already honored // retriever returned (retriever already honored
@@ -573,15 +568,11 @@ impl App {
// Step 1: shorten snippets progressively to a 60-char floor. // Step 1: shorten snippets progressively to a 60-char floor.
const SNIPPET_FLOOR: usize = 60; const SNIPPET_FLOOR: usize = 60;
let mut current_snippet_cap = snippet_chars; let mut current_snippet_cap = snippet_chars;
while estimate_chars(&hits) > max_chars while estimate_chars(&hits) > max_chars && current_snippet_cap > SNIPPET_FLOOR {
&& current_snippet_cap > SNIPPET_FLOOR current_snippet_cap = (current_snippet_cap / 2).max(SNIPPET_FLOOR);
{
current_snippet_cap =
(current_snippet_cap / 2).max(SNIPPET_FLOOR);
for h in &mut hits { for h in &mut hits {
if h.snippet.chars().count() > current_snippet_cap { if h.snippet.chars().count() > current_snippet_cap {
h.snippet = h.snippet = trim_to_chars(&h.snippet, current_snippet_cap);
trim_to_chars(&h.snippet, current_snippet_cap);
truncated = true; truncated = true;
} }
} }
@@ -651,8 +642,7 @@ impl App {
retriever: Arc<dyn Retriever>, retriever: Arc<dyn Retriever>,
llm: Arc<dyn LanguageModel>, llm: Arc<dyn LanguageModel>,
) -> RagPipeline { ) -> RagPipeline {
let pipeline = let pipeline = RagPipeline::new(self.config.clone(), retriever, llm, self.sqlite.clone());
RagPipeline::new(self.config.clone(), retriever, llm, self.sqlite.clone());
match &self.pipeline_verifier { match &self.pipeline_verifier {
Some(v) => pipeline.with_verifier(v.clone()), Some(v) => pipeline.with_verifier(v.clone()),
None => pipeline, None => pipeline,
@@ -723,12 +713,7 @@ impl App {
/// returns; on persistence error, the answer is still returned /// returns; on persistence error, the answer is still returned
/// (don't lose the user's compute) but the error is logged so /// (don't lose the user's compute) but the error is logged so
/// the operator notices. /// the operator notices.
pub fn ask_with_session( pub fn ask_with_session(&self, session_id: &str, query: &str, opts: AskOpts) -> Result<Answer> {
&self,
session_id: &str,
query: &str,
opts: AskOpts,
) -> Result<Answer> {
use kebab_core::traits::{ChatSessionRepo, ChatSessionRow, ChatTurnRow}; use kebab_core::traits::{ChatSessionRepo, ChatSessionRow, ChatTurnRow};
use std::time::{SystemTime, UNIX_EPOCH}; use std::time::{SystemTime, UNIX_EPOCH};
@@ -766,13 +751,8 @@ impl App {
let retriever = self.build_retriever(opts.mode)?; let retriever = self.build_retriever(opts.mode)?;
let llm = self.llm()?; let llm = self.llm()?;
let pipeline = self.build_pipeline(retriever, llm); let pipeline = self.build_pipeline(retriever, llm);
let answer = pipeline.ask_with_history( let answer =
query, pipeline.ask_with_history(query, history, session_id.to_string(), next_index, opts)?;
history,
session_id.to_string(),
next_index,
opts,
)?;
// Auto-create the session header on first use. Title from // Auto-create the session header on first use. Title from
// the first question (≤40 chars after trim). // the first question (≤40 chars after trim).
@@ -813,7 +793,8 @@ impl App {
turn_index: next_index, turn_index: next_index,
question: query.to_string(), question: query.to_string(),
answer: answer.answer.clone(), answer: answer.answer.clone(),
citations_json: serde_json::to_string(&answer.citations).unwrap_or_else(|_| "[]".to_string()), citations_json: serde_json::to_string(&answer.citations)
.unwrap_or_else(|_| "[]".to_string()),
created_at: now_unix, created_at: now_unix,
}; };
if let Err(e) = self.sqlite.append_turn(&turn_row) { if let Err(e) = self.sqlite.append_turn(&turn_row) {
@@ -848,8 +829,7 @@ impl App {
return Ok(Some(e.clone())); return Ok(Some(e.clone()));
} }
let emb: Arc<dyn Embedder + Send + Sync> = Arc::new( let emb: Arc<dyn Embedder + Send + Sync> = Arc::new(
FastembedEmbedder::new(&self.config) FastembedEmbedder::new(&self.config).context("kb-app: load FastembedEmbedder")?,
.context("kb-app: load FastembedEmbedder")?,
); );
// `set` returns Err if another thread won the race; in that case // `set` returns Err if another thread won the race; in that case
// the loser still returns the (now-cached) winner via `get()`. // the loser still returns the (now-cached) winner via `get()`.
@@ -925,7 +905,9 @@ impl App {
/// clear` admin command). No-op when the cache is disabled. /// clear` admin command). No-op when the cache is disabled.
pub fn clear_search_cache(&self) { pub fn clear_search_cache(&self) {
if let Some(cache) = self.search_cache.as_ref() { if let Some(cache) = self.search_cache.as_ref() {
let mut guard = cache.lock().unwrap_or_else(std::sync::PoisonError::into_inner); let mut guard = cache
.lock()
.unwrap_or_else(std::sync::PoisonError::into_inner);
guard.clear(); guard.clear();
} }
} }
@@ -946,8 +928,8 @@ impl App {
/// git tree) correctly keep `repo: None` — `Metadata.repo` is already /// git tree) correctly keep `repo: None` — `Metadata.repo` is already
/// `None` for those, so the assignment is a no-op. /// `None` for those, so the assignment is a no-op.
fn backfill_repo(&self, hits: &mut [SearchHit]) { fn backfill_repo(&self, hits: &mut [SearchHit]) {
use std::collections::HashMap;
use kebab_core::DocumentId; use kebab_core::DocumentId;
use std::collections::HashMap;
// doc_id → Option<String> where None means "not found / no repo" // doc_id → Option<String> where None means "not found / no repo"
let mut cache: HashMap<DocumentId, Option<String>> = HashMap::new(); let mut cache: HashMap<DocumentId, Option<String>> = HashMap::new();
@@ -956,26 +938,24 @@ impl App {
if hit.repo.is_some() { if hit.repo.is_some() {
continue; continue;
} }
let repo_val = cache let repo_val = cache.entry(hit.doc_id.clone()).or_insert_with(|| {
.entry(hit.doc_id.clone()) // Deliberately non-aborting: a failed store lookup for
.or_insert_with(|| { // one hit must not abort the whole search response. Log
// Deliberately non-aborting: a failed store lookup for // the error so it's observable rather than silently
// one hit must not abort the whole search response. Log // dropped (review #140 round 1).
// the error so it's observable rather than silently match self.sqlite.get_document(&hit.doc_id) {
// dropped (review #140 round 1). Ok(opt) => opt.and_then(|doc| doc.metadata.repo),
match self.sqlite.get_document(&hit.doc_id) { Err(e) => {
Ok(opt) => opt.and_then(|doc| doc.metadata.repo), tracing::warn!(
Err(e) => { target: "kebab-app",
tracing::warn!( doc_id = %hit.doc_id,
target: "kebab-app", error = %e,
doc_id = %hit.doc_id, "backfill_repo: get_document failed; leaving hit.repo = None"
error = %e, );
"backfill_repo: get_document failed; leaving hit.repo = None" None
);
None
}
} }
}); }
});
if let Some(r) = repo_val { if let Some(r) = repo_val {
hit.repo = Some(r.clone()); hit.repo = Some(r.clone());
} }
@@ -986,10 +966,7 @@ impl App {
/// "switch to --mode lexical" error when embeddings are disabled. /// "switch to --mode lexical" error when embeddings are disabled.
fn require_embeddings( fn require_embeddings(
&self, &self,
) -> Result<( ) -> Result<(Arc<dyn Embedder + Send + Sync>, Arc<LanceVectorStore>)> {
Arc<dyn Embedder + Send + Sync>,
Arc<LanceVectorStore>,
)> {
let emb = self.embedder()?.ok_or_else(|| { let emb = self.embedder()?.ok_or_else(|| {
anyhow!( anyhow!(
"embeddings disabled (config.models.embedding.provider == \"none\" \ "embeddings disabled (config.models.embedding.provider == \"none\" \
@@ -1278,8 +1255,8 @@ mod tests_extractor_dispatch {
MediaType::Code("kotlin".into()), MediaType::Code("kotlin".into()),
MediaType::Code("c".into()), MediaType::Code("c".into()),
MediaType::Code("cpp".into()), MediaType::Code("cpp".into()),
MediaType::Code("yaml".into()), // registry NOT cover MediaType::Code("yaml".into()), // registry NOT cover
MediaType::Code("shell".into()), // registry NOT cover MediaType::Code("shell".into()), // registry NOT cover
MediaType::Audio(AudioType::Wav), // registry NOT cover MediaType::Audio(AudioType::Wav), // registry NOT cover
]; ];
for sample in &samples { for sample in &samples {

View File

@@ -215,7 +215,10 @@ fn parse_one(raw: &Value) -> Result<(SearchQuery, SearchOpts), String> {
.and_then(serde_json::Value::as_u64) .and_then(serde_json::Value::as_u64)
.map(|n| n as usize), .map(|n| n as usize),
cursor: obj.get("cursor").and_then(|v| v.as_str()).map(String::from), cursor: obj.get("cursor").and_then(|v| v.as_str()).map(String::from),
trace: obj.get("trace").and_then(serde_json::Value::as_bool).unwrap_or(false), trace: obj
.get("trace")
.and_then(serde_json::Value::as_bool)
.unwrap_or(false),
}; };
Ok(( Ok((

View File

@@ -10,6 +10,6 @@
pub use crate::doctor_signal::{DoctorUnhealthy, NoHitSignal, RefusalSignal}; pub use crate::doctor_signal::{DoctorUnhealthy, NoHitSignal, RefusalSignal};
pub use kebab_llm_local::LlmError;
pub use kebab_config::{ConfigInvalid, ConfigNotFound}; pub use kebab_config::{ConfigInvalid, ConfigNotFound};
pub use kebab_llm_local::LlmError;
pub use kebab_store_sqlite::NotIndexed; pub use kebab_store_sqlite::NotIndexed;

View File

@@ -172,7 +172,10 @@ mod tests {
}); });
let v1 = classify(&err, false); let v1 = classify(&err, false);
assert_eq!(v1.code, "config_invalid"); assert_eq!(v1.code, "config_invalid");
assert_eq!(v1.details.get("path").and_then(|p| p.as_str()), Some("/tmp/x.toml")); assert_eq!(
v1.details.get("path").and_then(|p| p.as_str()),
Some("/tmp/x.toml")
);
assert!(v1.hint.is_some()); assert!(v1.hint.is_some());
} }
@@ -196,7 +199,8 @@ mod tests {
// the resulting LlmError::Unreachable maps to "model_unreachable". // the resulting LlmError::Unreachable maps to "model_unreachable".
let client = reqwest::blocking::Client::builder() let client = reqwest::blocking::Client::builder()
.timeout(std::time::Duration::from_millis(500)) .timeout(std::time::Duration::from_millis(500))
.build().unwrap(); .build()
.unwrap();
let err = client.get("http://127.0.0.1:1").send().unwrap_err(); let err = client.get("http://127.0.0.1:1").send().unwrap_err();
let llm = LlmError::Unreachable { let llm = LlmError::Unreachable {
endpoint: "http://127.0.0.1:1".to_string(), endpoint: "http://127.0.0.1:1".to_string(),
@@ -212,7 +216,10 @@ mod tests {
let llm = LlmError::ModelNotPulled("gemma4:e4b".to_string()); let llm = LlmError::ModelNotPulled("gemma4:e4b".to_string());
let v1 = classify(&anyhow::Error::new(llm), false); let v1 = classify(&anyhow::Error::new(llm), false);
assert_eq!(v1.code, "model_not_pulled"); assert_eq!(v1.code, "model_not_pulled");
assert_eq!(v1.details.get("model").and_then(|p| p.as_str()), Some("gemma4:e4b")); assert_eq!(
v1.details.get("model").and_then(|p| p.as_str()),
Some("gemma4:e4b")
);
} }
#[test] #[test]
@@ -249,7 +256,10 @@ mod tests {
// (single source of truth). classify must not pattern-match on // (single source of truth). classify must not pattern-match on
// anyhow string contents — that would create two sources of // anyhow string contents — that would create two sources of
// truth. The bare anyhow string falls through to "generic". // truth. The bare anyhow string falls through to "generic".
assert_ne!(v1.code, "stale_cursor", "classify must not produce stale_cursor from bare anyhow string"); assert_ne!(
v1.code, "stale_cursor",
"classify must not produce stale_cursor from bare anyhow string"
);
} }
#[test] #[test]

View File

@@ -36,9 +36,7 @@ pub fn ensure_kebabignore_entry(workspace_root: &Path) -> Result<()> {
} else { } else {
String::new() String::new()
}; };
let already = existing let already = existing.lines().any(|line| line.trim() == KEBABIGNORE_LINE);
.lines()
.any(|line| line.trim() == KEBABIGNORE_LINE);
if already { if already {
return Ok(()); return Ok(());
} }
@@ -57,11 +55,7 @@ pub fn ensure_kebabignore_entry(workspace_root: &Path) -> Result<()> {
/// Copy bytes to `<external_dir>/<blake3-12>.<ext>`. Idempotent — if the /// Copy bytes to `<external_dir>/<blake3-12>.<ext>`. Idempotent — if the
/// destination file already exists with the expected hash, the existing /// destination file already exists with the expected hash, the existing
/// file is reused (no second write). Returns the destination path. /// file is reused (no second write). Returns the destination path.
pub fn copy_to_external( pub fn copy_to_external(external_dir: &Path, bytes: &[u8], ext: &str) -> Result<PathBuf> {
external_dir: &Path,
bytes: &[u8],
ext: &str,
) -> Result<PathBuf> {
let hash = blake3::hash(bytes); let hash = blake3::hash(bytes);
let hex = hash.to_hex(); let hex = hash.to_hex();
let prefix = &hex.as_str()[..12]; let prefix = &hex.as_str()[..12];
@@ -82,11 +76,7 @@ pub fn copy_to_external(
/// Internal `yaml_quote` always uses double-quoted YAML form with backslash /// Internal `yaml_quote` always uses double-quoted YAML form with backslash
/// escapes for `"` / `\` / control chars — agent-supplied titles with /// escapes for `"` / `\` / control chars — agent-supplied titles with
/// special characters are safe. /// special characters are safe.
pub fn inject_frontmatter( pub fn inject_frontmatter(body: &str, title: &str, source_uri: Option<&str>) -> Result<String> {
body: &str,
title: &str,
source_uri: Option<&str>,
) -> Result<String> {
let head = body.trim_start(); let head = body.trim_start();
if head.starts_with("---\n") || head.starts_with("---\r\n") || head.starts_with("---\r") { if head.starts_with("---\n") || head.starts_with("---\r\n") || head.starts_with("---\r") {
anyhow::bail!( anyhow::bail!(

View File

@@ -50,14 +50,14 @@ impl App {
fn fetch_chunk(app: &App, id: ChunkId, opts: FetchOpts) -> Result<FetchResult> { fn fetch_chunk(app: &App, id: ChunkId, opts: FetchOpts) -> Result<FetchResult> {
let target = <kebab_store_sqlite::SqliteStore as DocumentStore>::get_chunk(&app.sqlite, &id)? let target = <kebab_store_sqlite::SqliteStore as DocumentStore>::get_chunk(&app.sqlite, &id)?
.ok_or_else(|| { .ok_or_else(|| {
anyhow::Error::new(StructuredError(ErrorV1 { anyhow::Error::new(StructuredError(ErrorV1 {
schema_version: ERROR_V1_ID.to_string(), schema_version: ERROR_V1_ID.to_string(),
code: "chunk_not_found".to_string(), code: "chunk_not_found".to_string(),
message: format!("chunk_id '{}' not found", id.0), message: format!("chunk_id '{}' not found", id.0),
details: serde_json::Value::Null, details: serde_json::Value::Null,
hint: None, hint: None,
})) }))
})?; })?;
let doc_id = target.doc_id.clone(); let doc_id = target.doc_id.clone();
let doc = let doc =
@@ -107,14 +107,14 @@ fn fetch_chunk(app: &App, id: ChunkId, opts: FetchOpts) -> Result<FetchResult> {
fn fetch_doc(app: &App, id: DocumentId, opts: FetchOpts) -> Result<FetchResult> { fn fetch_doc(app: &App, id: DocumentId, opts: FetchOpts) -> Result<FetchResult> {
let doc = <kebab_store_sqlite::SqliteStore as DocumentStore>::get_document(&app.sqlite, &id)? let doc = <kebab_store_sqlite::SqliteStore as DocumentStore>::get_document(&app.sqlite, &id)?
.ok_or_else(|| { .ok_or_else(|| {
anyhow::Error::new(StructuredError(ErrorV1 { anyhow::Error::new(StructuredError(ErrorV1 {
schema_version: ERROR_V1_ID.to_string(), schema_version: ERROR_V1_ID.to_string(),
code: "doc_not_found".to_string(), code: "doc_not_found".to_string(),
message: format!("doc_id '{}' not found", id.0), message: format!("doc_id '{}' not found", id.0),
details: serde_json::Value::Null, details: serde_json::Value::Null,
hint: None, hint: None,
})) }))
})?; })?;
let mut text = fmt_canonical_to_markdown(&doc); let mut text = fmt_canonical_to_markdown(&doc);
let mut truncated = false; let mut truncated = false;
@@ -176,14 +176,14 @@ fn fetch_span(
) -> Result<FetchResult> { ) -> Result<FetchResult> {
let doc = <kebab_store_sqlite::SqliteStore as DocumentStore>::get_document(&app.sqlite, &id)? let doc = <kebab_store_sqlite::SqliteStore as DocumentStore>::get_document(&app.sqlite, &id)?
.ok_or_else(|| { .ok_or_else(|| {
anyhow::Error::new(StructuredError(ErrorV1 { anyhow::Error::new(StructuredError(ErrorV1 {
schema_version: ERROR_V1_ID.to_string(), schema_version: ERROR_V1_ID.to_string(),
code: "doc_not_found".to_string(), code: "doc_not_found".to_string(),
message: format!("doc_id '{}' not found", id.0), message: format!("doc_id '{}' not found", id.0),
details: serde_json::Value::Null, details: serde_json::Value::Null,
hint: None, hint: None,
})) }))
})?; })?;
// Reject line-incompatible media types (PDF / audio). `SourceType` // Reject line-incompatible media types (PDF / audio). `SourceType`
// (markdown / note / paper / reference / inbox) is the *user-facing* // (markdown / note / paper / reference / inbox) is the *user-facing*

View File

@@ -81,7 +81,9 @@ fn generate_run_id() -> String {
use time::macros::format_description; use time::macros::format_description;
let now = time::OffsetDateTime::now_utc(); let now = time::OffsetDateTime::now_utc();
let ts = now let ts = now
.format(format_description!("[year][month][day]T[hour][minute][second]Z")) .format(format_description!(
"[year][month][day]T[hour][minute][second]Z"
))
.unwrap_or_else(|_| "19700101T000000Z".to_string()); .unwrap_or_else(|_| "19700101T000000Z".to_string());
let uid = uuid::Uuid::now_v7().simple().to_string(); let uid = uuid::Uuid::now_v7().simple().to_string();
let suffix = &uid[uid.len() - 8..]; let suffix = &uid[uid.len() - 8..];
@@ -211,8 +213,8 @@ pub(crate) fn percentiles(samples: &[u64]) -> (Option<u64>, Option<u64>, Option<
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::*; use super::*;
use tempfile::TempDir;
use kebab_config::LoggingCfg; use kebab_config::LoggingCfg;
use tempfile::TempDir;
#[test] #[test]
fn generate_run_id_has_iso_prefix_and_8_hex_suffix() { fn generate_run_id_has_iso_prefix_and_8_hex_suffix() {
@@ -224,7 +226,10 @@ mod tests {
assert!(prefix.contains('T'), "prefix should contain T: {prefix}"); assert!(prefix.contains('T'), "prefix should contain T: {prefix}");
assert!(prefix.ends_with('Z'), "prefix should end with Z: {prefix}"); assert!(prefix.ends_with('Z'), "prefix should end with Z: {prefix}");
assert_eq!(suffix.len(), 8, "suffix should be 8 chars: {suffix}"); assert_eq!(suffix.len(), 8, "suffix should be 8 chars: {suffix}");
assert!(suffix.chars().all(|c| c.is_ascii_hexdigit()), "suffix should be hex: {suffix}"); assert!(
suffix.chars().all(|c| c.is_ascii_hexdigit()),
"suffix should be hex: {suffix}"
);
} }
#[test] #[test]
@@ -256,31 +261,43 @@ mod tests {
let mut writer = IngestLogWriter::open(&cfg).unwrap().unwrap(); let mut writer = IngestLogWriter::open(&cfg).unwrap().unwrap();
let path = writer.path().to_path_buf(); let path = writer.path().to_path_buf();
writer.write_event(&LogEvent::Skip { writer
ts: now_ts(), .write_event(&LogEvent::Skip {
doc_path: "a.zip", ts: now_ts(),
reason: "builtin_blacklist", doc_path: "a.zip",
detail: Some(".zip extension"), reason: "builtin_blacklist",
}).unwrap(); detail: Some(".zip extension"),
writer.write_event(&LogEvent::Error { })
ts: now_ts(), .unwrap();
code: "ingest_fatal", writer
message: "something bad", .write_event(&LogEvent::Error {
}).unwrap(); ts: now_ts(),
writer.write_event(&LogEvent::ParseError { code: "ingest_fatal",
ts: now_ts(), message: "something bad",
doc_path: "weird.pdf", })
reason: "lopdf_error", .unwrap();
message: "unexpected EOF", writer
}).unwrap(); .write_event(&LogEvent::ParseError {
ts: now_ts(),
doc_path: "weird.pdf",
reason: "lopdf_error",
message: "unexpected EOF",
})
.unwrap();
writer.flush().unwrap(); writer.flush().unwrap();
let contents = std::fs::read_to_string(&path).unwrap(); let contents = std::fs::read_to_string(&path).unwrap();
let lines: Vec<&str> = contents.lines().collect(); let lines: Vec<&str> = contents.lines().collect();
assert_eq!(lines.len(), 3, "expected 3 lines, got: {}", lines.len()); assert_eq!(lines.len(), 3, "expected 3 lines, got: {}", lines.len());
for line in &lines { for line in &lines {
assert!(line.starts_with('{'), "each line should be JSON object: {line}"); assert!(
assert!(line.contains("\"kind\""), "each line should have 'kind': {line}"); line.starts_with('{'),
"each line should be JSON object: {line}"
);
assert!(
line.contains("\"kind\""),
"each line should have 'kind': {line}"
);
} }
} }
@@ -293,14 +310,19 @@ mod tests {
}; };
let mut writer = IngestLogWriter::open(&cfg).unwrap().unwrap(); let mut writer = IngestLogWriter::open(&cfg).unwrap().unwrap();
let path = writer.path().to_path_buf(); let path = writer.path().to_path_buf();
writer.write_event(&LogEvent::Error { writer
ts: now_ts(), .write_event(&LogEvent::Error {
code: "test", ts: now_ts(),
message: "drop flush test", code: "test",
}).unwrap(); message: "drop flush test",
})
.unwrap();
// Drop without explicit flush — Drop impl should flush BufWriter. // Drop without explicit flush — Drop impl should flush BufWriter.
drop(writer); drop(writer);
let contents = std::fs::read_to_string(&path).unwrap(); let contents = std::fs::read_to_string(&path).unwrap();
assert!(contents.lines().count() >= 1, "file should have at least 1 line after drop"); assert!(
contents.lines().count() >= 1,
"file should have at least 1 line after drop"
);
} }
} }

View File

@@ -145,10 +145,7 @@ pub fn render_skipped_breakdown(map: &std::collections::BTreeMap<String, u32>) -
/// Best-effort send into an optional `mpsc::Sender`. A dropped receiver /// Best-effort send into an optional `mpsc::Sender`. A dropped receiver
/// is silently absorbed — the ingest hot path must not stall on a slow /// is silently absorbed — the ingest hot path must not stall on a slow
/// consumer. Logged at `trace` for diagnostics. /// consumer. Logged at `trace` for diagnostics.
pub(crate) fn emit( pub(crate) fn emit(progress: Option<&std::sync::mpsc::Sender<IngestEvent>>, event: IngestEvent) {
progress: Option<&std::sync::mpsc::Sender<IngestEvent>>,
event: IngestEvent,
) {
if let Some(tx) = progress { if let Some(tx) = progress {
if tx.send(event).is_err() { if tx.send(event).is_err() {
tracing::trace!( tracing::trace!(
@@ -192,7 +189,10 @@ mod tests {
media: "markdown".into(), media: "markdown".into(),
}; };
let v = serde_json::to_value(&ev).unwrap(); let v = serde_json::to_value(&ev).unwrap();
assert_eq!(v.get("kind").and_then(|s| s.as_str()), Some("asset_started")); assert_eq!(
v.get("kind").and_then(|s| s.as_str()),
Some("asset_started")
);
assert_eq!(v.get("idx").and_then(serde_json::Value::as_u64), Some(1)); assert_eq!(v.get("idx").and_then(serde_json::Value::as_u64), Some(1));
assert_eq!(v.get("total").and_then(serde_json::Value::as_u64), Some(10)); assert_eq!(v.get("total").and_then(serde_json::Value::as_u64), Some(10));
assert_eq!(v.get("path").and_then(|s| s.as_str()), Some("notes/foo.md")); assert_eq!(v.get("path").and_then(|s| s.as_str()), Some("notes/foo.md"));
@@ -211,8 +211,14 @@ mod tests {
let v = serde_json::to_value(&ev).unwrap(); let v = serde_json::to_value(&ev).unwrap();
assert_eq!(v.get("kind").and_then(|s| s.as_str()), Some("completed")); assert_eq!(v.get("kind").and_then(|s| s.as_str()), Some("completed"));
let counts = v.get("counts").unwrap(); let counts = v.get("counts").unwrap();
assert_eq!(counts.get("scanned").and_then(serde_json::Value::as_u64), Some(5)); assert_eq!(
assert_eq!(counts.get("new").and_then(serde_json::Value::as_u64), Some(2)); counts.get("scanned").and_then(serde_json::Value::as_u64),
Some(5)
);
assert_eq!(
counts.get("new").and_then(serde_json::Value::as_u64),
Some(2)
);
} }
#[test] #[test]

View File

@@ -39,13 +39,17 @@ use std::sync::{Arc, Mutex};
use anyhow::{Context, anyhow}; use anyhow::{Context, anyhow};
use serde::{Deserialize, Serialize}; use serde::{Deserialize, Serialize};
use kebab_chunk::{CodeCAstV1Chunker, CodeCppAstV1Chunker, CodeGoAstV1Chunker, CodeJavaAstV1Chunker, CodeJsAstV1Chunker, CodeKotlinAstV1Chunker, CodePythonAstV1Chunker, CodeRustAstV1Chunker, CodeTextParagraphV1Chunker, CodeTsAstV1Chunker, DockerfileFileV1Chunker, K8sManifestResourceV1Chunker, ManifestFileV1Chunker, MdHeadingV1Chunker, PdfPageV1Chunker}; use kebab_chunk::{
CodeCAstV1Chunker, CodeCppAstV1Chunker, CodeGoAstV1Chunker, CodeJavaAstV1Chunker,
CodeJsAstV1Chunker, CodeKotlinAstV1Chunker, CodePythonAstV1Chunker, CodeRustAstV1Chunker,
CodeTextParagraphV1Chunker, CodeTsAstV1Chunker, DockerfileFileV1Chunker,
K8sManifestResourceV1Chunker, ManifestFileV1Chunker, MdHeadingV1Chunker, PdfPageV1Chunker,
};
use kebab_core::{ use kebab_core::{
Answer, Block, CanonicalDocument, Chunk, ChunkId, ChunkPolicy, ChunkerVersion, Chunker, Answer, Block, CanonicalDocument, Chunk, ChunkId, ChunkPolicy, Chunker, ChunkerVersion,
DocFilter, DocSummary, DocumentId, DocumentStore, Embedder, EmbeddingInput, DocFilter, DocSummary, DocumentId, DocumentStore, Embedder, EmbeddingInput, EmbeddingKind,
EmbeddingKind, ExtractContext, IngestReport, Lang, LanguageModel, MediaType, ExtractContext, IngestReport, Lang, LanguageModel, MediaType, ParserVersion, RawAsset,
ParserVersion, RawAsset, SearchHit, SearchQuery, SourceScope, SearchHit, SearchQuery, SourceScope, SourceUri, VectorRecord, VectorStore,
SourceUri, VectorRecord, VectorStore,
}; };
use kebab_llm_local::OllamaLanguageModel; use kebab_llm_local::OllamaLanguageModel;
use kebab_parse_image::{OcrEngine, OllamaVisionOcr, apply_caption, apply_ocr}; use kebab_parse_image::{OcrEngine, OllamaVisionOcr, apply_caption, apply_ocr};
@@ -69,15 +73,17 @@ pub mod schema;
mod staleness; mod staleness;
pub use app::{App, SearchResponse, short_query_hint}; pub use app::{App, SearchResponse, short_query_hint};
pub use ingest_log::{IngestLogWriter, IngestSummary, LogEvent};
pub use ingest_progress::{AggregateCounts, IngestEvent, render_skipped_breakdown};
pub use reset::{ResetReport, ResetScope, enumerate_orphans};
pub use error_wire::{ERROR_V1_ID, ErrorV1, StructuredError, classify};
pub use kebab_config::{ConfigInvalid, ConfigNotFound};
pub use fetch::fetch_with_config;
#[doc(hidden)] #[doc(hidden)]
pub use bulk::{BULK_QUERIES_MAX, bulk_search_with_config}; pub use bulk::{BULK_QUERIES_MAX, bulk_search_with_config};
pub use schema::{Capabilities, Models, SCHEMA_V1_ID, SchemaV1, Stats, WireBlock, schema_with_config}; pub use error_wire::{ERROR_V1_ID, ErrorV1, StructuredError, classify};
pub use fetch::fetch_with_config;
pub use ingest_log::{IngestLogWriter, IngestSummary, LogEvent};
pub use ingest_progress::{AggregateCounts, IngestEvent, render_skipped_breakdown};
pub use kebab_config::{ConfigInvalid, ConfigNotFound};
pub use reset::{ResetReport, ResetScope, enumerate_orphans};
pub use schema::{
Capabilities, Models, SCHEMA_V1_ID, SchemaV1, Stats, WireBlock, schema_with_config,
};
pub use staleness::{compute_stale, mark_stale_in_place}; pub use staleness::{compute_stale, mark_stale_in_place};
/// p9-fb-25: sentinel for files without an extension in /// p9-fb-25: sentinel for files without an extension in
@@ -322,8 +328,8 @@ pub fn ingest_with_config_opts(
root: scope.root.to_string_lossy().into_owned(), root: scope.root.to_string_lossy().into_owned(),
}, },
); );
let connector = FsSourceConnector::new(&app.config) let connector =
.context("kb-app::ingest: build FsSourceConnector")?; FsSourceConnector::new(&app.config).context("kb-app::ingest: build FsSourceConnector")?;
let (assets, fs_skips) = connector let (assets, fs_skips) = connector
.scan_with_skips(&scope) .scan_with_skips(&scope)
.context("kb-app::ingest: scan workspace")?; .context("kb-app::ingest: scan workspace")?;
@@ -372,18 +378,14 @@ pub fn ingest_with_config_opts(
// endpoint) aborts ingest fail-fast — better than silently disabling // endpoint) aborts ingest fail-fast — better than silently disabling
// OCR/caption mid-run. // OCR/caption mid-run.
let ocr_engine: Option<OllamaVisionOcr> = if app.config.image.ocr.enabled { let ocr_engine: Option<OllamaVisionOcr> = if app.config.image.ocr.enabled {
Some( Some(OllamaVisionOcr::new(&app.config).context("kb-app::ingest: build OllamaVisionOcr")?)
OllamaVisionOcr::new(&app.config)
.context("kb-app::ingest: build OllamaVisionOcr")?,
)
} else { } else {
None None
}; };
let caption_llm: Option<Box<dyn LanguageModel>> = if app.config.image.caption.enabled { let caption_llm: Option<Box<dyn LanguageModel>> = if app.config.image.caption.enabled {
Some(Box::new( Some(Box::new(OllamaLanguageModel::new(&app.config).context(
OllamaLanguageModel::new(&app.config) "kb-app::ingest: build OllamaLanguageModel for caption",
.context("kb-app::ingest: build OllamaLanguageModel for caption")?, )?))
))
} else { } else {
None None
}; };
@@ -440,10 +442,8 @@ pub fn ingest_with_config_opts(
// current walker scope (config narrowing / include-glob change) is // current walker scope (config narrowing / include-glob change) is
// NOT purged — we leave it in place to protect against accidental // NOT purged — we leave it in place to protect against accidental
// data loss via config edits. // data loss via config edits.
let scanned_paths: std::collections::HashSet<kebab_core::WorkspacePath> = assets let scanned_paths: std::collections::HashSet<kebab_core::WorkspacePath> =
.iter() assets.iter().map(|a| a.workspace_path.clone()).collect();
.map(|a| a.workspace_path.clone())
.collect();
let purged_deleted_files = sweep_deleted_files( let purged_deleted_files = sweep_deleted_files(
&app, &app,
&scanned_paths, &scanned_paths,
@@ -659,8 +659,7 @@ pub fn ingest_with_config_opts(
} }
} }
let duration_ms = u32::try_from(started_instant.elapsed().as_millis()) let duration_ms = u32::try_from(started_instant.elapsed().as_millis()).unwrap_or(u32::MAX);
.unwrap_or(u32::MAX);
let finished_at = time::OffsetDateTime::now_utc(); let finished_at = time::OffsetDateTime::now_utc();
// Record the ingest_runs row with aggregate counts. // Record the ingest_runs row with aggregate counts.
@@ -941,8 +940,8 @@ fn try_skip_unchanged(
if stored_is_tier3_fallback { if stored_is_tier3_fallback {
// Embedder version still must match. // Embedder version still must match.
let embedder_match = existing_doc.last_embedding_version.as_ref() let embedder_match =
== current_embedding_version; existing_doc.last_embedding_version.as_ref() == current_embedding_version;
if !embedder_match { if !embedder_match {
return Ok(None); return Ok(None);
} }
@@ -986,23 +985,17 @@ fn try_skip_unchanged(
// sentinel removes every doc at this path (the new doc_id is // sentinel removes every doc at this path (the new doc_id is
// not yet known here — it's computed downstream from the new // not yet known here — it's computed downstream from the new
// PARSER_VERSION). // PARSER_VERSION).
purge_workspace_path_for_parser_bump(app, asset).with_context(|| { purge_workspace_path_for_parser_bump(app, asset)
format!( .with_context(|| format!("parser-bump orphan purge at {}", asset.workspace_path.0))?;
"parser-bump orphan purge at {}",
asset.workspace_path.0
)
})?;
return Ok(None); return Ok(None);
} }
// 3. Chunker unchanged. // 3. Chunker unchanged.
let chunker_match = existing_doc.last_chunker_version.as_ref() let chunker_match = existing_doc.last_chunker_version.as_ref() == Some(current_chunker_version);
== Some(current_chunker_version);
if !chunker_match { if !chunker_match {
return Ok(None); return Ok(None);
} }
// 4. Embedder unchanged. // 4. Embedder unchanged.
let embedder_match = existing_doc.last_embedding_version.as_ref() let embedder_match = existing_doc.last_embedding_version.as_ref() == current_embedding_version;
== current_embedding_version;
if !embedder_match { if !embedder_match {
return Ok(None); return Ok(None);
} }
@@ -1038,7 +1031,8 @@ fn try_skip_unchanged(
fn ext_for_skip_warning(path: &str) -> String { fn ext_for_skip_warning(path: &str) -> String {
std::path::Path::new(path) std::path::Path::new(path)
.extension() .extension()
.and_then(|s| s.to_str()).map_or_else(|| NO_EXT_SENTINEL.to_string(), str::to_ascii_lowercase) .and_then(|s| s.to_str())
.map_or_else(|| NO_EXT_SENTINEL.to_string(), str::to_ascii_lowercase)
} }
/// p9-fb-25: render the `IngestItem.warnings` line for a Skipped /// p9-fb-25: render the `IngestItem.warnings` line for a Skipped
@@ -1121,10 +1115,26 @@ fn ingest_one_asset(
} }
// p10-1A-2 / 1B: code ingest dispatch. p10-2: Tier 2 langs added. p10-3: shell added. p10-1D: c/cpp added. // p10-1A-2 / 1B: code ingest dispatch. p10-2: Tier 2 langs added. p10-3: shell added. p10-1D: c/cpp added.
MediaType::Code(lang) MediaType::Code(lang)
if matches!(lang.as_str(), if matches!(
"rust" | "python" | "typescript" | "javascript" | "go" | "java" | "kotlin" lang.as_str(),
| "yaml" | "dockerfile" | "toml" | "json" | "xml" | "groovy" | "go-mod" "rust"
| "shell" | "c" | "cpp") => | "python"
| "typescript"
| "javascript"
| "go"
| "java"
| "kotlin"
| "yaml"
| "dockerfile"
| "toml"
| "json"
| "xml"
| "groovy"
| "go-mod"
| "shell"
| "c"
| "cpp"
) =>
{ {
return ingest_one_code_asset( return ingest_one_code_asset(
app, app,
@@ -1204,16 +1214,17 @@ fn ingest_one_asset(
// Frontmatter — `parse_frontmatter` returns Ok even on malformed // Frontmatter — `parse_frontmatter` returns Ok even on malformed
// frontmatter (warnings are surfaced through the `Vec<Warning>`). // frontmatter (warnings are surfaced through the `Vec<Warning>`).
let (metadata, fm_span, fm_warns) = parse_frontmatter(&bytes, &body_hints) let (metadata, fm_span, fm_warns) =
.context("kb-parse-md::parse_frontmatter")?; parse_frontmatter(&bytes, &body_hints).context("kb-parse-md::parse_frontmatter")?;
let body_offset_lines = match fm_span { let body_offset_lines = match fm_span {
Some(span) => count_lines_in(&bytes[..span.end]), Some(span) => count_lines_in(&bytes[..span.end]),
None => 0, None => 0,
}; };
let (parsed_blocks, blk_warns) = parse_blocks(&bytes[fm_span_end(fm_span)..], body_offset_lines) let (parsed_blocks, blk_warns) =
.context("kb-parse-md::parse_blocks")?; parse_blocks(&bytes[fm_span_end(fm_span)..], body_offset_lines)
.context("kb-parse-md::parse_blocks")?;
let mut all_warnings = Vec::with_capacity(fm_warns.len() + blk_warns.len()); let mut all_warnings = Vec::with_capacity(fm_warns.len() + blk_warns.len());
all_warnings.extend(fm_warns); all_warnings.extend(fm_warns);
@@ -1226,14 +1237,9 @@ fn ingest_one_asset(
.map(|w| format!("{:?}: {}", w.kind, w.note)) .map(|w| format!("{:?}: {}", w.kind, w.note))
.collect(); .collect();
let mut canonical = build_canonical_document( let mut canonical =
asset, build_canonical_document(asset, metadata, parsed_blocks, parser_version, all_warnings)
metadata, .context("kb-parse-md::build_canonical_document")?;
parsed_blocks,
parser_version,
all_warnings,
)
.context("kb-parse-md::build_canonical_document")?;
let chunks = MdHeadingV1Chunker let chunks = MdHeadingV1Chunker
.chunk(&canonical, chunk_policy) .chunk(&canonical, chunk_policy)
@@ -1300,9 +1306,7 @@ fn ingest_one_asset(
dimensions, dimensions,
}) })
.collect(); .collect();
vec_store vec_store.upsert(&records).context("VectorStore::upsert")?;
.upsert(&records)
.context("VectorStore::upsert")?;
} }
} }
@@ -1367,9 +1371,7 @@ fn ingest_one_image_asset(
chunk_count: None, chunk_count: None,
parser_version: None, parser_version: None,
chunker_version: None, chunker_version: None,
warnings: vec![ warnings: vec!["kb:// URI not yet supported".to_string()],
"kb:// URI not yet supported".to_string(),
],
pdf_ocr_pages: None, pdf_ocr_pages: None,
pdf_ocr_ms_total: None, pdf_ocr_ms_total: None,
error: None, error: None,
@@ -1481,17 +1483,19 @@ fn ingest_one_image_asset(
"image document missing leading ImageRef block — OCR/caption skipped (first block: {:?})", "image document missing leading ImageRef block — OCR/caption skipped (first block: {:?})",
other.map(|b| std::mem::discriminant(b)) other.map(|b| std::mem::discriminant(b))
); );
canonical.provenance.events.push(kebab_core::ProvenanceEvent { canonical
at: now, .provenance
agent: "kb-app".to_string(), .events
kind: kebab_core::ProvenanceKind::Warning, .push(kebab_core::ProvenanceEvent {
note: Some( at: now,
"image document missing leading ImageRef block — OCR/caption skipped" agent: "kb-app".to_string(),
.to_string(), kind: kebab_core::ProvenanceKind::Warning,
), note: Some(
}); "image document missing leading ImageRef block — OCR/caption skipped"
warning_notes .to_string(),
.push("ImageDispatchAnomaly: missing ImageRef block".to_string()); ),
});
warning_notes.push("ImageDispatchAnomaly: missing ImageRef block".to_string());
} }
} }
@@ -1639,10 +1643,7 @@ fn record_image_analysis_failure(
/// 3. Sweeps the SQLite `documents` row (CASCADE drops `blocks` / /// 3. Sweeps the SQLite `documents` row (CASCADE drops `blocks` /
/// `chunks` / `embedding_records`). The `assets` row stays — same /// `chunks` / `embedding_records`). The `assets` row stays — same
/// bytes, same asset_id, only the derived `doc_id` changed. /// bytes, same asset_id, only the derived `doc_id` changed.
fn purge_workspace_path_for_parser_bump( fn purge_workspace_path_for_parser_bump(app: &App, asset: &RawAsset) -> anyhow::Result<()> {
app: &App,
asset: &RawAsset,
) -> anyhow::Result<()> {
let path = &asset.workspace_path.0; let path = &asset.workspace_path.0;
let stale = app let stale = app
.sqlite .sqlite
@@ -1777,21 +1778,19 @@ fn sweep_deleted_files(
} }
// File is truly absent → purge. // File is truly absent → purge.
let chunk_ids = match kebab_store_sqlite::purge_deleted_workspace_path( let chunk_ids =
&app.sqlite, match kebab_store_sqlite::purge_deleted_workspace_path(&app.sqlite, &stored_path) {
&stored_path, Ok(ids) => ids,
) { Err(e) => {
Ok(ids) => ids, tracing::warn!(
Err(e) => { target: "kebab-app",
tracing::warn!( path = %stored_path.0,
target: "kebab-app", error = %e,
path = %stored_path.0, "sweep_deleted_files: purge failed; skipping this path"
error = %e, );
"sweep_deleted_files: purge failed; skipping this path" continue;
); }
continue; };
}
};
// Purge associated vectors (best-effort; partial failure // Purge associated vectors (best-effort; partial failure
// acceptable — orphan vectors get cleaned by `kebab reset // acceptable — orphan vectors get cleaned by `kebab reset
@@ -1875,9 +1874,7 @@ fn ingest_one_pdf_asset(
chunk_count: None, chunk_count: None,
parser_version: None, parser_version: None,
chunker_version: None, chunker_version: None,
warnings: vec![ warnings: vec!["kb:// URI not yet supported".to_string()],
"kb:// URI not yet supported".to_string(),
],
pdf_ocr_pages: None, pdf_ocr_pages: None,
pdf_ocr_ms_total: None, pdf_ocr_ms_total: None,
error: None, error: None,
@@ -1946,9 +1943,7 @@ fn ingest_one_pdf_asset(
crate::pdf_ocr_apply::PdfOcrProgress::Started { page } => { crate::pdf_ocr_apply::PdfOcrProgress::Started { page } => {
if let Some(sender) = progress { if let Some(sender) = progress {
let _ = sender.send( let _ = sender.send(
crate::ingest_progress::IngestEvent::PdfOcrStarted { crate::ingest_progress::IngestEvent::PdfOcrStarted { page },
page,
},
); );
} }
} }
@@ -1996,9 +1991,13 @@ fn ingest_one_pdf_asset(
}); });
} }
} }
if let Ok(mut p) = pages_for_ocr.lock() { *p += 1; } if let Ok(mut p) = pages_for_ocr.lock() {
*p += 1;
}
if success { if success {
if let Ok(mut s) = samples_for_ocr.lock() { s.push(ms); } if let Ok(mut s) = samples_for_ocr.lock() {
s.push(ms);
}
} else if let Ok(mut f) = failures_for_ocr.lock() { } else if let Ok(mut f) = failures_for_ocr.lock() {
*f += 1; *f += 1;
} }
@@ -2053,9 +2052,7 @@ fn ingest_one_pdf_asset(
kind: EmbeddingKind::Document, kind: EmbeddingKind::Document,
}) })
.collect(); .collect();
let vectors = emb let vectors = emb.embed(&inputs).context("Embedder::embed (pdf chunks)")?;
.embed(&inputs)
.context("Embedder::embed (pdf chunks)")?;
let model_id = emb.model_id(); let model_id = emb.model_id();
let model_version = emb.model_version(); let model_version = emb.model_version();
let dimensions = emb.dimensions(); let dimensions = emb.dimensions();
@@ -2139,7 +2136,7 @@ fn ingest_one_code_asset(
vector_store: Option<&Arc<kebab_store_vector::LanceVectorStore>>, vector_store: Option<&Arc<kebab_store_vector::LanceVectorStore>>,
existing_doc_ids: &std::collections::HashSet<String>, existing_doc_ids: &std::collections::HashSet<String>,
force_reingest: bool, force_reingest: bool,
code_lang: &str, // <-- NEW (p10-1b Task D) code_lang: &str, // <-- NEW (p10-1b Task D)
) -> anyhow::Result<kebab_core::IngestItem> { ) -> anyhow::Result<kebab_core::IngestItem> {
let path = match &asset.source_uri { let path = match &asset.source_uri {
SourceUri::File(p) => p.clone(), SourceUri::File(p) => p.clone(),
@@ -2154,9 +2151,7 @@ fn ingest_one_code_asset(
chunk_count: None, chunk_count: None,
parser_version: None, parser_version: None,
chunker_version: None, chunker_version: None,
warnings: vec![ warnings: vec!["kb:// URI not yet supported".to_string()],
"kb:// URI not yet supported".to_string(),
],
pdf_ocr_pages: None, pdf_ocr_pages: None,
pdf_ocr_ms_total: None, pdf_ocr_ms_total: None,
error: None, error: None,
@@ -2166,43 +2161,43 @@ fn ingest_one_code_asset(
// p10-1b Task D/G/J: parser_version per-lang. // p10-1b Task D/G/J: parser_version per-lang.
let parser_version = match code_lang { let parser_version = match code_lang {
"rust" => ParserVersion(kebab_parse_code::RUST_PARSER_VERSION.to_string()), "rust" => ParserVersion(kebab_parse_code::RUST_PARSER_VERSION.to_string()),
"python" => ParserVersion(kebab_parse_code::PYTHON_PARSER_VERSION.to_string()), "python" => ParserVersion(kebab_parse_code::PYTHON_PARSER_VERSION.to_string()),
"typescript" => ParserVersion(kebab_parse_code::TS_PARSER_VERSION.to_string()), "typescript" => ParserVersion(kebab_parse_code::TS_PARSER_VERSION.to_string()),
"javascript" => ParserVersion(kebab_parse_code::JS_PARSER_VERSION.to_string()), "javascript" => ParserVersion(kebab_parse_code::JS_PARSER_VERSION.to_string()),
"go" => ParserVersion(kebab_parse_code::GO_PARSER_VERSION.to_string()), "go" => ParserVersion(kebab_parse_code::GO_PARSER_VERSION.to_string()),
"java" => ParserVersion(kebab_parse_code::JAVA_PARSER_VERSION.to_string()), "java" => ParserVersion(kebab_parse_code::JAVA_PARSER_VERSION.to_string()),
"kotlin" => ParserVersion(kebab_parse_code::KOTLIN_PARSER_VERSION.to_string()), "kotlin" => ParserVersion(kebab_parse_code::KOTLIN_PARSER_VERSION.to_string()),
// p10-2: Tier 2 has no parse step — sentinel "none-v1". // p10-2: Tier 2 has no parse step — sentinel "none-v1".
"yaml" | "dockerfile" | "toml" | "json" | "xml" | "groovy" | "go-mod" "yaml" | "dockerfile" | "toml" | "json" | "xml" | "groovy" | "go-mod" => {
=> ParserVersion("none-v1".to_string()), ParserVersion("none-v1".to_string())
}
// p10-3: shell direct routes to Tier 3 (no parse step). // p10-3: shell direct routes to Tier 3 (no parse step).
"shell" => ParserVersion("none-v1".to_string()), "shell" => ParserVersion("none-v1".to_string()),
// p10-1D: C + C++ AST extractors. // p10-1D: C + C++ AST extractors.
"c" => ParserVersion(kebab_parse_code::C_PARSER_VERSION.to_string()), "c" => ParserVersion(kebab_parse_code::C_PARSER_VERSION.to_string()),
"cpp" => ParserVersion(kebab_parse_code::CPP_PARSER_VERSION.to_string()), "cpp" => ParserVersion(kebab_parse_code::CPP_PARSER_VERSION.to_string()),
other => anyhow::bail!("unsupported code_lang: {other}"), other => anyhow::bail!("unsupported code_lang: {other}"),
}; };
// p10-1b Task D/G/J/L: chunker_version per-lang. // p10-1b Task D/G/J/L: chunker_version per-lang.
let mut chunker_version = match code_lang { let mut chunker_version = match code_lang {
"rust" => CodeRustAstV1Chunker.chunker_version(), "rust" => CodeRustAstV1Chunker.chunker_version(),
"python" => CodePythonAstV1Chunker.chunker_version(), "python" => CodePythonAstV1Chunker.chunker_version(),
"typescript" => CodeTsAstV1Chunker.chunker_version(), "typescript" => CodeTsAstV1Chunker.chunker_version(),
"javascript" => CodeJsAstV1Chunker.chunker_version(), "javascript" => CodeJsAstV1Chunker.chunker_version(),
"go" => CodeGoAstV1Chunker.chunker_version(), "go" => CodeGoAstV1Chunker.chunker_version(),
"java" => CodeJavaAstV1Chunker.chunker_version(), "java" => CodeJavaAstV1Chunker.chunker_version(),
"kotlin" => CodeKotlinAstV1Chunker.chunker_version(), "kotlin" => CodeKotlinAstV1Chunker.chunker_version(),
// p10-2 Tier 2: // p10-2 Tier 2:
"yaml" => K8sManifestResourceV1Chunker.chunker_version(), "yaml" => K8sManifestResourceV1Chunker.chunker_version(),
"dockerfile" => DockerfileFileV1Chunker.chunker_version(), "dockerfile" => DockerfileFileV1Chunker.chunker_version(),
"toml" | "json" | "xml" | "groovy" | "go-mod" "toml" | "json" | "xml" | "groovy" | "go-mod" => ManifestFileV1Chunker.chunker_version(),
=> ManifestFileV1Chunker.chunker_version(),
// p10-3: // p10-3:
"shell" => CodeTextParagraphV1Chunker.chunker_version(), "shell" => CodeTextParagraphV1Chunker.chunker_version(),
// p10-1D: C + C++ AST chunkers. // p10-1D: C + C++ AST chunkers.
"c" => CodeCAstV1Chunker.chunker_version(), "c" => CodeCAstV1Chunker.chunker_version(),
"cpp" => CodeCppAstV1Chunker.chunker_version(), "cpp" => CodeCppAstV1Chunker.chunker_version(),
other => anyhow::bail!("unreachable chunker_version: {other}"), other => anyhow::bail!("unreachable chunker_version: {other}"),
}; };
@@ -2265,8 +2260,12 @@ fn ingest_one_code_asset(
// Tier 2 (yaml/dockerfile/…) and shell errors are real (e.g. non-UTF-8) — propagate. // Tier 2 (yaml/dockerfile/…) and shell errors are real (e.g. non-UTF-8) — propagate.
let mut canonical = match canonical_result { let mut canonical = match canonical_result {
Ok(d) => d, Ok(d) => d,
Err(e) if code_lang == "shell" Err(e)
|| matches!(code_lang, "yaml" | "dockerfile" | "toml" | "json" | "xml" | "groovy" | "go-mod") => if code_lang == "shell"
|| matches!(
code_lang,
"yaml" | "dockerfile" | "toml" | "json" | "xml" | "groovy" | "go-mod"
) =>
{ {
return Err(e).context("synthesize_tier2_document failed for tier 2/3 lang"); return Err(e).context("synthesize_tier2_document failed for tier 2/3 lang");
} }
@@ -2290,7 +2289,10 @@ fn ingest_one_code_asset(
// Tier 2 langs already have "none-v1" parser_version normally, so exclude them // Tier 2 langs already have "none-v1" parser_version normally, so exclude them
// from the extract_fell_back guard with the !matches! exclusion. // from the extract_fell_back guard with the !matches! exclusion.
let extract_fell_back = canonical.parser_version.0 == "none-v1" let extract_fell_back = canonical.parser_version.0 == "none-v1"
&& !matches!(code_lang, "yaml" | "dockerfile" | "toml" | "json" | "xml" | "groovy" | "go-mod" | "shell"); && !matches!(
code_lang,
"yaml" | "dockerfile" | "toml" | "json" | "xml" | "groovy" | "go-mod" | "shell"
);
let chunks_result: anyhow::Result<Vec<Chunk>> = if extract_fell_back { let chunks_result: anyhow::Result<Vec<Chunk>> = if extract_fell_back {
// Tier 1 lang whose extractor errored — go straight to Tier 3 chunker. // Tier 1 lang whose extractor errored — go straight to Tier 3 chunker.
@@ -2349,7 +2351,7 @@ fn ingest_one_code_asset(
// "shell" direct path is already Tier 3 — don't retry-double-up. // "shell" direct path is already Tier 3 — don't retry-double-up.
let chunks: Vec<Chunk> = match chunks_result { let chunks: Vec<Chunk> = match chunks_result {
Ok(v) if !v.is_empty() => v, Ok(v) if !v.is_empty() => v,
other if code_lang == "shell" => other?, // shell propagates directly other if code_lang == "shell" => other?, // shell propagates directly
Ok(_empty) => { Ok(_empty) => {
tracing::warn!( tracing::warn!(
workspace_path = %asset.workspace_path.0, workspace_path = %asset.workspace_path.0,
@@ -2373,7 +2375,9 @@ fn ingest_one_code_asset(
canonical.parser_version = ParserVersion("none-v1".to_string()); canonical.parser_version = ParserVersion("none-v1".to_string());
CodeTextParagraphV1Chunker CodeTextParagraphV1Chunker
.chunk(&canonical, chunk_policy) .chunk(&canonical, chunk_policy)
.context("kb-chunk::CodeTextParagraphV1Chunker::chunk (tier 3 fallback after error)")? .context(
"kb-chunk::CodeTextParagraphV1Chunker::chunk (tier 3 fallback after error)",
)?
} }
}; };
@@ -2501,13 +2505,7 @@ fn synthesize_tier2_document(
symbol: Some("<file>".to_string()), symbol: Some("<file>".to_string()),
lang: Some(code_lang.to_string()), lang: Some(code_lang.to_string()),
}; };
let block_id: BlockId = id_for_block( let block_id: BlockId = id_for_block(&doc_id, "code", &[], 0, &span);
&doc_id,
"code",
&[],
0,
&span,
);
let block = kebab_core::Block::Code(CodeBlock { let block = kebab_core::Block::Code(CodeBlock {
common: CommonBlock { common: CommonBlock {
block_id, block_id,
@@ -2553,7 +2551,9 @@ fn synthesize_tier2_document(
}; };
let title = { let title = {
let fname = asset.workspace_path.0 let fname = asset
.workspace_path
.0
.rsplit('/') .rsplit('/')
.next() .next()
.unwrap_or(&asset.workspace_path.0); .unwrap_or(&asset.workspace_path.0);
@@ -2799,7 +2799,9 @@ pub fn ask_with_session_with_config(
/// `data_dir_writable` check probes the resolved `storage.data_dir` /// `data_dir_writable` check probes the resolved `storage.data_dir`
/// from that config (so `--config` users see their custom paths /// from that config (so `--config` users see their custom paths
/// reflected in the report rather than the XDG defaults). /// reflected in the report rather than the XDG defaults).
pub fn doctor_with_config_path(config_path: Option<&std::path::Path>) -> anyhow::Result<DoctorReport> { pub fn doctor_with_config_path(
config_path: Option<&std::path::Path>,
) -> anyhow::Result<DoctorReport> {
tracing::debug!("doctor() invoked"); tracing::debug!("doctor() invoked");
let mut checks = Vec::new(); let mut checks = Vec::new();
@@ -2817,11 +2819,7 @@ pub fn doctor_with_config_path(config_path: Option<&std::path::Path>) -> anyhow:
} else if config_path.is_some() { } else if config_path.is_some() {
// Explicit `--config <path>` that doesn't exist is a hard error // Explicit `--config <path>` that doesn't exist is a hard error
// — defaults would silently mask the user's intent. // — defaults would silently mask the user's intent.
( (false, format!("{} (not found)", cfg_path.display()), None)
false,
format!("{} (not found)", cfg_path.display()),
None,
)
} else { } else {
// No `--config` and no XDG file: defaults are always loadable. // No `--config` and no XDG file: defaults are always loadable.
(true, format!("{} (defaults)", cfg_path.display()), None) (true, format!("{} (defaults)", cfg_path.display()), None)
@@ -2907,16 +2905,18 @@ pub fn ingest_file_with_config(
path: &std::path::Path, path: &std::path::Path,
) -> anyhow::Result<IngestReport> { ) -> anyhow::Result<IngestReport> {
if !path.exists() { if !path.exists() {
anyhow::bail!("ingest-file: source path does not exist: {}", path.display()); anyhow::bail!(
"ingest-file: source path does not exist: {}",
path.display()
);
} }
if !path.is_file() { if !path.is_file() {
anyhow::bail!("ingest-file: not a regular file: {}", path.display()); anyhow::bail!("ingest-file: not a regular file: {}", path.display());
} }
let ext_raw = path let ext_raw = path.extension().and_then(|e| e.to_str()).ok_or_else(|| {
.extension() anyhow::anyhow!("ingest-file: source has no extension: {}", path.display())
.and_then(|e| e.to_str()) })?;
.ok_or_else(|| anyhow::anyhow!("ingest-file: source has no extension: {}", path.display()))?;
let ext = ext_raw.to_lowercase(); let ext = ext_raw.to_lowercase();
const SUPPORTED_EXTS: &[&str] = &["md", "pdf", "png", "jpg", "jpeg"]; const SUPPORTED_EXTS: &[&str] = &["md", "pdf", "png", "jpg", "jpeg"];
@@ -2993,11 +2993,7 @@ pub fn ingest_stdin_with_config(
let external_dir = crate::external::ensure_external_dir(&workspace_root)?; let external_dir = crate::external::ensure_external_dir(&workspace_root)?;
crate::external::ensure_kebabignore_entry(&workspace_root)?; crate::external::ensure_kebabignore_entry(&workspace_root)?;
let dest = crate::external::copy_to_external( let dest = crate::external::copy_to_external(&external_dir, wrapped.as_bytes(), "md")?;
&external_dir,
wrapped.as_bytes(),
"md",
)?;
ingest_file_with_config(config, &dest) ingest_file_with_config(config, &dest)
} }
@@ -3005,7 +3001,10 @@ pub fn ingest_stdin_with_config(
/// Returns true if `source_path` matches any `.kebabignore` pattern /// Returns true if `source_path` matches any `.kebabignore` pattern
/// rooted at `workspace_root`. Used by `ingest_file_with_config` to /// rooted at `workspace_root`. Used by `ingest_file_with_config` to
/// emit a stderr warn before bypassing the ignore. /// emit a stderr warn before bypassing the ignore.
fn check_kebabignore_match(workspace_root: &std::path::Path, source_path: &std::path::Path) -> bool { fn check_kebabignore_match(
workspace_root: &std::path::Path,
source_path: &std::path::Path,
) -> bool {
let kebabignore = workspace_root.join(".kebabignore"); let kebabignore = workspace_root.join(".kebabignore");
if !kebabignore.exists() { if !kebabignore.exists() {
return false; return false;
@@ -3026,5 +3025,7 @@ fn check_kebabignore_match(workspace_root: &std::path::Path, source_path: &std::
Ok(m) => m, Ok(m) => m,
Err(_) => return false, Err(_) => return false,
}; };
matcher.matched(source_path, source_path.is_dir()).is_ignore() matcher
.matched(source_path, source_path.is_dir())
.is_ignore()
} }

View File

@@ -26,7 +26,9 @@ pub fn init(level: LogLevel) -> Result<WorkerGuard> {
let (nb, guard) = tracing_appender::non_blocking(file_appender); let (nb, guard) = tracing_appender::non_blocking(file_appender);
let env_filter = match level { let env_filter = match level {
LogLevel::Default => EnvFilter::try_from_default_env().unwrap_or_else(|_| EnvFilter::new("warn")), LogLevel::Default => {
EnvFilter::try_from_default_env().unwrap_or_else(|_| EnvFilter::new("warn"))
}
LogLevel::Verbose => EnvFilter::new("info"), LogLevel::Verbose => EnvFilter::new("info"),
LogLevel::Debug => EnvFilter::new("debug"), LogLevel::Debug => EnvFilter::new("debug"),
}; };

View File

@@ -13,8 +13,8 @@ use std::time::Instant;
use anyhow::{Context, Result}; use anyhow::{Context, Result};
use kebab_core::{ use kebab_core::{
Block, CanonicalDocument, CommonBlock, Inline, Lang, ProvenanceEvent, Block, CanonicalDocument, CommonBlock, Inline, Lang, ProvenanceEvent, ProvenanceKind,
ProvenanceKind, SourceSpan, TextBlock, id_for_block, SourceSpan, TextBlock, id_for_block,
}; };
use kebab_parse_image::OcrEngine; use kebab_parse_image::OcrEngine;
use kebab_parse_pdf::{compute_valid_char_ratio, extract_dctdecode_page_image}; use kebab_parse_pdf::{compute_valid_char_ratio, extract_dctdecode_page_image};
@@ -88,7 +88,10 @@ where
F: FnMut(PdfOcrProgress), F: FnMut(PdfOcrProgress),
{ {
if !opts.enabled { if !opts.enabled {
return Ok(PdfOcrSummary { pages_ocrd: 0, ms_total: 0 }); return Ok(PdfOcrSummary {
pages_ocrd: 0,
ms_total: 0,
});
} }
let pdf_doc = LopdfDocument::load_mem(pdf_bytes) let pdf_doc = LopdfDocument::load_mem(pdf_bytes)
.context("kb-app::pdf_ocr_apply: re-parse PDF for image extract")?; .context("kb-app::pdf_ocr_apply: re-parse PDF for image extract")?;
@@ -117,8 +120,7 @@ where
}; };
let chars = text.chars().count() as u32; let chars = text.chars().count() as u32;
let valid_ratio = compute_valid_char_ratio(&text); let valid_ratio = compute_valid_char_ratio(&text);
let needs_ocr = let needs_ocr = chars < opts.min_char_count || valid_ratio < opts.valid_ratio_threshold;
chars < opts.min_char_count || valid_ratio < opts.valid_ratio_threshold;
// 결정 matrix: // 결정 matrix:
// always_on=true → 모든 page OCR (dual-block). // always_on=true → 모든 page OCR (dual-block).
@@ -131,7 +133,9 @@ where
emit_progress(PdfOcrProgress::Started { page: page_num }); emit_progress(PdfOcrProgress::Started { page: page_num });
let page_image_bytes = if let Some(b) = extract_dctdecode_page_image(&pdf_doc, page_num)? { b } else { let page_image_bytes = if let Some(b) = extract_dctdecode_page_image(&pdf_doc, page_num)? {
b
} else {
let note = format!( let note = format!(
"page={page_num} skipped: no DCTDecode image XObject (vector PDF page or unsupported /Filter — v1 supports DCTDecode passthrough only; see release notes for normalization guidance)" "page={page_num} skipped: no DCTDecode image XObject (vector PDF page or unsupported /Filter — v1 supports DCTDecode passthrough only; see release notes for normalization guidance)"
); );
@@ -266,7 +270,10 @@ where
canonical.blocks.extend(ocr_blocks); canonical.blocks.extend(ocr_blocks);
canonical.provenance.events.extend(new_events); canonical.provenance.events.extend(new_events);
Ok(PdfOcrSummary { pages_ocrd, ms_total }) Ok(PdfOcrSummary {
pages_ocrd,
ms_total,
})
} }
fn find_paragraph_block_idx(blocks: &[Block], page_num: u32) -> usize { fn find_paragraph_block_idx(blocks: &[Block], page_num: u32) -> usize {

View File

@@ -85,8 +85,7 @@ pub fn enumerate_paths(scope: ResetScope, cfg: &Config) -> Vec<PathBuf> {
ResetScope::All => vec![cfg_dir, data_dir, cache_dir, state_dir], ResetScope::All => vec![cfg_dir, data_dir, cache_dir, state_dir],
ResetScope::DataOnly => vec![data_dir, cache_dir, state_dir], ResetScope::DataOnly => vec![data_dir, cache_dir, state_dir],
ResetScope::VectorOnly => { ResetScope::VectorOnly => {
let vector_dir = let vector_dir = expand_path(&cfg.storage.vector_dir, &data_dir.to_string_lossy());
expand_path(&cfg.storage.vector_dir, &data_dir.to_string_lossy());
vec![vector_dir] vec![vector_dir]
} }
ResetScope::ConfigOnly => vec![cfg_dir], ResetScope::ConfigOnly => vec![cfg_dir],
@@ -137,8 +136,8 @@ pub fn estimate_size_bytes(paths: &[PathBuf]) -> u64 {
/// the double scan is acceptable for a rare destructive operation. /// the double scan is acceptable for a rare destructive operation.
pub fn enumerate_orphans(cfg: &Config) -> Result<Vec<WorkspacePath>> { pub fn enumerate_orphans(cfg: &Config) -> Result<Vec<WorkspacePath>> {
use kebab_core::DocumentStore as _; use kebab_core::DocumentStore as _;
use kebab_source_fs::FsSourceConnector;
use kebab_core::SourceScope; use kebab_core::SourceScope;
use kebab_source_fs::FsSourceConnector;
let store = kebab_store_sqlite::SqliteStore::open(cfg) let store = kebab_store_sqlite::SqliteStore::open(cfg)
.context("enumerate_orphans: open SqliteStore")?; .context("enumerate_orphans: open SqliteStore")?;
@@ -160,16 +159,13 @@ pub fn enumerate_orphans(cfg: &Config) -> Result<Vec<WorkspacePath>> {
..Default::default() ..Default::default()
}; };
let connector = FsSourceConnector::new(cfg) let connector =
.context("enumerate_orphans: build FsSourceConnector")?; FsSourceConnector::new(cfg).context("enumerate_orphans: build FsSourceConnector")?;
let (assets, _skips) = connector let (assets, _skips) = connector
.scan_with_skips(&scope) .scan_with_skips(&scope)
.context("enumerate_orphans: scan workspace")?; .context("enumerate_orphans: scan workspace")?;
let scanned: HashSet<WorkspacePath> = assets let scanned: HashSet<WorkspacePath> = assets.into_iter().map(|a| a.workspace_path).collect();
.into_iter()
.map(|a| a.workspace_path)
.collect();
let mut orphans: Vec<WorkspacePath> = stored let mut orphans: Vec<WorkspacePath> = stored
.into_iter() .into_iter()
@@ -206,8 +202,7 @@ pub fn execute(scope: ResetScope, cfg: &Config) -> Result<ResetReport> {
if !p.exists() { if !p.exists() {
continue; continue;
} }
std::fs::remove_dir_all(p) std::fs::remove_dir_all(p).with_context(|| format!("remove {}", p.display()))?;
.with_context(|| format!("remove {}", p.display()))?;
removed.push(p.clone()); removed.push(p.clone());
} }
@@ -229,8 +224,7 @@ pub fn execute(scope: ResetScope, cfg: &Config) -> Result<ResetReport> {
/// Execute the `OrphansOnly` variant: reconcile stored docs against the /// Execute the `OrphansOnly` variant: reconcile stored docs against the
/// current walker scope without touching any filesystem directory. /// current walker scope without touching any filesystem directory.
fn execute_orphans_only(cfg: &Config) -> Result<ResetReport> { fn execute_orphans_only(cfg: &Config) -> Result<ResetReport> {
let orphans = enumerate_orphans(cfg) let orphans = enumerate_orphans(cfg).context("execute_orphans_only: enumerate orphans")?;
.context("execute_orphans_only: enumerate orphans")?;
if orphans.is_empty() { if orphans.is_empty() {
return Ok(ResetReport { return Ok(ResetReport {

View File

@@ -168,12 +168,8 @@ fn open_store_for_stats(cfg: &Config) -> anyhow::Result<kebab_store_sqlite::Sqli
kebab_store_sqlite::SqliteStore::open_existing(&db_path) kebab_store_sqlite::SqliteStore::open_existing(&db_path)
} }
fn collect_stats( fn collect_stats(cfg: &Config, store: &kebab_store_sqlite::SqliteStore) -> anyhow::Result<Stats> {
cfg: &Config, let counts = store.count_summary_with_threshold(u64::from(cfg.search.stale_threshold_days))?;
store: &kebab_store_sqlite::SqliteStore,
) -> anyhow::Result<Stats> {
let counts = store
.count_summary_with_threshold(u64::from(cfg.search.stale_threshold_days))?;
let data_dir = kebab_config::expand_path(&cfg.storage.data_dir, ""); let data_dir = kebab_config::expand_path(&cfg.storage.data_dir, "");
let index_bytes = kebab_store_sqlite::stats_ext::index_bytes(&data_dir) let index_bytes = kebab_store_sqlite::stats_ext::index_bytes(&data_dir)
.map_err(|e| anyhow::anyhow!("index_bytes: {e}"))?; .map_err(|e| anyhow::anyhow!("index_bytes: {e}"))?;
@@ -298,6 +294,9 @@ mod tests_capabilities {
// Bug #9: kebab ingest-file <path> + kebab ingest-stdin --title <T> 양쪽 모두 // Bug #9: kebab ingest-file <path> + kebab ingest-stdin --title <T> 양쪽 모두
// ingest_report.v1 정상 emit → capabilities.single_file_ingest 가 true 여야 함. // ingest_report.v1 정상 emit → capabilities.single_file_ingest 가 true 여야 함.
let caps = capabilities_snapshot(); let caps = capabilities_snapshot();
assert!(caps.single_file_ingest, "single_file_ingest must be true (Bug #9)"); assert!(
caps.single_file_ingest,
"single_file_ingest must be true (Bug #9)"
);
} }
} }

View File

@@ -10,11 +10,7 @@ use kebab_core::SearchHit;
/// ///
/// p9-fb-32: mirrored in `kebab_rag::pipeline::compute_stale` (dep-boundary /// p9-fb-32: mirrored in `kebab_rag::pipeline::compute_stale` (dep-boundary
/// rule prevents `kebab-rag → kebab-app`). Update both together. /// rule prevents `kebab-rag → kebab-app`). Update both together.
pub fn compute_stale( pub fn compute_stale(indexed_at: OffsetDateTime, now: OffsetDateTime, threshold_days: u32) -> bool {
indexed_at: OffsetDateTime,
now: OffsetDateTime,
threshold_days: u32,
) -> bool {
if threshold_days == 0 { if threshold_days == 0 {
return false; return false;
} }
@@ -23,11 +19,7 @@ pub fn compute_stale(
} }
/// Sets `stale` on each hit in place using `compute_stale`. /// Sets `stale` on each hit in place using `compute_stale`.
pub fn mark_stale_in_place( pub fn mark_stale_in_place(hits: &mut [SearchHit], now: OffsetDateTime, threshold_days: u32) {
hits: &mut [SearchHit],
now: OffsetDateTime,
threshold_days: u32,
) {
for h in hits { for h in hits {
h.stale = compute_stale(h.indexed_at, now, threshold_days); h.stale = compute_stale(h.indexed_at, now, threshold_days);
} }

View File

@@ -29,9 +29,8 @@ fn rust_file_ingests_and_searches_as_code_citation() {
) )
.unwrap(); .unwrap();
let report = let report = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false)
kebab_app::ingest_with_config(env.config.clone(), env.scope(), false) .expect("ingest must succeed");
.expect("ingest must succeed");
assert_eq!(report.errors, 0, "no errors expected: {report:?}"); assert_eq!(report.errors, 0, "no errors expected: {report:?}");
let items = report.items.as_ref().expect("items present"); let items = report.items.as_ref().expect("items present");
@@ -127,9 +126,8 @@ fn rust_code_search_hit_has_repo() {
) )
.unwrap(); .unwrap();
let report = let report = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false)
kebab_app::ingest_with_config(env.config.clone(), env.scope(), false) .expect("ingest must succeed");
.expect("ingest must succeed");
assert_eq!(report.errors, 0, "no ingest errors: {report:?}"); assert_eq!(report.errors, 0, "no ingest errors: {report:?}");
let hits = kebab_app::search_with_config(env.config.clone(), lexical_query("mul")) let hits = kebab_app::search_with_config(env.config.clone(), lexical_query("mul"))
@@ -147,8 +145,7 @@ fn rust_code_search_hit_has_repo() {
.and_then(|n| n.to_str()) .and_then(|n| n.to_str())
.map(str::to_owned); .map(str::to_owned);
assert_eq!( assert_eq!(
h.repo, h.repo, expected_repo,
expected_repo,
"SearchHit.repo must match the workspace dir name (detect_repo result)" "SearchHit.repo must match the workspace dir name (detect_repo result)"
); );
// Also sanity-check code_lang is still filled. // Also sanity-check code_lang is still filled.
@@ -177,9 +174,8 @@ fn python_file_ingests_and_searches_as_code_citation() {
) )
.unwrap(); .unwrap();
let report = let report = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false)
kebab_app::ingest_with_config(env.config.clone(), env.scope(), false) .expect("ingest must succeed");
.expect("ingest must succeed");
assert!(report.new >= 1, "python file ingested: {report:?}"); assert!(report.new >= 1, "python file ingested: {report:?}");
@@ -254,9 +250,8 @@ fn typescript_file_ingests_and_searches_as_code_citation() {
) )
.unwrap(); .unwrap();
let report = let report = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false)
kebab_app::ingest_with_config(env.config.clone(), env.scope(), false) .expect("ingest must succeed");
.expect("ingest must succeed");
assert!(report.new >= 1, "ts file ingested: {report:?}"); assert!(report.new >= 1, "ts file ingested: {report:?}");
@@ -331,9 +326,8 @@ fn javascript_file_ingests_and_searches_as_code_citation() {
) )
.unwrap(); .unwrap();
let report = let report = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false)
kebab_app::ingest_with_config(env.config.clone(), env.scope(), false) .expect("ingest must succeed");
.expect("ingest must succeed");
assert!(report.new >= 1, "js file ingested: {report:?}"); assert!(report.new >= 1, "js file ingested: {report:?}");
@@ -515,7 +509,11 @@ fn java_file_ingests_and_searches_as_code_citation() {
line_start, line_start,
.. ..
} => { } => {
assert_eq!(lang.as_deref(), Some("java"), "citation.lang must be 'java'"); assert_eq!(
lang.as_deref(),
Some("java"),
"citation.lang must be 'java'"
);
assert_eq!( assert_eq!(
symbol.as_deref(), symbol.as_deref(),
Some("com.foo.Foo.bar"), Some("com.foo.Foo.bar"),
@@ -586,7 +584,11 @@ fn kotlin_file_ingests_and_searches_as_code_citation() {
line_start, line_start,
.. ..
} => { } => {
assert_eq!(lang.as_deref(), Some("kotlin"), "citation.lang must be 'kotlin'"); assert_eq!(
lang.as_deref(),
Some("kotlin"),
"citation.lang must be 'kotlin'"
);
assert_eq!( assert_eq!(
symbol.as_deref(), symbol.as_deref(),
Some("com.foo.Foo.bar"), Some("com.foo.Foo.bar"),
@@ -651,8 +653,8 @@ fn tier2_k8s_yaml_ingest_searchable() {
..Default::default() ..Default::default()
}, },
}; };
let hits = kebab_app::search_with_config(env.config.clone(), query) let hits =
.expect("search must succeed"); kebab_app::search_with_config(env.config.clone(), query).expect("search must succeed");
let h = hits let h = hits
.iter() .iter()
@@ -666,7 +668,11 @@ fn tier2_k8s_yaml_ingest_searchable() {
line_start, line_start,
.. ..
} => { } => {
assert_eq!(lang.as_deref(), Some("yaml"), "citation.lang must be 'yaml'"); assert_eq!(
lang.as_deref(),
Some("yaml"),
"citation.lang must be 'yaml'"
);
assert_eq!( assert_eq!(
symbol.as_deref(), symbol.as_deref(),
Some("Deployment/prod/api"), Some("Deployment/prod/api"),
@@ -730,8 +736,8 @@ fn tier2_dockerfile_ingest_searchable() {
..Default::default() ..Default::default()
}, },
}; };
let hits = kebab_app::search_with_config(env.config.clone(), query) let hits =
.expect("search must succeed"); kebab_app::search_with_config(env.config.clone(), query).expect("search must succeed");
let h = hits let h = hits
.iter() .iter()
@@ -813,8 +819,8 @@ fn tier2_cargo_toml_ingest_searchable() {
..Default::default() ..Default::default()
}, },
}; };
let hits = kebab_app::search_with_config(env.config.clone(), query) let hits =
.expect("search must succeed"); kebab_app::search_with_config(env.config.clone(), query).expect("search must succeed");
let h = hits let h = hits
.iter() .iter()
@@ -896,8 +902,8 @@ fn tier3_shell_ingest_searchable() {
..Default::default() ..Default::default()
}, },
}; };
let hits = kebab_app::search_with_config(env.config.clone(), query) let hits =
.expect("search must succeed"); kebab_app::search_with_config(env.config.clone(), query).expect("search must succeed");
let h = hits let h = hits
.iter() .iter()
@@ -987,8 +993,8 @@ fn tier3_yaml_fallback_picks_up_non_k8s_yaml() {
..Default::default() ..Default::default()
}, },
}; };
let hits = kebab_app::search_with_config(env.config.clone(), query) let hits =
.expect("search must succeed"); kebab_app::search_with_config(env.config.clone(), query).expect("search must succeed");
let h = hits let h = hits
.iter() .iter()
@@ -1031,14 +1037,9 @@ fn tier3_yaml_fallback_picks_up_non_k8s_yaml() {
fn rust_file_re_ingest_is_unchanged() { fn rust_file_re_ingest_is_unchanged() {
let env = TestEnv::lexical_only(); let env = TestEnv::lexical_only();
std::fs::write( std::fs::write(env.workspace_root.join("stable.rs"), "pub fn noop() {}\n").unwrap();
env.workspace_root.join("stable.rs"),
"pub fn noop() {}\n",
)
.unwrap();
let r1 = let r1 = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
kebab_app::ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
let item1 = r1 let item1 = r1
.items .items
.as_ref() .as_ref()
@@ -1049,8 +1050,7 @@ fn rust_file_re_ingest_is_unchanged() {
.unwrap(); .unwrap();
assert_eq!(item1.kind, IngestItemKind::New); assert_eq!(item1.kind, IngestItemKind::New);
let r2 = let r2 = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
kebab_app::ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
let item2 = r2 let item2 = r2
.items .items
.unwrap() .unwrap()
@@ -1081,9 +1081,8 @@ fn tier3_yaml_fallback_reingest_is_unchanged() {
) )
.unwrap(); .unwrap();
let report1 = let report1 = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false)
kebab_app::ingest_with_config(env.config.clone(), env.scope(), false) .expect("first ingest");
.expect("first ingest");
let item1 = report1 let item1 = report1
.items .items
.as_ref() .as_ref()
@@ -1093,7 +1092,8 @@ fn tier3_yaml_fallback_reingest_is_unchanged() {
.expect("docker-compose.yml in first report"); .expect("docker-compose.yml in first report");
assert!( assert!(
matches!(item1.kind, IngestItemKind::New), matches!(item1.kind, IngestItemKind::New),
"first ingest must be New, got {:?}", item1.kind "first ingest must be New, got {:?}",
item1.kind
); );
assert_eq!( assert_eq!(
item1.chunker_version.as_ref().map(|c| c.0.as_str()), item1.chunker_version.as_ref().map(|c| c.0.as_str()),
@@ -1101,9 +1101,8 @@ fn tier3_yaml_fallback_reingest_is_unchanged() {
"first ingest must use Tier 3 fallback chunker" "first ingest must use Tier 3 fallback chunker"
); );
let report2 = let report2 = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false)
kebab_app::ingest_with_config(env.config.clone(), env.scope(), false) .expect("second ingest");
.expect("second ingest");
let item2 = report2 let item2 = report2
.items .items
.as_ref() .as_ref()
@@ -1113,7 +1112,8 @@ fn tier3_yaml_fallback_reingest_is_unchanged() {
.expect("docker-compose.yml in second report"); .expect("docker-compose.yml in second report");
assert!( assert!(
matches!(item2.kind, IngestItemKind::Unchanged), matches!(item2.kind, IngestItemKind::Unchanged),
"second ingest must be Unchanged, got {:?}", item2.kind "second ingest must be Unchanged, got {:?}",
item2.kind
); );
} }
@@ -1163,8 +1163,8 @@ fn tier1_c_ingest_searchable() {
..Default::default() ..Default::default()
}, },
}; };
let hits = kebab_app::search_with_config(env.config.clone(), query) let hits =
.expect("search must succeed"); kebab_app::search_with_config(env.config.clone(), query).expect("search must succeed");
let h = hits let h = hits
.iter() .iter()
@@ -1247,8 +1247,8 @@ fn tier1_cpp_ingest_searchable() {
..Default::default() ..Default::default()
}, },
}; };
let hits = kebab_app::search_with_config(env.config.clone(), query) let hits =
.expect("search must succeed"); kebab_app::search_with_config(env.config.clone(), query).expect("search must succeed");
let h = hits let h = hits
.iter() .iter()
@@ -1266,7 +1266,9 @@ fn tier1_cpp_ingest_searchable() {
// Symbol could be "kebab::chunk::Foo" (class) or "kebab::chunk::Foo::bar" // Symbol could be "kebab::chunk::Foo" (class) or "kebab::chunk::Foo::bar"
// (method) depending on which chunk ranks first. // (method) depending on which chunk ranks first.
assert!( assert!(
symbol.as_deref().is_some_and(|s| s.starts_with("kebab::chunk::Foo")), symbol
.as_deref()
.is_some_and(|s| s.starts_with("kebab::chunk::Foo")),
"C++ symbol must start with namespace::Class prefix, got {symbol:?}" "C++ symbol must start with namespace::Class prefix, got {symbol:?}"
); );
assert!(*line_start >= 1, "line_start must be >=1"); assert!(*line_start >= 1, "line_start must be >=1");
@@ -1335,8 +1337,8 @@ fn tier2_k8s_multi_resource_yaml_ingests_without_collision() {
..Default::default() ..Default::default()
}, },
}; };
let hits = kebab_app::search_with_config(env.config.clone(), query) let hits =
.expect("search must succeed"); kebab_app::search_with_config(env.config.clone(), query).expect("search must succeed");
assert!( assert!(
hits.len() >= 2, hits.len() >= 2,
"expected ≥2 hits (Deployment + Service), got {}", "expected ≥2 hits (Deployment + Service), got {}",
@@ -1359,9 +1361,8 @@ fn tier3_shell_reingest_is_unchanged() {
) )
.unwrap(); .unwrap();
let report1 = let report1 = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false)
kebab_app::ingest_with_config(env.config.clone(), env.scope(), false) .expect("first ingest");
.expect("first ingest");
let item1 = report1 let item1 = report1
.items .items
.as_ref() .as_ref()
@@ -1371,12 +1372,12 @@ fn tier3_shell_reingest_is_unchanged() {
.expect("deploy.sh in first report"); .expect("deploy.sh in first report");
assert!( assert!(
matches!(item1.kind, IngestItemKind::New), matches!(item1.kind, IngestItemKind::New),
"first ingest must be New, got {:?}", item1.kind "first ingest must be New, got {:?}",
item1.kind
); );
let report2 = let report2 = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false)
kebab_app::ingest_with_config(env.config.clone(), env.scope(), false) .expect("second ingest");
.expect("second ingest");
let item2 = report2 let item2 = report2
.items .items
.as_ref() .as_ref()
@@ -1386,6 +1387,7 @@ fn tier3_shell_reingest_is_unchanged() {
.expect("deploy.sh in second report"); .expect("deploy.sh in second report");
assert!( assert!(
matches!(item2.kind, IngestItemKind::Unchanged), matches!(item2.kind, IngestItemKind::Unchanged),
"shell reingest must be Unchanged, got {:?}", item2.kind "shell reingest must be Unchanged, got {:?}",
item2.kind
); );
} }

View File

@@ -93,8 +93,7 @@ impl TestEnv {
/// directly. Caller can invoke this multiple times to simulate /// directly. Caller can invoke this multiple times to simulate
/// re-opening the binary after a corpus revision bump. /// re-opening the binary after a corpus revision bump.
pub fn app(&self) -> kebab_app::App { pub fn app(&self) -> kebab_app::App {
kebab_app::App::open_with_config(self.config.clone()) kebab_app::App::open_with_config(self.config.clone()).expect("App::open_with_config")
.expect("App::open_with_config")
} }
} }

View File

@@ -12,7 +12,11 @@ fn open(env: &common::TestEnv) -> App {
#[test] #[test]
fn fetch_chunk_returns_target_only_when_no_context() { fn fetch_chunk_returns_target_only_when_no_context() {
let env = common::TestEnv::new(); let env = common::TestEnv::new();
common::ingest_md(&env, "a.md", "# Title\n\nFirst paragraph.\n\n## Section\n\nSecond.\n"); common::ingest_md(
&env,
"a.md",
"# Title\n\nFirst paragraph.\n\n## Section\n\nSecond.\n",
);
let app = open(&env); let app = open(&env);
// Find a chunk via search to obtain its id. // Find a chunk via search to obtain its id.
@@ -42,7 +46,8 @@ fn fetch_chunk_with_context_returns_neighbors() {
// match. The earlier fixture used 2-char tokens like `A1`/`A3` for // match. The earlier fixture used 2-char tokens like `A1`/`A3` for
// section bodies — those zero-hit under trigram. Use 5-char unique // section bodies — those zero-hit under trigram. Use 5-char unique
// words per section so the query can pin one chunk deterministically. // words per section so the query can pin one chunk deterministically.
let body = "# H1\n\napples\n\n# H2\n\nbanana\n\n# H3\n\ncherry\n\n# H4\n\ndurian\n\n# H5\n\nelder\n"; let body =
"# H1\n\napples\n\n# H2\n\nbanana\n\n# H3\n\ncherry\n\n# H4\n\ndurian\n\n# H5\n\nelder\n";
common::ingest_md(&env, "multi.md", body); common::ingest_md(&env, "multi.md", body);
let app = env.app(); let app = env.app();
@@ -110,7 +115,10 @@ fn fetch_doc_returns_serialized_markdown() {
.unwrap(); .unwrap();
assert_eq!(result.kind, FetchKind::Doc); assert_eq!(result.kind, FetchKind::Doc);
let text = result.text.expect("doc text"); let text = result.text.expect("doc text");
assert!(text.contains("Heading One"), "doc text contains heading: {text:?}"); assert!(
text.contains("Heading One"),
"doc text contains heading: {text:?}"
);
assert!(text.contains("First paragraph"), "doc text contains body"); assert!(text.contains("First paragraph"), "doc text contains body");
assert!(!result.truncated); assert!(!result.truncated);
} }
@@ -155,7 +163,11 @@ fn fetch_doc_with_max_tokens_truncates() {
.unwrap(); .unwrap();
assert!(result.truncated); assert!(result.truncated);
let text = result.text.expect("doc text"); let text = result.text.expect("doc text");
assert!(text.chars().count() <= 100, "trimmed text len {}", text.chars().count()); assert!(
text.chars().count() <= 100,
"trimmed text len {}",
text.chars().count()
);
} }
#[test] #[test]
@@ -292,8 +304,7 @@ fn fetch_span_line_start_beyond_total_returns_empty_text() {
fn fetch_chunk_context_at_first_chunk_clamps_lower_bound() { fn fetch_chunk_context_at_first_chunk_clamps_lower_bound() {
let env = common::TestEnv::new(); let env = common::TestEnv::new();
// Multi-chunk markdown so context ±N has neighbors. // Multi-chunk markdown so context ±N has neighbors.
let body = let body = "# H1\n\nFirst chunk text body.\n\n# H2\n\nSecond chunk.\n\n# H3\n\nThird chunk.\n";
"# H1\n\nFirst chunk text body.\n\n# H2\n\nSecond chunk.\n\n# H3\n\nThird chunk.\n";
common::ingest_md(&env, "boundary.md", body); common::ingest_md(&env, "boundary.md", body);
let app = env.app(); let app = env.app();
let q = kebab_core::SearchQuery { let q = kebab_core::SearchQuery {

View File

@@ -16,8 +16,8 @@
mod common; mod common;
use common::TestEnv; use common::TestEnv;
use kebab_app::ingest_with_config_opts;
use kebab_app::IngestOpts; use kebab_app::IngestOpts;
use kebab_app::ingest_with_config_opts;
use kebab_core::{DocFilter, DocumentStore, SearchMode, SearchQuery, SourceScope}; use kebab_core::{DocFilter, DocumentStore, SearchMode, SearchQuery, SourceScope};
/// Helper: open the store via `TestEnv` and run `list_documents`. /// Helper: open the store via `TestEnv` and run `list_documents`.
@@ -125,17 +125,10 @@ fn include_scope_narrowing_does_not_purge() {
include: vec!["**/*.rs".to_string()], include: vec!["**/*.rs".to_string()],
exclude: env.config.workspace.exclude.clone(), exclude: env.config.workspace.exclude.clone(),
}; };
let first = ingest_with_config_opts( let first =
env.config.clone(), ingest_with_config_opts(env.config.clone(), wide_scope, false, IngestOpts::default())
wide_scope, .expect("first ingest (wide) must succeed");
false, assert!(first.new >= 2, "expected at least 2 new docs: {first:?}");
IngestOpts::default(),
)
.expect("first ingest (wide) must succeed");
assert!(
first.new >= 2,
"expected at least 2 new docs: {first:?}"
);
assert_eq!( assert_eq!(
first.purged_deleted_files, 0, first.purged_deleted_files, 0,
"no purges on first ingest: {first:?}" "no purges on first ingest: {first:?}"

View File

@@ -24,8 +24,7 @@ use wiremock::{Mock, MockServer, ResponseTemplate};
/// inspectable in stored DB rows. /// inspectable in stored DB rows.
fn write_red_png(root: &Path, name: &str) -> std::path::PathBuf { fn write_red_png(root: &Path, name: &str) -> std::path::PathBuf {
use image::{ImageBuffer, Rgb}; use image::{ImageBuffer, Rgb};
let img: ImageBuffer<Rgb<u8>, _> = let img: ImageBuffer<Rgb<u8>, _> = ImageBuffer::from_fn(100, 50, |_, _| Rgb([255, 0, 0]));
ImageBuffer::from_fn(100, 50, |_, _| Rgb([255, 0, 0]));
let path = root.join(name); let path = root.join(name);
img.save(&path).expect("write PNG fixture"); img.save(&path).expect("write PNG fixture");
path path
@@ -80,7 +79,12 @@ async fn ingest_image_with_ocr_produces_chunk_containing_ocr_text() {
// Counters: scanned should include the PNG; new ≥ 1 (markdown // Counters: scanned should include the PNG; new ≥ 1 (markdown
// fixtures from the workspace tree may also count). // fixtures from the workspace tree may also count).
assert!(report.scanned >= 1, "scanned={}, items={:?}", report.scanned, report.items); assert!(
report.scanned >= 1,
"scanned={}, items={:?}",
report.scanned,
report.items
);
assert_eq!(report.errors, 0, "no errors on lenient OCR path"); assert_eq!(report.errors, 0, "no errors on lenient OCR path");
// Locate the image doc in the report items. // Locate the image doc in the report items.
@@ -94,7 +98,11 @@ async fn ingest_image_with_ocr_produces_chunk_containing_ocr_text() {
kebab_core::IngestItemKind::New, kebab_core::IngestItemKind::New,
"image asset must be classified New on first ingest" "image asset must be classified New on first ingest"
); );
assert_eq!(img_item.chunk_count, Some(1), "image emits exactly one chunk"); assert_eq!(
img_item.chunk_count,
Some(1),
"image emits exactly one chunk"
);
// Inspect the stored chunk text via kb-app's inspect_chunk facade. // Inspect the stored chunk text via kb-app's inspect_chunk facade.
let doc_id = img_item.doc_id.clone().expect("image doc id"); let doc_id = img_item.doc_id.clone().expect("image doc id");
@@ -117,10 +125,12 @@ async fn ingest_image_with_ocr_produces_chunk_containing_ocr_text() {
// Sanity: the doc was actually persisted into SQLite (kb-app's // Sanity: the doc was actually persisted into SQLite (kb-app's
// list_docs facade reads the same store the chunker writes to). // list_docs facade reads the same store the chunker writes to).
let summaries = kebab_app::list_docs_with_config(cfg, kebab_core::DocFilter::default()) let summaries =
.expect("list_docs"); kebab_app::list_docs_with_config(cfg, kebab_core::DocFilter::default()).expect("list_docs");
assert!( assert!(
summaries.iter().any(|s| s.doc_path.0.ends_with("diagram.png")), summaries
.iter()
.any(|s| s.doc_path.0.ends_with("diagram.png")),
"image doc must appear in list_docs" "image doc must appear in list_docs"
); );
@@ -171,8 +181,7 @@ async fn ingest_image_with_ocr_and_caption_populates_both_fields() {
.iter() .iter()
.find(|i| i.doc_path.0.ends_with("diagram.png")) .find(|i| i.doc_path.0.ends_with("diagram.png"))
.unwrap(); .unwrap();
let doc = kebab_app::inspect_doc_with_config(cfg, img_item.doc_id.as_ref().unwrap()) let doc = kebab_app::inspect_doc_with_config(cfg, img_item.doc_id.as_ref().unwrap()).unwrap();
.unwrap();
let block = match &doc.blocks[0] { let block = match &doc.blocks[0] {
kebab_core::Block::ImageRef(b) => b, kebab_core::Block::ImageRef(b) => b,
_ => unreachable!(), _ => unreachable!(),
@@ -267,8 +276,7 @@ async fn image_indexed_with_filename_when_ocr_and_caption_disabled() {
let cfg_clone = cfg.clone(); let cfg_clone = cfg.clone();
let scope = env.scope(); let scope = env.scope();
let report = spawn_blocking(move || { let report = spawn_blocking(move || {
kebab_app::ingest_with_config(cfg_clone, scope, false) kebab_app::ingest_with_config(cfg_clone, scope, false).expect("ingest with no OCR/caption")
.expect("ingest with no OCR/caption")
}) })
.await .await
.expect("task"); .expect("task");
@@ -282,8 +290,7 @@ async fn image_indexed_with_filename_when_ocr_and_caption_disabled() {
.find(|i| i.doc_path.0.ends_with("raw.png")) .find(|i| i.doc_path.0.ends_with("raw.png"))
.unwrap(); .unwrap();
assert_eq!(img_item.chunk_count, Some(1), "image emits one chunk"); assert_eq!(img_item.chunk_count, Some(1), "image emits one chunk");
let doc = kebab_app::inspect_doc_with_config(cfg, img_item.doc_id.as_ref().unwrap()) let doc = kebab_app::inspect_doc_with_config(cfg, img_item.doc_id.as_ref().unwrap()).unwrap();
.unwrap();
let block = match &doc.blocks[0] { let block = match &doc.blocks[0] {
kebab_core::Block::ImageRef(b) => b, kebab_core::Block::ImageRef(b) => b,
_ => unreachable!(), _ => unreachable!(),
@@ -392,16 +399,12 @@ async fn re_ingest_image_produces_unchanged_with_same_doc_id() {
let scope1 = scope.clone(); let scope1 = scope.clone();
let scope2 = scope.clone(); let scope2 = scope.clone();
let r1 = spawn_blocking(move || { let r1 = spawn_blocking(move || kebab_app::ingest_with_config(cfg1, scope1, false).unwrap())
kebab_app::ingest_with_config(cfg1, scope1, false).unwrap() .await
}) .unwrap();
.await let r2 = spawn_blocking(move || kebab_app::ingest_with_config(cfg2, scope2, false).unwrap())
.unwrap(); .await
let r2 = spawn_blocking(move || { .unwrap();
kebab_app::ingest_with_config(cfg2, scope2, false).unwrap()
})
.await
.unwrap();
let id1 = r1 let id1 = r1
.items .items

View File

@@ -21,11 +21,16 @@ fn second_ingest_of_unchanged_corpus_marks_all_unchanged() {
// First ingest — populates the DB. Use the legacy entry so the // First ingest — populates the DB. Use the legacy entry so the
// assertions cover the "previously ingested" set without needing // assertions cover the "previously ingested" set without needing
// IngestOpts::default() to behave identically. // IngestOpts::default() to behave identically.
let first = let first = ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
assert_eq!(first.errors, 0, "first ingest must not error: {first:?}"); assert_eq!(first.errors, 0, "first ingest must not error: {first:?}");
assert!(first.new >= 1, "first ingest must create new docs: {first:?}"); assert!(
assert_eq!(first.unchanged, 0, "first ingest cannot have unchanged: {first:?}"); first.new >= 1,
"first ingest must create new docs: {first:?}"
);
assert_eq!(
first.unchanged, 0,
"first ingest cannot have unchanged: {first:?}"
);
let scanned = first.scanned; let scanned = first.scanned;
@@ -38,9 +43,15 @@ fn second_ingest_of_unchanged_corpus_marks_all_unchanged() {
IngestOpts::default(), IngestOpts::default(),
) )
.unwrap(); .unwrap();
assert_eq!(second.scanned, scanned, "second scanned matches first: {second:?}"); assert_eq!(
second.scanned, scanned,
"second scanned matches first: {second:?}"
);
assert_eq!(second.new, 0, "no new docs on re-ingest: {second:?}"); assert_eq!(second.new, 0, "no new docs on re-ingest: {second:?}");
assert_eq!(second.updated, 0, "nothing should be marked updated: {second:?}"); assert_eq!(
second.updated, 0,
"nothing should be marked updated: {second:?}"
);
assert_eq!( assert_eq!(
second.unchanged, scanned, second.unchanged, scanned,
"every doc must be Unchanged: {second:?}" "every doc must be Unchanged: {second:?}"
@@ -52,10 +63,12 @@ fn second_ingest_of_unchanged_corpus_marks_all_unchanged() {
fn force_reingest_bypasses_skip() { fn force_reingest_bypasses_skip() {
let env = TestEnv::lexical_only(); let env = TestEnv::lexical_only();
let first = let first = ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
assert_eq!(first.errors, 0, "first ingest must not error: {first:?}"); assert_eq!(first.errors, 0, "first ingest must not error: {first:?}");
assert!(first.new >= 1, "first ingest must create new docs: {first:?}"); assert!(
first.new >= 1,
"first ingest must create new docs: {first:?}"
);
let scanned = first.scanned; let scanned = first.scanned;
let second = ingest_with_config_opts( let second = ingest_with_config_opts(

View File

@@ -107,13 +107,9 @@ fn cancel_none_is_uncancellable_default() {
// ingest_with_config_progress (no cancel) runs to completion. // ingest_with_config_progress (no cancel) runs to completion.
let env = TestEnv::lexical_only(); let env = TestEnv::lexical_only();
let (tx, rx) = mpsc::channel::<IngestEvent>(); let (tx, rx) = mpsc::channel::<IngestEvent>();
let report = kebab_app::ingest_with_config_progress( let report =
env.config.clone(), kebab_app::ingest_with_config_progress(env.config.clone(), env.scope(), true, Some(tx))
env.scope(), .unwrap();
true,
Some(tx),
)
.unwrap();
assert_eq!(report.scanned, 3); assert_eq!(report.scanned, 3);
assert_eq!(report.new, 3); assert_eq!(report.new, 3);

View File

@@ -107,5 +107,8 @@ fn ingest_file_errors_on_unsupported_extension() {
let err = kebab_app::ingest_file_with_config(cfg, &docx).unwrap_err(); let err = kebab_app::ingest_file_with_config(cfg, &docx).unwrap_err();
assert!(err.to_string().contains("unsupported extension"), "{err}"); assert!(err.to_string().contains("unsupported extension"), "{err}");
assert!(err.to_string().contains(".docx") || err.to_string().contains("docx"), "{err}"); assert!(
err.to_string().contains(".docx") || err.to_string().contains("docx"),
"{err}"
);
} }

View File

@@ -8,8 +8,7 @@ use common::TestEnv;
#[test] #[test]
fn ingest_then_list_inspects_round_trip() { fn ingest_then_list_inspects_round_trip() {
let env = TestEnv::lexical_only(); let env = TestEnv::lexical_only();
let report = let report = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
kebab_app::ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
// The fixture has 3 markdown files; first ingest should label them // The fixture has 3 markdown files; first ingest should label them
// all as New. // all as New.
@@ -27,17 +26,14 @@ fn ingest_then_list_inspects_round_trip() {
} }
// list_docs returns the 3 docs. // list_docs returns the 3 docs.
let docs = kebab_app::list_docs_with_config( let docs =
env.config.clone(), kebab_app::list_docs_with_config(env.config.clone(), kebab_core::DocFilter::default())
kebab_core::DocFilter::default(), .unwrap();
)
.unwrap();
assert_eq!(docs.len(), 3, "docs: {docs:?}"); assert_eq!(docs.len(), 3, "docs: {docs:?}");
// inspect_doc round-trips one of them. // inspect_doc round-trips one of them.
let any_doc_id = docs[0].doc_id.clone(); let any_doc_id = docs[0].doc_id.clone();
let canonical = kebab_app::inspect_doc_with_config(env.config.clone(), &any_doc_id) let canonical = kebab_app::inspect_doc_with_config(env.config.clone(), &any_doc_id).unwrap();
.unwrap();
assert_eq!(canonical.doc_id, any_doc_id); assert_eq!(canonical.doc_id, any_doc_id);
assert!(!canonical.blocks.is_empty(), "blocks empty"); assert!(!canonical.blocks.is_empty(), "blocks empty");
} }
@@ -46,12 +42,10 @@ fn ingest_then_list_inspects_round_trip() {
fn ingest_idempotent_on_second_run() { fn ingest_idempotent_on_second_run() {
let env = TestEnv::lexical_only(); let env = TestEnv::lexical_only();
let r1 = let r1 = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
kebab_app::ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
assert_eq!(r1.new, 3); assert_eq!(r1.new, 3);
let r2 = let r2 = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
kebab_app::ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
// Same files re-ingested — p9-fb-23 task 7 introduced the early-skip // Same files re-ingested — p9-fb-23 task 7 introduced the early-skip
// path: when checksum + parser/chunker/embedding versions all match, // path: when checksum + parser/chunker/embedding versions all match,
// the second run reports `Unchanged` rather than `Updated`. Pre-p9-fb-23 // the second run reports `Unchanged` rather than `Updated`. Pre-p9-fb-23
@@ -63,19 +57,16 @@ fn ingest_idempotent_on_second_run() {
assert_eq!(r2.unchanged, 3, "second run unchanged: {r2:?}"); assert_eq!(r2.unchanged, 3, "second run unchanged: {r2:?}");
// list_docs still has 3 docs (no duplicates). // list_docs still has 3 docs (no duplicates).
let docs = kebab_app::list_docs_with_config( let docs =
env.config.clone(), kebab_app::list_docs_with_config(env.config.clone(), kebab_core::DocFilter::default())
kebab_core::DocFilter::default(), .unwrap();
)
.unwrap();
assert_eq!(docs.len(), 3); assert_eq!(docs.len(), 3);
} }
#[test] #[test]
fn ingest_summary_only_drops_items() { fn ingest_summary_only_drops_items() {
let env = TestEnv::lexical_only(); let env = TestEnv::lexical_only();
let report = let report = kebab_app::ingest_with_config(env.config.clone(), env.scope(), true).unwrap();
kebab_app::ingest_with_config(env.config.clone(), env.scope(), true).unwrap();
assert_eq!(report.scanned, 3); assert_eq!(report.scanned, 3);
assert!(report.items.is_none(), "summary-only should null items"); assert!(report.items.is_none(), "summary-only should null items");
} }
@@ -87,12 +78,10 @@ fn ingest_records_ingest_runs_row_with_aggregate_counts() {
// of every run. `summary_only=true` writes `items_json=NULL`; the // of every run. `summary_only=true` writes `items_json=NULL`; the
// counts MUST still be present. // counts MUST still be present.
let env = TestEnv::lexical_only(); let env = TestEnv::lexical_only();
let report = kebab_app::ingest_with_config(env.config.clone(), env.scope(), true) let report = kebab_app::ingest_with_config(env.config.clone(), env.scope(), true).unwrap();
.unwrap();
assert_eq!(report.scanned, 3); assert_eq!(report.scanned, 3);
let db_path = std::path::PathBuf::from(&env.config.storage.data_dir) let db_path = std::path::PathBuf::from(&env.config.storage.data_dir).join("kebab.sqlite");
.join("kebab.sqlite");
let conn = rusqlite::Connection::open(&db_path).expect("open kebab.sqlite"); let conn = rusqlite::Connection::open(&db_path).expect("open kebab.sqlite");
let (scanned, new_c, updated, skipped, errors, items_json): ( let (scanned, new_c, updated, skipped, errors, items_json): (
i64, i64,
@@ -141,25 +130,18 @@ fn ingest_provider_none_skips_lance() {
// tree shape (no `<data_dir>/lancedb` directory, or no `*.lance` // tree shape (no `<data_dir>/lancedb` directory, or no `*.lance`
// tables under it). // tables under it).
let env = TestEnv::lexical_only(); let env = TestEnv::lexical_only();
let report = let report = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
kebab_app::ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
assert_eq!(report.errors, 0, "lexical-only run must not error"); assert_eq!(report.errors, 0, "lexical-only run must not error");
assert_eq!(report.new, 3); assert_eq!(report.new, 3);
let lance_dir = std::path::PathBuf::from(&env.config.storage.data_dir) let lance_dir = std::path::PathBuf::from(&env.config.storage.data_dir).join("lancedb");
.join("lancedb");
if lance_dir.exists() { if lance_dir.exists() {
// If the dir was created (e.g., by an earlier consumer touching // If the dir was created (e.g., by an earlier consumer touching
// the path), it MUST contain no `.lance` tables. // the path), it MUST contain no `.lance` tables.
let mut had_lance_table = false; let mut had_lance_table = false;
for entry in std::fs::read_dir(&lance_dir).expect("read lance_dir") { for entry in std::fs::read_dir(&lance_dir).expect("read lance_dir") {
let entry = entry.unwrap(); let entry = entry.unwrap();
if entry if entry.path().extension().and_then(|s| s.to_str()) == Some("lance") {
.path()
.extension()
.and_then(|s| s.to_str())
== Some("lance")
{
had_lance_table = true; had_lance_table = true;
break; break;
} }
@@ -189,8 +171,7 @@ fn list_docs_filters_by_tags_any() {
tags_any: vec!["rust".to_string()], tags_any: vec!["rust".to_string()],
..Default::default() ..Default::default()
}; };
let rust_docs = let rust_docs = kebab_app::list_docs_with_config(env.config.clone(), rust_filter).unwrap();
kebab_app::list_docs_with_config(env.config.clone(), rust_filter).unwrap();
// intro.md and notes/cargo.md both tag "rust". // intro.md and notes/cargo.md both tag "rust".
assert_eq!(rust_docs.len(), 2, "expected 2 rust docs: {rust_docs:?}"); assert_eq!(rust_docs.len(), 2, "expected 2 rust docs: {rust_docs:?}");
} }
@@ -198,8 +179,9 @@ fn list_docs_filters_by_tags_any() {
#[test] #[test]
fn inspect_doc_not_found_returns_actionable_error() { fn inspect_doc_not_found_returns_actionable_error() {
let env = TestEnv::lexical_only(); let env = TestEnv::lexical_only();
let bogus = let bogus = kebab_core::DocumentId(
kebab_core::DocumentId("0000000000000000000000000000000000000000000000000000000000000000".to_string()); "0000000000000000000000000000000000000000000000000000000000000000".to_string(),
);
let err = kebab_app::inspect_doc_with_config(env.config.clone(), &bogus).unwrap_err(); let err = kebab_app::inspect_doc_with_config(env.config.clone(), &bogus).unwrap_err();
let msg = format!("{err:#}"); let msg = format!("{err:#}");
assert!( assert!(
@@ -218,8 +200,7 @@ fn inspect_chunk_not_found_returns_actionable_error() {
let bogus = kebab_core::ChunkId( let bogus = kebab_core::ChunkId(
"0000000000000000000000000000000000000000000000000000000000000000".to_string(), "0000000000000000000000000000000000000000000000000000000000000000".to_string(),
); );
let err = kebab_app::inspect_chunk_with_config(env.config.clone(), &bogus) let err = kebab_app::inspect_chunk_with_config(env.config.clone(), &bogus).unwrap_err();
.unwrap_err();
let msg = format!("{err:#}"); let msg = format!("{err:#}");
assert!(msg.contains("not found"), "got: {msg}"); assert!(msg.contains("not found"), "got: {msg}");
} }
@@ -251,22 +232,18 @@ fn ingest_with_config_opts_default_matches_legacy_behaviour() {
#[test] #[test]
fn ingest_stamps_chunker_version_on_document() { fn ingest_stamps_chunker_version_on_document() {
let env = TestEnv::lexical_only(); let env = TestEnv::lexical_only();
let report = let report = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
kebab_app::ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
assert!(report.new >= 1, "expected at least one new doc: {report:?}"); assert!(report.new >= 1, "expected at least one new doc: {report:?}");
assert_eq!(report.errors, 0, "no errors expected: {report:?}"); assert_eq!(report.errors, 0, "no errors expected: {report:?}");
let docs = kebab_app::list_docs_with_config( let docs =
env.config.clone(), kebab_app::list_docs_with_config(env.config.clone(), kebab_core::DocFilter::default())
kebab_core::DocFilter::default(), .unwrap();
)
.unwrap();
assert!(!docs.is_empty(), "no docs after ingest"); assert!(!docs.is_empty(), "no docs after ingest");
for doc_entry in &docs { for doc_entry in &docs {
let canonical = let canonical =
kebab_app::inspect_doc_with_config(env.config.clone(), &doc_entry.doc_id) kebab_app::inspect_doc_with_config(env.config.clone(), &doc_entry.doc_id).unwrap();
.unwrap();
assert!( assert!(
canonical.last_chunker_version.is_some(), canonical.last_chunker_version.is_some(),
"last_chunker_version must be stamped for doc {}: got {:?}", "last_chunker_version must be stamped for doc {}: got {:?}",

View File

@@ -17,8 +17,7 @@ use std::sync::atomic::AtomicBool;
use common::TestEnv; use common::TestEnv;
fn ollama_endpoint() -> String { fn ollama_endpoint() -> String {
std::env::var("KEBAB_PDF_OCR_ENDPOINT") std::env::var("KEBAB_PDF_OCR_ENDPOINT").unwrap_or_else(|_| "http://localhost:11434".to_string())
.unwrap_or_else(|_| "http://localhost:11434".to_string())
} }
fn make_ocr_env_real() -> TestEnv { fn make_ocr_env_real() -> TestEnv {
@@ -43,8 +42,8 @@ fn make_ocr_env_real() -> TestEnv {
fn ingest_with_mock_ocr_yields_pdf_ocr_summary() { fn ingest_with_mock_ocr_yields_pdf_ocr_summary() {
let env = make_ocr_env_real(); let env = make_ocr_env_real();
let report = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false) let report =
.expect("ingest"); kebab_app::ingest_with_config(env.config.clone(), env.scope(), false).expect("ingest");
assert!(report.new >= 1, "at least one PDF ingested: {report:?}"); assert!(report.new >= 1, "at least one PDF ingested: {report:?}");
@@ -72,15 +71,13 @@ fn ingest_with_mock_ocr_yields_pdf_ocr_summary() {
fn ocr_text_indexed_and_searchable() { fn ocr_text_indexed_and_searchable() {
let env = make_ocr_env_real(); let env = make_ocr_env_real();
kebab_app::ingest_with_config(env.config.clone(), env.scope(), false) kebab_app::ingest_with_config(env.config.clone(), env.scope(), false).expect("ingest");
.expect("ingest");
// Search for a Korean morpheme expected to appear in qwen2.5vl:3b OCR // Search for a Korean morpheme expected to appear in qwen2.5vl:3b OCR
// output of the PoC ground-truth page. "다음" is a high-frequency token // output of the PoC ground-truth page. "다음" is a high-frequency token
// in page1.txt truth file. // in page1.txt truth file.
let query = common::lexical_query("다음"); let query = common::lexical_query("다음");
let hits = let hits = kebab_app::search_with_config(env.config.clone(), query).expect("search");
kebab_app::search_with_config(env.config.clone(), query).expect("search");
assert!( assert!(
!hits.is_empty(), !hits.is_empty(),

View File

@@ -13,13 +13,9 @@ use kebab_core::IngestItemKind;
fn run_with_progress() -> Vec<IngestEvent> { fn run_with_progress() -> Vec<IngestEvent> {
let env = TestEnv::lexical_only(); let env = TestEnv::lexical_only();
let (tx, rx) = mpsc::channel::<IngestEvent>(); let (tx, rx) = mpsc::channel::<IngestEvent>();
let report = kebab_app::ingest_with_config_progress( let report =
env.config.clone(), kebab_app::ingest_with_config_progress(env.config.clone(), env.scope(), false, Some(tx))
env.scope(), .unwrap();
false,
Some(tx),
)
.unwrap();
assert_eq!(report.scanned, 3); assert_eq!(report.scanned, 3);
assert_eq!(report.new, 3); assert_eq!(report.new, 3);
@@ -116,13 +112,9 @@ fn ingest_with_config_progress_none_matches_ingest_with_config() {
// `ingest_with_config_progress(..., None)` must produce identical // `ingest_with_config_progress(..., None)` must produce identical
// reports modulo wall-clock duration. // reports modulo wall-clock duration.
let env = TestEnv::lexical_only(); let env = TestEnv::lexical_only();
let r_none = kebab_app::ingest_with_config_progress( let r_none =
env.config.clone(), kebab_app::ingest_with_config_progress(env.config.clone(), env.scope(), true, None)
env.scope(), .unwrap();
true,
None,
)
.unwrap();
assert_eq!(r_none.scanned, 3); assert_eq!(r_none.scanned, 3);
assert_eq!(r_none.new, 3); assert_eq!(r_none.new, 3);
} }
@@ -134,13 +126,9 @@ fn dropped_receiver_does_not_panic_or_fail_ingest() {
let env = TestEnv::lexical_only(); let env = TestEnv::lexical_only();
let (tx, rx) = mpsc::channel::<IngestEvent>(); let (tx, rx) = mpsc::channel::<IngestEvent>();
drop(rx); drop(rx);
let report = kebab_app::ingest_with_config_progress( let report =
env.config.clone(), kebab_app::ingest_with_config_progress(env.config.clone(), env.scope(), true, Some(tx))
env.scope(), .unwrap();
true,
Some(tx),
)
.unwrap();
assert_eq!(report.scanned, 3); assert_eq!(report.scanned, 3);
} }
@@ -185,13 +173,8 @@ fn pdf_ocr_progress_emits_started_finished_events() {
}; };
let (tx, rx) = mpsc::channel::<IngestEvent>(); let (tx, rx) = mpsc::channel::<IngestEvent>();
let _report = kebab_app::ingest_with_config_progress( let _report = kebab_app::ingest_with_config_progress(config, scope, false, Some(tx))
config, .expect("ingest_with_config_progress");
scope,
false,
Some(tx),
)
.expect("ingest_with_config_progress");
let events: Vec<_> = rx.iter().collect(); let events: Vec<_> = rx.iter().collect();
@@ -204,7 +187,16 @@ fn pdf_ocr_progress_emits_started_finished_events() {
.filter(|e| matches!(e, IngestEvent::PdfOcrFinished { .. })) .filter(|e| matches!(e, IngestEvent::PdfOcrFinished { .. }))
.count(); .count();
assert!(started_count >= 1, "PdfOcrStarted 가 ≥ 1 emit 됨 (got {started_count})"); assert!(
assert!(finished_count >= 1, "PdfOcrFinished 가 ≥ 1 emit 됨 (got {finished_count})"); started_count >= 1,
assert_eq!(started_count, finished_count, "Started 와 Finished 의 count 일치"); "PdfOcrStarted 가 ≥ 1 emit 됨 (got {started_count})"
);
assert!(
finished_count >= 1,
"PdfOcrFinished 가 ≥ 1 emit 됨 (got {finished_count})"
);
assert_eq!(
started_count, finished_count,
"Started 와 Finished 의 count 일치"
);
} }

View File

@@ -29,12 +29,14 @@ fn ingest_stdin_writes_frontmatter_and_reports_new() {
"## Body content\n\nMore.", "## Body content\n\nMore.",
"Article X", "Article X",
Some("https://example.com/x"), Some("https://example.com/x"),
).unwrap(); )
.unwrap();
assert_eq!(report.new, 1, "{report:?}"); assert_eq!(report.new, 1, "{report:?}");
// _external/ contains exactly one .md file with frontmatter. // _external/ contains exactly one .md file with frontmatter.
let ext_dir = std::path::PathBuf::from(&cfg.workspace.root).join("_external"); let ext_dir = std::path::PathBuf::from(&cfg.workspace.root).join("_external");
let entries: Vec<_> = fs::read_dir(&ext_dir).unwrap() let entries: Vec<_> = fs::read_dir(&ext_dir)
.unwrap()
.filter_map(std::result::Result::ok) .filter_map(std::result::Result::ok)
.collect(); .collect();
assert_eq!(entries.len(), 1); assert_eq!(entries.len(), 1);
@@ -50,16 +52,13 @@ fn ingest_stdin_without_source_uri() {
let dir = tempfile::tempdir().unwrap(); let dir = tempfile::tempdir().unwrap();
let cfg = fresh_cfg(dir.path()); let cfg = fresh_cfg(dir.path());
let report = kebab_app::ingest_stdin_with_config( let report =
cfg.clone(), kebab_app::ingest_stdin_with_config(cfg.clone(), "## Body", "Title", None).unwrap();
"## Body",
"Title",
None,
).unwrap();
assert_eq!(report.new, 1); assert_eq!(report.new, 1);
let ext_dir = std::path::PathBuf::from(&cfg.workspace.root).join("_external"); let ext_dir = std::path::PathBuf::from(&cfg.workspace.root).join("_external");
let entries: Vec<_> = fs::read_dir(&ext_dir).unwrap() let entries: Vec<_> = fs::read_dir(&ext_dir)
.unwrap()
.filter_map(std::result::Result::ok) .filter_map(std::result::Result::ok)
.collect(); .collect();
let content = fs::read_to_string(entries[0].path()).unwrap(); let content = fs::read_to_string(entries[0].path()).unwrap();

View File

@@ -17,9 +17,8 @@ fn init_workspace_header_lists_supported_extensions() {
} }
kebab_app::init_workspace(true).expect("init_workspace"); kebab_app::init_workspace(true).expect("init_workspace");
let cfg_path = kebab_config::Config::xdg_config_path(); let cfg_path = kebab_config::Config::xdg_config_path();
let body = std::fs::read_to_string(&cfg_path).unwrap_or_else(|e| { let body = std::fs::read_to_string(&cfg_path)
panic!("read config at {}: {e}", cfg_path.display()) .unwrap_or_else(|e| panic!("read config at {}: {e}", cfg_path.display()));
});
assert!( assert!(
body.contains("처리 가능한 형식"), body.contains("처리 가능한 형식"),
"header lists supported types section: body=\n{body}" "header lists supported types section: body=\n{body}"

View File

@@ -9,9 +9,8 @@ use std::sync::atomic::AtomicBool;
use common::mock_ocr::MockOcrEngine; use common::mock_ocr::MockOcrEngine;
use kebab_app::pdf_ocr_apply::{PdfOcrOpts, apply_ocr_to_pdf_pages}; use kebab_app::pdf_ocr_apply::{PdfOcrOpts, apply_ocr_to_pdf_pages};
use kebab_core::{ use kebab_core::{
AssetStorage, Block, CanonicalDocument, Checksum, ExtractConfig, ExtractContext, AssetStorage, Block, CanonicalDocument, Checksum, ExtractConfig, ExtractContext, Extractor,
Extractor, Inline, Lang, MediaType, RawAsset, SourceSpan, Inline, Lang, MediaType, RawAsset, SourceSpan, SourceUri, WorkspacePath, id_for_asset,
SourceUri, WorkspacePath, id_for_asset,
}; };
use kebab_parse_pdf::PdfTextExtractor; use kebab_parse_pdf::PdfTextExtractor;
use time::OffsetDateTime; use time::OffsetDateTime;
@@ -258,8 +257,8 @@ fn f6_flatedecode_skipped_with_warning() {
// Test 7: F7 CCITTFax → skip + warning (verifier M-4 split) // Test 7: F7 CCITTFax → skip + warning (verifier M-4 split)
#[test] #[test]
fn f7_ccittfax_skipped_with_warning() { fn f7_ccittfax_skipped_with_warning() {
let bytes = std::fs::read("../kebab-parse-pdf/tests/fixtures/ccitt.pdf") let bytes =
.expect("F7 fixture missing"); std::fs::read("../kebab-parse-pdf/tests/fixtures/ccitt.pdf").expect("F7 fixture missing");
let mut canonical = canonical_with_empty_block(); // page-1 block from F1 let mut canonical = canonical_with_empty_block(); // page-1 block from F1
let engine = MockOcrEngine::single("SHOULD_NOT_BE_CALLED", false); let engine = MockOcrEngine::single("SHOULD_NOT_BE_CALLED", false);
let opts = default_opts(true); let opts = default_opts(true);

View File

@@ -46,17 +46,13 @@ fn build_text_pdf(pages: &[Option<&str>]) -> Vec<u8> {
operations: vec![ operations: vec![
Operation::new("BT", vec![]), Operation::new("BT", vec![]),
Operation::new("Tf", vec!["F1".into(), 24.into()]), Operation::new("Tf", vec!["F1".into(), 24.into()]),
Operation::new( Operation::new("Td", vec![Object::Integer(100), Object::Integer(700)]),
"Td",
vec![Object::Integer(100), Object::Integer(700)],
),
Operation::new("Tj", vec![Object::string_literal(*text)]), Operation::new("Tj", vec![Object::string_literal(*text)]),
Operation::new("ET", vec![]), Operation::new("ET", vec![]),
], ],
}; };
let stream_data = content.encode().expect("content encode"); let stream_data = content.encode().expect("content encode");
let content_id = let content_id = doc.add_object(Stream::new(dictionary! {}, stream_data));
doc.add_object(Stream::new(dictionary! {}, stream_data));
page_dict.set("Contents", content_id); page_dict.set("Contents", content_id);
} }
let page_id = doc.add_object(page_dict); let page_id = doc.add_object(page_dict);
@@ -76,8 +72,7 @@ fn build_text_pdf(pages: &[Option<&str>]) -> Vec<u8> {
Object::Integer(842), Object::Integer(842),
], ],
}; };
doc.objects doc.objects.insert(pages_id, Object::Dictionary(pages_dict));
.insert(pages_id, Object::Dictionary(pages_dict));
let catalog_id = doc.add_object(dictionary! { let catalog_id = doc.add_object(dictionary! {
"Type" => "Catalog", "Type" => "Catalog",
@@ -146,9 +141,8 @@ fn ingest_3_page_pdf_produces_one_doc_and_per_page_chunks() {
write_pdf(&env.workspace_root, "three.pdf", &bytes); write_pdf(&env.workspace_root, "three.pdf", &bytes);
let cfg = cfg_with_pdf(&env); let cfg = cfg_with_pdf(&env);
let report = let report = kebab_app::ingest_with_config(cfg.clone(), env.scope(), false)
kebab_app::ingest_with_config(cfg.clone(), env.scope(), false) .expect("PDF ingest must succeed");
.expect("PDF ingest must succeed");
assert_eq!(report.errors, 0); assert_eq!(report.errors, 0);
let items = report.items.as_ref().expect("items present"); let items = report.items.as_ref().expect("items present");
@@ -157,8 +151,16 @@ fn ingest_3_page_pdf_produces_one_doc_and_per_page_chunks() {
.find(|i| i.doc_path.0.ends_with("three.pdf")) .find(|i| i.doc_path.0.ends_with("three.pdf"))
.expect("PDF item present"); .expect("PDF item present");
assert_eq!(pdf_item.kind, IngestItemKind::New); assert_eq!(pdf_item.kind, IngestItemKind::New);
assert_eq!(pdf_item.block_count, Some(3), "one Block::Paragraph per page"); assert_eq!(
assert_eq!(pdf_item.chunk_count, Some(3), "one chunk per non-empty page"); pdf_item.block_count,
Some(3),
"one Block::Paragraph per page"
);
assert_eq!(
pdf_item.chunk_count,
Some(3),
"one chunk per non-empty page"
);
assert_eq!( assert_eq!(
pdf_item.parser_version.as_ref().map(|p| p.0.as_str()), pdf_item.parser_version.as_ref().map(|p| p.0.as_str()),
Some("pdf-text-v1") Some("pdf-text-v1")
@@ -169,11 +171,8 @@ fn ingest_3_page_pdf_produces_one_doc_and_per_page_chunks() {
); );
// Inspect the stored doc to confirm SourceSpan::Page round-trip. // Inspect the stored doc to confirm SourceSpan::Page round-trip.
let doc = kebab_app::inspect_doc_with_config( let doc = kebab_app::inspect_doc_with_config(cfg, pdf_item.doc_id.as_ref().unwrap())
cfg, .expect("inspect_doc returns the PDF document");
pdf_item.doc_id.as_ref().unwrap(),
)
.expect("inspect_doc returns the PDF document");
assert_eq!(doc.blocks.len(), 3); assert_eq!(doc.blocks.len(), 3);
for (i, block) in doc.blocks.iter().enumerate() { for (i, block) in doc.blocks.iter().enumerate() {
let want_page = (i as u32) + 1; let want_page = (i as u32) + 1;
@@ -202,8 +201,7 @@ fn re_ingest_identical_pdf_produces_unchanged_with_same_doc_id() {
write_pdf(&env.workspace_root, "stable.pdf", &bytes); write_pdf(&env.workspace_root, "stable.pdf", &bytes);
let cfg = cfg_with_pdf(&env); let cfg = cfg_with_pdf(&env);
let report1 = let report1 = kebab_app::ingest_with_config(cfg.clone(), env.scope(), false).unwrap();
kebab_app::ingest_with_config(cfg.clone(), env.scope(), false).unwrap();
let item1 = report1 let item1 = report1
.items .items
.as_ref() .as_ref()
@@ -214,8 +212,7 @@ fn re_ingest_identical_pdf_produces_unchanged_with_same_doc_id() {
.unwrap(); .unwrap();
assert_eq!(item1.kind, IngestItemKind::New); assert_eq!(item1.kind, IngestItemKind::New);
let report2 = let report2 = kebab_app::ingest_with_config(cfg.clone(), env.scope(), false).unwrap();
kebab_app::ingest_with_config(cfg.clone(), env.scope(), false).unwrap();
let item2 = report2 let item2 = report2
.items .items
.unwrap() .unwrap()
@@ -239,8 +236,7 @@ fn re_ingest_edited_pdf_produces_new_doc_id() {
std::fs::write(&path, &bytes_v1).unwrap(); std::fs::write(&path, &bytes_v1).unwrap();
let cfg = cfg_with_pdf(&env); let cfg = cfg_with_pdf(&env);
let report_v1 = let report_v1 = kebab_app::ingest_with_config(cfg.clone(), env.scope(), false).unwrap();
kebab_app::ingest_with_config(cfg.clone(), env.scope(), false).unwrap();
let id_v1 = report_v1 let id_v1 = report_v1
.items .items
.as_ref() .as_ref()
@@ -252,12 +248,10 @@ fn re_ingest_edited_pdf_produces_new_doc_id() {
.clone() .clone()
.unwrap(); .unwrap();
let bytes_v2 = let bytes_v2 = build_text_pdf(&[Some("VERSION TWO entirely different body content.")]);
build_text_pdf(&[Some("VERSION TWO entirely different body content.")]);
std::fs::write(&path, &bytes_v2).unwrap(); std::fs::write(&path, &bytes_v2).unwrap();
let report_v2 = let report_v2 = kebab_app::ingest_with_config(cfg.clone(), env.scope(), false).unwrap();
kebab_app::ingest_with_config(cfg.clone(), env.scope(), false).unwrap();
let item_v2 = report_v2 let item_v2 = report_v2
.items .items
.as_ref() .as_ref()
@@ -282,9 +276,11 @@ fn encrypted_pdf_fails_with_qpdf_hint() {
write_pdf(&env.workspace_root, "secret.pdf", &bytes); write_pdf(&env.workspace_root, "secret.pdf", &bytes);
let cfg = cfg_with_pdf(&env); let cfg = cfg_with_pdf(&env);
let report = let report = kebab_app::ingest_with_config(cfg, env.scope(), false).unwrap();
kebab_app::ingest_with_config(cfg, env.scope(), false).unwrap(); assert_eq!(
assert_eq!(report.errors, 1, "encrypted PDF must increment errors exactly once"); report.errors, 1,
"encrypted PDF must increment errors exactly once"
);
let items = report.items.as_ref().unwrap(); let items = report.items.as_ref().unwrap();
let pdf_item = items let pdf_item = items
.iter() .iter()
@@ -310,9 +306,11 @@ fn corrupt_pdf_fails_without_storing() {
write_pdf(&env.workspace_root, "corrupt.pdf", &bytes); write_pdf(&env.workspace_root, "corrupt.pdf", &bytes);
let cfg = cfg_with_pdf(&env); let cfg = cfg_with_pdf(&env);
let report = let report = kebab_app::ingest_with_config(cfg.clone(), env.scope(), false).unwrap();
kebab_app::ingest_with_config(cfg.clone(), env.scope(), false).unwrap(); assert_eq!(
assert_eq!(report.errors, 1, "corrupt PDF must increment errors exactly once"); report.errors, 1,
"corrupt PDF must increment errors exactly once"
);
let items = report.items.as_ref().unwrap(); let items = report.items.as_ref().unwrap();
let pdf_item = items let pdf_item = items
.iter() .iter()
@@ -322,11 +320,8 @@ fn corrupt_pdf_fails_without_storing() {
// Confirm the doc was NOT stored — list_docs returns nothing for // Confirm the doc was NOT stored — list_docs returns nothing for
// this path. // this path.
let summaries = kebab_app::list_docs_with_config( let summaries =
cfg, kebab_app::list_docs_with_config(cfg, kebab_core::DocFilter::default()).unwrap();
kebab_core::DocFilter::default(),
)
.unwrap();
assert!( assert!(
!summaries !summaries
.iter() .iter()
@@ -341,14 +336,15 @@ fn corrupt_pdf_fails_without_storing() {
#[test] #[test]
fn mixed_page_pdf_stores_asset_with_scanned_candidate_warning() { fn mixed_page_pdf_stores_asset_with_scanned_candidate_warning() {
let env = TestEnv::lexical_only(); let env = TestEnv::lexical_only();
let bytes = let bytes = build_text_pdf(&[Some("first page"), None, Some("third page")]);
build_text_pdf(&[Some("first page"), None, Some("third page")]);
write_pdf(&env.workspace_root, "mixed.pdf", &bytes); write_pdf(&env.workspace_root, "mixed.pdf", &bytes);
let cfg = cfg_with_pdf(&env); let cfg = cfg_with_pdf(&env);
let report = let report = kebab_app::ingest_with_config(cfg.clone(), env.scope(), false).unwrap();
kebab_app::ingest_with_config(cfg.clone(), env.scope(), false).unwrap(); assert_eq!(
assert_eq!(report.errors, 0, "scanned candidate is a Warning, not Error"); report.errors, 0,
"scanned candidate is a Warning, not Error"
);
let pdf_item = report let pdf_item = report
.items .items
.as_ref() .as_ref()
@@ -368,11 +364,7 @@ fn mixed_page_pdf_stores_asset_with_scanned_candidate_warning() {
"pdf-page-v1.1 emits 0 chunks for the empty page; total = 2" "pdf-page-v1.1 emits 0 chunks for the empty page; total = 2"
); );
let doc = kebab_app::inspect_doc_with_config( let doc = kebab_app::inspect_doc_with_config(cfg, pdf_item.doc_id.as_ref().unwrap()).unwrap();
cfg,
pdf_item.doc_id.as_ref().unwrap(),
)
.unwrap();
let warnings: Vec<_> = doc let warnings: Vec<_> = doc
.provenance .provenance
.events .events
@@ -419,8 +411,7 @@ fn ingest_report_arithmetic_invariant_holds_with_corrupt_pdf() {
write_pdf(&env.workspace_root, "broken.pdf", &corrupt_pdf()); write_pdf(&env.workspace_root, "broken.pdf", &corrupt_pdf());
let cfg = cfg_with_pdf(&env); let cfg = cfg_with_pdf(&env);
let report = let report = kebab_app::ingest_with_config(cfg, env.scope(), false).unwrap();
kebab_app::ingest_with_config(cfg, env.scope(), false).unwrap();
let total = report.new + report.updated + report.skipped + report.errors; let total = report.new + report.updated + report.skipped + report.errors;
assert_eq!( assert_eq!(
report.scanned, total, report.scanned, total,
@@ -441,14 +432,12 @@ fn long_pdf_round_trips_through_lexical_pipeline() {
let pages: Vec<String> = (1..=50) let pages: Vec<String> = (1..=50)
.map(|i| format!("Page {i} body — lorem ipsum dolor sit amet.")) .map(|i| format!("Page {i} body — lorem ipsum dolor sit amet."))
.collect(); .collect();
let page_refs: Vec<Option<&str>> = let page_refs: Vec<Option<&str>> = pages.iter().map(|s| Some(s.as_str())).collect();
pages.iter().map(|s| Some(s.as_str())).collect();
let bytes = build_text_pdf(&page_refs); let bytes = build_text_pdf(&page_refs);
write_pdf(&env.workspace_root, "long.pdf", &bytes); write_pdf(&env.workspace_root, "long.pdf", &bytes);
let cfg = cfg_with_pdf(&env); let cfg = cfg_with_pdf(&env);
let report = let report = kebab_app::ingest_with_config(cfg.clone(), env.scope(), false).unwrap();
kebab_app::ingest_with_config(cfg.clone(), env.scope(), false).unwrap();
assert_eq!(report.errors, 0); assert_eq!(report.errors, 0);
let pdf_item = report let pdf_item = report
.items .items
@@ -466,8 +455,7 @@ fn long_pdf_round_trips_through_lexical_pipeline() {
// Round-trip: list_docs sees the long PDF. // Round-trip: list_docs sees the long PDF.
let summaries = let summaries =
kebab_app::list_docs_with_config(cfg, kebab_core::DocFilter::default()) kebab_app::list_docs_with_config(cfg, kebab_core::DocFilter::default()).unwrap();
.unwrap();
assert!(summaries.iter().any(|s| s.doc_path.0.ends_with("long.pdf"))); assert!(summaries.iter().any(|s| s.doc_path.0.ends_with("long.pdf")));
} }
@@ -476,13 +464,11 @@ fn long_pdf_round_trips_through_lexical_pipeline() {
#[test] #[test]
fn inspect_doc_surfaces_page_spans() { fn inspect_doc_surfaces_page_spans() {
let env = TestEnv::lexical_only(); let env = TestEnv::lexical_only();
let bytes = let bytes = build_text_pdf(&[Some("alpha body"), Some("beta body"), Some("gamma body")]);
build_text_pdf(&[Some("alpha body"), Some("beta body"), Some("gamma body")]);
write_pdf(&env.workspace_root, "inspect.pdf", &bytes); write_pdf(&env.workspace_root, "inspect.pdf", &bytes);
let cfg = cfg_with_pdf(&env); let cfg = cfg_with_pdf(&env);
let report = let report = kebab_app::ingest_with_config(cfg.clone(), env.scope(), false).unwrap();
kebab_app::ingest_with_config(cfg.clone(), env.scope(), false).unwrap();
let pdf_item = report let pdf_item = report
.items .items
.as_ref() .as_ref()
@@ -490,19 +476,12 @@ fn inspect_doc_surfaces_page_spans() {
.iter() .iter()
.find(|i| i.doc_path.0.ends_with("inspect.pdf")) .find(|i| i.doc_path.0.ends_with("inspect.pdf"))
.unwrap(); .unwrap();
let doc = kebab_app::inspect_doc_with_config( let doc = kebab_app::inspect_doc_with_config(cfg, pdf_item.doc_id.as_ref().unwrap()).unwrap();
cfg,
pdf_item.doc_id.as_ref().unwrap(),
)
.unwrap();
assert_eq!(doc.parser_version.0, "pdf-text-v1"); assert_eq!(doc.parser_version.0, "pdf-text-v1");
assert_eq!(doc.blocks.len(), 3); assert_eq!(doc.blocks.len(), 3);
for block in &doc.blocks { for block in &doc.blocks {
match block { match block {
Block::Paragraph(p) => assert!(matches!( Block::Paragraph(p) => assert!(matches!(p.common.source_span, SourceSpan::Page { .. })),
p.common.source_span,
SourceSpan::Page { .. }
)),
other => panic!("expected Paragraph, got {other:?}"), other => panic!("expected Paragraph, got {other:?}"),
} }
} }

View File

@@ -78,19 +78,15 @@ fn reset_orphans_only_purges_out_of_scope_docs() {
narrow_cfg.workspace.exclude = vec!["b.rs".to_string(), "c.rs".to_string()]; narrow_cfg.workspace.exclude = vec!["b.rs".to_string(), "c.rs".to_string()];
// Run orphans-only reset. // Run orphans-only reset.
let report = execute(ResetScope::OrphansOnly, &narrow_cfg) let report =
.expect("orphans-only reset must succeed"); execute(ResetScope::OrphansOnly, &narrow_cfg).expect("orphans-only reset must succeed");
assert_eq!( assert_eq!(
report.orphans_purged, 2, report.orphans_purged, 2,
"expected 2 orphans purged (b.rs + c.rs): {report:?}" "expected 2 orphans purged (b.rs + c.rs): {report:?}"
); );
let mut purged: Vec<String> = report let mut purged: Vec<String> = report.purged_paths.iter().map(|p| p.0.clone()).collect();
.purged_paths
.iter()
.map(|p| p.0.clone())
.collect();
purged.sort(); purged.sort();
assert_eq!( assert_eq!(
purged, purged,

View File

@@ -37,8 +37,14 @@ fn schema_models_active_arrays_empty_on_empty_corpus() {
drop(store); drop(store);
let s = schema_with_config(&cfg).unwrap(); let s = schema_with_config(&cfg).unwrap();
assert!(s.models.active_parsers.is_empty(), "empty corpus → no parsers"); assert!(
assert!(s.models.active_chunkers.is_empty(), "empty corpus → no chunkers"); s.models.active_parsers.is_empty(),
"empty corpus → no parsers"
);
assert!(
s.models.active_chunkers.is_empty(),
"empty corpus → no chunkers"
);
// backward compat: 기존 단일 field 는 markdown default 보존. // backward compat: 기존 단일 field 는 markdown default 보존.
assert_eq!(s.models.parser_version, kebab_parse_md::PARSER_VERSION); assert_eq!(s.models.parser_version, kebab_parse_md::PARSER_VERSION);
} }
@@ -55,10 +61,19 @@ fn schema_emits_active_parsers_and_chunkers_array_after_ingest() {
kebab_app::ingest_with_config(cfg.clone(), scope, false).unwrap(); kebab_app::ingest_with_config(cfg.clone(), scope, false).unwrap();
let s = schema_with_config(&cfg).unwrap(); let s = schema_with_config(&cfg).unwrap();
assert!(!s.models.active_parsers.is_empty(), "active_parsers populated after ingest"); assert!(
assert!(!s.models.active_chunkers.is_empty(), "active_chunkers populated after ingest"); !s.models.active_parsers.is_empty(),
"active_parsers populated after ingest"
);
assert!(
!s.models.active_chunkers.is_empty(),
"active_chunkers populated after ingest"
);
// active arrays must be sorted (ORDER BY in SQL). // active arrays must be sorted (ORDER BY in SQL).
let mut sorted = s.models.active_parsers.clone(); let mut sorted = s.models.active_parsers.clone();
sorted.sort(); sorted.sort();
assert_eq!(s.models.active_parsers, sorted, "active_parsers must be sorted"); assert_eq!(
s.models.active_parsers, sorted,
"active_parsers must be sorted"
);
} }

View File

@@ -27,7 +27,10 @@ fn search_with_opts_no_budget_matches_search() {
assert_eq!(resp.hits.len(), baseline.len()); assert_eq!(resp.hits.len(), baseline.len());
assert!(!resp.truncated); assert!(!resp.truncated);
assert!(resp.next_cursor.is_none(), "k=5 against 1 doc → no next page"); assert!(
resp.next_cursor.is_none(),
"k=5 against 1 doc → no next page"
);
} }
#[test] #[test]
@@ -62,7 +65,11 @@ fn budget_truncates_snippets_when_below_threshold() {
fn cursor_paginates_to_next_page() { fn cursor_paginates_to_next_page() {
let env = common::TestEnv::new(); let env = common::TestEnv::new();
for i in 0..6 { for i in 0..6 {
common::ingest_md(&env, &format!("d{i}.md"), &format!("# T{i}\n\nrust topic {i}\n")); common::ingest_md(
&env,
&format!("d{i}.md"),
&format!("# T{i}\n\nrust topic {i}\n"),
);
} }
let app = env.app(); let app = env.app();
@@ -88,7 +95,10 @@ fn cursor_paginates_to_next_page() {
page1.hits.iter().map(|h| h.chunk_id.0.clone()).collect(); page1.hits.iter().map(|h| h.chunk_id.0.clone()).collect();
let p2_ids: std::collections::HashSet<_> = let p2_ids: std::collections::HashSet<_> =
page2.hits.iter().map(|h| h.chunk_id.0.clone()).collect(); page2.hits.iter().map(|h| h.chunk_id.0.clone()).collect();
assert!(p1_ids.is_disjoint(&p2_ids), "page 2 must not repeat page 1 hits"); assert!(
p1_ids.is_disjoint(&p2_ids),
"page 2 must not repeat page 1 hits"
);
} }
#[test] #[test]

View File

@@ -75,11 +75,9 @@ fn lexical_multi_token_korean_query_hits() {
kebab_app::ingest_with_config(env.config.clone(), env.scope(), true) kebab_app::ingest_with_config(env.config.clone(), env.scope(), true)
.expect("ingest must succeed"); .expect("ingest must succeed");
let hits = kebab_app::search_with_config( let hits =
env.config.clone(), kebab_app::search_with_config(env.config.clone(), common::lexical_query("해시 충돌"))
common::lexical_query("해시 충돌"), .expect("search must succeed");
)
.expect("search must succeed");
assert!( assert!(
!hits.is_empty(), !hits.is_empty(),
@@ -113,11 +111,9 @@ fn lexical_mixed_korean_english_multi_token_query_hits() {
kebab_app::ingest_with_config(env.config.clone(), env.scope(), true) kebab_app::ingest_with_config(env.config.clone(), env.scope(), true)
.expect("ingest must succeed"); .expect("ingest must succeed");
let hits = kebab_app::search_with_config( let hits =
env.config.clone(), kebab_app::search_with_config(env.config.clone(), common::lexical_query("Rust 충돌은"))
common::lexical_query("Rust 충돌은"), .expect("search must succeed");
)
.expect("search must succeed");
assert!( assert!(
!hits.is_empty(), !hits.is_empty(),

View File

@@ -35,8 +35,8 @@ fn lexical_search_returns_hits_after_ingest() {
fn lexical_search_empty_query_returns_empty() { fn lexical_search_empty_query_returns_empty() {
let env = TestEnv::lexical_only(); let env = TestEnv::lexical_only();
kebab_app::ingest_with_config(env.config.clone(), env.scope(), true).unwrap(); kebab_app::ingest_with_config(env.config.clone(), env.scope(), true).unwrap();
let hits = kebab_app::search_with_config(env.config.clone(), common::lexical_query(" ")) let hits =
.unwrap(); kebab_app::search_with_config(env.config.clone(), common::lexical_query(" ")).unwrap();
assert!(hits.is_empty(), "blank query must short-circuit empty"); assert!(hits.is_empty(), "blank query must short-circuit empty");
} }
@@ -107,17 +107,17 @@ fn search_uncached_returns_same_hits_as_cached() {
#[test] #[test]
fn first_ingest_bumps_corpus_revision() { fn first_ingest_bumps_corpus_revision() {
let env = TestEnv::lexical_only(); let env = TestEnv::lexical_only();
let store_before = let store_before = kebab_store_sqlite::SqliteStore::open(&env.config).unwrap();
kebab_store_sqlite::SqliteStore::open(&env.config).unwrap();
store_before.run_migrations().unwrap(); store_before.run_migrations().unwrap();
assert_eq!(store_before.corpus_revision(), 0, "fresh store seeds 0"); assert_eq!(store_before.corpus_revision(), 0, "fresh store seeds 0");
let report = let report = kebab_app::ingest_with_config(env.config.clone(), env.scope(), true).unwrap();
kebab_app::ingest_with_config(env.config.clone(), env.scope(), true).unwrap(); assert!(
assert!(report.new + report.updated > 0, "first ingest must commit ≥1 doc"); report.new + report.updated > 0,
"first ingest must commit ≥1 doc"
);
let store_after = let store_after = kebab_store_sqlite::SqliteStore::open(&env.config).unwrap();
kebab_store_sqlite::SqliteStore::open(&env.config).unwrap();
assert!( assert!(
store_after.corpus_revision() >= 1, store_after.corpus_revision() >= 1,
"ingest commit must bump corpus_revision (got {})", "ingest commit must bump corpus_revision (got {})",

View File

@@ -29,7 +29,9 @@ fn fresh_doc_is_not_stale_with_default_threshold() {
assert!( assert!(
hits.iter().all(|h| !h.stale), hits.iter().all(|h| !h.stale),
"freshly-ingested doc must not be stale at default 30d threshold: {:?}", "freshly-ingested doc must not be stale at default 30d threshold: {:?}",
hits.iter().map(|h| (h.doc_path.0.clone(), h.stale)).collect::<Vec<_>>() hits.iter()
.map(|h| (h.doc_path.0.clone(), h.stale))
.collect::<Vec<_>>()
); );
} }
@@ -50,7 +52,9 @@ fn threshold_zero_disables_staleness() {
assert!( assert!(
hits.iter().all(|h| !h.stale), hits.iter().all(|h| !h.stale),
"threshold=0 disables staleness even for year-old docs: {:?}", "threshold=0 disables staleness even for year-old docs: {:?}",
hits.iter().map(|h| (h.doc_path.0.clone(), h.stale)).collect::<Vec<_>>() hits.iter()
.map(|h| (h.doc_path.0.clone(), h.stale))
.collect::<Vec<_>>()
); );
} }

View File

@@ -14,7 +14,8 @@ use common::TestEnv;
fn require_avx_or_panic() { fn require_avx_or_panic() {
#[cfg(target_arch = "x86_64")] #[cfg(target_arch = "x86_64")]
{ {
assert!(std::is_x86_feature_detected!("avx"), assert!(
std::is_x86_feature_detected!("avx"),
"kb-app vector integration test requires AVX-capable hardware; \ "kb-app vector integration test requires AVX-capable hardware; \
host CPU lacks AVX. Run on an AVX-capable machine." host CPU lacks AVX. Run on an AVX-capable machine."
); );
@@ -28,8 +29,7 @@ fn ingest_then_hybrid_search_returns_hits() {
require_avx_or_panic(); require_avx_or_panic();
let env = TestEnv::with_embeddings(); let env = TestEnv::with_embeddings();
let report = let report = kebab_app::ingest_with_config(env.config.clone(), env.scope(), true).unwrap();
kebab_app::ingest_with_config(env.config.clone(), env.scope(), true).unwrap();
assert_eq!(report.errors, 0, "no per-file errors: {report:?}"); assert_eq!(report.errors, 0, "no per-file errors: {report:?}");
assert_eq!(report.new, 3); assert_eq!(report.new, 3);
@@ -55,8 +55,7 @@ fn ingest_then_vector_search_carries_embedding_model() {
require_avx_or_panic(); require_avx_or_panic();
let env = TestEnv::with_embeddings(); let env = TestEnv::with_embeddings();
let report = let report = kebab_app::ingest_with_config(env.config.clone(), env.scope(), true).unwrap();
kebab_app::ingest_with_config(env.config.clone(), env.scope(), true).unwrap();
assert_eq!(report.errors, 0, "no per-file errors: {report:?}"); assert_eq!(report.errors, 0, "no per-file errors: {report:?}");
assert_eq!(report.new, 3); assert_eq!(report.new, 3);

View File

@@ -13,11 +13,7 @@ fn unsupported_extension_skip_carries_warning_and_is_aggregated() {
std::fs::write(workspace_root.join("legacy.docx"), b"unsupported").unwrap(); std::fs::write(workspace_root.join("legacy.docx"), b"unsupported").unwrap();
std::fs::write(workspace_root.join("Makefile"), b"unsupported").unwrap(); std::fs::write(workspace_root.join("Makefile"), b"unsupported").unwrap();
let report = kebab_app::ingest_with_config( let report = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false).unwrap();
env.config.clone(),
env.scope(),
false,
).unwrap();
let items = report.items.as_ref().expect("items array populated"); let items = report.items.as_ref().expect("items array populated");
let docx_item = items let docx_item = items
@@ -39,5 +35,8 @@ fn unsupported_extension_skip_carries_warning_and_is_aggregated() {
vec!["unsupported media type: <no-ext>".to_string()], vec!["unsupported media type: <no-ext>".to_string()],
); );
assert_eq!(report.skipped_by_extension.get("docx").copied(), Some(1)); assert_eq!(report.skipped_by_extension.get("docx").copied(), Some(1));
assert_eq!(report.skipped_by_extension.get("<no-ext>").copied(), Some(1)); assert_eq!(
report.skipped_by_extension.get("<no-ext>").copied(),
Some(1)
);
} }

View File

@@ -44,8 +44,8 @@ fn twin_files_fetch_span_uses_correct_asset() {
std::fs::write(dir_b.join("note.md"), content).unwrap(); std::fs::write(dir_b.join("note.md"), content).unwrap();
// Ingest all files (fixture workspace + our two new twins). // Ingest all files (fixture workspace + our two new twins).
let report = ingest_with_config(env.config.clone(), env.scope(), false) let report =
.expect("ingest must succeed"); ingest_with_config(env.config.clone(), env.scope(), false).expect("ingest must succeed");
assert_eq!(report.errors, 0, "no ingest errors; report={report:?}"); assert_eq!(report.errors, 0, "no ingest errors; report={report:?}");
// Both twin paths must appear as New in the report. // Both twin paths must appear as New in the report.
@@ -53,8 +53,7 @@ fn twin_files_fetch_span_uses_correct_asset() {
let twin_items: Vec<_> = items let twin_items: Vec<_> = items
.iter() .iter()
.filter(|i| { .filter(|i| {
i.doc_path.0.ends_with("src_a/note.md") i.doc_path.0.ends_with("src_a/note.md") || i.doc_path.0.ends_with("src_b/note.md")
|| i.doc_path.0.ends_with("src_b/note.md")
}) })
.collect(); .collect();
assert_eq!( assert_eq!(
@@ -149,7 +148,10 @@ fn twin_files_fetch_span_uses_correct_asset() {
// at either twin, making one twin's span fetch behave incorrectly. // at either twin, making one twin's span fetch behave incorrectly.
let report2 = ingest_with_config(env.config.clone(), env.scope(), false) let report2 = ingest_with_config(env.config.clone(), env.scope(), false)
.expect("second ingest must succeed"); .expect("second ingest must succeed");
assert_eq!(report2.errors, 0, "no ingest errors on second run; report={report2:?}"); assert_eq!(
report2.errors, 0,
"no ingest errors on second run; report={report2:?}"
);
// Re-open app after second ingest and verify span still works on both. // Re-open app after second ingest and verify span still works on both.
let app2 = env.app(); let app2 = env.app();

View File

@@ -43,9 +43,7 @@ fn twin_files_second_ingest_is_unchanged() {
let items = first.items.as_ref().expect("items must be present"); let items = first.items.as_ref().expect("items must be present");
let twin_items: Vec<_> = items let twin_items: Vec<_> = items
.iter() .iter()
.filter(|i| { .filter(|i| i.doc_path.0.ends_with("__init__.py"))
i.doc_path.0.ends_with("__init__.py")
})
.collect(); .collect();
assert_eq!( assert_eq!(
twin_items.len(), twin_items.len(),
@@ -63,8 +61,14 @@ fn twin_files_second_ingest_is_unchanged() {
// Second ingest — same files, same content → both must be Unchanged. // Second ingest — same files, same content → both must be Unchanged.
let second = ingest_with_config(env.config.clone(), env.scope(), false) let second = ingest_with_config(env.config.clone(), env.scope(), false)
.expect("second ingest must succeed"); .expect("second ingest must succeed");
assert_eq!(second.errors, 0, "second ingest: no errors; report={second:?}"); assert_eq!(
assert_eq!(second.new, 0, "second ingest: no new docs; report={second:?}"); second.errors, 0,
"second ingest: no errors; report={second:?}"
);
assert_eq!(
second.new, 0,
"second ingest: no new docs; report={second:?}"
);
assert_eq!( assert_eq!(
second.updated, 0, second.updated, 0,
"second ingest: no updated docs (twin-file bug would set this to 2); report={second:?}" "second ingest: no updated docs (twin-file bug would set this to 2); report={second:?}"

View File

@@ -39,17 +39,11 @@ impl Chunker for CodeCAstV1Chunker {
hex[..POLICY_HASH_HEX_LEN].to_string() hex[..POLICY_HASH_HEX_LEN].to_string()
} }
fn chunk( fn chunk(&self, doc: &CanonicalDocument, policy: &ChunkPolicy) -> anyhow::Result<Vec<Chunk>> {
&self,
doc: &CanonicalDocument,
policy: &ChunkPolicy,
) -> anyhow::Result<Vec<Chunk>> {
for b in &doc.blocks { for b in &doc.blocks {
let c = match b { let c = match b {
Block::Code(c) => c, Block::Code(c) => c,
_ => anyhow::bail!( _ => anyhow::bail!("CodeCAstV1Chunker only handles code docs (got non-Code block)"),
"CodeCAstV1Chunker only handles code docs (got non-Code block)"
),
}; };
if !matches!(c.common.source_span, SourceSpan::Code { .. }) { if !matches!(c.common.source_span, SourceSpan::Code { .. }) {
anyhow::bail!( anyhow::bail!(
@@ -68,9 +62,12 @@ impl Chunker for CodeCAstV1Chunker {
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let (ls, le, symbol, lang) = match &cb.common.source_span { let (ls, le, symbol, lang) = match &cb.common.source_span {
SourceSpan::Code { line_start, line_end, symbol, lang } => { SourceSpan::Code {
(*line_start, *line_end, symbol.clone(), lang.clone()) line_start,
} line_end,
symbol,
lang,
} => (*line_start, *line_end, symbol.clone(), lang.clone()),
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()]; let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
@@ -84,8 +81,13 @@ impl Chunker for CodeCAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
None, span, cb.code.clone(), &chunker_version,
&block_ids,
&base_policy_hash,
None,
span,
cb.code.clone(),
)); ));
} else { } else {
let parts = split_oversize(&cb.code); let parts = split_oversize(&cb.code);
@@ -93,9 +95,7 @@ impl Chunker for CodeCAstV1Chunker {
for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() { for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
let part_ls = ls + off_start; let part_ls = ls + off_start;
let part_le = ls + off_end; let part_le = ls + off_end;
let part_sym = symbol let part_sym = symbol.as_ref().map(|s| format!("{s} [part {}/{n}]", i + 1));
.as_ref()
.map(|s| format!("{s} [part {}/{n}]", i + 1));
let span = SourceSpan::Code { let span = SourceSpan::Code {
line_start: part_ls, line_start: part_ls,
line_end: part_le, line_end: part_le,
@@ -103,8 +103,13 @@ impl Chunker for CodeCAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
Some(part_ls), span, text, &chunker_version,
&block_ids,
&base_policy_hash,
Some(part_ls),
span,
text,
)); ));
} }
} }
@@ -183,9 +188,9 @@ fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
mod tests { mod tests {
use super::*; use super::*;
use kebab_core::{ use kebab_core::{
Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
SourceType, TrustLevel, WorkspacePath, WorkspacePath, id_for_block, id_for_doc,
}; };
use time::OffsetDateTime; use time::OffsetDateTime;
@@ -206,39 +211,60 @@ mod tests {
}; };
let bid = id_for_block(&doc_id, "code", &[], i as u32, &span); let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
Block::Code(CodeBlock { Block::Code(CodeBlock {
common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span }, common: CommonBlock {
block_id: bid,
heading_path: vec![],
source_span: span,
},
lang: Some("c".into()), lang: Some("c".into()),
code: (*code).to_string(), code: (*code).to_string(),
}) })
}) })
.collect(); .collect();
CanonicalDocument { CanonicalDocument {
doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(), doc_id,
lang: Lang("und".into()), blocks, source_asset_id: aid,
workspace_path: wp,
title: "a".into(),
lang: Lang("und".into()),
blocks,
metadata: Metadata { metadata: Metadata {
aliases: vec![], tags: vec![], aliases: vec![],
tags: vec![],
created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
source_type: SourceType::Note, trust_level: TrustLevel::Primary, source_type: SourceType::Note,
user_id_alias: None, user: Default::default(), trust_level: TrustLevel::Primary,
repo: Some("kebab".into()), git_branch: Some("main".into()), user_id_alias: None,
git_commit: Some("0".repeat(40)), code_lang: Some("c".into()), user: Default::default(),
repo: Some("kebab".into()),
git_branch: Some("main".into()),
git_commit: Some("0".repeat(40)),
code_lang: Some("c".into()),
}, },
provenance: Provenance { events: vec![] }, provenance: Provenance { events: vec![] },
parser_version: pv, schema_version: 1, doc_version: 1, parser_version: pv,
last_chunker_version: None, last_embedding_version: None, schema_version: 1,
doc_version: 1,
last_chunker_version: None,
last_embedding_version: None,
} }
} }
fn policy() -> ChunkPolicy { fn policy() -> ChunkPolicy {
ChunkPolicy { target_tokens: 500, overlap_tokens: 80, ChunkPolicy {
target_tokens: 500,
overlap_tokens: 80,
respect_markdown_headings: false, respect_markdown_headings: false,
chunker_version: ChunkerVersion(VERSION_LABEL.into()) } chunker_version: ChunkerVersion(VERSION_LABEL.into()),
}
} }
#[test] #[test]
fn chunker_version_is_code_c_ast_v1() { fn chunker_version_is_code_c_ast_v1() {
assert_eq!(CodeCAstV1Chunker.chunker_version(), assert_eq!(
ChunkerVersion("code-c-ast-v1".into())); CodeCAstV1Chunker.chunker_version(),
ChunkerVersion("code-c-ast-v1".into())
);
} }
#[test] #[test]
@@ -256,7 +282,12 @@ mod tests {
assert_eq!(c.chunker_version.0, "code-c-ast-v1"); assert_eq!(c.chunker_version.0, "code-c-ast-v1");
} }
match &chunks[0].source_spans[0] { match &chunks[0].source_spans[0] {
SourceSpan::Code { symbol, line_start, line_end, .. } => { SourceSpan::Code {
symbol,
line_start,
line_end,
..
} => {
assert_eq!(symbol.as_deref(), Some("parse")); assert_eq!(symbol.as_deref(), Some("parse"));
assert_eq!((*line_start, *line_end), (1, 3)); assert_eq!((*line_start, *line_end), (1, 3));
} }
@@ -266,22 +297,32 @@ mod tests {
#[test] #[test]
fn oversize_unit_splits_into_parts_with_unique_ids() { fn oversize_unit_splits_into_parts_with_unique_ids() {
let body = (0..500).map(|i| format!("\tx{i} = {i};\n")).collect::<String>(); let body = (0..500)
.map(|i| format!("\tx{i} = {i};\n"))
.collect::<String>();
let code = format!("int big() {{\n{body}\n}}"); let code = format!("int big() {{\n{body}\n}}");
let doc = code_doc(&[("big", 1, 502, &code)]); let doc = code_doc(&[("big", 1, 502, &code)]);
let chunks = CodeCAstV1Chunker.chunk(&doc, &policy()).unwrap(); let chunks = CodeCAstV1Chunker.chunk(&doc, &policy()).unwrap();
assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len()); assert!(
chunks.len() >= 2,
"oversize unit must split, got {}",
chunks.len()
);
for c in &chunks { for c in &chunks {
match &c.source_spans[0] { match &c.source_spans[0] {
SourceSpan::Code { symbol, .. } => { SourceSpan::Code { symbol, .. } => {
assert!(symbol.as_deref().unwrap().starts_with("big [part "), assert!(
"part-numbered symbol, got {symbol:?}"); symbol.as_deref().unwrap().starts_with("big [part "),
"part-numbered symbol, got {symbol:?}"
);
} }
_ => unreachable!(), _ => unreachable!(),
} }
} }
let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect(); let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
let n = ids.len(); ids.sort_unstable(); ids.dedup(); let n = ids.len();
ids.sort_unstable();
ids.dedup();
assert_eq!(ids.len(), n, "chunk_ids unique across split parts"); assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
} }
@@ -295,7 +336,8 @@ mod tests {
heading_path: vec![], heading_path: vec![],
source_span: SourceSpan::Line { start: 1, end: 1 }, source_span: SourceSpan::Line { start: 1, end: 1 },
}, },
text: "x".into(), inlines: vec![], text: "x".into(),
inlines: vec![],
})]; })];
let err = CodeCAstV1Chunker.chunk(&doc, &policy()).unwrap_err(); let err = CodeCAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
assert!(err.to_string().contains("CodeCAstV1Chunker")); assert!(err.to_string().contains("CodeCAstV1Chunker"));
@@ -304,11 +346,19 @@ mod tests {
#[test] #[test]
fn deterministic_chunk_ids_1000() { fn deterministic_chunk_ids_1000() {
let doc = code_doc(&[("parse", 1, 2, "int parse() {}\n")]); let doc = code_doc(&[("parse", 1, 2, "int parse() {}\n")]);
let base: Vec<String> = CodeCAstV1Chunker.chunk(&doc, &policy()) let base: Vec<String> = CodeCAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
for _ in 0..1000 { for _ in 0..1000 {
let again: Vec<String> = CodeCAstV1Chunker.chunk(&doc, &policy()) let again: Vec<String> = CodeCAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
assert_eq!(again, base); assert_eq!(again, base);
} }
} }
@@ -316,7 +366,9 @@ mod tests {
#[test] #[test]
fn policy_hash_matches_md_heading_v1() { fn policy_hash_matches_md_heading_v1() {
let p = policy(); let p = policy();
assert_eq!(CodeCAstV1Chunker.policy_hash(&p), assert_eq!(
crate::MdHeadingV1Chunker.policy_hash(&p)); CodeCAstV1Chunker.policy_hash(&p),
crate::MdHeadingV1Chunker.policy_hash(&p)
);
} }
} }

View File

@@ -39,17 +39,13 @@ impl Chunker for CodeCppAstV1Chunker {
hex[..POLICY_HASH_HEX_LEN].to_string() hex[..POLICY_HASH_HEX_LEN].to_string()
} }
fn chunk( fn chunk(&self, doc: &CanonicalDocument, policy: &ChunkPolicy) -> anyhow::Result<Vec<Chunk>> {
&self,
doc: &CanonicalDocument,
policy: &ChunkPolicy,
) -> anyhow::Result<Vec<Chunk>> {
for b in &doc.blocks { for b in &doc.blocks {
let c = match b { let c = match b {
Block::Code(c) => c, Block::Code(c) => c,
_ => anyhow::bail!( _ => {
"CodeCppAstV1Chunker only handles code docs (got non-Code block)" anyhow::bail!("CodeCppAstV1Chunker only handles code docs (got non-Code block)")
), }
}; };
if !matches!(c.common.source_span, SourceSpan::Code { .. }) { if !matches!(c.common.source_span, SourceSpan::Code { .. }) {
anyhow::bail!( anyhow::bail!(
@@ -68,9 +64,12 @@ impl Chunker for CodeCppAstV1Chunker {
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let (ls, le, symbol, lang) = match &cb.common.source_span { let (ls, le, symbol, lang) = match &cb.common.source_span {
SourceSpan::Code { line_start, line_end, symbol, lang } => { SourceSpan::Code {
(*line_start, *line_end, symbol.clone(), lang.clone()) line_start,
} line_end,
symbol,
lang,
} => (*line_start, *line_end, symbol.clone(), lang.clone()),
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()]; let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
@@ -84,8 +83,13 @@ impl Chunker for CodeCppAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
None, span, cb.code.clone(), &chunker_version,
&block_ids,
&base_policy_hash,
None,
span,
cb.code.clone(),
)); ));
} else { } else {
let parts = split_oversize(&cb.code); let parts = split_oversize(&cb.code);
@@ -93,9 +97,7 @@ impl Chunker for CodeCppAstV1Chunker {
for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() { for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
let part_ls = ls + off_start; let part_ls = ls + off_start;
let part_le = ls + off_end; let part_le = ls + off_end;
let part_sym = symbol let part_sym = symbol.as_ref().map(|s| format!("{s} [part {}/{n}]", i + 1));
.as_ref()
.map(|s| format!("{s} [part {}/{n}]", i + 1));
let span = SourceSpan::Code { let span = SourceSpan::Code {
line_start: part_ls, line_start: part_ls,
line_end: part_le, line_end: part_le,
@@ -103,8 +105,13 @@ impl Chunker for CodeCppAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
Some(part_ls), span, text, &chunker_version,
&block_ids,
&base_policy_hash,
Some(part_ls),
span,
text,
)); ));
} }
} }
@@ -183,9 +190,9 @@ fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
mod tests { mod tests {
use super::*; use super::*;
use kebab_core::{ use kebab_core::{
Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
SourceType, TrustLevel, WorkspacePath, WorkspacePath, id_for_block, id_for_doc,
}; };
use time::OffsetDateTime; use time::OffsetDateTime;
@@ -206,39 +213,60 @@ mod tests {
}; };
let bid = id_for_block(&doc_id, "code", &[], i as u32, &span); let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
Block::Code(CodeBlock { Block::Code(CodeBlock {
common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span }, common: CommonBlock {
block_id: bid,
heading_path: vec![],
source_span: span,
},
lang: Some("cpp".into()), lang: Some("cpp".into()),
code: (*code).to_string(), code: (*code).to_string(),
}) })
}) })
.collect(); .collect();
CanonicalDocument { CanonicalDocument {
doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(), doc_id,
lang: Lang("und".into()), blocks, source_asset_id: aid,
workspace_path: wp,
title: "a".into(),
lang: Lang("und".into()),
blocks,
metadata: Metadata { metadata: Metadata {
aliases: vec![], tags: vec![], aliases: vec![],
tags: vec![],
created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
source_type: SourceType::Note, trust_level: TrustLevel::Primary, source_type: SourceType::Note,
user_id_alias: None, user: Default::default(), trust_level: TrustLevel::Primary,
repo: Some("kebab".into()), git_branch: Some("main".into()), user_id_alias: None,
git_commit: Some("0".repeat(40)), code_lang: Some("cpp".into()), user: Default::default(),
repo: Some("kebab".into()),
git_branch: Some("main".into()),
git_commit: Some("0".repeat(40)),
code_lang: Some("cpp".into()),
}, },
provenance: Provenance { events: vec![] }, provenance: Provenance { events: vec![] },
parser_version: pv, schema_version: 1, doc_version: 1, parser_version: pv,
last_chunker_version: None, last_embedding_version: None, schema_version: 1,
doc_version: 1,
last_chunker_version: None,
last_embedding_version: None,
} }
} }
fn policy() -> ChunkPolicy { fn policy() -> ChunkPolicy {
ChunkPolicy { target_tokens: 500, overlap_tokens: 80, ChunkPolicy {
target_tokens: 500,
overlap_tokens: 80,
respect_markdown_headings: false, respect_markdown_headings: false,
chunker_version: ChunkerVersion(VERSION_LABEL.into()) } chunker_version: ChunkerVersion(VERSION_LABEL.into()),
}
} }
#[test] #[test]
fn chunker_version_is_code_cpp_ast_v1() { fn chunker_version_is_code_cpp_ast_v1() {
assert_eq!(CodeCppAstV1Chunker.chunker_version(), assert_eq!(
ChunkerVersion("code-cpp-ast-v1".into())); CodeCppAstV1Chunker.chunker_version(),
ChunkerVersion("code-cpp-ast-v1".into())
);
} }
#[test] #[test]
@@ -256,7 +284,12 @@ mod tests {
assert_eq!(c.chunker_version.0, "code-cpp-ast-v1"); assert_eq!(c.chunker_version.0, "code-cpp-ast-v1");
} }
match &chunks[0].source_spans[0] { match &chunks[0].source_spans[0] {
SourceSpan::Code { symbol, line_start, line_end, .. } => { SourceSpan::Code {
symbol,
line_start,
line_end,
..
} => {
assert_eq!(symbol.as_deref(), Some("parse")); assert_eq!(symbol.as_deref(), Some("parse"));
assert_eq!((*line_start, *line_end), (1, 3)); assert_eq!((*line_start, *line_end), (1, 3));
} }
@@ -266,22 +299,32 @@ mod tests {
#[test] #[test]
fn oversize_unit_splits_into_parts_with_unique_ids() { fn oversize_unit_splits_into_parts_with_unique_ids() {
let body = (0..500).map(|i| format!("\tx{i} = {i};\n")).collect::<String>(); let body = (0..500)
.map(|i| format!("\tx{i} = {i};\n"))
.collect::<String>();
let code = format!("int big() {{\n{body}\n}}"); let code = format!("int big() {{\n{body}\n}}");
let doc = code_doc(&[("big", 1, 502, &code)]); let doc = code_doc(&[("big", 1, 502, &code)]);
let chunks = CodeCppAstV1Chunker.chunk(&doc, &policy()).unwrap(); let chunks = CodeCppAstV1Chunker.chunk(&doc, &policy()).unwrap();
assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len()); assert!(
chunks.len() >= 2,
"oversize unit must split, got {}",
chunks.len()
);
for c in &chunks { for c in &chunks {
match &c.source_spans[0] { match &c.source_spans[0] {
SourceSpan::Code { symbol, .. } => { SourceSpan::Code { symbol, .. } => {
assert!(symbol.as_deref().unwrap().starts_with("big [part "), assert!(
"part-numbered symbol, got {symbol:?}"); symbol.as_deref().unwrap().starts_with("big [part "),
"part-numbered symbol, got {symbol:?}"
);
} }
_ => unreachable!(), _ => unreachable!(),
} }
} }
let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect(); let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
let n = ids.len(); ids.sort_unstable(); ids.dedup(); let n = ids.len();
ids.sort_unstable();
ids.dedup();
assert_eq!(ids.len(), n, "chunk_ids unique across split parts"); assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
} }
@@ -295,7 +338,8 @@ mod tests {
heading_path: vec![], heading_path: vec![],
source_span: SourceSpan::Line { start: 1, end: 1 }, source_span: SourceSpan::Line { start: 1, end: 1 },
}, },
text: "x".into(), inlines: vec![], text: "x".into(),
inlines: vec![],
})]; })];
let err = CodeCppAstV1Chunker.chunk(&doc, &policy()).unwrap_err(); let err = CodeCppAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
assert!(err.to_string().contains("CodeCppAstV1Chunker")); assert!(err.to_string().contains("CodeCppAstV1Chunker"));
@@ -304,11 +348,19 @@ mod tests {
#[test] #[test]
fn deterministic_chunk_ids_1000() { fn deterministic_chunk_ids_1000() {
let doc = code_doc(&[("parse", 1, 2, "int parse() {}\n")]); let doc = code_doc(&[("parse", 1, 2, "int parse() {}\n")]);
let base: Vec<String> = CodeCppAstV1Chunker.chunk(&doc, &policy()) let base: Vec<String> = CodeCppAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
for _ in 0..1000 { for _ in 0..1000 {
let again: Vec<String> = CodeCppAstV1Chunker.chunk(&doc, &policy()) let again: Vec<String> = CodeCppAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
assert_eq!(again, base); assert_eq!(again, base);
} }
} }
@@ -316,7 +368,9 @@ mod tests {
#[test] #[test]
fn policy_hash_matches_md_heading_v1() { fn policy_hash_matches_md_heading_v1() {
let p = policy(); let p = policy();
assert_eq!(CodeCppAstV1Chunker.policy_hash(&p), assert_eq!(
crate::MdHeadingV1Chunker.policy_hash(&p)); CodeCppAstV1Chunker.policy_hash(&p),
crate::MdHeadingV1Chunker.policy_hash(&p)
);
} }
} }

View File

@@ -39,17 +39,13 @@ impl Chunker for CodeGoAstV1Chunker {
hex[..POLICY_HASH_HEX_LEN].to_string() hex[..POLICY_HASH_HEX_LEN].to_string()
} }
fn chunk( fn chunk(&self, doc: &CanonicalDocument, policy: &ChunkPolicy) -> anyhow::Result<Vec<Chunk>> {
&self,
doc: &CanonicalDocument,
policy: &ChunkPolicy,
) -> anyhow::Result<Vec<Chunk>> {
for b in &doc.blocks { for b in &doc.blocks {
let c = match b { let c = match b {
Block::Code(c) => c, Block::Code(c) => c,
_ => anyhow::bail!( _ => {
"CodeGoAstV1Chunker only handles code docs (got non-Code block)" anyhow::bail!("CodeGoAstV1Chunker only handles code docs (got non-Code block)")
), }
}; };
if !matches!(c.common.source_span, SourceSpan::Code { .. }) { if !matches!(c.common.source_span, SourceSpan::Code { .. }) {
anyhow::bail!( anyhow::bail!(
@@ -68,9 +64,12 @@ impl Chunker for CodeGoAstV1Chunker {
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let (ls, le, symbol, lang) = match &cb.common.source_span { let (ls, le, symbol, lang) = match &cb.common.source_span {
SourceSpan::Code { line_start, line_end, symbol, lang } => { SourceSpan::Code {
(*line_start, *line_end, symbol.clone(), lang.clone()) line_start,
} line_end,
symbol,
lang,
} => (*line_start, *line_end, symbol.clone(), lang.clone()),
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()]; let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
@@ -84,8 +83,13 @@ impl Chunker for CodeGoAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
None, span, cb.code.clone(), &chunker_version,
&block_ids,
&base_policy_hash,
None,
span,
cb.code.clone(),
)); ));
} else { } else {
let parts = split_oversize(&cb.code); let parts = split_oversize(&cb.code);
@@ -93,9 +97,7 @@ impl Chunker for CodeGoAstV1Chunker {
for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() { for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
let part_ls = ls + off_start; let part_ls = ls + off_start;
let part_le = ls + off_end; let part_le = ls + off_end;
let part_sym = symbol let part_sym = symbol.as_ref().map(|s| format!("{s} [part {}/{n}]", i + 1));
.as_ref()
.map(|s| format!("{s} [part {}/{n}]", i + 1));
let span = SourceSpan::Code { let span = SourceSpan::Code {
line_start: part_ls, line_start: part_ls,
line_end: part_le, line_end: part_le,
@@ -103,8 +105,13 @@ impl Chunker for CodeGoAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
Some(part_ls), span, text, &chunker_version,
&block_ids,
&base_policy_hash,
Some(part_ls),
span,
text,
)); ));
} }
} }
@@ -183,9 +190,9 @@ fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
mod tests { mod tests {
use super::*; use super::*;
use kebab_core::{ use kebab_core::{
Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
SourceType, TrustLevel, WorkspacePath, WorkspacePath, id_for_block, id_for_doc,
}; };
use time::OffsetDateTime; use time::OffsetDateTime;
@@ -206,46 +213,72 @@ mod tests {
}; };
let bid = id_for_block(&doc_id, "code", &[], i as u32, &span); let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
Block::Code(CodeBlock { Block::Code(CodeBlock {
common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span }, common: CommonBlock {
block_id: bid,
heading_path: vec![],
source_span: span,
},
lang: Some("go".into()), lang: Some("go".into()),
code: (*code).to_string(), code: (*code).to_string(),
}) })
}) })
.collect(); .collect();
CanonicalDocument { CanonicalDocument {
doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(), doc_id,
lang: Lang("und".into()), blocks, source_asset_id: aid,
workspace_path: wp,
title: "a".into(),
lang: Lang("und".into()),
blocks,
metadata: Metadata { metadata: Metadata {
aliases: vec![], tags: vec![], aliases: vec![],
tags: vec![],
created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
source_type: SourceType::Note, trust_level: TrustLevel::Primary, source_type: SourceType::Note,
user_id_alias: None, user: Default::default(), trust_level: TrustLevel::Primary,
repo: Some("kebab".into()), git_branch: Some("main".into()), user_id_alias: None,
git_commit: Some("0".repeat(40)), code_lang: Some("go".into()), user: Default::default(),
repo: Some("kebab".into()),
git_branch: Some("main".into()),
git_commit: Some("0".repeat(40)),
code_lang: Some("go".into()),
}, },
provenance: Provenance { events: vec![] }, provenance: Provenance { events: vec![] },
parser_version: pv, schema_version: 1, doc_version: 1, parser_version: pv,
last_chunker_version: None, last_embedding_version: None, schema_version: 1,
doc_version: 1,
last_chunker_version: None,
last_embedding_version: None,
} }
} }
fn policy() -> ChunkPolicy { fn policy() -> ChunkPolicy {
ChunkPolicy { target_tokens: 500, overlap_tokens: 80, ChunkPolicy {
target_tokens: 500,
overlap_tokens: 80,
respect_markdown_headings: false, respect_markdown_headings: false,
chunker_version: ChunkerVersion(VERSION_LABEL.into()) } chunker_version: ChunkerVersion(VERSION_LABEL.into()),
}
} }
#[test] #[test]
fn chunker_version_is_code_go_ast_v1() { fn chunker_version_is_code_go_ast_v1() {
assert_eq!(CodeGoAstV1Chunker.chunker_version(), assert_eq!(
ChunkerVersion("code-go-ast-v1".into())); CodeGoAstV1Chunker.chunker_version(),
ChunkerVersion("code-go-ast-v1".into())
);
} }
#[test] #[test]
fn one_chunk_per_unit_preserves_code_span() { fn one_chunk_per_unit_preserves_code_span() {
let doc = code_doc(&[ let doc = code_doc(&[
("parse", 1, 3, "func parse() {\n\t// x\n}"), ("parse", 1, 3, "func parse() {\n\t// x\n}"),
("Foo.double", 5, 7, "func double() int {\n\t//\n\treturn 0\n}"), (
"Foo.double",
5,
7,
"func double() int {\n\t//\n\treturn 0\n}",
),
]); ]);
let chunks = CodeGoAstV1Chunker.chunk(&doc, &policy()).unwrap(); let chunks = CodeGoAstV1Chunker.chunk(&doc, &policy()).unwrap();
assert_eq!(chunks.len(), 2); assert_eq!(chunks.len(), 2);
@@ -256,7 +289,12 @@ mod tests {
assert_eq!(c.chunker_version.0, "code-go-ast-v1"); assert_eq!(c.chunker_version.0, "code-go-ast-v1");
} }
match &chunks[0].source_spans[0] { match &chunks[0].source_spans[0] {
SourceSpan::Code { symbol, line_start, line_end, .. } => { SourceSpan::Code {
symbol,
line_start,
line_end,
..
} => {
assert_eq!(symbol.as_deref(), Some("parse")); assert_eq!(symbol.as_deref(), Some("parse"));
assert_eq!((*line_start, *line_end), (1, 3)); assert_eq!((*line_start, *line_end), (1, 3));
} }
@@ -266,22 +304,33 @@ mod tests {
#[test] #[test]
fn oversize_unit_splits_into_parts_with_unique_ids() { fn oversize_unit_splits_into_parts_with_unique_ids() {
let body = (0..500).map(|i| format!("\tx{i} := {i}")).collect::<Vec<_>>().join("\n"); let body = (0..500)
.map(|i| format!("\tx{i} := {i}"))
.collect::<Vec<_>>()
.join("\n");
let code = format!("func big() {{\n{body}\n}}"); let code = format!("func big() {{\n{body}\n}}");
let doc = code_doc(&[("big", 1, 502, &code)]); let doc = code_doc(&[("big", 1, 502, &code)]);
let chunks = CodeGoAstV1Chunker.chunk(&doc, &policy()).unwrap(); let chunks = CodeGoAstV1Chunker.chunk(&doc, &policy()).unwrap();
assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len()); assert!(
chunks.len() >= 2,
"oversize unit must split, got {}",
chunks.len()
);
for c in &chunks { for c in &chunks {
match &c.source_spans[0] { match &c.source_spans[0] {
SourceSpan::Code { symbol, .. } => { SourceSpan::Code { symbol, .. } => {
assert!(symbol.as_deref().unwrap().starts_with("big [part "), assert!(
"part-numbered symbol, got {symbol:?}"); symbol.as_deref().unwrap().starts_with("big [part "),
"part-numbered symbol, got {symbol:?}"
);
} }
_ => unreachable!(), _ => unreachable!(),
} }
} }
let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect(); let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
let n = ids.len(); ids.sort_unstable(); ids.dedup(); let n = ids.len();
ids.sort_unstable();
ids.dedup();
assert_eq!(ids.len(), n, "chunk_ids unique across split parts"); assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
} }
@@ -295,7 +344,8 @@ mod tests {
heading_path: vec![], heading_path: vec![],
source_span: SourceSpan::Line { start: 1, end: 1 }, source_span: SourceSpan::Line { start: 1, end: 1 },
}, },
text: "x".into(), inlines: vec![], text: "x".into(),
inlines: vec![],
})]; })];
let err = CodeGoAstV1Chunker.chunk(&doc, &policy()).unwrap_err(); let err = CodeGoAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
assert!(err.to_string().contains("CodeGoAstV1Chunker")); assert!(err.to_string().contains("CodeGoAstV1Chunker"));
@@ -304,11 +354,19 @@ mod tests {
#[test] #[test]
fn deterministic_chunk_ids_1000() { fn deterministic_chunk_ids_1000() {
let doc = code_doc(&[("parse", 1, 2, "func parse() {}\n")]); let doc = code_doc(&[("parse", 1, 2, "func parse() {}\n")]);
let base: Vec<String> = CodeGoAstV1Chunker.chunk(&doc, &policy()) let base: Vec<String> = CodeGoAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
for _ in 0..1000 { for _ in 0..1000 {
let again: Vec<String> = CodeGoAstV1Chunker.chunk(&doc, &policy()) let again: Vec<String> = CodeGoAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
assert_eq!(again, base); assert_eq!(again, base);
} }
} }
@@ -316,7 +374,9 @@ mod tests {
#[test] #[test]
fn policy_hash_matches_md_heading_v1() { fn policy_hash_matches_md_heading_v1() {
let p = policy(); let p = policy();
assert_eq!(CodeGoAstV1Chunker.policy_hash(&p), assert_eq!(
crate::MdHeadingV1Chunker.policy_hash(&p)); CodeGoAstV1Chunker.policy_hash(&p),
crate::MdHeadingV1Chunker.policy_hash(&p)
);
} }
} }

View File

@@ -39,11 +39,7 @@ impl Chunker for CodeJavaAstV1Chunker {
hex[..POLICY_HASH_HEX_LEN].to_string() hex[..POLICY_HASH_HEX_LEN].to_string()
} }
fn chunk( fn chunk(&self, doc: &CanonicalDocument, policy: &ChunkPolicy) -> anyhow::Result<Vec<Chunk>> {
&self,
doc: &CanonicalDocument,
policy: &ChunkPolicy,
) -> anyhow::Result<Vec<Chunk>> {
for b in &doc.blocks { for b in &doc.blocks {
let c = match b { let c = match b {
Block::Code(c) => c, Block::Code(c) => c,
@@ -68,9 +64,12 @@ impl Chunker for CodeJavaAstV1Chunker {
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let (ls, le, symbol, lang) = match &cb.common.source_span { let (ls, le, symbol, lang) = match &cb.common.source_span {
SourceSpan::Code { line_start, line_end, symbol, lang } => { SourceSpan::Code {
(*line_start, *line_end, symbol.clone(), lang.clone()) line_start,
} line_end,
symbol,
lang,
} => (*line_start, *line_end, symbol.clone(), lang.clone()),
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()]; let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
@@ -84,8 +83,13 @@ impl Chunker for CodeJavaAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
None, span, cb.code.clone(), &chunker_version,
&block_ids,
&base_policy_hash,
None,
span,
cb.code.clone(),
)); ));
} else { } else {
let parts = split_oversize(&cb.code); let parts = split_oversize(&cb.code);
@@ -93,9 +97,7 @@ impl Chunker for CodeJavaAstV1Chunker {
for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() { for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
let part_ls = ls + off_start; let part_ls = ls + off_start;
let part_le = ls + off_end; let part_le = ls + off_end;
let part_sym = symbol let part_sym = symbol.as_ref().map(|s| format!("{s} [part {}/{n}]", i + 1));
.as_ref()
.map(|s| format!("{s} [part {}/{n}]", i + 1));
let span = SourceSpan::Code { let span = SourceSpan::Code {
line_start: part_ls, line_start: part_ls,
line_end: part_le, line_end: part_le,
@@ -103,8 +105,13 @@ impl Chunker for CodeJavaAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
Some(part_ls), span, text, &chunker_version,
&block_ids,
&base_policy_hash,
Some(part_ls),
span,
text,
)); ));
} }
} }
@@ -183,9 +190,9 @@ fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
mod tests { mod tests {
use super::*; use super::*;
use kebab_core::{ use kebab_core::{
Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
SourceType, TrustLevel, WorkspacePath, WorkspacePath, id_for_block, id_for_doc,
}; };
use time::OffsetDateTime; use time::OffsetDateTime;
@@ -206,39 +213,60 @@ mod tests {
}; };
let bid = id_for_block(&doc_id, "code", &[], i as u32, &span); let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
Block::Code(CodeBlock { Block::Code(CodeBlock {
common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span }, common: CommonBlock {
block_id: bid,
heading_path: vec![],
source_span: span,
},
lang: Some("java".into()), lang: Some("java".into()),
code: (*code).to_string(), code: (*code).to_string(),
}) })
}) })
.collect(); .collect();
CanonicalDocument { CanonicalDocument {
doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(), doc_id,
lang: Lang("und".into()), blocks, source_asset_id: aid,
workspace_path: wp,
title: "a".into(),
lang: Lang("und".into()),
blocks,
metadata: Metadata { metadata: Metadata {
aliases: vec![], tags: vec![], aliases: vec![],
tags: vec![],
created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
source_type: SourceType::Note, trust_level: TrustLevel::Primary, source_type: SourceType::Note,
user_id_alias: None, user: Default::default(), trust_level: TrustLevel::Primary,
repo: Some("kebab".into()), git_branch: Some("main".into()), user_id_alias: None,
git_commit: Some("0".repeat(40)), code_lang: Some("java".into()), user: Default::default(),
repo: Some("kebab".into()),
git_branch: Some("main".into()),
git_commit: Some("0".repeat(40)),
code_lang: Some("java".into()),
}, },
provenance: Provenance { events: vec![] }, provenance: Provenance { events: vec![] },
parser_version: pv, schema_version: 1, doc_version: 1, parser_version: pv,
last_chunker_version: None, last_embedding_version: None, schema_version: 1,
doc_version: 1,
last_chunker_version: None,
last_embedding_version: None,
} }
} }
fn policy() -> ChunkPolicy { fn policy() -> ChunkPolicy {
ChunkPolicy { target_tokens: 500, overlap_tokens: 80, ChunkPolicy {
target_tokens: 500,
overlap_tokens: 80,
respect_markdown_headings: false, respect_markdown_headings: false,
chunker_version: ChunkerVersion(VERSION_LABEL.into()) } chunker_version: ChunkerVersion(VERSION_LABEL.into()),
}
} }
#[test] #[test]
fn chunker_version_is_code_java_ast_v1() { fn chunker_version_is_code_java_ast_v1() {
assert_eq!(CodeJavaAstV1Chunker.chunker_version(), assert_eq!(
ChunkerVersion("code-java-ast-v1".into())); CodeJavaAstV1Chunker.chunker_version(),
ChunkerVersion("code-java-ast-v1".into())
);
} }
#[test] #[test]
@@ -256,7 +284,12 @@ mod tests {
assert_eq!(c.chunker_version.0, "code-java-ast-v1"); assert_eq!(c.chunker_version.0, "code-java-ast-v1");
} }
match &chunks[0].source_spans[0] { match &chunks[0].source_spans[0] {
SourceSpan::Code { symbol, line_start, line_end, .. } => { SourceSpan::Code {
symbol,
line_start,
line_end,
..
} => {
assert_eq!(symbol.as_deref(), Some("parse")); assert_eq!(symbol.as_deref(), Some("parse"));
assert_eq!((*line_start, *line_end), (1, 3)); assert_eq!((*line_start, *line_end), (1, 3));
} }
@@ -266,22 +299,33 @@ mod tests {
#[test] #[test]
fn oversize_unit_splits_into_parts_with_unique_ids() { fn oversize_unit_splits_into_parts_with_unique_ids() {
let body = (0..500).map(|i| format!("\tint x{i} = {i};")).collect::<Vec<_>>().join("\n"); let body = (0..500)
.map(|i| format!("\tint x{i} = {i};"))
.collect::<Vec<_>>()
.join("\n");
let code = format!("void big() {{\n{body}\n}}"); let code = format!("void big() {{\n{body}\n}}");
let doc = code_doc(&[("big", 1, 502, &code)]); let doc = code_doc(&[("big", 1, 502, &code)]);
let chunks = CodeJavaAstV1Chunker.chunk(&doc, &policy()).unwrap(); let chunks = CodeJavaAstV1Chunker.chunk(&doc, &policy()).unwrap();
assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len()); assert!(
chunks.len() >= 2,
"oversize unit must split, got {}",
chunks.len()
);
for c in &chunks { for c in &chunks {
match &c.source_spans[0] { match &c.source_spans[0] {
SourceSpan::Code { symbol, .. } => { SourceSpan::Code { symbol, .. } => {
assert!(symbol.as_deref().unwrap().starts_with("big [part "), assert!(
"part-numbered symbol, got {symbol:?}"); symbol.as_deref().unwrap().starts_with("big [part "),
"part-numbered symbol, got {symbol:?}"
);
} }
_ => unreachable!(), _ => unreachable!(),
} }
} }
let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect(); let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
let n = ids.len(); ids.sort_unstable(); ids.dedup(); let n = ids.len();
ids.sort_unstable();
ids.dedup();
assert_eq!(ids.len(), n, "chunk_ids unique across split parts"); assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
} }
@@ -295,7 +339,8 @@ mod tests {
heading_path: vec![], heading_path: vec![],
source_span: SourceSpan::Line { start: 1, end: 1 }, source_span: SourceSpan::Line { start: 1, end: 1 },
}, },
text: "x".into(), inlines: vec![], text: "x".into(),
inlines: vec![],
})]; })];
let err = CodeJavaAstV1Chunker.chunk(&doc, &policy()).unwrap_err(); let err = CodeJavaAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
assert!(err.to_string().contains("CodeJavaAstV1Chunker")); assert!(err.to_string().contains("CodeJavaAstV1Chunker"));
@@ -304,11 +349,19 @@ mod tests {
#[test] #[test]
fn deterministic_chunk_ids_1000() { fn deterministic_chunk_ids_1000() {
let doc = code_doc(&[("parse", 1, 2, "void parse() {}\n")]); let doc = code_doc(&[("parse", 1, 2, "void parse() {}\n")]);
let base: Vec<String> = CodeJavaAstV1Chunker.chunk(&doc, &policy()) let base: Vec<String> = CodeJavaAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
for _ in 0..1000 { for _ in 0..1000 {
let again: Vec<String> = CodeJavaAstV1Chunker.chunk(&doc, &policy()) let again: Vec<String> = CodeJavaAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
assert_eq!(again, base); assert_eq!(again, base);
} }
} }
@@ -316,7 +369,9 @@ mod tests {
#[test] #[test]
fn policy_hash_matches_md_heading_v1() { fn policy_hash_matches_md_heading_v1() {
let p = policy(); let p = policy();
assert_eq!(CodeJavaAstV1Chunker.policy_hash(&p), assert_eq!(
crate::MdHeadingV1Chunker.policy_hash(&p)); CodeJavaAstV1Chunker.policy_hash(&p),
crate::MdHeadingV1Chunker.policy_hash(&p)
);
} }
} }

View File

@@ -39,17 +39,13 @@ impl Chunker for CodeJsAstV1Chunker {
hex[..POLICY_HASH_HEX_LEN].to_string() hex[..POLICY_HASH_HEX_LEN].to_string()
} }
fn chunk( fn chunk(&self, doc: &CanonicalDocument, policy: &ChunkPolicy) -> anyhow::Result<Vec<Chunk>> {
&self,
doc: &CanonicalDocument,
policy: &ChunkPolicy,
) -> anyhow::Result<Vec<Chunk>> {
for b in &doc.blocks { for b in &doc.blocks {
let c = match b { let c = match b {
Block::Code(c) => c, Block::Code(c) => c,
_ => anyhow::bail!( _ => {
"CodeJsAstV1Chunker only handles code docs (got non-Code block)" anyhow::bail!("CodeJsAstV1Chunker only handles code docs (got non-Code block)")
), }
}; };
if !matches!(c.common.source_span, SourceSpan::Code { .. }) { if !matches!(c.common.source_span, SourceSpan::Code { .. }) {
anyhow::bail!( anyhow::bail!(
@@ -68,9 +64,12 @@ impl Chunker for CodeJsAstV1Chunker {
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let (ls, le, symbol, lang) = match &cb.common.source_span { let (ls, le, symbol, lang) = match &cb.common.source_span {
SourceSpan::Code { line_start, line_end, symbol, lang } => { SourceSpan::Code {
(*line_start, *line_end, symbol.clone(), lang.clone()) line_start,
} line_end,
symbol,
lang,
} => (*line_start, *line_end, symbol.clone(), lang.clone()),
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()]; let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
@@ -84,8 +83,13 @@ impl Chunker for CodeJsAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
None, span, cb.code.clone(), &chunker_version,
&block_ids,
&base_policy_hash,
None,
span,
cb.code.clone(),
)); ));
} else { } else {
let parts = split_oversize(&cb.code); let parts = split_oversize(&cb.code);
@@ -93,9 +97,7 @@ impl Chunker for CodeJsAstV1Chunker {
for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() { for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
let part_ls = ls + off_start; let part_ls = ls + off_start;
let part_le = ls + off_end; let part_le = ls + off_end;
let part_sym = symbol let part_sym = symbol.as_ref().map(|s| format!("{s} [part {}/{n}]", i + 1));
.as_ref()
.map(|s| format!("{s} [part {}/{n}]", i + 1));
let span = SourceSpan::Code { let span = SourceSpan::Code {
line_start: part_ls, line_start: part_ls,
line_end: part_le, line_end: part_le,
@@ -103,8 +105,13 @@ impl Chunker for CodeJsAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
Some(part_ls), span, text, &chunker_version,
&block_ids,
&base_policy_hash,
Some(part_ls),
span,
text,
)); ));
} }
} }
@@ -183,9 +190,9 @@ fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
mod tests { mod tests {
use super::*; use super::*;
use kebab_core::{ use kebab_core::{
Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
SourceType, TrustLevel, WorkspacePath, WorkspacePath, id_for_block, id_for_doc,
}; };
use time::OffsetDateTime; use time::OffsetDateTime;
@@ -206,46 +213,72 @@ mod tests {
}; };
let bid = id_for_block(&doc_id, "code", &[], i as u32, &span); let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
Block::Code(CodeBlock { Block::Code(CodeBlock {
common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span }, common: CommonBlock {
block_id: bid,
heading_path: vec![],
source_span: span,
},
lang: Some("javascript".into()), lang: Some("javascript".into()),
code: (*code).to_string(), code: (*code).to_string(),
}) })
}) })
.collect(); .collect();
CanonicalDocument { CanonicalDocument {
doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(), doc_id,
lang: Lang("und".into()), blocks, source_asset_id: aid,
workspace_path: wp,
title: "a".into(),
lang: Lang("und".into()),
blocks,
metadata: Metadata { metadata: Metadata {
aliases: vec![], tags: vec![], aliases: vec![],
tags: vec![],
created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
source_type: SourceType::Note, trust_level: TrustLevel::Primary, source_type: SourceType::Note,
user_id_alias: None, user: Default::default(), trust_level: TrustLevel::Primary,
repo: Some("kebab".into()), git_branch: Some("main".into()), user_id_alias: None,
git_commit: Some("0".repeat(40)), code_lang: Some("javascript".into()), user: Default::default(),
repo: Some("kebab".into()),
git_branch: Some("main".into()),
git_commit: Some("0".repeat(40)),
code_lang: Some("javascript".into()),
}, },
provenance: Provenance { events: vec![] }, provenance: Provenance { events: vec![] },
parser_version: pv, schema_version: 1, doc_version: 1, parser_version: pv,
last_chunker_version: None, last_embedding_version: None, schema_version: 1,
doc_version: 1,
last_chunker_version: None,
last_embedding_version: None,
} }
} }
fn policy() -> ChunkPolicy { fn policy() -> ChunkPolicy {
ChunkPolicy { target_tokens: 500, overlap_tokens: 80, ChunkPolicy {
target_tokens: 500,
overlap_tokens: 80,
respect_markdown_headings: false, respect_markdown_headings: false,
chunker_version: ChunkerVersion(VERSION_LABEL.into()) } chunker_version: ChunkerVersion(VERSION_LABEL.into()),
}
} }
#[test] #[test]
fn chunker_version_is_code_js_ast_v1() { fn chunker_version_is_code_js_ast_v1() {
assert_eq!(CodeJsAstV1Chunker.chunker_version(), assert_eq!(
ChunkerVersion("code-js-ast-v1".into())); CodeJsAstV1Chunker.chunker_version(),
ChunkerVersion("code-js-ast-v1".into())
);
} }
#[test] #[test]
fn one_chunk_per_unit_preserves_code_span() { fn one_chunk_per_unit_preserves_code_span() {
let doc = code_doc(&[ let doc = code_doc(&[
("parse", 1, 3, "function parse() {\n // x\n}"), ("parse", 1, 3, "function parse() {\n // x\n}"),
("Foo.double", 5, 7, "function double() {\n //\n return 0;\n}"), (
"Foo.double",
5,
7,
"function double() {\n //\n return 0;\n}",
),
]); ]);
let chunks = CodeJsAstV1Chunker.chunk(&doc, &policy()).unwrap(); let chunks = CodeJsAstV1Chunker.chunk(&doc, &policy()).unwrap();
assert_eq!(chunks.len(), 2); assert_eq!(chunks.len(), 2);
@@ -256,7 +289,12 @@ mod tests {
assert_eq!(c.chunker_version.0, "code-js-ast-v1"); assert_eq!(c.chunker_version.0, "code-js-ast-v1");
} }
match &chunks[0].source_spans[0] { match &chunks[0].source_spans[0] {
SourceSpan::Code { symbol, line_start, line_end, .. } => { SourceSpan::Code {
symbol,
line_start,
line_end,
..
} => {
assert_eq!(symbol.as_deref(), Some("parse")); assert_eq!(symbol.as_deref(), Some("parse"));
assert_eq!((*line_start, *line_end), (1, 3)); assert_eq!((*line_start, *line_end), (1, 3));
} }
@@ -266,22 +304,33 @@ mod tests {
#[test] #[test]
fn oversize_unit_splits_into_parts_with_unique_ids() { fn oversize_unit_splits_into_parts_with_unique_ids() {
let body = (0..500).map(|i| format!(" const x{i} = {i};")).collect::<Vec<_>>().join("\n"); let body = (0..500)
.map(|i| format!(" const x{i} = {i};"))
.collect::<Vec<_>>()
.join("\n");
let code = format!("function big() {{\n{body}\n}}"); let code = format!("function big() {{\n{body}\n}}");
let doc = code_doc(&[("big", 1, 502, &code)]); let doc = code_doc(&[("big", 1, 502, &code)]);
let chunks = CodeJsAstV1Chunker.chunk(&doc, &policy()).unwrap(); let chunks = CodeJsAstV1Chunker.chunk(&doc, &policy()).unwrap();
assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len()); assert!(
chunks.len() >= 2,
"oversize unit must split, got {}",
chunks.len()
);
for c in &chunks { for c in &chunks {
match &c.source_spans[0] { match &c.source_spans[0] {
SourceSpan::Code { symbol, .. } => { SourceSpan::Code { symbol, .. } => {
assert!(symbol.as_deref().unwrap().starts_with("big [part "), assert!(
"part-numbered symbol, got {symbol:?}"); symbol.as_deref().unwrap().starts_with("big [part "),
"part-numbered symbol, got {symbol:?}"
);
} }
_ => unreachable!(), _ => unreachable!(),
} }
} }
let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect(); let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
let n = ids.len(); ids.sort_unstable(); ids.dedup(); let n = ids.len();
ids.sort_unstable();
ids.dedup();
assert_eq!(ids.len(), n, "chunk_ids unique across split parts"); assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
} }
@@ -295,7 +344,8 @@ mod tests {
heading_path: vec![], heading_path: vec![],
source_span: SourceSpan::Line { start: 1, end: 1 }, source_span: SourceSpan::Line { start: 1, end: 1 },
}, },
text: "x".into(), inlines: vec![], text: "x".into(),
inlines: vec![],
})]; })];
let err = CodeJsAstV1Chunker.chunk(&doc, &policy()).unwrap_err(); let err = CodeJsAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
assert!(err.to_string().contains("CodeJsAstV1Chunker")); assert!(err.to_string().contains("CodeJsAstV1Chunker"));
@@ -304,11 +354,19 @@ mod tests {
#[test] #[test]
fn deterministic_chunk_ids_1000() { fn deterministic_chunk_ids_1000() {
let doc = code_doc(&[("parse", 1, 2, "function parse() {}\n")]); let doc = code_doc(&[("parse", 1, 2, "function parse() {}\n")]);
let base: Vec<String> = CodeJsAstV1Chunker.chunk(&doc, &policy()) let base: Vec<String> = CodeJsAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
for _ in 0..1000 { for _ in 0..1000 {
let again: Vec<String> = CodeJsAstV1Chunker.chunk(&doc, &policy()) let again: Vec<String> = CodeJsAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
assert_eq!(again, base); assert_eq!(again, base);
} }
} }
@@ -316,7 +374,9 @@ mod tests {
#[test] #[test]
fn policy_hash_matches_md_heading_v1() { fn policy_hash_matches_md_heading_v1() {
let p = policy(); let p = policy();
assert_eq!(CodeJsAstV1Chunker.policy_hash(&p), assert_eq!(
crate::MdHeadingV1Chunker.policy_hash(&p)); CodeJsAstV1Chunker.policy_hash(&p),
crate::MdHeadingV1Chunker.policy_hash(&p)
);
} }
} }

View File

@@ -39,11 +39,7 @@ impl Chunker for CodeKotlinAstV1Chunker {
hex[..POLICY_HASH_HEX_LEN].to_string() hex[..POLICY_HASH_HEX_LEN].to_string()
} }
fn chunk( fn chunk(&self, doc: &CanonicalDocument, policy: &ChunkPolicy) -> anyhow::Result<Vec<Chunk>> {
&self,
doc: &CanonicalDocument,
policy: &ChunkPolicy,
) -> anyhow::Result<Vec<Chunk>> {
for b in &doc.blocks { for b in &doc.blocks {
let c = match b { let c = match b {
Block::Code(c) => c, Block::Code(c) => c,
@@ -68,9 +64,12 @@ impl Chunker for CodeKotlinAstV1Chunker {
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let (ls, le, symbol, lang) = match &cb.common.source_span { let (ls, le, symbol, lang) = match &cb.common.source_span {
SourceSpan::Code { line_start, line_end, symbol, lang } => { SourceSpan::Code {
(*line_start, *line_end, symbol.clone(), lang.clone()) line_start,
} line_end,
symbol,
lang,
} => (*line_start, *line_end, symbol.clone(), lang.clone()),
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()]; let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
@@ -84,8 +83,13 @@ impl Chunker for CodeKotlinAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
None, span, cb.code.clone(), &chunker_version,
&block_ids,
&base_policy_hash,
None,
span,
cb.code.clone(),
)); ));
} else { } else {
let parts = split_oversize(&cb.code); let parts = split_oversize(&cb.code);
@@ -93,9 +97,7 @@ impl Chunker for CodeKotlinAstV1Chunker {
for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() { for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
let part_ls = ls + off_start; let part_ls = ls + off_start;
let part_le = ls + off_end; let part_le = ls + off_end;
let part_sym = symbol let part_sym = symbol.as_ref().map(|s| format!("{s} [part {}/{n}]", i + 1));
.as_ref()
.map(|s| format!("{s} [part {}/{n}]", i + 1));
let span = SourceSpan::Code { let span = SourceSpan::Code {
line_start: part_ls, line_start: part_ls,
line_end: part_le, line_end: part_le,
@@ -103,8 +105,13 @@ impl Chunker for CodeKotlinAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
Some(part_ls), span, text, &chunker_version,
&block_ids,
&base_policy_hash,
Some(part_ls),
span,
text,
)); ));
} }
} }
@@ -183,9 +190,9 @@ fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
mod tests { mod tests {
use super::*; use super::*;
use kebab_core::{ use kebab_core::{
Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
SourceType, TrustLevel, WorkspacePath, WorkspacePath, id_for_block, id_for_doc,
}; };
use time::OffsetDateTime; use time::OffsetDateTime;
@@ -206,46 +213,72 @@ mod tests {
}; };
let bid = id_for_block(&doc_id, "code", &[], i as u32, &span); let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
Block::Code(CodeBlock { Block::Code(CodeBlock {
common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span }, common: CommonBlock {
block_id: bid,
heading_path: vec![],
source_span: span,
},
lang: Some("kotlin".into()), lang: Some("kotlin".into()),
code: (*code).to_string(), code: (*code).to_string(),
}) })
}) })
.collect(); .collect();
CanonicalDocument { CanonicalDocument {
doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(), doc_id,
lang: Lang("und".into()), blocks, source_asset_id: aid,
workspace_path: wp,
title: "a".into(),
lang: Lang("und".into()),
blocks,
metadata: Metadata { metadata: Metadata {
aliases: vec![], tags: vec![], aliases: vec![],
tags: vec![],
created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
source_type: SourceType::Note, trust_level: TrustLevel::Primary, source_type: SourceType::Note,
user_id_alias: None, user: Default::default(), trust_level: TrustLevel::Primary,
repo: Some("kebab".into()), git_branch: Some("main".into()), user_id_alias: None,
git_commit: Some("0".repeat(40)), code_lang: Some("kotlin".into()), user: Default::default(),
repo: Some("kebab".into()),
git_branch: Some("main".into()),
git_commit: Some("0".repeat(40)),
code_lang: Some("kotlin".into()),
}, },
provenance: Provenance { events: vec![] }, provenance: Provenance { events: vec![] },
parser_version: pv, schema_version: 1, doc_version: 1, parser_version: pv,
last_chunker_version: None, last_embedding_version: None, schema_version: 1,
doc_version: 1,
last_chunker_version: None,
last_embedding_version: None,
} }
} }
fn policy() -> ChunkPolicy { fn policy() -> ChunkPolicy {
ChunkPolicy { target_tokens: 500, overlap_tokens: 80, ChunkPolicy {
target_tokens: 500,
overlap_tokens: 80,
respect_markdown_headings: false, respect_markdown_headings: false,
chunker_version: ChunkerVersion(VERSION_LABEL.into()) } chunker_version: ChunkerVersion(VERSION_LABEL.into()),
}
} }
#[test] #[test]
fn chunker_version_is_code_kotlin_ast_v1() { fn chunker_version_is_code_kotlin_ast_v1() {
assert_eq!(CodeKotlinAstV1Chunker.chunker_version(), assert_eq!(
ChunkerVersion("code-kotlin-ast-v1".into())); CodeKotlinAstV1Chunker.chunker_version(),
ChunkerVersion("code-kotlin-ast-v1".into())
);
} }
#[test] #[test]
fn one_chunk_per_unit_preserves_code_span() { fn one_chunk_per_unit_preserves_code_span() {
let doc = code_doc(&[ let doc = code_doc(&[
("parse", 1, 3, "fun parse() {\n\t// x\n}"), ("parse", 1, 3, "fun parse() {\n\t// x\n}"),
("Foo.double", 5, 7, "fun double(): Int {\n\t//\n\treturn 0\n}"), (
"Foo.double",
5,
7,
"fun double(): Int {\n\t//\n\treturn 0\n}",
),
]); ]);
let chunks = CodeKotlinAstV1Chunker.chunk(&doc, &policy()).unwrap(); let chunks = CodeKotlinAstV1Chunker.chunk(&doc, &policy()).unwrap();
assert_eq!(chunks.len(), 2); assert_eq!(chunks.len(), 2);
@@ -256,7 +289,12 @@ mod tests {
assert_eq!(c.chunker_version.0, "code-kotlin-ast-v1"); assert_eq!(c.chunker_version.0, "code-kotlin-ast-v1");
} }
match &chunks[0].source_spans[0] { match &chunks[0].source_spans[0] {
SourceSpan::Code { symbol, line_start, line_end, .. } => { SourceSpan::Code {
symbol,
line_start,
line_end,
..
} => {
assert_eq!(symbol.as_deref(), Some("parse")); assert_eq!(symbol.as_deref(), Some("parse"));
assert_eq!((*line_start, *line_end), (1, 3)); assert_eq!((*line_start, *line_end), (1, 3));
} }
@@ -266,22 +304,33 @@ mod tests {
#[test] #[test]
fn oversize_unit_splits_into_parts_with_unique_ids() { fn oversize_unit_splits_into_parts_with_unique_ids() {
let body = (0..500).map(|i| format!("\tval x{i} = {i}")).collect::<Vec<_>>().join("\n"); let body = (0..500)
.map(|i| format!("\tval x{i} = {i}"))
.collect::<Vec<_>>()
.join("\n");
let code = format!("fun big() {{\n{body}\n}}"); let code = format!("fun big() {{\n{body}\n}}");
let doc = code_doc(&[("big", 1, 502, &code)]); let doc = code_doc(&[("big", 1, 502, &code)]);
let chunks = CodeKotlinAstV1Chunker.chunk(&doc, &policy()).unwrap(); let chunks = CodeKotlinAstV1Chunker.chunk(&doc, &policy()).unwrap();
assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len()); assert!(
chunks.len() >= 2,
"oversize unit must split, got {}",
chunks.len()
);
for c in &chunks { for c in &chunks {
match &c.source_spans[0] { match &c.source_spans[0] {
SourceSpan::Code { symbol, .. } => { SourceSpan::Code { symbol, .. } => {
assert!(symbol.as_deref().unwrap().starts_with("big [part "), assert!(
"part-numbered symbol, got {symbol:?}"); symbol.as_deref().unwrap().starts_with("big [part "),
"part-numbered symbol, got {symbol:?}"
);
} }
_ => unreachable!(), _ => unreachable!(),
} }
} }
let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect(); let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
let n = ids.len(); ids.sort_unstable(); ids.dedup(); let n = ids.len();
ids.sort_unstable();
ids.dedup();
assert_eq!(ids.len(), n, "chunk_ids unique across split parts"); assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
} }
@@ -295,7 +344,8 @@ mod tests {
heading_path: vec![], heading_path: vec![],
source_span: SourceSpan::Line { start: 1, end: 1 }, source_span: SourceSpan::Line { start: 1, end: 1 },
}, },
text: "x".into(), inlines: vec![], text: "x".into(),
inlines: vec![],
})]; })];
let err = CodeKotlinAstV1Chunker.chunk(&doc, &policy()).unwrap_err(); let err = CodeKotlinAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
assert!(err.to_string().contains("CodeKotlinAstV1Chunker")); assert!(err.to_string().contains("CodeKotlinAstV1Chunker"));
@@ -304,11 +354,19 @@ mod tests {
#[test] #[test]
fn deterministic_chunk_ids_1000() { fn deterministic_chunk_ids_1000() {
let doc = code_doc(&[("parse", 1, 2, "fun parse() {}\n")]); let doc = code_doc(&[("parse", 1, 2, "fun parse() {}\n")]);
let base: Vec<String> = CodeKotlinAstV1Chunker.chunk(&doc, &policy()) let base: Vec<String> = CodeKotlinAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
for _ in 0..1000 { for _ in 0..1000 {
let again: Vec<String> = CodeKotlinAstV1Chunker.chunk(&doc, &policy()) let again: Vec<String> = CodeKotlinAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
assert_eq!(again, base); assert_eq!(again, base);
} }
} }
@@ -316,7 +374,9 @@ mod tests {
#[test] #[test]
fn policy_hash_matches_md_heading_v1() { fn policy_hash_matches_md_heading_v1() {
let p = policy(); let p = policy();
assert_eq!(CodeKotlinAstV1Chunker.policy_hash(&p), assert_eq!(
crate::MdHeadingV1Chunker.policy_hash(&p)); CodeKotlinAstV1Chunker.policy_hash(&p),
crate::MdHeadingV1Chunker.policy_hash(&p)
);
} }
} }

View File

@@ -39,11 +39,7 @@ impl Chunker for CodePythonAstV1Chunker {
hex[..POLICY_HASH_HEX_LEN].to_string() hex[..POLICY_HASH_HEX_LEN].to_string()
} }
fn chunk( fn chunk(&self, doc: &CanonicalDocument, policy: &ChunkPolicy) -> anyhow::Result<Vec<Chunk>> {
&self,
doc: &CanonicalDocument,
policy: &ChunkPolicy,
) -> anyhow::Result<Vec<Chunk>> {
for b in &doc.blocks { for b in &doc.blocks {
let c = match b { let c = match b {
Block::Code(c) => c, Block::Code(c) => c,
@@ -68,9 +64,12 @@ impl Chunker for CodePythonAstV1Chunker {
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let (ls, le, symbol, lang) = match &cb.common.source_span { let (ls, le, symbol, lang) = match &cb.common.source_span {
SourceSpan::Code { line_start, line_end, symbol, lang } => { SourceSpan::Code {
(*line_start, *line_end, symbol.clone(), lang.clone()) line_start,
} line_end,
symbol,
lang,
} => (*line_start, *line_end, symbol.clone(), lang.clone()),
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()]; let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
@@ -84,8 +83,13 @@ impl Chunker for CodePythonAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
None, span, cb.code.clone(), &chunker_version,
&block_ids,
&base_policy_hash,
None,
span,
cb.code.clone(),
)); ));
} else { } else {
let parts = split_oversize(&cb.code); let parts = split_oversize(&cb.code);
@@ -93,9 +97,7 @@ impl Chunker for CodePythonAstV1Chunker {
for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() { for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
let part_ls = ls + off_start; let part_ls = ls + off_start;
let part_le = ls + off_end; let part_le = ls + off_end;
let part_sym = symbol let part_sym = symbol.as_ref().map(|s| format!("{s} [part {}/{n}]", i + 1));
.as_ref()
.map(|s| format!("{s} [part {}/{n}]", i + 1));
let span = SourceSpan::Code { let span = SourceSpan::Code {
line_start: part_ls, line_start: part_ls,
line_end: part_le, line_end: part_le,
@@ -103,8 +105,13 @@ impl Chunker for CodePythonAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
Some(part_ls), span, text, &chunker_version,
&block_ids,
&base_policy_hash,
Some(part_ls),
span,
text,
)); ));
} }
} }
@@ -183,9 +190,9 @@ fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
mod tests { mod tests {
use super::*; use super::*;
use kebab_core::{ use kebab_core::{
Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
SourceType, TrustLevel, WorkspacePath, WorkspacePath, id_for_block, id_for_doc,
}; };
use time::OffsetDateTime; use time::OffsetDateTime;
@@ -206,39 +213,60 @@ mod tests {
}; };
let bid = id_for_block(&doc_id, "code", &[], i as u32, &span); let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
Block::Code(CodeBlock { Block::Code(CodeBlock {
common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span }, common: CommonBlock {
block_id: bid,
heading_path: vec![],
source_span: span,
},
lang: Some("python".into()), lang: Some("python".into()),
code: (*code).to_string(), code: (*code).to_string(),
}) })
}) })
.collect(); .collect();
CanonicalDocument { CanonicalDocument {
doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(), doc_id,
lang: Lang("und".into()), blocks, source_asset_id: aid,
workspace_path: wp,
title: "a".into(),
lang: Lang("und".into()),
blocks,
metadata: Metadata { metadata: Metadata {
aliases: vec![], tags: vec![], aliases: vec![],
tags: vec![],
created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
source_type: SourceType::Note, trust_level: TrustLevel::Primary, source_type: SourceType::Note,
user_id_alias: None, user: Default::default(), trust_level: TrustLevel::Primary,
repo: Some("kebab".into()), git_branch: Some("main".into()), user_id_alias: None,
git_commit: Some("0".repeat(40)), code_lang: Some("python".into()), user: Default::default(),
repo: Some("kebab".into()),
git_branch: Some("main".into()),
git_commit: Some("0".repeat(40)),
code_lang: Some("python".into()),
}, },
provenance: Provenance { events: vec![] }, provenance: Provenance { events: vec![] },
parser_version: pv, schema_version: 1, doc_version: 1, parser_version: pv,
last_chunker_version: None, last_embedding_version: None, schema_version: 1,
doc_version: 1,
last_chunker_version: None,
last_embedding_version: None,
} }
} }
fn policy() -> ChunkPolicy { fn policy() -> ChunkPolicy {
ChunkPolicy { target_tokens: 500, overlap_tokens: 80, ChunkPolicy {
target_tokens: 500,
overlap_tokens: 80,
respect_markdown_headings: false, respect_markdown_headings: false,
chunker_version: ChunkerVersion(VERSION_LABEL.into()) } chunker_version: ChunkerVersion(VERSION_LABEL.into()),
}
} }
#[test] #[test]
fn chunker_version_is_code_python_ast_v1() { fn chunker_version_is_code_python_ast_v1() {
assert_eq!(CodePythonAstV1Chunker.chunker_version(), assert_eq!(
ChunkerVersion("code-python-ast-v1".into())); CodePythonAstV1Chunker.chunker_version(),
ChunkerVersion("code-python-ast-v1".into())
);
} }
#[test] #[test]
@@ -256,7 +284,12 @@ mod tests {
assert_eq!(c.chunker_version.0, "code-python-ast-v1"); assert_eq!(c.chunker_version.0, "code-python-ast-v1");
} }
match &chunks[0].source_spans[0] { match &chunks[0].source_spans[0] {
SourceSpan::Code { symbol, line_start, line_end, .. } => { SourceSpan::Code {
symbol,
line_start,
line_end,
..
} => {
assert_eq!(symbol.as_deref(), Some("parse")); assert_eq!(symbol.as_deref(), Some("parse"));
assert_eq!((*line_start, *line_end), (1, 3)); assert_eq!((*line_start, *line_end), (1, 3));
} }
@@ -266,22 +299,33 @@ mod tests {
#[test] #[test]
fn oversize_unit_splits_into_parts_with_unique_ids() { fn oversize_unit_splits_into_parts_with_unique_ids() {
let body = (0..500).map(|i| format!(" x{i} = {i}")).collect::<Vec<_>>().join("\n"); let body = (0..500)
.map(|i| format!(" x{i} = {i}"))
.collect::<Vec<_>>()
.join("\n");
let code = format!("def big():\n{body}\n"); let code = format!("def big():\n{body}\n");
let doc = code_doc(&[("big", 1, 502, &code)]); let doc = code_doc(&[("big", 1, 502, &code)]);
let chunks = CodePythonAstV1Chunker.chunk(&doc, &policy()).unwrap(); let chunks = CodePythonAstV1Chunker.chunk(&doc, &policy()).unwrap();
assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len()); assert!(
chunks.len() >= 2,
"oversize unit must split, got {}",
chunks.len()
);
for c in &chunks { for c in &chunks {
match &c.source_spans[0] { match &c.source_spans[0] {
SourceSpan::Code { symbol, .. } => { SourceSpan::Code { symbol, .. } => {
assert!(symbol.as_deref().unwrap().starts_with("big [part "), assert!(
"part-numbered symbol, got {symbol:?}"); symbol.as_deref().unwrap().starts_with("big [part "),
"part-numbered symbol, got {symbol:?}"
);
} }
_ => unreachable!(), _ => unreachable!(),
} }
} }
let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect(); let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
let n = ids.len(); ids.sort_unstable(); ids.dedup(); let n = ids.len();
ids.sort_unstable();
ids.dedup();
assert_eq!(ids.len(), n, "chunk_ids unique across split parts"); assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
} }
@@ -295,7 +339,8 @@ mod tests {
heading_path: vec![], heading_path: vec![],
source_span: SourceSpan::Line { start: 1, end: 1 }, source_span: SourceSpan::Line { start: 1, end: 1 },
}, },
text: "x".into(), inlines: vec![], text: "x".into(),
inlines: vec![],
})]; })];
let err = CodePythonAstV1Chunker.chunk(&doc, &policy()).unwrap_err(); let err = CodePythonAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
assert!(err.to_string().contains("CodePythonAstV1Chunker")); assert!(err.to_string().contains("CodePythonAstV1Chunker"));
@@ -304,11 +349,19 @@ mod tests {
#[test] #[test]
fn deterministic_chunk_ids_1000() { fn deterministic_chunk_ids_1000() {
let doc = code_doc(&[("parse", 1, 2, "def parse(): pass\n")]); let doc = code_doc(&[("parse", 1, 2, "def parse(): pass\n")]);
let base: Vec<String> = CodePythonAstV1Chunker.chunk(&doc, &policy()) let base: Vec<String> = CodePythonAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
for _ in 0..1000 { for _ in 0..1000 {
let again: Vec<String> = CodePythonAstV1Chunker.chunk(&doc, &policy()) let again: Vec<String> = CodePythonAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
assert_eq!(again, base); assert_eq!(again, base);
} }
} }
@@ -316,7 +369,9 @@ mod tests {
#[test] #[test]
fn policy_hash_matches_md_heading_v1() { fn policy_hash_matches_md_heading_v1() {
let p = policy(); let p = policy();
assert_eq!(CodePythonAstV1Chunker.policy_hash(&p), assert_eq!(
crate::MdHeadingV1Chunker.policy_hash(&p)); CodePythonAstV1Chunker.policy_hash(&p),
crate::MdHeadingV1Chunker.policy_hash(&p)
);
} }
} }

View File

@@ -39,11 +39,7 @@ impl Chunker for CodeRustAstV1Chunker {
hex[..POLICY_HASH_HEX_LEN].to_string() hex[..POLICY_HASH_HEX_LEN].to_string()
} }
fn chunk( fn chunk(&self, doc: &CanonicalDocument, policy: &ChunkPolicy) -> anyhow::Result<Vec<Chunk>> {
&self,
doc: &CanonicalDocument,
policy: &ChunkPolicy,
) -> anyhow::Result<Vec<Chunk>> {
for b in &doc.blocks { for b in &doc.blocks {
let c = match b { let c = match b {
Block::Code(c) => c, Block::Code(c) => c,
@@ -68,9 +64,12 @@ impl Chunker for CodeRustAstV1Chunker {
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let (ls, le, symbol, lang) = match &cb.common.source_span { let (ls, le, symbol, lang) = match &cb.common.source_span {
SourceSpan::Code { line_start, line_end, symbol, lang } => { SourceSpan::Code {
(*line_start, *line_end, symbol.clone(), lang.clone()) line_start,
} line_end,
symbol,
lang,
} => (*line_start, *line_end, symbol.clone(), lang.clone()),
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()]; let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
@@ -84,8 +83,13 @@ impl Chunker for CodeRustAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
None, span, cb.code.clone(), &chunker_version,
&block_ids,
&base_policy_hash,
None,
span,
cb.code.clone(),
)); ));
} else { } else {
let parts = split_oversize(&cb.code); let parts = split_oversize(&cb.code);
@@ -93,9 +97,7 @@ impl Chunker for CodeRustAstV1Chunker {
for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() { for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
let part_ls = ls + off_start; let part_ls = ls + off_start;
let part_le = ls + off_end; let part_le = ls + off_end;
let part_sym = symbol let part_sym = symbol.as_ref().map(|s| format!("{s} [part {}/{n}]", i + 1));
.as_ref()
.map(|s| format!("{s} [part {}/{n}]", i + 1));
let span = SourceSpan::Code { let span = SourceSpan::Code {
line_start: part_ls, line_start: part_ls,
line_end: part_le, line_end: part_le,
@@ -103,8 +105,13 @@ impl Chunker for CodeRustAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
Some(part_ls), span, text, &chunker_version,
&block_ids,
&base_policy_hash,
Some(part_ls),
span,
text,
)); ));
} }
} }
@@ -183,9 +190,9 @@ fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
mod tests { mod tests {
use super::*; use super::*;
use kebab_core::{ use kebab_core::{
Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
SourceType, TrustLevel, WorkspacePath, WorkspacePath, id_for_block, id_for_doc,
}; };
use time::OffsetDateTime; use time::OffsetDateTime;
@@ -206,39 +213,60 @@ mod tests {
}; };
let bid = id_for_block(&doc_id, "code", &[], i as u32, &span); let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
Block::Code(CodeBlock { Block::Code(CodeBlock {
common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span }, common: CommonBlock {
block_id: bid,
heading_path: vec![],
source_span: span,
},
lang: Some("rust".into()), lang: Some("rust".into()),
code: (*code).to_string(), code: (*code).to_string(),
}) })
}) })
.collect(); .collect();
CanonicalDocument { CanonicalDocument {
doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(), doc_id,
lang: Lang("und".into()), blocks, source_asset_id: aid,
workspace_path: wp,
title: "a".into(),
lang: Lang("und".into()),
blocks,
metadata: Metadata { metadata: Metadata {
aliases: vec![], tags: vec![], aliases: vec![],
tags: vec![],
created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
source_type: SourceType::Note, trust_level: TrustLevel::Primary, source_type: SourceType::Note,
user_id_alias: None, user: Default::default(), trust_level: TrustLevel::Primary,
repo: Some("kebab".into()), git_branch: Some("main".into()), user_id_alias: None,
git_commit: Some("0".repeat(40)), code_lang: Some("rust".into()), user: Default::default(),
repo: Some("kebab".into()),
git_branch: Some("main".into()),
git_commit: Some("0".repeat(40)),
code_lang: Some("rust".into()),
}, },
provenance: Provenance { events: vec![] }, provenance: Provenance { events: vec![] },
parser_version: pv, schema_version: 1, doc_version: 1, parser_version: pv,
last_chunker_version: None, last_embedding_version: None, schema_version: 1,
doc_version: 1,
last_chunker_version: None,
last_embedding_version: None,
} }
} }
fn policy() -> ChunkPolicy { fn policy() -> ChunkPolicy {
ChunkPolicy { target_tokens: 500, overlap_tokens: 80, ChunkPolicy {
target_tokens: 500,
overlap_tokens: 80,
respect_markdown_headings: false, respect_markdown_headings: false,
chunker_version: ChunkerVersion(VERSION_LABEL.into()) } chunker_version: ChunkerVersion(VERSION_LABEL.into()),
}
} }
#[test] #[test]
fn chunker_version_is_code_rust_ast_v1() { fn chunker_version_is_code_rust_ast_v1() {
assert_eq!(CodeRustAstV1Chunker.chunker_version(), assert_eq!(
ChunkerVersion("code-rust-ast-v1".into())); CodeRustAstV1Chunker.chunker_version(),
ChunkerVersion("code-rust-ast-v1".into())
);
} }
#[test] #[test]
@@ -256,7 +284,12 @@ mod tests {
assert_eq!(c.chunker_version.0, "code-rust-ast-v1"); assert_eq!(c.chunker_version.0, "code-rust-ast-v1");
} }
match &chunks[0].source_spans[0] { match &chunks[0].source_spans[0] {
SourceSpan::Code { symbol, line_start, line_end, .. } => { SourceSpan::Code {
symbol,
line_start,
line_end,
..
} => {
assert_eq!(symbol.as_deref(), Some("parse")); assert_eq!(symbol.as_deref(), Some("parse"));
assert_eq!((*line_start, *line_end), (1, 3)); assert_eq!((*line_start, *line_end), (1, 3));
} }
@@ -266,22 +299,33 @@ mod tests {
#[test] #[test]
fn oversize_unit_splits_into_parts_with_unique_ids() { fn oversize_unit_splits_into_parts_with_unique_ids() {
let body = (0..500).map(|i| format!(" let x{i} = {i};")).collect::<Vec<_>>().join("\n"); let body = (0..500)
.map(|i| format!(" let x{i} = {i};"))
.collect::<Vec<_>>()
.join("\n");
let code = format!("pub fn big() {{\n{body}\n}}"); let code = format!("pub fn big() {{\n{body}\n}}");
let doc = code_doc(&[("big", 1, 502, &code)]); let doc = code_doc(&[("big", 1, 502, &code)]);
let chunks = CodeRustAstV1Chunker.chunk(&doc, &policy()).unwrap(); let chunks = CodeRustAstV1Chunker.chunk(&doc, &policy()).unwrap();
assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len()); assert!(
chunks.len() >= 2,
"oversize unit must split, got {}",
chunks.len()
);
for c in &chunks { for c in &chunks {
match &c.source_spans[0] { match &c.source_spans[0] {
SourceSpan::Code { symbol, .. } => { SourceSpan::Code { symbol, .. } => {
assert!(symbol.as_deref().unwrap().starts_with("big [part "), assert!(
"part-numbered symbol, got {symbol:?}"); symbol.as_deref().unwrap().starts_with("big [part "),
"part-numbered symbol, got {symbol:?}"
);
} }
_ => unreachable!(), _ => unreachable!(),
} }
} }
let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect(); let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
let n = ids.len(); ids.sort_unstable(); ids.dedup(); let n = ids.len();
ids.sort_unstable();
ids.dedup();
assert_eq!(ids.len(), n, "chunk_ids unique across split parts"); assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
} }
@@ -295,7 +339,8 @@ mod tests {
heading_path: vec![], heading_path: vec![],
source_span: SourceSpan::Line { start: 1, end: 1 }, source_span: SourceSpan::Line { start: 1, end: 1 },
}, },
text: "x".into(), inlines: vec![], text: "x".into(),
inlines: vec![],
})]; })];
let err = CodeRustAstV1Chunker.chunk(&doc, &policy()).unwrap_err(); let err = CodeRustAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
assert!(err.to_string().contains("CodeRustAstV1Chunker")); assert!(err.to_string().contains("CodeRustAstV1Chunker"));
@@ -304,11 +349,19 @@ mod tests {
#[test] #[test]
fn deterministic_chunk_ids_1000() { fn deterministic_chunk_ids_1000() {
let doc = code_doc(&[("parse", 1, 2, "fn parse(){}\n}")]); let doc = code_doc(&[("parse", 1, 2, "fn parse(){}\n}")]);
let base: Vec<String> = CodeRustAstV1Chunker.chunk(&doc, &policy()) let base: Vec<String> = CodeRustAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
for _ in 0..1000 { for _ in 0..1000 {
let again: Vec<String> = CodeRustAstV1Chunker.chunk(&doc, &policy()) let again: Vec<String> = CodeRustAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
assert_eq!(again, base); assert_eq!(again, base);
} }
} }
@@ -316,7 +369,9 @@ mod tests {
#[test] #[test]
fn policy_hash_matches_md_heading_v1() { fn policy_hash_matches_md_heading_v1() {
let p = policy(); let p = policy();
assert_eq!(CodeRustAstV1Chunker.policy_hash(&p), assert_eq!(
crate::MdHeadingV1Chunker.policy_hash(&p)); CodeRustAstV1Chunker.policy_hash(&p),
crate::MdHeadingV1Chunker.policy_hash(&p)
);
} }
} }

View File

@@ -9,7 +9,7 @@
use crate::tier2_shared::{build_chunk_no_symbol, policy_hash}; use crate::tier2_shared::{build_chunk_no_symbol, policy_hash};
use anyhow::Result; use anyhow::Result;
use kebab_core::{Block, CanonicalDocument, Chunk, ChunkPolicy, ChunkerVersion, Chunker}; use kebab_core::{Block, CanonicalDocument, Chunk, ChunkPolicy, Chunker, ChunkerVersion};
pub const VERSION_LABEL: &str = "code-text-paragraph-v1"; pub const VERSION_LABEL: &str = "code-text-paragraph-v1";

View File

@@ -39,17 +39,13 @@ impl Chunker for CodeTsAstV1Chunker {
hex[..POLICY_HASH_HEX_LEN].to_string() hex[..POLICY_HASH_HEX_LEN].to_string()
} }
fn chunk( fn chunk(&self, doc: &CanonicalDocument, policy: &ChunkPolicy) -> anyhow::Result<Vec<Chunk>> {
&self,
doc: &CanonicalDocument,
policy: &ChunkPolicy,
) -> anyhow::Result<Vec<Chunk>> {
for b in &doc.blocks { for b in &doc.blocks {
let c = match b { let c = match b {
Block::Code(c) => c, Block::Code(c) => c,
_ => anyhow::bail!( _ => {
"CodeTsAstV1Chunker only handles code docs (got non-Code block)" anyhow::bail!("CodeTsAstV1Chunker only handles code docs (got non-Code block)")
), }
}; };
if !matches!(c.common.source_span, SourceSpan::Code { .. }) { if !matches!(c.common.source_span, SourceSpan::Code { .. }) {
anyhow::bail!( anyhow::bail!(
@@ -68,9 +64,12 @@ impl Chunker for CodeTsAstV1Chunker {
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let (ls, le, symbol, lang) = match &cb.common.source_span { let (ls, le, symbol, lang) = match &cb.common.source_span {
SourceSpan::Code { line_start, line_end, symbol, lang } => { SourceSpan::Code {
(*line_start, *line_end, symbol.clone(), lang.clone()) line_start,
} line_end,
symbol,
lang,
} => (*line_start, *line_end, symbol.clone(), lang.clone()),
_ => unreachable!("validated above"), _ => unreachable!("validated above"),
}; };
let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()]; let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
@@ -84,8 +83,13 @@ impl Chunker for CodeTsAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
None, span, cb.code.clone(), &chunker_version,
&block_ids,
&base_policy_hash,
None,
span,
cb.code.clone(),
)); ));
} else { } else {
let parts = split_oversize(&cb.code); let parts = split_oversize(&cb.code);
@@ -93,9 +97,7 @@ impl Chunker for CodeTsAstV1Chunker {
for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() { for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
let part_ls = ls + off_start; let part_ls = ls + off_start;
let part_le = ls + off_end; let part_le = ls + off_end;
let part_sym = symbol let part_sym = symbol.as_ref().map(|s| format!("{s} [part {}/{n}]", i + 1));
.as_ref()
.map(|s| format!("{s} [part {}/{n}]", i + 1));
let span = SourceSpan::Code { let span = SourceSpan::Code {
line_start: part_ls, line_start: part_ls,
line_end: part_le, line_end: part_le,
@@ -103,8 +105,13 @@ impl Chunker for CodeTsAstV1Chunker {
lang: lang.clone(), lang: lang.clone(),
}; };
out.push(make_chunk( out.push(make_chunk(
doc, &chunker_version, &block_ids, &base_policy_hash, doc,
Some(part_ls), span, text, &chunker_version,
&block_ids,
&base_policy_hash,
Some(part_ls),
span,
text,
)); ));
} }
} }
@@ -183,9 +190,9 @@ fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
mod tests { mod tests {
use super::*; use super::*;
use kebab_core::{ use kebab_core::{
Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
SourceType, TrustLevel, WorkspacePath, WorkspacePath, id_for_block, id_for_doc,
}; };
use time::OffsetDateTime; use time::OffsetDateTime;
@@ -206,46 +213,72 @@ mod tests {
}; };
let bid = id_for_block(&doc_id, "code", &[], i as u32, &span); let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
Block::Code(CodeBlock { Block::Code(CodeBlock {
common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span }, common: CommonBlock {
block_id: bid,
heading_path: vec![],
source_span: span,
},
lang: Some("typescript".into()), lang: Some("typescript".into()),
code: (*code).to_string(), code: (*code).to_string(),
}) })
}) })
.collect(); .collect();
CanonicalDocument { CanonicalDocument {
doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(), doc_id,
lang: Lang("und".into()), blocks, source_asset_id: aid,
workspace_path: wp,
title: "a".into(),
lang: Lang("und".into()),
blocks,
metadata: Metadata { metadata: Metadata {
aliases: vec![], tags: vec![], aliases: vec![],
tags: vec![],
created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
source_type: SourceType::Note, trust_level: TrustLevel::Primary, source_type: SourceType::Note,
user_id_alias: None, user: Default::default(), trust_level: TrustLevel::Primary,
repo: Some("kebab".into()), git_branch: Some("main".into()), user_id_alias: None,
git_commit: Some("0".repeat(40)), code_lang: Some("typescript".into()), user: Default::default(),
repo: Some("kebab".into()),
git_branch: Some("main".into()),
git_commit: Some("0".repeat(40)),
code_lang: Some("typescript".into()),
}, },
provenance: Provenance { events: vec![] }, provenance: Provenance { events: vec![] },
parser_version: pv, schema_version: 1, doc_version: 1, parser_version: pv,
last_chunker_version: None, last_embedding_version: None, schema_version: 1,
doc_version: 1,
last_chunker_version: None,
last_embedding_version: None,
} }
} }
fn policy() -> ChunkPolicy { fn policy() -> ChunkPolicy {
ChunkPolicy { target_tokens: 500, overlap_tokens: 80, ChunkPolicy {
target_tokens: 500,
overlap_tokens: 80,
respect_markdown_headings: false, respect_markdown_headings: false,
chunker_version: ChunkerVersion(VERSION_LABEL.into()) } chunker_version: ChunkerVersion(VERSION_LABEL.into()),
}
} }
#[test] #[test]
fn chunker_version_is_code_ts_ast_v1() { fn chunker_version_is_code_ts_ast_v1() {
assert_eq!(CodeTsAstV1Chunker.chunker_version(), assert_eq!(
ChunkerVersion("code-ts-ast-v1".into())); CodeTsAstV1Chunker.chunker_version(),
ChunkerVersion("code-ts-ast-v1".into())
);
} }
#[test] #[test]
fn one_chunk_per_unit_preserves_code_span() { fn one_chunk_per_unit_preserves_code_span() {
let doc = code_doc(&[ let doc = code_doc(&[
("parse", 1, 3, "function parse(): void {\n // x\n}"), ("parse", 1, 3, "function parse(): void {\n // x\n}"),
("Foo.double", 5, 7, "function double(): number {\n //\n return 0;\n}"), (
"Foo.double",
5,
7,
"function double(): number {\n //\n return 0;\n}",
),
]); ]);
let chunks = CodeTsAstV1Chunker.chunk(&doc, &policy()).unwrap(); let chunks = CodeTsAstV1Chunker.chunk(&doc, &policy()).unwrap();
assert_eq!(chunks.len(), 2); assert_eq!(chunks.len(), 2);
@@ -256,7 +289,12 @@ mod tests {
assert_eq!(c.chunker_version.0, "code-ts-ast-v1"); assert_eq!(c.chunker_version.0, "code-ts-ast-v1");
} }
match &chunks[0].source_spans[0] { match &chunks[0].source_spans[0] {
SourceSpan::Code { symbol, line_start, line_end, .. } => { SourceSpan::Code {
symbol,
line_start,
line_end,
..
} => {
assert_eq!(symbol.as_deref(), Some("parse")); assert_eq!(symbol.as_deref(), Some("parse"));
assert_eq!((*line_start, *line_end), (1, 3)); assert_eq!((*line_start, *line_end), (1, 3));
} }
@@ -266,22 +304,33 @@ mod tests {
#[test] #[test]
fn oversize_unit_splits_into_parts_with_unique_ids() { fn oversize_unit_splits_into_parts_with_unique_ids() {
let body = (0..500).map(|i| format!(" const x{i} = {i};")).collect::<Vec<_>>().join("\n"); let body = (0..500)
.map(|i| format!(" const x{i} = {i};"))
.collect::<Vec<_>>()
.join("\n");
let code = format!("function big(): void {{\n{body}\n}}"); let code = format!("function big(): void {{\n{body}\n}}");
let doc = code_doc(&[("big", 1, 502, &code)]); let doc = code_doc(&[("big", 1, 502, &code)]);
let chunks = CodeTsAstV1Chunker.chunk(&doc, &policy()).unwrap(); let chunks = CodeTsAstV1Chunker.chunk(&doc, &policy()).unwrap();
assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len()); assert!(
chunks.len() >= 2,
"oversize unit must split, got {}",
chunks.len()
);
for c in &chunks { for c in &chunks {
match &c.source_spans[0] { match &c.source_spans[0] {
SourceSpan::Code { symbol, .. } => { SourceSpan::Code { symbol, .. } => {
assert!(symbol.as_deref().unwrap().starts_with("big [part "), assert!(
"part-numbered symbol, got {symbol:?}"); symbol.as_deref().unwrap().starts_with("big [part "),
"part-numbered symbol, got {symbol:?}"
);
} }
_ => unreachable!(), _ => unreachable!(),
} }
} }
let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect(); let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
let n = ids.len(); ids.sort_unstable(); ids.dedup(); let n = ids.len();
ids.sort_unstable();
ids.dedup();
assert_eq!(ids.len(), n, "chunk_ids unique across split parts"); assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
} }
@@ -295,7 +344,8 @@ mod tests {
heading_path: vec![], heading_path: vec![],
source_span: SourceSpan::Line { start: 1, end: 1 }, source_span: SourceSpan::Line { start: 1, end: 1 },
}, },
text: "x".into(), inlines: vec![], text: "x".into(),
inlines: vec![],
})]; })];
let err = CodeTsAstV1Chunker.chunk(&doc, &policy()).unwrap_err(); let err = CodeTsAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
assert!(err.to_string().contains("CodeTsAstV1Chunker")); assert!(err.to_string().contains("CodeTsAstV1Chunker"));
@@ -304,11 +354,19 @@ mod tests {
#[test] #[test]
fn deterministic_chunk_ids_1000() { fn deterministic_chunk_ids_1000() {
let doc = code_doc(&[("parse", 1, 2, "function parse(): void {}\n")]); let doc = code_doc(&[("parse", 1, 2, "function parse(): void {}\n")]);
let base: Vec<String> = CodeTsAstV1Chunker.chunk(&doc, &policy()) let base: Vec<String> = CodeTsAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
for _ in 0..1000 { for _ in 0..1000 {
let again: Vec<String> = CodeTsAstV1Chunker.chunk(&doc, &policy()) let again: Vec<String> = CodeTsAstV1Chunker
.unwrap().into_iter().map(|c| c.chunk_id.0).collect(); .chunk(&doc, &policy())
.unwrap()
.into_iter()
.map(|c| c.chunk_id.0)
.collect();
assert_eq!(again, base); assert_eq!(again, base);
} }
} }
@@ -316,7 +374,9 @@ mod tests {
#[test] #[test]
fn policy_hash_matches_md_heading_v1() { fn policy_hash_matches_md_heading_v1() {
let p = policy(); let p = policy();
assert_eq!(CodeTsAstV1Chunker.policy_hash(&p), assert_eq!(
crate::MdHeadingV1Chunker.policy_hash(&p)); CodeTsAstV1Chunker.policy_hash(&p),
crate::MdHeadingV1Chunker.policy_hash(&p)
);
} }
} }

View File

@@ -7,7 +7,7 @@
use crate::tier2_shared::{policy_hash, push_chunks_with_oversize}; use crate::tier2_shared::{policy_hash, push_chunks_with_oversize};
use anyhow::Result; use anyhow::Result;
use kebab_core::{Block, CanonicalDocument, Chunk, ChunkPolicy, ChunkerVersion, Chunker}; use kebab_core::{Block, CanonicalDocument, Chunk, ChunkPolicy, Chunker, ChunkerVersion};
pub const VERSION_LABEL: &str = "dockerfile-file-v1"; pub const VERSION_LABEL: &str = "dockerfile-file-v1";

View File

@@ -8,7 +8,7 @@
use crate::tier2_shared::{policy_hash, push_chunks_with_oversize}; use crate::tier2_shared::{policy_hash, push_chunks_with_oversize};
use anyhow::Result; use anyhow::Result;
use kebab_core::{Block, CanonicalDocument, Chunk, ChunkPolicy, ChunkerVersion, Chunker}; use kebab_core::{Block, CanonicalDocument, Chunk, ChunkPolicy, Chunker, ChunkerVersion};
pub const VERSION_LABEL: &str = "k8s-manifest-resource-v1"; pub const VERSION_LABEL: &str = "k8s-manifest-resource-v1";
@@ -49,19 +49,14 @@ impl Chunker for K8sManifestResourceV1Chunker {
.get("apiVersion") .get("apiVersion")
.and_then(|v| v.as_str()) .and_then(|v| v.as_str())
.unwrap_or(""); .unwrap_or("");
let kind = mapping let kind = mapping.get("kind").and_then(|v| v.as_str()).unwrap_or("");
.get("kind")
.and_then(|v| v.as_str())
.unwrap_or("");
// Skip non-k8s documents. // Skip non-k8s documents.
if api.is_empty() || kind.is_empty() { if api.is_empty() || kind.is_empty() {
continue; continue;
} }
let metadata = mapping let metadata = mapping.get("metadata").and_then(|v| v.as_mapping());
.get("metadata")
.and_then(|v| v.as_mapping());
let name = metadata let name = metadata
.and_then(|m| m.get("name")) .and_then(|m| m.get("name"))
.and_then(|v| v.as_str()) .and_then(|v| v.as_str())
@@ -118,10 +113,7 @@ fn split_yaml_documents(text: &str) -> Vec<YamlSlice<'_>> {
.enumerate() .enumerate()
.filter_map(|(i, l)| { .filter_map(|(i, l)| {
let trimmed = l.trim_end(); let trimmed = l.trim_end();
if trimmed == "---" if trimmed == "---" || trimmed.starts_with("--- ") || trimmed.starts_with("---\t") {
|| trimmed.starts_with("--- ")
|| trimmed.starts_with("---\t")
{
Some(i) Some(i)
} else { } else {
None None

View File

@@ -23,14 +23,14 @@ mod code_js_ast_v1;
mod code_kotlin_ast_v1; mod code_kotlin_ast_v1;
mod code_python_ast_v1; mod code_python_ast_v1;
mod code_rust_ast_v1; mod code_rust_ast_v1;
pub mod code_text_paragraph_v1;
mod code_ts_ast_v1; mod code_ts_ast_v1;
pub mod dockerfile_file_v1;
pub mod k8s_manifest_resource_v1;
pub mod manifest_file_v1;
mod md_heading_v1; mod md_heading_v1;
mod pdf_page_v1; mod pdf_page_v1;
mod tier2_shared; mod tier2_shared;
pub mod k8s_manifest_resource_v1;
pub mod dockerfile_file_v1;
pub mod manifest_file_v1;
pub mod code_text_paragraph_v1;
pub use code_c_ast_v1::CodeCAstV1Chunker; pub use code_c_ast_v1::CodeCAstV1Chunker;
pub use code_cpp_ast_v1::CodeCppAstV1Chunker; pub use code_cpp_ast_v1::CodeCppAstV1Chunker;
@@ -40,10 +40,10 @@ pub use code_js_ast_v1::CodeJsAstV1Chunker;
pub use code_kotlin_ast_v1::CodeKotlinAstV1Chunker; pub use code_kotlin_ast_v1::CodeKotlinAstV1Chunker;
pub use code_python_ast_v1::CodePythonAstV1Chunker; pub use code_python_ast_v1::CodePythonAstV1Chunker;
pub use code_rust_ast_v1::CodeRustAstV1Chunker; pub use code_rust_ast_v1::CodeRustAstV1Chunker;
pub use code_text_paragraph_v1::CodeTextParagraphV1Chunker;
pub use code_ts_ast_v1::CodeTsAstV1Chunker; pub use code_ts_ast_v1::CodeTsAstV1Chunker;
pub use dockerfile_file_v1::DockerfileFileV1Chunker;
pub use k8s_manifest_resource_v1::K8sManifestResourceV1Chunker;
pub use manifest_file_v1::ManifestFileV1Chunker;
pub use md_heading_v1::MdHeadingV1Chunker; pub use md_heading_v1::MdHeadingV1Chunker;
pub use pdf_page_v1::PdfPageV1Chunker; pub use pdf_page_v1::PdfPageV1Chunker;
pub use k8s_manifest_resource_v1::K8sManifestResourceV1Chunker;
pub use dockerfile_file_v1::DockerfileFileV1Chunker;
pub use manifest_file_v1::ManifestFileV1Chunker;
pub use code_text_paragraph_v1::CodeTextParagraphV1Chunker;

View File

@@ -8,7 +8,7 @@
use crate::tier2_shared::{policy_hash, push_chunks_with_oversize}; use crate::tier2_shared::{policy_hash, push_chunks_with_oversize};
use anyhow::Result; use anyhow::Result;
use kebab_core::{Block, CanonicalDocument, Chunk, ChunkPolicy, ChunkerVersion, Chunker}; use kebab_core::{Block, CanonicalDocument, Chunk, ChunkPolicy, Chunker, ChunkerVersion};
pub const VERSION_LABEL: &str = "manifest-file-v1"; pub const VERSION_LABEL: &str = "manifest-file-v1";

View File

@@ -1,8 +1,8 @@
//! `md-heading-v1` — heading-aware Markdown chunker. //! `md-heading-v1` — heading-aware Markdown chunker.
use kebab_core::{ use kebab_core::{
Block, BlockId, CanonicalDocument, Chunk, ChunkPolicy, Chunker, Block, BlockId, CanonicalDocument, Chunk, ChunkPolicy, Chunker, ChunkerVersion, DocumentId,
ChunkerVersion, DocumentId, SourceSpan, id_for_chunk, SourceSpan, id_for_chunk,
}; };
/// Version label emitted by [`MdHeadingV1Chunker`]. Bumping this label /// Version label emitted by [`MdHeadingV1Chunker`]. Bumping this label
@@ -99,11 +99,7 @@ impl Chunker for MdHeadingV1Chunker {
hex[..POLICY_HASH_HEX_LEN].to_string() hex[..POLICY_HASH_HEX_LEN].to_string()
} }
fn chunk( fn chunk(&self, doc: &CanonicalDocument, policy: &ChunkPolicy) -> anyhow::Result<Vec<Chunk>> {
&self,
doc: &CanonicalDocument,
policy: &ChunkPolicy,
) -> anyhow::Result<Vec<Chunk>> {
let policy_hash = self.policy_hash(policy); let policy_hash = self.policy_hash(policy);
let chunker_version = self.chunker_version(); let chunker_version = self.chunker_version();
let mut out: Vec<Chunk> = Vec::new(); let mut out: Vec<Chunk> = Vec::new();
@@ -152,22 +148,12 @@ impl Chunker for MdHeadingV1Chunker {
// `collect_overlap_seed` keeps seed ≤ target/2, so // `collect_overlap_seed` keeps seed ≤ target/2, so
// a flush here never produces a chunk smaller than // a flush here never produces a chunk smaller than
// the seed budget. // the seed budget.
let would_exceed = acc.text_tokens + next_tokens let would_exceed = acc.text_tokens + next_tokens > policy.target_tokens
> policy.target_tokens
&& acc.has_non_heading_content(); && acc.has_non_heading_content();
if would_exceed { if would_exceed {
let overlap_seed = collect_overlap_seed( let overlap_seed =
&acc, collect_overlap_seed(&acc, policy.overlap_tokens, policy.target_tokens);
policy.overlap_tokens, flush(&mut acc, doc, &chunker_version, &policy_hash, &mut out);
policy.target_tokens,
);
flush(
&mut acc,
doc,
&chunker_version,
&policy_hash,
&mut out,
);
// Seed next accumulator with the prior chunk's // Seed next accumulator with the prior chunk's
// tail blocks (paragraph-level overlap). The // tail blocks (paragraph-level overlap). The
// heading is *not* re-included here — it lives // heading is *not* re-included here — it lives
@@ -292,10 +278,11 @@ fn build_chunk(
) -> Chunk { ) -> Chunk {
debug_assert!(!blocks.is_empty(), "build_chunk requires ≥1 block"); debug_assert!(!blocks.is_empty(), "build_chunk requires ≥1 block");
let block_ids: Vec<BlockId> = let block_ids: Vec<BlockId> = blocks.iter().map(|b| common(b).block_id.clone()).collect();
blocks.iter().map(|b| common(b).block_id.clone()).collect(); let source_spans: Vec<SourceSpan> = blocks
let source_spans: Vec<SourceSpan> = .iter()
blocks.iter().map(|b| common(b).source_span.clone()).collect(); .map(|b| common(b).source_span.clone())
.collect();
// heading_path: pick the first non-Heading block's heading_path // heading_path: pick the first non-Heading block's heading_path
// (which already includes every parent heading per kb-normalize). // (which already includes every parent heading per kb-normalize).
@@ -339,12 +326,7 @@ fn build_chunk(
text.len().div_ceil(BYTES_PER_TOKEN) text.len().div_ceil(BYTES_PER_TOKEN)
}; };
let chunk_id = id_for_chunk( let chunk_id = id_for_chunk(&doc.doc_id, chunker_version, &block_ids, policy_hash);
&doc.doc_id,
chunker_version,
&block_ids,
policy_hash,
);
Chunk { Chunk {
chunk_id, chunk_id,
@@ -400,14 +382,8 @@ fn render_block_text(b: &Block) -> String {
} else { } else {
i.alt.clone() i.alt.clone()
}; };
let ocr = i let ocr = i.ocr.as_ref().map_or("", |o| o.joined.as_str());
.ocr let cap = i.caption.as_ref().map_or("", |c| c.text.as_str());
.as_ref()
.map_or("", |o| o.joined.as_str());
let cap = i
.caption
.as_ref()
.map_or("", |c| c.text.as_str());
[alt.as_str(), ocr, cap] [alt.as_str(), ocr, cap]
.iter() .iter()
.filter(|s| !s.is_empty()) .filter(|s| !s.is_empty())
@@ -447,9 +423,8 @@ fn common(b: &Block) -> &kebab_core::CommonBlock {
mod tests { mod tests {
use super::*; use super::*;
use kebab_core::{ use kebab_core::{
AssetId, CodeBlock, CommonBlock, HeadingBlock, ImageRefBlock, Lang, AssetId, CodeBlock, CommonBlock, HeadingBlock, ImageRefBlock, Lang, Metadata, Provenance,
Metadata, Provenance, SourceType, TableBlock, TextBlock, TrustLevel, SourceType, TableBlock, TextBlock, TrustLevel, WorkspacePath, id_for_block,
WorkspacePath, id_for_block,
}; };
use time::OffsetDateTime; use time::OffsetDateTime;
@@ -492,12 +467,7 @@ mod tests {
SourceSpan::Line { start, end } SourceSpan::Line { start, end }
} }
fn common_for( fn common_for(kind: &str, heading_path: &[String], ordinal: u32, s: SourceSpan) -> CommonBlock {
kind: &str,
heading_path: &[String],
ordinal: u32,
s: SourceSpan,
) -> CommonBlock {
CommonBlock { CommonBlock {
block_id: id_for_block(&doc_id(), kind, heading_path, ordinal, &s), block_id: id_for_block(&doc_id(), kind, heading_path, ordinal, &s),
heading_path: heading_path.to_vec(), heading_path: heading_path.to_vec(),
@@ -532,12 +502,7 @@ mod tests {
}) })
} }
fn paragraph( fn paragraph(text: &str, heading_path: &[&str], ordinal: u32, line: u32) -> Block {
text: &str,
heading_path: &[&str],
ordinal: u32,
line: u32,
) -> Block {
let hp: Vec<String> = heading_path.iter().map(|s| (*s).into()).collect(); let hp: Vec<String> = heading_path.iter().map(|s| (*s).into()).collect();
Block::Paragraph(TextBlock { Block::Paragraph(TextBlock {
common: common_for("paragraph", &hp, ordinal, span(line, line)), common: common_for("paragraph", &hp, ordinal, span(line, line)),
@@ -546,12 +511,7 @@ mod tests {
}) })
} }
fn code_block( fn code_block(code: &str, heading_path: &[&str], ordinal: u32, s: SourceSpan) -> Block {
code: &str,
heading_path: &[&str],
ordinal: u32,
s: SourceSpan,
) -> Block {
let hp: Vec<String> = heading_path.iter().map(|s| (*s).into()).collect(); let hp: Vec<String> = heading_path.iter().map(|s| (*s).into()).collect();
Block::Code(CodeBlock { Block::Code(CodeBlock {
common: common_for("code", &hp, ordinal, s), common: common_for("code", &hp, ordinal, s),
@@ -578,12 +538,7 @@ mod tests {
}) })
} }
fn image_ref( fn image_ref(alt: &str, heading_path: &[&str], ordinal: u32, line: u32) -> Block {
alt: &str,
heading_path: &[&str],
ordinal: u32,
line: u32,
) -> Block {
let hp: Vec<String> = heading_path.iter().map(|s| (*s).into()).collect(); let hp: Vec<String> = heading_path.iter().map(|s| (*s).into()).collect();
Block::ImageRef(ImageRefBlock { Block::ImageRef(ImageRefBlock {
common: common_for("imageref", &hp, ordinal, span(line, line)), common: common_for("imageref", &hp, ordinal, span(line, line)),

View File

@@ -92,11 +92,7 @@ impl Chunker for PdfPageV1Chunker {
hex[..POLICY_HASH_HEX_LEN].to_string() hex[..POLICY_HASH_HEX_LEN].to_string()
} }
fn chunk( fn chunk(&self, doc: &CanonicalDocument, policy: &ChunkPolicy) -> anyhow::Result<Vec<Chunk>> {
&self,
doc: &CanonicalDocument,
policy: &ChunkPolicy,
) -> anyhow::Result<Vec<Chunk>> {
// Validate up front — every block must be a Paragraph carrying // Validate up front — every block must be a Paragraph carrying
// SourceSpan::Page. A mixed document signals a routing bug in // SourceSpan::Page. A mixed document signals a routing bug in
// the caller (e.g. running this chunker on Markdown) and is // the caller (e.g. running this chunker on Markdown) and is
@@ -109,18 +105,13 @@ impl Chunker for PdfPageV1Chunker {
), ),
}; };
if !matches!(common.source_span, SourceSpan::Page { .. }) { if !matches!(common.source_span, SourceSpan::Page { .. }) {
anyhow::bail!( anyhow::bail!("PdfPageV1Chunker only handles PDF docs (got non-Page source_span)");
"PdfPageV1Chunker only handles PDF docs (got non-Page source_span)"
);
} }
} }
let base_policy_hash = self.policy_hash(policy); let base_policy_hash = self.policy_hash(policy);
let chunker_version = self.chunker_version(); let chunker_version = self.chunker_version();
let target_bytes = policy let target_bytes = policy.target_tokens.saturating_mul(BYTES_PER_TOKEN).max(1);
.target_tokens
.saturating_mul(BYTES_PER_TOKEN)
.max(1);
// Clamp the overlap to half the target. Without this, a policy // Clamp the overlap to half the target. Without this, a policy
// with `overlap_tokens >= target_tokens` would make every chunk // with `overlap_tokens >= target_tokens` would make every chunk
// fully re-emit the previous chunk's text — mirrors // fully re-emit the previous chunk's text — mirrors
@@ -157,10 +148,8 @@ impl Chunker for PdfPageV1Chunker {
// typography); silent `as u32` truncation would only // typography); silent `as u32` truncation would only
// surface on corrupted input, where an explicit panic // surface on corrupted input, where an explicit panic
// is preferable to an off-by-2^32 span. // is preferable to an off-by-2^32 span.
let char_start_u32 = u32::try_from(char_start) let char_start_u32 = u32::try_from(char_start).expect("page chars fit in u32");
.expect("page chars fit in u32"); let char_end_u32 = u32::try_from(char_end).expect("page chars fit in u32");
let char_end_u32 =
u32::try_from(char_end).expect("page chars fit in u32");
let span = SourceSpan::Page { let span = SourceSpan::Page {
page: page_num, page: page_num,
char_start: Some(char_start_u32), char_start: Some(char_start_u32),
@@ -213,7 +202,11 @@ impl Chunker for PdfPageV1Chunker {
/// - `chunk_end` = chunk's end char index (exclusive). /// - `chunk_end` = chunk's end char index (exclusive).
/// ///
/// Returns an empty vector when `text` is empty or whitespace-only. /// Returns an empty vector when `text` is empty or whitespace-only.
fn chunk_page(text: &str, target_bytes: usize, overlap_bytes: usize) -> Vec<(usize, usize, usize, String)> { fn chunk_page(
text: &str,
target_bytes: usize,
overlap_bytes: usize,
) -> Vec<(usize, usize, usize, String)> {
let chars: Vec<char> = text.chars().collect(); let chars: Vec<char> = text.chars().collect();
let n = chars.len(); let n = chars.len();
if n == 0 { if n == 0 {
@@ -233,8 +226,7 @@ fn chunk_page(text: &str, target_bytes: usize, overlap_bytes: usize) -> Vec<(usi
let c = chars[k]; let c = chars[k];
let nx = chars[k + 1]; let nx = chars[k + 1];
let is_paragraph_break = c == '\n' && nx == '\n'; let is_paragraph_break = c == '\n' && nx == '\n';
let is_sentence_end = let is_sentence_end = matches!(c, '.' | '?' | '!') && nx.is_whitespace();
matches!(c, '.' | '?' | '!') && nx.is_whitespace();
if (is_paragraph_break || is_sentence_end) && k + 2 <= n { if (is_paragraph_break || is_sentence_end) && k + 2 <= n {
bounds.push(k + 2); bounds.push(k + 2);
} }
@@ -246,9 +238,7 @@ fn chunk_page(text: &str, target_bytes: usize, overlap_bytes: usize) -> Vec<(usi
bounds.dedup(); bounds.dedup();
// UTF-8 byte length of the slice between two char indices. // UTF-8 byte length of the slice between two char indices.
let byte_len = |a: usize, b: usize| -> usize { let byte_len = |a: usize, b: usize| -> usize { chars[a..b].iter().map(|c| c.len_utf8()).sum() };
chars[a..b].iter().map(|c| c.len_utf8()).sum()
};
let mut chunks: Vec<(usize, usize, usize, String)> = Vec::new(); let mut chunks: Vec<(usize, usize, usize, String)> = Vec::new();
let mut seg_idx: usize = 0; let mut seg_idx: usize = 0;
@@ -403,7 +393,11 @@ mod tests {
assert_eq!(c.heading_path, Vec::<String>::new()); assert_eq!(c.heading_path, Vec::<String>::new());
assert_eq!(c.source_spans.len(), 1); assert_eq!(c.source_spans.len(), 1);
match c.source_spans[0] { match c.source_spans[0] {
SourceSpan::Page { page, char_start, char_end } => { SourceSpan::Page {
page,
char_start,
char_end,
} => {
assert_eq!(page, (i as u32) + 1); assert_eq!(page, (i as u32) + 1);
assert_eq!(char_start, Some(0)); assert_eq!(char_start, Some(0));
assert!(char_end.unwrap() > 0); assert!(char_end.unwrap() > 0);
@@ -448,11 +442,16 @@ mod tests {
// N-1's char_end). // N-1's char_end).
for w in chunks.windows(2) { for w in chunks.windows(2) {
let prev_end = match w[0].source_spans[0] { let prev_end = match w[0].source_spans[0] {
SourceSpan::Page { char_end: Some(e), .. } => e, SourceSpan::Page {
char_end: Some(e), ..
} => e,
_ => panic!("missing char_end"), _ => panic!("missing char_end"),
}; };
let next_start = match w[1].source_spans[0] { let next_start = match w[1].source_spans[0] {
SourceSpan::Page { char_start: Some(s), .. } => s, SourceSpan::Page {
char_start: Some(s),
..
} => s,
_ => panic!("missing char_start"), _ => panic!("missing char_start"),
}; };
assert!( assert!(
@@ -666,11 +665,17 @@ mod tests {
// overlap) is the failure mode. // overlap) is the failure mode.
for w in chunks.windows(2) { for w in chunks.windows(2) {
let prev_start = match w[0].source_spans[0] { let prev_start = match w[0].source_spans[0] {
SourceSpan::Page { char_start: Some(s), .. } => s, SourceSpan::Page {
char_start: Some(s),
..
} => s,
_ => panic!("missing char_start"), _ => panic!("missing char_start"),
}; };
let next_start = match w[1].source_spans[0] { let next_start = match w[1].source_spans[0] {
SourceSpan::Page { char_start: Some(s), .. } => s, SourceSpan::Page {
char_start: Some(s),
..
} => s,
_ => panic!("missing char_start"), _ => panic!("missing char_start"),
}; };
assert!( assert!(
@@ -703,7 +708,7 @@ mod tests {
let page_text = format!("{early_seg}. {tail}"); let page_text = format!("{early_seg}. {tail}");
let doc = make_pdf_doc(&[&page_text]); let doc = make_pdf_doc(&[&page_text]);
let policy = default_policy(500, 80); // target=1500 byte, overlap=240 byte let policy = default_policy(500, 80); // target=1500 byte, overlap=240 byte
let chunks = PdfPageV1Chunker.chunk(&doc, &policy).unwrap(); let chunks = PdfPageV1Chunker.chunk(&doc, &policy).unwrap();
assert!( assert!(

View File

@@ -113,7 +113,14 @@ pub(crate) fn build_chunk(
symbol: Some(symbol.to_string()), symbol: Some(symbol.to_string()),
lang: Some(lang.to_string()), lang: Some(lang.to_string()),
}; };
build_chunk_from_span(doc, chunker_version, base_policy_hash, text, span, split_key) build_chunk_from_span(
doc,
chunker_version,
base_policy_hash,
text,
span,
split_key,
)
} }
/// Like `build_chunk` but emits `symbol: None`. Used by Tier 3 (per spec §9.3). /// Like `build_chunk` but emits `symbol: None`. Used by Tier 3 (per spec §9.3).

View File

@@ -13,9 +13,9 @@ use std::path::PathBuf;
use kebab_chunk::CodeCAstV1Chunker; use kebab_chunk::CodeCAstV1Chunker;
use kebab_core::{ use kebab_core::{
AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
id_for_block, id_for_doc, WorkspacePath, id_for_block, id_for_doc,
}; };
use serde_json::Value; use serde_json::Value;
use time::OffsetDateTime; use time::OffsetDateTime;

View File

@@ -15,9 +15,9 @@ use std::path::PathBuf;
use kebab_chunk::CodeCppAstV1Chunker; use kebab_chunk::CodeCppAstV1Chunker;
use kebab_core::{ use kebab_core::{
AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
id_for_block, id_for_doc, WorkspacePath, id_for_block, id_for_doc,
}; };
use kebab_parse_code::CppAstExtractor; use kebab_parse_code::CppAstExtractor;
use serde_json::Value; use serde_json::Value;
@@ -171,7 +171,9 @@ fn extract_cpp_fixture() -> CanonicalDocument {
workspace_root: &root, workspace_root: &root,
config: &cfg, config: &cfg,
}; };
CppAstExtractor::new().extract(&ctx, src.as_bytes()).unwrap() CppAstExtractor::new()
.extract(&ctx, src.as_bytes())
.unwrap()
} }
// --------------------------------------------------------------------------- // ---------------------------------------------------------------------------
@@ -261,43 +263,61 @@ fn code_cpp_ast_extractor_snapshot() {
let doc = extract_cpp_fixture(); let doc = extract_cpp_fixture();
// Verify the extractor emits all expected named units. // Verify the extractor emits all expected named units.
let block_syms: Vec<Option<String>> = doc.blocks.iter().filter_map(|b| match b { let block_syms: Vec<Option<String>> = doc
Block::Code(c) => match &c.common.source_span { .blocks
SourceSpan::Code { symbol, .. } => Some(symbol.clone()), .iter()
.filter_map(|b| match b {
Block::Code(c) => match &c.common.source_span {
SourceSpan::Code { symbol, .. } => Some(symbol.clone()),
_ => None,
},
_ => None, _ => None,
}, })
_ => None, .collect();
}).collect();
// Must include namespace-qualified class and its methods // Must include namespace-qualified class and its methods
assert!( assert!(
block_syms.iter().any(|s| s.as_deref() == Some("kebab::chunk::MdHeadingV1Chunker")), block_syms
.iter()
.any(|s| s.as_deref() == Some("kebab::chunk::MdHeadingV1Chunker")),
"class unit missing: {block_syms:?}" "class unit missing: {block_syms:?}"
); );
assert!( assert!(
block_syms.iter().any(|s| s.as_deref() == Some("kebab::chunk::MdHeadingV1Chunker::MdHeadingV1Chunker")), block_syms
.iter()
.any(|s| s.as_deref() == Some("kebab::chunk::MdHeadingV1Chunker::MdHeadingV1Chunker")),
"ctor unit missing: {block_syms:?}" "ctor unit missing: {block_syms:?}"
); );
assert!( assert!(
block_syms.iter().any(|s| s.as_deref() == Some("kebab::chunk::MdHeadingV1Chunker::~MdHeadingV1Chunker")), block_syms
.iter()
.any(|s| s.as_deref() == Some("kebab::chunk::MdHeadingV1Chunker::~MdHeadingV1Chunker")),
"dtor unit missing: {block_syms:?}" "dtor unit missing: {block_syms:?}"
); );
assert!( assert!(
block_syms.iter().any(|s| s.as_deref() == Some("kebab::chunk::MdHeadingV1Chunker::chunk_doc")), block_syms
.iter()
.any(|s| s.as_deref() == Some("kebab::chunk::MdHeadingV1Chunker::chunk_doc")),
"chunk_doc unit missing: {block_syms:?}" "chunk_doc unit missing: {block_syms:?}"
); );
assert!( assert!(
block_syms.iter().any(|s| s.as_deref() == Some("kebab::chunk::MdHeadingV1Chunker::operator()")), block_syms
.iter()
.any(|s| s.as_deref() == Some("kebab::chunk::MdHeadingV1Chunker::operator()")),
"operator() unit missing: {block_syms:?}" "operator() unit missing: {block_syms:?}"
); );
// Template function (inside kebab::chunk namespace in the fixture) // Template function (inside kebab::chunk namespace in the fixture)
assert!( assert!(
block_syms.iter().any(|s| s.as_deref() == Some("kebab::chunk::identity")), block_syms
.iter()
.any(|s| s.as_deref() == Some("kebab::chunk::identity")),
"identity template fn unit missing: {block_syms:?}" "identity template fn unit missing: {block_syms:?}"
); );
// Free function in outer namespace // Free function in outer namespace
assert!( assert!(
block_syms.iter().any(|s| s.as_deref() == Some("kebab::global_helper")), block_syms
.iter()
.any(|s| s.as_deref() == Some("kebab::global_helper")),
"global_helper unit missing: {block_syms:?}" "global_helper unit missing: {block_syms:?}"
); );
// Global main // Global main
@@ -312,14 +332,23 @@ fn code_cpp_ast_extractor_snapshot() {
fn code_cpp_ast_extractor_chunks_deterministic() { fn code_cpp_ast_extractor_chunks_deterministic() {
let doc1 = extract_cpp_fixture(); let doc1 = extract_cpp_fixture();
let doc2 = extract_cpp_fixture(); let doc2 = extract_cpp_fixture();
assert_eq!(doc1.blocks, doc2.blocks, "extractor output non-deterministic"); assert_eq!(
doc1.blocks, doc2.blocks,
"extractor output non-deterministic"
);
let policy = fixed_policy(); let policy = fixed_policy();
let chunks1 = CodeCppAstV1Chunker.chunk(&doc1, &policy).unwrap(); let chunks1 = CodeCppAstV1Chunker.chunk(&doc1, &policy).unwrap();
let chunks2 = CodeCppAstV1Chunker.chunk(&doc2, &policy).unwrap(); let chunks2 = CodeCppAstV1Chunker.chunk(&doc2, &policy).unwrap();
assert_eq!( assert_eq!(
chunks1.iter().map(|c| c.chunk_id.0.clone()).collect::<Vec<_>>(), chunks1
chunks2.iter().map(|c| c.chunk_id.0.clone()).collect::<Vec<_>>(), .iter()
.map(|c| c.chunk_id.0.clone())
.collect::<Vec<_>>(),
chunks2
.iter()
.map(|c| c.chunk_id.0.clone())
.collect::<Vec<_>>(),
"chunker output non-deterministic" "chunker output non-deterministic"
); );
} }

View File

@@ -13,9 +13,9 @@ use std::path::PathBuf;
use kebab_chunk::CodeGoAstV1Chunker; use kebab_chunk::CodeGoAstV1Chunker;
use kebab_core::{ use kebab_core::{
AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
id_for_block, id_for_doc, WorkspacePath, id_for_block, id_for_doc,
}; };
use serde_json::Value; use serde_json::Value;
use time::OffsetDateTime; use time::OffsetDateTime;

View File

@@ -13,9 +13,9 @@ use std::path::PathBuf;
use kebab_chunk::CodeJavaAstV1Chunker; use kebab_chunk::CodeJavaAstV1Chunker;
use kebab_core::{ use kebab_core::{
AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
id_for_block, id_for_doc, WorkspacePath, id_for_block, id_for_doc,
}; };
use serde_json::Value; use serde_json::Value;
use time::OffsetDateTime; use time::OffsetDateTime;

View File

@@ -13,9 +13,9 @@ use std::path::PathBuf;
use kebab_chunk::CodeJsAstV1Chunker; use kebab_chunk::CodeJsAstV1Chunker;
use kebab_core::{ use kebab_core::{
AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
id_for_block, id_for_doc, WorkspacePath, id_for_block, id_for_doc,
}; };
use serde_json::Value; use serde_json::Value;
use time::OffsetDateTime; use time::OffsetDateTime;

View File

@@ -13,9 +13,9 @@ use std::path::PathBuf;
use kebab_chunk::CodeKotlinAstV1Chunker; use kebab_chunk::CodeKotlinAstV1Chunker;
use kebab_core::{ use kebab_core::{
AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
id_for_block, id_for_doc, WorkspacePath, id_for_block, id_for_doc,
}; };
use serde_json::Value; use serde_json::Value;
use time::OffsetDateTime; use time::OffsetDateTime;

View File

@@ -13,9 +13,9 @@ use std::path::PathBuf;
use kebab_chunk::CodePythonAstV1Chunker; use kebab_chunk::CodePythonAstV1Chunker;
use kebab_core::{ use kebab_core::{
AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
id_for_block, id_for_doc, WorkspacePath, id_for_block, id_for_doc,
}; };
use serde_json::Value; use serde_json::Value;
use time::OffsetDateTime; use time::OffsetDateTime;

View File

@@ -13,9 +13,9 @@ use std::path::PathBuf;
use kebab_chunk::CodeRustAstV1Chunker; use kebab_chunk::CodeRustAstV1Chunker;
use kebab_core::{ use kebab_core::{
AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
id_for_block, id_for_doc, WorkspacePath, id_for_block, id_for_doc,
}; };
use serde_json::Value; use serde_json::Value;
use time::OffsetDateTime; use time::OffsetDateTime;

View File

@@ -13,9 +13,9 @@ use std::path::PathBuf;
use kebab_chunk::CodeTsAstV1Chunker; use kebab_chunk::CodeTsAstV1Chunker;
use kebab_core::{ use kebab_core::{
AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock, AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath, CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
id_for_block, id_for_doc, WorkspacePath, id_for_block, id_for_doc,
}; };
use serde_json::Value; use serde_json::Value;
use time::OffsetDateTime; use time::OffsetDateTime;

View File

@@ -124,7 +124,11 @@ fn dockerfile_emits_single_chunk() {
Some("<dockerfile>"), Some("<dockerfile>"),
"symbol must be '<dockerfile>'" "symbol must be '<dockerfile>'"
); );
assert_eq!(lang.as_deref(), Some("dockerfile"), "lang must be 'dockerfile'"); assert_eq!(
lang.as_deref(),
Some("dockerfile"),
"lang must be 'dockerfile'"
);
} }
other => panic!("expected SourceSpan::Code, got {other:?}"), other => panic!("expected SourceSpan::Code, got {other:?}"),
} }

View File

@@ -110,13 +110,11 @@ fn k8s_multi_doc_emits_one_chunk_per_resource() {
let symbols: Vec<&str> = chunks let symbols: Vec<&str> = chunks
.iter() .iter()
.map(|c| { .map(|c| match &c.source_spans[0] {
match &c.source_spans[0] { SourceSpan::Code { symbol, .. } => symbol
SourceSpan::Code { symbol, .. } => { .as_deref()
symbol.as_deref().expect("symbol must be Some for k8s chunks") .expect("symbol must be Some for k8s chunks"),
} other => panic!("expected Code span, got {other:?}"),
other => panic!("expected Code span, got {other:?}"),
}
}) })
.collect(); .collect();
@@ -270,7 +268,11 @@ fn k8s_oversize_splits_into_line_windows_sharing_symbol() {
let ranges: Vec<(u32, u32)> = chunks let ranges: Vec<(u32, u32)> = chunks
.iter() .iter()
.map(|c| match &c.source_spans[0] { .map(|c| match &c.source_spans[0] {
SourceSpan::Code { line_start, line_end, .. } => (*line_start, *line_end), SourceSpan::Code {
line_start,
line_end,
..
} => (*line_start, *line_end),
other => panic!("expected Code span, got {other:?}"), other => panic!("expected Code span, got {other:?}"),
}) })
.collect(); .collect();

View File

@@ -15,7 +15,7 @@ use std::path::PathBuf;
use kebab_chunk::MdHeadingV1Chunker; use kebab_chunk::MdHeadingV1Chunker;
use kebab_core::{ use kebab_core::{
AssetId, AssetStorage, Checksum, ChunkPolicy, ChunkerVersion, Chunker, MediaType, AssetId, AssetStorage, Checksum, ChunkPolicy, Chunker, ChunkerVersion, MediaType,
ParserVersion, RawAsset, SourceUri, WorkspacePath, ParserVersion, RawAsset, SourceUri, WorkspacePath,
}; };
use kebab_parse_md::{BodyHints, build_canonical_document, parse_blocks, parse_frontmatter}; use kebab_parse_md::{BodyHints, build_canonical_document, parse_blocks, parse_frontmatter};
@@ -65,8 +65,7 @@ fn long_section_chunks_snapshot() {
Some(span) => bytes[..span.end].iter().filter(|b| **b == b'\n').count() as u32 + 1, Some(span) => bytes[..span.end].iter().filter(|b| **b == b'\n').count() as u32 + 1,
None => 1, None => 1,
}; };
let (blocks, parse_warns) = let (blocks, parse_warns) = parse_blocks(&bytes, body_offset_lines).expect("blocks parse");
parse_blocks(&bytes, body_offset_lines).expect("blocks parse");
// Pin parser_version so doc_id / block_ids are reproducible. // Pin parser_version so doc_id / block_ids are reproducible.
let parser_version = ParserVersion("kb-chunk-snapshot-test-0".into()); let parser_version = ParserVersion("kb-chunk-snapshot-test-0".into());
@@ -74,9 +73,8 @@ fn long_section_chunks_snapshot() {
metadata.aliases.sort(); metadata.aliases.sort();
metadata.tags.sort(); metadata.tags.sort();
let doc = let doc = build_canonical_document(&asset, metadata, blocks, &parser_version, parse_warns)
build_canonical_document(&asset, metadata, blocks, &parser_version, parse_warns) .expect("build_canonical_document");
.expect("build_canonical_document");
// Pin policy so policy_hash and chunk_ids are reproducible. // Pin policy so policy_hash and chunk_ids are reproducible.
let policy = ChunkPolicy { let policy = ChunkPolicy {
@@ -102,8 +100,7 @@ fn long_section_chunks_snapshot() {
baseline_path.display() baseline_path.display()
), ),
}; };
let expected: Value = let expected: Value = serde_json::from_str(&baseline_text).expect("baseline parses as json");
serde_json::from_str(&baseline_text).expect("baseline parses as json");
if actual != expected { if actual != expected {
if std::env::var("UPDATE_SNAPSHOTS").is_ok() { if std::env::var("UPDATE_SNAPSHOTS").is_ok() {
@@ -154,14 +151,8 @@ fn long_section_chunks_are_deterministic() {
let mut metadata = metadata; let mut metadata = metadata;
metadata.aliases.sort(); metadata.aliases.sort();
metadata.tags.sort(); metadata.tags.sort();
let doc = build_canonical_document( let doc = build_canonical_document(&asset, metadata, blocks, &parser_version, parse_warns)
&asset, .expect("build_canonical_document");
metadata,
blocks,
&parser_version,
parse_warns,
)
.expect("build_canonical_document");
let ids: Vec<String> = MdHeadingV1Chunker let ids: Vec<String> = MdHeadingV1Chunker
.chunk(&doc, &policy) .chunk(&doc, &policy)
.unwrap() .unwrap()

View File

@@ -107,9 +107,7 @@ fn cargo_toml_single_chunk_with_toml_lang() {
.unwrap_or_else(|e| panic!("cannot read fixture {}: {e}", fixture_path.display())); .unwrap_or_else(|e| panic!("cannot read fixture {}: {e}", fixture_path.display()));
let doc = manifest_doc("toml", &text); let doc = manifest_doc("toml", &text);
let chunks = ManifestFileV1Chunker let chunks = ManifestFileV1Chunker.chunk(&doc, &policy()).expect("chunk");
.chunk(&doc, &policy())
.expect("chunk");
assert_eq!( assert_eq!(
chunks.len(), chunks.len(),
@@ -149,9 +147,7 @@ fn package_json_single_chunk_with_json_lang() {
.unwrap_or_else(|e| panic!("cannot read fixture {}: {e}", fixture_path.display())); .unwrap_or_else(|e| panic!("cannot read fixture {}: {e}", fixture_path.display()));
let doc = manifest_doc("json", &text); let doc = manifest_doc("json", &text);
let chunks = ManifestFileV1Chunker let chunks = ManifestFileV1Chunker.chunk(&doc, &policy()).expect("chunk");
.chunk(&doc, &policy())
.expect("chunk");
assert_eq!( assert_eq!(
chunks.len(), chunks.len(),
@@ -191,9 +187,7 @@ fn pom_xml_single_chunk_with_xml_lang() {
.unwrap_or_else(|e| panic!("cannot read fixture {}: {e}", fixture_path.display())); .unwrap_or_else(|e| panic!("cannot read fixture {}: {e}", fixture_path.display()));
let doc = manifest_doc("xml", &text); let doc = manifest_doc("xml", &text);
let chunks = ManifestFileV1Chunker let chunks = ManifestFileV1Chunker.chunk(&doc, &policy()).expect("chunk");
.chunk(&doc, &policy())
.expect("chunk");
assert_eq!( assert_eq!(
chunks.len(), chunks.len(),
@@ -233,9 +227,7 @@ fn go_mod_single_chunk_with_go_mod_lang() {
.unwrap_or_else(|e| panic!("cannot read fixture {}: {e}", fixture_path.display())); .unwrap_or_else(|e| panic!("cannot read fixture {}: {e}", fixture_path.display()));
let doc = manifest_doc("go-mod", &text); let doc = manifest_doc("go-mod", &text);
let chunks = ManifestFileV1Chunker let chunks = ManifestFileV1Chunker.chunk(&doc, &policy()).expect("chunk");
.chunk(&doc, &policy())
.expect("chunk");
assert_eq!( assert_eq!(
chunks.len(), chunks.len(),

View File

@@ -179,7 +179,12 @@ enum Cmd {
/// canonical). Repeatable or comma-separated. /// canonical). Repeatable or comma-separated.
/// Examples: `rust`, `python`, `typescript`. /// Examples: `rust`, `python`, `typescript`.
/// Unknown values produce empty hits. /// Unknown values produce empty hits.
#[arg(long = "code-lang", value_name = "LANG", num_args = 1, value_delimiter = ',')] #[arg(
long = "code-lang",
value_name = "LANG",
num_args = 1,
value_delimiter = ','
)]
code_lang: Vec<String>, code_lang: Vec<String>,
/// p9-fb-37: emit pre-fusion lexical / vector / RRF candidate /// p9-fb-37: emit pre-fusion lexical / vector / RRF candidate
@@ -464,7 +469,9 @@ fn parse_bool_env(s: &str) -> Result<bool, String> {
match s.to_ascii_lowercase().as_str() { match s.to_ascii_lowercase().as_str() {
"1" | "true" | "yes" | "on" => Ok(true), "1" | "true" | "yes" | "on" => Ok(true),
"0" | "false" | "no" | "off" => Ok(false), "0" | "false" | "no" | "off" => Ok(false),
other => Err(format!("expected 1/0/true/false/yes/no/on/off, got {other:?}")), other => Err(format!(
"expected 1/0/true/false/yes/no/on/off, got {other:?}"
)),
} }
} }
@@ -551,8 +558,14 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
"created {}", "created {}",
kebab_config::Config::xdg_config_path().display() kebab_config::Config::xdg_config_path().display()
); );
println!("created {}", kebab_config::Config::xdg_data_dir().display()); println!(
println!("created {}", kebab_config::Config::xdg_state_dir().display()); "created {}",
kebab_config::Config::xdg_data_dir().display()
);
println!(
"created {}",
kebab_config::Config::xdg_state_dir().display()
);
println!("hint edit the config above, then `kebab ingest`"); println!("hint edit the config above, then `kebab ingest`");
} }
Ok(()) Ok(())
@@ -565,7 +578,9 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
} => { } => {
let cfg = kebab_config::Config::load(cli.config.as_deref())?; let cfg = kebab_config::Config::load(cli.config.as_deref())?;
let scope = kebab_core::SourceScope { let scope = kebab_core::SourceScope {
root: root.clone().unwrap_or_else(|| PathBuf::from(&cfg.workspace.root)), root: root
.clone()
.unwrap_or_else(|| PathBuf::from(&cfg.workspace.root)),
exclude: cfg.workspace.exclude.clone(), exclude: cfg.workspace.exclude.clone(),
..Default::default() ..Default::default()
}; };
@@ -580,9 +595,8 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
.unwrap_or(false); .unwrap_or(false);
let mode = progress::ProgressMode::from_flags(cli.json, cli.quiet, plain_env); let mode = progress::ProgressMode::from_flags(cli.json, cli.quiet, plain_env);
let (tx, rx) = std::sync::mpsc::channel::<kebab_app::IngestEvent>(); let (tx, rx) = std::sync::mpsc::channel::<kebab_app::IngestEvent>();
let display_handle = std::thread::spawn(move || { let display_handle =
progress::ProgressDisplay::new(mode).run(rx) std::thread::spawn(move || progress::ProgressDisplay::new(mode).run(rx));
});
// p9-fb-04: register a Ctrl-C handler that flips the same // p9-fb-04: register a Ctrl-C handler that flips the same
// AtomicBool the facade polls at each step boundary. The // AtomicBool the facade polls at each step boundary. The
@@ -614,7 +628,8 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
if cli.json { if cli.json {
println!("{}", serde_json::to_string(&wire::wire_ingest(&report))?); println!("{}", serde_json::to_string(&wire::wire_ingest(&report))?);
} else { } else {
let skipped_breakdown = kebab_app::render_skipped_breakdown(&report.skipped_by_extension); let skipped_breakdown =
kebab_app::render_skipped_breakdown(&report.skipped_by_extension);
let purged_suffix = if report.purged_deleted_files > 0 { let purged_suffix = if report.purged_deleted_files > 0 {
format!(" purged {}", report.purged_deleted_files) format!(" purged {}", report.purged_deleted_files)
} else { } else {
@@ -640,7 +655,10 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
let cfg = kebab_config::Config::load(cli.config.as_deref())?; let cfg = kebab_config::Config::load(cli.config.as_deref())?;
let docs = kebab_app::list_docs_with_config(cfg, kebab_core::DocFilter::default())?; let docs = kebab_app::list_docs_with_config(cfg, kebab_core::DocFilter::default())?;
if cli.json { if cli.json {
println!("{}", serde_json::to_string(&wire::wire_doc_summaries(&docs))?); println!(
"{}",
serde_json::to_string(&wire::wire_doc_summaries(&docs))?
);
} else { } else {
for d in &docs { for d in &docs {
println!("{}\t{}", d.doc_id, d.doc_path.0); println!("{}\t{}", d.doc_id, d.doc_path.0);
@@ -667,7 +685,10 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
let cfg = kebab_config::Config::load(cli.config.as_deref())?; let cfg = kebab_config::Config::load(cli.config.as_deref())?;
let chunk_id: kebab_core::ChunkId = id.parse()?; let chunk_id: kebab_core::ChunkId = id.parse()?;
let chunk = kebab_app::inspect_chunk_with_config(cfg, &chunk_id)?; let chunk = kebab_app::inspect_chunk_with_config(cfg, &chunk_id)?;
println!("{}", serde_json::to_string(&wire::wire_chunk_inspection(&chunk))?); println!(
"{}",
serde_json::to_string(&wire::wire_chunk_inspection(&chunk))?
);
Ok(()) Ok(())
} }
}, },
@@ -708,7 +729,10 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
}; };
let result = kebab_app::fetch_with_config(cfg, query, opts)?; let result = kebab_app::fetch_with_config(cfg, query, opts)?;
if cli.json { if cli.json {
println!("{}", serde_json::to_string(&wire::wire_fetch_result(&result))?); println!(
"{}",
serde_json::to_string(&wire::wire_fetch_result(&result))?
);
} else { } else {
render_fetch_plain(&result); render_fetch_plain(&result);
} }
@@ -752,30 +776,21 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
if line.trim().is_empty() { if line.trim().is_empty() {
continue; continue;
} }
let v: serde_json::Value = let v: serde_json::Value = serde_json::from_str(&line).map_err(|e| {
serde_json::from_str(&line).map_err(|e| { anyhow::Error::new(kebab_app::StructuredError(kebab_app::ErrorV1 {
anyhow::Error::new(kebab_app::StructuredError( schema_version: kebab_app::ERROR_V1_ID.to_string(),
kebab_app::ErrorV1 { code: "config_invalid".to_string(),
schema_version: kebab_app::ERROR_V1_ID message: format!("stdin ndjson line {} parse error: {e}", lineno + 1),
.to_string(), details: serde_json::Value::Null,
code: "config_invalid".to_string(), hint: Some(
message: format!( "each line must be a JSON object with at least `query`".to_string(),
"stdin ndjson line {} parse error: {e}", ),
lineno + 1 }))
), })?;
details: serde_json::Value::Null,
hint: Some(
"each line must be a JSON object with at least `query`"
.to_string(),
),
},
))
})?;
raw_items.push(v); raw_items.push(v);
} }
let (items, summary) = let (items, summary) = kebab_app::bulk_search_with_config(cfg, raw_items)?;
kebab_app::bulk_search_with_config(cfg, raw_items)?;
if cli.json { if cli.json {
let mut stdout = std::io::stdout().lock(); let mut stdout = std::io::stdout().lock();
@@ -799,11 +814,7 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
if let Some(err) = &item.error { if let Some(err) = &item.error {
writeln!(stdout, "error: {err}")?; writeln!(stdout, "error: {err}")?;
} else if let Some(resp) = &item.response { } else if let Some(resp) = &item.response {
writeln!( writeln!(stdout, "{}", serde_json::to_string_pretty(resp)?)?;
stdout,
"{}",
serde_json::to_string_pretty(resp)?
)?;
} }
writeln!(stdout)?; writeln!(stdout)?;
} }
@@ -843,8 +854,7 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
other => other.to_string(), other => other.to_string(),
} }
} }
let media_norm: Vec<String> = let media_norm: Vec<String> = media.iter().map(|s| normalize_media_alias(s)).collect();
media.iter().map(|s| normalize_media_alias(s)).collect();
// p9-fb-36: parse --ingested-after as RFC3339; structured error on failure. // p9-fb-36: parse --ingested-after as RFC3339; structured error on failure.
let ingested_after_parsed: Option<time::OffsetDateTime> = let ingested_after_parsed: Option<time::OffsetDateTime> =
@@ -856,8 +866,8 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
) { ) {
Ok(ts) => Some(ts), Ok(ts) => Some(ts),
Err(e) => { Err(e) => {
return Err(anyhow::Error::new( return Err(anyhow::Error::new(kebab_app::StructuredError(
kebab_app::StructuredError(kebab_app::ErrorV1 { kebab_app::ErrorV1 {
schema_version: kebab_app::ERROR_V1_ID.to_string(), schema_version: kebab_app::ERROR_V1_ID.to_string(),
code: "config_invalid".to_string(), code: "config_invalid".to_string(),
message: format!( message: format!(
@@ -867,8 +877,8 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
hint: Some( hint: Some(
"expected format like 2026-04-01T00:00:00Z".to_string(), "expected format like 2026-04-01T00:00:00Z".to_string(),
), ),
}), },
)); )));
} }
} }
} }
@@ -943,11 +953,7 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
}; };
println!( println!(
"{:>2}. {:.4} {}{}{}", "{:>2}. {:.4} {}{}{}",
h.rank, h.rank, h.retrieval.fusion_score, stale_tag, h.doc_path.0, heading,
h.retrieval.fusion_score,
stale_tag,
h.doc_path.0,
heading,
); );
} }
// p9-fb-34: truncation hint goes to stderr so it // p9-fb-34: truncation hint goes to stderr so it
@@ -969,15 +975,33 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
if let Some(t) = &resp.trace { if let Some(t) = &resp.trace {
eprintln!(); eprintln!();
eprintln!("Trace:"); eprintln!("Trace:");
eprintln!(" lexical ({} hits, {}ms):", t.lexical.len(), t.timing.lexical_ms); eprintln!(
" lexical ({} hits, {}ms):",
t.lexical.len(),
t.timing.lexical_ms
);
for c in t.lexical.iter().take(3) { for c in t.lexical.iter().take(3) {
eprintln!(" rank={} score={:.4} chunk={}", c.rank, c.score, c.chunk_id.0); eprintln!(
" rank={} score={:.4} chunk={}",
c.rank, c.score, c.chunk_id.0
);
} }
eprintln!(" vector ({} hits, {}ms):", t.vector.len(), t.timing.vector_ms); eprintln!(
" vector ({} hits, {}ms):",
t.vector.len(),
t.timing.vector_ms
);
for c in t.vector.iter().take(3) { for c in t.vector.iter().take(3) {
eprintln!(" rank={} score={:.4} chunk={}", c.rank, c.score, c.chunk_id.0); eprintln!(
" rank={} score={:.4} chunk={}",
c.rank, c.score, c.chunk_id.0
);
} }
eprintln!(" fusion ({} inputs, {}ms)", t.rrf_inputs.len(), t.timing.fusion_ms); eprintln!(
" fusion ({} inputs, {}ms)",
t.rrf_inputs.len(),
t.timing.fusion_ms
);
eprintln!(" total: {}ms", t.timing.total_ms); eprintln!(" total: {}ms", t.timing.total_ms);
} }
} }
@@ -1039,16 +1063,12 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
let cfg2 = cfg.clone(); let cfg2 = cfg.clone();
let q = query.clone(); let q = query.clone();
let session2 = session.clone(); let session2 = session.clone();
let handle = std::thread::spawn( let handle = std::thread::spawn(move || -> anyhow::Result<kebab_core::Answer> {
move || -> anyhow::Result<kebab_core::Answer> { match session2.as_deref() {
match session2.as_deref() { Some(sid) => kebab_app::ask_with_session_with_config(cfg2, sid, &q, opts),
Some(sid) => kebab_app::ask_with_session_with_config( None => kebab_app::ask_with_config(cfg2, &q, opts),
cfg2, sid, &q, opts, }
), });
None => kebab_app::ask_with_config(cfg2, &q, opts),
}
},
);
// Drain receiver, write ndjson to stderr until // Drain receiver, write ndjson to stderr until
// completion or BrokenPipe. // completion or BrokenPipe.
@@ -1324,9 +1344,18 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
println!("{}", serde_json::to_string_pretty(&agg)?); println!("{}", serde_json::to_string_pretty(&agg)?);
} else { } else {
println!("run_id: {run_id}"); println!("run_id: {run_id}");
println!("queries: {} ({} failed)", agg.total_queries, agg.failed_queries); println!(
println!("hit@1: {:.4}", agg.hit_at_k.get(&1).copied().unwrap_or(0.0)); "queries: {} ({} failed)",
println!("hit@5: {:.4}", agg.hit_at_k.get(&5).copied().unwrap_or(0.0)); agg.total_queries, agg.failed_queries
);
println!(
"hit@1: {:.4}",
agg.hit_at_k.get(&1).copied().unwrap_or(0.0)
);
println!(
"hit@5: {:.4}",
agg.hit_at_k.get(&5).copied().unwrap_or(0.0)
);
println!("MRR: {:.4}", agg.mrr); println!("MRR: {:.4}", agg.mrr);
} }
Ok(()) Ok(())
@@ -1376,8 +1405,12 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
} else { } else {
println!( println!(
"ingest-file: scanned={} new={} updated={} unchanged={} skipped={} errors={}", "ingest-file: scanned={} new={} updated={} unchanged={} skipped={} errors={}",
report.scanned, report.new, report.updated, report.scanned,
report.unchanged, report.skipped, report.errors report.new,
report.updated,
report.unchanged,
report.skipped,
report.errors
); );
} }
Ok(()) Ok(())
@@ -1390,20 +1423,20 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
.read_to_string(&mut body) .read_to_string(&mut body)
.context("kebab ingest-stdin: read stdin")?; .context("kebab ingest-stdin: read stdin")?;
let cfg = kebab_config::Config::load(cli.config.as_deref())?; let cfg = kebab_config::Config::load(cli.config.as_deref())?;
let report = kebab_app::ingest_stdin_with_config( let report =
cfg, kebab_app::ingest_stdin_with_config(cfg, &body, title, source_uri.as_deref())?;
&body,
title,
source_uri.as_deref(),
)?;
if cli.json { if cli.json {
let v = wire::wire_ingest(&report); let v = wire::wire_ingest(&report);
println!("{}", serde_json::to_string(&v)?); println!("{}", serde_json::to_string(&v)?);
} else { } else {
println!( println!(
"ingest-stdin: scanned={} new={} updated={} unchanged={} skipped={} errors={}", "ingest-stdin: scanned={} new={} updated={} unchanged={} skipped={} errors={}",
report.scanned, report.new, report.updated, report.scanned,
report.unchanged, report.skipped, report.errors report.new,
report.updated,
report.unchanged,
report.skipped,
report.errors
); );
} }
Ok(()) Ok(())
@@ -1432,10 +1465,7 @@ fn render_ask_plain_citations(
writeln!(w)?; writeln!(w)?;
writeln!(w, "근거:")?; writeln!(w, "근거:")?;
for (idx, c) in ans.citations.iter().enumerate() { for (idx, c) in ans.citations.iter().enumerate() {
let marker = c let marker = c.marker.clone().unwrap_or_else(|| format!("{}", idx + 1));
.marker
.clone()
.unwrap_or_else(|| format!("{}", idx + 1));
// p9-fb-32: `[stale]` prefix on the URI for citations whose // p9-fb-32: `[stale]` prefix on the URI for citations whose
// `stale: true`. Yellow on TTY, plain otherwise — mirrors the // `stale: true`. Yellow on TTY, plain otherwise — mirrors the
// search-plain renderer in `Cmd::Search`. // search-plain renderer in `Cmd::Search`.
@@ -1496,7 +1526,10 @@ fn print_schema_text(s: &kebab_app::SchemaV1) {
println!(" parser_version {}", s.models.parser_version); println!(" parser_version {}", s.models.parser_version);
println!(" chunker_version {}", s.models.chunker_version); println!(" chunker_version {}", s.models.chunker_version);
println!(" embedding_version {}", s.models.embedding_version); println!(" embedding_version {}", s.models.embedding_version);
println!(" prompt_template_version {}", s.models.prompt_template_version); println!(
" prompt_template_version {}",
s.models.prompt_template_version
);
println!(" index_version {}", s.models.index_version); println!(" index_version {}", s.models.index_version);
println!(" corpus_revision {}", s.models.corpus_revision); println!(" corpus_revision {}", s.models.corpus_revision);
println!(); println!();
@@ -1545,9 +1578,7 @@ fn confirm_destructive(
/// Confirm prompt for `--orphans-only`: shows the orphan count + a /// Confirm prompt for `--orphans-only`: shows the orphan count + a
/// sample of up to 5 paths so the user knows what will be purged before /// sample of up to 5 paths so the user knows what will be purged before
/// committing. No filesystem paths are removed — only store records. /// committing. No filesystem paths are removed — only store records.
fn confirm_orphans_only( fn confirm_orphans_only(orphan_paths: &[kebab_core::WorkspacePath]) -> anyhow::Result<bool> {
orphan_paths: &[kebab_core::WorkspacePath],
) -> anyhow::Result<bool> {
use std::io::Write; use std::io::Write;
let n = orphan_paths.len(); let n = orphan_paths.len();
let mut out = std::io::stderr().lock(); let mut out = std::io::stderr().lock();
@@ -1560,11 +1591,7 @@ fn confirm_orphans_only(
return Ok(true); return Ok(true);
} }
let sample: Vec<&str> = orphan_paths let sample: Vec<&str> = orphan_paths.iter().take(5).map(|p| p.0.as_str()).collect();
.iter()
.take(5)
.map(|p| p.0.as_str())
.collect();
let sample_str = sample.join(", "); let sample_str = sample.join(", ");
let ellipsis = if n > 5 { ", …" } else { "" }; let ellipsis = if n > 5 { ", …" } else { "" };
@@ -1593,19 +1620,28 @@ fn render_fetch_plain(r: &kebab_core::FetchResult) {
if !r.context_before.is_empty() { if !r.context_before.is_empty() {
println!("\n=== before ==="); println!("\n=== before ===");
for c in &r.context_before { for c in &r.context_before {
let heading = c.heading_path.last().map_or("", std::string::String::as_str); let heading = c
.heading_path
.last()
.map_or("", std::string::String::as_str);
println!("[{} § {}]\n{}\n", c.chunk_id.0, heading, c.text); println!("[{} § {}]\n{}\n", c.chunk_id.0, heading, c.text);
} }
} }
if let Some(c) = &r.chunk { if let Some(c) = &r.chunk {
println!("\n=== target ==="); println!("\n=== target ===");
let heading = c.heading_path.last().map_or("", std::string::String::as_str); let heading = c
.heading_path
.last()
.map_or("", std::string::String::as_str);
println!("[{} § {}]\n{}\n", c.chunk_id.0, heading, c.text); println!("[{} § {}]\n{}\n", c.chunk_id.0, heading, c.text);
} }
if !r.context_after.is_empty() { if !r.context_after.is_empty() {
println!("\n=== after ==="); println!("\n=== after ===");
for c in &r.context_after { for c in &r.context_after {
let heading = c.heading_path.last().map_or("", std::string::String::as_str); let heading = c
.heading_path
.last()
.map_or("", std::string::String::as_str);
println!("[{} § {}]\n{}\n", c.chunk_id.0, heading, c.text); println!("[{} § {}]\n{}\n", c.chunk_id.0, heading, c.text);
} }
} }
@@ -1637,8 +1673,8 @@ mod tests {
//! against a synthetic `Answer` instead. //! against a synthetic `Answer` instead.
use super::*; use super::*;
use kebab_core::{ use kebab_core::{
Answer, AnswerCitation, AnswerRetrievalSummary, Citation, ModelRef, Answer, AnswerCitation, AnswerRetrievalSummary, Citation, ModelRef, PromptTemplateVersion,
PromptTemplateVersion, SearchMode, TokenUsage, TraceId, WorkspacePath, SearchMode, TokenUsage, TraceId, WorkspacePath,
}; };
use time::OffsetDateTime; use time::OffsetDateTime;
@@ -1734,4 +1770,3 @@ mod tests {
); );
} }
} }

View File

@@ -124,11 +124,9 @@ impl ProgressDisplay {
bar.set_length(u64::from(*total)); bar.set_length(u64::from(*total));
bar.set_position(0); bar.set_position(0);
bar.set_style( bar.set_style(
ProgressStyle::with_template( ProgressStyle::with_template("ingest [{bar:30}] {pos}/{len} {wide_msg}")
"ingest [{bar:30}] {pos}/{len} {wide_msg}", .unwrap()
) .progress_chars("=> "),
.unwrap()
.progress_chars("=> "),
); );
bar.set_message(""); bar.set_message("");
} }
@@ -170,11 +168,7 @@ impl ProgressDisplay {
let _ = writeln!( let _ = writeln!(
err, err,
"ingest: complete (scanned={} new={} updated={} skipped={} errors={})", "ingest: complete (scanned={} new={} updated={} skipped={} errors={})",
counts.scanned, counts.scanned, counts.new, counts.updated, counts.skipped, counts.errors,
counts.new,
counts.updated,
counts.skipped,
counts.errors,
); );
} }
} }
@@ -193,11 +187,7 @@ impl ProgressDisplay {
let _ = writeln!( let _ = writeln!(
err, err,
"ingest: aborted (scanned={} new={} updated={} skipped={} errors={})", "ingest: aborted (scanned={} new={} updated={} skipped={} errors={})",
counts.scanned, counts.scanned, counts.new, counts.updated, counts.skipped, counts.errors,
counts.new,
counts.updated,
counts.skipped,
counts.errors,
); );
} }
} }
@@ -210,13 +200,26 @@ impl ProgressDisplay {
let _ = writeln!(err, " 📷 OCR page {page}..."); let _ = writeln!(err, " 📷 OCR page {page}...");
} }
} }
IngestEvent::PdfOcrFinished { page, ms, chars, ocr_engine, skipped, .. } => { IngestEvent::PdfOcrFinished {
page,
ms,
chars,
ocr_engine,
skipped,
..
} => {
if !quiet { if !quiet {
let mut err = std::io::stderr().lock(); let mut err = std::io::stderr().lock();
if *skipped { if *skipped {
let _ = writeln!(err, " ⊘ OCR page {page} skipped (no DCTDecode or engine fail, {ms}ms)"); let _ = writeln!(
err,
" ⊘ OCR page {page} skipped (no DCTDecode or engine fail, {ms}ms)"
);
} else { } else {
let _ = writeln!(err, " ✓ OCR page {page} ({chars} chars, {ms}ms via {ocr_engine})"); let _ = writeln!(
err,
" ✓ OCR page {page} ({chars} chars, {ms}ms via {ocr_engine})"
);
} }
} }
} }
@@ -250,7 +253,10 @@ mod tests {
#[test] #[test]
fn from_flags_json_takes_priority_over_tty() { fn from_flags_json_takes_priority_over_tty() {
assert_eq!(ProgressMode::from_flags(true, false, false), ProgressMode::Json); assert_eq!(
ProgressMode::from_flags(true, false, false),
ProgressMode::Json
);
} }
#[test] #[test]

View File

@@ -114,10 +114,7 @@ pub fn wire_answer(a: &Answer) -> Value {
/// The timestamp is added at emit time (caller fills `ts`), since the /// The timestamp is added at emit time (caller fills `ts`), since the
/// pipeline doesn't carry one in the in-process enum — mirrors the /// pipeline doesn't carry one in the in-process enum — mirrors the
/// `wire_ingest_progress` pattern (§2 ingest_progress.v1). /// `wire_ingest_progress` pattern (§2 ingest_progress.v1).
pub fn wire_answer_event( pub fn wire_answer_event(ev: &kebab_app::StreamEvent, ts: time::OffsetDateTime) -> Value {
ev: &kebab_app::StreamEvent,
ts: time::OffsetDateTime,
) -> Value {
let mut v = serde_json::to_value(ev).expect("StreamEvent serializes"); let mut v = serde_json::to_value(ev).expect("StreamEvent serializes");
let ts_str = ts let ts_str = ts
.format(&time::format_description::well_known::Rfc3339) .format(&time::format_description::well_known::Rfc3339)
@@ -161,9 +158,7 @@ pub fn wire_reset(r: &kebab_app::ResetReport) -> Value {
/// wall-clock — the emit site is the only place that knows the moment /// wall-clock — the emit site is the only place that knows the moment
/// of emission, so the timestamp is stamped here rather than carried /// of emission, so the timestamp is stamped here rather than carried
/// on the event itself. /// on the event itself.
pub fn wire_ingest_progress( pub fn wire_ingest_progress(event: &kebab_app::IngestEvent) -> anyhow::Result<Value> {
event: &kebab_app::IngestEvent,
) -> anyhow::Result<Value> {
let mut v = serde_json::to_value(event)?; let mut v = serde_json::to_value(event)?;
if let Value::Object(ref mut map) = v { if let Value::Object(ref mut map) = v {
map.insert( map.insert(
@@ -305,15 +300,15 @@ mod tests {
let v = wire_search_response(&r); let v = wire_search_response(&r);
assert_eq!(schema_of(&v), Some("search_response.v1")); assert_eq!(schema_of(&v), Some("search_response.v1"));
assert!(v.get("hits").and_then(|h| h.as_array()).is_some()); assert!(v.get("hits").and_then(|h| h.as_array()).is_some());
assert_eq!( assert_eq!(v.get("hits").and_then(|h| h.as_array()).unwrap().len(), 0);
v.get("hits").and_then(|h| h.as_array()).unwrap().len(),
0
);
assert_eq!( assert_eq!(
v.get("next_cursor").and_then(|c| c.as_str()), v.get("next_cursor").and_then(|c| c.as_str()),
Some("opaque-cursor-abc") Some("opaque-cursor-abc")
); );
assert_eq!(v.get("truncated").and_then(serde_json::Value::as_bool), Some(true)); assert_eq!(
v.get("truncated").and_then(serde_json::Value::as_bool),
Some(true)
);
} }
#[test] #[test]
@@ -322,12 +317,21 @@ mod tests {
let schema = SchemaV1 { let schema = SchemaV1 {
schema_version: "schema.v1".to_string(), schema_version: "schema.v1".to_string(),
kebab_version: "0.2.1".to_string(), kebab_version: "0.2.1".to_string(),
wire: WireBlock { schemas: vec!["answer.v1".to_string()] }, wire: WireBlock {
schemas: vec!["answer.v1".to_string()],
},
capabilities: Capabilities { capabilities: Capabilities {
json_mode: true, ingest_progress: true, ingest_cancellation: true, json_mode: true,
rag_multi_turn: true, search_cache: true, incremental_ingest: true, ingest_progress: true,
streaming_ask: false, http_daemon: false, mcp_server: false, ingest_cancellation: true,
single_file_ingest: false, bulk_search: true, rag_multi_turn: true,
search_cache: true,
incremental_ingest: true,
streaming_ask: false,
http_daemon: false,
mcp_server: false,
single_file_ingest: false,
bulk_search: true,
}, },
models: Models { models: Models {
parser_version: "x".to_string(), parser_version: "x".to_string(),
@@ -340,7 +344,9 @@ mod tests {
corpus_revision: 7, corpus_revision: 7,
}, },
stats: Stats { stats: Stats {
doc_count: 1, chunk_count: 2, asset_count: 1, doc_count: 1,
chunk_count: 2,
asset_count: 1,
last_ingest_at: None, last_ingest_at: None,
media_breakdown: Default::default(), media_breakdown: Default::default(),
lang_breakdown: Default::default(), lang_breakdown: Default::default(),
@@ -352,7 +358,10 @@ mod tests {
}; };
let v = wire_schema(&schema); let v = wire_schema(&schema);
assert_eq!(schema_of(&v), Some("schema.v1")); assert_eq!(schema_of(&v), Some("schema.v1"));
assert_eq!(v.get("kebab_version").and_then(Value::as_str), Some("0.2.1")); assert_eq!(
v.get("kebab_version").and_then(Value::as_str),
Some("0.2.1")
);
} }
#[test] #[test]
@@ -367,7 +376,10 @@ mod tests {
}; };
let v = wire_error_v1(&err); let v = wire_error_v1(&err);
assert_eq!(schema_of(&v), Some("error.v1")); assert_eq!(schema_of(&v), Some("error.v1"));
assert_eq!(v.get("code").and_then(Value::as_str), Some("config_invalid")); assert_eq!(
v.get("code").and_then(Value::as_str),
Some("config_invalid")
);
} }
#[test] #[test]
@@ -393,8 +405,10 @@ mod tests {
#[test] #[test]
fn search_response_with_trace_serializes_trace_field() { fn search_response_with_trace_serializes_trace_field() {
use kebab_core::{SearchTrace, TraceCandidate, TraceFusionInput, use kebab_core::{
TraceTiming, ChunkId, DocumentId, WorkspacePath}; ChunkId, DocumentId, SearchTrace, TraceCandidate, TraceFusionInput, TraceTiming,
WorkspacePath,
};
let r = kebab_app::SearchResponse { let r = kebab_app::SearchResponse {
hits: vec![], hits: vec![],
next_cursor: None, next_cursor: None,
@@ -414,7 +428,12 @@ mod tests {
vector_rank: None, vector_rank: None,
fusion_score: 0.0, fusion_score: 0.0,
}], }],
timing: TraceTiming { lexical_ms: 5, vector_ms: 0, fusion_ms: 1, total_ms: 7 }, timing: TraceTiming {
lexical_ms: 5,
vector_ms: 0,
fusion_ms: 1,
total_ms: 7,
},
}), }),
hint: None, hint: None,
}; };

View File

@@ -2,15 +2,18 @@
//! must fail with exit≠0 and error.v1 code=config_not_found (not silently fall //! must fail with exit≠0 and error.v1 code=config_not_found (not silently fall
//! back to XDG defaults). //! back to XDG defaults).
use std::process::Command;
use serde_json::Value; use serde_json::Value;
use std::process::Command;
fn kebab_bin() -> String { fn kebab_bin() -> String {
env!("CARGO_BIN_EXE_kebab").to_string() env!("CARGO_BIN_EXE_kebab").to_string()
} }
fn parse_error_v1(stderr: &str) -> Value { fn parse_error_v1(stderr: &str) -> Value {
let last = stderr.lines().last().expect("expected error.v1 ndjson on stderr"); let last = stderr
.lines()
.last()
.expect("expected error.v1 ndjson on stderr");
serde_json::from_str(last) serde_json::from_str(last)
.unwrap_or_else(|e| panic!("expected ndjson on stderr: {e}\nstderr={stderr}")) .unwrap_or_else(|e| panic!("expected ndjson on stderr: {e}\nstderr={stderr}"))
} }
@@ -25,7 +28,11 @@ fn invalid_config_path_emits_error_v1_with_nonzero_exit() {
.output() .output()
.expect("spawn kebab"); .expect("spawn kebab");
assert_ne!(out.status.code(), Some(0), "exit must be nonzero on missing --config"); assert_ne!(
out.status.code(),
Some(0),
"exit must be nonzero on missing --config"
);
let stderr = String::from_utf8_lossy(&out.stderr); let stderr = String::from_utf8_lossy(&out.stderr);
let v = parse_error_v1(&stderr); let v = parse_error_v1(&stderr);
assert_eq!(v["schema_version"], "error.v1"); assert_eq!(v["schema_version"], "error.v1");
@@ -38,7 +45,13 @@ fn invalid_relative_config_path_emits_config_not_found() {
// Bug #10 spec §6 R-1: relative path も cwd-relative で cover. // Bug #10 spec §6 R-1: relative path も cwd-relative で cover.
let tmp = tempfile::tempdir().unwrap(); let tmp = tempfile::tempdir().unwrap();
let out = Command::new(kebab_bin()) let out = Command::new(kebab_bin())
.args(["search", "rust", "--config", "nonexistent-rel.toml", "--json"]) .args([
"search",
"rust",
"--config",
"nonexistent-rel.toml",
"--json",
])
.current_dir(tmp.path()) .current_dir(tmp.path())
.output() .output()
.expect("spawn kebab"); .expect("spawn kebab");

View File

@@ -1,15 +1,18 @@
//! Integration tests for Bug #14: empty or whitespace-only query must emit //! Integration tests for Bug #14: empty or whitespace-only query must emit
//! error.v1 code=invalid_input and exit nonzero (not silent 0-hit return). //! error.v1 code=invalid_input and exit nonzero (not silent 0-hit return).
use std::process::Command;
use serde_json::Value; use serde_json::Value;
use std::process::Command;
fn kebab_bin() -> String { fn kebab_bin() -> String {
env!("CARGO_BIN_EXE_kebab").to_string() env!("CARGO_BIN_EXE_kebab").to_string()
} }
fn parse_error_v1(stderr: &str) -> Value { fn parse_error_v1(stderr: &str) -> Value {
let last = stderr.lines().last().expect("expected error.v1 ndjson on stderr"); let last = stderr
.lines()
.last()
.expect("expected error.v1 ndjson on stderr");
serde_json::from_str(last) serde_json::from_str(last)
.unwrap_or_else(|e| panic!("expected ndjson on stderr: {e}\nstderr={stderr}")) .unwrap_or_else(|e| panic!("expected ndjson on stderr: {e}\nstderr={stderr}"))
} }

View File

@@ -36,12 +36,7 @@ fn json_mode_emits_error_v1_on_config_invalid() {
std::fs::write(&bad_config, b"this is not { valid toml !!!").unwrap(); std::fs::write(&bad_config, b"this is not { valid toml !!!").unwrap();
let mut cmd = Command::new(kebab_bin()); let mut cmd = Command::new(kebab_bin());
cmd.args([ cmd.args(["--json", "--config", bad_config.to_str().unwrap(), "ingest"]);
"--json",
"--config",
bad_config.to_str().unwrap(),
"ingest",
]);
for (k, v) in xdg_envs(tmp.path()) { for (k, v) in xdg_envs(tmp.path()) {
cmd.env(k, v); cmd.env(k, v);
} }
@@ -55,7 +50,10 @@ fn json_mode_emits_error_v1_on_config_invalid() {
assert_eq!(exit_code, 2, "expected exit code 2, got {exit_code}"); assert_eq!(exit_code, 2, "expected exit code 2, got {exit_code}");
let stderr = String::from_utf8(out.stderr).unwrap(); let stderr = String::from_utf8(out.stderr).unwrap();
let first_line = stderr.lines().next().expect("stderr must have at least one line"); let first_line = stderr
.lines()
.next()
.expect("stderr must have at least one line");
let v: serde_json::Value = let v: serde_json::Value =
serde_json::from_str(first_line).expect("stderr first line must be valid JSON"); serde_json::from_str(first_line).expect("stderr first line must be valid JSON");

View File

@@ -72,21 +72,34 @@ max_context_tokens = 8000
workspace = workspace.display(), workspace = workspace.display(),
data = data.display(), data = data.display(),
), ),
).unwrap(); )
.unwrap();
let src = dir.path().join("doc.md"); let src = dir.path().join("doc.md");
fs::write(&src, "# A\n\nbody.").unwrap(); fs::write(&src, "# A\n\nbody.").unwrap();
let bin = env!("CARGO_BIN_EXE_kebab"); let bin = env!("CARGO_BIN_EXE_kebab");
let out = Command::new(bin) let out = Command::new(bin)
.args(["--json", "--config", cfg_path.to_str().unwrap(), "ingest-file"]) .args([
"--json",
"--config",
cfg_path.to_str().unwrap(),
"ingest-file",
])
.arg(&src) .arg(&src)
.output() .output()
.unwrap(); .unwrap();
assert!(out.status.success(), "stderr: {}", String::from_utf8_lossy(&out.stderr)); assert!(
out.status.success(),
"stderr: {}",
String::from_utf8_lossy(&out.stderr)
);
let stdout = String::from_utf8_lossy(&out.stdout); let stdout = String::from_utf8_lossy(&out.stdout);
let v: serde_json::Value = serde_json::from_str(stdout.trim()).unwrap(); let v: serde_json::Value = serde_json::from_str(stdout.trim()).unwrap();
assert_eq!(v.get("schema_version").and_then(|s| s.as_str()), Some("ingest_report.v1")); assert_eq!(
v.get("schema_version").and_then(|s| s.as_str()),
Some("ingest_report.v1")
);
assert_eq!(v.get("new").and_then(serde_json::Value::as_u64), Some(1)); assert_eq!(v.get("new").and_then(serde_json::Value::as_u64), Some(1));
} }

View File

@@ -73,13 +73,18 @@ max_context_tokens = 8000
workspace = workspace.display(), workspace = workspace.display(),
data = data.display(), data = data.display(),
), ),
).unwrap(); )
.unwrap();
let bin = env!("CARGO_BIN_EXE_kebab"); let bin = env!("CARGO_BIN_EXE_kebab");
let mut child = Command::new(bin) let mut child = Command::new(bin)
.args([ .args([
"--json", "--config", cfg_path.to_str().unwrap(), "--json",
"ingest-stdin", "--title", "X", "--config",
cfg_path.to_str().unwrap(),
"ingest-stdin",
"--title",
"X",
]) ])
.stdin(Stdio::piped()) .stdin(Stdio::piped())
.stdout(Stdio::piped()) .stdout(Stdio::piped())
@@ -91,10 +96,17 @@ max_context_tokens = 8000
stdin.write_all(b"## Body\n\nbody text.\n").unwrap(); stdin.write_all(b"## Body\n\nbody text.\n").unwrap();
} }
let out = child.wait_with_output().unwrap(); let out = child.wait_with_output().unwrap();
assert!(out.status.success(), "stderr: {}", String::from_utf8_lossy(&out.stderr)); assert!(
out.status.success(),
"stderr: {}",
String::from_utf8_lossy(&out.stderr)
);
let stdout = String::from_utf8_lossy(&out.stdout); let stdout = String::from_utf8_lossy(&out.stdout);
let v: serde_json::Value = serde_json::from_str(stdout.trim()).unwrap(); let v: serde_json::Value = serde_json::from_str(stdout.trim()).unwrap();
assert_eq!(v.get("schema_version").and_then(|s| s.as_str()), Some("ingest_report.v1")); assert_eq!(
v.get("schema_version").and_then(|s| s.as_str()),
Some("ingest_report.v1")
);
assert_eq!(v.get("new").and_then(serde_json::Value::as_u64), Some(1)); assert_eq!(v.get("new").and_then(serde_json::Value::as_u64), Some(1));
} }

View File

@@ -112,7 +112,13 @@ fn kebab_readonly_env_blocks_ingest() {
fn readonly_json_mode_emits_error_v1() { fn readonly_json_mode_emits_error_v1() {
let (tmp, ws) = fixture_workspace(); let (tmp, ws) = fixture_workspace();
let out = Command::new(kebab_bin()) let out = Command::new(kebab_bin())
.args(["--readonly", "--json", "ingest", "--root", ws.to_str().unwrap()]) .args([
"--readonly",
"--json",
"ingest",
"--root",
ws.to_str().unwrap(),
])
.envs(xdg_envs(tmp.path())) .envs(xdg_envs(tmp.path()))
.output() .output()
.unwrap(); .unwrap();
@@ -164,12 +170,22 @@ fn quiet_flag_suppresses_progress_stderr() {
fn quiet_with_json_stdout_has_report_stderr_is_empty() { fn quiet_with_json_stdout_has_report_stderr_is_empty() {
let (tmp, ws) = fixture_workspace(); let (tmp, ws) = fixture_workspace();
let out = Command::new(kebab_bin()) let out = Command::new(kebab_bin())
.args(["--quiet", "--json", "ingest", "--root", ws.to_str().unwrap()]) .args([
"--quiet",
"--json",
"ingest",
"--root",
ws.to_str().unwrap(),
])
.envs(xdg_envs(tmp.path())) .envs(xdg_envs(tmp.path()))
.output() .output()
.unwrap(); .unwrap();
assert!(out.status.success(), "stderr: {}", String::from_utf8_lossy(&out.stderr)); assert!(
out.status.success(),
"stderr: {}",
String::from_utf8_lossy(&out.stderr)
);
let stderr = String::from_utf8_lossy(&out.stderr); let stderr = String::from_utf8_lossy(&out.stderr);
assert!(stderr.is_empty(), "expected empty stderr, got: {stderr}"); assert!(stderr.is_empty(), "expected empty stderr, got: {stderr}");
let stdout = String::from_utf8_lossy(&out.stdout); let stdout = String::from_utf8_lossy(&out.stdout);

View File

@@ -90,12 +90,7 @@ fn ingest_human_non_tty_emits_progress_lines_to_stderr() {
// target is `hidden` and progress lines go to stderr instead. // target is `hidden` and progress lines go to stderr instead.
let (tmp, ws) = fixture_workspace(); let (tmp, ws) = fixture_workspace();
let mut cmd = Command::new(kebab_bin()); let mut cmd = Command::new(kebab_bin());
cmd.args([ cmd.args(["ingest", "--root", ws.to_str().unwrap(), "--summary-only"]);
"ingest",
"--root",
ws.to_str().unwrap(),
"--summary-only",
]);
for (k, v) in xdg_envs(tmp.path()) { for (k, v) in xdg_envs(tmp.path()) {
cmd.env(k, v); cmd.env(k, v);
} }
@@ -155,8 +150,14 @@ fn ingest_json_progress_lines_carry_kind_and_ts() {
saw_completed = true; saw_completed = true;
// Counts mirror the report. // Counts mirror the report.
let counts = v.get("counts").unwrap(); let counts = v.get("counts").unwrap();
assert_eq!(counts.get("scanned").and_then(serde_json::Value::as_u64), Some(2)); assert_eq!(
assert_eq!(counts.get("new").and_then(serde_json::Value::as_u64), Some(2)); counts.get("scanned").and_then(serde_json::Value::as_u64),
Some(2)
);
assert_eq!(
counts.get("new").and_then(serde_json::Value::as_u64),
Some(2)
);
} }
} }
assert!(saw_scan_started, "missing scan_started event"); assert!(saw_scan_started, "missing scan_started event");

View File

@@ -50,9 +50,18 @@ fn reset_data_only_yes_removes_data_dir_and_keeps_config() {
); );
assert!(!xdg_data.join("kebab").exists(), "data dir should be gone"); assert!(!xdg_data.join("kebab").exists(), "data dir should be gone");
assert!(!xdg_cache.join("kebab").exists(), "cache dir should be gone"); assert!(
assert!(!xdg_state.join("kebab").exists(), "state dir should be gone"); !xdg_cache.join("kebab").exists(),
assert!(xdg_cfg.join("kebab/marker").exists(), "config dir preserved"); "cache dir should be gone"
);
assert!(
!xdg_state.join("kebab").exists(),
"state dir should be gone"
);
assert!(
xdg_cfg.join("kebab/marker").exists(),
"config dir preserved"
);
} }
#[test] #[test]
@@ -101,7 +110,11 @@ fn reset_data_only_yes_json_emits_reset_report_v1() {
.env("XDG_STATE_HOME", tmp.path().join("state")) .env("XDG_STATE_HOME", tmp.path().join("state"))
.output() .output()
.unwrap(); .unwrap();
assert!(out.status.success(), "stderr: {}", String::from_utf8_lossy(&out.stderr)); assert!(
out.status.success(),
"stderr: {}",
String::from_utf8_lossy(&out.stderr)
);
let v: serde_json::Value = serde_json::from_slice(&out.stdout).unwrap(); let v: serde_json::Value = serde_json::from_slice(&out.stdout).unwrap();
assert_eq!( assert_eq!(

View File

@@ -32,10 +32,9 @@ fn schema_path(name: &str) -> PathBuf {
} }
fn parse_schema(name: &str) -> serde_json::Value { fn parse_schema(name: &str) -> serde_json::Value {
let text = std::fs::read_to_string(schema_path(name)) let text =
.unwrap_or_else(|e| panic!("read {name}: {e}")); std::fs::read_to_string(schema_path(name)).unwrap_or_else(|e| panic!("read {name}: {e}"));
serde_json::from_str(&text) serde_json::from_str(&text).unwrap_or_else(|e| panic!("{name} must parse as valid JSON: {e}"))
.unwrap_or_else(|e| panic!("{name} must parse as valid JSON: {e}"))
} }
#[test] #[test]

View File

@@ -41,8 +41,7 @@ fn relax_score_gate(cfg: &Path) {
#[ignore = "requires real Ollama on 127.0.0.1:11434"] #[ignore = "requires real Ollama on 127.0.0.1:11434"]
fn stream_emits_ndjson_events_on_stderr() { fn stream_emits_ndjson_events_on_stderr() {
let dir = tempfile::tempdir().unwrap(); let dir = tempfile::tempdir().unwrap();
let (cfg, workspace, _data) = let (cfg, workspace, _data) = common::write_config_with_llm_model(dir.path(), 30, "gemma4:e4b");
common::write_config_with_llm_model(dir.path(), 30, "gemma4:e4b");
relax_score_gate(&cfg); relax_score_gate(&cfg);
fs::write( fs::write(
workspace.join("a.md"), workspace.join("a.md"),
@@ -93,12 +92,8 @@ fn stream_emits_ndjson_events_on_stderr() {
// stdout: last line is answer.v1 (backwards compat with the // stdout: last line is answer.v1 (backwards compat with the
// non-streaming path — same wire shape, just emitted after the // non-streaming path — same wire shape, just emitted after the
// ndjson event stream rather than instead of it). // ndjson event stream rather than instead of it).
let final_line = stdout let final_line = stdout.lines().last().expect("stdout has at least one line");
.lines() let answer: Value = serde_json::from_str(final_line).expect("stdout final line = answer.v1");
.last()
.expect("stdout has at least one line");
let answer: Value =
serde_json::from_str(final_line).expect("stdout final line = answer.v1");
assert_eq!(answer["schema_version"], "answer.v1"); assert_eq!(answer["schema_version"], "answer.v1");
} }
@@ -109,8 +104,7 @@ fn non_stream_path_unchanged() {
// emits a single `answer.v1` line on stdout — fb-33 must not // emits a single `answer.v1` line on stdout — fb-33 must not
// perturb the existing wire surface. // perturb the existing wire surface.
let dir = tempfile::tempdir().unwrap(); let dir = tempfile::tempdir().unwrap();
let (cfg, workspace, _data) = let (cfg, workspace, _data) = common::write_config_with_llm_model(dir.path(), 30, "gemma4:e4b");
common::write_config_with_llm_model(dir.path(), 30, "gemma4:e4b");
relax_score_gate(&cfg); relax_score_gate(&cfg);
fs::write( fs::write(
workspace.join("a.md"), workspace.join("a.md"),
@@ -140,8 +134,7 @@ fn stream_cancels_when_stderr_closes() {
use std::process::{Command, Stdio}; use std::process::{Command, Stdio};
let dir = tempfile::tempdir().unwrap(); let dir = tempfile::tempdir().unwrap();
let (cfg, workspace, _data) = let (cfg, workspace, _data) = common::write_config_with_llm_model(dir.path(), 30, "gemma4:e4b");
common::write_config_with_llm_model(dir.path(), 30, "gemma4:e4b");
relax_score_gate(&cfg); relax_score_gate(&cfg);
fs::write( fs::write(
workspace.join("a.md"), workspace.join("a.md"),
@@ -198,15 +191,10 @@ fn stream_cancels_when_stderr_closes() {
#[ignore = "requires real Ollama on 127.0.0.1:11434"] #[ignore = "requires real Ollama on 127.0.0.1:11434"]
fn stream_score_gate_refusal_emits_only_retrieval_done() { fn stream_score_gate_refusal_emits_only_retrieval_done() {
let dir = tempfile::tempdir().unwrap(); let dir = tempfile::tempdir().unwrap();
let (cfg, workspace, _data) = let (cfg, workspace, _data) = common::write_config_with_llm_model(dir.path(), 30, "gemma4:e4b");
common::write_config_with_llm_model(dir.path(), 30, "gemma4:e4b");
// Intentionally NO relax_score_gate — keep the default 0.30 // Intentionally NO relax_score_gate — keep the default 0.30
// so the thin-doc + unrelated-query combo trips refusal. // so the thin-doc + unrelated-query combo trips refusal.
fs::write( fs::write(workspace.join("a.md"), "# Title\n\nrust is a language.\n").unwrap();
workspace.join("a.md"),
"# Title\n\nrust is a language.\n",
)
.unwrap();
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
let (stdout, stderr) = let (stdout, stderr) =
@@ -230,12 +218,8 @@ fn stream_score_gate_refusal_emits_only_retrieval_done() {
); );
// Stdout still has answer.v1 with grounded=false. // Stdout still has answer.v1 with grounded=false.
let final_line = stdout let final_line = stdout.lines().last().expect("stdout has at least one line");
.lines() let answer: Value = serde_json::from_str(final_line).expect("answer.v1");
.last()
.expect("stdout has at least one line");
let answer: Value =
serde_json::from_str(final_line).expect("answer.v1");
assert_eq!(answer["schema_version"], "answer.v1"); assert_eq!(answer["schema_version"], "answer.v1");
assert_eq!(answer["grounded"], false); assert_eq!(answer["grounded"], false);
} }

View File

@@ -21,7 +21,11 @@ fn cargo_bin() -> &'static str {
env!("CARGO_BIN_EXE_kebab") env!("CARGO_BIN_EXE_kebab")
} }
fn run_bulk_with_stdin(cfg: &std::path::Path, stdin_body: &str, json: bool) -> std::process::Output { fn run_bulk_with_stdin(
cfg: &std::path::Path,
stdin_body: &str,
json: bool,
) -> std::process::Output {
let mut cmd = Command::new(cargo_bin()); let mut cmd = Command::new(cargo_bin());
cmd.arg("--config").arg(cfg).arg("search").arg("--bulk"); cmd.arg("--config").arg(cfg).arg("search").arg("--bulk");
if json { if json {
@@ -94,7 +98,10 @@ fn empty_stdin_returns_empty_results_with_zero_summary() {
let out = run_bulk_with_stdin(&cfg, "", true); let out = run_bulk_with_stdin(&cfg, "", true);
assert!(out.status.success()); assert!(out.status.success());
let stdout = String::from_utf8_lossy(&out.stdout); let stdout = String::from_utf8_lossy(&out.stdout);
assert!(stdout.trim().is_empty(), "expected empty stdout, got: {stdout}"); assert!(
stdout.trim().is_empty(),
"expected empty stdout, got: {stdout}"
);
let stderr = String::from_utf8_lossy(&out.stderr); let stderr = String::from_utf8_lossy(&out.stderr);
assert!(stderr.contains("bulk_summary: total=0 succeeded=0 failed=0")); assert!(stderr.contains("bulk_summary: total=0 succeeded=0 failed=0"));
} }

View File

@@ -19,7 +19,10 @@ fn line_variant_serialization_unchanged() {
assert_eq!(v["end"], 2); assert_eq!(v["end"], 2);
assert_eq!(v["section"], "§14"); assert_eq!(v["section"], "§14");
// Must not bleed Code-variant keys. // Must not bleed Code-variant keys.
assert!(v.get("line_start").is_none(), "line_start must be absent: {v}"); assert!(
v.get("line_start").is_none(),
"line_start must be absent: {v}"
);
assert!(v.get("symbol").is_none(), "symbol must be absent: {v}"); assert!(v.get("symbol").is_none(), "symbol must be absent: {v}");
assert!(v.get("code").is_none(), "code must be absent: {v}"); assert!(v.get("code").is_none(), "code must be absent: {v}");
} }
@@ -48,7 +51,10 @@ fn page_variant_serialization_unchanged() {
let v = serde_json::to_value(&c).unwrap(); let v = serde_json::to_value(&c).unwrap();
assert_eq!(v["kind"], "page"); assert_eq!(v["kind"], "page");
assert_eq!(v["page"], 13); assert_eq!(v["page"], 13);
assert!(v.get("line_start").is_none(), "line_start must be absent: {v}"); assert!(
v.get("line_start").is_none(),
"line_start must be absent: {v}"
);
assert!(v.get("symbol").is_none(), "symbol must be absent: {v}"); assert!(v.get("symbol").is_none(), "symbol must be absent: {v}");
} }
@@ -67,7 +73,10 @@ fn region_variant_serialization_unchanged() {
assert_eq!(v["y"], 20); assert_eq!(v["y"], 20);
assert_eq!(v["w"], 100); assert_eq!(v["w"], 100);
assert_eq!(v["h"], 200); assert_eq!(v["h"], 200);
assert!(v.get("line_start").is_none(), "line_start must be absent: {v}"); assert!(
v.get("line_start").is_none(),
"line_start must be absent: {v}"
);
} }
#[test] #[test]
@@ -79,7 +88,10 @@ fn caption_variant_serialization_unchanged() {
let v = serde_json::to_value(&c).unwrap(); let v = serde_json::to_value(&c).unwrap();
assert_eq!(v["kind"], "caption"); assert_eq!(v["kind"], "caption");
assert_eq!(v["model"], "qwen2.5-vl:7b"); assert_eq!(v["model"], "qwen2.5-vl:7b");
assert!(v.get("line_start").is_none(), "line_start must be absent: {v}"); assert!(
v.get("line_start").is_none(),
"line_start must be absent: {v}"
);
} }
#[test] #[test]
@@ -95,6 +107,9 @@ fn time_variant_serialization_unchanged() {
assert_eq!(v["start_ms"], 1000); assert_eq!(v["start_ms"], 1000);
assert_eq!(v["end_ms"], 5000); assert_eq!(v["end_ms"], 5000);
assert_eq!(v["speaker"], "Alice"); assert_eq!(v["speaker"], "Alice");
assert!(v.get("line_start").is_none(), "line_start must be absent: {v}"); assert!(
v.get("line_start").is_none(),
"line_start must be absent: {v}"
);
assert!(v.get("symbol").is_none(), "symbol must be absent: {v}"); assert!(v.get("symbol").is_none(), "symbol must be absent: {v}");
} }

View File

@@ -24,10 +24,8 @@ fn fetch_chunk_json_emits_fetch_result_v1() {
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
// Find chunk_id via search. // Find chunk_id via search.
let (search_stdout, _) = common::run_search_with_args( let (search_stdout, _) =
&cfg, common::run_search_with_args(&cfg, &["--json", "--mode", "lexical", "--k", "1", "apples"]);
&["--json", "--mode", "lexical", "--k", "1", "apples"],
);
let search: Value = serde_json::from_str(search_stdout.trim()) let search: Value = serde_json::from_str(search_stdout.trim())
.unwrap_or_else(|e| panic!("search not JSON: {search_stdout:?}: {e}")); .unwrap_or_else(|e| panic!("search not JSON: {search_stdout:?}: {e}"));
let chunk_id = search["hits"][0]["chunk_id"] let chunk_id = search["hits"][0]["chunk_id"]
@@ -35,10 +33,7 @@ fn fetch_chunk_json_emits_fetch_result_v1() {
.expect("chunk_id on first hit") .expect("chunk_id on first hit")
.to_string(); .to_string();
let (stdout, _) = common::run_fetch_with_args( let (stdout, _) = common::run_fetch_with_args(&cfg, &["--json", "chunk", &chunk_id]);
&cfg,
&["--json", "chunk", &chunk_id],
);
let v: Value = serde_json::from_str(stdout.trim()) let v: Value = serde_json::from_str(stdout.trim())
.unwrap_or_else(|e| panic!("fetch not JSON: {stdout:?}: {e}")); .unwrap_or_else(|e| panic!("fetch not JSON: {stdout:?}: {e}"));
assert_eq!(v["schema_version"], "fetch_result.v1"); assert_eq!(v["schema_version"], "fetch_result.v1");
@@ -59,10 +54,8 @@ fn fetch_doc_json_with_max_tokens_truncates() {
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
// Find doc_id via search. // Find doc_id via search.
let (search_stdout, _) = common::run_search_with_args( let (search_stdout, _) =
&cfg, common::run_search_with_args(&cfg, &["--json", "--mode", "lexical", "--k", "1", "Lorem"]);
&["--json", "--mode", "lexical", "--k", "1", "Lorem"],
);
let search: Value = serde_json::from_str(search_stdout.trim()) let search: Value = serde_json::from_str(search_stdout.trim())
.unwrap_or_else(|e| panic!("search not JSON: {search_stdout:?}: {e}")); .unwrap_or_else(|e| panic!("search not JSON: {search_stdout:?}: {e}"));
let doc_id = search["hits"][0]["doc_id"] let doc_id = search["hits"][0]["doc_id"]
@@ -70,10 +63,8 @@ fn fetch_doc_json_with_max_tokens_truncates() {
.expect("doc_id on first hit") .expect("doc_id on first hit")
.to_string(); .to_string();
let (stdout, _) = common::run_fetch_with_args( let (stdout, _) =
&cfg, common::run_fetch_with_args(&cfg, &["--json", "doc", &doc_id, "--max-tokens", "20"]);
&["--json", "doc", &doc_id, "--max-tokens", "20"],
);
let v: Value = serde_json::from_str(stdout.trim()) let v: Value = serde_json::from_str(stdout.trim())
.unwrap_or_else(|e| panic!("fetch not JSON: {stdout:?}: {e}")); .unwrap_or_else(|e| panic!("fetch not JSON: {stdout:?}: {e}"));
assert_eq!(v["kind"], "doc"); assert_eq!(v["kind"], "doc");

View File

@@ -32,12 +32,9 @@ fn search_with_doc_id_filter_returns_only_target_doc() {
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
// First, search without a doc-id filter to find what doc_ids exist. // First, search without a doc-id filter to find what doc_ids exist.
let (stdout, _) = common::run_search_with_args( let (stdout, _) = common::run_search_with_args(&cfg, &["--json", "--mode", "lexical", "rust"]);
&cfg, let resp: Value =
&["--json", "--mode", "lexical", "rust"], serde_json::from_str(stdout.trim()).unwrap_or_else(|e| panic!("not JSON: {stdout:?}: {e}"));
);
let resp: Value = serde_json::from_str(stdout.trim())
.unwrap_or_else(|e| panic!("not JSON: {stdout:?}: {e}"));
let hits = resp["hits"].as_array().expect("hits array"); let hits = resp["hits"].as_array().expect("hits array");
assert!( assert!(
hits.len() >= 2, hits.len() >= 2,
@@ -147,15 +144,19 @@ fn search_with_media_filter_md_alias_normalizes_to_markdown() {
let (cfg, workspace, _data) = common::write_config(dir.path(), 30); let (cfg, workspace, _data) = common::write_config(dir.path(), 30);
// Only a markdown file — the `md` alias should match it. // Only a markdown file — the `md` alias should match it.
fs::write(workspace.join("notes.md"), "# Notes\n\nrust async programming\n").unwrap(); fs::write(
workspace.join("notes.md"),
"# Notes\n\nrust async programming\n",
)
.unwrap();
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
let (stdout, _) = common::run_search_with_args( let (stdout, _) = common::run_search_with_args(
&cfg, &cfg,
&["--json", "--mode", "lexical", "--media", "md", "rust"], &["--json", "--mode", "lexical", "--media", "md", "rust"],
); );
let resp: Value = serde_json::from_str(stdout.trim()) let resp: Value =
.unwrap_or_else(|e| panic!("not JSON: {stdout:?}: {e}")); serde_json::from_str(stdout.trim()).unwrap_or_else(|e| panic!("not JSON: {stdout:?}: {e}"));
let hits = resp["hits"].as_array().expect("hits array"); let hits = resp["hits"].as_array().expect("hits array");
assert!( assert!(
@@ -189,10 +190,8 @@ fn search_with_tag_filter_matches_frontmatter_tags() {
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
// Without filter — both docs must produce hits. // Without filter — both docs must produce hits.
let (unfiltered, _) = common::run_search_with_args( let (unfiltered, _) =
&cfg, common::run_search_with_args(&cfg, &["--json", "--mode", "lexical", "rust"]);
&["--json", "--mode", "lexical", "rust"],
);
let uresp: Value = serde_json::from_str(unfiltered.trim()) let uresp: Value = serde_json::from_str(unfiltered.trim())
.unwrap_or_else(|e| panic!("not JSON (unfiltered): {unfiltered:?}: {e}")); .unwrap_or_else(|e| panic!("not JSON (unfiltered): {unfiltered:?}: {e}"));
let uhits = uresp["hits"].as_array().expect("unfiltered hits array"); let uhits = uresp["hits"].as_array().expect("unfiltered hits array");
@@ -254,10 +253,8 @@ fn search_with_two_tag_filters_returns_or_within_tags() {
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
// Without filter: all three docs produce hits. // Without filter: all three docs produce hits.
let (unfiltered, _) = common::run_search_with_args( let (unfiltered, _) =
&cfg, common::run_search_with_args(&cfg, &["--json", "--mode", "lexical", "rust"]);
&["--json", "--mode", "lexical", "rust"],
);
let uresp: Value = serde_json::from_str(unfiltered.trim()) let uresp: Value = serde_json::from_str(unfiltered.trim())
.unwrap_or_else(|e| panic!("not JSON (unfiltered): {unfiltered:?}: {e}")); .unwrap_or_else(|e| panic!("not JSON (unfiltered): {unfiltered:?}: {e}"));
let uhits = uresp["hits"].as_array().expect("unfiltered hits array"); let uhits = uresp["hits"].as_array().expect("unfiltered hits array");
@@ -270,10 +267,7 @@ fn search_with_two_tag_filters_returns_or_within_tags() {
let (filtered, _) = common::run_search_with_args( let (filtered, _) = common::run_search_with_args(
&cfg, &cfg,
&[ &[
"--json", "--mode", "lexical", "--json", "--mode", "lexical", "--tag", "rust", "--tag", "async", "rust",
"--tag", "rust",
"--tag", "async",
"rust",
], ],
); );
let fresp: Value = serde_json::from_str(filtered.trim()) let fresp: Value = serde_json::from_str(filtered.trim())
@@ -301,6 +295,12 @@ fn search_with_two_tag_filters_returns_or_within_tags() {
.collect(); .collect();
let has_a = paths.iter().any(|p| p.ends_with("a.md")); let has_a = paths.iter().any(|p| p.ends_with("a.md"));
let has_b = paths.iter().any(|p| p.ends_with("b.md")); let has_b = paths.iter().any(|p| p.ends_with("b.md"));
assert!(has_a, "--tag rust must include a.md (rust-tagged): paths={paths:?}"); assert!(
assert!(has_b, "--tag async must include b.md (async-tagged): paths={paths:?}"); has_a,
"--tag rust must include a.md (rust-tagged): paths={paths:?}"
);
assert!(
has_b,
"--tag async must include b.md (async-tagged): paths={paths:?}"
);
} }

View File

@@ -5,7 +5,7 @@
//! inject spurious keys into the existing markdown corpus wire shape. //! inject spurious keys into the existing markdown corpus wire shape.
use kebab_core::{ use kebab_core::{
Citation, ChunkId, ChunkerVersion, DocumentId, IndexVersion, RetrievalDetail, ScoreKind, ChunkId, ChunkerVersion, Citation, DocumentId, IndexVersion, RetrievalDetail, ScoreKind,
SearchHit, WorkspacePath, SearchHit, WorkspacePath,
}; };

View File

@@ -23,12 +23,10 @@ fn search_json_emits_search_response_v1_wrapper() {
fs::write(workspace.join("a.md"), "# T\n\napples are red.\n").unwrap(); fs::write(workspace.join("a.md"), "# T\n\napples are red.\n").unwrap();
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
let (stdout, _stderr) = common::run_search_with_args( let (stdout, _stderr) =
&cfg, common::run_search_with_args(&cfg, &["--json", "--mode", "lexical", "apples"]);
&["--json", "--mode", "lexical", "apples"], let v: Value =
); serde_json::from_str(stdout.trim()).unwrap_or_else(|e| panic!("not JSON: {stdout:?}: {e}"));
let v: Value = serde_json::from_str(stdout.trim())
.unwrap_or_else(|e| panic!("not JSON: {stdout:?}: {e}"));
assert_eq!(v["schema_version"], "search_response.v1"); assert_eq!(v["schema_version"], "search_response.v1");
assert!(v["hits"].is_array(), "hits must be array, got {v}"); assert!(v["hits"].is_array(), "hits must be array, got {v}");
assert!( assert!(
@@ -67,8 +65,8 @@ fn search_json_truncates_with_max_tokens() {
&cfg, &cfg,
&["--json", "--mode", "lexical", "--max-tokens", "30", "rust"], &["--json", "--mode", "lexical", "--max-tokens", "30", "rust"],
); );
let v: Value = serde_json::from_str(stdout.trim()) let v: Value =
.unwrap_or_else(|e| panic!("not JSON: {stdout:?}: {e}")); serde_json::from_str(stdout.trim()).unwrap_or_else(|e| panic!("not JSON: {stdout:?}: {e}"));
assert_eq!( assert_eq!(
v["truncated"], true, v["truncated"], true,
"30-token cap must trip truncation: {v}" "30-token cap must trip truncation: {v}"
@@ -88,10 +86,8 @@ fn search_json_cursor_paginates() {
} }
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
let (page1, _) = common::run_search_with_args( let (page1, _) =
&cfg, common::run_search_with_args(&cfg, &["--json", "--mode", "lexical", "--k", "2", "rust"]);
&["--json", "--mode", "lexical", "--k", "2", "rust"],
);
let v1: Value = serde_json::from_str(page1.trim()) let v1: Value = serde_json::from_str(page1.trim())
.unwrap_or_else(|e| panic!("page1 not JSON: {page1:?}: {e}")); .unwrap_or_else(|e| panic!("page1 not JSON: {page1:?}: {e}"));
let cursor = v1["next_cursor"] let cursor = v1["next_cursor"]
@@ -101,14 +97,7 @@ fn search_json_cursor_paginates() {
let (page2, _) = common::run_search_with_args( let (page2, _) = common::run_search_with_args(
&cfg, &cfg,
&[ &[
"--json", "--json", "--mode", "lexical", "--k", "2", "--cursor", cursor, "rust",
"--mode",
"lexical",
"--k",
"2",
"--cursor",
cursor,
"rust",
], ],
); );
let v2: Value = serde_json::from_str(page2.trim()) let v2: Value = serde_json::from_str(page2.trim())
@@ -118,23 +107,13 @@ fn search_json_cursor_paginates() {
.as_array() .as_array()
.expect("page1 hits array") .expect("page1 hits array")
.iter() .iter()
.map(|h| { .map(|h| h["chunk_id"].as_str().expect("chunk_id string").to_string())
h["chunk_id"]
.as_str()
.expect("chunk_id string")
.to_string()
})
.collect(); .collect();
let p2_ids: Vec<String> = v2["hits"] let p2_ids: Vec<String> = v2["hits"]
.as_array() .as_array()
.expect("page2 hits array") .expect("page2 hits array")
.iter() .iter()
.map(|h| { .map(|h| h["chunk_id"].as_str().expect("chunk_id string").to_string())
h["chunk_id"]
.as_str()
.expect("chunk_id string")
.to_string()
})
.collect(); .collect();
assert!( assert!(
!p2_ids.is_empty(), !p2_ids.is_empty(),
@@ -161,10 +140,8 @@ fn search_stale_cursor_returns_error_v1_with_stale_cursor_code() {
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
// Get a valid cursor first. // Get a valid cursor first.
let (page1_stdout, _) = common::run_search_with_args( let (page1_stdout, _) =
&cfg, common::run_search_with_args(&cfg, &["--mode", "lexical", "--json", "--k", "1", "apples"]);
&["--mode", "lexical", "--json", "--k", "1", "apples"],
);
let v1: Value = serde_json::from_str(page1_stdout.trim()).expect("json"); let v1: Value = serde_json::from_str(page1_stdout.trim()).expect("json");
let cursor = v1["next_cursor"] let cursor = v1["next_cursor"]
.as_str() .as_str()
@@ -181,16 +158,8 @@ fn search_stale_cursor_returns_error_v1_with_stale_cursor_code() {
let cfg_str = cfg.to_str().expect("utf8"); let cfg_str = cfg.to_str().expect("utf8");
let out = std::process::Command::new(exe) let out = std::process::Command::new(exe)
.args([ .args([
"--config", "--config", cfg_str, "--json", "search", "--mode", "lexical", "--json", "--cursor",
cfg_str, &cursor, "apples",
"--json",
"search",
"--mode",
"lexical",
"--json",
"--cursor",
&cursor,
"apples",
]) ])
.output() .output()
.expect("kebab search --cursor"); .expect("kebab search --cursor");
@@ -234,10 +203,8 @@ fn search_plain_emits_truncated_hint_to_stderr() {
} }
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
let (_stdout, stderr) = common::run_search_with_args( let (_stdout, stderr) =
&cfg, common::run_search_with_args(&cfg, &["--mode", "lexical", "--max-tokens", "30", "rust"]);
&["--mode", "lexical", "--max-tokens", "30", "rust"],
);
assert!( assert!(
stderr.contains("[truncated;"), stderr.contains("[truncated;"),
"stderr must carry truncated hint: {stderr:?}" "stderr must carry truncated hint: {stderr:?}"
@@ -254,10 +221,7 @@ fn search_plain_emits_short_query_hint_to_stderr() {
let (cfg, workspace, _data) = common::write_config(dir.path(), 30); let (cfg, workspace, _data) = common::write_config(dir.path(), 30);
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
let (_stdout, stderr) = common::run_search_with_args( let (_stdout, stderr) = common::run_search_with_args(&cfg, &["--mode", "lexical", "ab"]);
&cfg,
&["--mode", "lexical", "ab"],
);
assert!( assert!(
stderr.contains("[hint]"), stderr.contains("[hint]"),
"stderr must carry short-query hint: {stderr:?}" "stderr must carry short-query hint: {stderr:?}"
@@ -278,18 +242,18 @@ fn search_json_emits_hint_field_for_short_query() {
let (cfg, workspace, _data) = common::write_config(dir.path(), 30); let (cfg, workspace, _data) = common::write_config(dir.path(), 30);
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
let (stdout, _stderr) = common::run_search_with_args( let (stdout, _stderr) =
&cfg, common::run_search_with_args(&cfg, &["--json", "--mode", "lexical", "ab"]);
&["--json", "--mode", "lexical", "ab"], let v: Value =
); serde_json::from_str(stdout.trim()).unwrap_or_else(|e| panic!("not JSON: {stdout:?}: {e}"));
let v: Value = serde_json::from_str(stdout.trim())
.unwrap_or_else(|e| panic!("not JSON: {stdout:?}: {e}"));
assert!( assert!(
v["hits"].as_array().unwrap().is_empty(), v["hits"].as_array().unwrap().is_empty(),
"empty hits expected for short query in empty KB: {v}" "empty hits expected for short query in empty KB: {v}"
); );
assert_eq!( assert_eq!(
v["hint"].as_str().expect("hint field set on short empty result"), v["hint"]
.as_str()
.expect("hint field set on short empty result"),
"3자 이상 키워드 권장 (trigram tokenizer 제약)", "3자 이상 키워드 권장 (trigram tokenizer 제약)",
"hint must carry the standard advisory: {v}" "hint must carry the standard advisory: {v}"
); );
@@ -305,12 +269,10 @@ fn search_json_omits_hint_field_when_query_is_long_enough() {
let (cfg, workspace, _data) = common::write_config(dir.path(), 30); let (cfg, workspace, _data) = common::write_config(dir.path(), 30);
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
let (stdout, _stderr) = common::run_search_with_args( let (stdout, _stderr) =
&cfg, common::run_search_with_args(&cfg, &["--json", "--mode", "lexical", "abc"]);
&["--json", "--mode", "lexical", "abc"], let v: Value =
); serde_json::from_str(stdout.trim()).unwrap_or_else(|e| panic!("not JSON: {stdout:?}: {e}"));
let v: Value = serde_json::from_str(stdout.trim())
.unwrap_or_else(|e| panic!("not JSON: {stdout:?}: {e}"));
assert!( assert!(
v.get("hint").is_none(), v.get("hint").is_none(),
"hint must be absent for ≥3-char queries: {v}" "hint must be absent for ≥3-char queries: {v}"

View File

@@ -16,10 +16,8 @@ fn lexical_mode_hits_carry_bm25_score_kind() {
doc_with_term(&workspace); doc_with_term(&workspace);
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
let (stdout, _stderr) = common::run_search_with_args( let (stdout, _stderr) =
&cfg, common::run_search_with_args(&cfg, &["--mode", "lexical", "--json", "rust"]);
&["--mode", "lexical", "--json", "rust"],
);
let v: Value = serde_json::from_str(stdout.trim()).expect("valid JSON"); let v: Value = serde_json::from_str(stdout.trim()).expect("valid JSON");
let hits = v["hits"].as_array().expect("hits array"); let hits = v["hits"].as_array().expect("hits array");
assert!(!hits.is_empty(), "expected at least 1 hit"); assert!(!hits.is_empty(), "expected at least 1 hit");
@@ -40,10 +38,8 @@ fn old_wire_reader_compat_score_kind_optional_field() {
doc_with_term(&workspace); doc_with_term(&workspace);
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
let (stdout, _stderr) = common::run_search_with_args( let (stdout, _stderr) =
&cfg, common::run_search_with_args(&cfg, &["--mode", "lexical", "--json", "rust"]);
&["--mode", "lexical", "--json", "rust"],
);
let v: Value = serde_json::from_str(stdout.trim()).unwrap(); let v: Value = serde_json::from_str(stdout.trim()).unwrap();
let hit = &v["hits"][0]; let hit = &v["hits"][0];
assert!(hit.get("score_kind").is_some(), "score_kind always emitted"); assert!(hit.get("score_kind").is_some(), "score_kind always emitted");

View File

@@ -59,15 +59,14 @@ fn search_json_includes_indexed_at_and_stale() {
.get("hits") .get("hits")
.and_then(|h| h.as_array()) .and_then(|h| h.as_array())
.unwrap_or_else(|| panic!("expected hits array, got {stdout}")); .unwrap_or_else(|| panic!("expected hits array, got {stdout}"));
let first = arr.first().unwrap_or_else(|| panic!("expected ≥1 hit, got empty hits: {stdout}")); let first = arr
.first()
.unwrap_or_else(|| panic!("expected ≥1 hit, got empty hits: {stdout}"));
assert!( assert!(
first.get("indexed_at").is_some(), first.get("indexed_at").is_some(),
"missing indexed_at in {first}" "missing indexed_at in {first}"
); );
assert!( assert!(first.get("stale").is_some(), "missing stale in {first}");
first.get("stale").is_some(),
"missing stale in {first}"
);
assert_eq!( assert_eq!(
first["stale"], false, first["stale"], false,
"freshly ingested doc must not be stale at default 30d threshold" "freshly ingested doc must not be stale at default 30d threshold"

View File

@@ -12,10 +12,8 @@ fn search_trace_json_includes_trace_block() {
fs::write(workspace.join("doc1.md"), "# Title\n\nrust async hello\n").unwrap(); fs::write(workspace.join("doc1.md"), "# Title\n\nrust async hello\n").unwrap();
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
let (stdout, _stderr) = common::run_search_with_args( let (stdout, _stderr) =
&cfg, common::run_search_with_args(&cfg, &["--mode", "lexical", "--trace", "--json", "rust"]);
&["--mode", "lexical", "--trace", "--json", "rust"],
);
let v: Value = serde_json::from_str(stdout.trim()).expect("valid JSON"); let v: Value = serde_json::from_str(stdout.trim()).expect("valid JSON");
assert_eq!(v["schema_version"], "search_response.v1"); assert_eq!(v["schema_version"], "search_response.v1");
assert!(v["trace"].is_object(), "trace block present"); assert!(v["trace"].is_object(), "trace block present");
@@ -33,12 +31,13 @@ fn search_without_trace_omits_trace_field() {
fs::write(workspace.join("doc1.md"), "# Title\n\nrust async hello\n").unwrap(); fs::write(workspace.join("doc1.md"), "# Title\n\nrust async hello\n").unwrap();
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
let (stdout, _stderr) = common::run_search_with_args( let (stdout, _stderr) =
&cfg, common::run_search_with_args(&cfg, &["--mode", "lexical", "--json", "rust"]);
&["--mode", "lexical", "--json", "rust"],
);
let v: Value = serde_json::from_str(stdout.trim()).expect("valid JSON"); let v: Value = serde_json::from_str(stdout.trim()).expect("valid JSON");
assert!(v.get("trace").is_none(), "trace field absent without --trace"); assert!(
v.get("trace").is_none(),
"trace field absent without --trace"
);
} }
#[test] #[test]
@@ -48,10 +47,8 @@ fn search_trace_lexical_mode_vector_list_empty() {
fs::write(workspace.join("doc1.md"), "# Title\n\nrust async hello\n").unwrap(); fs::write(workspace.join("doc1.md"), "# Title\n\nrust async hello\n").unwrap();
common::ingest(&cfg, &workspace); common::ingest(&cfg, &workspace);
let (stdout, _stderr) = common::run_search_with_args( let (stdout, _stderr) =
&cfg, common::run_search_with_args(&cfg, &["--mode", "lexical", "--trace", "--json", "rust"]);
&["--mode", "lexical", "--trace", "--json", "rust"],
);
let v: Value = serde_json::from_str(stdout.trim()).expect("valid JSON"); let v: Value = serde_json::from_str(stdout.trim()).expect("valid JSON");
assert_eq!(v["trace"]["vector"].as_array().unwrap().len(), 0); assert_eq!(v["trace"]["vector"].as_array().unwrap().len(), 0);
assert_eq!(v["trace"]["timing"]["vector_ms"], 0); assert_eq!(v["trace"]["timing"]["vector_ms"], 0);

View File

@@ -420,12 +420,16 @@ pub struct PdfCfg {
impl PdfCfg { impl PdfCfg {
pub fn defaults() -> Self { pub fn defaults() -> Self {
Self { ocr: PdfOcrCfg::defaults() } Self {
ocr: PdfOcrCfg::defaults(),
}
} }
} }
impl Default for PdfCfg { impl Default for PdfCfg {
fn default() -> Self { Self::defaults() } fn default() -> Self {
Self::defaults()
}
} }
/// v0.20.x ingest log surface: structured ndjson log written per ingest run. /// v0.20.x ingest log surface: structured ndjson log written per ingest run.
@@ -444,7 +448,9 @@ pub struct LoggingCfg {
pub ingest_log_dir: PathBuf, pub ingest_log_dir: PathBuf,
} }
fn default_ingest_log_enabled() -> bool { true } fn default_ingest_log_enabled() -> bool {
true
}
fn default_ingest_log_dir() -> PathBuf { fn default_ingest_log_dir() -> PathBuf {
PathBuf::from("{state_dir}/logs") PathBuf::from("{state_dir}/logs")
} }
@@ -531,10 +537,18 @@ impl PdfOcrCfg {
/// metro-korea.pdf page 8/9/13) 의 OCR 을 강제 timeout 시켜 본문 indexed 손실. /// metro-korea.pdf page 8/9/13) 의 OCR 을 강제 timeout 시켜 본문 indexed 손실.
/// **conservative starting point 180s 로 재조정** + dogfood evidence 기반 sweet spot /// **conservative starting point 180s 로 재조정** + dogfood evidence 기반 sweet spot
/// 점진적 축소 정책. user 가 `[pdf.ocr] request_timeout_secs = N` 으로 직접 tune. /// 점진적 축소 정책. user 가 `[pdf.ocr] request_timeout_secs = N` 으로 직접 tune.
fn default_pdf_ocr_request_timeout_secs() -> u64 { 180 } fn default_pdf_ocr_request_timeout_secs() -> u64 {
fn default_pdf_ocr_valid_ratio() -> f32 { 0.5 } 180
fn default_pdf_ocr_min_char_count() -> u32 { 20 } }
fn default_pdf_ocr_lang_hint() -> Option<String> { Some("kor".to_string()) } fn default_pdf_ocr_valid_ratio() -> f32 {
0.5
}
fn default_pdf_ocr_min_char_count() -> u32 {
20
}
fn default_pdf_ocr_lang_hint() -> Option<String> {
Some("kor".to_string())
}
/// p9-fb-14: TUI-only configuration. Currently a single `theme` /// p9-fb-14: TUI-only configuration. Currently a single `theme`
/// selector (`"dark"` / `"light"`); future fields (custom role /// selector (`"dark"` / `"light"`); future fields (custom role
@@ -675,8 +689,7 @@ impl Config {
explain_default: false, explain_default: false,
max_context_tokens: 8000, max_context_tokens: 8000,
multi_hop_max_depth: default_multi_hop_max_depth(), multi_hop_max_depth: default_multi_hop_max_depth(),
multi_hop_max_sub_queries_per_iter: multi_hop_max_sub_queries_per_iter: default_multi_hop_max_sub_queries_per_iter(),
default_multi_hop_max_sub_queries_per_iter(),
multi_hop_max_pool_chunks: default_multi_hop_max_pool_chunks(), multi_hop_max_pool_chunks: default_multi_hop_max_pool_chunks(),
nli_threshold: default_nli_threshold(), nli_threshold: default_nli_threshold(),
}, },
@@ -1015,11 +1028,7 @@ impl Config {
"KEBAB_IMAGE_OCR_ENDPOINT" => { "KEBAB_IMAGE_OCR_ENDPOINT" => {
// Empty env value is treated the same as "fall back // Empty env value is treated the same as "fall back
// to models.llm.endpoint" — i.e. set None. // to models.llm.endpoint" — i.e. set None.
self.image.ocr.endpoint = if v.is_empty() { self.image.ocr.endpoint = if v.is_empty() { None } else { Some(v.clone()) };
None
} else {
Some(v.clone())
};
} }
"KEBAB_IMAGE_OCR_LANGUAGES" => { "KEBAB_IMAGE_OCR_LANGUAGES" => {
// Comma-separated list, e.g. "eng,kor". // Comma-separated list, e.g. "eng,kor".
@@ -1319,7 +1328,10 @@ theme = "dark"
#[test] #[test]
fn env_overrides_chunking_target_tokens() { fn env_overrides_chunking_target_tokens() {
let mut env = HashMap::new(); let mut env = HashMap::new();
env.insert("KEBAB_CHUNKING_TARGET_TOKENS".to_string(), "777".to_string()); env.insert(
"KEBAB_CHUNKING_TARGET_TOKENS".to_string(),
"777".to_string(),
);
let c = Config::defaults().apply_env(&env); let c = Config::defaults().apply_env(&env);
assert_eq!(c.chunking.target_tokens, 777); assert_eq!(c.chunking.target_tokens, 777);
} }
@@ -1331,7 +1343,10 @@ theme = "dark"
"KEBAB_MODELS_LLM_ENDPOINT".to_string(), "KEBAB_MODELS_LLM_ENDPOINT".to_string(),
"http://10.0.0.1:11434".to_string(), "http://10.0.0.1:11434".to_string(),
); );
env.insert("KEBAB_MODELS_LLM_TEMPERATURE".to_string(), "0.7".to_string()); env.insert(
"KEBAB_MODELS_LLM_TEMPERATURE".to_string(),
"0.7".to_string(),
);
let c = Config::defaults().apply_env(&env); let c = Config::defaults().apply_env(&env);
assert_eq!(c.models.llm.endpoint, "http://10.0.0.1:11434"); assert_eq!(c.models.llm.endpoint, "http://10.0.0.1:11434");
assert!((c.models.llm.temperature - 0.7).abs() < 1e-6); assert!((c.models.llm.temperature - 0.7).abs() < 1e-6);
@@ -1361,8 +1376,7 @@ theme = "dark"
/// shared with the OCR-side invariant via [`LEGACY_PRE_TIMEOUT_TOML`]. /// shared with the OCR-side invariant via [`LEGACY_PRE_TIMEOUT_TOML`].
#[test] #[test]
fn legacy_config_without_request_timeout_secs_uses_default() { fn legacy_config_without_request_timeout_secs_uses_default() {
let c: Config = toml::from_str(LEGACY_PRE_TIMEOUT_TOML) let c: Config = toml::from_str(LEGACY_PRE_TIMEOUT_TOML).expect("parse legacy config");
.expect("parse legacy config");
assert_eq!(c.models.llm.request_timeout_secs, 300); assert_eq!(c.models.llm.request_timeout_secs, 300);
} }
@@ -1391,10 +1405,7 @@ theme = "dark"
/// existing configs that omit the new field keep behaving identically. /// existing configs that omit the new field keep behaving identically.
#[test] #[test]
fn default_ocr_request_timeout_secs_is_300() { fn default_ocr_request_timeout_secs_is_300() {
assert_eq!( assert_eq!(Config::defaults().image.ocr.request_timeout_secs, 300);
Config::defaults().image.ocr.request_timeout_secs,
300
);
} }
#[test] #[test]
@@ -1414,8 +1425,7 @@ theme = "dark"
/// with the LLM-side invariant via [`LEGACY_PRE_TIMEOUT_TOML`]. /// with the LLM-side invariant via [`LEGACY_PRE_TIMEOUT_TOML`].
#[test] #[test]
fn legacy_config_without_ocr_request_timeout_secs_uses_default() { fn legacy_config_without_ocr_request_timeout_secs_uses_default() {
let c: Config = toml::from_str(LEGACY_PRE_TIMEOUT_TOML) let c: Config = toml::from_str(LEGACY_PRE_TIMEOUT_TOML).expect("parse legacy config");
.expect("parse legacy config");
assert_eq!(c.image.ocr.request_timeout_secs, 300); assert_eq!(c.image.ocr.request_timeout_secs, 300);
} }
@@ -1428,10 +1438,7 @@ theme = "dark"
#[test] #[test]
fn default_multi_hop_max_sub_queries_per_iter_is_5() { fn default_multi_hop_max_sub_queries_per_iter_is_5() {
assert_eq!( assert_eq!(Config::defaults().rag.multi_hop_max_sub_queries_per_iter, 5);
Config::defaults().rag.multi_hop_max_sub_queries_per_iter,
5
);
} }
#[test] #[test]
@@ -1445,10 +1452,7 @@ theme = "dark"
#[test] #[test]
fn env_overrides_multi_hop_knobs() { fn env_overrides_multi_hop_knobs() {
let mut env = HashMap::new(); let mut env = HashMap::new();
env.insert( env.insert("KEBAB_RAG_MULTI_HOP_MAX_DEPTH".to_string(), "5".to_string());
"KEBAB_RAG_MULTI_HOP_MAX_DEPTH".to_string(),
"5".to_string(),
);
env.insert( env.insert(
"KEBAB_RAG_MULTI_HOP_MAX_SUB_QUERIES_PER_ITER".to_string(), "KEBAB_RAG_MULTI_HOP_MAX_SUB_QUERIES_PER_ITER".to_string(),
"7".to_string(), "7".to_string(),
@@ -1470,8 +1474,7 @@ theme = "dark"
/// (that fixture also predates the multi_hop_* fields). /// (that fixture also predates the multi_hop_* fields).
#[test] #[test]
fn legacy_config_without_multi_hop_knobs_uses_defaults() { fn legacy_config_without_multi_hop_knobs_uses_defaults() {
let c: Config = toml::from_str(LEGACY_PRE_TIMEOUT_TOML) let c: Config = toml::from_str(LEGACY_PRE_TIMEOUT_TOML).expect("parse legacy config");
.expect("parse legacy config");
assert_eq!(c.rag.multi_hop_max_depth, 3); assert_eq!(c.rag.multi_hop_max_depth, 3);
assert_eq!(c.rag.multi_hop_max_sub_queries_per_iter, 5); assert_eq!(c.rag.multi_hop_max_sub_queries_per_iter, 5);
// v0.18 dogfood (post-PR-7): pool default 30 → 15. // v0.18 dogfood (post-PR-7): pool default 30 → 15.
@@ -1504,8 +1507,7 @@ theme = "dark"
/// all PR-9c-1 fields). /// all PR-9c-1 fields).
#[test] #[test]
fn legacy_config_without_nli_uses_defaults() { fn legacy_config_without_nli_uses_defaults() {
let c: Config = toml::from_str(LEGACY_PRE_TIMEOUT_TOML) let c: Config = toml::from_str(LEGACY_PRE_TIMEOUT_TOML).expect("parse legacy config");
.expect("parse legacy config");
assert_eq!(c.rag.nli_threshold, 0.0); assert_eq!(c.rag.nli_threshold, 0.0);
assert_eq!( assert_eq!(
c.models.nli.model, c.models.nli.model,
@@ -1705,7 +1707,11 @@ max_context_tokens = 8000
"[workspace]\ninclude = [\"**/*.md\", \"**/*.txt\"]", "[workspace]\ninclude = [\"**/*.md\", \"**/*.txt\"]",
); );
let parsed: Result<Config, _> = toml::from_str(&toml_text); let parsed: Result<Config, _> = toml::from_str(&toml_text);
assert!(parsed.is_ok(), "legacy include must not break load: {:?}", parsed.err()); assert!(
parsed.is_ok(),
"legacy include must not break load: {:?}",
parsed.err()
);
let cfg = parsed.unwrap(); let cfg = parsed.unwrap();
assert_eq!(cfg.workspace.root, "/tmp/kebab-legacy"); assert_eq!(cfg.workspace.root, "/tmp/kebab-legacy");
} }
@@ -1715,7 +1721,10 @@ max_context_tokens = 8000
#[test] #[test]
fn workspace_cfg_has_only_root_and_exclude_fields() { fn workspace_cfg_has_only_root_and_exclude_fields() {
let ws = Config::defaults().workspace; let ws = Config::defaults().workspace;
let WorkspaceCfg { root: _, exclude: _ } = &ws; let WorkspaceCfg {
root: _,
exclude: _,
} = &ws;
} }
#[test] #[test]
@@ -1727,9 +1736,10 @@ max_context_tokens = 8000
#[test] #[test]
fn env_override_stale_threshold() { fn env_override_stale_threshold() {
let c = Config::defaults(); let c = Config::defaults();
let env: HashMap<String, String> = [ let env: HashMap<String, String> = [(
("KEBAB_SEARCH_STALE_THRESHOLD_DAYS".to_string(), "7".to_string()), "KEBAB_SEARCH_STALE_THRESHOLD_DAYS".to_string(),
] "7".to_string(),
)]
.into_iter() .into_iter()
.collect(); .collect();
let c = c.apply_env(&env); let c = c.apply_env(&env);
@@ -1744,9 +1754,10 @@ max_context_tokens = 8000
// `fb27_tests::file_negative_stale_threshold_returns_config_invalid`) // `fb27_tests::file_negative_stale_threshold_returns_config_invalid`)
// is the spec-required hard error surface. // is the spec-required hard error surface.
let c = Config::defaults(); let c = Config::defaults();
let env: HashMap<String, String> = [ let env: HashMap<String, String> = [(
("KEBAB_SEARCH_STALE_THRESHOLD_DAYS".to_string(), "-5".to_string()), "KEBAB_SEARCH_STALE_THRESHOLD_DAYS".to_string(),
] "-5".to_string(),
)]
.into_iter() .into_iter()
.collect(); .collect();
let c = c.apply_env(&env); let c = c.apply_env(&env);
@@ -1765,7 +1776,10 @@ max_context_tokens = 8000
std::env::set_var("XDG_CONFIG_HOME", "/tmp/kebabtest-xdg-config"); std::env::set_var("XDG_CONFIG_HOME", "/tmp/kebabtest-xdg-config");
} }
let p = Config::xdg_config_path(); let p = Config::xdg_config_path();
assert_eq!(p, PathBuf::from("/tmp/kebabtest-xdg-config/kebab/config.toml")); assert_eq!(
p,
PathBuf::from("/tmp/kebabtest-xdg-config/kebab/config.toml")
);
// SAFETY: scope-local restore. // SAFETY: scope-local restore.
unsafe { unsafe {
match prev { match prev {
@@ -1810,10 +1824,7 @@ max_context_tokens = 8000
let base = Config::defaults(); let base = Config::defaults();
let mut toml_text = toml::to_string(&base).unwrap(); let mut toml_text = toml::to_string(&base).unwrap();
// Inject max_file_bytes override into the [ingest.code] table. // Inject max_file_bytes override into the [ingest.code] table.
toml_text = toml_text.replace( toml_text = toml_text.replace("max_file_bytes = 262144", "max_file_bytes = 524288");
"max_file_bytes = 262144",
"max_file_bytes = 524288",
);
let cfg: Config = toml::from_str(&toml_text).unwrap(); let cfg: Config = toml::from_str(&toml_text).unwrap();
assert_eq!(cfg.ingest.code.max_file_bytes, 524_288); assert_eq!(cfg.ingest.code.max_file_bytes, 524_288);
} }
@@ -1828,7 +1839,8 @@ mod fb27_tests {
fn config_invalid_carries_path_and_cause() { fn config_invalid_carries_path_and_cause() {
let nonexistent = PathBuf::from("/this/path/should/not/exist/kebab.toml"); let nonexistent = PathBuf::from("/this/path/should/not/exist/kebab.toml");
let err = Config::from_file(&nonexistent).unwrap_err(); let err = Config::from_file(&nonexistent).unwrap_err();
let signal = err.downcast_ref::<ConfigInvalid>() let signal = err
.downcast_ref::<ConfigInvalid>()
.expect("from_file error should downcast to ConfigInvalid"); .expect("from_file error should downcast to ConfigInvalid");
assert_eq!(signal.path, nonexistent); assert_eq!(signal.path, nonexistent);
assert!(!signal.cause.is_empty(), "cause should be non-empty"); assert!(!signal.cause.is_empty(), "cause should be non-empty");
@@ -1840,7 +1852,8 @@ mod fb27_tests {
let p = dir.path().join("bad.toml"); let p = dir.path().join("bad.toml");
std::fs::write(&p, "this is not [valid toml").unwrap(); std::fs::write(&p, "this is not [valid toml").unwrap();
let err = Config::from_file(&p).unwrap_err(); let err = Config::from_file(&p).unwrap_err();
let signal = err.downcast_ref::<ConfigInvalid>() let signal = err
.downcast_ref::<ConfigInvalid>()
.expect("malformed TOML should downcast to ConfigInvalid"); .expect("malformed TOML should downcast to ConfigInvalid");
assert_eq!(signal.path, p); assert_eq!(signal.path, p);
assert!(!signal.cause.is_empty(), "cause should be non-empty"); assert!(!signal.cause.is_empty(), "cause should be non-empty");
@@ -1864,13 +1877,11 @@ mod fb27_tests {
toml_text.contains("stale_threshold_days = 30"), toml_text.contains("stale_threshold_days = 30"),
"default value drifted; update test fixture" "default value drifted; update test fixture"
); );
toml_text = toml_text.replace( toml_text = toml_text.replace("stale_threshold_days = 30", "stale_threshold_days = -5");
"stale_threshold_days = 30",
"stale_threshold_days = -5",
);
std::fs::write(&p, &toml_text).unwrap(); std::fs::write(&p, &toml_text).unwrap();
let err = Config::from_file(&p).unwrap_err(); let err = Config::from_file(&p).unwrap_err();
let signal = err.downcast_ref::<ConfigInvalid>() let signal = err
.downcast_ref::<ConfigInvalid>()
.expect("negative stale_threshold_days should downcast to ConfigInvalid"); .expect("negative stale_threshold_days should downcast to ConfigInvalid");
assert_eq!(signal.path, p); assert_eq!(signal.path, p);
assert!( assert!(

View File

@@ -157,7 +157,9 @@ mod tests {
#[test] #[test]
fn xdg_data_home_set_replaces_var() { fn xdg_data_home_set_replaces_var() {
let _lock = ENV_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); let _lock = ENV_LOCK
.lock()
.unwrap_or_else(std::sync::PoisonError::into_inner);
let _guard = XdgGuard::capture(); let _guard = XdgGuard::capture();
// SAFETY: lock held for the duration of this test. // SAFETY: lock held for the duration of this test.
unsafe { std::env::set_var("XDG_DATA_HOME", "/custom/path") }; unsafe { std::env::set_var("XDG_DATA_HOME", "/custom/path") };
@@ -168,7 +170,9 @@ mod tests {
#[test] #[test]
fn xdg_data_home_unset_uses_default() { fn xdg_data_home_unset_uses_default() {
let _lock = ENV_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); let _lock = ENV_LOCK
.lock()
.unwrap_or_else(std::sync::PoisonError::into_inner);
let _guard = XdgGuard::capture(); let _guard = XdgGuard::capture();
// SAFETY: lock held for the duration of this test. // SAFETY: lock held for the duration of this test.
unsafe { std::env::remove_var("XDG_DATA_HOME") }; unsafe { std::env::remove_var("XDG_DATA_HOME") };
@@ -181,7 +185,9 @@ mod tests {
#[test] #[test]
fn xdg_with_no_default_resolves_to_empty_when_unset() { fn xdg_with_no_default_resolves_to_empty_when_unset() {
let _lock = ENV_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); let _lock = ENV_LOCK
.lock()
.unwrap_or_else(std::sync::PoisonError::into_inner);
let _guard = XdgGuard::capture(); let _guard = XdgGuard::capture();
// SAFETY: lock held for the duration of this test. // SAFETY: lock held for the duration of this test.
unsafe { std::env::remove_var("XDG_DATA_HOME") }; unsafe { std::env::remove_var("XDG_DATA_HOME") };
@@ -193,7 +199,9 @@ mod tests {
#[test] #[test]
fn leading_tilde_expands_to_home() { fn leading_tilde_expands_to_home() {
let _lock = ENV_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); let _lock = ENV_LOCK
.lock()
.unwrap_or_else(std::sync::PoisonError::into_inner);
let home = std::env::var("HOME").expect("HOME must be set in tests"); let home = std::env::var("HOME").expect("HOME must be set in tests");
let p = expand_path("~/runs", ""); let p = expand_path("~/runs", "");
assert_eq!(p, PathBuf::from(home).join("runs")); assert_eq!(p, PathBuf::from(home).join("runs"));
@@ -229,7 +237,9 @@ mod tests {
#[test] #[test]
fn tilde_path_ignores_base_dir() { fn tilde_path_ignores_base_dir() {
let _lock = ENV_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); let _lock = ENV_LOCK
.lock()
.unwrap_or_else(std::sync::PoisonError::into_inner);
let home = std::env::var("HOME").expect("HOME must be set in tests"); let home = std::env::var("HOME").expect("HOME must be set in tests");
let base = Path::new("/tmp/ignored-cfg"); let base = Path::new("/tmp/ignored-cfg");
let p = expand_path_with_base("~/x", "", base); let p = expand_path_with_base("~/x", "", base);
@@ -238,7 +248,9 @@ mod tests {
#[test] #[test]
fn xdg_var_path_ignores_base_dir() { fn xdg_var_path_ignores_base_dir() {
let _lock = ENV_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); let _lock = ENV_LOCK
.lock()
.unwrap_or_else(std::sync::PoisonError::into_inner);
let _guard = XdgGuard::capture(); let _guard = XdgGuard::capture();
// SAFETY: lock held for the duration of this test. // SAFETY: lock held for the duration of this test.
unsafe { std::env::set_var("XDG_DATA_HOME", "/xdg/data") }; unsafe { std::env::set_var("XDG_DATA_HOME", "/xdg/data") };
@@ -255,7 +267,9 @@ mod tests {
// Order matters: substitute `{data_dir}` (which itself contains // Order matters: substitute `{data_dir}` (which itself contains
// an unexpanded `${XDG_DATA_HOME}` and `~`), then the other two // an unexpanded `${XDG_DATA_HOME}` and `~`), then the other two
// resolve the result. // resolve the result.
let _lock = ENV_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner); let _lock = ENV_LOCK
.lock()
.unwrap_or_else(std::sync::PoisonError::into_inner);
let _guard = XdgGuard::capture(); let _guard = XdgGuard::capture();
// SAFETY: lock held for the duration of this test. // SAFETY: lock held for the duration of this test.
unsafe { std::env::set_var("XDG_DATA_HOME", "/xdg/data") }; unsafe { std::env::set_var("XDG_DATA_HOME", "/xdg/data") };

View File

@@ -2,13 +2,15 @@
// //
// Integration tests for [pdf.ocr] config section (v0.20.0 sub-item 1). // Integration tests for [pdf.ocr] config section (v0.20.0 sub-item 1).
use std::collections::HashMap;
use kebab_config::{Config, PdfCfg}; use kebab_config::{Config, PdfCfg};
use std::collections::HashMap;
// Test 1: toml roundtrip — spec §4.5 line 1034-1047 example block. // Test 1: toml roundtrip — spec §4.5 line 1034-1047 example block.
// Config requires many required fields; test the [pdf] section via PdfCfg wrapper. // Config requires many required fields; test the [pdf] section via PdfCfg wrapper.
#[derive(serde::Deserialize)] #[derive(serde::Deserialize)]
struct PdfWrapper { pdf: PdfCfg } struct PdfWrapper {
pdf: PdfCfg,
}
#[test] #[test]
fn pdf_ocr_toml_roundtrip() { fn pdf_ocr_toml_roundtrip() {
@@ -50,7 +52,10 @@ fn pdf_ocr_defaults_off_with_qwen_3b() {
assert_eq!(cfg.pdf.ocr.engine, "ollama-vision"); assert_eq!(cfg.pdf.ocr.engine, "ollama-vision");
assert_eq!(cfg.pdf.ocr.model, "qwen2.5vl:3b"); assert_eq!(cfg.pdf.ocr.model, "qwen2.5vl:3b");
assert!(cfg.pdf.ocr.endpoint.is_none()); assert!(cfg.pdf.ocr.endpoint.is_none());
assert_eq!(cfg.pdf.ocr.languages, vec!["eng".to_string(), "kor".to_string()]); assert_eq!(
cfg.pdf.ocr.languages,
vec!["eng".to_string(), "kor".to_string()]
);
assert_eq!(cfg.pdf.ocr.max_pixels, 2048); assert_eq!(cfg.pdf.ocr.max_pixels, 2048);
assert_eq!(cfg.pdf.ocr.request_timeout_secs, 180); // Bug #11: 600 → 60 → 180 (HOTFIXES 2026-05-28) assert_eq!(cfg.pdf.ocr.request_timeout_secs, 180); // Bug #11: 600 → 60 → 180 (HOTFIXES 2026-05-28)
assert!((cfg.pdf.ocr.valid_ratio_threshold - 0.5).abs() < 1e-6); assert!((cfg.pdf.ocr.valid_ratio_threshold - 0.5).abs() < 1e-6);
@@ -63,9 +68,15 @@ fn pdf_ocr_defaults_off_with_qwen_3b() {
fn pdf_ocr_env_overrides() { fn pdf_ocr_env_overrides() {
let mut env: HashMap<String, String> = HashMap::new(); let mut env: HashMap<String, String> = HashMap::new();
env.insert("KEBAB_PDF_OCR_ENABLED".to_string(), "true".to_string()); env.insert("KEBAB_PDF_OCR_ENABLED".to_string(), "true".to_string());
env.insert("KEBAB_PDF_OCR_MODEL".to_string(), "qwen2.5vl:7b".to_string()); env.insert(
"KEBAB_PDF_OCR_MODEL".to_string(),
"qwen2.5vl:7b".to_string(),
);
env.insert("KEBAB_PDF_OCR_ALWAYS_ON".to_string(), "true".to_string()); env.insert("KEBAB_PDF_OCR_ALWAYS_ON".to_string(), "true".to_string());
env.insert("KEBAB_PDF_OCR_VALID_RATIO_THRESHOLD".to_string(), "0.75".to_string()); env.insert(
"KEBAB_PDF_OCR_VALID_RATIO_THRESHOLD".to_string(),
"0.75".to_string(),
);
let cfg = Config::defaults().apply_env(&env); let cfg = Config::defaults().apply_env(&env);

View File

@@ -63,7 +63,9 @@ impl Citation {
/// fragment; they live in the structured wire object. /// fragment; they live in the structured wire object.
pub fn to_uri(&self) -> String { pub fn to_uri(&self) -> String {
match self { match self {
Citation::Line { path, start, end, .. } => { Citation::Line {
path, start, end, ..
} => {
if start == end { if start == end {
format!("{}#L{}", path.0, start) format!("{}#L{}", path.0, start)
} else { } else {

View File

@@ -235,7 +235,9 @@ mod tests {
href: "h".into(), href: "h".into(),
}, },
Inline::Strong { Inline::Strong {
children: vec![Inline::Text { text: "bold".into() }], children: vec![Inline::Text {
text: "bold".into(),
}],
}, },
Inline::Emph { Inline::Emph {
children: vec![Inline::Text { text: "em".into() }], children: vec![Inline::Text { text: "em".into() }],

View File

@@ -14,8 +14,7 @@ use crate::asset::WorkspacePath;
use crate::document::SourceSpan; use crate::document::SourceSpan;
use crate::errors::CoreError; use crate::errors::CoreError;
use crate::versions::{ use crate::versions::{
ChunkerVersion, EmbeddingModelId, EmbeddingVersion, IndexVersion, ChunkerVersion, EmbeddingModelId, EmbeddingVersion, IndexVersion, ParserVersion,
ParserVersion,
}; };
macro_rules! newtype_id { macro_rules! newtype_id {
@@ -54,9 +53,7 @@ fn validate_hex32(s: &str) -> Result<(), CoreError> {
))); )));
} }
if !s.bytes().all(|b| b.is_ascii_hexdigit()) { if !s.bytes().all(|b| b.is_ascii_hexdigit()) {
return Err(CoreError::InvalidId(format!( return Err(CoreError::InvalidId(format!("non-hex character in {s:?}")));
"non-hex character in {s:?}"
)));
} }
Ok(()) Ok(())
} }

View File

@@ -7,67 +7,63 @@
//! See `docs/superpowers/specs/2026-04-27-kebab-final-form-design.md` for //! See `docs/superpowers/specs/2026-04-27-kebab-final-form-design.md` for
//! the canonical type bodies — this crate is the byte-for-byte mirror. //! the canonical type bodies — this crate is the byte-for-byte mirror.
pub mod ids; pub mod answer;
pub mod versions;
pub mod media;
pub mod asset; pub mod asset;
pub mod document;
pub mod chunk; pub mod chunk;
pub mod citation; pub mod citation;
pub mod metadata; pub mod document;
pub mod search; pub mod errors;
pub mod answer; pub mod fetch;
pub mod ids;
pub mod ingest; pub mod ingest;
pub mod jobs; pub mod jobs;
pub mod vector; pub mod media;
pub mod errors; pub mod metadata;
pub mod traits;
pub mod normalize; pub mod normalize;
pub mod fetch; pub mod search;
pub mod traits;
pub mod vector;
pub mod versions;
// Re-export the most commonly used items at the crate root, mirroring the // Re-export the most commonly used items at the crate root, mirroring the
// public surface listed in the task spec. // public surface listed in the task spec.
pub use ids::{ pub use answer::{
AssetId, BlockId, ChunkId, DocumentId, EmbeddingId, IndexId, Answer, AnswerCitation, AnswerRetrievalSummary, HopKind, HopRecord, ModelRef, RefusalReason,
id_for_asset, id_for_block, id_for_chunk, id_for_doc, id_for_embedding, TokenUsage, TraceId, Turn, VerificationSummary,
id_for_index, id_from,
}; };
pub use versions::{
ChunkerVersion, EmbeddingModelId, EmbeddingVersion, IndexVersion,
ParserVersion, PromptTemplateVersion, SchemaVersion,
};
pub use media::{AudioType, Checksum, ImageType, Lang, MediaType};
pub use asset::{AssetStorage, RawAsset, SourceUri, WorkspacePath}; pub use asset::{AssetStorage, RawAsset, SourceUri, WorkspacePath};
pub use document::{
AudioRefBlock, Block, CanonicalDocument, CodeBlock, CommonBlock,
HeadingBlock, ImageRefBlock, Inline, ListBlock, ModelCaption, OcrRegion,
OcrText, SourceSpan, TableBlock, TextBlock, Transcript, TranscriptSegment,
};
pub use chunk::Chunk; pub use chunk::Chunk;
pub use citation::Citation; pub use citation::Citation;
pub use metadata::{ pub use document::{
Metadata, Provenance, ProvenanceEvent, ProvenanceKind, SourceType, AudioRefBlock, Block, CanonicalDocument, CodeBlock, CommonBlock, HeadingBlock, ImageRefBlock,
TrustLevel, Inline, ListBlock, ModelCaption, OcrRegion, OcrText, SourceSpan, TableBlock, TextBlock,
Transcript, TranscriptSegment,
}; };
pub use search::{ pub use errors::CoreError;
BulkSearchItem, BulkSearchResponse, BulkSearchSummary, DocFilter, DocSummary, IndexBytes, MEDIA_KINDS, pub use fetch::{FetchKind, FetchOpts, FetchQuery, FetchResult};
RetrievalDetail, ScoreKind, SearchFilters, SearchHit, SearchMode, SearchOpts, SearchQuery, SearchTrace, pub use ids::{
TraceCandidate, TraceFusionInput, TraceTiming, AssetId, BlockId, ChunkId, DocumentId, EmbeddingId, IndexId, id_for_asset, id_for_block,
}; id_for_chunk, id_for_doc, id_for_embedding, id_for_index, id_from,
pub use answer::{
Answer, AnswerCitation, AnswerRetrievalSummary, HopKind, HopRecord, ModelRef,
RefusalReason, TokenUsage, TraceId, Turn, VerificationSummary,
}; };
pub use ingest::{IngestItem, IngestItemKind, IngestReport, SkipExamples}; pub use ingest::{IngestItem, IngestItemKind, IngestReport, SkipExamples};
pub use jobs::{JobFilter, JobId, JobKind, JobRow, JobStatus}; pub use jobs::{JobFilter, JobId, JobKind, JobRow, JobStatus};
pub use vector::{VectorHit, VectorRecord}; pub use media::{AudioType, Checksum, ImageType, Lang, MediaType};
pub use errors::CoreError; pub use metadata::{Metadata, Provenance, ProvenanceEvent, ProvenanceKind, SourceType, TrustLevel};
pub use traits::{
ChatSessionRepo, ChatSessionRow, ChatTurnRow, ChunkPolicy, Chunker, DocumentStore,
Embedder, EmbeddingInput, EmbeddingKind, ExtractConfig, ExtractContext, Extractor,
FinishReason, GenerateRequest, JobRepo, LanguageModel, Retriever, SourceConnector,
SourceScope, TokenChunk, VectorStore,
};
pub use normalize::{nfc, to_posix}; pub use normalize::{nfc, to_posix};
pub use fetch::{FetchKind, FetchOpts, FetchQuery, FetchResult}; pub use search::{
BulkSearchItem, BulkSearchResponse, BulkSearchSummary, DocFilter, DocSummary, IndexBytes,
MEDIA_KINDS, RetrievalDetail, ScoreKind, SearchFilters, SearchHit, SearchMode, SearchOpts,
SearchQuery, SearchTrace, TraceCandidate, TraceFusionInput, TraceTiming,
};
pub use traits::{
ChatSessionRepo, ChatSessionRow, ChatTurnRow, ChunkPolicy, Chunker, DocumentStore, Embedder,
EmbeddingInput, EmbeddingKind, ExtractConfig, ExtractContext, Extractor, FinishReason,
GenerateRequest, JobRepo, LanguageModel, Retriever, SourceConnector, SourceScope, TokenChunk,
VectorStore,
};
pub use vector::{VectorHit, VectorRecord};
pub use versions::{
ChunkerVersion, EmbeddingModelId, EmbeddingVersion, IndexVersion, ParserVersion,
PromptTemplateVersion, SchemaVersion,
};

View File

@@ -317,7 +317,10 @@ mod tests {
#[test] #[test]
fn search_filters_serialize_with_serde_default_compat() { fn search_filters_serialize_with_serde_default_compat() {
let old: SearchFilters = serde_json::from_str(r#"{"tags_any":[],"lang":null,"path_glob":null,"trust_min":null}"#).unwrap(); let old: SearchFilters = serde_json::from_str(
r#"{"tags_any":[],"lang":null,"path_glob":null,"trust_min":null}"#,
)
.unwrap();
assert!(old.media.is_empty()); assert!(old.media.is_empty());
assert!(old.ingested_after.is_none()); assert!(old.ingested_after.is_none());
assert!(old.doc_id.is_none()); assert!(old.doc_id.is_none());
@@ -349,10 +352,7 @@ mod tests {
}; };
let v = serde_json::to_value(&t).unwrap(); let v = serde_json::to_value(&t).unwrap();
assert_eq!(v["timing"]["lexical_ms"], 12); assert_eq!(v["timing"]["lexical_ms"], 12);
assert_eq!( assert_eq!(v["lexical"][0]["score"].as_f64().unwrap() as f32, 0.42_f32);
v["lexical"][0]["score"].as_f64().unwrap() as f32,
0.42_f32
);
let back: SearchTrace = serde_json::from_value(v).unwrap(); let back: SearchTrace = serde_json::from_value(v).unwrap();
assert_eq!(back, t); assert_eq!(back, t);
} }
@@ -490,7 +490,10 @@ mod tests {
}; };
let v = serde_json::to_value(&hit).unwrap(); let v = serde_json::to_value(&hit).unwrap();
assert!(v.get("repo").is_none(), "repo should be omitted when None"); assert!(v.get("repo").is_none(), "repo should be omitted when None");
assert!(v.get("code_lang").is_none(), "code_lang should be omitted when None"); assert!(
v.get("code_lang").is_none(),
"code_lang should be omitted when None"
);
} }
#[test] #[test]

Some files were not shown because too many files have changed in this diff Show More