diff --git a/Cargo.lock b/Cargo.lock index 60477fa..b3b7ef6 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4751,7 +4751,7 @@ dependencies = [ [[package]] name = "kebab-app" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "base64 0.22.1", @@ -4799,7 +4799,7 @@ dependencies = [ [[package]] name = "kebab-chunk" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "blake3", @@ -4817,7 +4817,7 @@ dependencies = [ [[package]] name = "kebab-cli" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "clap", @@ -4838,7 +4838,7 @@ dependencies = [ [[package]] name = "kebab-config" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "dirs 5.0.1", @@ -4854,7 +4854,7 @@ dependencies = [ [[package]] name = "kebab-core" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "blake3", @@ -4868,7 +4868,7 @@ dependencies = [ [[package]] name = "kebab-embed" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "blake3", @@ -4882,7 +4882,7 @@ dependencies = [ [[package]] name = "kebab-embed-candle" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "candle-core", @@ -4902,7 +4902,7 @@ dependencies = [ [[package]] name = "kebab-embed-local" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "fastembed", @@ -4915,7 +4915,7 @@ dependencies = [ [[package]] name = "kebab-embed-ollama" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "kebab-config", @@ -4930,7 +4930,7 @@ dependencies = [ [[package]] name = "kebab-eval" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "kebab-app", @@ -4949,7 +4949,7 @@ dependencies = [ [[package]] name = "kebab-llm" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "kebab-core", @@ -4958,7 +4958,7 @@ dependencies = [ [[package]] name = "kebab-llm-local" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "kebab-config", @@ -4975,7 +4975,7 @@ dependencies = [ [[package]] name = "kebab-mcp" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "kebab-app", @@ -4993,7 +4993,7 @@ dependencies = [ [[package]] name = "kebab-nli" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "hf-hub", @@ -5008,7 +5008,7 @@ dependencies = [ [[package]] name = "kebab-parse-code" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "gix", @@ -5031,7 +5031,7 @@ dependencies = [ [[package]] name = "kebab-parse-image" -version = "0.29.0" +version = "0.30.0" dependencies = [ "ab_glyph", "anyhow", @@ -5059,7 +5059,7 @@ dependencies = [ [[package]] name = "kebab-parse-md" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "kebab-core", @@ -5076,7 +5076,7 @@ dependencies = [ [[package]] name = "kebab-parse-pdf" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "blake3", @@ -5091,7 +5091,7 @@ dependencies = [ [[package]] name = "kebab-rag" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "blake3", @@ -5113,7 +5113,7 @@ dependencies = [ [[package]] name = "kebab-search" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "globset", @@ -5132,7 +5132,7 @@ dependencies = [ [[package]] name = "kebab-source-fs" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "blake3", @@ -5150,7 +5150,7 @@ dependencies = [ [[package]] name = "kebab-store-sqlite" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "blake3", @@ -5170,7 +5170,7 @@ dependencies = [ [[package]] name = "kebab-store-vector" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "arrow", @@ -5194,7 +5194,7 @@ dependencies = [ [[package]] name = "kebab-tui" -version = "0.29.0" +version = "0.30.0" dependencies = [ "anyhow", "crossterm", diff --git a/Cargo.toml b/Cargo.toml index bc350a7..de90e46 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -32,7 +32,7 @@ edition = "2024" rust-version = "1.85" license = "MIT OR Apache-2.0" repository = "https://github.com/altair823/kebab" -version = "0.29.0" # v0.29.0 — provenance 출처 필터: `[[workspace.sources]]` 멀티소스 + 검색 `--source ` / `--source-type `(lexical+vector 두 site, OR). `documents.source_id` 컬럼(V014, additive·DEFAULT 'default'·재색인 0) + config v3→v4 migration(`step_3_to_4`, 단일 root→implicit `default` source 미러, 멱등). per-source `trust_level`/`source_type` 기본값(우선순위 frontmatter > source 기본값 > Primary). 단일 root 사용자 무영향. 설계 근거: 전역 trust 곱셈가중(weighted-RRF)은 A/B 반증(incident MRR 절벽), 출처 필터가 see-saw 없는 레버. 신규 CLI flag + config 키 + migration → minor. — CLAUDE.md §Release +version = "0.30.0" # v0.30.0 — md-heading-v2 청커: 예산 초과 청크 일반 분할. v1 의 "블록 미분할" 한계(거대 list/code/table/paragraph 가 한 청크→임베더 ctx 초과)를 일반화 — `token_estimate > max_chunk_tokens`(신규 config, byte/3, default 4000)인 청크만 줄(→UTF-8 char) 경계로 분할, 각 조각 ≤ 예산. 분할 조각 chunk_id 는 `#seg{i}` 접미사로 충돌 회피, `max_chunk_tokens` 는 v2 policy_hash 에 fold(공유 ChunkPolicy 미변경). 미분할 청크는 v1 과 byte-identical. `chunker_version` v1→v2 → 다음 ingest 에서 markdown 1회 자동 재청크(코드/PDF 무영향). 동기: strict 임베더(AMD Lemonade)는 oversize 입력을 truncate 아닌 거부 → 청커가 애초에 안 만드는 게 옳음. 신규 config 키 + 청커 동작(검색 hit 분할) 변경 → minor + 도그푸딩 트리거. — CLAUDE.md §Release # pre-v0.18 workspace-wide cleanup: enable clippy::pedantic group with # intentional allow-list. The allowed lints are either cosmetic (doc style), diff --git a/HANDOFF.md b/HANDOFF.md index cc64166..75d6d82 100644 --- a/HANDOFF.md +++ b/HANDOFF.md @@ -35,6 +35,7 @@ P0~P5 직렬. P6~P9 P5 이후 병렬 가능. 머지 후 발견된 모든 deviation / hotfix 의 dated 로그는 [tasks/HOTFIXES.md](tasks/HOTFIXES.md). 본 요약은 \"누군가가 인수받을 때 알아두면 시간을 많이 절약하는\" 항목만: +- **2026-06-24 md-heading-v2: 예산 초과 청크 일반 분할** — v0.30.0. markdown 청커가 v1 의 "블록 미분할" 한계를 일반화 — 거대 list/code/table/paragraph 가 한 청크로 임베더 ctx 를 초과하던 문제를, `token_estimate > max_chunk_tokens`(신규 config, byte/3, default 4000)인 청크만 줄(→UTF-8 char) 경계로 분할해 해소. 미분할 청크는 v1 과 byte-identical. 분할 조각 chunk_id 는 `#seg{i}` 접미사로 충돌 회피, `max_chunk_tokens` 는 v2 policy_hash 에 fold(공유 ChunkPolicy 미변경). `chunker_version` v1→v2 라 다음 plain ingest 에서 markdown 1회 자동 재청크(코드/PDF 무영향). **동기**: strict 임베더(AMD Lemonade `/api/embed`)는 oversize 입력을 truncate 아닌 거부(`500 too large`) — ollama 가 조용히 truncate 하던 걸 청커가 애초에 안 만들도록. **known limitation**: 분할 조각 citation 은 블록 단위(sub-line 정밀 아님). 도그푸딩(실험 KB, arctic@Lemonade): v2 전 620 중 2 doc 임베드 실패 → v2 후 620/620·7114 청크 전부 ≤4000, "WiredTiger excessive memory" 질의에 거대 doc SERVER-22906 가 1위(0.977). 자세한 내용: `tasks/HOTFIXES.md` (2026-06-24), 설계 `docs/superpowers/plans/2026-06-24-md-heading-v2-oversize-split.md`. - **2026-06-21 provenance 출처 필터: `[[workspace.sources]]` 멀티소스 + `--source`/`--source-type`** — v0.29.0. 혼합 출처 KB(위키+jira 등)에서 색인은 전부 하되 질의 시 출처로 좁히는 레버. config `[[workspace.sources]]`(각 id/root/trust_level/source_type) + `documents.source_id` 컬럼(V014, additive, 재색인 0) + config v3→v4 migration(`step_3_to_4`, 단일 root→implicit `default` source, 멱등) + 검색 `--source ` / `--source-type `(lexical+vector 두 site, OR). trust precedence = frontmatter > per-source 기본값 > Primary. **설계 근거**: 전역 trust 곱셈가중(weighted-RRF)은 A/B 에서 반증(θ=0.85 만으로 incident MRR 0.918→0.340 절벽) — 필터가 see-saw 없는 올바른 레버. 도그푸딩(620 doc, jira400+wiki220): `--source wiki` concept 0.780→0.810, `--source jira` incident 0.918→0.975. **follow-up**: MCP search 필터 미노출 · `kebab list` source_id 미표시 · RAG provenance 라벨 미구현. 자세한 내용: `tasks/HOTFIXES.md` (2026-06-21). - **2026-06-04 PP-OCRv5 ONNX Rust 네이티브 OCR** — v0.27.0. `[image.ocr] engine = "paddle-onnx"` 로 PP-OCRv5(검출+인식) ONNX 를 in-process(`ort` =2.0.0-rc.9) 실행 — Python 런타임/원격 호출 없이 큰 페이지 CPU <4초(Ollama vision ~50초 대비). default 는 여전히 `"ollama-vision"`. 후처리(min-area rect/unclip)는 pure-Rust. **함정**: unclip 은 corner 를 centroid 에서 방사 확장하면 안 되고 edge 별 polygon offset 이어야 함(방사 확장 시 wide/short 텍스트 박스 높이가 안 커져 글자 윗부분 잘림 → ㄷ→ㄴ, e2e CER 0.26). 수정 후 CER 0.005. 모델 ONNX 는 `crates/kebab-parse-image/assets/paddleocr-onnx/`(LFS). 자세한 내용: `tasks/HOTFIXES.md` (2026-06-04 PP-OCRv5 ONNX), spec/plan `docs/superpowers/{specs,plans}/2026-06-04-rust-native-ocr-*.md`. - **2026-06-03 ingest 설정 변경 자동 재색인** — v0.26.2. ingest 산출에 영향 주는 설정(청킹/이미지 OCR·caption/pdf.ocr/`[ingest.code]`)을 변경하면 `--force-reingest` 없이 영향 자산만 자동 재색인. 그 설정들의 결정적 서명(`ingest_config_signature`)을 effective parser_version(skip 비교 + 저장 doc 필드 양쪽)에 폴딩 → 다음 ingest 비교가 mismatch. 비산출 설정(search/rag/ui/log + max_pixels/languages/timeout)은 제외(과도 무효화 회피), doc_id 는 base 로 안정 유지. **업그레이드 후 첫 ingest 는 전 자산 1회 재색인**(저장된 상수 parser_version ≠ 새 composite; embedding 은 V012 캐시 히트). 결과 포맷·CLI·wire 불변(내부 skip 판정 정정). 자세한 내용: `tasks/HOTFIXES.md` (2026-06-03 ingest 설정 변경 자동 재색인), spec/plan `docs/superpowers/{specs,plans}/2026-06-03-*invalidation*.md`. diff --git a/README.md b/README.md index 10ddc17..9a0be8b 100644 --- a/README.md +++ b/README.md @@ -196,6 +196,7 @@ nli_threshold = 0.0 # >0 (예: 0.5) 면 mDeBERTa XNLI groundedn ``` - **`[ingest]`** (v0.28.0) — 모든 형식 ingest 설정의 우산. 병렬도(`max_parallel_extractors`/`max_parallel_embeddings`/`watch_filesystem`, ← 옛 `[indexing]`)와 형식별 하위 절(`[ingest.chunking]` ← 옛 `[chunking]`, `[ingest.code]`, `[ingest.image.ocr]` ← 옛 `[image.ocr]`, `[ingest.pdf.ocr]` ← 옛 `[pdf.ocr]`)이 전부 이 아래로 모인다. 기존 v2 `config.toml` 은 그대로 둬도 로드 시 메모리에서 자동 변환되며, 파일을 새 레이아웃으로 갱신하려면 `kebab config migrate` (값·주석 보존). +- **`[ingest.chunking]`** — 청크 크기·오버랩·heading 존중. `chunker_version` 기본 `"md-heading-v2"` (v0.30.0). **`max_chunk_tokens`** (default 4000, byte/3 토큰) — 이 값을 넘는 청크는 줄(→UTF-8 char) 경계로 분할해 각 조각이 예산 이하가 되게 한다. 거대 list/code/log 덤프가 한 청크로 임베더 컨텍스트를 초과하던 문제를 막는다(미분할 청크는 v0.29.0 `md-heading-v1` 과 출력 동일). 이 값을 바꾸면 markdown 자산이 자동 재청크된다. - **파생물 캐시** — embedding 결과를 내용 해시로 자동 캐싱한다 (위 「핵심 기능」 참고). 설정 항목 없음. - **`[ingest.code]`** — code ingest 의 skip 정책 (`skip_generated_header`, `max_file_bytes`, `extra_skip_globs`). `.gitignore` 자동 honor, `.kebabignore` 는 추가 layer. - **`[ingest.image.ocr]`** — 이미지 OCR (default off / opt-in). `engine` 으로 백엔드 선택: `"ollama-vision"` (default, 원격 vision LM) 또는 `"paddle-onnx"` (PP-OCRv5 ONNX 를 in-process 로 실행, Python 런타임 불필요, 큰 페이지 CPU <4초, 오프라인). `paddle-onnx` 는 워크스페이스에 번들된 모델을 쓰며 `det_model`/`rec_model`/`dict` 로 경로 override, `score_thresh`(0.3)/`unclip_ratio`(1.5)/`max_boxes`(1000) 로 검출 튜닝 가능 (`KEBAB_IMAGE_OCR_*` env 동일 지원 — env 이름은 v3 에서도 불변). engine 또는 모델을 바꾸면 영향 이미지가 자동 재색인된다. diff --git a/crates/kebab-app/src/lib.rs b/crates/kebab-app/src/lib.rs index d896c3e..7ed4a4e 100644 --- a/crates/kebab-app/src/lib.rs +++ b/crates/kebab-app/src/lib.rs @@ -43,7 +43,7 @@ use kebab_chunk::{ CodeCAstV1Chunker, CodeCppAstV1Chunker, CodeGoAstV1Chunker, CodeJavaAstV1Chunker, CodeJsAstV1Chunker, CodeKotlinAstV1Chunker, CodePythonAstV1Chunker, CodeRustAstV1Chunker, CodeTextParagraphV1Chunker, CodeTsAstV1Chunker, DockerfileFileV1Chunker, - K8sManifestResourceV1Chunker, ManifestFileV1Chunker, MdHeadingV1Chunker, PdfPageV1Chunker, + K8sManifestResourceV1Chunker, ManifestFileV1Chunker, MdHeadingV2Chunker, PdfPageV1Chunker, }; use kebab_core::{ Answer, Block, CanonicalDocument, Chunk, ChunkId, ChunkPolicy, Chunker, ChunkerVersion, @@ -1388,7 +1388,7 @@ fn ingest_one_asset( app, asset, &eff_parser_version, - &MdHeadingV1Chunker.chunker_version(), + &md_chunker_from_config(&app.config).chunker_version(), embedder.map(|e| e.model_version()).as_ref(), force_reingest, None, @@ -1441,9 +1441,9 @@ fn ingest_one_asset( let parse_ms = u64::try_from(t_parse.elapsed().as_millis()).unwrap_or(u64::MAX); let t_chunk = std::time::Instant::now(); - let chunks = MdHeadingV1Chunker + let chunks = md_chunker_from_config(&app.config) .chunk(&canonical, chunk_policy) - .context("kb-chunk::MdHeadingV1Chunker::chunk")?; + .context("kb-chunk::MdHeadingV2Chunker::chunk")?; let chunk_ms = u64::try_from(t_chunk.elapsed().as_millis()).unwrap_or(u64::MAX); // v0.24.0: surface the chunk count immediately, before the (potentially @@ -1465,7 +1465,7 @@ fn ingest_one_asset( // Stamp chunker + embedding versions so Task 7's skip detection has // data on the second run. - canonical.last_chunker_version = Some(MdHeadingV1Chunker.chunker_version()); + canonical.last_chunker_version = Some(md_chunker_from_config(&app.config).chunker_version()); if let Some(emb) = embedder { canonical.last_embedding_version = Some(emb.model_version()); } @@ -1607,7 +1607,7 @@ fn ingest_one_asset( block_count: u32::try_from(canonical.blocks.len()).ok(), chunk_count: u32::try_from(chunks.len()).ok(), parser_version: Some(parser_version.clone()), - chunker_version: Some(MdHeadingV1Chunker.chunker_version()), + chunker_version: Some(md_chunker_from_config(&app.config).chunker_version()), warnings: warning_notes, pdf_ocr_pages: None, pdf_ocr_ms_total: None, @@ -1666,7 +1666,7 @@ fn ingest_one_image_asset( }; // p9-fb-23 task 7: incremental-ingest early-skip for the image flow. // Image docs use the `image-meta-v1` parser_version + the same - // MdHeadingV1Chunker as the markdown flow (single-block doc). The + // MdHeadingV2Chunker as the markdown flow (single-block doc). The // embedding-version check matches the markdown path: when the // active embedder's model_version equals what was stamped on the // existing doc, the asset is Unchanged. @@ -1679,7 +1679,7 @@ fn ingest_one_image_asset( app, asset, &eff_parser_version, - &MdHeadingV1Chunker.chunker_version(), + &md_chunker_from_config(&app.config).chunker_version(), embedder.map(|e| e.model_version()).as_ref(), force_reingest, None, @@ -1828,14 +1828,17 @@ fn ingest_one_image_asset( } } - // 4. Chunk via the same `MdHeadingV1Chunker` markdown uses — its + // 4. Chunk via the same `MdHeadingV2Chunker` markdown uses — its // `Block::ImageRef` arm already produces a single chunk per - // image (P1-5). The chunk text now follows the (β) plain-concat - // contract per the kebab-chunk render_block_text update. + // image (P1-5). The chunk text follows the (β) plain-concat + // contract per the kebab-chunk render_block_text update. Using v2 + // here keeps the markdown family consistent: a pathologically + // large OCR text dump splits at line boundaries just like a giant + // fenced code block would, instead of overflowing the embedder. let t_chunk = std::time::Instant::now(); - let chunks = MdHeadingV1Chunker + let chunks = md_chunker_from_config(&app.config) .chunk(&canonical, chunk_policy) - .context("kb-chunk::MdHeadingV1Chunker::chunk (image)")?; + .context("kb-chunk::MdHeadingV2Chunker::chunk (image)")?; let chunk_ms = u64::try_from(t_chunk.elapsed().as_millis()).unwrap_or(u64::MAX); // v0.24.0: surface chunk count for the image path too. @@ -1849,9 +1852,9 @@ fn ingest_one_image_asset( ); // 5. Persist + embed — identical sequence to markdown. - // Stamp chunker + embedding versions (image uses MdHeadingV1Chunker + // Stamp chunker + embedding versions (image uses MdHeadingV2Chunker // for its single-block doc, so we record that version). - canonical.last_chunker_version = Some(MdHeadingV1Chunker.chunker_version()); + canonical.last_chunker_version = Some(md_chunker_from_config(&app.config).chunker_version()); if let Some(emb) = embedder { canonical.last_embedding_version = Some(emb.model_version()); } @@ -1956,7 +1959,7 @@ fn ingest_one_image_asset( block_count: u32::try_from(canonical.blocks.len()).ok(), chunk_count: u32::try_from(chunks.len()).ok(), parser_version: Some(canonical.parser_version.clone()), - chunker_version: Some(MdHeadingV1Chunker.chunker_version()), + chunker_version: Some(md_chunker_from_config(&app.config).chunker_version()), warnings: warning_notes, pdf_ocr_pages: None, pdf_ocr_ms_total: None, @@ -3154,6 +3157,19 @@ fn chunk_policy_from_config(config: &kebab_config::Config) -> ChunkPolicy { } } +/// Construct the markdown chunker (the hardcoded `md-heading-v2`) with the +/// split budget threaded from config. Used by the markdown ingest path +/// AND the image-OCR / caption path (which flows its synthetic +/// `Block::ImageRef` text through the same chunker), so a giant OCR dump +/// is split like any other oversize chunk. The PDF path stays pinned to +/// `pdf-page-v1` and code paths keep their own AST chunkers — only the +/// markdown-family default moved v1 → v2. +fn md_chunker_from_config(config: &kebab_config::Config) -> MdHeadingV2Chunker { + MdHeadingV2Chunker { + max_chunk_tokens: config.ingest.chunking.max_chunk_tokens, + } +} + /// v0.26.2: deterministic signature of the **ingest-output-affecting** /// config for an asset's media type, folded into the effective /// `parser_version` (both the `try_skip_unchanged` compare field AND the @@ -3240,9 +3256,20 @@ fn ingest_config_signature(config: &kebab_config::Config, media: &MediaType) -> // boundaries. `target_tokens` / `overlap_tokens` change re-chunking for // markdown / image / pdf / code alike, so a change re-indexes all types. let c = &config.ingest.chunking; + // `max_chunk_tokens` is appended as a 5th field: md-heading-v2 + // splits any oversize chunk (list, code, paragraph, table) at this + // budget, so changing it moves markdown chunk boundaries and must + // re-index. It also folds into the v2 policy_hash, but the signature + // is what the no-`--force` skip-check compares, so it must be here + // too. Appended (not inserted) so the existing 4-field prefix + // `chunk:T:O:H:V` stays a stable substring for any existing golden. let mut sig = format!( - "chunk:{}:{}:{}:{}", - c.target_tokens, c.overlap_tokens, c.respect_markdown_headings, c.chunker_version + "chunk:{}:{}:{}:{}:{}", + c.target_tokens, + c.overlap_tokens, + c.respect_markdown_headings, + c.chunker_version, + c.max_chunk_tokens ); match media { MediaType::Image(_) => { diff --git a/crates/kebab-app/tests/config_invalidation.rs b/crates/kebab-app/tests/config_invalidation.rs index 237fdcf..8f0c6d1 100644 --- a/crates/kebab-app/tests/config_invalidation.rs +++ b/crates/kebab-app/tests/config_invalidation.rs @@ -160,8 +160,10 @@ fn ingest_signature_image_paddle_byte_stable() { &kebab_core::MediaType::Image(kebab_core::ImageType::Png), ); // 골든: chunk:... |ocr:1:paddle-onnx: |cap:0 + // md-heading-v2 가 markdown 기본값 + max_chunk_tokens(4000) 가 + // signature 5번째 필드로 추가됐다 — budget 변경 시 자동 재색인용. assert!( - sig.starts_with("chunk:500:80:true:md-heading-v1"), + sig.starts_with("chunk:500:80:true:md-heading-v2:4000"), "chunk prefix drift: {sig}" ); assert!(sig.contains("|ocr:1:paddle-onnx:"), "ocr token drift: {sig}"); diff --git a/crates/kebab-chunk/src/lib.rs b/crates/kebab-chunk/src/lib.rs index 00256d3..e3ae88d 100644 --- a/crates/kebab-chunk/src/lib.rs +++ b/crates/kebab-chunk/src/lib.rs @@ -7,8 +7,15 @@ //! //! * [`MdHeadingV1Chunker`] — heading-aware chunker for Markdown //! `CanonicalDocument`s, emitting `chunker_version = "md-heading-v1"`. +//! * [`MdHeadingV2Chunker`] — byte-identical to v1 in its chunking pass, +//! then applies a generic post-pass: any chunk whose byte/3 estimate +//! exceeds `max_chunk_tokens` is split at line (then UTF-8 char) +//! boundaries. Covers all block kinds (list, code, paragraph, table). +//! Emits `chunker_version = "md-heading-v2"`; the hardcoded markdown +//! default (design §9 label bump). //! -//! Behavior contract is enumerated on [`MdHeadingV1Chunker`]. +//! Behavior contract is enumerated on [`MdHeadingV1Chunker`] (v2 inherits +//! it; the divergence is the generic post-pass documented on [`MdHeadingV2Chunker`]). //! //! This crate must NOT depend on any parser implementation //! (`kb-parse-md`, `kb-parse-pdf`, …), the document/vector store, the @@ -29,6 +36,7 @@ pub mod dockerfile_file_v1; pub mod k8s_manifest_resource_v1; pub mod manifest_file_v1; mod md_heading_v1; +mod md_heading_v2; mod pdf_page_v1; mod tier2_shared; @@ -46,6 +54,7 @@ pub use dockerfile_file_v1::DockerfileFileV1Chunker; pub use k8s_manifest_resource_v1::K8sManifestResourceV1Chunker; pub use manifest_file_v1::ManifestFileV1Chunker; pub use md_heading_v1::MdHeadingV1Chunker; +pub use md_heading_v2::MdHeadingV2Chunker; pub use pdf_page_v1::PdfPageV1Chunker; // ── Korean morphological tokenizer ─────────────────────────────────────────── diff --git a/crates/kebab-chunk/src/md_heading_v2.rs b/crates/kebab-chunk/src/md_heading_v2.rs new file mode 100644 index 0000000..274a896 --- /dev/null +++ b/crates/kebab-chunk/src/md_heading_v2.rs @@ -0,0 +1,1078 @@ +//! `md-heading-v2` — heading-aware Markdown chunker with a generic +//! oversize-chunk post-pass. +//! +//! ## Why a new variant (design §9 label-bump) +//! +//! `md-heading-v1` emits every block — code, table, list, paragraph — as +//! a single chunk regardless of size. Pathologically large blocks observed +//! in real Jira exports: a `Block::List` of 76 189 tokens (SERVER-22906, +//! source lines 60–1303, a log dump rendered as bullet points) and a +//! `Block::List` of 20 643 tokens (SERVER-23097). These exceed any +//! practical embedder context window and fail to embed on strict servers. +//! +//! `md-heading-v2`'s main `chunk()` is **byte-identical to v1** — it +//! produces the same chunk set v1 would (same block merging, overlap, +//! heading boundaries). The difference is a **generic post-pass**: after +//! producing v1-equivalent chunks, any chunk whose `token_estimate` +//! exceeds `self.max_chunk_tokens` is replaced by its line-split (then, +//! for single-line content, UTF-8 char-split) sub-pieces. Chunks at/below +//! the budget pass through entirely unchanged, guaranteeing v1 parity for +//! all normal content. +//! +//! ## Split disambiguation +//! +//! All pieces of one oversize chunk share the same `block_ids` (the chunk +//! may span multiple blocks). We disambiguate chunk_ids with a `#seg{i}` +//! suffix on the policy-hash argument to `id_for_chunk`, while storing the +//! bare base hash in `Chunk.policy_hash` — the same recipe +//! `code_rust_ast_v1` uses with `#L{line}`. +//! +//! ## Budget in `policy_hash` +//! +//! The budget changes output, so it MUST participate in the chunk-id +//! cascade (design §9). The shared `ChunkPolicy` has no budget field +//! (adding one would widen the cascade to every chunker including pdf/code). +//! Instead, v2 overrides `policy_hash()` to concatenate the canonical +//! `ChunkPolicy` bytes with the 8 LE bytes of `self.max_chunk_tokens`. +//! v1's `policy_hash` is left untouched. +//! +//! ## Source-span citation +//! +//! Split pieces inherit the **original chunk's `source_spans`** (block- +//! granular citation). This is intentional: we do not have sub-line source +//! map data for arbitrary block kinds, and a citation to the enclosing +//! block/region is always correct — never wrong, just not sub-line-precise. + +use kebab_core::{ + Block, BlockId, CanonicalDocument, Chunk, ChunkPolicy, Chunker, ChunkerVersion, DocumentId, + SourceSpan, id_for_chunk, +}; + +/// Version label emitted by [`MdHeadingV2Chunker`]. Distinct from +/// `md-heading-v1` so the version cascade (design §9) re-chunks every +/// markdown doc on the first ingest after v2 becomes the default. +const VERSION_LABEL: &str = "md-heading-v2"; + +/// Bytes-per-token proxy — identical to v1. 3 bytes/token over-estimates +/// token count for both Korean (E5 ≈ 3) and English (BPE ≈ 4) so chunks +/// sized against this proxy always fit a real tokenizer's budget. +const BYTES_PER_TOKEN: usize = 3; + +/// Maximum hex characters of the policy hash. 16 hex = 64 bits, matching v1. +const POLICY_HASH_HEX_LEN: usize = 16; + +/// Heading-aware Markdown chunker with a generic oversize-chunk post-pass. +/// +/// Main `chunk()` is byte-identical to v1. After building the v1-equivalent +/// chunk list, any chunk whose `token_estimate > self.max_chunk_tokens` is +/// replaced by line-split (then char-split for single-line content) +/// sub-pieces each ≤ budget. Chunks at/below budget pass through unchanged. +/// +/// Not a unit struct — it carries the split budget threaded from +/// `config.ingest.chunking.max_chunk_tokens`. +#[derive(Clone, Copy, Debug)] +pub struct MdHeadingV2Chunker { + /// Max byte/3 token estimate per emitted chunk. Any chunk produced by + /// the v1-equivalent pass whose `token_estimate` exceeds this is split + /// at line (then UTF-8 char) boundaries. Folded into `policy_hash` so + /// changing this budget triggers a re-chunk via the cascade (design §9). + pub max_chunk_tokens: usize, +} + +impl Chunker for MdHeadingV2Chunker { + fn chunker_version(&self) -> ChunkerVersion { + ChunkerVersion(VERSION_LABEL.to_string()) + } + + /// Policy hash with the v2-only budget folded in. + /// + /// We append the 8 LE bytes of `self.max_chunk_tokens` to the canonical + /// `ChunkPolicy` JSON before hashing. The canonical JSON is + /// self-delimiting (a balanced JSON object), so there is no structural + /// ambiguity between the policy bytes and the trailing 8 budget bytes. + /// + /// This means two v2 instances with different `max_chunk_tokens` produce + /// different `policy_hash` values — and therefore different chunk_ids — + /// for the same document, ensuring a budget change re-indexes all md + /// docs rather than leaving stale oversized chunks from the previous run. + /// + /// # Panics + /// + /// Panics if canonical JSON serialization of `ChunkPolicy` fails — + /// unreachable in practice (see `MdHeadingV1Chunker::policy_hash` for + /// the full guard rationale). + fn policy_hash(&self, policy: &ChunkPolicy) -> String { + let mut bytes = serde_json_canonicalizer::to_vec(policy) + .expect("canonical JSON serialization of ChunkPolicy must not fail"); + bytes.extend_from_slice(&self.max_chunk_tokens.to_le_bytes()); + let hex = blake3::hash(&bytes).to_hex().to_string(); + hex[..POLICY_HASH_HEX_LEN].to_string() + } + + fn chunk(&self, doc: &CanonicalDocument, policy: &ChunkPolicy) -> anyhow::Result> { + let policy_hash = self.policy_hash(policy); + let chunker_version = self.chunker_version(); + + // ── Phase 1: v1-equivalent chunking ───────────────────────────── + // Byte-identical to MdHeadingV1Chunker::chunk() — same block-merge + // logic, same heading boundaries, same overlap seeding, same atomic + // Code/Table arm. The only callers that see a difference are those + // that hold chunks exceeding `self.max_chunk_tokens`. + let v1_chunks = chunk_v1_equivalent(doc, &chunker_version, &policy_hash, policy); + + // ── Phase 2: generic oversize post-pass ───────────────────────── + // Any chunk at/below budget passes through UNTOUCHED (this is what + // guarantees byte-identical output to v1 for all normal content). + // Any chunk above budget is replaced by its split sub-pieces. + let budget = self.max_chunk_tokens; + let mut out: Vec = Vec::with_capacity(v1_chunks.len()); + for chunk in v1_chunks { + // Decide on the chunk's ACTUAL embedded text size, NOT the stored + // `token_estimate`. For a text chunk the two are equal + // (`token_estimate == text.len()/BYTES_PER_TOKEN`), so non-image + // output is unchanged. But ImageRef/AudioRef chunks report + // `token_estimate = 0` by the image-only convention (build_chunk), + // while their `text` (alt + OCR + caption) can be arbitrarily large + // — e.g. a dense screenshot OCR'd to tens of KB. Keying the split + // on `text.len()` is what the embedder actually receives, so an + // oversize image/audio chunk splits like any other. + let embed_tokens = chunk.text.len().div_ceil(BYTES_PER_TOKEN); + if embed_tokens <= budget { + out.push(chunk); + } else { + // Oversize: split into line-then-char pieces each ≤ budget. + // The original chunk's block_ids / source_spans / heading_path + // are shared across all pieces (see module-level doc comment + // on source-span citation rationale). + let pieces = split_oversize_chunk(&chunk, budget, &chunker_version, &policy_hash); + out.extend(pieces); + } + } + + tracing::debug!( + target: "kebab-chunk", + doc_id = %doc.doc_id, + chunks = out.len(), + "md-heading-v2 chunked", + ); + + Ok(out) + } +} + +// ── v1-equivalent chunking (phase 1) ──────────────────────────────────────── + +/// Produce the same `Vec` that `MdHeadingV1Chunker` would produce. +/// Extracted as a free function so v2's `chunk()` can call it then apply +/// the post-pass cleanly. +fn chunk_v1_equivalent( + doc: &CanonicalDocument, + chunker_version: &ChunkerVersion, + policy_hash: &str, + policy: &ChunkPolicy, +) -> Vec { + let mut out: Vec = Vec::new(); + let mut acc = ChunkAcc::default(); + + for block in &doc.blocks { + match block { + Block::Heading(_) => { + flush(&mut acc, doc, chunker_version, policy_hash, &mut out); + acc.push_block(block); + } + // Atomic blocks — Code never splits per v1 priority 2; Table + // stays single per priority 3. The generic post-pass (phase 2) + // handles splitting if either turns out oversize. + Block::Code(_) | Block::Table(_) => { + flush(&mut acc, doc, chunker_version, policy_hash, &mut out); + let mut single = ChunkAcc::default(); + single.push_block(block); + flush(&mut single, doc, chunker_version, policy_hash, &mut out); + } + Block::ImageRef(_) | Block::AudioRef(_) => { + flush(&mut acc, doc, chunker_version, policy_hash, &mut out); + let mut single = ChunkAcc::default(); + single.push_block(block); + flush(&mut single, doc, chunker_version, policy_hash, &mut out); + } + Block::Paragraph(_) | Block::List(_) | Block::Quote(_) => { + let next_tokens = estimate_block_tokens(block); + let would_exceed = acc.text_tokens + next_tokens > policy.target_tokens + && acc.has_non_heading_content(); + if would_exceed { + let overlap_seed = + collect_overlap_seed(&acc, policy.overlap_tokens, policy.target_tokens); + flush(&mut acc, doc, chunker_version, policy_hash, &mut out); + for b in overlap_seed { + acc.push_block(b); + } + } + acc.push_block(block); + } + } + } + flush(&mut acc, doc, chunker_version, policy_hash, &mut out); + out +} + +// ── Generic oversize-chunk splitter (phase 2) ─────────────────────────────── + +/// Split an oversize `Chunk` into sub-pieces each with +/// `token_estimate <= budget`. +/// +/// ## Splitting strategy (two tiers): +/// +/// 1. **Line split**: split `chunk.text` on `'\n'` and greedily accumulate +/// whole lines into pieces each ≤ budget (byte/3). A single line that +/// alone exceeds the budget drops to the next tier. +/// +/// 2. **Char split (lone-oversize-line fallback)**: if a single line's +/// byte/3 estimate still exceeds `budget`, split it at UTF-8 char +/// boundaries, accumulating chars until the NEXT char would push the +/// piece over `budget * BYTES_PER_TOKEN` bytes. This guarantees the +/// budget bound for every piece, including pathological inputs like a +/// mega-paragraph rendered as one line. +/// +/// ## Piece metadata +/// +/// Each piece is a new `Chunk` that inherits: +/// - `doc_id`, `block_ids`, `heading_path`, `chunker_version`, +/// `source_spans`: copied from the original (block-granular citation). +/// - `text`: the piece text. +/// - `token_estimate`: piece `len().div_ceil(BYTES_PER_TOKEN)`. +/// - `tokenized_korean_text`: recomputed for the piece. +/// - `policy_hash`: the BARE base hash (the `#seg{i}` suffix lives only +/// in the id-input hash, never in the persisted field). +/// - `chunk_id`: `id_for_chunk(doc_id, version, block_ids, +/// "{base_hash}#seg{i}")` where `i` is the 0-based piece index — +/// the sole id disambiguator since `block_ids` is identical across pieces. +fn split_oversize_chunk( + chunk: &Chunk, + budget: usize, + chunker_version: &ChunkerVersion, + base_policy_hash: &str, +) -> Vec { + // Collect all sub-piece texts first. + let pieces: Vec = text_pieces(&chunk.text, budget); + + // Safety invariant: split must produce ≥1 piece. + debug_assert!(!pieces.is_empty(), "text_pieces must return ≥1 piece"); + + pieces + .into_iter() + .enumerate() + .map(|(i, text)| { + let id_hash = format!("{base_policy_hash}#seg{i}"); + let chunk_id = + id_for_chunk(&chunk.doc_id, chunker_version, &chunk.block_ids, &id_hash); + let token_estimate = text.len().div_ceil(BYTES_PER_TOKEN); + Chunk { + chunk_id, + doc_id: chunk.doc_id.clone(), + block_ids: chunk.block_ids.clone(), + tokenized_korean_text: crate::tokenize_korean_morphological(&text), + text, + heading_path: chunk.heading_path.clone(), + // Source spans are cloned from the original chunk: block- + // granular citation is always correct for a split piece (it + // cites the enclosing block/region). We don't have sub-line + // source-map data for arbitrary block kinds. + source_spans: chunk.source_spans.clone(), + token_estimate, + chunker_version: chunker_version.clone(), + // Store the BARE base hash; the #seg suffix is id-input only. + policy_hash: base_policy_hash.to_string(), + } + }) + .collect() +} + +/// Decompose `text` into sub-pieces each with `len().div_ceil(BYTES_PER_TOKEN) +/// <= budget`. Returns ≥1 piece. The pieces join back to the original with +/// `\n` (line splits) or direct concatenation (char splits within a single +/// line), preserving the full text. +/// +/// The returned vec is ordered; concatenating the pieces in order with `\n` +/// reconstructs `text` exactly when `text` contains newlines. For a +/// single-line (newline-free) text, the pieces concatenate directly. +fn text_pieces(text: &str, budget: usize) -> Vec { + // A budget of 0 is degenerate — treat as 1 to avoid infinite loops. + let budget = budget.max(1); + + // Split on '\n' first. A trailing '\n' yields a final empty element + // which we preserve so joining with '\n' reconstructs the original. + let lines: Vec<&str> = text.split('\n').collect(); + + let mut result: Vec = Vec::new(); + let mut current_piece: Vec<&str> = Vec::new(); + let mut current_bytes: usize = 0; + + for line in &lines { + let line_bytes = line.len(); + // +1 for the '\n' that re-joins this line to the previous one. + let sep_bytes = usize::from(!current_piece.is_empty()); + + if line_bytes.div_ceil(BYTES_PER_TOKEN) > budget { + // This single line alone exceeds the budget → flush any + // accumulated piece, then char-split the line. + if !current_piece.is_empty() { + result.push(current_piece.join("\n")); + current_piece.clear(); + current_bytes = 0; + } + result.extend(char_pieces(line, budget)); + } else if !current_piece.is_empty() + && (current_bytes + sep_bytes + line_bytes).div_ceil(BYTES_PER_TOKEN) > budget + { + // Adding this line would push the current piece over budget + // → flush, then start a new piece with this line. + result.push(current_piece.join("\n")); + current_piece = vec![line]; + current_bytes = line_bytes; + } else { + // Fits: accumulate. + current_bytes += sep_bytes + line_bytes; + current_piece.push(line); + } + } + if !current_piece.is_empty() { + result.push(current_piece.join("\n")); + } + if result.is_empty() { + // Degenerate (empty text) — return one empty piece so the caller + // always gets ≥1 chunk. + result.push(String::new()); + } + result +} + +/// Char-split a single newline-free string `s` into sub-pieces each with +/// `len() <= budget * BYTES_PER_TOKEN`, cutting at UTF-8 char boundaries. +/// Returns ≥1 piece; direct concatenation of all pieces reconstructs `s`. +fn char_pieces(s: &str, budget: usize) -> Vec { + let byte_budget = budget * BYTES_PER_TOKEN; + let mut result: Vec = Vec::new(); + let mut piece_start = 0usize; + let mut piece_bytes = 0usize; + + for (byte_idx, ch) in s.char_indices() { + let ch_bytes = ch.len_utf8(); + if piece_bytes > 0 && piece_bytes + ch_bytes > byte_budget { + // Flush current piece. + result.push(s[piece_start..byte_idx].to_string()); + piece_start = byte_idx; + piece_bytes = 0; + } + piece_bytes += ch_bytes; + } + // Remaining tail. + result.push(s[piece_start..].to_string()); + if result.is_empty() { + result.push(String::new()); + } + result +} + +// ── v1-equivalent helpers (verbatim from md_heading_v1) ───────────────────── + +#[derive(Default)] +struct ChunkAcc<'a> { + blocks: Vec<&'a Block>, + text_tokens: usize, +} + +impl<'a> ChunkAcc<'a> { + fn push_block(&mut self, b: &'a Block) { + self.text_tokens += estimate_block_tokens(b); + self.blocks.push(b); + } + + fn is_empty(&self) -> bool { + self.blocks.is_empty() + } + + fn has_non_heading_content(&self) -> bool { + self.blocks.iter().any(|b| !matches!(b, Block::Heading(_))) + } +} + +fn flush( + acc: &mut ChunkAcc<'_>, + doc: &CanonicalDocument, + chunker_version: &ChunkerVersion, + policy_hash: &str, + out: &mut Vec, +) { + if acc.is_empty() { + return; + } + let blocks = std::mem::take(&mut acc.blocks); + acc.text_tokens = 0; + out.push(build_chunk(doc, &blocks, chunker_version, policy_hash)); +} + +fn collect_overlap_seed<'a>( + acc: &ChunkAcc<'a>, + overlap_tokens: usize, + target_tokens: usize, +) -> Vec<&'a Block> { + let seed_budget = overlap_tokens.min(target_tokens / 2); + if seed_budget == 0 { + return Vec::new(); + } + let mut taken = Vec::new(); + let mut budget = seed_budget; + for b in acc.blocks.iter().rev() { + if matches!(b, Block::Heading(_)) { + continue; + } + let est = estimate_block_tokens(b); + if est > budget && !taken.is_empty() { + break; + } + taken.push(*b); + budget = budget.saturating_sub(est); + if budget == 0 { + break; + } + } + taken.reverse(); + taken +} + +fn build_chunk( + doc: &CanonicalDocument, + blocks: &[&Block], + chunker_version: &ChunkerVersion, + policy_hash: &str, +) -> Chunk { + debug_assert!(!blocks.is_empty(), "build_chunk requires ≥1 block"); + + let block_ids: Vec = blocks.iter().map(|b| common(b).block_id.clone()).collect(); + let source_spans: Vec = blocks + .iter() + .map(|b| common(b).source_span.clone()) + .collect(); + + let heading_path = match blocks[0] { + Block::Heading(h) => { + let mut path = h.common.heading_path.clone(); + path.push(h.text.clone()); + path + } + _ => common(blocks[0]).heading_path.clone(), + }; + + let mut text = String::new(); + let mut is_image_or_audio_only = true; + for (i, b) in blocks.iter().enumerate() { + let part = render_block_text(b); + if !matches!(b, Block::ImageRef(_) | Block::AudioRef(_)) { + is_image_or_audio_only = false; + } + if i > 0 { + text.push_str("\n\n"); + } + text.push_str(&part); + } + + let token_estimate = if is_image_or_audio_only { + 0 + } else { + text.len().div_ceil(BYTES_PER_TOKEN) + }; + + let chunk_id = id_for_chunk(&doc.doc_id, chunker_version, &block_ids, policy_hash); + + Chunk { + chunk_id, + doc_id: DocumentId(doc.doc_id.0.clone()), + block_ids, + tokenized_korean_text: crate::tokenize_korean_morphological(&text), + text, + heading_path, + source_spans, + token_estimate, + chunker_version: chunker_version.clone(), + policy_hash: policy_hash.to_string(), + } +} + +fn render_block_text(b: &Block) -> String { + match b { + Block::Heading(h) => h.text.clone(), + Block::Paragraph(p) | Block::Quote(p) => p.text.clone(), + Block::List(l) => l + .items + .iter() + .map(|it| it.text.as_str()) + .collect::>() + .join("\n"), + Block::Code(c) => c.code.clone(), + Block::Table(t) => { + let mut s = t.headers.join(" | "); + for row in &t.rows { + s.push('\n'); + s.push_str(&row.join(" | ")); + } + s + } + Block::ImageRef(i) => { + let alt = if i.alt.is_empty() { + i.src + .rsplit('/') + .next() + .filter(|s| !s.is_empty()) + .unwrap_or("[image]") + .to_string() + } else { + i.alt.clone() + }; + let ocr = i.ocr.as_ref().map_or("", |o| o.joined.as_str()); + let cap = i.caption.as_ref().map_or("", |c| c.text.as_str()); + [alt.as_str(), ocr, cap] + .iter() + .filter(|s| !s.is_empty()) + .copied() + .collect::>() + .join("\n\n") + } + Block::AudioRef(_) => String::new(), + } +} + +fn estimate_block_tokens(b: &Block) -> usize { + match b { + Block::ImageRef(_) | Block::AudioRef(_) => 0, + _ => render_block_text(b).len().div_ceil(BYTES_PER_TOKEN), + } +} + +fn common(b: &Block) -> &kebab_core::CommonBlock { + match b { + Block::Heading(h) => &h.common, + Block::Paragraph(t) | Block::Quote(t) => &t.common, + Block::List(l) => &l.common, + Block::Code(c) => &c.common, + Block::Table(t) => &t.common, + Block::ImageRef(i) => &i.common, + Block::AudioRef(a) => &a.common, + } +} + +// ── Tests ──────────────────────────────────────────────────────────────────── + +#[cfg(test)] +mod tests { + use super::*; + use crate::MdHeadingV1Chunker; + use kebab_core::{ + AssetId, CodeBlock, CommonBlock, HeadingBlock, ImageRefBlock, Lang, ListBlock, + Metadata, OcrText, Provenance, SourceType, TextBlock, TrustLevel, WorkspacePath, + id_for_block, + }; + use time::OffsetDateTime; + + // ── Document / block helpers ───────────────────────────────────────── + + fn make_doc(blocks: Vec) -> CanonicalDocument { + CanonicalDocument { + doc_id: kebab_core::DocumentId("d".repeat(32)), + source_asset_id: AssetId("a".repeat(32)), + workspace_path: WorkspacePath::new("notes/test.md".into()).unwrap(), + title: "Test".into(), + lang: Lang("en".into()), + blocks, + metadata: Metadata { + aliases: vec![], + tags: vec![], + created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), + updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(), + source_type: SourceType::Note, + trust_level: TrustLevel::Primary, + user_id_alias: None, + user: Default::default(), + repo: None, + git_branch: None, + git_commit: None, + code_lang: None, + source_id: None, + }, + provenance: Provenance { events: vec![] }, + parser_version: kebab_core::ParserVersion("test-parser-0".into()), + schema_version: 1, + doc_version: 1, + last_chunker_version: None, + last_embedding_version: None, + } + } + + fn doc_id() -> kebab_core::DocumentId { + kebab_core::DocumentId("d".repeat(32)) + } + + fn span(start: u32, end: u32) -> SourceSpan { + SourceSpan::Line { start, end } + } + + fn common_for( + kind: &str, + heading_path: &[String], + ordinal: u32, + s: SourceSpan, + ) -> CommonBlock { + CommonBlock { + block_id: id_for_block(&doc_id(), kind, heading_path, ordinal, &s), + heading_path: heading_path.to_vec(), + source_span: s, + } + } + + fn heading(level: u8, text: &str, ordinal: u32, line: u32) -> Block { + Block::Heading(HeadingBlock { + common: common_for("heading", &[], ordinal, span(line, line)), + level, + text: text.into(), + }) + } + + fn paragraph(text: &str, heading_path: &[&str], ordinal: u32, line: u32) -> Block { + let hp: Vec = heading_path.iter().map(|s| (*s).into()).collect(); + Block::Paragraph(TextBlock { + common: common_for("paragraph", &hp, ordinal, span(line, line)), + text: text.into(), + inlines: vec![], + }) + } + + /// Build a `Block::List` whose rendered text is `items.join("\n")`. + /// Each item becomes a `TextBlock`; `render_block_text` joins them + /// with `"\n"` matching the v1/v2 list rendering contract. + fn list_block(items: &[&str], heading_path: &[&str], ordinal: u32, s: SourceSpan) -> Block { + let hp: Vec = heading_path.iter().map(|s| (*s).into()).collect(); + Block::List(ListBlock { + common: common_for("list", &hp, ordinal, s.clone()), + ordered: false, + items: items + .iter() + .enumerate() + .map(|(i, t)| TextBlock { + common: common_for("list_item", &hp, i as u32, s.clone()), + text: (*t).to_string(), + inlines: vec![], + }) + .collect(), + }) + } + + fn code_block(code: &str, heading_path: &[&str], ordinal: u32, s: SourceSpan) -> Block { + let hp: Vec = heading_path.iter().map(|s| (*s).into()).collect(); + Block::Code(CodeBlock { + common: common_for("code", &hp, ordinal, s), + lang: Some("rust".into()), + code: code.into(), + }) + } + + /// Build a `Block::ImageRef` whose OCR `joined` text is `ocr_text`. + /// `render_block_text(ImageRef)` = `[alt, ocr.joined, caption]` joined by + /// `\n\n`, and `build_chunk` forces `token_estimate = 0` for an + /// image-only chunk — so this models a dense screenshot whose OCR output + /// is large yet reports a zero token_estimate. + fn image_block_with_ocr(ocr_text: &str, ordinal: u32, line: u32) -> Block { + Block::ImageRef(ImageRefBlock { + common: common_for("imageref", &[], ordinal, span(line, line)), + asset_id: None, + src: "dense.png".into(), + alt: "dense".into(), + ocr: Some(OcrText { + joined: ocr_text.into(), + regions: vec![], + engine: "paddle-onnx".into(), + engine_version: "ppocrv5".into(), + }), + caption: None, + }) + } + + fn policy_v2(budget: usize) -> (ChunkPolicy, MdHeadingV2Chunker) { + let p = ChunkPolicy { + target_tokens: 500, + overlap_tokens: 80, + respect_markdown_headings: true, + chunker_version: ChunkerVersion(VERSION_LABEL.into()), + }; + let c = MdHeadingV2Chunker { + max_chunk_tokens: budget, + }; + (p, c) + } + + fn policy_v1() -> ChunkPolicy { + ChunkPolicy { + target_tokens: 500, + overlap_tokens: 80, + respect_markdown_headings: true, + chunker_version: ChunkerVersion("md-heading-v1".into()), + } + } + + // ── Tests ───────────────────────────────────────────────────────────── + + #[test] + fn chunker_version_is_md_heading_v2() { + let (_, c) = policy_v2(4000); + assert_eq!(c.chunker_version(), ChunkerVersion(VERSION_LABEL.to_string())); + } + + /// Regression for the image-OCR hole found during the matching dogfood + /// re-test: an `ImageRef` chunk reports `token_estimate = 0` (image-only + /// convention) even when its OCR text is huge. The oversize post-pass must + /// key the split on the ACTUAL embedded `text` length, not the stored + /// `token_estimate`, or a dense screenshot's OCR would reach the embedder + /// whole and fail on a strict backend (e.g. AMD Lemonade). + #[test] + fn oversize_image_ocr_chunk_splits() { + let budget = 200; + let (policy, chunker) = policy_v2(budget); + // ~3000 bytes of OCR text → byte/3 ≈ 1000 tokens ≫ 200 budget, + // but the source ImageRef chunk's token_estimate is 0. + let ocr = "WiredTiger cache eviction stalls under heavy write load. ".repeat(54); + let doc = make_doc(vec![image_block_with_ocr(&ocr, 0, 1)]); + let chunks = chunker.chunk(&doc, &policy).unwrap(); + + // The single image chunk's stored token_estimate is 0 (image-only)... + assert_eq!( + estimate_block_tokens(&Block::ImageRef(match &doc.blocks[0] { + Block::ImageRef(i) => i.clone(), + _ => unreachable!(), + })), + 0, + "ImageRef token_estimate must be 0 (image-only convention)" + ); + // ...yet the oversize OCR text is split into ≥2 pieces, each ≤ budget. + assert!( + chunks.len() >= 2, + "oversize image OCR must split, got {} chunk(s)", + chunks.len() + ); + for ch in &chunks { + assert!( + ch.text.len().div_ceil(BYTES_PER_TOKEN) <= budget, + "every split piece's embedded text must be ≤ budget" + ); + } + // chunk_ids are unique across the image-OCR split pieces. + let mut ids: Vec = chunks.iter().map(|c| c.chunk_id.0.clone()).collect(); + let n = ids.len(); + ids.sort_unstable(); + ids.dedup(); + assert_eq!(ids.len(), n, "split piece chunk_ids must be unique"); + } + + /// The primary regression the pivot was designed to fix: a Jira-style + /// log dump rendered as a `Block::List` (items joined by `\n`) whose + /// rendered text vastly exceeds any embedder budget. + /// + /// Asserts: ≥2 chunks produced; every chunk ≤ budget; concatenating + /// pieces' text with `\n` reconstructs the list's full rendered text; + /// all chunk_ids are unique. + #[test] + fn oversize_list_block_splits() { + // 60 items × 30 bytes ≈ 1800 bytes ≈ 600 tokens, well above budget=50. + let items: Vec = (0..60) + .map(|i| format!("log line {i:03}: some event description here")) + .collect(); + let item_refs: Vec<&str> = items.iter().map(String::as_str).collect(); + let blocks = vec![list_block(&item_refs, &[], 0, span(1, 60))]; + let doc = make_doc(blocks); + let (p, c) = policy_v2(50); + let chunks = c.chunk(&doc, &p).unwrap(); + + assert!( + chunks.len() >= 2, + "oversize list must split, got {}: {chunks:#?}", + chunks.len() + ); + for ch in &chunks { + assert!( + ch.token_estimate <= 50, + "piece exceeds budget: {} > 50", + ch.token_estimate + ); + } + + // Concatenating piece texts with '\n' reconstructs the original + // rendered list text (items.join("\n")). + let expected = item_refs.join("\n"); + let rejoined = chunks + .iter() + .map(|c| c.text.as_str()) + .collect::>() + .join("\n"); + assert_eq!(rejoined, expected, "pieces must reconstruct the list text"); + + // Chunk_ids unique. + let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect(); + let n = ids.len(); + ids.sort_unstable(); + ids.dedup(); + assert_eq!(ids.len(), n, "chunk_ids must be unique"); + } + + /// Single-line paragraph that vastly exceeds budget when rendered — + /// triggers the UTF-8 char-split fallback. Includes multibyte Korean + /// to verify no mid-codepoint split. + #[test] + fn oversize_paragraph_single_line_char_splits() { + // "가나다라마바사아자차카타파하" is 14 Korean syllables × 3 bytes each. + // Repeat to well exceed a small budget. + let korean_base = "가나다라마바사아자차카타파하"; + let text = korean_base.repeat(30); // 420 chars × 3 bytes = 1260 bytes ≈ 420 tokens + let blocks = vec![paragraph(&text, &[], 0, 1)]; + let doc = make_doc(blocks); + let (p, c) = policy_v2(20); // budget = 20 tokens ≈ 60 bytes + let chunks = c.chunk(&doc, &p).unwrap(); + + assert!( + chunks.len() >= 2, + "single-line oversize paragraph must char-split, got {}: {chunks:#?}", + chunks.len() + ); + for ch in &chunks { + assert!( + ch.token_estimate <= 20, + "piece exceeds budget: {} > 20", + ch.token_estimate + ); + // Verify no replacement chars from a mid-codepoint cut. + assert!( + !ch.text.contains('\u{FFFD}'), + "replacement char detected — mid-codepoint split!" + ); + // Every piece is valid UTF-8 (Rust strings always are if + // they are constructed correctly, but this confirms no panic). + assert!(std::str::from_utf8(ch.text.as_bytes()).is_ok()); + } + + // Concatenation reconstructs the original (no newlines introduced). + let rejoined: String = chunks.iter().map(|c| c.text.as_str()).collect(); + assert_eq!(rejoined, text, "char pieces must reconstruct original"); + } + + /// A code block that exceeds budget is now handled by the generic + /// post-pass (not a code-specific branch). Same guarantees apply. + #[test] + fn oversize_code_block_still_splits() { + // 30 lines × ~30 bytes ≈ 900 bytes ≈ 300 tokens. budget = 50. + let body: String = (0..30) + .map(|i| format!(" let x{i:02} = compute({i});")) + .collect::>() + .join("\n"); + let blocks = vec![code_block(&body, &[], 0, span(5, 34))]; + let doc = make_doc(blocks); + let (p, c) = policy_v2(50); + let chunks = c.chunk(&doc, &p).unwrap(); + + assert!( + chunks.len() >= 2, + "oversize code must split via post-pass, got {}: {chunks:#?}", + chunks.len() + ); + for ch in &chunks { + assert!( + ch.token_estimate <= 50, + "piece exceeds budget: {} > 50", + ch.token_estimate + ); + } + + // Rejoining with '\n' reconstructs the original code. + let rejoined = chunks + .iter() + .map(|c| c.text.as_str()) + .collect::>() + .join("\n"); + assert_eq!(rejoined, body, "pieces must reconstruct the code text"); + } + + /// For a doc with headings + paragraphs + a small (≤budget) code + /// block, v2 output (chunk count, texts, heading_paths, source_spans, + /// token_estimate, block_ids) equals v1 output. This is the parity + /// contract: v2 ≡ v1 for all under-budget content. + /// + /// Note: chunk_ids DIFFER by design because chunker_version is in the + /// id recipe (`"md-heading-v1"` vs `"md-heading-v2"`). We compare the + /// non-id fields that should be stable. + #[test] + fn non_oversize_identical_to_v1() { + let blocks = vec![ + heading(2, "First", 0, 1), + paragraph("body of the first section here", &["First"], 0, 2), + paragraph("more first-section prose", &["First"], 1, 3), + heading(2, "Second", 1, 4), + paragraph("second section body text", &["Second"], 0, 5), + // Small code block — below budget so post-pass leaves it alone. + code_block("fn small() {}", &["Second"], 0, span(6, 6)), + ]; + let doc = make_doc(blocks); + + let v1 = MdHeadingV1Chunker + .chunk(&doc, &policy_v1()) + .unwrap(); + let (p, c) = policy_v2(4000); + let v2 = c.chunk(&doc, &p).unwrap(); + + assert_eq!(v1.len(), v2.len(), "chunk count must match v1"); + for (a, b) in v1.iter().zip(v2.iter()) { + assert_eq!(a.text, b.text, "text must match v1"); + assert_eq!(a.heading_path, b.heading_path, "heading_path must match v1"); + assert_eq!(a.source_spans, b.source_spans, "source_spans must match v1"); + assert_eq!(a.token_estimate, b.token_estimate, "token_estimate must match v1"); + assert_eq!(a.block_ids, b.block_ids, "block_ids must match v1"); + } + } + + /// Split pieces have unique chunk_ids; running chunk() twice produces + /// a byte-identical id sequence (1000-iteration determinism check). + #[test] + fn split_pieces_unique_deterministic_ids() { + let items: Vec = (0..40) + .map(|i| format!("item {i:02}: line number with some padding text here")) + .collect(); + let item_refs: Vec<&str> = items.iter().map(String::as_str).collect(); + let blocks = vec![list_block(&item_refs, &[], 0, span(1, 40))]; + let doc = make_doc(blocks); + let (p, c) = policy_v2(40); + + let baseline: Vec = c + .chunk(&doc, &p) + .unwrap() + .into_iter() + .map(|ch| ch.chunk_id.0) + .collect(); + assert!(baseline.len() >= 2, "must split into ≥2 pieces"); + + // Unique. + let mut sorted = baseline.clone(); + let n = sorted.len(); + sorted.sort_unstable(); + sorted.dedup(); + assert_eq!(sorted.len(), n, "chunk_ids unique across split pieces"); + + // Deterministic. + for _ in 0..1000 { + let again: Vec = c + .chunk(&doc, &p) + .unwrap() + .into_iter() + .map(|ch| ch.chunk_id.0) + .collect(); + assert_eq!(again, baseline, "chunk_id sequence must be deterministic"); + } + } + + /// Two MdHeadingV2Chunker instances with different `max_chunk_tokens` + /// produce different `policy_hash` values for the same policy, causing + /// different chunk_ids — so a budget change triggers a re-chunk cascade. + #[test] + fn budget_in_policy_hash() { + let p = ChunkPolicy { + target_tokens: 500, + overlap_tokens: 80, + respect_markdown_headings: true, + chunker_version: ChunkerVersion(VERSION_LABEL.into()), + }; + let a = MdHeadingV2Chunker { + max_chunk_tokens: 100, + }; + let b = MdHeadingV2Chunker { + max_chunk_tokens: 200, + }; + assert_ne!( + a.policy_hash(&p), + b.policy_hash(&p), + "different budgets must yield different policy_hash" + ); + + // Propagates to chunk_ids on an oversize doc. + let items: Vec = (0..30) + .map(|i| format!(" let x{i:02} = compute({i});")) + .collect(); + let item_refs: Vec<&str> = items.iter().map(String::as_str).collect(); + let blocks = vec![list_block(&item_refs, &[], 0, span(1, 30))]; + let doc = make_doc(blocks); + let ids_a: Vec = a + .chunk(&doc, &p) + .unwrap() + .into_iter() + .map(|c| c.chunk_id.0) + .collect(); + let ids_b: Vec = b + .chunk(&doc, &p) + .unwrap() + .into_iter() + .map(|c| c.chunk_id.0) + .collect(); + assert_ne!(ids_a, ids_b, "different budgets must produce different chunk_ids"); + } + + /// A below-budget chunk stores the BARE policy_hash (no `#seg` suffix). + #[test] + fn under_budget_stores_bare_policy_hash() { + let blocks = vec![code_block("fn x() {}", &[], 0, span(1, 1))]; + let doc = make_doc(blocks); + let (p, c) = policy_v2(4000); + let chunks = c.chunk(&doc, &p).unwrap(); + assert_eq!(chunks.len(), 1); + assert_eq!(chunks[0].policy_hash.len(), POLICY_HASH_HEX_LEN); + assert!( + !chunks[0].policy_hash.contains('#'), + "stored hash has no #seg suffix: {}", + chunks[0].policy_hash + ); + } + + // ── Unit tests for the split helpers ───────────────────────────────── + + /// `text_pieces` on a multi-line string reconstructs with join("\n"). + #[test] + fn text_pieces_multiline_roundtrip() { + let lines: Vec = (0..20) + .map(|i| format!("line {i:02}: some content here")) + .collect(); + let text = lines.join("\n"); + let budget = 10usize; // small to force splits + let pieces = text_pieces(&text, budget); + assert!(pieces.len() >= 2, "must split multi-line text"); + for p in &pieces { + assert!( + p.len().div_ceil(BYTES_PER_TOKEN) <= budget, + "piece exceeds budget: {} bytes / 3 = {} > {budget}", + p.len(), + p.len().div_ceil(BYTES_PER_TOKEN) + ); + } + assert_eq!(pieces.join("\n"), text, "pieces must reconstruct original"); + } + + /// `char_pieces` on a newline-free string reconstructs by concatenation. + #[test] + fn char_pieces_utf8_roundtrip() { + // Mix of ASCII and 3-byte Korean. + let s = "hello가나다world마바사".repeat(10); + let budget = 5usize; + let pieces = char_pieces(&s, budget); + assert!(pieces.len() >= 2); + for p in &pieces { + assert!( + p.len() <= budget * BYTES_PER_TOKEN, + "char piece too long: {} > {}", + p.len(), + budget * BYTES_PER_TOKEN + ); + assert!(std::str::from_utf8(p.as_bytes()).is_ok(), "not valid UTF-8"); + } + assert_eq!(pieces.concat(), s, "char pieces must reconstruct original"); + } +} diff --git a/crates/kebab-config/src/lib.rs b/crates/kebab-config/src/lib.rs index 723488e..1d72d4f 100644 --- a/crates/kebab-config/src/lib.rs +++ b/crates/kebab-config/src/lib.rs @@ -175,6 +175,25 @@ pub struct ChunkingCfg { pub overlap_tokens: usize, pub respect_markdown_headings: bool, pub chunker_version: String, + /// Max byte/3 token estimate per emitted chunk (md-heading-v2). + /// After the v1-equivalent chunking pass, any chunk whose estimate + /// exceeds this value is split at line (then UTF-8 char) boundaries + /// into sub-pieces each ≤ budget. Covers all block kinds: list, code, + /// paragraph, table. The default (4000) is large enough to keep normal + /// content atomic while splitting pathological log / stacktrace / Jira + /// list dumps that would otherwise overflow an embedder context window. + /// `#[serde(default)]` so pre-v2 config files that predate the key + /// still load (migration injects it additively). + #[serde(default = "default_max_chunk_tokens")] + pub max_chunk_tokens: usize, +} + +/// Default md-heading-v2 chunk split budget. 4000 byte/3 tokens +/// (~12 KB) keeps ordinary source files and prose atomic while +/// splitting the pathological 20k–76k token Jira log blocks that fail +/// to embed on strict servers. +fn default_max_chunk_tokens() -> usize { + 4000 } impl ChunkingCfg { @@ -183,7 +202,12 @@ impl ChunkingCfg { target_tokens: 500, overlap_tokens: 80, respect_markdown_headings: true, - chunker_version: "md-heading-v1".to_string(), + // md-heading-v2 is the hardcoded markdown default (it splits + // oversize chunks of any block kind; v1 never did). Stamping it + // here means the lib.rs skip-check re-chunks md docs on next + // ingest via the version cascade (design §9). + chunker_version: "md-heading-v2".to_string(), + max_chunk_tokens: default_max_chunk_tokens(), } } } @@ -1251,6 +1275,11 @@ impl Config { self.ingest.chunking.respect_markdown_headings = parse_bool(v); } "KEBAB_CHUNKING_CHUNKER_VERSION" => self.ingest.chunking.chunker_version = v.clone(), + "KEBAB_CHUNKING_MAX_CHUNK_TOKENS" => { + if let Ok(n) = v.parse::() { + self.ingest.chunking.max_chunk_tokens = n; + } + } // models.embedding "KEBAB_MODELS_EMBEDDING_PROVIDER" => self.models.embedding.provider = v.clone(), diff --git a/crates/kebab-config/src/migrate.rs b/crates/kebab-config/src/migrate.rs index 7c1fbdd..a8d6a92 100644 --- a/crates/kebab-config/src/migrate.rs +++ b/crates/kebab-config/src/migrate.rs @@ -108,6 +108,9 @@ fn key_comment(path: &str) -> Option<&'static str> { "ingest.max_parallel_embeddings" => "동시 임베딩 수.", "ingest.chunking.target_tokens" => "청크 목표 토큰(전 형식 공통).", "ingest.chunking.respect_markdown_headings" => "markdown heading 경계 존중.", + "ingest.chunking.max_chunk_tokens" => { + "md-heading-v2 청크 최대 토큰(byte/3). 초과 시 줄/문자 경계로 분할(list·code·단락 공통)." + } "ingest.image.ocr.enabled" => "이미지 OCR(기본 off, asset 당 비용).", "ingest.image.ocr.engine" => "ollama-vision | paddle-onnx.", "ingest.image.ocr.model" => "ollama-vision 전용. paddle-onnx 는 번들 모델 사용(이 값 무시).", diff --git a/docs/SMOKE.md b/docs/SMOKE.md index 7a4c9bf..9dd5ffb 100644 --- a/docs/SMOKE.md +++ b/docs/SMOKE.md @@ -106,7 +106,8 @@ watch_filesystem = false target_tokens = 500 overlap_tokens = 80 respect_markdown_headings = true -chunker_version = "md-heading-v1" +chunker_version = "md-heading-v2" +max_chunk_tokens = 4000 # v0.30.0 — 이 byte/3 토큰 초과 청크는 줄(→UTF-8 char) 경계로 분할(거대 list/code/log 덤프 대비) [models.embedding] provider = "fastembed" # "fastembed"(기본, onnxruntime) / "candle"(순수 Rust, NUMA-안전) diff --git a/docs/components/normalize-chunk/README.md b/docs/components/normalize-chunk/README.md index 08e5b2f..16cc04a 100644 --- a/docs/components/normalize-chunk/README.md +++ b/docs/components/normalize-chunk/README.md @@ -7,7 +7,7 @@ | Crate | 역할 | |-------|------| | `kebab-normalize` | `ParsedBlock` (markdown only) → `CanonicalDocument` lift. NFC + heading-path ordinal + provenance 합성 + title fallback chain (p9-fb-07). | -| `kebab-chunk` | `CanonicalDocument` → `Vec`. v1 두 변종: `md-heading-v1` (markdown + image), `pdf-page-v1` (PDF). | +| `kebab-chunk` | `CanonicalDocument` → `Vec`. markdown 기본 `md-heading-v2` (v1 + 예산 초과 청크 일반 분할; v0.30.0), `pdf-page-v1` (PDF). `md-heading-v1` 은 historical 변종으로 잔존. | ## 구조 @@ -30,6 +30,11 @@ classDiagram BYTES_PER_TOKEN = 3 POLICY_HASH_HEX_LEN = 16 } + class MdHeadingV2Chunker { + VERSION = "md-heading-v2" + max_chunk_tokens = 4000 + split_oversize_chunk(line→char) + } class PdfPageV1Chunker { VERSION = "pdf-page-v1" BYTES_PER_TOKEN = 3 @@ -42,11 +47,20 @@ classDiagram chunker_version } Chunker <|.. MdHeadingV1Chunker + Chunker <|.. MdHeadingV2Chunker Chunker <|.. PdfPageV1Chunker - MdHeadingV1Chunker ..> ChunkPolicy + MdHeadingV2Chunker ..> ChunkPolicy PdfPageV1Chunker ..> ChunkPolicy ``` +`md-heading-v2` (기본, v0.30.0) 는 v1 과 모든 출력이 동일하되, 마지막에 +`token_estimate > max_chunk_tokens` 인 청크만 줄(`\n`) 경계로 — 단일 거대 줄은 +UTF-8 char 경계로 — 잘라 각 조각이 예산 이하가 되도록 한다. 분할 조각은 동일 +`block_ids` 를 공유하므로 chunk_id 충돌을 막기 위해 id-input 해시에 `#seg{i}` +접미사를 붙이고(저장 `policy_hash` 는 bare), `max_chunk_tokens` 는 v2 의 +`policy_hash` 에 fold 된다(공유 `ChunkPolicy` 미변경). 분할 조각의 +`source_spans` 는 원 블록 범위를 그대로 보존(블록 단위 citation). + ## Data flow ```mermaid diff --git a/docs/release-notes/v0.30.0-draft.md b/docs/release-notes/v0.30.0-draft.md new file mode 100644 index 0000000..5a6536f --- /dev/null +++ b/docs/release-notes/v0.30.0-draft.md @@ -0,0 +1,97 @@ +--- +title: kebab v0.30.0 release notes (draft) +created: 2026-06-24 +status: draft +release_trigger: + - 신규 config `[ingest.chunking] max_chunk_tokens` — 인터페이스 추가 (pre-1.0 minor) + - markdown 청커 동작 변경 `md-heading-v1` → `md-heading-v2` (거대 청크 분할 → 검색 hit 분할) — 도그푸딩 트리거 +--- + +# kebab v0.30.0 — md-heading-v2: 거대 청크가 임베딩을 깨지 않도록 + +v0.29.0(provenance 출처 필터) 후속 minor release. markdown 청커가 **하나의 거대 +블록이 임베더 컨텍스트를 초과해 임베딩을 통째로 실패시키던** 문제를 해소한다. +긴 로그·스택트레이스·표가 통으로 들어간 문서(예: jira 이슈)를 색인할 때 그 +문서가 검색에서 통째로 사라지던 회귀를 막는다. **일반 사용자는 업그레이드 후 +다음 `kebab ingest` 한 번이면 끝** — markdown 자산만 1회 자동 재청크된다. + +--- + +## 변경 사실 + +**1) `md-heading-v2` 청커가 기본값이 됐다.** v1은 코드/테이블 블록을 (크기와 +무관하게) 절대 분할하지 않았고, list·paragraph도 블록 하나가 곧 청크 하나였다. +그래서 거대한 단일 블록 — 예컨대 로그 덤프가 1000줄짜리 불릿 리스트로 변환된 것 — +이 임베더의 입력 한도를 넘으면 그 문서 전체의 벡터 색인이 실패했다. v2는 v1과 +**모든 출력이 동일**하되, 마지막에 **예산을 넘는 청크만** 줄(`\n`) 경계로, +한 줄이 홀로 예산을 넘으면 UTF-8 문자 경계로 잘라 각 조각이 예산 이하가 되게 한다. + +**2) 신규 config `[ingest.chunking] max_chunk_tokens`** (byte/3 토큰, default +**4000**). 이 값을 넘는 청크가 분할 대상이 된다. 4000은 8192-context 임베더(예: +snowflake-arctic-embed2)에 약 2배 안전마진을 두면서 일반적인 코드·문단은 그대로 +한 청크로 유지하는 값이다. `kebab config migrate`로 기존 config에 additive 주입. + +```toml +[ingest.chunking] +chunker_version = "md-heading-v2" +max_chunk_tokens = 4000 +``` + +**3) 분할 조각의 chunk_id·citation.** 한 블록에서 나온 분할 조각들은 같은 +`block_ids`를 공유하므로 chunk_id에 `#seg{i}` 접미사를 붙여 충돌을 막는다. 분할 +조각의 citation은 **블록 단위**(원 블록 범위) — sub-line 정밀도는 아니다. +미분할 청크(대다수)는 v0.29.0과 출력·citation이 완전히 동일하다. + +## Trade-off + +- **거대 블록이 여러 검색 hit로 나뉜다.** 이전엔 그 문서가 (임베딩 실패로) + 벡터 검색에서 아예 안 나왔거나, ollama처럼 조용히 잘린 채 한 hit였다. 이제는 + 여러 조각으로 정상 색인되어 각각 hit로 잡힐 수 있다 — 같은 문서가 상위에 여러 + 번 보일 수 있다. +- **분할 조각 citation은 블록 단위다.** fenced 코드의 줄 span이 fence를 포함하는 + 비대칭 때문에 조각별 정확한 줄 범위를 (span, code)만으로 복원할 수 없어, + "절대 틀리지 않되 블록 단위"를 택했다. 정밀 sub-line citation이 필요하면 향후 + 파서가 content 줄 범위를 직접 제공하는 개선이 필요하다. +- **byte/3 휴리스틱.** 토큰 수는 실제 토크나이저가 아니라 바이트/3 근사다. CJK는 + 과대추정(더 안전), 영문 코드는 실토큰과 비슷. strict 임베더에 대해 보수적이다. + +## Mitigation + +- **결과·CLI·wire 포맷 불변.** `--json` 스키마, exit code, citation 모양 모두 + 동일. 내부적으로 거대 블록이 분할될 뿐이다. +- **공유 `ChunkPolicy` 미변경.** `max_chunk_tokens`는 v2 청커의 policy_hash에만 + fold되어 코드·PDF 청커의 cascade를 건드리지 않는다 — 코드/PDF 자산은 재청크되지 + 않는다. +- **strict 임베더 호환.** v1 시절 ollama가 조용히 truncate하던 것을 청커가 애초에 + 예산 이하로 만들므로, oversize 입력을 truncate가 아닌 거부(`500 too large`)로 + 처리하는 백엔드(예: AMD Lemonade)에서도 색인이 깨지지 않는다. +- **이미지 OCR 텍스트도 보호.** 분할 판정은 청크의 실제 임베드 text 크기로 한다 — + 이미지 청크는 내부적으로 token_estimate가 0이지만 그 OCR/캡션 text는 클 수 있어 + (빽빽한 스크린샷), 실제 text 길이로 판정해야 oversize 이미지 OCR도 분할된다. + +### Known limitation + +PDF는 별도 청커(`pdf-page-v1.1`)를 쓰므로 이 oversize-split이 적용되지 않는다 — +초고밀도 scanned page 한 장의 OCR이 한 청크로 예산을 넘으면 그대로 임베드된다 +(일반 PDF는 무관). 후속 작업으로 PDF 청커에도 동일 분할을 넣을 수 있다. + +## 업그레이드 절차 + +1. 새 바이너리로 교체. +2. (선택) `kebab config migrate` — config 파일에 `max_chunk_tokens = 4000`을 + additive 주입(주석·값 보존). 안 해도 기본값으로 동작한다. +3. `kebab ingest` — `chunker_version`이 `md-heading-v1` → `md-heading-v2`로 + 바뀌었으므로 **markdown 자산이 1회 자동 재청크**된다(`--force-reingest` + 불필요). 코드/PDF 자산은 영향 없음. embedding은 파생물 캐시(V012) 히트로 + 대부분 재계산을 피하지만, 분할된 거대 블록의 새 조각은 새로 임베드된다. +4. 대용량 KB라면 첫 ingest가 markdown 재청크로 평소보다 길어질 수 있다(1회성). + +### 도그푸딩 evidence + +실험 KB(MongoDB 문서 220 + jira 400, arctic-embed-l-v2 @ Lemonade GPU). v2 전: +620 문서 중 2개(`SERVER-22906`=76189토큰 list, `SERVER-23097`=20643토큰 list)가 +임베드 실패. v2 후: **620/620, errors=0**, 전 코퍼스 7114 청크가 모두 ≤ 4000 +(초과 0). `SERVER-22906`은 14 청크로, `SERVER-23097`은 5 청크로 분할. "WiredTiger +excessive memory cache size" 질의에서 그동안 색인조차 안 되던 `SERVER-22906`이 +**1위 결과(0.977)**로 — "임베드 불가"에서 "최상위 검색 결과"로 전환됐다. 출처 +필터(`--trust-min primary` 등)는 정상 유지. diff --git a/docs/superpowers/plans/2026-06-24-md-heading-v2-oversize-split.md b/docs/superpowers/plans/2026-06-24-md-heading-v2-oversize-split.md new file mode 100644 index 0000000..b76a8ae --- /dev/null +++ b/docs/superpowers/plans/2026-06-24-md-heading-v2-oversize-split.md @@ -0,0 +1,140 @@ +--- +title: "md-heading-v2 — 예산 초과 청크 일반 분할 (oversize-chunk split)" +created: 2026-06-24 +status: implemented +extends: tasks/p1/p1-5-chunk.md (md-heading-v1, frozen) +contract_sections: [§3.5 Chunk, §4.2 chunk_id recipe, §7.2 Chunker, §9 versioning] +design_doc_change: none # 설계 §9 가 라벨 bump 를 변경 메커니즘으로 이미 명시 +--- + +# md-heading-v2 — 예산 초과 청크 일반 분할 + +## 문제 + +`md-heading-v1`(p1-5)은 규칙 2로 "코드/테이블 블록은 `target_tokens`를 넘어도 +절대 분할하지 않는다"를 둔다. 실제로는 **모든** atomic/single 블록(코드뿐 아니라 +`list`·`table`·거대 `paragraph`)이 한 청크가 될 수 있고, 그 청크가 임베더 +컨텍스트를 초과하면 임베딩이 실패한다. + +도그푸딩에서 jira 이슈 일부(예: `SERVER-22906`)가 임베드에 실패했다. 긴 MongoDB +로그/스택트레이스가 md 변환 시 **하나의 거대 `list` 블록**(76189 byte/3 토큰, +소스 60–1303 줄)으로 렌더된 것이 원인이었다. + +- 기존 ollama(`/api/embed`)는 이런 입력을 서버측 8192로 **조용히 truncate** → + 0 errors 였지만 사실상 잘려 색인됨. +- AMD Lemonade 같은 strict 백엔드는 + `500 "input (N tokens) is too large ... increase the physical batch size"` 로 + **거부**한다. + +임베더-무관하게 견고하려면 청커가 애초에 예산 초과 청크를 만들지 않아야 한다. +임베더 측 전역 truncate(silently 잘림)는 reject. + +## 결정 — 왜 v2(새 라벨)인가, 왜 일반(generic) 분할인가 + +- **새 변종 `md-heading-v2`, frozen doc 미변경.** 설계 doc은 "코드 블록 미분할"을 + 계약 불변식으로 못박지 않는다(예제 출력 텍스트일 뿐). 규칙은 frozen task spec + p1-5 소유. 설계 §9가 `chunk boundary/policy 변화 → 라벨(md-heading-v2)`을 변경 + 메커니즘으로 **명시**한다. → 설계 §1/§3.5/§7.2/§9 byte-identical, p1-5 frozen + 유지. 선례: pdf-page-v1 → pdf-page-v1.1(HOTFIXES, frozen doc 미변경). +- **코드 한정이 아닌 일반 분할.** 실패의 실제 원인은 코드가 아니라 list 블록. + 블록 종류별 특수 로직 대신 "예산 초과 청크"를 일반적으로 분할하면 list·code· + table·paragraph를 균일하게 덮고, fenced-code의 span 비대칭 문제(아래)도 피한다. + +## 설계 + +`md-heading-v2`의 `chunk()`는 v1과 **출력이 동일**하다(같은 블록 처리, 같은 +soft-split). 마지막에 후처리 패스를 둔다: + +``` +let chunks = ; +chunks.into_iter().flat_map(|c| { + // 판정 기준 = 실제 임베드 text 크기 (token_estimate 아님) + let embed_tokens = c.text.len().div_ceil(BYTES_PER_TOKEN); + if embed_tokens <= max_chunk_tokens { vec![c] } // v1 parity + else { split_oversize_chunk(c, max_chunk_tokens) } // 분할 +}) +``` + +**왜 `token_estimate` 가 아니라 `text.len()` 인가.** text 청크는 둘이 같다 +(`token_estimate == text.len()/BYTES_PER_TOKEN`) → 비이미지 출력 불변. 하지만 +`ImageRef`/`AudioRef` 청크는 `build_chunk` 가 image-only 규약으로 +`token_estimate=0` 을 박는데, 그 `text`(alt+OCR+caption)는 임의로 클 수 있다 +(빽빽한 스크린샷이 수십 KB 로 OCR 되는 경우). 임베더가 실제로 받는 건 `text` 이므로 +거기에 맞춰 판정해야 oversize 이미지 OCR 도 분할된다. 이 구멍은 markdown-only +검증에선 안 보였고, **사용자 실 config(이미지 OCR ON) 재현 도그푸딩에서 발견**됐다 +(회귀 테스트 `oversize_image_ocr_chunk_splits`). + +`split_oversize_chunk`: + +1. `chunk.text`를 줄(`\n`) 경계로 그리디 누적 — 다음 줄을 더하면 예산(byte/3)을 + 넘을 때 조각을 닫는다. +2. **단일 줄이 홀로 예산을 넘으면**(거대 paragraph는 개행 없는 한 줄로 렌더) + 그 줄을 **UTF-8 char 경계**(`char_indices`)로 ≤ `budget * 3` 바이트씩 분할 — + codepoint 중간을 자르지 않는다. → 예산 상한이 모든 입력에 대해 **하드 보장**. +3. 각 조각 i → `Chunk`: 동일 `doc_id`/`block_ids`/`heading_path`/`chunker_version`, + `text`=조각, `token_estimate`=조각 byte/3, 저장 `policy_hash`=**bare** base 해시, + `chunk_id = id_for_chunk(doc_id, version, block_ids, "{base}#seg{i}")`. + +### chunk_id 충돌 회피 + +분할 조각은 동일 `block_ids`를 공유하므로 `id_for_chunk`의 기본 recipe로는 +충돌한다. id-input 해시에만 `#seg{i}`(i = 0-based 조각 인덱스, 단조증가) 접미사를 +붙여 disambiguate하고, 저장 `Chunk.policy_hash`에는 bare base를 남긴다 — +pdf-page-v1의 `#L` recipe(HOTFIXES 2026-05-02 P7-2)와 동형. + +### `max_chunk_tokens`를 policy_hash에 fold + +신규 config `[ingest.chunking] max_chunk_tokens`(byte/3, default 4000)를 공유 +`ChunkPolicy`에 넣으면 **모든** 청커(코드·PDF 포함)의 policy_hash가 바뀌어 +cascade가 번진다. 대신 v2의 `policy_hash()`만 override해 canonical `ChunkPolicy` +바이트 뒤에 `max_chunk_tokens.to_le_bytes()`를 이어 blake3에 먹인다 → 값을 바꾸면 +markdown만 재청크, v1·공유 ChunkPolicy 무영향. + +### citation 정밀도 (known limitation) + +분할 조각은 원 청크의 `source_spans`(원 블록 전체 범위)를 **그대로** 보존한다 → +거대 블록을 쪼갠 조각의 citation은 **블록 단위**(sub-line 정밀 아님). fenced +코드의 `SourceSpan::Line`은 fence 줄을 포함하는데 `code`는 content만 담아 +(`kebab-parse-md/src/blocks.rs` `span_for(full range)`), 조각별 줄 범위를 정확히 +좁히는 건 (span, code)만으로 일반적으로 불가능 → "절대 틀리지 않되 블록 단위"를 +택했다. 미분할 청크는 v1과 byte-identical이라 영향 없음. + +## cascade / 업그레이드 + +- 마크다운 청커 dispatch는 하드코딩(`kebab-app`) — config `chunker_version` + 문자열은 impl 선택에 쓰이지 않는다. v2는 type swap으로 기본값 승격. +- `chunker_version` 라벨 `md-heading-v1` → `md-heading-v2` → 다음 plain + `kebab ingest`에서 markdown 자산 1회 자동 재청크(`--force-reingest` 불필요, + skip 비교 mismatch). 코드/PDF는 chunker_version unchanged → 무영향. 마이그레이션 + 불필요(chunks 스키마 V001부터 동일, chunk_id 키 재계산). +- **wrinkle**: 기존 config가 `chunker_version = "md-heading-v1"`로 핀돼 있어도 + 실제로는 v2가 돈다(문자열은 정보성). `config migrate`/새 default config는 + `max_chunk_tokens`를 additive 주입. + +## 검증 (도그푸딩) + +실험 KB `/home/user/large_data/out/kebab-ab/xdg_sources`, arctic-embed-l-v2 @ +Lemonade `.243`. v2 전: 620 중 2 doc(`SERVER-22906`/`SERVER-23097`) 임베드 실패. +v2 후: **620/620, errors=0**, 7114 청크 전부 ≤ 4000(초과 0). `SERVER-22906` → +14 청크(max 3897), `SERVER-23097` → 5 청크(max 3850). 검색 payoff: "WiredTiger +excessive memory cache size" 질의에 `SERVER-22906`가 **1위(0.977)** — "임베드 +불가"에서 "최상위 검색 결과"로. `--trust-min primary` 등 출처 필터 정상. + +**일치 재테스트 (사용자 실 config 재현)**. 사용자 실 config 가 이미지 OCR + PDF +OCR 를 paddle-onnx 로 ON 함을 반영해, 미디어(생성 이미지 2 + repo scanned PDF 2)를 +같은 paddle-onnx + arctic@Lemonade 로 재인덱싱(624 자산 errors=0). PDF OCR → +청크(`pdf-page-v1.1`) → arctic 임베드 정상. **여기서 image-OCR 구멍 발견·수정** +(위 "왜 token_estimate 가 아닌가" 참조). **known limitation**: PDF 는 별도 청커 +`pdf-page-v1.1` 이라 v2 의 oversize-split 미적용 — 초고밀도 scanned page 가 한 +청크로 budget 초과 시 잔존. 후속 후보: pdf-page 청커에도 동일 oversize-split. + +단위 테스트(`crates/kebab-chunk/src/md_heading_v2.rs`): `oversize_list_block_splits`, +`oversize_paragraph_single_line_char_splits`(다국어 UTF-8 경계), `oversize_code_block_still_splits`, +`oversize_image_ocr_chunk_splits`(token_estimate=0 이미지 OCR), `non_oversize_identical_to_v1`(v1 parity), +`split_pieces_unique_deterministic_ids`(1000-iter 결정성), `budget_in_policy_hash`. +kebab-chunk / kebab-config / kebab-app 전체 pass, clippy `-D warnings` clean. + +## 버전 + +`Cargo.toml` 0.29.0 → **0.30.0**. 신규 config 키 + 청커 동작 변경(검색 hit 분할) += pre-1.0 minor + 도그푸딩 트리거(CLAUDE.md §Versioning/§Dogfood). diff --git a/tasks/HOTFIXES.md b/tasks/HOTFIXES.md index 33901dc..ce81758 100644 --- a/tasks/HOTFIXES.md +++ b/tasks/HOTFIXES.md @@ -14,6 +14,79 @@ historical contract that was implemented; this file accumulates the deltas so phase 5+ readers can find the live behavior without diffing git history. +## 2026-06-24 — md-heading-v2: 예산 초과 청크 일반 분할 (oversize-chunk split) (v0.30.0) + +**무엇을 바꿨나.** markdown 청커에 새 변종 `md-heading-v2` 를 추가하고 +기본값으로 승격했다. v1 의 규칙 2("코드/테이블 블록은 `target_tokens` 를 +넘어도 절대 분할하지 않는다")는 **모든** 블록 종류로 일반화된 한계였다 — 하나의 +거대 블록(코드뿐 아니라 list·table·paragraph 도)이 통째로 한 청크가 되어 +임베더 컨텍스트를 초과할 수 있었다. v2 는 v1 과 **모든 출력이 동일**하되, +마지막에 청크의 **실제 임베드 text 크기**(`text.len()/3`)가 `max_chunk_tokens` +를 넘는 청크만 줄(`\n`) 경계로, 단일 거대 줄은 UTF-8 char 경계로 잘라 **각 조각이 +예산 이하**가 되도록 분할한다. (판정 기준이 저장 `token_estimate` 가 **아니라** +실제 text 길이인 이유: ImageRef/AudioRef 청크는 image-only 규약으로 +`token_estimate=0` 인데 그 text(alt+OCR+caption)는 거대할 수 있다 — 빽빽한 +스크린샷 OCR 이 대표 사례. 도그푸딩 일치 재테스트에서 발견·수정.) +신규 config `[ingest.chunking] max_chunk_tokens` (byte/3 토큰, default **4000**). +분할 조각의 chunk_id 는 동일 `block_ids` 를 공유하므로 id-input 해시에 +`#seg{i}` 접미사를 붙여 충돌을 막는다(저장 `policy_hash` 는 bare — pdf-page-v1 +의 `#L` 레시피와 동형). `max_chunk_tokens` 는 v2 의 `policy_hash` 에 +folding 되어(공유 `ChunkPolicy` 는 미변경 → 코드/PDF 청커 cascade 무영향) +값을 바꾸면 markdown 만 재청크된다. + +**왜 — strict 임베더는 oversize 입력을 truncate 가 아니라 거부한다.** 도그푸딩 +도중 jira 이슈 일부가 임베드에 실패했다. 원인은 긴 MongoDB 로그/스택트레이스가 +md 변환 시 **하나의 거대 `list` 블록**(예: SERVER-22906 = 76189 토큰 / 소스 +60–1303 줄)으로 렌더된 것. 기존 ollama(`/api/embed`)는 이런 입력을 조용히 +서버측 8192 로 **truncate** 해서 0 errors 였지만(=사실상 잘려 색인됨), AMD +Lemonade 같은 strict 백엔드는 `500 "input (N tokens) is too large ... increase +the physical batch size"` 로 **거부**한다. 임베더-무관하게 견고하려면 청커가 +애초에 예산 초과 청크를 만들지 않는 게 옳다. (전역 truncate 를 임베더 측에 +넣는 대안은 silently-잘림이라 reject.) + +**cascade / 업그레이드.** `chunker_version` 라벨이 `md-heading-v1` → `md-heading-v2` +로 바뀌므로, **다음 plain `kebab ingest` 에서 markdown 자산이 1회 자동 +재청크**된다(`--force-reingest` 불필요 — skip 비교가 mismatch). 코드/PDF 자산은 +각자 chunker_version 이 unchanged 라 영향 없음. wire / CLI / `--json` 포맷 +불변(검색 hit 의 텍스트·citation 모양 동일, 단 거대 블록이 여러 hit 로 나뉠 수 +있음). **알아둘 wrinkle**: markdown 청커 dispatch 는 하드코딩이고 config +`chunker_version` 문자열은 impl 선택에 쓰이지 않는다 → 기존 config 가 +`chunker_version = "md-heading-v1"` 로 핀돼 있어도 실제로는 v2 가 돈다(문자열은 +정보성). 새 default config 와 `kebab config migrate` 는 `max_chunk_tokens` 를 +additive 로 주입한다. + +**citation 정밀도(known limitation).** 분할 조각은 원 블록의 `source_spans` +(블록 전체 범위)를 그대로 갖는다 — 즉 거대 블록을 쪼갠 조각의 인용은 +**블록 단위**(sub-line 정밀 아님)다. fenced 코드의 `SourceSpan::Line` 이 fence +줄을 포함하는 비대칭 때문에 조각별 줄 범위를 정확히 좁히는 건 (span,code) 만으로 +일반적으로 불가능 → "절대 틀리지 않되 블록 단위" 를 택했다. 일반 청크(미분할)는 +v1 과 byte-identical 이라 영향 없음. + +**도그푸딩 evidence** (실험 KB `/home/user/large_data/out/kebab-ab/xdg_sources`, +arctic-embed-l-v2 @ Lemonade `.243`). v2 전: 620 중 2 doc(SERVER-22906/23097) +임베드 실패. v2 후: **620/620, errors=0**, 7114 청크 전부 ≤ 4000(초과 0). +SERVER-22906 → 14 청크(max 3897), SERVER-23097 → 5 청크(max 3850). 검색 payoff: +"WiredTiger excessive memory cache size" 질의에 SERVER-22906 가 **1위(0.977)** — +"임베드 불가" 에서 "최상위 검색 결과" 로. `--trust-min primary` 등 출처 필터 +정상 유지. + +**일치 재테스트 (사용자 실 config 재현 — 이미지 OCR + PDF OCR ON, paddle-onnx)**. +초기 검증은 markdown-only 였으나 사용자 실 config 가 image/pdf OCR ON 임을 반영해 +미디어(이미지 2 + scanned PDF 2)를 같은 paddle-onnx + arctic@Lemonade 로 재인덱싱. +**여기서 image-OCR 구멍 발견·수정**: 분할 판정을 `token_estimate` 로 하면 +ImageRef 청크는 `token_estimate=0`(image-only) 이라 OCR text 가 거대해도 분할이 +안 됨 → 빽빽한 스크린샷 OCR 이 임베더 ctx 초과 시 strict 백엔드에서 실패 가능. +판정을 실제 text 길이로 교정 + 회귀 테스트(`oversize_image_ocr_chunk_splits`). +검증: scanned_page1/2.pdf OCR→청크(pdf-page-v1.1)→arctic 임베드 정상(max tok +443/672), 624 자산 errors=0. **known limitation**: PDF 는 별도 청커 `pdf-page-v1.1` +이라 이 oversize-split 미적용 — 초고밀도 scanned page 가 한 청크로 budget 초과 시 +잔존(현 fixture 는 무관). md 청커(이미지 OCR text 포함)만 v2 가 커버. 후속 후보: +pdf-page 청커에도 동일 oversize-split. + +p1-5(md-heading-v1) 를 확장하며 frozen 설계 doc / frozen p1-5 spec 은 +미변경(설계 §9 가 `md-heading-v2` 라벨 bump 를 이미 변경 메커니즘으로 명시 — +pdf-page-v1→v1.1 선례 동형). 설계: `docs/superpowers/plans/2026-06-24-md-heading-v2-oversize-split.md`. + ## 2026-06-21 — provenance 출처 필터: `[[workspace.sources]]` 멀티소스 + `--source` / `--source-type` (v0.29.0) **무엇을 추가했나.** 혼합 출처 KB(예: 위키 문서 + jira 이슈)에서 "출처별로