Merge pull request 'refactor(app): extract dispatch polymorphism — App.extract_for(...) + 11 Extractor registry' (#187 ) from refactor/extractor-dispatch-unification into main

Reviewed-on: #187
refactor(app): extract dispatch polymorphism — App.extract_for(...) + 11 Extractor registry
2026-05-26 21:07:20 +00:00 · 2026-05-26 17:43:44 +00:00 · 2026-05-26 15:37:21 +00:00 · 2026-05-26 15:00:59 +00:00 · 2026-05-26 12:31:29 +00:00 · 2026-05-26 12:19:32 +00:00
301 changed files with 48007 additions and 918 deletions
--- a/.gitignore
+++ b/.gitignore
@@ -1,6 +1,6 @@
 .superpowers/
 .worktrees/
 .claude/
-/target/
+/target
 **/*.rs.bk
 Cargo.lock.bak
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -27,7 +27,7 @@ cargo build --release                          # produces target/release/kebab

 `-j 1` for the full workspace test isn't optional: 18 integration-test binaries each link `lance` + `datafusion` + `arrow` + `tantivy` and the parallel link step exhausts memory (linker gets SIGKILL'd, build silently fails partway). Per-crate runs are fine in parallel.

-`target/` is 6–10 GB after a fresh build (DataFusion + Lance + fastembed + 18 × test-binary debug info). The dev/test profile is already trimmed (`debug = "line-tables-only"`, `split-debuginfo = "unpacked"` — see workspace `Cargo.toml`). Run `cargo clean` after phase merges if disk pressure shows up; backtraces still resolve to function + line.
+`target/` is 6–10 GB after a fresh build but **balloons to 90+ GB after a few task cycles** (each fb-* batch adds incremental compile artifacts on top of the existing 18 × test-binary debug info). The dev/test profile is already trimmed (`debug = "line-tables-only"`, `split-debuginfo = "unpacked"` — see workspace `Cargo.toml`). Run `cargo clean` **routinely after each merged PR**, not just "if pressure shows up" — disk space is tight and recovery via `cargo clean` is cheap (one re-link per crate on next build). Verified pattern: 92 GB → 0 GB in seconds, backtraces still resolve to function + line.

 ## The facade rule

--- a/Cargo.lock
+++ b/Cargo.lock
--- a/Cargo.toml
+++ b/Cargo.toml
@@ -2,11 +2,9 @@
 resolver = "3"
 members = [
    "crates/kebab-core",
-    "crates/kebab-parse-types",
    "crates/kebab-config",
    "crates/kebab-source-fs",
    "crates/kebab-parse-md",
-    "crates/kebab-normalize",
    "crates/kebab-chunk",
    "crates/kebab-store-sqlite",
    "crates/kebab-store-vector",
@@ -23,6 +21,8 @@ members = [
    "crates/kebab-parse-pdf",
    "crates/kebab-tui",
    "crates/kebab-mcp",
+    "crates/kebab-parse-code",
+    "crates/kebab-nli",
 ]

 [workspace.package]
@@ -30,7 +30,95 @@ edition       = "2024"
 rust-version  = "1.85"
 license       = "MIT OR Apache-2.0"
 repository    = "https://github.com/altair823/kebab"
-version       = "0.6.0"
+version       = "0.19.0"
+
+# pre-v0.18 workspace-wide cleanup: enable clippy::pedantic group with
+# intentional allow-list. The allowed lints are either cosmetic (doc style),
+# informational (function size), or carry intentional truncation we accept
+# (numeric casts in tokenizer/ONNX inputs, hash modular reduction, etc).
+[workspace.lints.clippy]
+pedantic = { level = "warn", priority = -1 }
+# Intentional u32 ↔ i64 casts in kebab-nli (ONNX i64 inputs from tokenizer u32 ids).
+# u64 ↔ usize across kebab-store-sqlite row counts. Wide truncation is auditable
+# at use site, not lint-wide.
+cast_possible_truncation = "allow"
+cast_possible_wrap       = "allow"
+cast_sign_loss           = "allow"
+cast_precision_loss      = "allow"
+# Doc markdown style is cosmetic; we run rustdoc on demand.
+doc_markdown            = "allow"
+missing_errors_doc      = "allow"
+missing_panics_doc      = "allow"
+# Informational only — splitting a long pipeline function isn't always cleaner.
+too_many_lines          = "allow"
+# `Foo::default()` is concise and idiomatic here; `<Foo as Default>::default()`
+# adds noise without surfacing intent.
+default_trait_access    = "allow"
+# Module name prefix on public items keeps the wire/log surface readable
+# (`refusal_reason::no_chunks` etc).
+module_name_repetitions = "allow"
+# We use `#[must_use]` deliberately on public results, not blanket.
+must_use_candidate      = "allow"
+# `String` arg sometimes signals "I'll consume this" — let signature decide.
+needless_pass_by_value  = "allow"
+# Idiomatic single-line bindings stay; let-else expansion isn't always clearer.
+manual_let_else         = "allow"
+# `use` after `let` is a common kebab pattern (scoped imports next to use site).
+items_after_statements  = "allow"
+# Naming pairs like `chunk_id` / `chunks_id` are intentional domain terms.
+similar_names           = "allow"
+# `iter.map(format!).collect::<String>()` is idiomatic when the per-element
+# string is genuinely independent — `fold` only wins on accumulation patterns.
+format_collect          = "allow"
+# Exhaustive `match` with explicit variant arms (vs `_`) catches future
+# variant additions at compile time (kebab core's `RefusalReason` pattern).
+match_wildcard_for_single_variants = "allow"
+# Copy types under `&self` keep call-site discipline; auto-deref noise > tiny perf gain.
+trivially_copy_pass_by_ref = "allow"
+# `unnecessary_wraps` flags helpers that could drop `Result`, but keeping the
+# Result allows future error variants without churning callers.
+unnecessary_wraps        = "allow"
+# NLI score / RRF fusion / similarity threshold comparisons are intentional —
+# floats live in the `[0, 1]` band and are compared with explicit thresholds.
+float_cmp                = "allow"
+# File-extension dispatch is keyed on ASCII conventions; case sensitivity
+# is part of the spec for `.md`, `.pdf`, etc.
+case_sensitive_file_extension_comparisons = "allow"
+# Config / opts structs intentionally bundle boolean flags (ingest options,
+# search modes, etc) — splitting them into enums would obscure the wire shape.
+struct_excessive_bools   = "allow"
+# `bytecount` crate would be a new dep just for one-off ASCII counts.
+naive_bytecount          = "allow"
+# `#[ignore]` annotations on tests document via the test name + nearby comment.
+ignore_without_reason    = "allow"
+# `format!` push patterns are a hot path for kebab-tui's progressive rendering;
+# `write!` rewrite needs a verified-equal benchmark before swapping.
+format_push_string       = "allow"
+# Builder-style `with_*` methods return `Self`; the existing `#[must_use]`
+# discipline lives on aggregate constructors, not every chainable setter.
+return_self_not_must_use = "allow"
+# Match arms grouped by side-effect over body equality (e.g. snake_case wire
+# label tables) — fanning them out keeps adding a new variant trivial.
+match_same_arms          = "allow"
+# Remaining style-only warnings: trailing `continue` is sometimes clearer than
+# rewriting, `_x` underscored bindings document intent at the use site, and
+# `!(a == b)` reads better than `a != b` when paired with a complementary check.
+needless_continue        = "allow"
+used_underscore_binding  = "allow"
+nonminimal_bool          = "allow"
+# Other one-off cosmetic items: large literal formatting, doc link quoting,
+# `Clone::clone_from` swap, `str::replace` chaining, `Iterator::any` ergonomics.
+unreadable_literal           = "allow"
+many_single_char_names       = "allow"
+doc_link_with_quotes         = "allow"
+assigning_clones             = "allow"
+collapsible_str_replace      = "allow"
+trivial_regex                = "allow"
+elidable_lifetime_names      = "allow"
+range_plus_one               = "allow"
+explicit_iter_loop           = "allow"
+implicit_hasher              = "allow"
+ref_option                   = "allow"

 [workspace.dependencies]
 anyhow       = "1"
@@ -81,6 +169,39 @@ rmcp         = { version = "1.6", default-features = false, features = ["server"
 # sync via reqwest::blocking — wiremock is dev-only there).
 wiremock     = "0.6"
 base64       = "0.22"
+# Pure-Rust git library for repo metadata detection (kebab-parse-code).
+# No `git` binary required. Default features include thread-safety + most
+# object-reading capabilities needed for HEAD name + commit SHA queries.
+gix          = { version = "0.70", default-features = false, features = ["revision"] }
+# Rust source parsing for code ingest (kebab-parse-code, p10-1A-2). The
+# chunker stays tree-sitter-free — AST work is parser-side per design §6.3.
+tree-sitter      = "0.26"
+tree-sitter-rust = "0.24"
+# Python / TS / JS grammars for code ingest (kebab-parse-code, p10-1B).
+tree-sitter-python     = "0.25.0"
+tree-sitter-typescript = "0.23.2"
+tree-sitter-javascript = "0.25.0"
+# Go grammar for code ingest (kebab-parse-code, p10-1C-Go).
+tree-sitter-go         = "0.25.0"
+# JVM family grammars for code ingest (kebab-parse-code, p10-1C-JK).
+tree-sitter-java       = "0.23.5"
+tree-sitter-kotlin-ng  = "1.1.0"   # bare tree-sitter-kotlin requires ts <0.23; -ng uses tree-sitter-language 0.1 (ts 0.26 compat)
+# C/C++ family grammars for code ingest (kebab-parse-code, p10-1D).
+tree-sitter-c         = "0.24.2"
+tree-sitter-cpp       = "0.23.4"
+# fb-41 PR-9 (kebab-nli): mDeBERTa-v3 XNLI verifier deps. Versions match
+# the fastembed 4.9 transitive set so the ONNX Runtime + tokenizer stack
+# stays single-versioned across the workspace. ort `default-features=false`
+# drops the bundled binary downloader (fastembed already provides one);
+# tokenizers `default-features=false, onig` swaps the default `esaxx` regex
+# backend for `onig` so the build doesn't need libstdc++ headers (verified
+# via PR-9a pre-flight: SentencePiece tokenizer.json loads + KR/EN encode).
+# hf-hub uses `ureq + rustls-tls` to stay aligned with kebab-embed-local's
+# pure-Rust TLS stack.
+ort          = { version = "=2.0.0-rc.9", default-features = false, features = ["ndarray"] }
+tokenizers   = { version = "0.21", default-features = false, features = ["onig"] }
+hf-hub       = { version = "0.4", default-features = false, features = ["ureq", "rustls-tls"] }
+ndarray      = "0.16"

 # Disk-footprint trim for dev / test builds. Codegen, opt-level, and
 # behavior are unchanged — only DWARF debug info is reduced (line
--- a/HANDOFF.md
+++ b/HANDOFF.md
@@ -4,7 +4,7 @@

 ## 한 줄 요약

-P0–P5 + P6 + P7 + P9-1/2/3/4 (Library / Search / Ask / Inspect) 머지 완료. `kebab ingest` 가 markdown / image / PDF 모두 처리. `kebab search` / `kebab ask` 가 매체 가로질러 결과 + page citation 반환. `kebab tui` 가 4 패널 (Library + Search + Ask + Inspect) 제공 — 사용자가 `?` 로 ask, `/` 로 search, Library Enter / Search `i` 로 inspect, Search `g` 로 editor jump. 다음 후보 = P9-5 (desktop tauri) 또는 보류 중인 P8 (audio) 의 시스템 dep brainstorm.
+P0–P5 + P6 + P7 + P9-1/2/3/4 (Library / Search / Ask / Inspect) + P10 전체 머지 완료 (현재 **v0.18.0**). `kebab ingest` 가 markdown / image / PDF / 소스코드 (Rust / Python / TS / JS / Go / Java / Kotlin / C / C++) / Tier 2 리소스 파일 (yaml/k8s / dockerfile / toml / json / xml / groovy / go-mod) + Tier 3 paragraph fallback (shell / 비-k8s YAML / AST 실패 케이스) 처리. `kebab search` / `kebab ask` 가 매체 가로질러 결과 + page / code citation 반환. `kebab tui` 가 4 패널 (Library + Search + Ask + Inspect) 제공. **v0.17.0 cut (2026-05-24)**: 한국어 trigram FTS5 tokenizer (PR #159) + C typedef alias unit (PR #160) + `code_lang_chunk_breakdown` additive (PR #161). **v0.17.1 cut (2026-05-25)**: 확장 도그푸딩 후 `[models.llm] request_timeout_secs` config 노브 (PR #162) + sudo 없이 ollama 설치 + `kebab ask --stream` UX 권장 docs (PR #163). **v0.17.2 cut (2026-05-25)**: v0.17.1 post-dogfood polish — `[image.ocr] request_timeout_secs` 별 노브 (PR #164, v0.17.1 미진행 closure) + `heading_path` FTS5 column filter 로 text-only 매칭 + raw-mode escape hatch (PR #165, 2026-05-24 v0.17.0 trigram entry 의 JSON 노이즈 closure). **v0.18.0 cut (2026-05-26)**: fb-41 multi-hop RAG + NLI verification ship (PR #176-180) — `kebab ask --multi-hop` 의 decompose → decide → synthesize loop + mDeBERTa-v3 XNLI ONNX post-synthesize entailment 검사. dogfood S7 caffeine hallucination 의 silent LLM-self-judge ceiling 해결 (nli_score 0.0035 graceful refuse). 추가 `chore: workspace-wide cleanup + post-PR9 refactor` (PR #181) — clippy::pedantic baseline + H1 config wiring + 9 new tests. 자세한 영향은 [v0.17.0 release notes](https://gitea.altair823.xyz/altair823-org/kebab/releases/tag/v0.17.0) + [v0.17.1 release notes](https://gitea.altair823.xyz/altair823-org/kebab/releases/tag/v0.17.1) + [v0.17.2 release notes](https://gitea.altair823.xyz/altair823-org/kebab/releases/tag/v0.17.2) + [v0.18.0 release notes](https://gitea.altair823.xyz/altair823-org/kebab/releases/tag/v0.18.0). 구조적으로 남은 component 는 P9-5 (desktop tauri) 하나뿐, P8 (audio) 는 사용자 보류.

 ## Phase 로드맵

@@ -20,17 +20,28 @@ P0–P5 + P6 + P7 + P9-1/2/3/4 (Library / Search / Ask / Inspect) 머지 완료.
 | **P7** | PDF text + page citation | `kebab-parse-pdf` | P5 | ✅ 완료 (3/3 component, page-level chunker + ingest wiring) |
 | **P8** | 음성 transcription + timestamp citation | `kebab-parse-audio` | P5 | ⏸ 보류 (whisper-rs 시스템 dep brainstorm 필요) |
 | **P9** | TUI + desktop app | `kebab-tui`, `kebab-desktop` | P5 | 🟡 진행 (4/5 component — P9-1/2/3/4 완료 [Library / Search / Ask / Inspect], P9-5 desktop 예정 · 도그푸딩 피드백 **20/20 ✅**) |
+| **P10** | code ingest framework | `kebab-parse-code` | P5 | 🟡 진행 중 — 1A-1 ✅ (wire schema + parse-code skeleton + filter flags), 1A-2 ✅ (Rust AST chunker, `code-rust-ast-v1` — v0.7.0), 1B ✅ (Python/TS/JS AST chunkers — v0.8.0 이후), **1C-Go ✅ (Go AST chunker, `code-go-ast-v1` — v0.12.0)**, **1C-JavaKotlin ✅ (Java + Kotlin AST chunkers, `code-java-ast-v1` / `code-kotlin-ast-v1` — v0.13.0)**, **2 ✅ (Tier 2 resource-aware: yaml/k8s + dockerfile + manifest, `k8s-manifest-resource-v1` / `dockerfile-file-v1` / `manifest-file-v1` — v0.14.0)**, **3 ✅ (Tier 3 paragraph fallback: code-text-paragraph-v1 — v0.15.0)**, **1D ✅ (C + C++ AST chunkers, code-c-ast-v1 + code-cpp-ast-v1 — v0.16.0)** |

 P0~P5 직렬. P6~P9 P5 이후 병렬 가능.

 ## Component 카운트

-총 33 component task — spec 시점 31 개 + 후속 wiring task 3 (P3-5 / P6-4 / P7-3) 가 머지 시점에 추가됨. per-component 진행 + status 는 [tasks/INDEX.md](tasks/INDEX.md).
+총 33 component task — spec 시점 31 개 + 후속 wiring task 3 (P3-5 / P6-4 / P7-3) 가 머지 시점에 추가됨. v0.18.0 cut 시점에 fb-41 multi-hop RAG + NLI verification (PR-9 5 sub-PRs) 가 P9 추가 component 로 ship — `kebab-nli` 신규 crate (mDeBERTa-v3 XNLI ONNX verifier) + `kebab-rag::ask_multi_hop` (decompose/decide/synthesize loop + step 8.5 NLI hook). per-component 진행 + status 는 [tasks/INDEX.md](tasks/INDEX.md).

 ## 머지 후 발견된 버그 / 결정 (요약)

 머지 후 발견된 모든 deviation / hotfix 의 dated 로그는 [tasks/HOTFIXES.md](tasks/HOTFIXES.md). 본 요약은 \"누군가가 인수받을 때 알아두면 시간을 많이 절약하는\" 항목만:

+- **2026-05-26 kebab-normalize + kebab-parse-types 흡수 (24 → 22 crates, design §3.7b 재작성)** — v0.19.0 cut. 4 parser 중 markdown 한 갈래만 lift 를 경유하는 reality 가 design §3.7b 의 fan-in ≥ 2 가정과 diverge → thin layer (`kebab-parse-types`) + `kebab-normalize` 두 crate 가 `kebab-parse-md` 로 흡수. 5 사용 type + 3 forward-declared struct 모두 `kebab-parse-md::{types,normalize}` module 의 `pub` re-export 로 보존. wire / surface impact = 0 (CLI / TUI / MCP / `--json` / config / XDG / parser_version 모두 unchanged). 자세한 내용: `tasks/HOTFIXES.md` (2026-05-26 design deviation entry).
+- **2026-05-26 v0.18.0 fb-41 multi-hop RAG + NLI verification ship (PR #176-180) + post-PR9 cleanup (PR #181)** — pre-v0.18.0 dogfood (`/build/cache/dogfood-v018/`, 33 assets / 205 chunks, gemma3:4b CPU only / 16 GB RAM) 에서 발견된 S7 caffeine hallucination 의 root cause = LLM-self-judge ceiling (synthesize 가 chunks 와 무관한 Adam optimizer gradient 식을 silent emit, self-judge 가 reject 못함). 학계 표준 (Self-RAG, CRAG, Auto-GDA, MedTrust-RAG) 결론 = deterministic post-synthesis verification. mDeBERTa-v3 XNLI ONNX (280 MB, Xenova HF) 가 `(packed_chunks, answer)` entailment 검사 — `[rag] nli_threshold > 0` (default 0.0 = disabled, production 권장 0.5) 일 때 활성. dogfood retest 측정 — S7 PR-8 baseline `grounded=true + Adam hallucination` → PR-9 `nli_verification_failed, nli_score 0.0035`. wire additive minor — `answer.v1.verification` field + `refusal_reason` 의 `nli_verification_failed` / `nli_model_unavailable` 추가, pre-v0.18 reader 무영향. 5 sub-PR 시퀀스 + cleanup PR (clippy::pedantic baseline + 의도적 30+ allow + H1 `[models.nli].model` config wiring + 9 new tests). post-refactor retest = PR-9d byte-identical (deterministic 확인). 자세한 내용: `tasks/HOTFIXES.md` (2026-05-25 fb-41 PR-9 closure entry + S3 follow-up).
+- **2026-05-25 v0.17.2 post-v0.17.1 polish (PR #164 + #165)** — v0.17.1 의 두 follow-up closure. (1) `[image.ocr] request_timeout_secs` 별 노브 — `crates/kebab-parse-image/src/ocr.rs::REQUEST_TIMEOUT` hard 300s 제거, LLM 쪽 패턴 (PR #162) 을 OCR 어댑터에 동일 적용. 사용자 결정으로 별 노브 분리 (OCR vs LLM 의 cold start 패턴이 달라 독립 조절). v0.17.1 미진행 항목 closure. (2) `chunks_fts` 의 `heading_path` 컬럼이 JSON 표기 + path 세그먼트 까지 trigram 색인 → query false positive 가능 문제 closure. `lexical.rs::build_match_string` 가 non-raw 분기 결과를 `text : (<expr>)` 로 wrap — heading 색인 V007 verbatim 유지, 매칭만 text 한정. 사용자가 명시 heading 검색 하려면 raw mode `'heading_path : <token>'` escape hatch (SKILL.md 갱신). 둘 다 additive (옛 config 호환) / re-ingest 불필요. 자세한 내용: `tasks/HOTFIXES.md` (2026-05-25 v0.17.2 두 entry).
+- **2026-05-25 v0.17.1 post-dogfood (PR #162 + #163)** — 확장 도그푸딩 (16 GB CPU only, gemma4:e4b 시도) 에서 발견된 두 follow-up 한 묶음. (1) `crates/kebab-llm-local/src/ollama.rs::REQUEST_TIMEOUT` hard 300s → `[models.llm] request_timeout_secs` config + env override (additive, default 300, `=0` 은 disable 아닌 "즉시 timeout" 이라 doc 명시). (2) README + SMOKE 에 sudo / systemd 없이 ollama 설치 + ≤4B Q4 권장 모델 + `kebab ask --stream` UX 권장 docs. additive only — 옛 config / wire 호환. 자세한 내용: `tasks/HOTFIXES.md` (2026-05-25).
+- **2026-05-24 v0.17.0 PR-C `code_lang_chunk_breakdown` additive (closure of 2026-05-22 LOW)** — `schema.v1.stats` 에 chunk 수 집계 신규 키. 기존 `code_lang_breakdown` (doc count) 와 sister. 또 기존 두 필드 JSON schema description 의 "chunk count" 오기재 → "doc count" 로 정정. wire additive — schema_version bump 불필요. 자세한 내용: `tasks/HOTFIXES.md` (2026-05-24 PR-C).
+- **2026-05-24 v0.17.0 PR-B C typedef alias unit (closure of 2026-05-21)** — `kebab-parse-code::c::extract_blocks` 의 `type_definition` 분기로 inner anonymous struct/enum/union → declarator 의 typedef alias 이름으로 synthetic unit 방출. `PARSER_VERSION code-c-v1` → `code-c-v2` bump + 같은-asset/다른-doc_id 케이스용 `purge_workspace_path_for_parser_bump` cascade (`stale_chunk_ids_for_workspace_path_except_doc_id` + `purge_document_at_workspace_path_except_doc_id` helper 신규). 사용자 작업 불필요 (다음 ingest 가 자동 재처리). 자세한 내용: `tasks/HOTFIXES.md` (2026-05-24 PR-B).
+- **2026-05-24 v0.17.0 PR-A 한국어 trigram tokenizer 채택 (closure of 2026-05-22 한국어 lexical)** — `chunks_fts` 가 FTS5 `unicode61` → `trigram` 으로 V007 migration (자동 backfill, re-ingest 불필요). `lexical.rs::build_match_string` trigram-aware 재설계 — multi-token 한국어 query (`해시 충돌`) 가 whole-phrase 후보로 hit, 한영 혼합 (`Rust 충돌은`) 도 OR-combined. 2자 이하 query 는 0-hit + CLI/TUI/wire `hint` 안내. 영어 lexical 도 substring 매칭으로 바뀜 (recall ↑ / 단어 경계 ↓). `kebab.sqlite` 크기 ~2-5배 증가 (trigram index). 자세한 내용: `tasks/HOTFIXES.md` (2026-05-24).
+- **2026-05-22 P10 종합 도그푸딩 round 2 (한국어 lexical 검색 한계)** — `kebab search --mode lexical` 의 한국어 query 가 FTS5 `unicode61` 토크나이저에서 거의 0 hit (어절 단위 토큰화 → 부분 매칭 불가). 기본 hybrid 모드는 `multilingual-e5-small` vector 가 carry 해 한국어 검색 정상. **closure**: 위 2026-05-24 v0.17.0 entry.
+- **2026-05-20 P10-1B (Rust 1A symbol path 비일관 + expression-level 함수 미방출)** — (a) Rust `code-rust-ast-v1` 은 file-scope nesting 만 (workspace path prefix 없음), 1B 의 Python/TypeScript/JavaScript 는 workspace 경로 → module path prefix 사용 (비일관 수용, retrofit = chunker_version bump + reindex 필요, 사용자 명시 요청까지 보류); (b) TS/JS 의 `const foo = () => {...}` 같은 expression-level 함수는 `<top-level>` glue 로 처리됨 (declaration-level 단위만 1B 1차 범위). 자세한 내용: `tasks/HOTFIXES.md` (2026-05-20) 두 항목.
+- **2026-05-19 P10-1A-2 (code_rust_ast_v1.rs + SourceType)** — `AST_CHUNK_MAX_LINES` 상수가 `IngestCodeCfg.ast_chunk_max_lines` 를 읽지 않고 모듈 상수 200 고정 (Chunker trait 이 per-medium config 미노출); `SourceType::Code` variant 부재로 code 파일이 `SourceType::Note` 로 분류됨 — 두 항목 모두 `tasks/HOTFIXES.md` (2026-05-19) 에 기록.
 - **2026-05-07 fb-26 (progress.rs)** — `Aborted` unconditional writeln (TTY duplicate) + `Completed` TTY no summary fixed; `KEBAB_PROGRESS=plain` env + quiet suppression added
 - **2026-05-07 fb-28 (main.rs)** — `--readonly` (KEBAB_READONLY) blocks Ingest/IngestFile/IngestStdin/Reset; `--quiet` suppresses progress stderr; error.v1 code: "readonly_mode"

@@ -78,13 +89,13 @@ P0~P5 직렬. P6~P9 P5 이후 병렬 가능.

 ## 다음 task 후보

- **P9-2 TUI search** — `App.search` slot 채움. Library 의 `/` 가 enable 됨.
- **P9-3 TUI ask** — `App.ask` slot 채움. `?` enable.
- **P9-4 TUI inspect** — `App.inspect` slot 채움. `Enter` enable.
- **P9-5 desktop tauri** — 별도 분기. PDF citation rendering UI 가치 큼.
- **P8 audio brainstorm** — whisper-rs 시스템 dep 받을지 / 외부 transcription endpoint 사용할지 사용자 결정 필요. 사용자 패턴 (책+PDF 위주, audio 의향 없음) 상 후순위.
+구조적으로 미완인 component 는 P9-5 하나뿐. 나머지는 도그푸딩 follow-up (아래 "P10 dogfooding 백로그") 또는 사용자 결정 대기.

-P9-2/3/4 는 P9-1 의 parallel-safety contract (sub-state slot 패턴) 덕에 병렬 진행 가능 — 같은 `App` 손대지 않음.
+- **P9-5 desktop tauri** — 마지막 남은 P9 component. `kebab-desktop` crate + Tauri 앱, 별도 분기. PDF citation rendering UI 가치 큼. 사용자 우선순위 (P9 우선 · 책/PDF 위주) 와 부합.
+- **P10 도그푸딩 round 2 follow-up** — ✅ v0.17.0 cut (2026-05-24) 으로 세 항목 모두 closure (한국어 trigram PR-A + C typedef alias PR-B + code_lang_chunk_breakdown additive PR-C). 상세 cross-link: 아래 "P10 dogfooding 백로그" 절 + `tasks/HOTFIXES.md` (2026-05-24 PR-A/B/C).
+- **P8 audio brainstorm** — whisper-rs 시스템 dep 받을지 / 외부 transcription endpoint 사용할지 사용자 결정 필요. 사용자 패턴 (책+PDF 위주, audio 의향 없음) 상 보류.
+- **fb-41 multi-hop reasoning** — ⏳ 미구현, XL, eval 인프라 선행 + brainstorm 필요.
+- **Rust symbol path retrofit** — Rust `code-rust-ast-v1` symbol 이 file-scope-only (1B+ 는 module prefix). `code-rust-ast-v2` bump + Rust corpus re-ingest 비용 → 사용자 명시 요청까지 보류. HOTFIXES `2026-05-20`.

 ### P9 dogfooding 백로그 (fb-26 ~ fb-42) — release 분할

@@ -93,11 +104,20 @@ P9-2/3/4 는 P9-1 의 parallel-safety contract (sub-state slot 패턴) 덕에
 - **0.3.0 — agent foundation** ✅ cut 2026-05-07: fb-26 (log), fb-27 (introspection/error wire), fb-28 (readonly/quiet). ~~fb-29 (daemon)~~ → 🚫 **deferred** — fb-30 stdio MCP 가 동일 가치를 daemon 복잡도 없이 제공.
 - **0.4.0 — agent integration (MCP)** ✅ cut: fb-30 (MCP stdio), fb-31 (single-file/stdin ingest).
 - **0.5.0 — agent surface refinement (additive)** ✅ cut 2026-05-10: fb-32 (stale doc indicator), fb-33 (streaming ask), fb-34 (output budget controls), fb-35 (verbatim fetch), fb-36 (search filter args), fb-37 (trace + stats). 모두 wire schema additive minor.
- **0.6.0 — RAG quality** 🟡 진행: fb-38 (score semantics) ✅ 머지 (2026-05-10), fb-40 (fact-grounded answer / rag-v2 prompt) ✅ 머지 (2026-05-10), fb-39 (retrieval precision tuning, embedding_version cascade) — 미진행 (eval golden set 선행 필요).
- **0.7.0 또는 P+**: fb-41 (multi-hop reasoning, XL), fb-42 (bulk multi-query / rerank, Nice).
+- **0.6.0 — RAG quality** ✅ 대부분 머지 (2026-05-10): fb-38 (score semantics) ✅, fb-39 (eval foundation — `precision_at_k_chunk` metric) ✅, fb-39b (embedding upgrade — multilingual-e5-large default) ✅, fb-40 (fact-grounded answer / rag-v2 prompt) ✅. 잔여 = fb-39 의 retrieval precision lever 실제 적용 (eval golden set 확장 선행 필요).
+- **0.7.0 또는 P+**: fb-41 (multi-hop reasoning, XL) — ⏳ 미구현 · brainstorm 필요; fb-42 (bulk multi-query) ✅ 머지 (2026-05-10, bulk only — rerank hint 은 deferred).

 각 fb spec frontmatter 의 `target_version` 필드가 source of truth. INDEX.md 의 release subheader 도 동일 grouping.

+### P10 dogfooding 백로그 (2026-05-22 round 2)
+
+P10 종합 도그푸딩 round 2 (`/build/cache/dogfood-p10b/`, OSS 8 repo + 한국어 위키 문서 10편) 에서 발견된 follow-up 후보. 자세한 내용 + 우선순위 근거는 `tasks/HOTFIXES.md` (2026-05-22).
+
+- **한국어 lexical tokenizer** — ✅ v0.17.0 (2026-05-24) PR-A 머지 (#159). V007 trigram migration 자동 backfill + `build_match_string` 재설계 + CLI/TUI/wire hint. HOTFIXES `2026-05-24 PR-A` 참조.
+- **code_lang_chunk_breakdown chunk 단위 집계 (LOW)** — ✅ v0.17.0 (2026-05-24) PR-C 머지 (#161). `schema.v1.stats` additive 필드. HOTFIXES `2026-05-24 PR-C` 참조.
+- **C typedef-wrapped struct (LOW)** — ✅ v0.17.0 (2026-05-24) PR-B 머지 (#160). `type_definition` 분기 + `PARSER_VERSION code-c-v2` bump + orphan purge cascade. HOTFIXES `2026-05-24 PR-B` 참조.
+- **ranking glue chunk 편향 (deferred)** — 자동 heuristic 은 user intent misalignment 위험. 사용자 명시 요청 전까지 surface 변경 0 유지. 1주+ 실사용 후 재 brainstorm.
+
 ## 검증된 운영 동작 (release binary, fastembed enabled)

 P7-3 머지 직후 25 시나리오 smoke 통과 — markdown + image + PDF 5 자산 워크스페이스에서 doctor / ingest / list / inspect / search (lex/vec/hybrid) / re-ingest / byte-edit re-ingest / corrupt PDF / RAG ask + page citation 모두. 자세한 시나리오 표는 conversation 기록 참조; 워크스페이스에 직접 돌려보는 절차는 [docs/SMOKE.md](docs/SMOKE.md).
--- a/README.md
+++ b/README.md
@@ -6,6 +6,20 @@

 - **Rust toolchain** ≥ 1.85 (workspace 가 edition 2024 + resolver 3 사용). [rustup](https://rustup.rs) 권장.
 - **Ollama** — `kebab ask` 와 이미지 OCR/caption 가 사용. `https://ollama.com/download` 에서 설치 후 `ollama serve` 실행. 기본 LLM 은 gemma4 계열 (`ollama pull gemma4:e4b`) — OCR / caption 도 같은 family 라 모델 하나만 pull 하면 됨. 더 큰 variant 원하면 `gemma4:26b` 등으로 config override. config 의 `[models.llm].endpoint` 에 host:port 명시.
+  - **CPU only / RAM ≤ 16 GB 환경 권장 모델**: gemma4:e4b (8B) 는 CPU 추론에 무거워 RAG 한 답변이 5분을 넘기기 쉽다 — `[models.llm] request_timeout_secs` 의 기본 300 s 한도에 걸려 `error: kb-rag: llm.generate_stream` 으로 떨어진다 (HOTFIXES 2026-05-25). `gemma3:4b` / `qwen2.5:3b` / `phi3:mini` 같은 ≤ 4B Q4 모델로 바꾸면 답변 1-3 분에 안정 동작 (확장 도그푸딩에서 검증). 모델 storage 가 부담이면 `OLLAMA_MODELS=/path` env 로 위치 분리 가능.
+  - **`request_timeout_secs` 노브 (v0.17.0)**: `[models.llm] request_timeout_secs = 1200` (또는 `KEBAB_MODELS_LLM_REQUEST_TIMEOUT_SECS=1200`) 로 한도를 늘려 큰 모델도 시도 가능. 단 응답 동안 RAM 점유가 길어진다. **`= 0` 은 disable 이 아니라 "즉시 timeout"** (reqwest 의 의미상) — "사실상 무제한" 의도면 `u64::MAX` 또는 `86400` 같이 큰 finite 값 사용.
+  - **sudo 없이 설치 (격리 디렉토리 사용)**: `install.sh` 가 `/usr/local/bin/ollama` + `systemd` 유닛까지 건드리는 게 부담이면 binary tarball 만 받아 사용자 디렉토리에 풀고 env 로 모델 위치 분리하면 된다.
+    ```bash
+    mkdir -p /opt/ollama/{models,logs}
+    curl -fL https://ollama.com/download/ollama-linux-amd64.tar.zst -o /tmp/ollama.tar.zst
+    zstd -d /tmp/ollama.tar.zst -o /tmp/ollama.tar && tar -xf /tmp/ollama.tar -C /opt/ollama/
+    # bin/ollama + lib/ollama/ 가 풀린다. 모델 디렉토리는 OLLAMA_MODELS 로 분리.
+    OLLAMA_MODELS=/opt/ollama/models OLLAMA_HOST=127.0.0.1:11434 \
+        /opt/ollama/bin/ollama serve > /opt/ollama/logs/serve.log 2>&1 &
+    /opt/ollama/bin/ollama pull gemma3:4b
+    ```
+    루트 디스크 부담을 분리하고 싶을 때 (`~/.ollama/models` 가 기본) 그대로 활용. systemd 가 없는 컨테이너 / WSL2 / 회사 머신 등에서 유용.
+  - **`kebab ask --stream` 권장 (fb-33)**: 모델 cold start 가 길 때 (8B+ 또는 첫 호출) `--stream` 으로 토큰을 stderr 에 ndjson 으로 흘려 받으면 5 분 timeout 한도 안에서도 첫 토큰이 빨리 보여 사용자 체감이 개선된다. 동일 inference 시간이라도 wait-and-pray 보다 progressive 가 안정적. CLI: `kebab ask "..." --stream 2> events.ndjson > final.json`. MCP host 도 `streaming_ask` capability flag 가 `true` 면 자동 사용 권장.
 - **빌드 디스크** — 첫 빌드 시 `target/` 가 6–10 GB (Lance + DataFusion + fastembed). 여유 확인.
 - **fastembed 모델** — 첫 `kebab ingest` 시 `multilingual-e5-large` (~1.3 GB, fb-39b) 자동 다운로드. `config.toml` 에서 `model = "multilingual-e5-small"` 로 명시하면 이전 모델 사용.

@@ -34,7 +48,7 @@ cargo install --git https://gitea.altair823.xyz/altair823-org/kebab.git --bin ke

 업데이트는 `git pull && cargo install --path crates/kebab-cli --locked --force` 또는 git URL 형식의 경우 `cargo install --git ... --force`.

-제거는 `cargo uninstall kebab-cli`. 이 명령은 binary 만 지우고 워크스페이스 데이터는 그대로 남는다. 데이터까지 정리하려면 `kebab reset --all --yes` (config + data + cache + state 4 개 XDG 경로 모두 wipe — **irreversible**, 재시작 시 `kebab init` 다시 실행). 부분 wipe 는 `kebab reset --data-only` (config 보존), `kebab reset --vector-only` (Lance + `embedding_records` 만, 다음 ingest 가 re-embed) 등.
+제거는 `cargo uninstall kebab-cli`. 이 명령은 binary 만 지우고 워크스페이스 데이터는 그대로 남는다. 데이터까지 정리하려면 `kebab reset --all --yes` (config + data + cache + state 4 개 XDG 경로 모두 wipe — **irreversible**, 재시작 시 `kebab init` 다시 실행). 부분 wipe 는 `kebab reset --data-only` (config 보존), `kebab reset --vector-only` (Lance + `embedding_records` 만, 다음 ingest 가 re-embed), **`kebab reset --orphans-only`** (현재 walker scope 밖에 있는 stored doc 만 정리 — `config.workspace.include` 좁히거나 sub-dir 옮긴 후 explicit reconcile; fs 의 file 은 건드리지 않음) 등.

 ## Quick start

@@ -42,7 +56,7 @@ cargo install --git https://gitea.altair823.xyz/altair823-org/kebab.git --bin ke
 # 첫 실행 — XDG 경로에 데이터 디렉토리 + config.toml 생성
 kebab init

-# config 손보고 — workspace.root, 모델 endpoint 등 설정 (지원 형식은 md / png / jpg / pdf 로 고정)
+# config 손보고 — workspace.root, 모델 endpoint 등 설정 (지원 형식: md / png / jpg / pdf / rs / py / ts / js / go)
 ${EDITOR:-vi} ~/.config/kebab/config.toml

 # 색인 (Markdown / 이미지 / PDF 모두 한 번에)
@@ -70,12 +84,12 @@ kebab doctor
 | 명령 | 동작 |
 |------|------|
 | `kebab init` | XDG 경로에 데이터 디렉토리 + config.toml 생성 |
-| `kebab ingest [<path>]` | Markdown / 이미지 / PDF 색인 (idempotent). TTY 에서는 stderr 진행 바, non-TTY (CI / pipe) 는 stderr 한 줄씩, `--json` 은 stdout 에 `ingest_progress.v1` 라인 streaming 후 마지막에 `ingest_report.v1`. Ctrl-C 한 번이면 현재 asset 마무리 후 abort (부분 commit 보존, idempotent re-run), 두 번째 Ctrl-C 는 hard exit. Markdown title 이 frontmatter 에 없어도 첫 H1 → H2 → 첫 paragraph 80 자 → 파일명 순으로 자동 채움 (parser_version `md-frontmatter-v2`) — 기존 색인된 doc 도 다음 ingest 에서 새 title 로 갱신. **Incremental** (p9-fb-23): 두 번째 이후의 ingest 는 변하지 않은 doc (blake3 + parser/chunker/embedder version 모두 동일) 의 parse/chunk/embed/vector upsert 를 자동 스킵. final summary 에 `N unchanged` 카운트 표시. `--force-reingest` 로 skip 무시 강제 재처리. **지원 형식** (extractor 자동 결정 — config 에 명시 불가): Markdown (`.md`), 이미지 (`.png` / `.jpg` / `.jpeg`, OCR + caption), PDF (`.pdf`). 다른 확장자는 자동 skip — `IngestItem.warnings` 에 사유 (`"unsupported media type: .docx"` 등), `IngestReport.skipped_by_extension` 에 카운트 분류, CLI / TUI summary 에 breakdown 표시. |
-| `kebab search --mode {lexical,vector,hybrid} "<query>" [--no-cache] [--max-tokens N] [--snippet-chars N] [--cursor <opaque>] [--tag T] [--lang L] [--path-glob G] [--trust-min LEVEL] [--media TYPE] [--ingested-after RFC3339] [--doc-id ID] [--trace] [--bulk]` | 검색. hybrid는 RRF fusion, citation 포함. 같은 process 안에서 동일 query (NFKC + trim + lowercase 정규화) 반복 시 in-process LRU 캐시 hit (capacity = `[search] cache_capacity`, default 256). `--no-cache` 로 강제 bypass — 디버깅용. ingest commit 발생 시 `kv['corpus_revision']` bump 으로 모든 entry 자동 stale. **`--max-tokens` / `--snippet-chars` / `--cursor` (p9-fb-34)** — agent budget controls. `--json` 출력은 `search_response.v1` wrapper (`{hits, next_cursor, truncated}`) — pre-fb-34 의 bare array 와 호환 안 됨. mismatched cursor → `error.v1.code = stale_cursor`. **filter flags (p9-fb-36):** `--tag` 는 반복 가능 flag (`--tag rust --tag async`) 로 OR 매칭, `--media` 는 `,` 구분 다중 값 OR 매칭, 나머지 flags 간은 AND 조합. `--trust-min` 은 `primary\|secondary\|generated` 중 하나 (해당 level 이상 포함). `--ingested-after` 는 RFC3339 UTC — 파싱 실패 시 `error.v1.code = config_invalid` (exit 2). `--media md` 는 `markdown` alias 로 정규화. 알 수 없는 `--media` 값은 무조건 empty hits (오류 아님). **`--trace` (p9-fb-37)** — `search_response.v1.trace` 에 lexical / vector pre-fusion 후보 + RRF union + per-stage timing (`lexical_ms` / `vector_ms` / `fusion_ms` / `total_ms`) 노출. trace 요청은 캐시 우회 (`--no-cache` 없이도 항상 cold). **`--bulk` (p9-fb-42)** — stdin ndjson 으로 N query 한 번에 실행. `--json` 면 stdout per-query ndjson (`bulk_search_item.v1`) + stderr summary (`bulk_summary: total=N succeeded=S failed=F`). Cap 100. agent 가 query decomposition 후 sub-query 일괄 실행 시 single round-trip — App instance 재사용으로 캐시 / embedder cold-start 비용 한 번만. Per-query failure 는 item 의 `error` (error.v1) 에 격리, 다른 query 계속 진행. |
+| `kebab ingest [<path>]` | Markdown / 이미지 / PDF / Rust 소스코드 색인 (idempotent). TTY 에서는 stderr 진행 바, non-TTY (CI / pipe) 는 stderr 한 줄씩, `--json` 은 stdout 에 `ingest_progress.v1` 라인 streaming 후 마지막에 `ingest_report.v1`. Ctrl-C 한 번이면 현재 asset 마무리 후 abort (부분 commit 보존, idempotent re-run), 두 번째 Ctrl-C 는 hard exit. Markdown title 이 frontmatter 에 없어도 첫 H1 → H2 → 첫 paragraph 80 자 → 파일명 순으로 자동 채움 (parser_version `md-frontmatter-v2`) — 기존 색인된 doc 도 다음 ingest 에서 새 title 로 갱신. **Incremental** (p9-fb-23): 두 번째 이후의 ingest 는 변하지 않은 doc (blake3 + parser/chunker/embedder version 모두 동일) 의 parse/chunk/embed/vector upsert 를 자동 스킵. final summary 에 `N unchanged` 카운트 표시. `--force-reingest` 로 skip 무시 강제 재처리. **지원 형식** (extractor 자동 결정 — config 에 명시 불가): Markdown (`.md`), 이미지 (`.png` / `.jpg` / `.jpeg`, OCR + caption), PDF (`.pdf`), **소스코드** (`.rs` → `code-rust-ast-v1`, `.py` → `code-python-ast-v1`, `.ts`/`.tsx` → `code-ts-ast-v1`, `.js`/`.mjs`/`.cjs`/`.jsx` → `code-js-ast-v1`, `.go` → `code-go-ast-v1`, `.java` → `code-java-ast-v1`, `.kt`/`.kts` → `code-kotlin-ast-v1`, `.c`/`.h` → `code-c-ast-v1`, `.cpp`/`.cc`/`.cxx`/`.hpp`/`.hh`/`.hxx` → `code-cpp-ast-v1` — 모두 tree-sitter AST chunker; **Tier 2 리소스 파일**: `.yaml`/`.yml` → `k8s-manifest-resource-v1` (apiVersion+kind 파싱), `Dockerfile`/`Dockerfile.*`/`*.dockerfile` → `dockerfile-file-v1` (전체 파일), `Cargo.toml`/`pyproject.toml`/`.toml`/`package.json`/`tsconfig.json`/`.json`/`pom.xml`/`.xml`/`build.gradle`/`.gradle`/`go.mod` → `manifest-file-v1` (전체 파일) — yaml (k8s) / dockerfile / toml / json / xml / groovy / go-mod 지원); **Tier 3 paragraph fallback** (`.sh`/`.bash`/`.zsh` → `code-text-paragraph-v1`, blank-line paragraph split + 80-line/20-overlap line-window. Tier 1/2 가 0 chunk 또는 Err 시 자동 fallback — 비-k8s YAML 같은 케이스 picked up. symbol = None, lang 은 원본 보존.). 다른 확장자는 자동 skip — `IngestItem.warnings` 에 사유 (`"unsupported media type: .docx"` 등), `IngestReport.skipped_by_extension` 에 카운트 분류, CLI / TUI summary 에 breakdown 표시. 코드 chunk 는 `citation.kind = "code"` 에 `citation.lang = "<lang>"` + `symbol` + line range 를 담고, SearchHit top-level 에 `code_lang` + `repo` (`.git/` walk-up 의 디렉토리 이름) 가 backfill 됨. `--code-lang rust` / `--code-lang python` / `--code-lang typescript` / `--code-lang javascript` / `--code-lang go` / `--code-lang java` / `--code-lang kotlin` / `--code-lang yaml` / `--code-lang dockerfile` / `--code-lang toml` / `--code-lang json` / `--code-lang xml` / `--code-lang groovy` / `--code-lang go-mod` / `--code-lang shell` / `--code-lang c` / `--code-lang cpp` / `--media code` filter 로 언어별·코드 전용 검색 가능 (p10-1A-1 filter flags). Python symbol 은 workspace 경로 → dotted module path prefix (예: `kebab_eval.metrics.compute_mrr`), TS/JS symbol 은 slash-style module path prefix (예: `src/Foo.Foo.search`), Go symbol 은 `package.Func` / `package.(*Receiver).Method` 형식, Java / Kotlin symbol 은 `com.foo.Foo.bar` 형식 (패키지 + 클래스 + 메서드/필드). |
+| `kebab search --mode {lexical,vector,hybrid} "<query>" [--no-cache] [--max-tokens N] [--snippet-chars N] [--cursor <opaque>] [--tag T] [--lang L] [--path-glob G] [--trust-min LEVEL] [--media TYPE] [--ingested-after RFC3339] [--doc-id ID] [--trace] [--bulk] [--repo NAME ...] [--code-lang LIST]` | 검색. hybrid는 RRF fusion, citation 포함. 같은 process 안에서 동일 query (NFKC + trim + lowercase 정규화) 반복 시 in-process LRU 캐시 hit (capacity = `[search] cache_capacity`, default 256). `--no-cache` 로 강제 bypass — 디버깅용. ingest commit 발생 시 `kv['corpus_revision']` bump 으로 모든 entry 자동 stale. **`--max-tokens` / `--snippet-chars` / `--cursor` (p9-fb-34)** — agent budget controls. `--json` 출력은 `search_response.v1` wrapper (`{hits, next_cursor, truncated}`) — pre-fb-34 의 bare array 와 호환 안 됨. mismatched cursor → `error.v1.code = stale_cursor`. **filter flags (p9-fb-36):** `--tag` 는 반복 가능 flag (`--tag rust --tag async`) 로 OR 매칭, `--media` 는 `,` 구분 다중 값 OR 매칭, 나머지 flags 간은 AND 조합. `--trust-min` 은 `primary\|secondary\|generated` 중 하나 (해당 level 이상 포함). `--ingested-after` 는 RFC3339 UTC — 파싱 실패 시 `error.v1.code = config_invalid` (exit 2). `--media md` 는 `markdown` alias 로 정규화. 알 수 없는 `--media` 값은 무조건 empty hits (오류 아님). **`--trace` (p9-fb-37)** — `search_response.v1.trace` 에 lexical / vector pre-fusion 후보 + RRF union + per-stage timing (`lexical_ms` / `vector_ms` / `fusion_ms` / `total_ms`) 노출. trace 요청은 캐시 우회 (`--no-cache` 없이도 항상 cold). **`--bulk` (p9-fb-42)** — stdin ndjson 으로 N query 한 번에 실행. `--json` 면 stdout per-query ndjson (`bulk_search_item.v1`) + stderr summary (`bulk_summary: total=N succeeded=S failed=F`). Cap 100. agent 가 query decomposition 후 sub-query 일괄 실행 시 single round-trip — App instance 재사용으로 캐시 / embedder cold-start 비용 한 번만. Per-query failure 는 item 의 `error` (error.v1) 에 격리, 다른 query 계속 진행. **code corpus filters (p10-1A-1):** `--repo` 는 반복 가능 (`--repo kebab --repo other`) OR 매칭. `--code-lang` 는 반복 또는 comma 다중 값 (`--code-lang rust,python`), 알 수 없는 값은 빈 hits. `--media code` 는 Tier 1/2/3 모든 code chunk 포함. 1A-1 시점에서는 indexed 된 code chunk 가 없어 filter 가 항상 빈 결과 — 1A-2 (Rust AST chunker) 머지 이후 실효. **v0.17.0 trigram tokenizer (한국어 + 영어 동작 변경):** `chunks_fts` 가 FTS5 `trigram` 으로 동작 — 한국어 query 는 3자 이상 substring 매칭 (`해시 충돌` 같은 multi-token 도 whole-phrase 후보로 hit), 영어도 substring 매칭 (`token` 이 `tokenizer` 도 hit, recall ↑ / 단어 경계 ↓). 2자 이하 query 는 0-hit + stderr `[hint] 3자 이상 키워드 권장` + `search_response.v1.hint` 필드 (raw FTS5 mode `'...'` 제외). `kebab.sqlite` 파일 크기는 trigram index 비대화로 ~2-5배 또는 수백 MB 증가 (V007 자동 backfill, re-ingest 불필요). |
 | `kebab list docs` | 색인된 문서 목록 |
 | `kebab inspect doc <id>` / `kebab inspect chunk <id>` | raw record 보기 |
 | `kebab fetch chunk <id> [--context N]` / `kebab fetch doc <id> [--max-tokens N]` / `kebab fetch span <doc_id> <ls> <le> [--max-tokens N]` | (p9-fb-35) verbatim text fetch from indexed corpus. wire = `fetch_result.v1` (kind discriminator). chunk: target + ±N ordinal-context chunks. doc: full normalized markdown. span: 1-based line range (PDF/audio rejected as `error.v1.code = span_not_supported`). chars/4 budget on doc/span. |
-| `kebab ask "<query>" [--show-citations / --hide-citations] [--session <id>] [--stream]` | RAG 답변 + 근거 인용. 답변 후 `근거:` block 으로 full path / line range / score 한 줄씩 (default ON — `--hide-citations` 로 끄기, pipe 시 유용). 근거 부족 시 거절. Ollama 필요. `--session <id>` 로 multi-turn — 첫 호출에서 SQLite `chat_sessions` 에 자동 생성, 이후 호출은 prior turns 를 history 로 받아 follow-up. session id 는 사용자 지정 (e.g. `kb-rust-async-2026-05`) — `kebab reset --data-only` 로 모든 session wipe. **`--stream` (p9-fb-33)** 로 ndjson `answer_event.v1` event (retrieval_done → token* → final) 를 stderr 에 흘리고 stdout 마지막 줄에 기존 `answer.v1` — agent 가 token 즉시 소비 가능 |
+| `kebab ask "<query>" [--show-citations / --hide-citations] [--session <id>] [--stream] [--multi-hop]` | RAG 답변 + 근거 인용. 답변 후 `근거:` block 으로 full path / line range / score 한 줄씩 (default ON — `--hide-citations` 로 끄기, pipe 시 유용). 근거 부족 시 거절. Ollama 필요. `--session <id>` 로 multi-turn — 첫 호출에서 SQLite `chat_sessions` 에 자동 생성, 이후 호출은 prior turns 를 history 로 받아 follow-up. session id 는 사용자 지정 (e.g. `kb-rust-async-2026-05`) — `kebab reset --data-only` 로 모든 session wipe. **`--stream` (p9-fb-33)** 로 ndjson `answer_event.v1` event (retrieval_done → token* → final) 를 stderr 에 흘리고 stdout 마지막 줄에 기존 `answer.v1` — agent 가 token 즉시 소비 가능. **`--multi-hop` (v0.18.0 fb-41)** — single-pass 대신 decompose → decide → synthesize 의 N-hop loop. compound 질문 (cross-doc / prereq chain) 에 효과적. 최종 답변 후 mDeBERTa-v3 XNLI 가 `(packed_chunks, generated_answer)` entailment 검사 — `[rag] nli_threshold > 0` (default 0.0 = disabled, production 권장 0.5) 일 때 활성. entailment < threshold → `refusal_reason = "nli_verification_failed"` (LLM-self-judge ceiling 극복, S7 caffeine hallucination 같은 케이스 catch). 첫 호출 시 ~280 MB ONNX model 자동 다운로드 + RAM peak ~7-8 GB (gemma3:4b 기준). model unavailable 시 `refusal_reason = "nli_model_unavailable"`, 우회는 `[rag] nli_threshold = 0` 임시 disable. |
 | `kebab doctor` | 설정/모델/DB 헬스 체크 |
 | `kebab tui` | Ratatui 셸 (Library + Search + Ask + Inspect 패널, desktop 진행 중). Library 에서 `r` 키로 background ingest 시작 — 화면 하단 status bar 가 진행 표시, 완료/abort 시 final 라인 잠시 유지 후 자동 hide. ingest 진행 중 `Esc` / `Ctrl-C` 가 cancel signal (그 외에는 quit). vim-style mode (header 우측 `-- NORMAL --` / `-- INSERT --`) — Library/Inspect 는 자동 NORMAL, Search/Ask 는 자동 INSERT. `i` 로 Normal→Insert (모든 pane — p9-fb-21), `Esc` 로 Insert→Normal 어디서나. mode-authoritative dispatch — Search 의 `j/k/o/g`, Ask 의 `e/j/k` 는 NORMAL 모드에서만 명령으로 동작, INSERT 에서는 입력 문자로 typing. (Search 의 chunk inspect 키는 `i`→`o` 로 rebind — `i` 가 universal Insert toggle.) **`F1` 로 cheatsheet popup** (현재 pane 의 키 매핑 + global 토글 표) — `Esc` / `F1` 로 닫기. Search 패널은 200ms debounce 후 background worker 가 검색 — 키 입력으로 UI freeze 안 됨, 사용자가 계속 타이핑하면 stale 결과 자동 폐기 (generation counter). Ask 패널은 multi-turn — 같은 conversation 안에서 Q1/A1, Q2/A2 transcript 누적, 다음 질문이 이전 턴을 history 로 받아 답변. 답변 본문은 markdown 렌더 (bold/italic/inline code/heading/list/code fence/table/blockquote, raw `**bold**` 가 실제 굵게 표시). `Ctrl-L` 로 새 conversation 시작. Search 의 `g` 키가 `$EDITOR` (기본 `vi`) 로 hit 의 citation 위치 열기 — 종료 후 TUI 화면이 자동으로 깨끗이 redraw. CLI `kebab ask` 는 raw markdown 그대로 (terminal 호환성 위해). Library 의 doc-list 가 한글 / 일본어 / 중국어 (CJK) 제목을 wide-char 정확한 column width 로 truncate — 한글 제목이 한 줄을 넘기지 않음 (CJK 1 자 = 2 col). Search/Ask/Filter 입력의 cursor 가 wide char 위에서 column 단위로 정렬 — 한글 입력 시 caret 이 글자 옆에 정확히 놓임. `← / →` 로 입력 문자열 중간 cursor 이동 (한글 한 글자 = 2 column 이라도 한 번에 이동), `Home / End` 로 양 끝 점프, `Delete` 로 cursor 위치 char 삭제 — 모든 input pane (Ask / Search / Library filter overlay) 동일 (p9-fb-22). Ask 트랜스크립트는 새 답변이 viewport 아래로 누적될 때 자동으로 tail 을 따라감 (auto-scroll); `j` / `k` 로 위로 스크롤하면 freeze, `Shift-G` 로 다시 bottom + auto-tail 재개. 화면 하단 hint line 은 한국어 동사구로 (`"위로"` / `"아래로"` / `"필터"` / `"타이핑 검색어"` / `"Esc 로 NORMAL 모드"` / `"i 입력모드"` 등) + 현재 (pane, mode) 조합에 맞춰 자동 분기, **첫 fragment 가 항상 `F1 도움말`** (cheatsheet 발견성 보장). 모든 모드에서 항상 떠 있는 상태바 — `kebab v<version> │ <pane> │ <docs> docs │ <state>` (state: streaming/searching/indexing/idle, ingest 진행 중에는 progress 가 같은 자리에 흡수됨). Ask 진입 시 conversation id 8 자 prefix 도 함께 표시. Ask 트랜스크립트와 Inspect 양쪽에서 `PgUp / PgDn` 으로 10 줄씩 페이지 스크롤. Library 의 doc list 위에는 `TITLE / TAGS / UPDATED / CHUNKS` 컬럼 헤더 행 표시 (display-width 정렬, Hangul / CJK 안전). |
 | `kebab reset [--all / --data-only / --vector-only / --config-only] [--yes]` | XDG 데이터 wipe. **Irreversible.** TTY 면 confirm prompt, 아니면 `--yes` 필수. `--vector-only` 는 SQLite `embedding_records` 도 함께 truncate (orphan 방지) |
@@ -131,8 +145,8 @@ flowchart TB
    end

    subgraph Pipeline["도메인 + 파이프라인"]
-        parse["parse-md / parse-pdf / parse-image"]
-        chunker["chunker (md-heading-v1, pdf-page-v1)"]
+        parse["parse-md / parse-pdf / parse-image / parse-code"]
+        chunker["chunker (md-heading-v1, pdf-page-v1, code-{rust,python,ts,js,go,java,kotlin,c,cpp}-ast-v1, k8s-manifest-resource-v1, dockerfile-file-v1, manifest-file-v1, code-text-paragraph-v1)"]
        embedder["embedder (fastembed multilingual-e5-large)"]
        retriever["retriever (lexical / vector / hybrid RRF)"]
        rag["RAG pipeline"]
@@ -184,6 +198,11 @@ flowchart TB
    - `dimensions` (default `1024`) — 모델의 embedding 차원. config 와 LanceDB stored dim 불일치 시 검색 결과 0 건 (orphan table). 모델 변경 시 `kebab reset --vector-only && kebab ingest` 로 vector index 재구축 권장.
  - `[ui] theme = "dark" | "light"` 로 TUI 팔레트 선택 (default `"dark"`, 알 수 없는 값은 dark fallback). 
  - `[search] stale_threshold_days = 30` (p9-fb-32) — search hit / RAG citation 의 `stale` 플래그 기준 (default 30 일, `0` 으로 비활성화). 옛 config 의 `workspace.include = [...]` 은 silently 무시 + 단발 deprecation warning (p9-fb-25).
+- `[ingest.code]` (p10-1A-1) — code ingest 의 skip 정책 + chunker 기본값.
+  - `skip_generated_header = true` — 첫 ~512 byte 의 generated marker (`@generated` / `DO NOT EDIT` 등) 감지 시 skip.
+  - `max_file_bytes = 262144` (256 KiB) / `max_file_lines = 5000` — 파일당 cap, 초과 시 skip.
+  - `extra_skip_globs = []` — 사용자 추가 skip 패턴 (`.gitignore` 문법).
+  - `.gitignore` honor: 자동 적용. `.kebabignore` 는 추가 layer. 우선순위: built-in safety net (`node_modules/` / `target/` / `__pycache__/` / `.venv/` / `venv/` / `env/`) > `.gitignore` > `.kebabignore`.
 - `[rag] prompt_template_version` (default `"rag-v2"`) — RAG system prompt version. `"rag-v1"` 은 legacy backwards-compat (사용자 명시 시 유지). v2 강화 규칙: (1) fact 인용 시 [#번호] 앞에 chunk 속 원문 큰따옴표 표기, (2) 학습 지식 동원 금지, (3) 근거 모호 시 "확실하지 않다" 명시.
 - `--config <path>` flag — 임시 워크스페이스 / 격리 테스트 시 사용. CLI / TUI 모두 honor.
 - `KEBAB_*` env — 일부 키 override (`KEBAB_RAG_SCORE_GATE`, `KEBAB_EVAL_GOLDEN`, `KEBAB_COMMIT_HASH` 등).
--- a/crates/kebab-app/Cargo.toml
+++ b/crates/kebab-app/Cargo.toml
@@ -12,8 +12,6 @@ kebab-core = { path = "../kebab-core" }
 kebab-config = { path = "../kebab-config" }
 kebab-source-fs = { path = "../kebab-source-fs" }
 kebab-parse-md = { path = "../kebab-parse-md" }
-kebab-parse-types = { path = "../kebab-parse-types" }
-kebab-normalize = { path = "../kebab-normalize" }
 kebab-chunk = { path = "../kebab-chunk" }
 kebab-store-sqlite = { path = "../kebab-store-sqlite" }
 kebab-store-vector = { path = "../kebab-store-vector" }
@@ -23,6 +21,11 @@ kebab-embed-local = { path = "../kebab-embed-local" }
 kebab-llm = { path = "../kebab-llm" }
 kebab-llm-local = { path = "../kebab-llm-local" }
 kebab-rag = { path = "../kebab-rag" }
+# p9-fb-41 PR-9c-2: facade construction of OnnxNliVerifier when
+# `[rag] nli_threshold > 0`. Trait-only consumption via kebab-rag's
+# `with_verifier`; no kebab-nli internals leak into kebab-app code
+# beyond the construction site in `open_with_config`.
+kebab-nli = { path = "../kebab-nli" }
 # P6-4: image extractor + OCR + caption adapters live here. App
 # threads them into the per-asset dispatch (see `ingest_one_asset`
 # image branch). Trait-only consumption — no `kebab-parse-image`
@@ -32,6 +35,10 @@ kebab-parse-image = { path = "../kebab-parse-image" }
 # per-asset dispatch (see `ingest_one_asset` PDF branch) and runs the
 # resulting `CanonicalDocument` through `kebab-chunk::PdfPageV1Chunker`.
 kebab-parse-pdf = { path = "../kebab-parse-pdf" }
+# p10-1A-2: Rust AST extractor lives here. App threads it into the
+# per-asset dispatch (see `ingest_one_asset` Code branch) and runs the
+# resulting `CanonicalDocument` through `kebab-chunk::CodeRustAstV1Chunker`.
+kebab-parse-code = { path = "../kebab-parse-code" }
 anyhow               = { workspace = true }
 blake3               = { workspace = true }
 serde                = { workspace = true }
@@ -72,3 +79,6 @@ lopdf                = "0.32"
 # error_wire::tests::llm_unreachable_classifies_to_model_unreachable needs a real
 # reqwest::Error (private constructor) — built from a connect-refused call.
 reqwest      = { version = "0.12", default-features = false, features = ["blocking", "rustls-tls"] }
+
+[lints]
+workspace = true
--- a/crates/kebab-app/src/app.rs
+++ b/crates/kebab-app/src/app.rs
@@ -40,11 +40,18 @@ use anyhow::{Context, Result, anyhow};
 use lru::LruCache;

 use kebab_core::{
-    Answer, Embedder, IndexVersion, LanguageModel, Retriever, SearchHit, SearchMode,
-    SearchOpts, SearchQuery, VectorStore,
+    Answer, DocumentStore, Embedder, ExtractContext, Extractor, IndexVersion, LanguageModel,
+    MediaType, Retriever, SearchHit, SearchMode, SearchOpts, SearchQuery, VectorStore,
 };
 use kebab_embed_local::FastembedEmbedder;
 use kebab_llm_local::OllamaLanguageModel;
+use kebab_parse_code::{
+    CAstExtractor, CppAstExtractor, GoAstExtractor, JavaAstExtractor,
+    JavascriptAstExtractor, KotlinAstExtractor, PythonAstExtractor, RustAstExtractor,
+    TypescriptAstExtractor,
+};
+use kebab_parse_image::ImageExtractor;
+use kebab_parse_pdf::PdfTextExtractor;
 use kebab_rag::{AskOpts, RagPipeline};
 use kebab_search::{HybridRetriever, LexicalRetriever, VectorRetriever};
 use kebab_store_sqlite::SqliteStore;
@@ -73,6 +80,37 @@ pub struct SearchResponse {
    /// p9-fb-37: present when caller passed `SearchOpts.trace = true`.
    /// Consumers that ignore trace should leave this `None`.
    pub trace: Option<kebab_core::SearchTrace>,
+    /// v0.17.0 A5 Step 4b: human / agent-readable advisory string set
+    /// when the empty hit list is likely due to a query shorter than the
+    /// FTS5 trigram tokenizer's 3-char minimum. `None` otherwise. CLI
+    /// surfaces it on stderr (text mode); MCP / `--json` consumers
+    /// surface it however they prefer. See
+    /// `docs/superpowers/specs/2026-05-22-korean-trigram-tokenizer-design.md`
+    /// §3.3.
+    pub hint: Option<String>,
+}
+
+/// v0.17.0 A5 Step 4b: decide whether to attach a "3자 이상 키워드 권장"
+/// hint to a `SearchResponse`. Fires only when the result set is empty
+/// *and* the trimmed query is shorter than the trigram tokenizer can
+/// resolve. Raw FTS5 mode (`'...'`) opts out — the user explicitly
+/// invoked FTS5 syntax. Identical condition powers the CLI stderr line
+/// and (separately) the TUI status bar.
+pub fn short_query_hint(query_text: &str, hits_empty: bool) -> Option<String> {
+    if !hits_empty {
+        return None;
+    }
+    let trimmed = query_text.trim();
+    let bytes = trimmed.as_bytes();
+    // Raw single-quote mode: user opted into FTS5 syntax, no advisory.
+    if bytes.len() >= 2 && bytes[0] == b'\'' && bytes[bytes.len() - 1] == b'\'' {
+        return None;
+    }
+    if trimmed.chars().count() < 3 {
+        Some("3자 이상 키워드 권장 (trigram tokenizer 제약)".to_string())
+    } else {
+        None
+    }
 }

 /// Facade state — see module docs for lifetime rules.
@@ -84,6 +122,12 @@ pub struct SearchResponse {
 pub struct App {
    pub(crate) config: kebab_config::Config,
    pub(crate) sqlite: Arc<SqliteStore>,
+    /// post-v0.18.0 extractor-dispatch-unification: polymorphic Extractor
+    /// registry. App init 시 1회 등록되어 `extract_for(...)` 가 lookup
+    /// 한다. 현재 11 entry (ImageExtractor + PdfTextExtractor + 9 AST).
+    /// MarkdownExtractor 는 별 PR 에서 추가 — markdown ingest path 는
+    /// 본 PR 에서 free-function 그대로 유지.
+    pub(crate) extractors: Vec<Box<dyn Extractor + Send + Sync>>,
    /// Memoized embedder — built lazily on first `embedder()` call when
    /// embeddings are enabled. `OnceLock` keeps the struct `Sync` and
    /// the build path cold-only-once.
@@ -102,6 +146,17 @@ pub struct App {
    /// `corpus_revision` snapshot embedded in `SearchCacheKey`
    /// invalidates every entry the moment a new ingest commit lands.
    search_cache: Option<Mutex<LruCache<SearchCacheKey, Vec<SearchHit>>>>,
+    /// p9-fb-41 PR-9c-2: NLI verifier built eagerly at
+    /// `open_with_config` time when `config.rag.nli_threshold > 0`,
+    /// consumed by `RagPipeline::with_verifier` on every `ask` /
+    /// `ask_with_session` call. `None` when the gate is disabled
+    /// (default, threshold = 0) — multi-hop skips step 8.5 entirely
+    /// and single-pass never touches the verifier.
+    ///
+    /// Built eagerly (not lazy) so the `open_with_config` `?`
+    /// propagation surfaces NLI model construction errors at App
+    /// boot time, before any user query runs.
+    pipeline_verifier: Option<Arc<dyn kebab_nli::NliVerifier>>,
 }

 /// p9-fb-19: cache key for `App::search`. Includes every field that
@@ -162,16 +217,76 @@ impl App {
        // `None` (cache disabled — every search hits the retrievers).
        let search_cache = NonZeroUsize::new(config.search.cache_capacity)
            .map(|cap| Mutex::new(LruCache::new(cap)));
+        // post-v0.18.0 extractor-dispatch-unification: build the 11-entry
+        // Extractor registry. All entries are state-less unit structs with
+        // zero-cost `new()`, so init cost is effectively 0 and side effects
+        // are 0 — `pipeline_verifier` fallible `?` below may bail but the
+        // already-constructed `extractors` Vec drops without cost. Markdown
+        // is NOT registered (see field doc).
+        let extractors: Vec<Box<dyn Extractor + Send + Sync>> = vec![
+            Box::new(ImageExtractor::new()),
+            Box::new(PdfTextExtractor::new()),
+            Box::new(RustAstExtractor::new()),
+            Box::new(PythonAstExtractor::new()),
+            Box::new(TypescriptAstExtractor::new()),
+            Box::new(JavascriptAstExtractor::new()),
+            Box::new(GoAstExtractor::new()),
+            Box::new(JavaAstExtractor::new()),
+            Box::new(KotlinAstExtractor::new()),
+            Box::new(CAstExtractor::new()),
+            Box::new(CppAstExtractor::new()),
+        ];
+        // p9-fb-41 PR-9c-2: build the NLI verifier when the gate is
+        // enabled. App carries it on `RagPipeline` via
+        // `with_verifier` so the rag crate doesn't have to know about
+        // kebab-nli construction. Failure (`?`) surfaces as a user-
+        // facing error at App boot — never a panic in the pipeline's
+        // `expect("verifier must be Some when nli_threshold > 0.0")`.
+        let pipeline_verifier: Option<Arc<dyn kebab_nli::NliVerifier>> =
+            if config.rag.nli_threshold > 0.0 {
+                let v = kebab_nli::OnnxNliVerifier::new(&config).context(
+                    "kebab-app: construct OnnxNliVerifier (config.rag.nli_threshold > 0)",
+                )?;
+                Some(Arc::new(v))
+            } else {
+                None
+            };
        Ok(Self {
            config,
            sqlite: Arc::new(sqlite),
+            extractors,
            embedder: OnceLock::new(),
            vector: OnceLock::new(),
            llm: OnceLock::new(),
            search_cache,
+            pipeline_verifier,
        })
    }

+    /// Polymorphic dispatcher for the [`Extractor`] trait. Looks up the
+    /// first Extractor whose `supports(media)` returns true and invokes
+    /// `extract(ctx, bytes)` on it.
+    ///
+    /// Errors with `anyhow!("no Extractor for media_type {media:?}")`
+    /// when no matching Extractor is registered. Callers in
+    /// `ingest_one_*_asset` reach this only after the outer 4-arm
+    /// dispatch (`MediaType::Markdown` / `Image` / `Pdf` / `Code(lang)`)
+    /// has matched, so a miss is a programming error — NOT a user-
+    /// facing skip.
+    pub(crate) fn extract_for(
+        &self,
+        media: &MediaType,
+        ctx: &ExtractContext<'_>,
+        bytes: &[u8],
+    ) -> Result<kebab_core::CanonicalDocument> {
+        let extractor = self
+            .extractors
+            .iter()
+            .find(|e| e.supports(media))
+            .ok_or_else(|| anyhow!("no Extractor for media_type {media:?}"))?;
+        extractor.extract(ctx, bytes)
+    }
+
    /// Run a [`SearchQuery`] through the configured retriever stack and
    /// return the top-k hits. p9-fb-19: result is served from the
    /// in-process LRU cache when the same `(query_norm, mode, k,
@@ -235,7 +350,7 @@ impl App {
        // so other in-flight searches can use the cache concurrently.
        drop(guard);
        let hits = self.search_uncached(query)?;
-        let mut guard = cache.lock().unwrap_or_else(|e| e.into_inner());
+        let mut guard = cache.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
        guard.put(key, hits.clone());
        Ok(hits)
    }
@@ -296,6 +411,15 @@ impl App {
            now,
            self.config.search.stale_threshold_days,
        );
+        // p10-1A-2: backfill `code_lang` from the Citation::Code `lang`
+        // field. The search layer (kebab-search) constructs SearchHit with
+        // `code_lang: None`; we own the post-processing here in kebab-app
+        // and can fill it cheaply from data already present in the hit.
+        backfill_code_lang(&mut hits);
+        // p10-1A-2 Task 8b: backfill `repo` from the document's
+        // `Metadata.repo`. Unlike `code_lang`, this cannot be derived from
+        // the Citation alone — it requires a store lookup by `doc_id`.
+        self.backfill_repo(&mut hits);
        Ok(hits)
    }

@@ -387,6 +511,10 @@ impl App {
                now,
                self.config.search.stale_threshold_days,
            );
+            // p10-1A-2: backfill code_lang — same as search_uncached.
+            backfill_code_lang(&mut traced_hits);
+            // p10-1A-2 Task 8b: backfill repo — same as search_uncached.
+            self.backfill_repo(&mut traced_hits);

            // Apply offset + k_effective truncation (mirrors non-trace path).
            let drop_n = offset.min(traced_hits.len());
@@ -396,7 +524,7 @@ impl App {

            // Snippet truncation if opts.snippet_chars set (mirror non-trace path).
            if opts.snippet_chars.is_some() {
-                for h in hits.iter_mut() {
+                for h in &mut hits {
                    if h.snippet.chars().count() > snippet_chars {
                        h.snippet = trim_to_chars(&h.snippet, snippet_chars);
                    }
@@ -405,14 +533,19 @@ impl App {

            // Trace path skips the budget loop. Caller will inspect
            // `hits.len()` and `trace.timing` rather than paginate.
+            let hint = short_query_hint(&query.text, hits.is_empty());
            return Ok(SearchResponse {
                hits,
                next_cursor: None,
                truncated: false,
                trace: Some(trace),
+                hint,
            });
        }

+        // backfill_code_lang + backfill_repo are applied inside `search`
+        // via `search_uncached` — no explicit call needed here. Trace
+        // branch above calls them directly because it bypasses `search`.
        let mut all_hits = self.search(fetch_query)?;

        // Skip offset.
@@ -426,7 +559,7 @@ impl App {
        // `config.search.snippet_chars`; this only kicks in when the
        // caller asked for *less*).
        if opts.snippet_chars.is_some() {
-            for h in hits.iter_mut() {
+            for h in &mut hits {
                if h.snippet.chars().count() > snippet_chars {
                    h.snippet = trim_to_chars(&h.snippet, snippet_chars);
                }
@@ -445,7 +578,7 @@ impl App {
            {
                current_snippet_cap =
                    (current_snippet_cap / 2).max(SNIPPET_FLOOR);
-                for h in hits.iter_mut() {
+                for h in &mut hits {
                    if h.snippet.chars().count() > current_snippet_cap {
                        h.snippet =
                            trim_to_chars(&h.snippet, current_snippet_cap);
@@ -489,11 +622,13 @@ impl App {
            None
        };

+        let hint = short_query_hint(&query.text, hits.is_empty());
        Ok(SearchResponse {
            hits,
            next_cursor,
            truncated,
            trace: None,
+            hint,
        })
    }

@@ -502,9 +637,26 @@ impl App {
    pub fn ask(&self, query: &str, opts: AskOpts) -> Result<Answer> {
        let retriever = self.build_retriever(opts.mode)?;
        let llm = self.llm()?;
+        let pipeline = self.build_pipeline(retriever, llm);
+        pipeline.ask(query, opts)
+    }
+
+    /// p9-fb-41 PR-9c-2: shared pipeline builder used by [`Self::ask`]
+    /// and [`Self::ask_with_session`]. Attaches the App-built NLI
+    /// verifier (when `cfg.rag.nli_threshold > 0`) via
+    /// `RagPipeline::with_verifier`, keeping the construction site in
+    /// a single place so the two call paths can't drift.
+    fn build_pipeline(
+        &self,
+        retriever: Arc<dyn Retriever>,
+        llm: Arc<dyn LanguageModel>,
+    ) -> RagPipeline {
        let pipeline =
            RagPipeline::new(self.config.clone(), retriever, llm, self.sqlite.clone());
-        pipeline.ask(query, opts)
+        match &self.pipeline_verifier {
+            Some(v) => pipeline.with_verifier(v.clone()),
+            None => pipeline,
+        }
    }

    /// p9-fb-18: shared retriever-stack builder used by [`Self::ask`]
@@ -609,10 +761,11 @@ impl App {

        // p9-fb-18 R1: shared retriever builder removes the prior
        // copy of `ask`'s 35-line stack — see [`Self::build_retriever`].
+        // p9-fb-41 PR-9c-2: shared `build_pipeline` attaches the NLI
+        // verifier when the gate is enabled.
        let retriever = self.build_retriever(opts.mode)?;
        let llm = self.llm()?;
-        let pipeline =
-            RagPipeline::new(self.config.clone(), retriever, llm, self.sqlite.clone());
+        let pipeline = self.build_pipeline(retriever, llm);
        let answer = pipeline.ask_with_history(
            query,
            history,
@@ -772,11 +925,63 @@ impl App {
    /// clear` admin command). No-op when the cache is disabled.
    pub fn clear_search_cache(&self) {
        if let Some(cache) = self.search_cache.as_ref() {
-            let mut guard = cache.lock().unwrap_or_else(|e| e.into_inner());
+            let mut guard = cache.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
            guard.clear();
        }
    }

+    /// p10-1A-2 Task 8b: back-fill `SearchHit.repo` from the originating
+    /// document's `Metadata.repo` for every hit whose `repo` field is
+    /// currently `None`. The search layer (kebab-search) constructs hits
+    /// with `repo: None` because it has no store access; we fill it here
+    /// in kebab-app post-retrieval via a per-distinct-`doc_id` store lookup.
+    ///
+    /// Deduplication: a small `HashMap` accumulates the
+    /// `(doc_id → Option<String>)` mapping so each unique document is
+    /// fetched at most once. Search result sets are small (default k ≤ 20),
+    /// so the map overhead is negligible. A `None` entry is cached too
+    /// (document not found or no repo in metadata) to avoid re-querying.
+    ///
+    /// Non-repo documents (markdown, PDF, plain text, code files outside a
+    /// git tree) correctly keep `repo: None` — `Metadata.repo` is already
+    /// `None` for those, so the assignment is a no-op.
+    fn backfill_repo(&self, hits: &mut [SearchHit]) {
+        use std::collections::HashMap;
+        use kebab_core::DocumentId;
+
+        // doc_id → Option<String> where None means "not found / no repo"
+        let mut cache: HashMap<DocumentId, Option<String>> = HashMap::new();
+
+        for hit in hits.iter_mut() {
+            if hit.repo.is_some() {
+                continue;
+            }
+            let repo_val = cache
+                .entry(hit.doc_id.clone())
+                .or_insert_with(|| {
+                    // Deliberately non-aborting: a failed store lookup for
+                    // one hit must not abort the whole search response. Log
+                    // the error so it's observable rather than silently
+                    // dropped (review #140 round 1).
+                    match self.sqlite.get_document(&hit.doc_id) {
+                        Ok(opt) => opt.and_then(|doc| doc.metadata.repo),
+                        Err(e) => {
+                            tracing::warn!(
+                                target: "kebab-app",
+                                doc_id = %hit.doc_id,
+                                error = %e,
+                                "backfill_repo: get_document failed; leaving hit.repo = None"
+                            );
+                            None
+                        }
+                    }
+                });
+            if let Some(r) = repo_val {
+                hit.repo = Some(r.clone());
+            }
+        }
+    }
+
    /// Resolve the embedder + vector store, surfacing the user-friendly
    /// "switch to --mode lexical" error when embeddings are disabled.
    fn require_embeddings(
@@ -896,6 +1101,21 @@ fn estimate_chars(hits: &[SearchHit]) -> usize {
        .sum()
 }

+/// p10-1A-2: back-fill `SearchHit.code_lang` from `Citation::Code.lang`
+/// for every code hit in the list. The search layer (kebab-search)
+/// constructs hits with `code_lang: None`; we fill it here in kebab-app
+/// post-retrieval so callers see the correct language identifier without
+/// requiring a second SQL query.
+fn backfill_code_lang(hits: &mut [SearchHit]) {
+    for hit in hits.iter_mut() {
+        if let kebab_core::Citation::Code { lang, .. } = &hit.citation {
+            if hit.code_lang.is_none() {
+                hit.code_lang = lang.clone();
+            }
+        }
+    }
+}
+
 #[cfg(test)]
 mod tests {
    use super::*;
@@ -994,3 +1214,128 @@ mod tests_trace {
        assert!(resp.trace.is_some(), "trace populated when opts.trace=true");
    }
 }
+
+/// post-v0.18.0 extractor-dispatch-unification: in-crate unit tests for
+/// the `App.extractors` registry + `App::extract_for` polymorphic
+/// dispatch. In-crate (not `tests/`) because `extractors` + `extract_for`
+/// are `pub(crate)` — integration tests cannot reach them.
+///
+/// Spec §5.1 + plan §2 Step 10 — 3 test class:
+/// 1. registry length = 11 (image + pdf + 9 AST).
+/// 2. mutually-exclusive `supports()` grid over 16 sample MediaTypes.
+/// 3. `extract_for` returns `Err("no Extractor ...")` for registry-NOT-cover
+///    MediaType (Audio).
+#[cfg(test)]
+mod tests_extractor_dispatch {
+    use super::*;
+    use kebab_core::{AudioType, ExtractConfig, ImageType};
+
+    /// helper: tempdir-isolated App for tests (mirrors `tests_trace`'s
+    /// `open_app_with_temp_dir` pattern).
+    fn open_app_with_temp_dir() -> (tempfile::TempDir, App) {
+        let dir = tempfile::tempdir().unwrap();
+        let mut cfg = kebab_config::Config::defaults();
+        cfg.storage.data_dir = dir.path().to_string_lossy().into_owned();
+        // Bring up migrations.
+        let store = kebab_store_sqlite::SqliteStore::open(&cfg).unwrap();
+        store.run_migrations().unwrap();
+        drop(store);
+        let app = App::open_with_config(cfg).unwrap();
+        (dir, app)
+    }
+
+    /// Registry length invariant: 11 Extractor (image + pdf + 9 AST).
+    /// Markdown is NOT registered (free-function path — defer to a
+    /// separate PR per spec §3.4).
+    #[test]
+    fn registry_has_eleven_extractors() {
+        let (_dir, app) = open_app_with_temp_dir();
+        assert_eq!(
+            app.extractors.len(),
+            11,
+            "registry must hold 11 Extractors (image + pdf + 9 AST). \
+             markdown 은 별 PR."
+        );
+    }
+
+    /// 11 Extractor 의 `supports()` 가 16 sample MediaType 에 대해
+    /// mutually exclusive — 어떤 두 Extractor 도 동일 MediaType 에
+    /// 대해 true 반환 안 됨.
+    #[test]
+    fn supports_grid_is_mutually_exclusive() {
+        let (_dir, app) = open_app_with_temp_dir();
+        let samples = vec![
+            MediaType::Markdown,
+            MediaType::Pdf,
+            MediaType::Image(ImageType::Png),
+            MediaType::Image(ImageType::Jpeg),
+            MediaType::Code("rust".into()),
+            MediaType::Code("python".into()),
+            MediaType::Code("typescript".into()),
+            MediaType::Code("javascript".into()),
+            MediaType::Code("go".into()),
+            MediaType::Code("java".into()),
+            MediaType::Code("kotlin".into()),
+            MediaType::Code("c".into()),
+            MediaType::Code("cpp".into()),
+            MediaType::Code("yaml".into()),  // registry NOT cover
+            MediaType::Code("shell".into()), // registry NOT cover
+            MediaType::Audio(AudioType::Wav), // registry NOT cover
+        ];
+        for sample in &samples {
+            let hits: Vec<_> = app
+                .extractors
+                .iter()
+                .filter(|e| e.supports(sample))
+                .collect();
+            assert!(
+                hits.len() <= 1,
+                "mutually exclusive violated for {sample:?}: {} hits",
+                hits.len()
+            );
+        }
+    }
+
+    /// `extract_for` 가 registry NOT cover MediaType (Audio) 에 대해
+    /// `Err("no Extractor for media_type ...")` 반환. Audio MediaType
+    /// 사용으로 RawAsset 의 actual content 의존 회피 — registry NOT
+    /// cover → 즉시 Err.
+    #[test]
+    fn extract_for_unsupported_media_errors() {
+        let (_dir, app) = open_app_with_temp_dir();
+
+        // Minimal RawAsset. Actual content never read — Audio MediaType
+        // 는 registry NOT cover → `extract_for` 가 dispatch loop 안에서
+        // 바로 Err 반환. RawAsset field set 은 `crates/kebab-core/src/
+        // asset.rs:62-73` 와 정합 (8 field).
+        let asset = kebab_core::RawAsset {
+            asset_id: kebab_core::AssetId("00".repeat(16)),
+            source_uri: kebab_core::SourceUri::File("/tmp/dummy.wav".into()),
+            workspace_path: kebab_core::WorkspacePath("dummy.wav".to_string()),
+            media_type: MediaType::Audio(AudioType::Wav),
+            byte_len: 0,
+            checksum: kebab_core::Checksum("00".repeat(32)),
+            discovered_at: time::OffsetDateTime::now_utc(),
+            // AssetStorage::Inline 미존재 — actual variant `Copied { path }`
+            // 사용 (kebab-core/src/asset.rs:55-60).
+            stored: kebab_core::AssetStorage::Copied {
+                path: std::path::PathBuf::from("/tmp/dummy.wav"),
+            },
+        };
+
+        let workspace_root: std::path::PathBuf = std::path::PathBuf::from("/tmp");
+        let cfg = ExtractConfig::default();
+        let ctx = ExtractContext {
+            asset: &asset,
+            workspace_root: &workspace_root,
+            config: &cfg,
+        };
+        let result = app.extract_for(&MediaType::Audio(AudioType::Wav), &ctx, &[]);
+        assert!(result.is_err(), "Audio 는 registry 미포함 → Err 기대");
+        let err_msg = format!("{:#}", result.unwrap_err());
+        assert!(
+            err_msg.contains("no Extractor"),
+            "unexpected err: {err_msg}"
+        );
+    }
+}
--- a/crates/kebab-app/src/bulk.rs
+++ b/crates/kebab-app/src/bulk.rs
@@ -96,6 +96,11 @@ fn serialize_search_response(r: &SearchResponse) -> Value {
            None => Value::Null,
        };
        map.insert("trace".to_string(), trace_v);
+        // v0.17.0 A5 Step 4b: only emit `hint` when set — matches
+        // the CLI wire wrapper's additive emit pattern.
+        if let Some(hint) = &r.hint {
+            map.insert("hint".to_string(), Value::String(hint.clone()));
+        }
    }
    v
 }
@@ -134,9 +139,8 @@ fn parse_one(raw: &Value) -> Result<(SearchQuery, SearchOpts), String> {

    let k = obj
        .get("k")
-        .and_then(|v| v.as_u64())
-        .map(|n| n as usize)
-        .unwrap_or(0); // 0 → use config default in app
+        .and_then(serde_json::Value::as_u64)
+        .map_or(0, |n| n as usize); // 0 → use config default in app

    let trust_min = match obj.get("trust_min").and_then(|v| v.as_str()) {
        None => None,
@@ -197,19 +201,21 @@ fn parse_one(raw: &Value) -> Result<(SearchQuery, SearchOpts), String> {
        media,
        ingested_after,
        doc_id,
+        repo: vec![],
+        code_lang: vec![],
    };

    let opts = SearchOpts {
        max_tokens: obj
            .get("max_tokens")
-            .and_then(|v| v.as_u64())
+            .and_then(serde_json::Value::as_u64)
            .map(|n| n as usize),
        snippet_chars: obj
            .get("snippet_chars")
-            .and_then(|v| v.as_u64())
+            .and_then(serde_json::Value::as_u64)
            .map(|n| n as usize),
        cursor: obj.get("cursor").and_then(|v| v.as_str()).map(String::from),
-        trace: obj.get("trace").and_then(|v| v.as_bool()).unwrap_or(false),
+        trace: obj.get("trace").and_then(serde_json::Value::as_bool).unwrap_or(false),
    };

    Ok((
--- a/crates/kebab-app/src/error_wire.rs
+++ b/crates/kebab-app/src/error_wire.rs
@@ -91,7 +91,7 @@ pub fn classify(err: &anyhow::Error, verbose: bool) -> ErrorV1 {
    }
    let mut details = json!({});
    if verbose {
-        let chain: Vec<String> = err.chain().map(|c| c.to_string()).collect();
+        let chain: Vec<String> = err.chain().map(std::string::ToString::to_string).collect();
        details = json!({"chain": chain});
    }
    ErrorV1 {
--- a/crates/kebab-app/src/external.rs
+++ b/crates/kebab-app/src/external.rs
@@ -50,7 +50,7 @@ pub fn ensure_kebabignore_entry(workspace_root: &Path) -> Result<()> {
    if !existing.is_empty() && !existing.ends_with('\n') {
        file.write_all(b"\n")?;
    }
-    writeln!(file, "{}", KEBABIGNORE_LINE)?;
+    writeln!(file, "{KEBABIGNORE_LINE}")?;
    Ok(())
 }

--- a/crates/kebab-app/src/fetch.rs
+++ b/crates/kebab-app/src/fetch.rs
@@ -189,10 +189,12 @@ fn fetch_span(
    // (markdown / note / paper / reference / inbox) is the *user-facing*
    // category, not the rendering format — the actual byte-level format
    // lives on the source `RawAsset.media_type`. Look it up via
-    // workspace_path (unique key per asset).
-    if let Some(asset) = <kebab_store_sqlite::SqliteStore as DocumentStore>::get_asset_by_workspace_path(
+    // doc.source_asset_id (PRIMARY KEY) so twin files (identical content
+    // at different paths) always read *this* document's own asset row,
+    // not whichever twin last wrote `assets.workspace_path`.
+    if let Some(asset) = <kebab_store_sqlite::SqliteStore as DocumentStore>::get_asset(
        &app.sqlite,
-        &doc.workspace_path,
+        &doc.source_asset_id,
    )? {
        if matches!(
            asset.media_type,
--- a/crates/kebab-app/src/ingest_progress.rs
+++ b/crates/kebab-app/src/ingest_progress.rs
@@ -96,6 +96,7 @@ pub fn media_label(media: &kebab_core::MediaType) -> &'static str {
        kebab_core::MediaType::Pdf => "pdf",
        kebab_core::MediaType::Image(_) => "image",
        kebab_core::MediaType::Audio(_) => "audio",
+        kebab_core::MediaType::Code(_) => "code",
        kebab_core::MediaType::Other(_) => "other",
    }
 }
@@ -148,6 +149,7 @@ mod tests {
            media_label(&MediaType::Audio(kebab_core::AudioType::Wav)),
            "audio"
        );
+        assert_eq!(media_label(&MediaType::Code("rust".into())), "code");
        assert_eq!(media_label(&MediaType::Other("x".into())), "other");
    }

@@ -164,8 +166,8 @@ mod tests {
        };
        let v = serde_json::to_value(&ev).unwrap();
        assert_eq!(v.get("kind").and_then(|s| s.as_str()), Some("asset_started"));
-        assert_eq!(v.get("idx").and_then(|n| n.as_u64()), Some(1));
-        assert_eq!(v.get("total").and_then(|n| n.as_u64()), Some(10));
+        assert_eq!(v.get("idx").and_then(serde_json::Value::as_u64), Some(1));
+        assert_eq!(v.get("total").and_then(serde_json::Value::as_u64), Some(10));
        assert_eq!(v.get("path").and_then(|s| s.as_str()), Some("notes/foo.md"));
        assert_eq!(v.get("media").and_then(|s| s.as_str()), Some("markdown"));
    }
@@ -182,8 +184,8 @@ mod tests {
        let v = serde_json::to_value(&ev).unwrap();
        assert_eq!(v.get("kind").and_then(|s| s.as_str()), Some("completed"));
        let counts = v.get("counts").unwrap();
-        assert_eq!(counts.get("scanned").and_then(|n| n.as_u64()), Some(5));
-        assert_eq!(counts.get("new").and_then(|n| n.as_u64()), Some(2));
+        assert_eq!(counts.get("scanned").and_then(serde_json::Value::as_u64), Some(5));
+        assert_eq!(counts.get("new").and_then(serde_json::Value::as_u64), Some(2));
    }

    #[test]
--- a/crates/kebab-app/src/lib.rs
+++ b/crates/kebab-app/src/lib.rs
--- a/crates/kebab-app/src/reset.rs
+++ b/crates/kebab-app/src/reset.rs
@@ -9,13 +9,19 @@
 //!
 //! `--vector-only` additionally truncates `embedding_records` in SQLite
 //! so the next `kebab ingest` re-embeds cleanly without orphan rows.
+//!
+//! `--orphans-only` purges stored docs that are outside the current walker
+//! scope (config narrowing / removed sub-directory). No filesystem paths are
+//! removed — this is purely a store-level reconciliation.

+use std::collections::HashSet;
 use std::path::PathBuf;

 use anyhow::{Context, Result};
 use serde::{Deserialize, Serialize};

 use kebab_config::{Config, expand_path};
+use kebab_core::WorkspacePath;

 /// What the user asked to remove. Mutually exclusive — picked by the CLI
 /// from a clap `ArgGroup`.
@@ -32,6 +38,13 @@ pub enum ResetScope {
    VectorOnly,
    /// Wipe only the config dir.
    ConfigOnly,
+    /// Purge stored docs that are outside the current walker scope (no
+    /// filesystem paths are removed). Filesystem existence is NOT checked —
+    /// anything the current walker would not visit is considered an orphan.
+    /// The explicit complement to the conservative `sweep_deleted_files`
+    /// that runs during ingest (which leaves on-disk-but-out-of-scope docs
+    /// alone for data safety).
+    OrphansOnly,
 }

 /// Result of a successful wipe — emitted as `reset_report.v1` by the
@@ -41,6 +54,16 @@ pub struct ResetReport {
    pub scope: ResetScope,
    pub removed_paths: Vec<PathBuf>,
    pub embedding_rows_truncated: u64,
+    /// Number of stored docs purged because they are outside the current
+    /// walker scope. Non-zero only when `scope == OrphansOnly`.
+    /// `#[serde(default)]` preserves back-compat with older callers that
+    /// do not include this field.
+    #[serde(default)]
+    pub orphans_purged: u32,
+    /// Paths of the orphaned docs that were purged. Sorted for deterministic
+    /// output. Non-empty only when `scope == OrphansOnly`.
+    #[serde(default)]
+    pub purged_paths: Vec<WorkspacePath>,
 }

 /// Compute the absolute on-disk paths a given scope will wipe, given a
@@ -67,6 +90,10 @@ pub fn enumerate_paths(scope: ResetScope, cfg: &Config) -> Vec<PathBuf> {
            vec![vector_dir]
        }
        ResetScope::ConfigOnly => vec![cfg_dir],
+        // OrphansOnly operates purely at the store level — no filesystem paths
+        // are removed. Return empty so `estimate_size_bytes` stays zero and
+        // the existing confirm UI path for directory wipes is skipped.
+        ResetScope::OrphansOnly => vec![],
    }
 }

@@ -96,16 +123,82 @@ pub fn estimate_size_bytes(paths: &[PathBuf]) -> u64 {
    paths.iter().map(|p| walk(p)).sum()
 }

+/// Compute the workspace paths stored in SQLite that are NOT visited by
+/// the current walker scope (i.e. they are "orphans" — on disk but
+/// outside the configured include/exclude rules, or from a sub-directory
+/// that has since been removed from the workspace).
+///
+/// Does NOT check filesystem existence — `OrphansOnly` is the explicit
+/// "I know what I'm doing" variant; callers that want the conservative
+/// fs-aware sweep should use `sweep_deleted_files` inside ingest.
+///
+/// Returns the list sorted for deterministic output. Called twice by the
+/// CLI path (once for the confirm UI preview, once inside `execute`);
+/// the double scan is acceptable for a rare destructive operation.
+pub fn enumerate_orphans(cfg: &Config) -> Result<Vec<WorkspacePath>> {
+    use kebab_core::DocumentStore as _;
+    use kebab_source_fs::FsSourceConnector;
+    use kebab_core::SourceScope;
+
+    let store = kebab_store_sqlite::SqliteStore::open(cfg)
+        .context("enumerate_orphans: open SqliteStore")?;
+
+    let stored = store
+        .all_workspace_paths()
+        .context("enumerate_orphans: all_workspace_paths")?;
+
+    if stored.is_empty() {
+        return Ok(Vec::new());
+    }
+
+    // Build the same SourceScope the CLI's ingest path uses: root from
+    // config, exclude list from config, no include override (full scope).
+    let root = cfg.resolve_workspace_root();
+    let scope = SourceScope {
+        root: root.clone(),
+        exclude: cfg.workspace.exclude.clone(),
+        ..Default::default()
+    };
+
+    let connector = FsSourceConnector::new(cfg)
+        .context("enumerate_orphans: build FsSourceConnector")?;
+    let (assets, _skips) = connector
+        .scan_with_skips(&scope)
+        .context("enumerate_orphans: scan workspace")?;
+
+    let scanned: HashSet<WorkspacePath> = assets
+        .into_iter()
+        .map(|a| a.workspace_path)
+        .collect();
+
+    let mut orphans: Vec<WorkspacePath> = stored
+        .into_iter()
+        .filter(|p| !scanned.contains(p))
+        .collect();
+    orphans.sort_by(|a, b| a.0.cmp(&b.0));
+    Ok(orphans)
+}
+
 /// Wipe every path from `enumerate_paths(scope, cfg)`. For
 /// `ResetScope::VectorOnly`, also truncates the SQLite
 /// `embedding_records` table so the store doesn't point at the Lance
 /// rows we just removed off-disk.
 ///
+/// For `ResetScope::OrphansOnly`, no filesystem directories are removed.
+/// Instead the store is reconciled: stored docs outside the current walker
+/// scope are purged from SQLite (+ vector store when configured). The
+/// caller is expected to have already shown the confirm UI using
+/// `enumerate_orphans`.
+///
 /// Idempotent: a missing path is treated as already-removed (success).
 /// Returns a `ResetReport` listing exactly what was removed (paths that
 /// existed before the call) so `--json` callers see the truth, not the
 /// request.
 pub fn execute(scope: ResetScope, cfg: &Config) -> Result<ResetReport> {
+    if matches!(scope, ResetScope::OrphansOnly) {
+        return execute_orphans_only(cfg);
+    }
+
    let paths = enumerate_paths(scope, cfg);
    let mut removed = Vec::new();

@@ -128,9 +221,100 @@ pub fn execute(scope: ResetScope, cfg: &Config) -> Result<ResetReport> {
        scope,
        removed_paths: removed,
        embedding_rows_truncated,
+        orphans_purged: 0,
+        purged_paths: Vec::new(),
    })
 }

+/// Execute the `OrphansOnly` variant: reconcile stored docs against the
+/// current walker scope without touching any filesystem directory.
+fn execute_orphans_only(cfg: &Config) -> Result<ResetReport> {
+    let orphans = enumerate_orphans(cfg)
+        .context("execute_orphans_only: enumerate orphans")?;
+
+    if orphans.is_empty() {
+        return Ok(ResetReport {
+            scope: ResetScope::OrphansOnly,
+            removed_paths: Vec::new(),
+            embedding_rows_truncated: 0,
+            orphans_purged: 0,
+            purged_paths: Vec::new(),
+        });
+    }
+
+    let store = std::sync::Arc::new(
+        kebab_store_sqlite::SqliteStore::open(cfg)
+            .context("execute_orphans_only: open SqliteStore")?,
+    );
+
+    // Open vector store if configured. Mirror the same guard the ingest
+    // path uses: only construct when the provider is not "none" / dims > 0.
+    let vector_store: Option<kebab_store_vector::LanceVectorStore> =
+        open_vector_store_if_configured(cfg, store.clone())?;
+
+    let mut purged_paths: Vec<WorkspacePath> = Vec::new();
+
+    for path in &orphans {
+        let chunk_ids = kebab_store_sqlite::purge_deleted_workspace_path(&store, path)
+            .with_context(|| format!("execute_orphans_only: purge {}", path.0))?;
+
+        if let Some(ref vs) = vector_store {
+            if !chunk_ids.is_empty() {
+                use kebab_core::VectorStore as _;
+                if let Err(e) = vs.delete_by_chunk_ids(&chunk_ids) {
+                    tracing::warn!(
+                        target: "kebab-app",
+                        path = %path.0,
+                        count = chunk_ids.len(),
+                        error = %e,
+                        "reset --orphans-only: vector delete failed; SQLite side already cleaned"
+                    );
+                }
+            }
+        }
+
+        tracing::info!(
+            target: "kebab-app",
+            path = %path.0,
+            "reset --orphans-only: purged orphan document"
+        );
+        purged_paths.push(path.clone());
+    }
+
+    let orphans_purged = u32::try_from(purged_paths.len()).unwrap_or(u32::MAX);
+
+    Ok(ResetReport {
+        scope: ResetScope::OrphansOnly,
+        removed_paths: Vec::new(),
+        embedding_rows_truncated: 0,
+        orphans_purged,
+        purged_paths,
+    })
+}
+
+/// Open the Lance vector store if the configured embedding provider is
+/// active (non-"none", dimensions > 0). Returns `None` for lexical-only
+/// configs. Mirrors the guard in `App::vector`.
+fn open_vector_store_if_configured(
+    cfg: &Config,
+    store: std::sync::Arc<kebab_store_sqlite::SqliteStore>,
+) -> Result<Option<kebab_store_vector::LanceVectorStore>> {
+    if cfg.models.embedding.provider == "none" || cfg.models.embedding.dimensions == 0 {
+        return Ok(None);
+    }
+    match kebab_store_vector::LanceVectorStore::new(cfg, store) {
+        Ok(vs) => Ok(Some(vs)),
+        Err(e) => {
+            tracing::warn!(
+                target: "kebab-app",
+                error = %e,
+                "reset --orphans-only: could not open vector store; skipping vector delete"
+            );
+            Ok(None)
+        }
+    }
+}
+
 /// Open the SQLite store at the configured path and run
 /// `truncate_embedding_records`. Returns the count of truncated rows
 /// (the helper itself reports `DELETE` rowcount). If the SQLite file
@@ -200,4 +384,14 @@ mod tests {
        let bytes = estimate_size_bytes(&[dir.path().to_path_buf()]);
        assert_eq!(bytes, 5 + 6);
    }
+
+    #[test]
+    fn enumerate_orphans_only_returns_empty_paths() {
+        let cfg = Config::defaults();
+        let paths = enumerate_paths(ResetScope::OrphansOnly, &cfg);
+        assert!(
+            paths.is_empty(),
+            "OrphansOnly must return empty vec from enumerate_paths"
+        );
+    }
 }
--- a/crates/kebab-app/src/schema.rs
+++ b/crates/kebab-app/src/schema.rs
@@ -45,7 +45,7 @@ pub struct Models {
    pub corpus_revision: u64,
 }

-#[derive(Debug, Clone, Serialize, Deserialize)]
+#[derive(Debug, Clone, Default, Serialize, Deserialize)]
 pub struct Stats {
    pub doc_count: u64,
    pub chunk_count: u64,
@@ -63,6 +63,26 @@ pub struct Stats {
    /// p9-fb-37: docs whose `updated_at` exceeds the staleness threshold.
    #[serde(default)]
    pub stale_doc_count: u64,
+    /// p10-1A-1: code language breakdown (**doc** counts by canonical
+    /// lowercase language identifier). Empty until 1A-2 produces code
+    /// docs. v0.17.0 PR-C: doc-count semantics corrected here (the
+    /// previous "chunk counts" wording was a longstanding mis-label —
+    /// implementation has always been `COUNT(*) FROM documents
+    /// GROUP BY code_lang`). Use `code_lang_chunk_breakdown` for the
+    /// chunk-level companion.
+    #[serde(default)]
+    pub code_lang_breakdown: std::collections::BTreeMap<String, u32>,
+    /// p10-1A-1: repo breakdown (**doc** counts by `metadata.repo`
+    /// value). Empty until 1A-2 produces code docs. v0.17.0 PR-C:
+    /// doc-count wording corrected (mirror of code_lang_breakdown).
+    #[serde(default)]
+    pub repo_breakdown: std::collections::BTreeMap<String, u32>,
+    /// v0.17.0 PR-C: sister of [`Self::code_lang_breakdown`] returning
+    /// chunk counts instead of doc counts. Indexing-pressure metric —
+    /// one PDF spec → 200 chunks vs one Rust file → 5 chunks shows up
+    /// here in a way `code_lang_breakdown` (doc count) hides.
+    #[serde(default)]
+    pub code_lang_chunk_breakdown: std::collections::BTreeMap<String, u32>,
 }

 const KEBAB_VERSION: &str = env!("CARGO_PKG_VERSION");
@@ -145,7 +165,7 @@ fn collect_stats(
    store: &kebab_store_sqlite::SqliteStore,
 ) -> anyhow::Result<Stats> {
    let counts = store
-        .count_summary_with_threshold(cfg.search.stale_threshold_days as u64)?;
+        .count_summary_with_threshold(u64::from(cfg.search.stale_threshold_days))?;
    let data_dir = kebab_config::expand_path(&cfg.storage.data_dir, "");
    let index_bytes = kebab_store_sqlite::stats_ext::index_bytes(&data_dir)
        .map_err(|e| anyhow::anyhow!("index_bytes: {e}"))?;
@@ -158,6 +178,14 @@ fn collect_stats(
        lang_breakdown: counts.lang_breakdown,
        index_bytes,
        stale_doc_count: counts.stale_doc_count,
+        // p10-1A-2: populated by the store query added in this task.
+        code_lang_breakdown: store.code_lang_breakdown()?,
+        // p10-1A-2 follow-up: dogfooding (2026-05-20) revealed this was a
+        // placeholder — mirror of code_lang_breakdown for the repo field.
+        repo_breakdown: store.repo_breakdown()?,
+        // v0.17.0 PR-C: chunk-level companion (closes HOTFIXES
+        // 2026-05-22 "code_lang_breakdown chunk granularity" LOW).
+        code_lang_chunk_breakdown: store.code_lang_chunk_breakdown()?,
    })
 }

@@ -182,6 +210,41 @@ fn collect_models(cfg: &Config, store: &kebab_store_sqlite::SqliteStore) -> Mode
 mod tests_stats_ext {
    use super::*;

+    /// p10-1A-1: Stats must serialize `code_lang_breakdown` and
+    /// `repo_breakdown` so downstream consumers (MCP skill, Claude Code)
+    /// can branch on their presence.
+    #[test]
+    fn stats_includes_code_lang_and_repo_breakdown_fields() {
+        let stats = Stats::default();
+        let v = serde_json::to_value(&stats).unwrap();
+        assert!(
+            v.get("code_lang_breakdown").is_some(),
+            "Stats JSON must include code_lang_breakdown: {v}"
+        );
+        assert!(
+            v.get("repo_breakdown").is_some(),
+            "Stats JSON must include repo_breakdown: {v}"
+        );
+        // v0.17.0 PR-C: chunk-level companion field.
+        assert!(
+            v.get("code_lang_chunk_breakdown").is_some(),
+            "Stats JSON must include code_lang_chunk_breakdown (v0.17.0 PR-C): {v}"
+        );
+        // Empty BTreeMap serializes as `{}` — confirm it's an object, not null.
+        assert!(
+            v["code_lang_breakdown"].is_object(),
+            "code_lang_breakdown must be an object: {v}"
+        );
+        assert!(
+            v["repo_breakdown"].is_object(),
+            "repo_breakdown must be an object: {v}"
+        );
+        assert!(
+            v["code_lang_chunk_breakdown"].is_object(),
+            "code_lang_chunk_breakdown must be an object: {v}"
+        );
+    }
+
    #[test]
    fn stats_includes_breakdowns_and_bytes_on_fresh_corpus() {
        let dir = tempfile::tempdir().unwrap();
--- a/crates/kebab-app/tests/ask_smoke.rs
+++ b/crates/kebab-app/tests/ask_smoke.rs
@@ -33,6 +33,7 @@ fn ask_lexical_smoke() {
        history: Vec::new(),
        conversation_id: None,
        turn_index: None,
+        multi_hop: false,
    };
    // The fixture workspace contains "ownership" content; the model's
    // citation behavior depends on its training, so we don't assert on
--- a/crates/kebab-app/tests/code_ingest_smoke.rs
+++ b/crates/kebab-app/tests/code_ingest_smoke.rs
--- a/crates/kebab-app/tests/fetch_integration.rs
+++ b/crates/kebab-app/tests/fetch_integration.rs
@@ -38,12 +38,16 @@ fn fetch_chunk_returns_target_only_when_no_context() {
 #[test]
 fn fetch_chunk_with_context_returns_neighbors() {
    let env = common::TestEnv::new();
-    let body = "# H1\n\nA1\n\n# H2\n\nA2\n\n# H3\n\nA3\n\n# H4\n\nA4\n\n# H5\n\nA5\n";
+    // v0.17.0 trigram tokenizer: terms must be ≥3 Unicode chars to
+    // match. The earlier fixture used 2-char tokens like `A1`/`A3` for
+    // section bodies — those zero-hit under trigram. Use 5-char unique
+    // words per section so the query can pin one chunk deterministically.
+    let body = "# H1\n\napples\n\n# H2\n\nbanana\n\n# H3\n\ncherry\n\n# H4\n\ndurian\n\n# H5\n\nelder\n";
    common::ingest_md(&env, "multi.md", body);
    let app = env.app();

    let q = kebab_core::SearchQuery {
-        text: "A3".to_string(),
+        text: "cherry".to_string(),
        mode: kebab_core::SearchMode::Lexical,
        k: 1,
        filters: kebab_core::SearchFilters::default(),
--- a/crates/kebab-app/tests/file_deletion_auto_purge.rs
+++ b/crates/kebab-app/tests/file_deletion_auto_purge.rs
@@ -0,0 +1,178 @@
+//! Dogfood: auto-purge stored docs for filesystem-deleted files.
+//!
+//! Two tests:
+//!
+//! 1. `file_deletion_auto_purge` — ingest 2 files, delete one, re-ingest.
+//!    The re-ingest must report `purged_deleted_files = 1`, the deleted
+//!    file must no longer appear in `list_docs`, and lexical search for
+//!    its unique content must return no hits.
+//!
+//! 2. `include_scope_narrowing_does_not_purge` — ingest 2 files under a
+//!    wide glob, narrow the walker scope to only one file, re-ingest.
+//!    The narrowed ingest must NOT purge the out-of-scope file because
+//!    the file is still on disk (just excluded from this run). Protects
+//!    users against accidental data loss via config edits.
+
+mod common;
+
+use common::TestEnv;
+use kebab_app::ingest_with_config_opts;
+use kebab_app::IngestOpts;
+use kebab_core::{DocFilter, DocumentStore, SearchMode, SearchQuery, SourceScope};
+
+/// Helper: open the store via `TestEnv` and run `list_documents`.
+fn list_doc_paths(env: &TestEnv) -> Vec<String> {
+    use kebab_store_sqlite::SqliteStore;
+    let store = SqliteStore::open(&env.config).unwrap();
+    store.run_migrations().unwrap();
+    store
+        .list_documents(&DocFilter::default())
+        .unwrap()
+        .into_iter()
+        .map(|d| d.doc_path.0)
+        .collect()
+}
+
+#[test]
+fn file_deletion_auto_purge() {
+    let env = TestEnv::lexical_only();
+
+    // Write two .rs files into the workspace.
+    let a_path = env.workspace_root.join("a.rs");
+    let b_path = env.workspace_root.join("b.rs");
+    std::fs::write(&a_path, "// file a\nfn alpha() {}\n").unwrap();
+    std::fs::write(&b_path, "// file b\nfn bravo() {}\n").unwrap();
+
+    // First ingest — both must be New.
+    let first = ingest_with_config_opts(
+        env.config.clone(),
+        env.scope(),
+        false,
+        IngestOpts::default(),
+    )
+    .expect("first ingest must succeed");
+    // Only count the .rs files we added (there may be fixture files too).
+    let first_new = first.new;
+    assert!(first_new >= 2, "expected at least 2 new docs: {first:?}");
+    assert_eq!(
+        first.purged_deleted_files, 0,
+        "no purges on first ingest: {first:?}"
+    );
+    assert_eq!(first.errors, 0, "no errors on first ingest: {first:?}");
+
+    // Delete one file from the filesystem.
+    std::fs::remove_file(&b_path).expect("remove b.rs");
+
+    // Second ingest — scanned count drops by 1; b.rs should be purged.
+    let second = ingest_with_config_opts(
+        env.config.clone(),
+        env.scope(),
+        false,
+        IngestOpts::default(),
+    )
+    .expect("second ingest must succeed");
+
+    assert_eq!(
+        second.purged_deleted_files, 1,
+        "exactly 1 file should be purged: {second:?}"
+    );
+    assert_eq!(second.new, 0, "no new docs after deletion: {second:?}");
+    assert_eq!(second.updated, 0, "no updated docs: {second:?}");
+    assert_eq!(second.errors, 0, "no errors: {second:?}");
+
+    // b.rs must no longer appear in list_docs.
+    let doc_paths = list_doc_paths(&env);
+    let b_ws_path = "b.rs";
+    assert!(
+        !doc_paths.iter().any(|p| p == b_ws_path),
+        "b.rs must be gone from list_docs; got: {doc_paths:?}"
+    );
+    // a.rs must still be present.
+    let a_ws_path = "a.rs";
+    assert!(
+        doc_paths.iter().any(|p| p == a_ws_path),
+        "a.rs must still be in list_docs; got: {doc_paths:?}"
+    );
+
+    // Lexical search for b.rs's unique content returns no hits.
+    let app = env.app();
+    let query = SearchQuery {
+        text: "bravo".to_string(),
+        mode: SearchMode::Lexical,
+        k: 10,
+        filters: kebab_core::SearchFilters::default(),
+    };
+    let hits = app.search(query).expect("search must not error");
+    assert!(
+        hits.is_empty(),
+        "search for deleted file's content must return no hits; got: {hits:?}"
+    );
+}
+
+#[test]
+fn include_scope_narrowing_does_not_purge() {
+    let env = TestEnv::lexical_only();
+
+    // Write two .rs files.
+    let a_path = env.workspace_root.join("a_narrow.rs");
+    let b_path = env.workspace_root.join("b_narrow.rs");
+    std::fs::write(&a_path, "// narrow a\nfn alpha_narrow() {}\n").unwrap();
+    std::fs::write(&b_path, "// narrow b\nfn bravo_narrow() {}\n").unwrap();
+
+    // Wide scope: first ingest — both must be New.
+    let wide_scope = SourceScope {
+        root: env.workspace_root.clone(),
+        include: vec!["**/*.rs".to_string()],
+        exclude: env.config.workspace.exclude.clone(),
+    };
+    let first = ingest_with_config_opts(
+        env.config.clone(),
+        wide_scope,
+        false,
+        IngestOpts::default(),
+    )
+    .expect("first ingest (wide) must succeed");
+    assert!(
+        first.new >= 2,
+        "expected at least 2 new docs: {first:?}"
+    );
+    assert_eq!(
+        first.purged_deleted_files, 0,
+        "no purges on first ingest: {first:?}"
+    );
+
+    // Narrow scope: only a_narrow.rs in include — b_narrow.rs is still
+    // on disk but excluded from the walker scope.
+    let narrow_scope = SourceScope {
+        root: env.workspace_root.clone(),
+        include: vec!["a_narrow.rs".to_string()],
+        exclude: env.config.workspace.exclude.clone(),
+    };
+    let second = ingest_with_config_opts(
+        env.config.clone(),
+        narrow_scope,
+        false,
+        IngestOpts::default(),
+    )
+    .expect("second ingest (narrow) must succeed");
+
+    // CRITICAL: b_narrow.rs is still on disk — must NOT be purged.
+    assert_eq!(
+        second.purged_deleted_files, 0,
+        "scope-narrowing must NOT purge on-disk files; got: {second:?}"
+    );
+    assert_eq!(second.errors, 0, "no errors: {second:?}");
+
+    // b_narrow.rs must still exist in the store.
+    let doc_paths = list_doc_paths(&env);
+    let b_ws_path = "b_narrow.rs";
+    assert!(
+        doc_paths.iter().any(|p| p == b_ws_path),
+        "b_narrow.rs must still be in list_docs after scope narrowing; got: {doc_paths:?}"
+    );
+    // And the file must still be on disk.
+    assert!(
+        b_path.exists(),
+        "b_narrow.rs must still be on disk (we didn't delete it)"
+    );
+}
--- a/crates/kebab-app/tests/ingest_file.rs
+++ b/crates/kebab-app/tests/ingest_file.rs
@@ -33,7 +33,7 @@ fn ingest_file_copies_external_md_and_reports_new() {
    assert!(ext_dir.is_dir());
    let entries: Vec<_> = fs::read_dir(&ext_dir)
        .unwrap()
-        .filter_map(|e| e.ok())
+        .filter_map(std::result::Result::ok)
        .collect();
    assert_eq!(entries.len(), 1, "exactly one file in _external/");
    let name = entries[0].file_name().to_string_lossy().into_owned();
--- a/crates/kebab-app/tests/ingest_stdin.rs
+++ b/crates/kebab-app/tests/ingest_stdin.rs
@@ -35,7 +35,7 @@ fn ingest_stdin_writes_frontmatter_and_reports_new() {
    // _external/ contains exactly one .md file with frontmatter.
    let ext_dir = std::path::PathBuf::from(&cfg.workspace.root).join("_external");
    let entries: Vec<_> = fs::read_dir(&ext_dir).unwrap()
-        .filter_map(|e| e.ok())
+        .filter_map(std::result::Result::ok)
        .collect();
    assert_eq!(entries.len(), 1);
    let content = fs::read_to_string(entries[0].path()).unwrap();
@@ -60,7 +60,7 @@ fn ingest_stdin_without_source_uri() {

    let ext_dir = std::path::PathBuf::from(&cfg.workspace.root).join("_external");
    let entries: Vec<_> = fs::read_dir(&ext_dir).unwrap()
-        .filter_map(|e| e.ok())
+        .filter_map(std::result::Result::ok)
        .collect();
    let content = fs::read_to_string(entries[0].path()).unwrap();
    assert!(content.contains("title: \"Title\""));
--- a/crates/kebab-app/tests/open_with_config_nli.rs
+++ b/crates/kebab-app/tests/open_with_config_nli.rs
@@ -0,0 +1,81 @@
+//! Tests for `App::open_with_config`'s NLI verifier construction path.
+//!
+//! Coverage:
+//! 1. `open_with_config_nli_fails_when_model_dir_unwritable_and_threshold_positive` —
+//!    when `rag.nli_threshold > 0` and `storage.model_dir` is unwritable,
+//!    `open_with_config` returns `Err` with "OnnxNliVerifier" in the
+//!    error chain.
+//! 2. `open_with_config_nli_skipped_when_threshold_zero` —
+//!    same bad `model_dir`, but `rag.nli_threshold = 0.0` (gate disabled),
+//!    so `OnnxNliVerifier::new` is never called and `open_with_config`
+//!    succeeds.
+//!
+//! `/proc/1/root` is the init process's filesystem root; on Linux it is
+//! owned by root and not traversable by unprivileged users, making
+//! `create_dir_all` fail with `EACCES` — a reliable "unwritable path"
+//! that requires no test setup beyond the path literal.
+
+use kebab_config::Config;
+
+/// Return a `Config` whose `data_dir` lives in a fresh `TempDir`
+/// (so `SqliteStore::open` succeeds) and whose `model_dir` is set to
+/// `/proc/1/root` (unwritable by non-root processes on Linux).
+///
+/// The `TempDir` is returned alongside the config so the caller keeps
+/// it alive until the test completes — dropping it early would delete
+/// the data directory before any assertions run.
+fn config_with_unwritable_model_dir() -> (tempfile::TempDir, Config) {
+    let tmp = tempfile::tempdir().expect("tempdir");
+    let mut cfg = Config::defaults();
+    // Valid data_dir → SqliteStore::open + run_migrations succeed.
+    cfg.storage.data_dir = tmp.path().to_string_lossy().into_owned();
+    // /proc/1/root is only accessible to root; create_dir_all will
+    // return EACCES for any unprivileged user, which is exactly the
+    // failure mode we want to exercise.
+    cfg.storage.model_dir = "/proc/1/root".to_string();
+    (tmp, cfg)
+}
+
+// ── 1. Failure path: threshold > 0 + unwritable model_dir ─────────────────
+
+#[test]
+fn open_with_config_nli_fails_when_model_dir_unwritable_and_threshold_positive() {
+    let (_tmp, mut cfg) = config_with_unwritable_model_dir();
+    cfg.rag.nli_threshold = 0.5; // gate enabled → OnnxNliVerifier::new runs
+
+    let result = kebab_app::App::open_with_config(cfg);
+
+    let Err(err) = result else {
+        panic!(
+            "App::open_with_config must fail when model_dir is unwritable and nli_threshold > 0"
+        );
+    };
+    // The error chain must identify the OnnxNliVerifier as the source so
+    // an operator reading logs can trace the failure to the NLI config.
+    let err_chain = format!("{err:?}");
+    assert!(
+        err_chain.contains("OnnxNliVerifier"),
+        "error chain must mention OnnxNliVerifier; full chain: {err_chain}"
+    );
+}
+
+// ── 2. Success path: threshold = 0.0 → NLI verifier never constructed ──────
+
+#[test]
+fn open_with_config_nli_skipped_when_threshold_zero() {
+    let (_tmp, cfg) = config_with_unwritable_model_dir();
+    // Default nli_threshold is 0.0 — gate disabled, verifier skipped.
+    assert!(
+        (cfg.rag.nli_threshold - 0.0).abs() < f32::EPSILON,
+        "precondition: default nli_threshold must be 0.0 (gate disabled)"
+    );
+
+    // A bad model_dir must NOT cause a failure when the NLI gate is off.
+    let result = kebab_app::App::open_with_config(cfg);
+    assert!(
+        result.is_ok(),
+        "App::open_with_config must succeed when nli_threshold = 0.0 \
+         (OnnxNliVerifier is never constructed); err: {:?}",
+        result.err()
+    );
+}
--- a/crates/kebab-app/tests/reset_orphans.rs
+++ b/crates/kebab-app/tests/reset_orphans.rs
@@ -0,0 +1,141 @@
+//! Integration test for `kebab reset --orphans-only`.
+//!
+//! Verifies that stored docs outside the current walker scope are purged
+//! from the store without removing any files from the filesystem.
+//!
+//! Test outline:
+//! 1. Ingest 3 .rs files (a.rs, b.rs, c.rs) — all New.
+//! 2. Narrow the config `include` to `["a.rs"]` only; b.rs and c.rs are
+//!    still on disk but outside the walker scope.
+//! 3. Run `execute(ResetScope::OrphansOnly, &cfg)` — report must show
+//!    `orphans_purged == 2` and `purged_paths` contains b.rs + c.rs.
+//! 4. `list docs` must show only a.rs.
+//! 5. b.rs and c.rs must still exist on disk (no filesystem removal).
+//! 6. Second reset → `orphans_purged == 0` (idempotent).
+
+mod common;
+
+use common::TestEnv;
+use kebab_app::IngestOpts;
+use kebab_app::reset::{ResetScope, execute};
+use kebab_core::{DocFilter, DocumentStore, SourceScope};
+
+/// Open the SqliteStore and list all `workspace_path` values.
+fn list_doc_paths(env: &TestEnv) -> Vec<String> {
+    use kebab_store_sqlite::SqliteStore;
+    let store = SqliteStore::open(&env.config).unwrap();
+    store.run_migrations().unwrap();
+    store
+        .list_documents(&DocFilter::default())
+        .unwrap()
+        .into_iter()
+        .map(|d| d.doc_path.0)
+        .collect()
+}
+
+#[test]
+fn reset_orphans_only_purges_out_of_scope_docs() {
+    let env = TestEnv::lexical_only();
+
+    // Write three .rs files into the workspace.
+    let a_path = env.workspace_root.join("a.rs");
+    let b_path = env.workspace_root.join("b.rs");
+    let c_path = env.workspace_root.join("c.rs");
+    std::fs::write(&a_path, "// file a\nfn alpha() {}\n").unwrap();
+    std::fs::write(&b_path, "// file b\nfn bravo() {}\n").unwrap();
+    std::fs::write(&c_path, "// file c\nfn charlie() {}\n").unwrap();
+
+    // Ingest all three with a wide scope.
+    let wide_scope = SourceScope {
+        root: env.workspace_root.clone(),
+        include: vec!["**/*.rs".to_string()],
+        exclude: env.config.workspace.exclude.clone(),
+    };
+    let first = kebab_app::ingest_with_config_opts(
+        env.config.clone(),
+        wide_scope,
+        false,
+        IngestOpts::default(),
+    )
+    .expect("first ingest must succeed");
+    // The fixture workspace may contain other .rs files — just assert we
+    // got at least 3 new docs (our a.rs, b.rs, c.rs).
+    assert!(first.new >= 3, "expected at least 3 new docs: {first:?}");
+    assert_eq!(first.errors, 0, "no errors on first ingest");
+
+    // Narrow config to include only a.rs; b.rs + c.rs are still on disk.
+    let mut narrow_cfg = env.config.clone();
+    narrow_cfg.workspace.exclude.clear();
+    // Re-point workspace root (already correct) and restrict include via
+    // the SourceScope in the connector. The config's `workspace.root` is
+    // used by `enumerate_orphans` to build its scope — we keep that
+    // pointing at the workspace root. We simulate narrowing by setting a
+    // glob that only matches a.rs.
+    //
+    // NOTE: `kebab_config::WorkspaceCfg` does not have an `include` field
+    // (it was removed in p9-fb-25). We narrow the scope via the walker
+    // exclude list: exclude b.rs and c.rs explicitly.
+    narrow_cfg.workspace.exclude = vec!["b.rs".to_string(), "c.rs".to_string()];
+
+    // Run orphans-only reset.
+    let report = execute(ResetScope::OrphansOnly, &narrow_cfg)
+        .expect("orphans-only reset must succeed");
+
+    assert_eq!(
+        report.orphans_purged, 2,
+        "expected 2 orphans purged (b.rs + c.rs): {report:?}"
+    );
+
+    let mut purged: Vec<String> = report
+        .purged_paths
+        .iter()
+        .map(|p| p.0.clone())
+        .collect();
+    purged.sort();
+    assert_eq!(
+        purged,
+        vec!["b.rs".to_string(), "c.rs".to_string()],
+        "purged_paths must list b.rs and c.rs in sorted order: {purged:?}"
+    );
+
+    // list docs must show only a.rs (and any pre-existing fixture files
+    // that are not excluded by the narrow config).
+    let doc_paths = list_doc_paths(&env);
+    // The narrow_cfg excludes b.rs + c.rs — they must no longer be in store.
+    assert!(
+        !doc_paths.iter().any(|p| p == "b.rs"),
+        "b.rs must be gone from store after orphans-only reset; got: {doc_paths:?}"
+    );
+    assert!(
+        !doc_paths.iter().any(|p| p == "c.rs"),
+        "c.rs must be gone from store after orphans-only reset; got: {doc_paths:?}"
+    );
+    assert!(
+        doc_paths.iter().any(|p| p == "a.rs"),
+        "a.rs must still be in store; got: {doc_paths:?}"
+    );
+
+    // Both b.rs and c.rs must still exist on the filesystem — no file
+    // removal is performed by orphans-only.
+    assert!(
+        b_path.exists(),
+        "b.rs must still be on disk after orphans-only reset"
+    );
+    assert!(
+        c_path.exists(),
+        "c.rs must still be on disk after orphans-only reset"
+    );
+
+    // Second reset must be idempotent: nothing left to purge.
+    let second = execute(ResetScope::OrphansOnly, &narrow_cfg)
+        .expect("second orphans-only reset must succeed");
+    assert_eq!(
+        second.orphans_purged, 0,
+        "second reset must be idempotent (orphans_purged == 0): {second:?}"
+    );
+    assert!(
+        second.purged_paths.is_empty(),
+        "second reset purged_paths must be empty: {:?}",
+        second.purged_paths
+    );
+}
--- a/crates/kebab-app/tests/search_korean.rs
+++ b/crates/kebab-app/tests/search_korean.rs
@@ -46,3 +46,88 @@ fn korean_lexical_query_returns_korean_document() {
        hits.iter().map(|h| &h.doc_path.0).collect::<Vec<_>>()
    );
 }
+
+/// A4 Step 1c — multi-token Korean query (`해시 충돌`) must hit when
+/// the lexical builder routes it through a whole-phrase MATCH candidate.
+///
+/// Expected: FAIL until A5 (`build_match_string` redesign) lands — the
+/// current builder emits `"해시" "충돌"` AND, but FTS5 trigram tokenizer
+/// has no 2-char terms so each side is 0-hit. A5 introduces a whole-
+/// phrase candidate (`"해시 충돌"`) OR'd with the token AND, restoring
+/// hits for the dominant Korean usage pattern.
+#[test]
+fn lexical_multi_token_korean_query_hits() {
+    let env = TestEnv::lexical_only();
+
+    // Copy the synthetic Korean fixture (introduced in A4 Step 0) into
+    // the test workspace. The fixture contains the exact phrase
+    // "해시 충돌" multiple times.
+    let dest = env.workspace_root.join("hash-table.md");
+    let src = std::path::PathBuf::from(env!("CARGO_MANIFEST_DIR"))
+        .join("..")
+        .join("..")
+        .join("fixtures")
+        .join("search")
+        .join("korean")
+        .join("hash-table.md");
+    std::fs::copy(&src, &dest).expect("copy korean fixture");
+
+    kebab_app::ingest_with_config(env.config.clone(), env.scope(), true)
+        .expect("ingest must succeed");
+
+    let hits = kebab_app::search_with_config(
+        env.config.clone(),
+        common::lexical_query("해시 충돌"),
+    )
+    .expect("search must succeed");
+
+    assert!(
+        !hits.is_empty(),
+        "multi-token Korean query '해시 충돌' must hit the hash-table fixture; got {:?}",
+        hits.iter().map(|h| &h.doc_path.0).collect::<Vec<_>>()
+    );
+    let any_hash_table = hits.iter().any(|h| h.doc_path.0.contains("hash-table"));
+    assert!(
+        any_hash_table,
+        "expected at least one hit on the hash-table fixture, got: {:?}",
+        hits.iter().map(|h| &h.doc_path.0).collect::<Vec<_>>()
+    );
+}
+
+/// A4 Step 1c — mixed Korean+English multi-token query (`Rust 충돌은`).
+/// Both tokens are ≥3 chars, so the redesigned builder (A5) emits
+/// `("Rust 충돌은") OR ("Rust" AND "충돌은")`. With trigram tokenizer
+/// each side has substring coverage in the document, so the AND branch
+/// alone is enough. Expected: FAIL pre-A5, PASS post-A5.
+#[test]
+fn lexical_mixed_korean_english_multi_token_query_hits() {
+    let env = TestEnv::lexical_only();
+    let doc_path = env.workspace_root.join("rust-hash.md");
+    std::fs::write(
+        &doc_path,
+        "# Rust 해시 테이블\n\nRust 의 std::collections::HashMap 에서 \
+         해시 충돌은 SipHash 로 완화한다.\n",
+    )
+    .expect("write rust-hash fixture");
+
+    kebab_app::ingest_with_config(env.config.clone(), env.scope(), true)
+        .expect("ingest must succeed");
+
+    let hits = kebab_app::search_with_config(
+        env.config.clone(),
+        common::lexical_query("Rust 충돌은"),
+    )
+    .expect("search must succeed");
+
+    assert!(
+        !hits.is_empty(),
+        "mixed Korean+English multi-token query 'Rust 충돌은' must hit the rust-hash fixture; got {:?}",
+        hits.iter().map(|h| &h.doc_path.0).collect::<Vec<_>>()
+    );
+    let any_rust_hash = hits.iter().any(|h| h.doc_path.0.contains("rust-hash"));
+    assert!(
+        any_rust_hash,
+        "expected at least one hit on the rust-hash fixture, got: {:?}",
+        hits.iter().map(|h| &h.doc_path.0).collect::<Vec<_>>()
+    );
+}
--- a/crates/kebab-app/tests/search_vector.rs
+++ b/crates/kebab-app/tests/search_vector.rs
@@ -14,12 +14,10 @@ use common::TestEnv;
 fn require_avx_or_panic() {
    #[cfg(target_arch = "x86_64")]
    {
-        if !std::is_x86_feature_detected!("avx") {
-            panic!(
-                "kb-app vector integration test requires AVX-capable hardware; \
-                 host CPU lacks AVX. Run on an AVX-capable machine."
-            );
-        }
+        assert!(std::is_x86_feature_detected!("avx"), 
+            "kb-app vector integration test requires AVX-capable hardware; \
+             host CPU lacks AVX. Run on an AVX-capable machine."
+        );
    }
 }

--- a/crates/kebab-app/tests/twin_files_fetch_span.rs
+++ b/crates/kebab-app/tests/twin_files_fetch_span.rs
@@ -0,0 +1,176 @@
+//! Regression test for the twin-file fetch_span media-type lookup bug.
+//!
+//! Twin files (identical content at different workspace paths) share one
+//! `assets` row whose PRIMARY KEY is the blake3 content hash. The old
+//! `fetch_span` implementation called
+//! `get_asset_by_workspace_path(&doc.workspace_path)` to check whether the
+//! media type was PDF/audio (and therefore reject span fetch). For a twin
+//! file that lookup could silently return the *other* twin's asset row if
+//! `assets.workspace_path` had been overwritten on the most recent ingest of
+//! the sibling — making the media-type branch decision incorrect.
+//!
+//! Fix: `fetch_span` now uses the 2-step lookup
+//!   `get_document_by_workspace_path` → `doc.source_asset_id` → `get_asset`
+//! so the result is always anchored to the requesting document, not
+//! whichever twin last updated `assets.workspace_path`.
+//!
+//! This test builds a twin-file scenario (two .md files at different paths
+//! with identical content), ingests both, then calls `fetch_span` on each
+//! twin's `doc_id` and asserts it succeeds. Before the fix, if the asset
+//! row's workspace_path happened to point at the wrong twin the span could
+//! return an incorrect `span_not_supported` for a non-PDF/audio file, or
+//! conversely allow span on a PDF twin by accident. After the fix, the
+//! lookup is always doc-specific.
+
+mod common;
+
+use common::TestEnv;
+use kebab_app::ingest_with_config;
+use kebab_core::{DocumentStore, FetchKind, FetchOpts, FetchQuery, IngestItemKind};
+
+#[test]
+fn twin_files_fetch_span_uses_correct_asset() {
+    let env = TestEnv::lexical_only();
+
+    // Write two markdown files with identical content at different paths.
+    let dir_a = env.workspace_root.join("src_a");
+    let dir_b = env.workspace_root.join("src_b");
+    std::fs::create_dir_all(&dir_a).unwrap();
+    std::fs::create_dir_all(&dir_b).unwrap();
+
+    // The content must produce at least 1 line so span fetch is non-trivial.
+    let content = "# Twin\n\nLine one.\n\nLine two.\n\nLine three.\n";
+    std::fs::write(dir_a.join("note.md"), content).unwrap();
+    std::fs::write(dir_b.join("note.md"), content).unwrap();
+
+    // Ingest all files (fixture workspace + our two new twins).
+    let report = ingest_with_config(env.config.clone(), env.scope(), false)
+        .expect("ingest must succeed");
+    assert_eq!(report.errors, 0, "no ingest errors; report={report:?}");
+
+    // Both twin paths must appear as New in the report.
+    let items = report.items.as_ref().expect("items must be present");
+    let twin_items: Vec<_> = items
+        .iter()
+        .filter(|i| {
+            i.doc_path.0.ends_with("src_a/note.md")
+                || i.doc_path.0.ends_with("src_b/note.md")
+        })
+        .collect();
+    assert_eq!(
+        twin_items.len(),
+        2,
+        "exactly 2 twin items expected; items={items:?}"
+    );
+    for item in &twin_items {
+        assert_eq!(
+            item.kind,
+            IngestItemKind::New,
+            "each twin must be New; item={item:?}"
+        );
+    }
+
+    // Resolve doc_ids for both workspace paths.
+    // The ingest layer normalises workspace_path to the path relative to
+    // workspace_root (e.g. "src_a/note.md"), so we look up by that form.
+    let store = kebab_store_sqlite::SqliteStore::open(&env.config).unwrap();
+    store.run_migrations().unwrap();
+
+    // Find the twin items by matching on suffix so the test is robust to
+    // however the workspace root is represented.
+    let items = report.items.as_ref().expect("items must be present");
+    let path_a_str = items
+        .iter()
+        .find(|i| i.doc_path.0.ends_with("src_a/note.md"))
+        .map(|i| i.doc_path.0.clone())
+        .expect("src_a/note.md must appear in ingest report");
+    let path_b_str = items
+        .iter()
+        .find(|i| i.doc_path.0.ends_with("src_b/note.md"))
+        .map(|i| i.doc_path.0.clone())
+        .expect("src_b/note.md must appear in ingest report");
+
+    let path_a = kebab_core::WorkspacePath(path_a_str);
+    let path_b = kebab_core::WorkspacePath(path_b_str);
+
+    let doc_a = store
+        .get_document_by_workspace_path(&path_a)
+        .expect("get_document_by_workspace_path path_a")
+        .expect("doc_a must exist after ingest");
+    let doc_b = store
+        .get_document_by_workspace_path(&path_b)
+        .expect("get_document_by_workspace_path path_b")
+        .expect("doc_b must exist after ingest");
+
+    // Both twins share one asset_id (same content hash).
+    assert_eq!(
+        doc_a.source_asset_id, doc_b.source_asset_id,
+        "twin files must share one asset_id"
+    );
+
+    // Open App and issue span fetch on each twin's doc_id.
+    let app = env.app();
+
+    let result_a = app
+        .fetch(
+            FetchQuery::Span {
+                doc_id: doc_a.doc_id.clone(),
+                line_start: 1,
+                line_end: 2,
+            },
+            FetchOpts::default(),
+        )
+        .expect("fetch_span on twin A must succeed for a markdown file");
+    assert_eq!(result_a.kind, FetchKind::Span);
+    assert!(
+        result_a.text.as_deref().is_some_and(|t| !t.is_empty()),
+        "span text for twin A must not be empty"
+    );
+
+    let result_b = app
+        .fetch(
+            FetchQuery::Span {
+                doc_id: doc_b.doc_id.clone(),
+                line_start: 1,
+                line_end: 2,
+            },
+            FetchOpts::default(),
+        )
+        .expect("fetch_span on twin B must succeed for a markdown file");
+    assert_eq!(result_b.kind, FetchKind::Span);
+    assert!(
+        result_b.text.as_deref().is_some_and(|t| !t.is_empty()),
+        "span text for twin B must not be empty"
+    );
+
+    // Ingest again to force the asset.workspace_path flip-flop, then
+    // re-check. Pre-fix this was the scenario that triggered the bug:
+    // after the second ingest the asset row's workspace_path could point
+    // at either twin, making one twin's span fetch behave incorrectly.
+    let report2 = ingest_with_config(env.config.clone(), env.scope(), false)
+        .expect("second ingest must succeed");
+    assert_eq!(report2.errors, 0, "no ingest errors on second run; report={report2:?}");
+
+    // Re-open app after second ingest and verify span still works on both.
+    let app2 = env.app();
+
+    app2.fetch(
+        FetchQuery::Span {
+            doc_id: doc_a.doc_id.clone(),
+            line_start: 1,
+            line_end: 3,
+        },
+        FetchOpts::default(),
+    )
+    .expect("fetch_span on twin A after flip-flop must still succeed");
+
+    app2.fetch(
+        FetchQuery::Span {
+            doc_id: doc_b.doc_id.clone(),
+            line_start: 1,
+            line_end: 3,
+        },
+        FetchOpts::default(),
+    )
+    .expect("fetch_span on twin B after flip-flop must still succeed");
+}
--- a/crates/kebab-app/tests/twin_files_idempotent.rs
+++ b/crates/kebab-app/tests/twin_files_idempotent.rs
@@ -0,0 +1,90 @@
+//! Regression test for the twin-file idempotency bug.
+//!
+//! Identical-content files at different workspace paths share one
+//! `assets` row (`asset_id` = blake3 content hash, PRIMARY KEY). The
+//! old UPSERT `ON CONFLICT(asset_id) DO UPDATE SET workspace_path =
+//! excluded.workspace_path` made each twin overwrite the other's path
+//! on every ingest, so `get_asset_by_workspace_path(path1)` returned
+//! None (or the wrong twin) → re-process every time.
+//!
+//! Fix: `try_skip_unchanged` now uses `get_document_by_workspace_path`
+//! instead.  `documents.workspace_path` is UNIQUE (V001) so each twin
+//! has its own stable document row.
+//!
+//! Assertion contract:
+//!   1st ingest → 2 New (one per twin)
+//!   2nd ingest → 0 New, 0 Updated, 2 Unchanged
+
+mod common;
+
+use common::TestEnv;
+use kebab_app::ingest_with_config;
+use kebab_core::IngestItemKind;
+
+#[test]
+fn twin_files_second_ingest_is_unchanged() {
+    let env = TestEnv::lexical_only();
+
+    // Write two files with identical content at different paths.
+    let pkg_a = env.workspace_root.join("pkg_a");
+    let pkg_b = env.workspace_root.join("pkg_b");
+    std::fs::create_dir_all(&pkg_a).unwrap();
+    std::fs::create_dir_all(&pkg_b).unwrap();
+
+    let content = b"# shared\nThis content is identical in both files.\n";
+    std::fs::write(pkg_a.join("__init__.py"), content).unwrap();
+    std::fs::write(pkg_b.join("__init__.py"), content).unwrap();
+
+    // First ingest — both files must be New.
+    let first = ingest_with_config(env.config.clone(), env.scope(), false)
+        .expect("first ingest must succeed");
+    assert_eq!(first.errors, 0, "first ingest: no errors; report={first:?}");
+
+    let items = first.items.as_ref().expect("items must be present");
+    let twin_items: Vec<_> = items
+        .iter()
+        .filter(|i| {
+            i.doc_path.0.ends_with("__init__.py")
+        })
+        .collect();
+    assert_eq!(
+        twin_items.len(),
+        2,
+        "first ingest: expected exactly 2 __init__.py items; items={items:?}"
+    );
+    for item in &twin_items {
+        assert_eq!(
+            item.kind,
+            IngestItemKind::New,
+            "first ingest: each twin must be New; item={item:?}"
+        );
+    }
+
+    // Second ingest — same files, same content → both must be Unchanged.
+    let second = ingest_with_config(env.config.clone(), env.scope(), false)
+        .expect("second ingest must succeed");
+    assert_eq!(second.errors, 0, "second ingest: no errors; report={second:?}");
+    assert_eq!(second.new, 0, "second ingest: no new docs; report={second:?}");
+    assert_eq!(
+        second.updated, 0,
+        "second ingest: no updated docs (twin-file bug would set this to 2); report={second:?}"
+    );
+
+    let second_items = second.items.as_ref().expect("items must be present");
+    let twin_items2: Vec<_> = second_items
+        .iter()
+        .filter(|i| i.doc_path.0.ends_with("__init__.py"))
+        .collect();
+    assert_eq!(
+        twin_items2.len(),
+        2,
+        "second ingest: expected exactly 2 __init__.py items; items={second_items:?}"
+    );
+    for item in &twin_items2 {
+        assert_eq!(
+            item.kind,
+            IngestItemKind::Unchanged,
+            "second ingest: each twin must be Unchanged; item={item:?}"
+        );
+    }
+}
--- a/crates/kebab-chunk/Cargo.toml
+++ b/crates/kebab-chunk/Cargo.toml
@@ -13,14 +13,19 @@ serde_json_canonicalizer   = "0.3"
 blake3                     = { workspace = true }
 anyhow                     = { workspace = true }
 tracing                    = { workspace = true }
+serde_yaml                 = { workspace = true }

 [dev-dependencies]
-# kb-parse-md / kb-normalize are dev-only — used by the snapshot integration
-# test to build a CanonicalDocument from a fixture Markdown file. Forbidden as
-# regular deps per design §8 (chunker consumes CanonicalDocument from kb-core
-# only); `cargo tree -p kb-chunk --depth 1` (default scope, excludes dev-deps)
+# kb-parse-md / kb-parse-code are dev-only — used by the snapshot integration
+# tests to build a CanonicalDocument from fixture files. kb-parse-md absorbed
+# kb-normalize in v0.19.0 (HOTFIXES.md 2026-05-26). Forbidden as regular deps
+# per design §8 (chunker consumes CanonicalDocument from kb-core only);
+# `cargo tree -p kb-chunk --depth 1` (default scope, excludes dev-deps)
 # confirms this.
-kebab-parse-md = { path = "../kebab-parse-md" }
-kebab-normalize = { path = "../kebab-normalize" }
-serde_json      = { workspace = true }
-time            = { workspace = true }
+kebab-parse-md   = { path = "../kebab-parse-md" }
+kebab-parse-code = { path = "../kebab-parse-code" }
+serde_json       = { workspace = true }
+time             = { workspace = true }
+
+[lints]
+workspace = true
--- a/crates/kebab-chunk/src/code_c_ast_v1.rs
+++ b/crates/kebab-chunk/src/code_c_ast_v1.rs
@@ -0,0 +1,322 @@
+//! `code-c-ast-v1` — maps a tree-sitter-derived C AST
+//! `CanonicalDocument` (one `Block::Code` per semantic unit, each with
+//! `SourceSpan::Code`) to chunks 1:1. A unit longer than
+//! `AST_CHUNK_MAX_LINES` is split into `<symbol> [part i/N]` sub-chunks
+//! at blank-line paragraph boundaries (design §9.1 oversize fallback).
+//!
+//! tree-sitter is intentionally NOT a dependency here: AST work is
+//! parser-side (`kebab-parse-code`, design §6.3). This chunker only
+//! consumes the `CanonicalDocument`.
+//!
+//! `AST_CHUNK_MAX_LINES` is a constant matching
+//! `IngestCodeCfg::default().ast_chunk_max_lines` (200). Per-medium
+//! config threading needs a chunker registry (P+); same deviation
+//! pattern as `pdf-page-v1`'s pinned `chunker_version`
+//! (`tasks/HOTFIXES.md`).
+
+use kebab_core::{
+    Block, BlockId, CanonicalDocument, Chunk, ChunkPolicy, Chunker, ChunkerVersion, DocumentId,
+    SourceSpan, id_for_chunk,
+};
+
+const VERSION_LABEL: &str = "code-c-ast-v1";
+const BYTES_PER_TOKEN: usize = 3;
+const POLICY_HASH_HEX_LEN: usize = 16;
+const AST_CHUNK_MAX_LINES: u32 = 200;
+
+#[derive(Clone, Copy, Debug, Default)]
+pub struct CodeCAstV1Chunker;
+
+impl Chunker for CodeCAstV1Chunker {
+    fn chunker_version(&self) -> ChunkerVersion {
+        ChunkerVersion(VERSION_LABEL.to_string())
+    }
+
+    fn policy_hash(&self, policy: &ChunkPolicy) -> String {
+        let bytes = serde_json_canonicalizer::to_vec(policy)
+            .expect("canonical JSON serialization of ChunkPolicy must not fail");
+        let hex = blake3::hash(&bytes).to_hex().to_string();
+        hex[..POLICY_HASH_HEX_LEN].to_string()
+    }
+
+    fn chunk(
+        &self,
+        doc: &CanonicalDocument,
+        policy: &ChunkPolicy,
+    ) -> anyhow::Result<Vec<Chunk>> {
+        for b in &doc.blocks {
+            let c = match b {
+                Block::Code(c) => c,
+                _ => anyhow::bail!(
+                    "CodeCAstV1Chunker only handles code docs (got non-Code block)"
+                ),
+            };
+            if !matches!(c.common.source_span, SourceSpan::Code { .. }) {
+                anyhow::bail!(
+                    "CodeCAstV1Chunker only handles code docs (got non-Code source_span)"
+                );
+            }
+        }
+
+        let base_policy_hash = self.policy_hash(policy);
+        let chunker_version = self.chunker_version();
+        let mut out: Vec<Chunk> = Vec::new();
+
+        for b in &doc.blocks {
+            let cb = match b {
+                Block::Code(c) => c,
+                _ => unreachable!("validated above"),
+            };
+            let (ls, le, symbol, lang) = match &cb.common.source_span {
+                SourceSpan::Code { line_start, line_end, symbol, lang } => {
+                    (*line_start, *line_end, symbol.clone(), lang.clone())
+                }
+                _ => unreachable!("validated above"),
+            };
+            let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
+            let span_lines = le.saturating_sub(ls) + 1;
+
+            if span_lines <= AST_CHUNK_MAX_LINES {
+                let span = SourceSpan::Code {
+                    line_start: ls,
+                    line_end: le,
+                    symbol: symbol.clone(),
+                    lang: lang.clone(),
+                };
+                out.push(make_chunk(
+                    doc, &chunker_version, &block_ids, &base_policy_hash,
+                    None, span, cb.code.clone(),
+                ));
+            } else {
+                let parts = split_oversize(&cb.code);
+                let n = parts.len();
+                for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
+                    let part_ls = ls + off_start;
+                    let part_le = ls + off_end;
+                    let part_sym = symbol
+                        .as_ref()
+                        .map(|s| format!("{s} [part {}/{n}]", i + 1));
+                    let span = SourceSpan::Code {
+                        line_start: part_ls,
+                        line_end: part_le,
+                        symbol: part_sym,
+                        lang: lang.clone(),
+                    };
+                    out.push(make_chunk(
+                        doc, &chunker_version, &block_ids, &base_policy_hash,
+                        Some(part_ls), span, text,
+                    ));
+                }
+            }
+        }
+
+        tracing::debug!(
+            target: "kebab-chunk",
+            doc_id = %doc.doc_id,
+            chunks = out.len(),
+            "code-c-ast-v1 chunked",
+        );
+        Ok(out)
+    }
+}
+
+#[allow(clippy::too_many_arguments)]
+fn make_chunk(
+    doc: &CanonicalDocument,
+    chunker_version: &ChunkerVersion,
+    block_ids: &[BlockId],
+    base_policy_hash: &str,
+    split_key: Option<u32>,
+    span: SourceSpan,
+    text: String,
+) -> Chunk {
+    let id_hash = match split_key {
+        Some(k) => format!("{base_policy_hash}#L{k}"),
+        None => base_policy_hash.to_string(),
+    };
+    let chunk_id = id_for_chunk(&doc.doc_id, chunker_version, block_ids, &id_hash);
+    let token_estimate = text.len().div_ceil(BYTES_PER_TOKEN);
+    Chunk {
+        chunk_id,
+        doc_id: DocumentId(doc.doc_id.0.clone()),
+        block_ids: block_ids.to_vec(),
+        text,
+        heading_path: Vec::new(),
+        source_spans: vec![span],
+        token_estimate,
+        chunker_version: chunker_version.clone(),
+        policy_hash: base_policy_hash.to_string(),
+    }
+}
+
+/// Split an oversize unit at blank-line paragraph boundaries, greedily
+/// gluing paragraphs until ~`AST_CHUNK_MAX_LINES` lines accumulate.
+/// Returns `(line_offset_start, line_offset_end, text)` where offsets are
+/// 0-based within the unit (caller adds the unit's absolute `line_start`).
+fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
+    let lines: Vec<&str> = code.split('\n').collect();
+    let total = lines.len() as u32;
+    let mut out: Vec<(u32, u32, String)> = Vec::new();
+    let mut start: u32 = 0;
+    while start < total {
+        let mut end = (start + AST_CHUNK_MAX_LINES).min(total);
+        let floor = start + (AST_CHUNK_MAX_LINES * 4 / 5);
+        if end < total {
+            if let Some(b) = (floor.min(end)..end)
+                .rev()
+                .find(|&i| lines[i as usize].trim().is_empty())
+            {
+                end = b + 1;
+            }
+        }
+        let text = lines[start as usize..end as usize].join("\n");
+        out.push((start, end.saturating_sub(1), text));
+        start = end;
+    }
+    if out.is_empty() {
+        out.push((0, total.saturating_sub(1), code.to_string()));
+    }
+    out
+}
+
+#[cfg(test)]
+mod tests {
+    use super::*;
+    use kebab_core::{
+        Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+        SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance,
+        SourceType, TrustLevel, WorkspacePath,
+    };
+    use time::OffsetDateTime;
+
+    fn code_doc(units: &[(&str, u32, u32, &str)]) -> CanonicalDocument {
+        let wp = WorkspacePath("crates/x/src/a.c".into());
+        let aid = AssetId("a".repeat(64));
+        let pv = ParserVersion("code-c-v1".into());
+        let doc_id = id_for_doc(&wp, &aid, &pv);
+        let blocks = units
+            .iter()
+            .enumerate()
+            .map(|(i, (sym, ls, le, code))| {
+                let span = SourceSpan::Code {
+                    line_start: *ls,
+                    line_end: *le,
+                    symbol: Some((*sym).to_string()),
+                    lang: Some("c".into()),
+                };
+                let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+                Block::Code(CodeBlock {
+                    common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span },
+                    lang: Some("c".into()),
+                    code: (*code).to_string(),
+                })
+            })
+            .collect();
+        CanonicalDocument {
+            doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(),
+            lang: Lang("und".into()), blocks,
+            metadata: Metadata {
+                aliases: vec![], tags: vec![],
+                created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                source_type: SourceType::Note, trust_level: TrustLevel::Primary,
+                user_id_alias: None, user: Default::default(),
+                repo: Some("kebab".into()), git_branch: Some("main".into()),
+                git_commit: Some("0".repeat(40)), code_lang: Some("c".into()),
+            },
+            provenance: Provenance { events: vec![] },
+            parser_version: pv, schema_version: 1, doc_version: 1,
+            last_chunker_version: None, last_embedding_version: None,
+        }
+    }
+    fn policy() -> ChunkPolicy {
+        ChunkPolicy { target_tokens: 500, overlap_tokens: 80,
+            respect_markdown_headings: false,
+            chunker_version: ChunkerVersion(VERSION_LABEL.into()) }
+    }
+
+    #[test]
+    fn chunker_version_is_code_c_ast_v1() {
+        assert_eq!(CodeCAstV1Chunker.chunker_version(),
+            ChunkerVersion("code-c-ast-v1".into()));
+    }
+
+    #[test]
+    fn one_chunk_per_unit_preserves_code_span() {
+        let doc = code_doc(&[
+            ("parse", 1, 3, "int parse() {\n\t// x\n}"),
+            ("print", 5, 7, "void print() {\n\t//\n\treturn;\n}"),
+        ]);
+        let chunks = CodeCAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert_eq!(chunks.len(), 2);
+        for c in &chunks {
+            assert_eq!(c.source_spans.len(), 1);
+            assert!(matches!(c.source_spans[0], SourceSpan::Code { .. }));
+            assert_eq!(c.heading_path, Vec::<String>::new());
+            assert_eq!(c.chunker_version.0, "code-c-ast-v1");
+        }
+        match &chunks[0].source_spans[0] {
+            SourceSpan::Code { symbol, line_start, line_end, .. } => {
+                assert_eq!(symbol.as_deref(), Some("parse"));
+                assert_eq!((*line_start, *line_end), (1, 3));
+            }
+            _ => unreachable!(),
+        }
+    }
+
+    #[test]
+    fn oversize_unit_splits_into_parts_with_unique_ids() {
+        let body = (0..500).map(|i| format!("\tx{i} = {i};\n")).collect::<String>();
+        let code = format!("int big() {{\n{body}\n}}");
+        let doc = code_doc(&[("big", 1, 502, &code)]);
+        let chunks = CodeCAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len());
+        for c in &chunks {
+            match &c.source_spans[0] {
+                SourceSpan::Code { symbol, .. } => {
+                    assert!(symbol.as_deref().unwrap().starts_with("big [part "),
+                        "part-numbered symbol, got {symbol:?}");
+                }
+                _ => unreachable!(),
+            }
+        }
+        let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
+        let n = ids.len(); ids.sort_unstable(); ids.dedup();
+        assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
+    }
+
+    #[test]
+    fn non_code_doc_errors() {
+        use kebab_core::TextBlock;
+        let mut doc = code_doc(&[("parse", 1, 1, "int parse() {}")]);
+        doc.blocks = vec![Block::Paragraph(TextBlock {
+            common: CommonBlock {
+                block_id: kebab_core::BlockId("b".into()),
+                heading_path: vec![],
+                source_span: SourceSpan::Line { start: 1, end: 1 },
+            },
+            text: "x".into(), inlines: vec![],
+        })];
+        let err = CodeCAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
+        assert!(err.to_string().contains("CodeCAstV1Chunker"));
+    }
+
+    #[test]
+    fn deterministic_chunk_ids_1000() {
+        let doc = code_doc(&[("parse", 1, 2, "int parse() {}\n")]);
+        let base: Vec<String> = CodeCAstV1Chunker.chunk(&doc, &policy())
+            .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+        for _ in 0..1000 {
+            let again: Vec<String> = CodeCAstV1Chunker.chunk(&doc, &policy())
+                .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+            assert_eq!(again, base);
+        }
+    }
+
+    #[test]
+    fn policy_hash_matches_md_heading_v1() {
+        let p = policy();
+        assert_eq!(CodeCAstV1Chunker.policy_hash(&p),
+            crate::MdHeadingV1Chunker.policy_hash(&p));
+    }
+}
--- a/crates/kebab-chunk/src/code_cpp_ast_v1.rs
+++ b/crates/kebab-chunk/src/code_cpp_ast_v1.rs
@@ -0,0 +1,322 @@
+//! `code-cpp-ast-v1` — maps a tree-sitter-derived C++ AST
+//! `CanonicalDocument` (one `Block::Code` per semantic unit, each with
+//! `SourceSpan::Code`) to chunks 1:1. A unit longer than
+//! `AST_CHUNK_MAX_LINES` is split into `<symbol> [part i/N]` sub-chunks
+//! at blank-line paragraph boundaries (design §9.1 oversize fallback).
+//!
+//! tree-sitter is intentionally NOT a dependency here: AST work is
+//! parser-side (`kebab-parse-code`, design §6.3). This chunker only
+//! consumes the `CanonicalDocument`.
+//!
+//! `AST_CHUNK_MAX_LINES` is a constant matching
+//! `IngestCodeCfg::default().ast_chunk_max_lines` (200). Per-medium
+//! config threading needs a chunker registry (P+); same deviation
+//! pattern as `pdf-page-v1`'s pinned `chunker_version`
+//! (`tasks/HOTFIXES.md`).
+
+use kebab_core::{
+    Block, BlockId, CanonicalDocument, Chunk, ChunkPolicy, Chunker, ChunkerVersion, DocumentId,
+    SourceSpan, id_for_chunk,
+};
+
+const VERSION_LABEL: &str = "code-cpp-ast-v1";
+const BYTES_PER_TOKEN: usize = 3;
+const POLICY_HASH_HEX_LEN: usize = 16;
+const AST_CHUNK_MAX_LINES: u32 = 200;
+
+#[derive(Clone, Copy, Debug, Default)]
+pub struct CodeCppAstV1Chunker;
+
+impl Chunker for CodeCppAstV1Chunker {
+    fn chunker_version(&self) -> ChunkerVersion {
+        ChunkerVersion(VERSION_LABEL.to_string())
+    }
+
+    fn policy_hash(&self, policy: &ChunkPolicy) -> String {
+        let bytes = serde_json_canonicalizer::to_vec(policy)
+            .expect("canonical JSON serialization of ChunkPolicy must not fail");
+        let hex = blake3::hash(&bytes).to_hex().to_string();
+        hex[..POLICY_HASH_HEX_LEN].to_string()
+    }
+
+    fn chunk(
+        &self,
+        doc: &CanonicalDocument,
+        policy: &ChunkPolicy,
+    ) -> anyhow::Result<Vec<Chunk>> {
+        for b in &doc.blocks {
+            let c = match b {
+                Block::Code(c) => c,
+                _ => anyhow::bail!(
+                    "CodeCppAstV1Chunker only handles code docs (got non-Code block)"
+                ),
+            };
+            if !matches!(c.common.source_span, SourceSpan::Code { .. }) {
+                anyhow::bail!(
+                    "CodeCppAstV1Chunker only handles code docs (got non-Code source_span)"
+                );
+            }
+        }
+
+        let base_policy_hash = self.policy_hash(policy);
+        let chunker_version = self.chunker_version();
+        let mut out: Vec<Chunk> = Vec::new();
+
+        for b in &doc.blocks {
+            let cb = match b {
+                Block::Code(c) => c,
+                _ => unreachable!("validated above"),
+            };
+            let (ls, le, symbol, lang) = match &cb.common.source_span {
+                SourceSpan::Code { line_start, line_end, symbol, lang } => {
+                    (*line_start, *line_end, symbol.clone(), lang.clone())
+                }
+                _ => unreachable!("validated above"),
+            };
+            let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
+            let span_lines = le.saturating_sub(ls) + 1;
+
+            if span_lines <= AST_CHUNK_MAX_LINES {
+                let span = SourceSpan::Code {
+                    line_start: ls,
+                    line_end: le,
+                    symbol: symbol.clone(),
+                    lang: lang.clone(),
+                };
+                out.push(make_chunk(
+                    doc, &chunker_version, &block_ids, &base_policy_hash,
+                    None, span, cb.code.clone(),
+                ));
+            } else {
+                let parts = split_oversize(&cb.code);
+                let n = parts.len();
+                for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
+                    let part_ls = ls + off_start;
+                    let part_le = ls + off_end;
+                    let part_sym = symbol
+                        .as_ref()
+                        .map(|s| format!("{s} [part {}/{n}]", i + 1));
+                    let span = SourceSpan::Code {
+                        line_start: part_ls,
+                        line_end: part_le,
+                        symbol: part_sym,
+                        lang: lang.clone(),
+                    };
+                    out.push(make_chunk(
+                        doc, &chunker_version, &block_ids, &base_policy_hash,
+                        Some(part_ls), span, text,
+                    ));
+                }
+            }
+        }
+
+        tracing::debug!(
+            target: "kebab-chunk",
+            doc_id = %doc.doc_id,
+            chunks = out.len(),
+            "code-cpp-ast-v1 chunked",
+        );
+        Ok(out)
+    }
+}
+
+#[allow(clippy::too_many_arguments)]
+fn make_chunk(
+    doc: &CanonicalDocument,
+    chunker_version: &ChunkerVersion,
+    block_ids: &[BlockId],
+    base_policy_hash: &str,
+    split_key: Option<u32>,
+    span: SourceSpan,
+    text: String,
+) -> Chunk {
+    let id_hash = match split_key {
+        Some(k) => format!("{base_policy_hash}#L{k}"),
+        None => base_policy_hash.to_string(),
+    };
+    let chunk_id = id_for_chunk(&doc.doc_id, chunker_version, block_ids, &id_hash);
+    let token_estimate = text.len().div_ceil(BYTES_PER_TOKEN);
+    Chunk {
+        chunk_id,
+        doc_id: DocumentId(doc.doc_id.0.clone()),
+        block_ids: block_ids.to_vec(),
+        text,
+        heading_path: Vec::new(),
+        source_spans: vec![span],
+        token_estimate,
+        chunker_version: chunker_version.clone(),
+        policy_hash: base_policy_hash.to_string(),
+    }
+}
+
+/// Split an oversize unit at blank-line paragraph boundaries, greedily
+/// gluing paragraphs until ~`AST_CHUNK_MAX_LINES` lines accumulate.
+/// Returns `(line_offset_start, line_offset_end, text)` where offsets are
+/// 0-based within the unit (caller adds the unit's absolute `line_start`).
+fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
+    let lines: Vec<&str> = code.split('\n').collect();
+    let total = lines.len() as u32;
+    let mut out: Vec<(u32, u32, String)> = Vec::new();
+    let mut start: u32 = 0;
+    while start < total {
+        let mut end = (start + AST_CHUNK_MAX_LINES).min(total);
+        let floor = start + (AST_CHUNK_MAX_LINES * 4 / 5);
+        if end < total {
+            if let Some(b) = (floor.min(end)..end)
+                .rev()
+                .find(|&i| lines[i as usize].trim().is_empty())
+            {
+                end = b + 1;
+            }
+        }
+        let text = lines[start as usize..end as usize].join("\n");
+        out.push((start, end.saturating_sub(1), text));
+        start = end;
+    }
+    if out.is_empty() {
+        out.push((0, total.saturating_sub(1), code.to_string()));
+    }
+    out
+}
+
+#[cfg(test)]
+mod tests {
+    use super::*;
+    use kebab_core::{
+        Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+        SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance,
+        SourceType, TrustLevel, WorkspacePath,
+    };
+    use time::OffsetDateTime;
+
+    fn code_doc(units: &[(&str, u32, u32, &str)]) -> CanonicalDocument {
+        let wp = WorkspacePath("crates/x/src/a.cpp".into());
+        let aid = AssetId("a".repeat(64));
+        let pv = ParserVersion("code-cpp-v1".into());
+        let doc_id = id_for_doc(&wp, &aid, &pv);
+        let blocks = units
+            .iter()
+            .enumerate()
+            .map(|(i, (sym, ls, le, code))| {
+                let span = SourceSpan::Code {
+                    line_start: *ls,
+                    line_end: *le,
+                    symbol: Some((*sym).to_string()),
+                    lang: Some("cpp".into()),
+                };
+                let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+                Block::Code(CodeBlock {
+                    common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span },
+                    lang: Some("cpp".into()),
+                    code: (*code).to_string(),
+                })
+            })
+            .collect();
+        CanonicalDocument {
+            doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(),
+            lang: Lang("und".into()), blocks,
+            metadata: Metadata {
+                aliases: vec![], tags: vec![],
+                created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                source_type: SourceType::Note, trust_level: TrustLevel::Primary,
+                user_id_alias: None, user: Default::default(),
+                repo: Some("kebab".into()), git_branch: Some("main".into()),
+                git_commit: Some("0".repeat(40)), code_lang: Some("cpp".into()),
+            },
+            provenance: Provenance { events: vec![] },
+            parser_version: pv, schema_version: 1, doc_version: 1,
+            last_chunker_version: None, last_embedding_version: None,
+        }
+    }
+    fn policy() -> ChunkPolicy {
+        ChunkPolicy { target_tokens: 500, overlap_tokens: 80,
+            respect_markdown_headings: false,
+            chunker_version: ChunkerVersion(VERSION_LABEL.into()) }
+    }
+
+    #[test]
+    fn chunker_version_is_code_cpp_ast_v1() {
+        assert_eq!(CodeCppAstV1Chunker.chunker_version(),
+            ChunkerVersion("code-cpp-ast-v1".into()));
+    }
+
+    #[test]
+    fn one_chunk_per_unit_preserves_code_span() {
+        let doc = code_doc(&[
+            ("parse", 1, 3, "int parse() {\n\t// x\n}"),
+            ("print", 5, 7, "void print() {\n\t//\n\treturn;\n}"),
+        ]);
+        let chunks = CodeCppAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert_eq!(chunks.len(), 2);
+        for c in &chunks {
+            assert_eq!(c.source_spans.len(), 1);
+            assert!(matches!(c.source_spans[0], SourceSpan::Code { .. }));
+            assert_eq!(c.heading_path, Vec::<String>::new());
+            assert_eq!(c.chunker_version.0, "code-cpp-ast-v1");
+        }
+        match &chunks[0].source_spans[0] {
+            SourceSpan::Code { symbol, line_start, line_end, .. } => {
+                assert_eq!(symbol.as_deref(), Some("parse"));
+                assert_eq!((*line_start, *line_end), (1, 3));
+            }
+            _ => unreachable!(),
+        }
+    }
+
+    #[test]
+    fn oversize_unit_splits_into_parts_with_unique_ids() {
+        let body = (0..500).map(|i| format!("\tx{i} = {i};\n")).collect::<String>();
+        let code = format!("int big() {{\n{body}\n}}");
+        let doc = code_doc(&[("big", 1, 502, &code)]);
+        let chunks = CodeCppAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len());
+        for c in &chunks {
+            match &c.source_spans[0] {
+                SourceSpan::Code { symbol, .. } => {
+                    assert!(symbol.as_deref().unwrap().starts_with("big [part "),
+                        "part-numbered symbol, got {symbol:?}");
+                }
+                _ => unreachable!(),
+            }
+        }
+        let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
+        let n = ids.len(); ids.sort_unstable(); ids.dedup();
+        assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
+    }
+
+    #[test]
+    fn non_code_doc_errors() {
+        use kebab_core::TextBlock;
+        let mut doc = code_doc(&[("parse", 1, 1, "int parse() {}")]);
+        doc.blocks = vec![Block::Paragraph(TextBlock {
+            common: CommonBlock {
+                block_id: kebab_core::BlockId("b".into()),
+                heading_path: vec![],
+                source_span: SourceSpan::Line { start: 1, end: 1 },
+            },
+            text: "x".into(), inlines: vec![],
+        })];
+        let err = CodeCppAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
+        assert!(err.to_string().contains("CodeCppAstV1Chunker"));
+    }
+
+    #[test]
+    fn deterministic_chunk_ids_1000() {
+        let doc = code_doc(&[("parse", 1, 2, "int parse() {}\n")]);
+        let base: Vec<String> = CodeCppAstV1Chunker.chunk(&doc, &policy())
+            .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+        for _ in 0..1000 {
+            let again: Vec<String> = CodeCppAstV1Chunker.chunk(&doc, &policy())
+                .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+            assert_eq!(again, base);
+        }
+    }
+
+    #[test]
+    fn policy_hash_matches_md_heading_v1() {
+        let p = policy();
+        assert_eq!(CodeCppAstV1Chunker.policy_hash(&p),
+            crate::MdHeadingV1Chunker.policy_hash(&p));
+    }
+}
--- a/crates/kebab-chunk/src/code_go_ast_v1.rs
+++ b/crates/kebab-chunk/src/code_go_ast_v1.rs
@@ -0,0 +1,322 @@
+//! `code-go-ast-v1` — maps a tree-sitter-derived Go AST
+//! `CanonicalDocument` (one `Block::Code` per semantic unit, each with
+//! `SourceSpan::Code`) to chunks 1:1. A unit longer than
+//! `AST_CHUNK_MAX_LINES` is split into `<symbol> [part i/N]` sub-chunks
+//! at blank-line paragraph boundaries (design §9.1 oversize fallback).
+//!
+//! tree-sitter is intentionally NOT a dependency here: AST work is
+//! parser-side (`kebab-parse-code`, design §6.3). This chunker only
+//! consumes the `CanonicalDocument`.
+//!
+//! `AST_CHUNK_MAX_LINES` is a constant matching
+//! `IngestCodeCfg::default().ast_chunk_max_lines` (200). Per-medium
+//! config threading needs a chunker registry (P+); same deviation
+//! pattern as `pdf-page-v1`'s pinned `chunker_version`
+//! (`tasks/HOTFIXES.md`).
+
+use kebab_core::{
+    Block, BlockId, CanonicalDocument, Chunk, ChunkPolicy, Chunker, ChunkerVersion, DocumentId,
+    SourceSpan, id_for_chunk,
+};
+
+const VERSION_LABEL: &str = "code-go-ast-v1";
+const BYTES_PER_TOKEN: usize = 3;
+const POLICY_HASH_HEX_LEN: usize = 16;
+const AST_CHUNK_MAX_LINES: u32 = 200;
+
+#[derive(Clone, Copy, Debug, Default)]
+pub struct CodeGoAstV1Chunker;
+
+impl Chunker for CodeGoAstV1Chunker {
+    fn chunker_version(&self) -> ChunkerVersion {
+        ChunkerVersion(VERSION_LABEL.to_string())
+    }
+
+    fn policy_hash(&self, policy: &ChunkPolicy) -> String {
+        let bytes = serde_json_canonicalizer::to_vec(policy)
+            .expect("canonical JSON serialization of ChunkPolicy must not fail");
+        let hex = blake3::hash(&bytes).to_hex().to_string();
+        hex[..POLICY_HASH_HEX_LEN].to_string()
+    }
+
+    fn chunk(
+        &self,
+        doc: &CanonicalDocument,
+        policy: &ChunkPolicy,
+    ) -> anyhow::Result<Vec<Chunk>> {
+        for b in &doc.blocks {
+            let c = match b {
+                Block::Code(c) => c,
+                _ => anyhow::bail!(
+                    "CodeGoAstV1Chunker only handles code docs (got non-Code block)"
+                ),
+            };
+            if !matches!(c.common.source_span, SourceSpan::Code { .. }) {
+                anyhow::bail!(
+                    "CodeGoAstV1Chunker only handles code docs (got non-Code source_span)"
+                );
+            }
+        }
+
+        let base_policy_hash = self.policy_hash(policy);
+        let chunker_version = self.chunker_version();
+        let mut out: Vec<Chunk> = Vec::new();
+
+        for b in &doc.blocks {
+            let cb = match b {
+                Block::Code(c) => c,
+                _ => unreachable!("validated above"),
+            };
+            let (ls, le, symbol, lang) = match &cb.common.source_span {
+                SourceSpan::Code { line_start, line_end, symbol, lang } => {
+                    (*line_start, *line_end, symbol.clone(), lang.clone())
+                }
+                _ => unreachable!("validated above"),
+            };
+            let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
+            let span_lines = le.saturating_sub(ls) + 1;
+
+            if span_lines <= AST_CHUNK_MAX_LINES {
+                let span = SourceSpan::Code {
+                    line_start: ls,
+                    line_end: le,
+                    symbol: symbol.clone(),
+                    lang: lang.clone(),
+                };
+                out.push(make_chunk(
+                    doc, &chunker_version, &block_ids, &base_policy_hash,
+                    None, span, cb.code.clone(),
+                ));
+            } else {
+                let parts = split_oversize(&cb.code);
+                let n = parts.len();
+                for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
+                    let part_ls = ls + off_start;
+                    let part_le = ls + off_end;
+                    let part_sym = symbol
+                        .as_ref()
+                        .map(|s| format!("{s} [part {}/{n}]", i + 1));
+                    let span = SourceSpan::Code {
+                        line_start: part_ls,
+                        line_end: part_le,
+                        symbol: part_sym,
+                        lang: lang.clone(),
+                    };
+                    out.push(make_chunk(
+                        doc, &chunker_version, &block_ids, &base_policy_hash,
+                        Some(part_ls), span, text,
+                    ));
+                }
+            }
+        }
+
+        tracing::debug!(
+            target: "kebab-chunk",
+            doc_id = %doc.doc_id,
+            chunks = out.len(),
+            "code-go-ast-v1 chunked",
+        );
+        Ok(out)
+    }
+}
+
+#[allow(clippy::too_many_arguments)]
+fn make_chunk(
+    doc: &CanonicalDocument,
+    chunker_version: &ChunkerVersion,
+    block_ids: &[BlockId],
+    base_policy_hash: &str,
+    split_key: Option<u32>,
+    span: SourceSpan,
+    text: String,
+) -> Chunk {
+    let id_hash = match split_key {
+        Some(k) => format!("{base_policy_hash}#L{k}"),
+        None => base_policy_hash.to_string(),
+    };
+    let chunk_id = id_for_chunk(&doc.doc_id, chunker_version, block_ids, &id_hash);
+    let token_estimate = text.len().div_ceil(BYTES_PER_TOKEN);
+    Chunk {
+        chunk_id,
+        doc_id: DocumentId(doc.doc_id.0.clone()),
+        block_ids: block_ids.to_vec(),
+        text,
+        heading_path: Vec::new(),
+        source_spans: vec![span],
+        token_estimate,
+        chunker_version: chunker_version.clone(),
+        policy_hash: base_policy_hash.to_string(),
+    }
+}
+
+/// Split an oversize unit at blank-line paragraph boundaries, greedily
+/// gluing paragraphs until ~`AST_CHUNK_MAX_LINES` lines accumulate.
+/// Returns `(line_offset_start, line_offset_end, text)` where offsets are
+/// 0-based within the unit (caller adds the unit's absolute `line_start`).
+fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
+    let lines: Vec<&str> = code.split('\n').collect();
+    let total = lines.len() as u32;
+    let mut out: Vec<(u32, u32, String)> = Vec::new();
+    let mut start: u32 = 0;
+    while start < total {
+        let mut end = (start + AST_CHUNK_MAX_LINES).min(total);
+        let floor = start + (AST_CHUNK_MAX_LINES * 4 / 5);
+        if end < total {
+            if let Some(b) = (floor.min(end)..end)
+                .rev()
+                .find(|&i| lines[i as usize].trim().is_empty())
+            {
+                end = b + 1;
+            }
+        }
+        let text = lines[start as usize..end as usize].join("\n");
+        out.push((start, end.saturating_sub(1), text));
+        start = end;
+    }
+    if out.is_empty() {
+        out.push((0, total.saturating_sub(1), code.to_string()));
+    }
+    out
+}
+
+#[cfg(test)]
+mod tests {
+    use super::*;
+    use kebab_core::{
+        Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+        SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance,
+        SourceType, TrustLevel, WorkspacePath,
+    };
+    use time::OffsetDateTime;
+
+    fn code_doc(units: &[(&str, u32, u32, &str)]) -> CanonicalDocument {
+        let wp = WorkspacePath("crates/x/src/a.go".into());
+        let aid = AssetId("a".repeat(64));
+        let pv = ParserVersion("code-go-v1".into());
+        let doc_id = id_for_doc(&wp, &aid, &pv);
+        let blocks = units
+            .iter()
+            .enumerate()
+            .map(|(i, (sym, ls, le, code))| {
+                let span = SourceSpan::Code {
+                    line_start: *ls,
+                    line_end: *le,
+                    symbol: Some((*sym).to_string()),
+                    lang: Some("go".into()),
+                };
+                let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+                Block::Code(CodeBlock {
+                    common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span },
+                    lang: Some("go".into()),
+                    code: (*code).to_string(),
+                })
+            })
+            .collect();
+        CanonicalDocument {
+            doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(),
+            lang: Lang("und".into()), blocks,
+            metadata: Metadata {
+                aliases: vec![], tags: vec![],
+                created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                source_type: SourceType::Note, trust_level: TrustLevel::Primary,
+                user_id_alias: None, user: Default::default(),
+                repo: Some("kebab".into()), git_branch: Some("main".into()),
+                git_commit: Some("0".repeat(40)), code_lang: Some("go".into()),
+            },
+            provenance: Provenance { events: vec![] },
+            parser_version: pv, schema_version: 1, doc_version: 1,
+            last_chunker_version: None, last_embedding_version: None,
+        }
+    }
+    fn policy() -> ChunkPolicy {
+        ChunkPolicy { target_tokens: 500, overlap_tokens: 80,
+            respect_markdown_headings: false,
+            chunker_version: ChunkerVersion(VERSION_LABEL.into()) }
+    }
+
+    #[test]
+    fn chunker_version_is_code_go_ast_v1() {
+        assert_eq!(CodeGoAstV1Chunker.chunker_version(),
+            ChunkerVersion("code-go-ast-v1".into()));
+    }
+
+    #[test]
+    fn one_chunk_per_unit_preserves_code_span() {
+        let doc = code_doc(&[
+            ("parse", 1, 3, "func parse() {\n\t// x\n}"),
+            ("Foo.double", 5, 7, "func double() int {\n\t//\n\treturn 0\n}"),
+        ]);
+        let chunks = CodeGoAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert_eq!(chunks.len(), 2);
+        for c in &chunks {
+            assert_eq!(c.source_spans.len(), 1);
+            assert!(matches!(c.source_spans[0], SourceSpan::Code { .. }));
+            assert_eq!(c.heading_path, Vec::<String>::new());
+            assert_eq!(c.chunker_version.0, "code-go-ast-v1");
+        }
+        match &chunks[0].source_spans[0] {
+            SourceSpan::Code { symbol, line_start, line_end, .. } => {
+                assert_eq!(symbol.as_deref(), Some("parse"));
+                assert_eq!((*line_start, *line_end), (1, 3));
+            }
+            _ => unreachable!(),
+        }
+    }
+
+    #[test]
+    fn oversize_unit_splits_into_parts_with_unique_ids() {
+        let body = (0..500).map(|i| format!("\tx{i} := {i}")).collect::<Vec<_>>().join("\n");
+        let code = format!("func big() {{\n{body}\n}}");
+        let doc = code_doc(&[("big", 1, 502, &code)]);
+        let chunks = CodeGoAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len());
+        for c in &chunks {
+            match &c.source_spans[0] {
+                SourceSpan::Code { symbol, .. } => {
+                    assert!(symbol.as_deref().unwrap().starts_with("big [part "),
+                        "part-numbered symbol, got {symbol:?}");
+                }
+                _ => unreachable!(),
+            }
+        }
+        let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
+        let n = ids.len(); ids.sort_unstable(); ids.dedup();
+        assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
+    }
+
+    #[test]
+    fn non_code_doc_errors() {
+        use kebab_core::TextBlock;
+        let mut doc = code_doc(&[("parse", 1, 1, "func parse() {}")]);
+        doc.blocks = vec![Block::Paragraph(TextBlock {
+            common: CommonBlock {
+                block_id: kebab_core::BlockId("b".into()),
+                heading_path: vec![],
+                source_span: SourceSpan::Line { start: 1, end: 1 },
+            },
+            text: "x".into(), inlines: vec![],
+        })];
+        let err = CodeGoAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
+        assert!(err.to_string().contains("CodeGoAstV1Chunker"));
+    }
+
+    #[test]
+    fn deterministic_chunk_ids_1000() {
+        let doc = code_doc(&[("parse", 1, 2, "func parse() {}\n")]);
+        let base: Vec<String> = CodeGoAstV1Chunker.chunk(&doc, &policy())
+            .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+        for _ in 0..1000 {
+            let again: Vec<String> = CodeGoAstV1Chunker.chunk(&doc, &policy())
+                .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+            assert_eq!(again, base);
+        }
+    }
+
+    #[test]
+    fn policy_hash_matches_md_heading_v1() {
+        let p = policy();
+        assert_eq!(CodeGoAstV1Chunker.policy_hash(&p),
+            crate::MdHeadingV1Chunker.policy_hash(&p));
+    }
+}
--- a/crates/kebab-chunk/src/code_java_ast_v1.rs
+++ b/crates/kebab-chunk/src/code_java_ast_v1.rs
@@ -0,0 +1,322 @@
+//! `code-java-ast-v1` — maps a tree-sitter-derived Java AST
+//! `CanonicalDocument` (one `Block::Code` per semantic unit, each with
+//! `SourceSpan::Code`) to chunks 1:1. A unit longer than
+//! `AST_CHUNK_MAX_LINES` is split into `<symbol> [part i/N]` sub-chunks
+//! at blank-line paragraph boundaries (design §9.1 oversize fallback).
+//!
+//! tree-sitter is intentionally NOT a dependency here: AST work is
+//! parser-side (`kebab-parse-code`, design §6.3). This chunker only
+//! consumes the `CanonicalDocument`.
+//!
+//! `AST_CHUNK_MAX_LINES` is a constant matching
+//! `IngestCodeCfg::default().ast_chunk_max_lines` (200). Per-medium
+//! config threading needs a chunker registry (P+); same deviation
+//! pattern as `pdf-page-v1`'s pinned `chunker_version`
+//! (`tasks/HOTFIXES.md`).
+
+use kebab_core::{
+    Block, BlockId, CanonicalDocument, Chunk, ChunkPolicy, Chunker, ChunkerVersion, DocumentId,
+    SourceSpan, id_for_chunk,
+};
+
+const VERSION_LABEL: &str = "code-java-ast-v1";
+const BYTES_PER_TOKEN: usize = 3;
+const POLICY_HASH_HEX_LEN: usize = 16;
+const AST_CHUNK_MAX_LINES: u32 = 200;
+
+#[derive(Clone, Copy, Debug, Default)]
+pub struct CodeJavaAstV1Chunker;
+
+impl Chunker for CodeJavaAstV1Chunker {
+    fn chunker_version(&self) -> ChunkerVersion {
+        ChunkerVersion(VERSION_LABEL.to_string())
+    }
+
+    fn policy_hash(&self, policy: &ChunkPolicy) -> String {
+        let bytes = serde_json_canonicalizer::to_vec(policy)
+            .expect("canonical JSON serialization of ChunkPolicy must not fail");
+        let hex = blake3::hash(&bytes).to_hex().to_string();
+        hex[..POLICY_HASH_HEX_LEN].to_string()
+    }
+
+    fn chunk(
+        &self,
+        doc: &CanonicalDocument,
+        policy: &ChunkPolicy,
+    ) -> anyhow::Result<Vec<Chunk>> {
+        for b in &doc.blocks {
+            let c = match b {
+                Block::Code(c) => c,
+                _ => anyhow::bail!(
+                    "CodeJavaAstV1Chunker only handles code docs (got non-Code block)"
+                ),
+            };
+            if !matches!(c.common.source_span, SourceSpan::Code { .. }) {
+                anyhow::bail!(
+                    "CodeJavaAstV1Chunker only handles code docs (got non-Code source_span)"
+                );
+            }
+        }
+
+        let base_policy_hash = self.policy_hash(policy);
+        let chunker_version = self.chunker_version();
+        let mut out: Vec<Chunk> = Vec::new();
+
+        for b in &doc.blocks {
+            let cb = match b {
+                Block::Code(c) => c,
+                _ => unreachable!("validated above"),
+            };
+            let (ls, le, symbol, lang) = match &cb.common.source_span {
+                SourceSpan::Code { line_start, line_end, symbol, lang } => {
+                    (*line_start, *line_end, symbol.clone(), lang.clone())
+                }
+                _ => unreachable!("validated above"),
+            };
+            let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
+            let span_lines = le.saturating_sub(ls) + 1;
+
+            if span_lines <= AST_CHUNK_MAX_LINES {
+                let span = SourceSpan::Code {
+                    line_start: ls,
+                    line_end: le,
+                    symbol: symbol.clone(),
+                    lang: lang.clone(),
+                };
+                out.push(make_chunk(
+                    doc, &chunker_version, &block_ids, &base_policy_hash,
+                    None, span, cb.code.clone(),
+                ));
+            } else {
+                let parts = split_oversize(&cb.code);
+                let n = parts.len();
+                for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
+                    let part_ls = ls + off_start;
+                    let part_le = ls + off_end;
+                    let part_sym = symbol
+                        .as_ref()
+                        .map(|s| format!("{s} [part {}/{n}]", i + 1));
+                    let span = SourceSpan::Code {
+                        line_start: part_ls,
+                        line_end: part_le,
+                        symbol: part_sym,
+                        lang: lang.clone(),
+                    };
+                    out.push(make_chunk(
+                        doc, &chunker_version, &block_ids, &base_policy_hash,
+                        Some(part_ls), span, text,
+                    ));
+                }
+            }
+        }
+
+        tracing::debug!(
+            target: "kebab-chunk",
+            doc_id = %doc.doc_id,
+            chunks = out.len(),
+            "code-java-ast-v1 chunked",
+        );
+        Ok(out)
+    }
+}
+
+#[allow(clippy::too_many_arguments)]
+fn make_chunk(
+    doc: &CanonicalDocument,
+    chunker_version: &ChunkerVersion,
+    block_ids: &[BlockId],
+    base_policy_hash: &str,
+    split_key: Option<u32>,
+    span: SourceSpan,
+    text: String,
+) -> Chunk {
+    let id_hash = match split_key {
+        Some(k) => format!("{base_policy_hash}#L{k}"),
+        None => base_policy_hash.to_string(),
+    };
+    let chunk_id = id_for_chunk(&doc.doc_id, chunker_version, block_ids, &id_hash);
+    let token_estimate = text.len().div_ceil(BYTES_PER_TOKEN);
+    Chunk {
+        chunk_id,
+        doc_id: DocumentId(doc.doc_id.0.clone()),
+        block_ids: block_ids.to_vec(),
+        text,
+        heading_path: Vec::new(),
+        source_spans: vec![span],
+        token_estimate,
+        chunker_version: chunker_version.clone(),
+        policy_hash: base_policy_hash.to_string(),
+    }
+}
+
+/// Split an oversize unit at blank-line paragraph boundaries, greedily
+/// gluing paragraphs until ~`AST_CHUNK_MAX_LINES` lines accumulate.
+/// Returns `(line_offset_start, line_offset_end, text)` where offsets are
+/// 0-based within the unit (caller adds the unit's absolute `line_start`).
+fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
+    let lines: Vec<&str> = code.split('\n').collect();
+    let total = lines.len() as u32;
+    let mut out: Vec<(u32, u32, String)> = Vec::new();
+    let mut start: u32 = 0;
+    while start < total {
+        let mut end = (start + AST_CHUNK_MAX_LINES).min(total);
+        let floor = start + (AST_CHUNK_MAX_LINES * 4 / 5);
+        if end < total {
+            if let Some(b) = (floor.min(end)..end)
+                .rev()
+                .find(|&i| lines[i as usize].trim().is_empty())
+            {
+                end = b + 1;
+            }
+        }
+        let text = lines[start as usize..end as usize].join("\n");
+        out.push((start, end.saturating_sub(1), text));
+        start = end;
+    }
+    if out.is_empty() {
+        out.push((0, total.saturating_sub(1), code.to_string()));
+    }
+    out
+}
+
+#[cfg(test)]
+mod tests {
+    use super::*;
+    use kebab_core::{
+        Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+        SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance,
+        SourceType, TrustLevel, WorkspacePath,
+    };
+    use time::OffsetDateTime;
+
+    fn code_doc(units: &[(&str, u32, u32, &str)]) -> CanonicalDocument {
+        let wp = WorkspacePath("crates/x/src/Main.java".into());
+        let aid = AssetId("a".repeat(64));
+        let pv = ParserVersion("code-java-v1".into());
+        let doc_id = id_for_doc(&wp, &aid, &pv);
+        let blocks = units
+            .iter()
+            .enumerate()
+            .map(|(i, (sym, ls, le, code))| {
+                let span = SourceSpan::Code {
+                    line_start: *ls,
+                    line_end: *le,
+                    symbol: Some((*sym).to_string()),
+                    lang: Some("java".into()),
+                };
+                let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+                Block::Code(CodeBlock {
+                    common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span },
+                    lang: Some("java".into()),
+                    code: (*code).to_string(),
+                })
+            })
+            .collect();
+        CanonicalDocument {
+            doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(),
+            lang: Lang("und".into()), blocks,
+            metadata: Metadata {
+                aliases: vec![], tags: vec![],
+                created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                source_type: SourceType::Note, trust_level: TrustLevel::Primary,
+                user_id_alias: None, user: Default::default(),
+                repo: Some("kebab".into()), git_branch: Some("main".into()),
+                git_commit: Some("0".repeat(40)), code_lang: Some("java".into()),
+            },
+            provenance: Provenance { events: vec![] },
+            parser_version: pv, schema_version: 1, doc_version: 1,
+            last_chunker_version: None, last_embedding_version: None,
+        }
+    }
+    fn policy() -> ChunkPolicy {
+        ChunkPolicy { target_tokens: 500, overlap_tokens: 80,
+            respect_markdown_headings: false,
+            chunker_version: ChunkerVersion(VERSION_LABEL.into()) }
+    }
+
+    #[test]
+    fn chunker_version_is_code_java_ast_v1() {
+        assert_eq!(CodeJavaAstV1Chunker.chunker_version(),
+            ChunkerVersion("code-java-ast-v1".into()));
+    }
+
+    #[test]
+    fn one_chunk_per_unit_preserves_code_span() {
+        let doc = code_doc(&[
+            ("parse", 1, 3, "void parse() {\n\t// x\n}"),
+            ("Foo.double", 5, 7, "int double() {\n\t//\n\treturn 0;\n}"),
+        ]);
+        let chunks = CodeJavaAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert_eq!(chunks.len(), 2);
+        for c in &chunks {
+            assert_eq!(c.source_spans.len(), 1);
+            assert!(matches!(c.source_spans[0], SourceSpan::Code { .. }));
+            assert_eq!(c.heading_path, Vec::<String>::new());
+            assert_eq!(c.chunker_version.0, "code-java-ast-v1");
+        }
+        match &chunks[0].source_spans[0] {
+            SourceSpan::Code { symbol, line_start, line_end, .. } => {
+                assert_eq!(symbol.as_deref(), Some("parse"));
+                assert_eq!((*line_start, *line_end), (1, 3));
+            }
+            _ => unreachable!(),
+        }
+    }
+
+    #[test]
+    fn oversize_unit_splits_into_parts_with_unique_ids() {
+        let body = (0..500).map(|i| format!("\tint x{i} = {i};")).collect::<Vec<_>>().join("\n");
+        let code = format!("void big() {{\n{body}\n}}");
+        let doc = code_doc(&[("big", 1, 502, &code)]);
+        let chunks = CodeJavaAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len());
+        for c in &chunks {
+            match &c.source_spans[0] {
+                SourceSpan::Code { symbol, .. } => {
+                    assert!(symbol.as_deref().unwrap().starts_with("big [part "),
+                        "part-numbered symbol, got {symbol:?}");
+                }
+                _ => unreachable!(),
+            }
+        }
+        let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
+        let n = ids.len(); ids.sort_unstable(); ids.dedup();
+        assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
+    }
+
+    #[test]
+    fn non_code_doc_errors() {
+        use kebab_core::TextBlock;
+        let mut doc = code_doc(&[("parse", 1, 1, "void parse() {}")]);
+        doc.blocks = vec![Block::Paragraph(TextBlock {
+            common: CommonBlock {
+                block_id: kebab_core::BlockId("b".into()),
+                heading_path: vec![],
+                source_span: SourceSpan::Line { start: 1, end: 1 },
+            },
+            text: "x".into(), inlines: vec![],
+        })];
+        let err = CodeJavaAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
+        assert!(err.to_string().contains("CodeJavaAstV1Chunker"));
+    }
+
+    #[test]
+    fn deterministic_chunk_ids_1000() {
+        let doc = code_doc(&[("parse", 1, 2, "void parse() {}\n")]);
+        let base: Vec<String> = CodeJavaAstV1Chunker.chunk(&doc, &policy())
+            .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+        for _ in 0..1000 {
+            let again: Vec<String> = CodeJavaAstV1Chunker.chunk(&doc, &policy())
+                .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+            assert_eq!(again, base);
+        }
+    }
+
+    #[test]
+    fn policy_hash_matches_md_heading_v1() {
+        let p = policy();
+        assert_eq!(CodeJavaAstV1Chunker.policy_hash(&p),
+            crate::MdHeadingV1Chunker.policy_hash(&p));
+    }
+}
--- a/crates/kebab-chunk/src/code_js_ast_v1.rs
+++ b/crates/kebab-chunk/src/code_js_ast_v1.rs
@@ -0,0 +1,322 @@
+//! `code-js-ast-v1` — maps a tree-sitter-derived JavaScript AST
+//! `CanonicalDocument` (one `Block::Code` per semantic unit, each with
+//! `SourceSpan::Code`) to chunks 1:1. A unit longer than
+//! `AST_CHUNK_MAX_LINES` is split into `<symbol> [part i/N]` sub-chunks
+//! at blank-line paragraph boundaries (design §9.1 oversize fallback).
+//!
+//! tree-sitter is intentionally NOT a dependency here: AST work is
+//! parser-side (`kebab-parse-code`, design §6.3). This chunker only
+//! consumes the `CanonicalDocument`.
+//!
+//! `AST_CHUNK_MAX_LINES` is a constant matching
+//! `IngestCodeCfg::default().ast_chunk_max_lines` (200). Per-medium
+//! config threading needs a chunker registry (P+); same deviation
+//! pattern as `pdf-page-v1`'s pinned `chunker_version`
+//! (`tasks/HOTFIXES.md`).
+
+use kebab_core::{
+    Block, BlockId, CanonicalDocument, Chunk, ChunkPolicy, Chunker, ChunkerVersion, DocumentId,
+    SourceSpan, id_for_chunk,
+};
+
+const VERSION_LABEL: &str = "code-js-ast-v1";
+const BYTES_PER_TOKEN: usize = 3;
+const POLICY_HASH_HEX_LEN: usize = 16;
+const AST_CHUNK_MAX_LINES: u32 = 200;
+
+#[derive(Clone, Copy, Debug, Default)]
+pub struct CodeJsAstV1Chunker;
+
+impl Chunker for CodeJsAstV1Chunker {
+    fn chunker_version(&self) -> ChunkerVersion {
+        ChunkerVersion(VERSION_LABEL.to_string())
+    }
+
+    fn policy_hash(&self, policy: &ChunkPolicy) -> String {
+        let bytes = serde_json_canonicalizer::to_vec(policy)
+            .expect("canonical JSON serialization of ChunkPolicy must not fail");
+        let hex = blake3::hash(&bytes).to_hex().to_string();
+        hex[..POLICY_HASH_HEX_LEN].to_string()
+    }
+
+    fn chunk(
+        &self,
+        doc: &CanonicalDocument,
+        policy: &ChunkPolicy,
+    ) -> anyhow::Result<Vec<Chunk>> {
+        for b in &doc.blocks {
+            let c = match b {
+                Block::Code(c) => c,
+                _ => anyhow::bail!(
+                    "CodeJsAstV1Chunker only handles code docs (got non-Code block)"
+                ),
+            };
+            if !matches!(c.common.source_span, SourceSpan::Code { .. }) {
+                anyhow::bail!(
+                    "CodeJsAstV1Chunker only handles code docs (got non-Code source_span)"
+                );
+            }
+        }
+
+        let base_policy_hash = self.policy_hash(policy);
+        let chunker_version = self.chunker_version();
+        let mut out: Vec<Chunk> = Vec::new();
+
+        for b in &doc.blocks {
+            let cb = match b {
+                Block::Code(c) => c,
+                _ => unreachable!("validated above"),
+            };
+            let (ls, le, symbol, lang) = match &cb.common.source_span {
+                SourceSpan::Code { line_start, line_end, symbol, lang } => {
+                    (*line_start, *line_end, symbol.clone(), lang.clone())
+                }
+                _ => unreachable!("validated above"),
+            };
+            let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
+            let span_lines = le.saturating_sub(ls) + 1;
+
+            if span_lines <= AST_CHUNK_MAX_LINES {
+                let span = SourceSpan::Code {
+                    line_start: ls,
+                    line_end: le,
+                    symbol: symbol.clone(),
+                    lang: lang.clone(),
+                };
+                out.push(make_chunk(
+                    doc, &chunker_version, &block_ids, &base_policy_hash,
+                    None, span, cb.code.clone(),
+                ));
+            } else {
+                let parts = split_oversize(&cb.code);
+                let n = parts.len();
+                for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
+                    let part_ls = ls + off_start;
+                    let part_le = ls + off_end;
+                    let part_sym = symbol
+                        .as_ref()
+                        .map(|s| format!("{s} [part {}/{n}]", i + 1));
+                    let span = SourceSpan::Code {
+                        line_start: part_ls,
+                        line_end: part_le,
+                        symbol: part_sym,
+                        lang: lang.clone(),
+                    };
+                    out.push(make_chunk(
+                        doc, &chunker_version, &block_ids, &base_policy_hash,
+                        Some(part_ls), span, text,
+                    ));
+                }
+            }
+        }
+
+        tracing::debug!(
+            target: "kebab-chunk",
+            doc_id = %doc.doc_id,
+            chunks = out.len(),
+            "code-js-ast-v1 chunked",
+        );
+        Ok(out)
+    }
+}
+
+#[allow(clippy::too_many_arguments)]
+fn make_chunk(
+    doc: &CanonicalDocument,
+    chunker_version: &ChunkerVersion,
+    block_ids: &[BlockId],
+    base_policy_hash: &str,
+    split_key: Option<u32>,
+    span: SourceSpan,
+    text: String,
+) -> Chunk {
+    let id_hash = match split_key {
+        Some(k) => format!("{base_policy_hash}#L{k}"),
+        None => base_policy_hash.to_string(),
+    };
+    let chunk_id = id_for_chunk(&doc.doc_id, chunker_version, block_ids, &id_hash);
+    let token_estimate = text.len().div_ceil(BYTES_PER_TOKEN);
+    Chunk {
+        chunk_id,
+        doc_id: DocumentId(doc.doc_id.0.clone()),
+        block_ids: block_ids.to_vec(),
+        text,
+        heading_path: Vec::new(),
+        source_spans: vec![span],
+        token_estimate,
+        chunker_version: chunker_version.clone(),
+        policy_hash: base_policy_hash.to_string(),
+    }
+}
+
+/// Split an oversize unit at blank-line paragraph boundaries, greedily
+/// gluing paragraphs until ~`AST_CHUNK_MAX_LINES` lines accumulate.
+/// Returns `(line_offset_start, line_offset_end, text)` where offsets are
+/// 0-based within the unit (caller adds the unit's absolute `line_start`).
+fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
+    let lines: Vec<&str> = code.split('\n').collect();
+    let total = lines.len() as u32;
+    let mut out: Vec<(u32, u32, String)> = Vec::new();
+    let mut start: u32 = 0;
+    while start < total {
+        let mut end = (start + AST_CHUNK_MAX_LINES).min(total);
+        let floor = start + (AST_CHUNK_MAX_LINES * 4 / 5);
+        if end < total {
+            if let Some(b) = (floor.min(end)..end)
+                .rev()
+                .find(|&i| lines[i as usize].trim().is_empty())
+            {
+                end = b + 1;
+            }
+        }
+        let text = lines[start as usize..end as usize].join("\n");
+        out.push((start, end.saturating_sub(1), text));
+        start = end;
+    }
+    if out.is_empty() {
+        out.push((0, total.saturating_sub(1), code.to_string()));
+    }
+    out
+}
+
+#[cfg(test)]
+mod tests {
+    use super::*;
+    use kebab_core::{
+        Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+        SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance,
+        SourceType, TrustLevel, WorkspacePath,
+    };
+    use time::OffsetDateTime;
+
+    fn code_doc(units: &[(&str, u32, u32, &str)]) -> CanonicalDocument {
+        let wp = WorkspacePath("crates/x/src/a.js".into());
+        let aid = AssetId("a".repeat(64));
+        let pv = ParserVersion("code-js-v1".into());
+        let doc_id = id_for_doc(&wp, &aid, &pv);
+        let blocks = units
+            .iter()
+            .enumerate()
+            .map(|(i, (sym, ls, le, code))| {
+                let span = SourceSpan::Code {
+                    line_start: *ls,
+                    line_end: *le,
+                    symbol: Some((*sym).to_string()),
+                    lang: Some("javascript".into()),
+                };
+                let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+                Block::Code(CodeBlock {
+                    common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span },
+                    lang: Some("javascript".into()),
+                    code: (*code).to_string(),
+                })
+            })
+            .collect();
+        CanonicalDocument {
+            doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(),
+            lang: Lang("und".into()), blocks,
+            metadata: Metadata {
+                aliases: vec![], tags: vec![],
+                created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                source_type: SourceType::Note, trust_level: TrustLevel::Primary,
+                user_id_alias: None, user: Default::default(),
+                repo: Some("kebab".into()), git_branch: Some("main".into()),
+                git_commit: Some("0".repeat(40)), code_lang: Some("javascript".into()),
+            },
+            provenance: Provenance { events: vec![] },
+            parser_version: pv, schema_version: 1, doc_version: 1,
+            last_chunker_version: None, last_embedding_version: None,
+        }
+    }
+    fn policy() -> ChunkPolicy {
+        ChunkPolicy { target_tokens: 500, overlap_tokens: 80,
+            respect_markdown_headings: false,
+            chunker_version: ChunkerVersion(VERSION_LABEL.into()) }
+    }
+
+    #[test]
+    fn chunker_version_is_code_js_ast_v1() {
+        assert_eq!(CodeJsAstV1Chunker.chunker_version(),
+            ChunkerVersion("code-js-ast-v1".into()));
+    }
+
+    #[test]
+    fn one_chunk_per_unit_preserves_code_span() {
+        let doc = code_doc(&[
+            ("parse", 1, 3, "function parse() {\n    // x\n}"),
+            ("Foo.double", 5, 7, "function double() {\n    //\n    return 0;\n}"),
+        ]);
+        let chunks = CodeJsAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert_eq!(chunks.len(), 2);
+        for c in &chunks {
+            assert_eq!(c.source_spans.len(), 1);
+            assert!(matches!(c.source_spans[0], SourceSpan::Code { .. }));
+            assert_eq!(c.heading_path, Vec::<String>::new());
+            assert_eq!(c.chunker_version.0, "code-js-ast-v1");
+        }
+        match &chunks[0].source_spans[0] {
+            SourceSpan::Code { symbol, line_start, line_end, .. } => {
+                assert_eq!(symbol.as_deref(), Some("parse"));
+                assert_eq!((*line_start, *line_end), (1, 3));
+            }
+            _ => unreachable!(),
+        }
+    }
+
+    #[test]
+    fn oversize_unit_splits_into_parts_with_unique_ids() {
+        let body = (0..500).map(|i| format!("    const x{i} = {i};")).collect::<Vec<_>>().join("\n");
+        let code = format!("function big() {{\n{body}\n}}");
+        let doc = code_doc(&[("big", 1, 502, &code)]);
+        let chunks = CodeJsAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len());
+        for c in &chunks {
+            match &c.source_spans[0] {
+                SourceSpan::Code { symbol, .. } => {
+                    assert!(symbol.as_deref().unwrap().starts_with("big [part "),
+                        "part-numbered symbol, got {symbol:?}");
+                }
+                _ => unreachable!(),
+            }
+        }
+        let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
+        let n = ids.len(); ids.sort_unstable(); ids.dedup();
+        assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
+    }
+
+    #[test]
+    fn non_code_doc_errors() {
+        use kebab_core::TextBlock;
+        let mut doc = code_doc(&[("parse", 1, 1, "function parse() {}")]);
+        doc.blocks = vec![Block::Paragraph(TextBlock {
+            common: CommonBlock {
+                block_id: kebab_core::BlockId("b".into()),
+                heading_path: vec![],
+                source_span: SourceSpan::Line { start: 1, end: 1 },
+            },
+            text: "x".into(), inlines: vec![],
+        })];
+        let err = CodeJsAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
+        assert!(err.to_string().contains("CodeJsAstV1Chunker"));
+    }
+
+    #[test]
+    fn deterministic_chunk_ids_1000() {
+        let doc = code_doc(&[("parse", 1, 2, "function parse() {}\n")]);
+        let base: Vec<String> = CodeJsAstV1Chunker.chunk(&doc, &policy())
+            .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+        for _ in 0..1000 {
+            let again: Vec<String> = CodeJsAstV1Chunker.chunk(&doc, &policy())
+                .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+            assert_eq!(again, base);
+        }
+    }
+
+    #[test]
+    fn policy_hash_matches_md_heading_v1() {
+        let p = policy();
+        assert_eq!(CodeJsAstV1Chunker.policy_hash(&p),
+            crate::MdHeadingV1Chunker.policy_hash(&p));
+    }
+}
--- a/crates/kebab-chunk/src/code_kotlin_ast_v1.rs
+++ b/crates/kebab-chunk/src/code_kotlin_ast_v1.rs
@@ -0,0 +1,322 @@
+//! `code-kotlin-ast-v1` — maps a tree-sitter-derived Kotlin AST
+//! `CanonicalDocument` (one `Block::Code` per semantic unit, each with
+//! `SourceSpan::Code`) to chunks 1:1. A unit longer than
+//! `AST_CHUNK_MAX_LINES` is split into `<symbol> [part i/N]` sub-chunks
+//! at blank-line paragraph boundaries (design §9.1 oversize fallback).
+//!
+//! tree-sitter is intentionally NOT a dependency here: AST work is
+//! parser-side (`kebab-parse-code`, design §6.3). This chunker only
+//! consumes the `CanonicalDocument`.
+//!
+//! `AST_CHUNK_MAX_LINES` is a constant matching
+//! `IngestCodeCfg::default().ast_chunk_max_lines` (200). Per-medium
+//! config threading needs a chunker registry (P+); same deviation
+//! pattern as `pdf-page-v1`'s pinned `chunker_version`
+//! (`tasks/HOTFIXES.md`).
+
+use kebab_core::{
+    Block, BlockId, CanonicalDocument, Chunk, ChunkPolicy, Chunker, ChunkerVersion, DocumentId,
+    SourceSpan, id_for_chunk,
+};
+
+const VERSION_LABEL: &str = "code-kotlin-ast-v1";
+const BYTES_PER_TOKEN: usize = 3;
+const POLICY_HASH_HEX_LEN: usize = 16;
+const AST_CHUNK_MAX_LINES: u32 = 200;
+
+#[derive(Clone, Copy, Debug, Default)]
+pub struct CodeKotlinAstV1Chunker;
+
+impl Chunker for CodeKotlinAstV1Chunker {
+    fn chunker_version(&self) -> ChunkerVersion {
+        ChunkerVersion(VERSION_LABEL.to_string())
+    }
+
+    fn policy_hash(&self, policy: &ChunkPolicy) -> String {
+        let bytes = serde_json_canonicalizer::to_vec(policy)
+            .expect("canonical JSON serialization of ChunkPolicy must not fail");
+        let hex = blake3::hash(&bytes).to_hex().to_string();
+        hex[..POLICY_HASH_HEX_LEN].to_string()
+    }
+
+    fn chunk(
+        &self,
+        doc: &CanonicalDocument,
+        policy: &ChunkPolicy,
+    ) -> anyhow::Result<Vec<Chunk>> {
+        for b in &doc.blocks {
+            let c = match b {
+                Block::Code(c) => c,
+                _ => anyhow::bail!(
+                    "CodeKotlinAstV1Chunker only handles code docs (got non-Code block)"
+                ),
+            };
+            if !matches!(c.common.source_span, SourceSpan::Code { .. }) {
+                anyhow::bail!(
+                    "CodeKotlinAstV1Chunker only handles code docs (got non-Code source_span)"
+                );
+            }
+        }
+
+        let base_policy_hash = self.policy_hash(policy);
+        let chunker_version = self.chunker_version();
+        let mut out: Vec<Chunk> = Vec::new();
+
+        for b in &doc.blocks {
+            let cb = match b {
+                Block::Code(c) => c,
+                _ => unreachable!("validated above"),
+            };
+            let (ls, le, symbol, lang) = match &cb.common.source_span {
+                SourceSpan::Code { line_start, line_end, symbol, lang } => {
+                    (*line_start, *line_end, symbol.clone(), lang.clone())
+                }
+                _ => unreachable!("validated above"),
+            };
+            let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
+            let span_lines = le.saturating_sub(ls) + 1;
+
+            if span_lines <= AST_CHUNK_MAX_LINES {
+                let span = SourceSpan::Code {
+                    line_start: ls,
+                    line_end: le,
+                    symbol: symbol.clone(),
+                    lang: lang.clone(),
+                };
+                out.push(make_chunk(
+                    doc, &chunker_version, &block_ids, &base_policy_hash,
+                    None, span, cb.code.clone(),
+                ));
+            } else {
+                let parts = split_oversize(&cb.code);
+                let n = parts.len();
+                for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
+                    let part_ls = ls + off_start;
+                    let part_le = ls + off_end;
+                    let part_sym = symbol
+                        .as_ref()
+                        .map(|s| format!("{s} [part {}/{n}]", i + 1));
+                    let span = SourceSpan::Code {
+                        line_start: part_ls,
+                        line_end: part_le,
+                        symbol: part_sym,
+                        lang: lang.clone(),
+                    };
+                    out.push(make_chunk(
+                        doc, &chunker_version, &block_ids, &base_policy_hash,
+                        Some(part_ls), span, text,
+                    ));
+                }
+            }
+        }
+
+        tracing::debug!(
+            target: "kebab-chunk",
+            doc_id = %doc.doc_id,
+            chunks = out.len(),
+            "code-kotlin-ast-v1 chunked",
+        );
+        Ok(out)
+    }
+}
+
+#[allow(clippy::too_many_arguments)]
+fn make_chunk(
+    doc: &CanonicalDocument,
+    chunker_version: &ChunkerVersion,
+    block_ids: &[BlockId],
+    base_policy_hash: &str,
+    split_key: Option<u32>,
+    span: SourceSpan,
+    text: String,
+) -> Chunk {
+    let id_hash = match split_key {
+        Some(k) => format!("{base_policy_hash}#L{k}"),
+        None => base_policy_hash.to_string(),
+    };
+    let chunk_id = id_for_chunk(&doc.doc_id, chunker_version, block_ids, &id_hash);
+    let token_estimate = text.len().div_ceil(BYTES_PER_TOKEN);
+    Chunk {
+        chunk_id,
+        doc_id: DocumentId(doc.doc_id.0.clone()),
+        block_ids: block_ids.to_vec(),
+        text,
+        heading_path: Vec::new(),
+        source_spans: vec![span],
+        token_estimate,
+        chunker_version: chunker_version.clone(),
+        policy_hash: base_policy_hash.to_string(),
+    }
+}
+
+/// Split an oversize unit at blank-line paragraph boundaries, greedily
+/// gluing paragraphs until ~`AST_CHUNK_MAX_LINES` lines accumulate.
+/// Returns `(line_offset_start, line_offset_end, text)` where offsets are
+/// 0-based within the unit (caller adds the unit's absolute `line_start`).
+fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
+    let lines: Vec<&str> = code.split('\n').collect();
+    let total = lines.len() as u32;
+    let mut out: Vec<(u32, u32, String)> = Vec::new();
+    let mut start: u32 = 0;
+    while start < total {
+        let mut end = (start + AST_CHUNK_MAX_LINES).min(total);
+        let floor = start + (AST_CHUNK_MAX_LINES * 4 / 5);
+        if end < total {
+            if let Some(b) = (floor.min(end)..end)
+                .rev()
+                .find(|&i| lines[i as usize].trim().is_empty())
+            {
+                end = b + 1;
+            }
+        }
+        let text = lines[start as usize..end as usize].join("\n");
+        out.push((start, end.saturating_sub(1), text));
+        start = end;
+    }
+    if out.is_empty() {
+        out.push((0, total.saturating_sub(1), code.to_string()));
+    }
+    out
+}
+
+#[cfg(test)]
+mod tests {
+    use super::*;
+    use kebab_core::{
+        Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+        SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance,
+        SourceType, TrustLevel, WorkspacePath,
+    };
+    use time::OffsetDateTime;
+
+    fn code_doc(units: &[(&str, u32, u32, &str)]) -> CanonicalDocument {
+        let wp = WorkspacePath("crates/x/src/Main.kt".into());
+        let aid = AssetId("a".repeat(64));
+        let pv = ParserVersion("code-kotlin-v1".into());
+        let doc_id = id_for_doc(&wp, &aid, &pv);
+        let blocks = units
+            .iter()
+            .enumerate()
+            .map(|(i, (sym, ls, le, code))| {
+                let span = SourceSpan::Code {
+                    line_start: *ls,
+                    line_end: *le,
+                    symbol: Some((*sym).to_string()),
+                    lang: Some("kotlin".into()),
+                };
+                let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+                Block::Code(CodeBlock {
+                    common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span },
+                    lang: Some("kotlin".into()),
+                    code: (*code).to_string(),
+                })
+            })
+            .collect();
+        CanonicalDocument {
+            doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(),
+            lang: Lang("und".into()), blocks,
+            metadata: Metadata {
+                aliases: vec![], tags: vec![],
+                created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                source_type: SourceType::Note, trust_level: TrustLevel::Primary,
+                user_id_alias: None, user: Default::default(),
+                repo: Some("kebab".into()), git_branch: Some("main".into()),
+                git_commit: Some("0".repeat(40)), code_lang: Some("kotlin".into()),
+            },
+            provenance: Provenance { events: vec![] },
+            parser_version: pv, schema_version: 1, doc_version: 1,
+            last_chunker_version: None, last_embedding_version: None,
+        }
+    }
+    fn policy() -> ChunkPolicy {
+        ChunkPolicy { target_tokens: 500, overlap_tokens: 80,
+            respect_markdown_headings: false,
+            chunker_version: ChunkerVersion(VERSION_LABEL.into()) }
+    }
+
+    #[test]
+    fn chunker_version_is_code_kotlin_ast_v1() {
+        assert_eq!(CodeKotlinAstV1Chunker.chunker_version(),
+            ChunkerVersion("code-kotlin-ast-v1".into()));
+    }
+
+    #[test]
+    fn one_chunk_per_unit_preserves_code_span() {
+        let doc = code_doc(&[
+            ("parse", 1, 3, "fun parse() {\n\t// x\n}"),
+            ("Foo.double", 5, 7, "fun double(): Int {\n\t//\n\treturn 0\n}"),
+        ]);
+        let chunks = CodeKotlinAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert_eq!(chunks.len(), 2);
+        for c in &chunks {
+            assert_eq!(c.source_spans.len(), 1);
+            assert!(matches!(c.source_spans[0], SourceSpan::Code { .. }));
+            assert_eq!(c.heading_path, Vec::<String>::new());
+            assert_eq!(c.chunker_version.0, "code-kotlin-ast-v1");
+        }
+        match &chunks[0].source_spans[0] {
+            SourceSpan::Code { symbol, line_start, line_end, .. } => {
+                assert_eq!(symbol.as_deref(), Some("parse"));
+                assert_eq!((*line_start, *line_end), (1, 3));
+            }
+            _ => unreachable!(),
+        }
+    }
+
+    #[test]
+    fn oversize_unit_splits_into_parts_with_unique_ids() {
+        let body = (0..500).map(|i| format!("\tval x{i} = {i}")).collect::<Vec<_>>().join("\n");
+        let code = format!("fun big() {{\n{body}\n}}");
+        let doc = code_doc(&[("big", 1, 502, &code)]);
+        let chunks = CodeKotlinAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len());
+        for c in &chunks {
+            match &c.source_spans[0] {
+                SourceSpan::Code { symbol, .. } => {
+                    assert!(symbol.as_deref().unwrap().starts_with("big [part "),
+                        "part-numbered symbol, got {symbol:?}");
+                }
+                _ => unreachable!(),
+            }
+        }
+        let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
+        let n = ids.len(); ids.sort_unstable(); ids.dedup();
+        assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
+    }
+
+    #[test]
+    fn non_code_doc_errors() {
+        use kebab_core::TextBlock;
+        let mut doc = code_doc(&[("parse", 1, 1, "fun parse() {}")]);
+        doc.blocks = vec![Block::Paragraph(TextBlock {
+            common: CommonBlock {
+                block_id: kebab_core::BlockId("b".into()),
+                heading_path: vec![],
+                source_span: SourceSpan::Line { start: 1, end: 1 },
+            },
+            text: "x".into(), inlines: vec![],
+        })];
+        let err = CodeKotlinAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
+        assert!(err.to_string().contains("CodeKotlinAstV1Chunker"));
+    }
+
+    #[test]
+    fn deterministic_chunk_ids_1000() {
+        let doc = code_doc(&[("parse", 1, 2, "fun parse() {}\n")]);
+        let base: Vec<String> = CodeKotlinAstV1Chunker.chunk(&doc, &policy())
+            .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+        for _ in 0..1000 {
+            let again: Vec<String> = CodeKotlinAstV1Chunker.chunk(&doc, &policy())
+                .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+            assert_eq!(again, base);
+        }
+    }
+
+    #[test]
+    fn policy_hash_matches_md_heading_v1() {
+        let p = policy();
+        assert_eq!(CodeKotlinAstV1Chunker.policy_hash(&p),
+            crate::MdHeadingV1Chunker.policy_hash(&p));
+    }
+}
--- a/crates/kebab-chunk/src/code_python_ast_v1.rs
+++ b/crates/kebab-chunk/src/code_python_ast_v1.rs
@@ -0,0 +1,322 @@
+//! `code-python-ast-v1` — maps a tree-sitter-derived Python AST
+//! `CanonicalDocument` (one `Block::Code` per semantic unit, each with
+//! `SourceSpan::Code`) to chunks 1:1. A unit longer than
+//! `AST_CHUNK_MAX_LINES` is split into `<symbol> [part i/N]` sub-chunks
+//! at blank-line paragraph boundaries (design §9.1 oversize fallback).
+//!
+//! tree-sitter is intentionally NOT a dependency here: AST work is
+//! parser-side (`kebab-parse-code`, design §6.3). This chunker only
+//! consumes the `CanonicalDocument`.
+//!
+//! `AST_CHUNK_MAX_LINES` is a constant matching
+//! `IngestCodeCfg::default().ast_chunk_max_lines` (200). Per-medium
+//! config threading needs a chunker registry (P+); same deviation
+//! pattern as `pdf-page-v1`'s pinned `chunker_version`
+//! (`tasks/HOTFIXES.md`).
+
+use kebab_core::{
+    Block, BlockId, CanonicalDocument, Chunk, ChunkPolicy, Chunker, ChunkerVersion, DocumentId,
+    SourceSpan, id_for_chunk,
+};
+
+const VERSION_LABEL: &str = "code-python-ast-v1";
+const BYTES_PER_TOKEN: usize = 3;
+const POLICY_HASH_HEX_LEN: usize = 16;
+const AST_CHUNK_MAX_LINES: u32 = 200;
+
+#[derive(Clone, Copy, Debug, Default)]
+pub struct CodePythonAstV1Chunker;
+
+impl Chunker for CodePythonAstV1Chunker {
+    fn chunker_version(&self) -> ChunkerVersion {
+        ChunkerVersion(VERSION_LABEL.to_string())
+    }
+
+    fn policy_hash(&self, policy: &ChunkPolicy) -> String {
+        let bytes = serde_json_canonicalizer::to_vec(policy)
+            .expect("canonical JSON serialization of ChunkPolicy must not fail");
+        let hex = blake3::hash(&bytes).to_hex().to_string();
+        hex[..POLICY_HASH_HEX_LEN].to_string()
+    }
+
+    fn chunk(
+        &self,
+        doc: &CanonicalDocument,
+        policy: &ChunkPolicy,
+    ) -> anyhow::Result<Vec<Chunk>> {
+        for b in &doc.blocks {
+            let c = match b {
+                Block::Code(c) => c,
+                _ => anyhow::bail!(
+                    "CodePythonAstV1Chunker only handles code docs (got non-Code block)"
+                ),
+            };
+            if !matches!(c.common.source_span, SourceSpan::Code { .. }) {
+                anyhow::bail!(
+                    "CodePythonAstV1Chunker only handles code docs (got non-Code source_span)"
+                );
+            }
+        }
+
+        let base_policy_hash = self.policy_hash(policy);
+        let chunker_version = self.chunker_version();
+        let mut out: Vec<Chunk> = Vec::new();
+
+        for b in &doc.blocks {
+            let cb = match b {
+                Block::Code(c) => c,
+                _ => unreachable!("validated above"),
+            };
+            let (ls, le, symbol, lang) = match &cb.common.source_span {
+                SourceSpan::Code { line_start, line_end, symbol, lang } => {
+                    (*line_start, *line_end, symbol.clone(), lang.clone())
+                }
+                _ => unreachable!("validated above"),
+            };
+            let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
+            let span_lines = le.saturating_sub(ls) + 1;
+
+            if span_lines <= AST_CHUNK_MAX_LINES {
+                let span = SourceSpan::Code {
+                    line_start: ls,
+                    line_end: le,
+                    symbol: symbol.clone(),
+                    lang: lang.clone(),
+                };
+                out.push(make_chunk(
+                    doc, &chunker_version, &block_ids, &base_policy_hash,
+                    None, span, cb.code.clone(),
+                ));
+            } else {
+                let parts = split_oversize(&cb.code);
+                let n = parts.len();
+                for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
+                    let part_ls = ls + off_start;
+                    let part_le = ls + off_end;
+                    let part_sym = symbol
+                        .as_ref()
+                        .map(|s| format!("{s} [part {}/{n}]", i + 1));
+                    let span = SourceSpan::Code {
+                        line_start: part_ls,
+                        line_end: part_le,
+                        symbol: part_sym,
+                        lang: lang.clone(),
+                    };
+                    out.push(make_chunk(
+                        doc, &chunker_version, &block_ids, &base_policy_hash,
+                        Some(part_ls), span, text,
+                    ));
+                }
+            }
+        }
+
+        tracing::debug!(
+            target: "kebab-chunk",
+            doc_id = %doc.doc_id,
+            chunks = out.len(),
+            "code-python-ast-v1 chunked",
+        );
+        Ok(out)
+    }
+}
+
+#[allow(clippy::too_many_arguments)]
+fn make_chunk(
+    doc: &CanonicalDocument,
+    chunker_version: &ChunkerVersion,
+    block_ids: &[BlockId],
+    base_policy_hash: &str,
+    split_key: Option<u32>,
+    span: SourceSpan,
+    text: String,
+) -> Chunk {
+    let id_hash = match split_key {
+        Some(k) => format!("{base_policy_hash}#L{k}"),
+        None => base_policy_hash.to_string(),
+    };
+    let chunk_id = id_for_chunk(&doc.doc_id, chunker_version, block_ids, &id_hash);
+    let token_estimate = text.len().div_ceil(BYTES_PER_TOKEN);
+    Chunk {
+        chunk_id,
+        doc_id: DocumentId(doc.doc_id.0.clone()),
+        block_ids: block_ids.to_vec(),
+        text,
+        heading_path: Vec::new(),
+        source_spans: vec![span],
+        token_estimate,
+        chunker_version: chunker_version.clone(),
+        policy_hash: base_policy_hash.to_string(),
+    }
+}
+
+/// Split an oversize unit at blank-line paragraph boundaries, greedily
+/// gluing paragraphs until ~`AST_CHUNK_MAX_LINES` lines accumulate.
+/// Returns `(line_offset_start, line_offset_end, text)` where offsets are
+/// 0-based within the unit (caller adds the unit's absolute `line_start`).
+fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
+    let lines: Vec<&str> = code.split('\n').collect();
+    let total = lines.len() as u32;
+    let mut out: Vec<(u32, u32, String)> = Vec::new();
+    let mut start: u32 = 0;
+    while start < total {
+        let mut end = (start + AST_CHUNK_MAX_LINES).min(total);
+        let floor = start + (AST_CHUNK_MAX_LINES * 4 / 5);
+        if end < total {
+            if let Some(b) = (floor.min(end)..end)
+                .rev()
+                .find(|&i| lines[i as usize].trim().is_empty())
+            {
+                end = b + 1;
+            }
+        }
+        let text = lines[start as usize..end as usize].join("\n");
+        out.push((start, end.saturating_sub(1), text));
+        start = end;
+    }
+    if out.is_empty() {
+        out.push((0, total.saturating_sub(1), code.to_string()));
+    }
+    out
+}
+
+#[cfg(test)]
+mod tests {
+    use super::*;
+    use kebab_core::{
+        Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+        SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance,
+        SourceType, TrustLevel, WorkspacePath,
+    };
+    use time::OffsetDateTime;
+
+    fn code_doc(units: &[(&str, u32, u32, &str)]) -> CanonicalDocument {
+        let wp = WorkspacePath("crates/x/src/a.py".into());
+        let aid = AssetId("a".repeat(64));
+        let pv = ParserVersion("code-python-v1".into());
+        let doc_id = id_for_doc(&wp, &aid, &pv);
+        let blocks = units
+            .iter()
+            .enumerate()
+            .map(|(i, (sym, ls, le, code))| {
+                let span = SourceSpan::Code {
+                    line_start: *ls,
+                    line_end: *le,
+                    symbol: Some((*sym).to_string()),
+                    lang: Some("python".into()),
+                };
+                let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+                Block::Code(CodeBlock {
+                    common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span },
+                    lang: Some("python".into()),
+                    code: (*code).to_string(),
+                })
+            })
+            .collect();
+        CanonicalDocument {
+            doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(),
+            lang: Lang("und".into()), blocks,
+            metadata: Metadata {
+                aliases: vec![], tags: vec![],
+                created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                source_type: SourceType::Note, trust_level: TrustLevel::Primary,
+                user_id_alias: None, user: Default::default(),
+                repo: Some("kebab".into()), git_branch: Some("main".into()),
+                git_commit: Some("0".repeat(40)), code_lang: Some("python".into()),
+            },
+            provenance: Provenance { events: vec![] },
+            parser_version: pv, schema_version: 1, doc_version: 1,
+            last_chunker_version: None, last_embedding_version: None,
+        }
+    }
+    fn policy() -> ChunkPolicy {
+        ChunkPolicy { target_tokens: 500, overlap_tokens: 80,
+            respect_markdown_headings: false,
+            chunker_version: ChunkerVersion(VERSION_LABEL.into()) }
+    }
+
+    #[test]
+    fn chunker_version_is_code_python_ast_v1() {
+        assert_eq!(CodePythonAstV1Chunker.chunker_version(),
+            ChunkerVersion("code-python-ast-v1".into()));
+    }
+
+    #[test]
+    fn one_chunk_per_unit_preserves_code_span() {
+        let doc = code_doc(&[
+            ("parse", 1, 3, "def parse():\n    pass\n    # x"),
+            ("Foo.double", 5, 7, "def double():\n    #\n    pass"),
+        ]);
+        let chunks = CodePythonAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert_eq!(chunks.len(), 2);
+        for c in &chunks {
+            assert_eq!(c.source_spans.len(), 1);
+            assert!(matches!(c.source_spans[0], SourceSpan::Code { .. }));
+            assert_eq!(c.heading_path, Vec::<String>::new());
+            assert_eq!(c.chunker_version.0, "code-python-ast-v1");
+        }
+        match &chunks[0].source_spans[0] {
+            SourceSpan::Code { symbol, line_start, line_end, .. } => {
+                assert_eq!(symbol.as_deref(), Some("parse"));
+                assert_eq!((*line_start, *line_end), (1, 3));
+            }
+            _ => unreachable!(),
+        }
+    }
+
+    #[test]
+    fn oversize_unit_splits_into_parts_with_unique_ids() {
+        let body = (0..500).map(|i| format!("    x{i} = {i}")).collect::<Vec<_>>().join("\n");
+        let code = format!("def big():\n{body}\n");
+        let doc = code_doc(&[("big", 1, 502, &code)]);
+        let chunks = CodePythonAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len());
+        for c in &chunks {
+            match &c.source_spans[0] {
+                SourceSpan::Code { symbol, .. } => {
+                    assert!(symbol.as_deref().unwrap().starts_with("big [part "),
+                        "part-numbered symbol, got {symbol:?}");
+                }
+                _ => unreachable!(),
+            }
+        }
+        let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
+        let n = ids.len(); ids.sort_unstable(); ids.dedup();
+        assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
+    }
+
+    #[test]
+    fn non_code_doc_errors() {
+        use kebab_core::TextBlock;
+        let mut doc = code_doc(&[("parse", 1, 1, "def parse(): pass")]);
+        doc.blocks = vec![Block::Paragraph(TextBlock {
+            common: CommonBlock {
+                block_id: kebab_core::BlockId("b".into()),
+                heading_path: vec![],
+                source_span: SourceSpan::Line { start: 1, end: 1 },
+            },
+            text: "x".into(), inlines: vec![],
+        })];
+        let err = CodePythonAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
+        assert!(err.to_string().contains("CodePythonAstV1Chunker"));
+    }
+
+    #[test]
+    fn deterministic_chunk_ids_1000() {
+        let doc = code_doc(&[("parse", 1, 2, "def parse(): pass\n")]);
+        let base: Vec<String> = CodePythonAstV1Chunker.chunk(&doc, &policy())
+            .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+        for _ in 0..1000 {
+            let again: Vec<String> = CodePythonAstV1Chunker.chunk(&doc, &policy())
+                .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+            assert_eq!(again, base);
+        }
+    }
+
+    #[test]
+    fn policy_hash_matches_md_heading_v1() {
+        let p = policy();
+        assert_eq!(CodePythonAstV1Chunker.policy_hash(&p),
+            crate::MdHeadingV1Chunker.policy_hash(&p));
+    }
+}
--- a/crates/kebab-chunk/src/code_rust_ast_v1.rs
+++ b/crates/kebab-chunk/src/code_rust_ast_v1.rs
@@ -0,0 +1,322 @@
+//! `code-rust-ast-v1` — maps a tree-sitter-derived Rust AST
+//! `CanonicalDocument` (one `Block::Code` per semantic unit, each with
+//! `SourceSpan::Code`) to chunks 1:1. A unit longer than
+//! `AST_CHUNK_MAX_LINES` is split into `<symbol> [part i/N]` sub-chunks
+//! at blank-line paragraph boundaries (design §9.1 oversize fallback).
+//!
+//! tree-sitter is intentionally NOT a dependency here: AST work is
+//! parser-side (`kebab-parse-code`, design §6.3). This chunker only
+//! consumes the `CanonicalDocument`.
+//!
+//! `AST_CHUNK_MAX_LINES` is a constant matching
+//! `IngestCodeCfg::default().ast_chunk_max_lines` (200). Per-medium
+//! config threading needs a chunker registry (P+); same deviation
+//! pattern as `pdf-page-v1`'s pinned `chunker_version`
+//! (`tasks/HOTFIXES.md`).
+
+use kebab_core::{
+    Block, BlockId, CanonicalDocument, Chunk, ChunkPolicy, Chunker, ChunkerVersion, DocumentId,
+    SourceSpan, id_for_chunk,
+};
+
+const VERSION_LABEL: &str = "code-rust-ast-v1";
+const BYTES_PER_TOKEN: usize = 3;
+const POLICY_HASH_HEX_LEN: usize = 16;
+const AST_CHUNK_MAX_LINES: u32 = 200;
+
+#[derive(Clone, Copy, Debug, Default)]
+pub struct CodeRustAstV1Chunker;
+
+impl Chunker for CodeRustAstV1Chunker {
+    fn chunker_version(&self) -> ChunkerVersion {
+        ChunkerVersion(VERSION_LABEL.to_string())
+    }
+
+    fn policy_hash(&self, policy: &ChunkPolicy) -> String {
+        let bytes = serde_json_canonicalizer::to_vec(policy)
+            .expect("canonical JSON serialization of ChunkPolicy must not fail");
+        let hex = blake3::hash(&bytes).to_hex().to_string();
+        hex[..POLICY_HASH_HEX_LEN].to_string()
+    }
+
+    fn chunk(
+        &self,
+        doc: &CanonicalDocument,
+        policy: &ChunkPolicy,
+    ) -> anyhow::Result<Vec<Chunk>> {
+        for b in &doc.blocks {
+            let c = match b {
+                Block::Code(c) => c,
+                _ => anyhow::bail!(
+                    "CodeRustAstV1Chunker only handles code docs (got non-Code block)"
+                ),
+            };
+            if !matches!(c.common.source_span, SourceSpan::Code { .. }) {
+                anyhow::bail!(
+                    "CodeRustAstV1Chunker only handles code docs (got non-Code source_span)"
+                );
+            }
+        }
+
+        let base_policy_hash = self.policy_hash(policy);
+        let chunker_version = self.chunker_version();
+        let mut out: Vec<Chunk> = Vec::new();
+
+        for b in &doc.blocks {
+            let cb = match b {
+                Block::Code(c) => c,
+                _ => unreachable!("validated above"),
+            };
+            let (ls, le, symbol, lang) = match &cb.common.source_span {
+                SourceSpan::Code { line_start, line_end, symbol, lang } => {
+                    (*line_start, *line_end, symbol.clone(), lang.clone())
+                }
+                _ => unreachable!("validated above"),
+            };
+            let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
+            let span_lines = le.saturating_sub(ls) + 1;
+
+            if span_lines <= AST_CHUNK_MAX_LINES {
+                let span = SourceSpan::Code {
+                    line_start: ls,
+                    line_end: le,
+                    symbol: symbol.clone(),
+                    lang: lang.clone(),
+                };
+                out.push(make_chunk(
+                    doc, &chunker_version, &block_ids, &base_policy_hash,
+                    None, span, cb.code.clone(),
+                ));
+            } else {
+                let parts = split_oversize(&cb.code);
+                let n = parts.len();
+                for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
+                    let part_ls = ls + off_start;
+                    let part_le = ls + off_end;
+                    let part_sym = symbol
+                        .as_ref()
+                        .map(|s| format!("{s} [part {}/{n}]", i + 1));
+                    let span = SourceSpan::Code {
+                        line_start: part_ls,
+                        line_end: part_le,
+                        symbol: part_sym,
+                        lang: lang.clone(),
+                    };
+                    out.push(make_chunk(
+                        doc, &chunker_version, &block_ids, &base_policy_hash,
+                        Some(part_ls), span, text,
+                    ));
+                }
+            }
+        }
+
+        tracing::debug!(
+            target: "kebab-chunk",
+            doc_id = %doc.doc_id,
+            chunks = out.len(),
+            "code-rust-ast-v1 chunked",
+        );
+        Ok(out)
+    }
+}
+
+#[allow(clippy::too_many_arguments)]
+fn make_chunk(
+    doc: &CanonicalDocument,
+    chunker_version: &ChunkerVersion,
+    block_ids: &[BlockId],
+    base_policy_hash: &str,
+    split_key: Option<u32>,
+    span: SourceSpan,
+    text: String,
+) -> Chunk {
+    let id_hash = match split_key {
+        Some(k) => format!("{base_policy_hash}#L{k}"),
+        None => base_policy_hash.to_string(),
+    };
+    let chunk_id = id_for_chunk(&doc.doc_id, chunker_version, block_ids, &id_hash);
+    let token_estimate = text.len().div_ceil(BYTES_PER_TOKEN);
+    Chunk {
+        chunk_id,
+        doc_id: DocumentId(doc.doc_id.0.clone()),
+        block_ids: block_ids.to_vec(),
+        text,
+        heading_path: Vec::new(),
+        source_spans: vec![span],
+        token_estimate,
+        chunker_version: chunker_version.clone(),
+        policy_hash: base_policy_hash.to_string(),
+    }
+}
+
+/// Split an oversize unit at blank-line paragraph boundaries, greedily
+/// gluing paragraphs until ~`AST_CHUNK_MAX_LINES` lines accumulate.
+/// Returns `(line_offset_start, line_offset_end, text)` where offsets are
+/// 0-based within the unit (caller adds the unit's absolute `line_start`).
+fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
+    let lines: Vec<&str> = code.split('\n').collect();
+    let total = lines.len() as u32;
+    let mut out: Vec<(u32, u32, String)> = Vec::new();
+    let mut start: u32 = 0;
+    while start < total {
+        let mut end = (start + AST_CHUNK_MAX_LINES).min(total);
+        let floor = start + (AST_CHUNK_MAX_LINES * 4 / 5);
+        if end < total {
+            if let Some(b) = (floor.min(end)..end)
+                .rev()
+                .find(|&i| lines[i as usize].trim().is_empty())
+            {
+                end = b + 1;
+            }
+        }
+        let text = lines[start as usize..end as usize].join("\n");
+        out.push((start, end.saturating_sub(1), text));
+        start = end;
+    }
+    if out.is_empty() {
+        out.push((0, total.saturating_sub(1), code.to_string()));
+    }
+    out
+}
+
+#[cfg(test)]
+mod tests {
+    use super::*;
+    use kebab_core::{
+        Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+        SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance,
+        SourceType, TrustLevel, WorkspacePath,
+    };
+    use time::OffsetDateTime;
+
+    fn code_doc(units: &[(&str, u32, u32, &str)]) -> CanonicalDocument {
+        let wp = WorkspacePath("crates/x/src/a.rs".into());
+        let aid = AssetId("a".repeat(64));
+        let pv = ParserVersion("code-rust-v1".into());
+        let doc_id = id_for_doc(&wp, &aid, &pv);
+        let blocks = units
+            .iter()
+            .enumerate()
+            .map(|(i, (sym, ls, le, code))| {
+                let span = SourceSpan::Code {
+                    line_start: *ls,
+                    line_end: *le,
+                    symbol: Some((*sym).to_string()),
+                    lang: Some("rust".into()),
+                };
+                let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+                Block::Code(CodeBlock {
+                    common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span },
+                    lang: Some("rust".into()),
+                    code: (*code).to_string(),
+                })
+            })
+            .collect();
+        CanonicalDocument {
+            doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(),
+            lang: Lang("und".into()), blocks,
+            metadata: Metadata {
+                aliases: vec![], tags: vec![],
+                created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                source_type: SourceType::Note, trust_level: TrustLevel::Primary,
+                user_id_alias: None, user: Default::default(),
+                repo: Some("kebab".into()), git_branch: Some("main".into()),
+                git_commit: Some("0".repeat(40)), code_lang: Some("rust".into()),
+            },
+            provenance: Provenance { events: vec![] },
+            parser_version: pv, schema_version: 1, doc_version: 1,
+            last_chunker_version: None, last_embedding_version: None,
+        }
+    }
+    fn policy() -> ChunkPolicy {
+        ChunkPolicy { target_tokens: 500, overlap_tokens: 80,
+            respect_markdown_headings: false,
+            chunker_version: ChunkerVersion(VERSION_LABEL.into()) }
+    }
+
+    #[test]
+    fn chunker_version_is_code_rust_ast_v1() {
+        assert_eq!(CodeRustAstV1Chunker.chunker_version(),
+            ChunkerVersion("code-rust-ast-v1".into()));
+    }
+
+    #[test]
+    fn one_chunk_per_unit_preserves_code_span() {
+        let doc = code_doc(&[
+            ("parse", 1, 3, "pub fn parse() {}\n// x\n}"),
+            ("Foo::double", 5, 7, "fn double() {}\n//\n}"),
+        ]);
+        let chunks = CodeRustAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert_eq!(chunks.len(), 2);
+        for c in &chunks {
+            assert_eq!(c.source_spans.len(), 1);
+            assert!(matches!(c.source_spans[0], SourceSpan::Code { .. }));
+            assert_eq!(c.heading_path, Vec::<String>::new());
+            assert_eq!(c.chunker_version.0, "code-rust-ast-v1");
+        }
+        match &chunks[0].source_spans[0] {
+            SourceSpan::Code { symbol, line_start, line_end, .. } => {
+                assert_eq!(symbol.as_deref(), Some("parse"));
+                assert_eq!((*line_start, *line_end), (1, 3));
+            }
+            _ => unreachable!(),
+        }
+    }
+
+    #[test]
+    fn oversize_unit_splits_into_parts_with_unique_ids() {
+        let body = (0..500).map(|i| format!("    let x{i} = {i};")).collect::<Vec<_>>().join("\n");
+        let code = format!("pub fn big() {{\n{body}\n}}");
+        let doc = code_doc(&[("big", 1, 502, &code)]);
+        let chunks = CodeRustAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len());
+        for c in &chunks {
+            match &c.source_spans[0] {
+                SourceSpan::Code { symbol, .. } => {
+                    assert!(symbol.as_deref().unwrap().starts_with("big [part "),
+                        "part-numbered symbol, got {symbol:?}");
+                }
+                _ => unreachable!(),
+            }
+        }
+        let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
+        let n = ids.len(); ids.sort_unstable(); ids.dedup();
+        assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
+    }
+
+    #[test]
+    fn non_code_doc_errors() {
+        use kebab_core::TextBlock;
+        let mut doc = code_doc(&[("parse", 1, 1, "fn parse(){}")]);
+        doc.blocks = vec![Block::Paragraph(TextBlock {
+            common: CommonBlock {
+                block_id: kebab_core::BlockId("b".into()),
+                heading_path: vec![],
+                source_span: SourceSpan::Line { start: 1, end: 1 },
+            },
+            text: "x".into(), inlines: vec![],
+        })];
+        let err = CodeRustAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
+        assert!(err.to_string().contains("CodeRustAstV1Chunker"));
+    }
+
+    #[test]
+    fn deterministic_chunk_ids_1000() {
+        let doc = code_doc(&[("parse", 1, 2, "fn parse(){}\n}")]);
+        let base: Vec<String> = CodeRustAstV1Chunker.chunk(&doc, &policy())
+            .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+        for _ in 0..1000 {
+            let again: Vec<String> = CodeRustAstV1Chunker.chunk(&doc, &policy())
+                .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+            assert_eq!(again, base);
+        }
+    }
+
+    #[test]
+    fn policy_hash_matches_md_heading_v1() {
+        let p = policy();
+        assert_eq!(CodeRustAstV1Chunker.policy_hash(&p),
+            crate::MdHeadingV1Chunker.policy_hash(&p));
+    }
+}
--- a/crates/kebab-chunk/src/code_text_paragraph_v1.rs
+++ b/crates/kebab-chunk/src/code_text_paragraph_v1.rs
@@ -0,0 +1,170 @@
+//! p10-3: Tier 3 paragraph + line-window fallback chunker.
+//!
+//! Splits code/text files on blank-line paragraph boundaries.  Paragraphs
+//! with more than 80 lines are further split into 80-line windows with a
+//! 20-line overlap (stride 60) — the same oversize pattern used by Tier 1/2
+//! chunkers but without AST structure, hence no symbol.
+//!
+//! Per spec §9.3: all emitted chunks carry `symbol: None`.
+
+use crate::tier2_shared::{build_chunk_no_symbol, policy_hash};
+use anyhow::Result;
+use kebab_core::{Block, CanonicalDocument, Chunk, ChunkPolicy, ChunkerVersion, Chunker};
+
+pub const VERSION_LABEL: &str = "code-text-paragraph-v1";
+
+/// Lines-per-window for the oversize fallback (Tier 3).
+const FALLBACK_LINES_PER_CHUNK: usize = 80;
+/// Overlap between consecutive windows.
+const FALLBACK_LINES_OVERLAP: usize = 20;
+// stride = FALLBACK_LINES_PER_CHUNK - FALLBACK_LINES_OVERLAP = 60.
+
+#[derive(Clone, Copy, Debug, Default)]
+pub struct CodeTextParagraphV1Chunker;
+
+impl Chunker for CodeTextParagraphV1Chunker {
+    fn chunker_version(&self) -> ChunkerVersion {
+        ChunkerVersion(VERSION_LABEL.to_string())
+    }
+
+    fn policy_hash(&self, policy: &ChunkPolicy) -> String {
+        policy_hash(policy)
+    }
+
+    fn chunk(&self, doc: &CanonicalDocument, policy: &ChunkPolicy) -> Result<Vec<Chunk>> {
+        // Expect a single Block::Code carrying the full source text.
+        let (text, lang_str) = match doc.blocks.first() {
+            Some(Block::Code(cb)) => (cb.code.as_str(), cb.lang.as_deref().unwrap_or("")),
+            _ => return Ok(vec![]),
+        };
+
+        let mut chunks = Vec::new();
+        for para in split_paragraphs(text) {
+            push_paragraph(&mut chunks, doc, policy, &para, lang_str)?;
+        }
+
+        tracing::debug!(
+            target: "kebab-chunk",
+            doc_id = %doc.doc_id,
+            chunks = chunks.len(),
+            "code-text-paragraph-v1 chunked",
+        );
+
+        Ok(chunks)
+    }
+}
+
+/// A contiguous run of non-blank lines from the source text.
+struct Paragraph {
+    /// Lines joined with `\n` (no trailing newline).
+    text: String,
+    /// 1-indexed line number of the first line in the source file.
+    line_start: u32,
+    /// 1-indexed line number of the last line in the source file.
+    line_end: u32,
+}
+
+/// Split `text` into `Paragraph`s separated by blank (all-whitespace) lines.
+///
+/// Blank lines are treated as boundaries and are NOT included in any
+/// paragraph's line range.  Paragraphs that would consist entirely of blank
+/// lines are skipped.
+fn split_paragraphs(text: &str) -> Vec<Paragraph> {
+    let mut paragraphs = Vec::new();
+    let mut current: Vec<&str> = Vec::new();
+    let mut current_start: Option<u32> = None;
+
+    for (idx, line) in text.lines().enumerate() {
+        let line_no = (idx + 1) as u32;
+        let is_blank = line.trim().is_empty();
+        if is_blank {
+            if let Some(start) = current_start.take() {
+                let end = start + current.len() as u32 - 1;
+                paragraphs.push(Paragraph {
+                    text: current.join("\n"),
+                    line_start: start,
+                    line_end: end,
+                });
+                current.clear();
+            }
+        } else {
+            if current_start.is_none() {
+                current_start = Some(line_no);
+            }
+            current.push(line);
+        }
+    }
+    // Flush any trailing paragraph not terminated by a blank line.
+    if let Some(start) = current_start {
+        let end = start + current.len() as u32 - 1;
+        paragraphs.push(Paragraph {
+            text: current.join("\n"),
+            line_start: start,
+            line_end: end,
+        });
+    }
+    paragraphs
+}
+
+/// Emit one or more chunks for a single paragraph.
+///
+/// Paragraphs with ≤ `FALLBACK_LINES_PER_CHUNK` lines become a single chunk.
+/// Larger paragraphs are split into overlapping windows of
+/// `FALLBACK_LINES_PER_CHUNK` lines with stride `FALLBACK_LINES_PER_CHUNK -
+/// FALLBACK_LINES_OVERLAP`.  The last window may be shorter.  Window starts
+/// are passed as `split_key` so `id_for_chunk` can produce distinct ids
+/// across windows.
+fn push_paragraph(
+    out: &mut Vec<Chunk>,
+    doc: &CanonicalDocument,
+    policy: &ChunkPolicy,
+    para: &Paragraph,
+    lang: &str,
+) -> Result<()> {
+    let n_lines = (para.line_end - para.line_start + 1) as usize;
+
+    if n_lines <= FALLBACK_LINES_PER_CHUNK {
+        // Use line_start as split_key so each paragraph gets a distinct
+        // chunk_id even when block_ids is empty (no symbol, no AST structure).
+        // Without this, all short paragraphs from the same doc share the same
+        // base_policy_hash and therefore the same id_for_chunk result.
+        out.push(build_chunk_no_symbol(
+            doc,
+            policy,
+            &para.text,
+            para.line_start,
+            para.line_end,
+            lang,
+            VERSION_LABEL,
+            Some(para.line_start),
+        ));
+        return Ok(());
+    }
+
+    // Oversize: line-window split with overlap.
+    let stride = FALLBACK_LINES_PER_CHUNK - FALLBACK_LINES_OVERLAP;
+    let lines: Vec<&str> = para.text.lines().collect();
+    let mut i = 0usize;
+    loop {
+        let end = (i + FALLBACK_LINES_PER_CHUNK).min(lines.len());
+        let window_text = lines[i..end].join("\n");
+        let window_start = para.line_start + i as u32;
+        let window_end = para.line_start + (end as u32) - 1;
+        // Use window_start as split_key so chunk_ids are unique across windows.
+        out.push(build_chunk_no_symbol(
+            doc,
+            policy,
+            &window_text,
+            window_start,
+            window_end,
+            lang,
+            VERSION_LABEL,
+            Some(window_start),
+        ));
+        if end == lines.len() {
+            break;
+        }
+        i += stride;
+    }
+    Ok(())
+}
--- a/crates/kebab-chunk/src/code_ts_ast_v1.rs
+++ b/crates/kebab-chunk/src/code_ts_ast_v1.rs
@@ -0,0 +1,322 @@
+//! `code-ts-ast-v1` — maps a tree-sitter-derived TypeScript AST
+//! `CanonicalDocument` (one `Block::Code` per semantic unit, each with
+//! `SourceSpan::Code`) to chunks 1:1. A unit longer than
+//! `AST_CHUNK_MAX_LINES` is split into `<symbol> [part i/N]` sub-chunks
+//! at blank-line paragraph boundaries (design §9.1 oversize fallback).
+//!
+//! tree-sitter is intentionally NOT a dependency here: AST work is
+//! parser-side (`kebab-parse-code`, design §6.3). This chunker only
+//! consumes the `CanonicalDocument`.
+//!
+//! `AST_CHUNK_MAX_LINES` is a constant matching
+//! `IngestCodeCfg::default().ast_chunk_max_lines` (200). Per-medium
+//! config threading needs a chunker registry (P+); same deviation
+//! pattern as `pdf-page-v1`'s pinned `chunker_version`
+//! (`tasks/HOTFIXES.md`).
+
+use kebab_core::{
+    Block, BlockId, CanonicalDocument, Chunk, ChunkPolicy, Chunker, ChunkerVersion, DocumentId,
+    SourceSpan, id_for_chunk,
+};
+
+const VERSION_LABEL: &str = "code-ts-ast-v1";
+const BYTES_PER_TOKEN: usize = 3;
+const POLICY_HASH_HEX_LEN: usize = 16;
+const AST_CHUNK_MAX_LINES: u32 = 200;
+
+#[derive(Clone, Copy, Debug, Default)]
+pub struct CodeTsAstV1Chunker;
+
+impl Chunker for CodeTsAstV1Chunker {
+    fn chunker_version(&self) -> ChunkerVersion {
+        ChunkerVersion(VERSION_LABEL.to_string())
+    }
+
+    fn policy_hash(&self, policy: &ChunkPolicy) -> String {
+        let bytes = serde_json_canonicalizer::to_vec(policy)
+            .expect("canonical JSON serialization of ChunkPolicy must not fail");
+        let hex = blake3::hash(&bytes).to_hex().to_string();
+        hex[..POLICY_HASH_HEX_LEN].to_string()
+    }
+
+    fn chunk(
+        &self,
+        doc: &CanonicalDocument,
+        policy: &ChunkPolicy,
+    ) -> anyhow::Result<Vec<Chunk>> {
+        for b in &doc.blocks {
+            let c = match b {
+                Block::Code(c) => c,
+                _ => anyhow::bail!(
+                    "CodeTsAstV1Chunker only handles code docs (got non-Code block)"
+                ),
+            };
+            if !matches!(c.common.source_span, SourceSpan::Code { .. }) {
+                anyhow::bail!(
+                    "CodeTsAstV1Chunker only handles code docs (got non-Code source_span)"
+                );
+            }
+        }
+
+        let base_policy_hash = self.policy_hash(policy);
+        let chunker_version = self.chunker_version();
+        let mut out: Vec<Chunk> = Vec::new();
+
+        for b in &doc.blocks {
+            let cb = match b {
+                Block::Code(c) => c,
+                _ => unreachable!("validated above"),
+            };
+            let (ls, le, symbol, lang) = match &cb.common.source_span {
+                SourceSpan::Code { line_start, line_end, symbol, lang } => {
+                    (*line_start, *line_end, symbol.clone(), lang.clone())
+                }
+                _ => unreachable!("validated above"),
+            };
+            let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
+            let span_lines = le.saturating_sub(ls) + 1;
+
+            if span_lines <= AST_CHUNK_MAX_LINES {
+                let span = SourceSpan::Code {
+                    line_start: ls,
+                    line_end: le,
+                    symbol: symbol.clone(),
+                    lang: lang.clone(),
+                };
+                out.push(make_chunk(
+                    doc, &chunker_version, &block_ids, &base_policy_hash,
+                    None, span, cb.code.clone(),
+                ));
+            } else {
+                let parts = split_oversize(&cb.code);
+                let n = parts.len();
+                for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
+                    let part_ls = ls + off_start;
+                    let part_le = ls + off_end;
+                    let part_sym = symbol
+                        .as_ref()
+                        .map(|s| format!("{s} [part {}/{n}]", i + 1));
+                    let span = SourceSpan::Code {
+                        line_start: part_ls,
+                        line_end: part_le,
+                        symbol: part_sym,
+                        lang: lang.clone(),
+                    };
+                    out.push(make_chunk(
+                        doc, &chunker_version, &block_ids, &base_policy_hash,
+                        Some(part_ls), span, text,
+                    ));
+                }
+            }
+        }
+
+        tracing::debug!(
+            target: "kebab-chunk",
+            doc_id = %doc.doc_id,
+            chunks = out.len(),
+            "code-ts-ast-v1 chunked",
+        );
+        Ok(out)
+    }
+}
+
+#[allow(clippy::too_many_arguments)]
+fn make_chunk(
+    doc: &CanonicalDocument,
+    chunker_version: &ChunkerVersion,
+    block_ids: &[BlockId],
+    base_policy_hash: &str,
+    split_key: Option<u32>,
+    span: SourceSpan,
+    text: String,
+) -> Chunk {
+    let id_hash = match split_key {
+        Some(k) => format!("{base_policy_hash}#L{k}"),
+        None => base_policy_hash.to_string(),
+    };
+    let chunk_id = id_for_chunk(&doc.doc_id, chunker_version, block_ids, &id_hash);
+    let token_estimate = text.len().div_ceil(BYTES_PER_TOKEN);
+    Chunk {
+        chunk_id,
+        doc_id: DocumentId(doc.doc_id.0.clone()),
+        block_ids: block_ids.to_vec(),
+        text,
+        heading_path: Vec::new(),
+        source_spans: vec![span],
+        token_estimate,
+        chunker_version: chunker_version.clone(),
+        policy_hash: base_policy_hash.to_string(),
+    }
+}
+
+/// Split an oversize unit at blank-line paragraph boundaries, greedily
+/// gluing paragraphs until ~`AST_CHUNK_MAX_LINES` lines accumulate.
+/// Returns `(line_offset_start, line_offset_end, text)` where offsets are
+/// 0-based within the unit (caller adds the unit's absolute `line_start`).
+fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
+    let lines: Vec<&str> = code.split('\n').collect();
+    let total = lines.len() as u32;
+    let mut out: Vec<(u32, u32, String)> = Vec::new();
+    let mut start: u32 = 0;
+    while start < total {
+        let mut end = (start + AST_CHUNK_MAX_LINES).min(total);
+        let floor = start + (AST_CHUNK_MAX_LINES * 4 / 5);
+        if end < total {
+            if let Some(b) = (floor.min(end)..end)
+                .rev()
+                .find(|&i| lines[i as usize].trim().is_empty())
+            {
+                end = b + 1;
+            }
+        }
+        let text = lines[start as usize..end as usize].join("\n");
+        out.push((start, end.saturating_sub(1), text));
+        start = end;
+    }
+    if out.is_empty() {
+        out.push((0, total.saturating_sub(1), code.to_string()));
+    }
+    out
+}
+
+#[cfg(test)]
+mod tests {
+    use super::*;
+    use kebab_core::{
+        Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+        SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance,
+        SourceType, TrustLevel, WorkspacePath,
+    };
+    use time::OffsetDateTime;
+
+    fn code_doc(units: &[(&str, u32, u32, &str)]) -> CanonicalDocument {
+        let wp = WorkspacePath("crates/x/src/a.ts".into());
+        let aid = AssetId("a".repeat(64));
+        let pv = ParserVersion("code-ts-v1".into());
+        let doc_id = id_for_doc(&wp, &aid, &pv);
+        let blocks = units
+            .iter()
+            .enumerate()
+            .map(|(i, (sym, ls, le, code))| {
+                let span = SourceSpan::Code {
+                    line_start: *ls,
+                    line_end: *le,
+                    symbol: Some((*sym).to_string()),
+                    lang: Some("typescript".into()),
+                };
+                let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+                Block::Code(CodeBlock {
+                    common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span },
+                    lang: Some("typescript".into()),
+                    code: (*code).to_string(),
+                })
+            })
+            .collect();
+        CanonicalDocument {
+            doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(),
+            lang: Lang("und".into()), blocks,
+            metadata: Metadata {
+                aliases: vec![], tags: vec![],
+                created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                source_type: SourceType::Note, trust_level: TrustLevel::Primary,
+                user_id_alias: None, user: Default::default(),
+                repo: Some("kebab".into()), git_branch: Some("main".into()),
+                git_commit: Some("0".repeat(40)), code_lang: Some("typescript".into()),
+            },
+            provenance: Provenance { events: vec![] },
+            parser_version: pv, schema_version: 1, doc_version: 1,
+            last_chunker_version: None, last_embedding_version: None,
+        }
+    }
+    fn policy() -> ChunkPolicy {
+        ChunkPolicy { target_tokens: 500, overlap_tokens: 80,
+            respect_markdown_headings: false,
+            chunker_version: ChunkerVersion(VERSION_LABEL.into()) }
+    }
+
+    #[test]
+    fn chunker_version_is_code_ts_ast_v1() {
+        assert_eq!(CodeTsAstV1Chunker.chunker_version(),
+            ChunkerVersion("code-ts-ast-v1".into()));
+    }
+
+    #[test]
+    fn one_chunk_per_unit_preserves_code_span() {
+        let doc = code_doc(&[
+            ("parse", 1, 3, "function parse(): void {\n    // x\n}"),
+            ("Foo.double", 5, 7, "function double(): number {\n    //\n    return 0;\n}"),
+        ]);
+        let chunks = CodeTsAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert_eq!(chunks.len(), 2);
+        for c in &chunks {
+            assert_eq!(c.source_spans.len(), 1);
+            assert!(matches!(c.source_spans[0], SourceSpan::Code { .. }));
+            assert_eq!(c.heading_path, Vec::<String>::new());
+            assert_eq!(c.chunker_version.0, "code-ts-ast-v1");
+        }
+        match &chunks[0].source_spans[0] {
+            SourceSpan::Code { symbol, line_start, line_end, .. } => {
+                assert_eq!(symbol.as_deref(), Some("parse"));
+                assert_eq!((*line_start, *line_end), (1, 3));
+            }
+            _ => unreachable!(),
+        }
+    }
+
+    #[test]
+    fn oversize_unit_splits_into_parts_with_unique_ids() {
+        let body = (0..500).map(|i| format!("    const x{i} = {i};")).collect::<Vec<_>>().join("\n");
+        let code = format!("function big(): void {{\n{body}\n}}");
+        let doc = code_doc(&[("big", 1, 502, &code)]);
+        let chunks = CodeTsAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len());
+        for c in &chunks {
+            match &c.source_spans[0] {
+                SourceSpan::Code { symbol, .. } => {
+                    assert!(symbol.as_deref().unwrap().starts_with("big [part "),
+                        "part-numbered symbol, got {symbol:?}");
+                }
+                _ => unreachable!(),
+            }
+        }
+        let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
+        let n = ids.len(); ids.sort_unstable(); ids.dedup();
+        assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
+    }
+
+    #[test]
+    fn non_code_doc_errors() {
+        use kebab_core::TextBlock;
+        let mut doc = code_doc(&[("parse", 1, 1, "function parse(): void {}")]);
+        doc.blocks = vec![Block::Paragraph(TextBlock {
+            common: CommonBlock {
+                block_id: kebab_core::BlockId("b".into()),
+                heading_path: vec![],
+                source_span: SourceSpan::Line { start: 1, end: 1 },
+            },
+            text: "x".into(), inlines: vec![],
+        })];
+        let err = CodeTsAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
+        assert!(err.to_string().contains("CodeTsAstV1Chunker"));
+    }
+
+    #[test]
+    fn deterministic_chunk_ids_1000() {
+        let doc = code_doc(&[("parse", 1, 2, "function parse(): void {}\n")]);
+        let base: Vec<String> = CodeTsAstV1Chunker.chunk(&doc, &policy())
+            .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+        for _ in 0..1000 {
+            let again: Vec<String> = CodeTsAstV1Chunker.chunk(&doc, &policy())
+                .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+            assert_eq!(again, base);
+        }
+    }
+
+    #[test]
+    fn policy_hash_matches_md_heading_v1() {
+        let p = policy();
+        assert_eq!(CodeTsAstV1Chunker.policy_hash(&p),
+            crate::MdHeadingV1Chunker.policy_hash(&p));
+    }
+}
--- a/crates/kebab-chunk/src/dockerfile_file_v1.rs
+++ b/crates/kebab-chunk/src/dockerfile_file_v1.rs
@@ -0,0 +1,58 @@
+//! p10-2: dockerfile whole-file chunker (Tier 2).
+//!
+//! Reads entire Dockerfile content and emits a single Chunk with symbol
+//! "<dockerfile>", code_lang "dockerfile", line range 1..EOF.
+//! Oversize >200 lines splits into line-windows sharing the symbol via
+//! tier2_shared::push_chunks_with_oversize.
+
+use crate::tier2_shared::{policy_hash, push_chunks_with_oversize};
+use anyhow::Result;
+use kebab_core::{Block, CanonicalDocument, Chunk, ChunkPolicy, ChunkerVersion, Chunker};
+
+pub const VERSION_LABEL: &str = "dockerfile-file-v1";
+
+#[derive(Clone, Copy, Debug, Default)]
+pub struct DockerfileFileV1Chunker;
+
+impl Chunker for DockerfileFileV1Chunker {
+    fn chunker_version(&self) -> ChunkerVersion {
+        ChunkerVersion(VERSION_LABEL.to_string())
+    }
+
+    fn policy_hash(&self, policy: &ChunkPolicy) -> String {
+        policy_hash(policy)
+    }
+
+    fn chunk(&self, doc: &CanonicalDocument, policy: &ChunkPolicy) -> Result<Vec<Chunk>> {
+        // Expect a single Block::Code carrying the full Dockerfile text.
+        let text = match doc.blocks.first() {
+            Some(Block::Code(cb)) => cb.code.as_str(),
+            _ => return Ok(vec![]),
+        };
+
+        let total_lines = text.lines().count().max(1) as u32;
+        let mut chunks = Vec::new();
+
+        push_chunks_with_oversize(
+            &mut chunks,
+            doc,
+            policy,
+            text,
+            1,
+            total_lines,
+            "<dockerfile>",
+            "dockerfile",
+            VERSION_LABEL,
+            None,
+        )?;
+
+        tracing::debug!(
+            target: "kebab-chunk",
+            doc_id = %doc.doc_id,
+            chunks = chunks.len(),
+            "dockerfile-file-v1 chunked",
+        );
+
+        Ok(chunks)
+    }
+}
--- a/crates/kebab-chunk/src/k8s_manifest_resource_v1.rs
+++ b/crates/kebab-chunk/src/k8s_manifest_resource_v1.rs
@@ -0,0 +1,170 @@
+//! p10-2: k8s manifest resource-aware chunker.
+//!
+//! Splits a multi-document YAML file on `^---\s*$` boundaries, recognises
+//! documents that have both `apiVersion` and `kind` string fields as k8s
+//! resources, and emits one `Chunk` per resource (with oversize >200-line
+//! fallback).  Non-k8s documents are skipped; invalid YAML yields 0 chunks
+//! for the entire file.
+
+use crate::tier2_shared::{policy_hash, push_chunks_with_oversize};
+use anyhow::Result;
+use kebab_core::{Block, CanonicalDocument, Chunk, ChunkPolicy, ChunkerVersion, Chunker};
+
+pub const VERSION_LABEL: &str = "k8s-manifest-resource-v1";
+
+#[derive(Clone, Copy, Debug, Default)]
+pub struct K8sManifestResourceV1Chunker;
+
+impl Chunker for K8sManifestResourceV1Chunker {
+    fn chunker_version(&self) -> ChunkerVersion {
+        ChunkerVersion(VERSION_LABEL.to_string())
+    }
+
+    fn policy_hash(&self, policy: &ChunkPolicy) -> String {
+        policy_hash(policy)
+    }
+
+    fn chunk(&self, doc: &CanonicalDocument, policy: &ChunkPolicy) -> Result<Vec<Chunk>> {
+        // Expect a single Block::Code carrying the full YAML text.
+        let text = match doc.blocks.first() {
+            Some(Block::Code(cb)) => cb.code.as_str(),
+            _ => return Ok(vec![]),
+        };
+
+        let slices = split_yaml_documents(text);
+        let mut chunks: Vec<Chunk> = Vec::new();
+
+        for slice in slices {
+            // Invalid YAML in any document → return 0 chunks for the file.
+            let value: serde_yaml::Value = match serde_yaml::from_str(slice.text) {
+                Ok(v) => v,
+                Err(_) => return Ok(vec![]),
+            };
+
+            let Some(mapping) = value.as_mapping() else {
+                continue;
+            };
+
+            let api = mapping
+                .get("apiVersion")
+                .and_then(|v| v.as_str())
+                .unwrap_or("");
+            let kind = mapping
+                .get("kind")
+                .and_then(|v| v.as_str())
+                .unwrap_or("");
+
+            // Skip non-k8s documents.
+            if api.is_empty() || kind.is_empty() {
+                continue;
+            }
+
+            let metadata = mapping
+                .get("metadata")
+                .and_then(|v| v.as_mapping());
+            let name = metadata
+                .and_then(|m| m.get("name"))
+                .and_then(|v| v.as_str())
+                .unwrap_or("<unnamed>");
+            let namespace = metadata
+                .and_then(|m| m.get("namespace"))
+                .and_then(|v| v.as_str());
+
+            let symbol = match namespace {
+                Some(ns) if !ns.is_empty() => format!("{kind}/{ns}/{name}"),
+                _ => format!("{kind}/{name}"),
+            };
+
+            push_chunks_with_oversize(
+                &mut chunks,
+                doc,
+                policy,
+                slice.text,
+                slice.line_start,
+                slice.line_end,
+                &symbol,
+                "yaml",
+                VERSION_LABEL,
+                Some(slice.line_start),
+            )?;
+        }
+
+        tracing::debug!(
+            target: "kebab-chunk",
+            doc_id = %doc.doc_id,
+            chunks = chunks.len(),
+            "k8s-manifest-resource-v1 chunked",
+        );
+
+        Ok(chunks)
+    }
+}
+
+struct YamlSlice<'a> {
+    text: &'a str,
+    line_start: u32,
+    line_end: u32,
+}
+
+/// Split raw YAML text into per-document slices on `---` separator lines.
+/// Line numbers are 1-indexed.
+fn split_yaml_documents(text: &str) -> Vec<YamlSlice<'_>> {
+    let lines: Vec<&str> = text.lines().collect();
+
+    // Collect indices of separator lines (0-based), then append a sentinel at
+    // the end so the last slice is always terminated.
+    let mut separators: Vec<usize> = lines
+        .iter()
+        .enumerate()
+        .filter_map(|(i, l)| {
+            let trimmed = l.trim_end();
+            if trimmed == "---"
+                || trimmed.starts_with("--- ")
+                || trimmed.starts_with("---\t")
+            {
+                Some(i)
+            } else {
+                None
+            }
+        })
+        .collect();
+    separators.push(lines.len());
+
+    let mut slices: Vec<YamlSlice<'_>> = Vec::new();
+    let mut doc_start_line: usize = 0; // 0-based index of current doc start
+
+    for sep_line in separators {
+        if sep_line > doc_start_line {
+            let start_byte = byte_offset_of_line(text, doc_start_line);
+            let end_byte = byte_offset_of_line(text, sep_line);
+            let slice_text = &text[start_byte..end_byte];
+            if !slice_text.trim().is_empty() {
+                slices.push(YamlSlice {
+                    text: slice_text,
+                    line_start: (doc_start_line + 1) as u32,
+                    line_end: sep_line as u32,
+                });
+            }
+        }
+        doc_start_line = sep_line + 1;
+    }
+
+    slices
+}
+
+/// Return the byte offset of the start of `line_idx` (0-based line index).
+fn byte_offset_of_line(text: &str, line_idx: usize) -> usize {
+    if line_idx == 0 {
+        return 0;
+    }
+    let mut count = 0usize;
+    for (i, c) in text.char_indices() {
+        if c == '\n' {
+            count += 1;
+            if count == line_idx {
+                return i + 1;
+            }
+        }
+    }
+    text.len()
+}
--- a/crates/kebab-chunk/src/lib.rs
+++ b/crates/kebab-chunk/src/lib.rs
@@ -15,8 +15,35 @@
 //! embedder, the retriever, the LLM, the RAG layer, or the UI layers.
 //! It consumes `CanonicalDocument` purely through `kb-core` types.

+mod code_c_ast_v1;
+mod code_cpp_ast_v1;
+mod code_go_ast_v1;
+mod code_java_ast_v1;
+mod code_js_ast_v1;
+mod code_kotlin_ast_v1;
+mod code_python_ast_v1;
+mod code_rust_ast_v1;
+mod code_ts_ast_v1;
 mod md_heading_v1;
 mod pdf_page_v1;
+mod tier2_shared;
+pub mod k8s_manifest_resource_v1;
+pub mod dockerfile_file_v1;
+pub mod manifest_file_v1;
+pub mod code_text_paragraph_v1;

+pub use code_c_ast_v1::CodeCAstV1Chunker;
+pub use code_cpp_ast_v1::CodeCppAstV1Chunker;
+pub use code_go_ast_v1::CodeGoAstV1Chunker;
+pub use code_java_ast_v1::CodeJavaAstV1Chunker;
+pub use code_js_ast_v1::CodeJsAstV1Chunker;
+pub use code_kotlin_ast_v1::CodeKotlinAstV1Chunker;
+pub use code_python_ast_v1::CodePythonAstV1Chunker;
+pub use code_rust_ast_v1::CodeRustAstV1Chunker;
+pub use code_ts_ast_v1::CodeTsAstV1Chunker;
 pub use md_heading_v1::MdHeadingV1Chunker;
 pub use pdf_page_v1::PdfPageV1Chunker;
+pub use k8s_manifest_resource_v1::K8sManifestResourceV1Chunker;
+pub use dockerfile_file_v1::DockerfileFileV1Chunker;
+pub use manifest_file_v1::ManifestFileV1Chunker;
+pub use code_text_paragraph_v1::CodeTextParagraphV1Chunker;
--- a/crates/kebab-chunk/src/manifest_file_v1.rs
+++ b/crates/kebab-chunk/src/manifest_file_v1.rs
@@ -0,0 +1,59 @@
+//! p10-2: manifest whole-file chunker (Tier 2).
+//!
+//! Reads entire manifest file (Cargo.toml / package.json / pom.xml / go.mod /
+//! build.gradle / pyproject.toml / tsconfig.json) and emits a single Chunk
+//! with symbol "<manifest>", code_lang read from Block::Code.lang, line range
+//! 1..EOF. Oversize >200 lines splits into line-windows sharing the symbol via
+//! tier2_shared::push_chunks_with_oversize.
+
+use crate::tier2_shared::{policy_hash, push_chunks_with_oversize};
+use anyhow::Result;
+use kebab_core::{Block, CanonicalDocument, Chunk, ChunkPolicy, ChunkerVersion, Chunker};
+
+pub const VERSION_LABEL: &str = "manifest-file-v1";
+
+#[derive(Clone, Copy, Debug, Default)]
+pub struct ManifestFileV1Chunker;
+
+impl Chunker for ManifestFileV1Chunker {
+    fn chunker_version(&self) -> ChunkerVersion {
+        ChunkerVersion(VERSION_LABEL.to_string())
+    }
+
+    fn policy_hash(&self, policy: &ChunkPolicy) -> String {
+        policy_hash(policy)
+    }
+
+    fn chunk(&self, doc: &CanonicalDocument, policy: &ChunkPolicy) -> Result<Vec<Chunk>> {
+        // Expect a single Block::Code carrying the full manifest text.
+        let (text, lang) = match doc.blocks.first() {
+            Some(Block::Code(cb)) => (cb.code.as_str(), cb.lang.as_deref().unwrap_or("")),
+            _ => return Ok(vec![]),
+        };
+
+        let total_lines = text.lines().count().max(1) as u32;
+        let mut chunks = Vec::new();
+
+        push_chunks_with_oversize(
+            &mut chunks,
+            doc,
+            policy,
+            text,
+            1,
+            total_lines,
+            "<manifest>",
+            lang,
+            VERSION_LABEL,
+            None,
+        )?;
+
+        tracing::debug!(
+            target: "kebab-chunk",
+            doc_id = %doc.doc_id,
+            chunks = chunks.len(),
+            "manifest-file-v1 chunked",
+        );
+
+        Ok(chunks)
+    }
+}
--- a/crates/kebab-chunk/src/md_heading_v1.rs
+++ b/crates/kebab-chunk/src/md_heading_v1.rs
@@ -387,9 +387,7 @@ fn render_block_text(b: &Block) -> String {
        // alt keeps lexical search hits on filenames working even when
        // P6-1's filename auto-fill is bypassed.
        Block::ImageRef(i) => {
-            let alt = if !i.alt.is_empty() {
-                i.alt.clone()
-            } else {
+            let alt = if i.alt.is_empty() {
                // P6-1 falls back to filename so this branch is
                // defensive — keep it lest a future test fixture or
                // synthetic block path skip the auto-fill.
@@ -399,17 +397,17 @@ fn render_block_text(b: &Block) -> String {
                    .filter(|s| !s.is_empty())
                    .unwrap_or("[image]")
                    .to_string()
+            } else {
+                i.alt.clone()
            };
            let ocr = i
                .ocr
                .as_ref()
-                .map(|o| o.joined.as_str())
-                .unwrap_or("");
+                .map_or("", |o| o.joined.as_str());
            let cap = i
                .caption
                .as_ref()
-                .map(|c| c.text.as_str())
-                .unwrap_or("");
+                .map_or("", |c| c.text.as_str());
            [alt.as_str(), ocr, cap]
                .iter()
                .filter(|s| !s.is_empty())
@@ -472,6 +470,10 @@ mod tests {
                trust_level: TrustLevel::Primary,
                user_id_alias: None,
                user: Default::default(),
+                repo: None,
+                git_branch: None,
+                git_commit: None,
+                code_lang: None,
            },
            provenance: Provenance { events: vec![] },
            parser_version: kebab_core::ParserVersion("test-parser-0".into()),
--- a/crates/kebab-chunk/src/pdf_page_v1.rs
+++ b/crates/kebab-chunk/src/pdf_page_v1.rs
@@ -347,6 +347,10 @@ mod tests {
                trust_level: TrustLevel::Primary,
                user_id_alias: None,
                user: Default::default(),
+                repo: None,
+                git_branch: None,
+                git_commit: None,
+                code_lang: None,
            },
            provenance: Provenance { events: vec![] },
            parser_version,
@@ -446,7 +450,7 @@ mod tests {
        // chunk_ids stay distinct despite identical block_ids — the
        // per-chunk policy_hash variant is doing its job.
        let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
-        ids.sort();
+        ids.sort_unstable();
        let total = ids.len();
        ids.dedup();
        assert_eq!(ids.len(), total, "all chunk_ids must be unique");
@@ -512,6 +516,10 @@ mod tests {
                trust_level: TrustLevel::Primary,
                user_id_alias: None,
                user: Default::default(),
+                repo: None,
+                git_branch: None,
+                git_commit: None,
+                code_lang: None,
            },
            provenance: Provenance { events: vec![] },
            parser_version,
@@ -660,7 +668,7 @@ mod tests {
        // chunk_ids stay distinct (the per-chunk hash variant keys off
        // char_start which is now strictly increasing).
        let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
-        ids.sort();
+        ids.sort_unstable();
        let total = ids.len();
        ids.dedup();
        assert_eq!(ids.len(), total, "chunk_ids must remain unique");
--- a/crates/kebab-chunk/src/tier2_shared.rs
+++ b/crates/kebab-chunk/src/tier2_shared.rs
@@ -0,0 +1,192 @@
+//! p10-2: Tier 2 chunker shared helpers (oversize fallback + Chunk build).
+//!
+//! Mirrors `code_rust_ast_v1`'s Chunk-construction pattern exactly so that
+//! id / hashes / token-count / ChunkPolicy semantics stay identical across
+//! Tier 1 (AST) and Tier 2 (resource-aware) chunkers.
+
+use anyhow::Result;
+use kebab_core::{
+    BlockId, CanonicalDocument, Chunk, ChunkPolicy, ChunkerVersion, DocumentId, SourceSpan,
+    id_for_chunk,
+};
+
+pub(crate) const AST_CHUNK_MAX_LINES: u32 = 200;
+const BYTES_PER_TOKEN: usize = 3;
+const POLICY_HASH_HEX_LEN: usize = 16;
+
+/// Compute the policy hash the same way `code_rust_ast_v1` does.
+pub(crate) fn policy_hash(policy: &ChunkPolicy) -> String {
+    let bytes = serde_json_canonicalizer::to_vec(policy)
+        .expect("canonical JSON serialization of ChunkPolicy must not fail");
+    let hex = blake3::hash(&bytes).to_hex().to_string();
+    hex[..POLICY_HASH_HEX_LEN].to_string()
+}
+
+/// Emit one chunk for `(text, line_start..=line_end, symbol, lang)`, splitting
+/// into line-windows of at most `AST_CHUNK_MAX_LINES` if the slice is oversize.
+/// Mirrors the oversize path in `code_rust_ast_v1`'s `chunk` impl.
+///
+/// `base_split_key` is used as the `split_key` for the non-oversize single-chunk
+/// case. Callers that emit multiple chunks from the same document (e.g.
+/// `K8sManifestResourceV1Chunker` — one call per k8s resource) MUST pass
+/// `Some(line_start)` so that each call produces a distinct `chunk_id`.
+/// Single-chunk callers (dockerfile-file-v1, manifest-file-v1) pass `None` to
+/// keep chunk_ids stable (no sibling can collide when there's only one chunk).
+#[allow(clippy::too_many_arguments)]
+pub(crate) fn push_chunks_with_oversize(
+    out: &mut Vec<Chunk>,
+    doc: &CanonicalDocument,
+    policy: &ChunkPolicy,
+    text: &str,
+    line_start: u32,
+    line_end: u32,
+    symbol: &str,
+    lang: &str,
+    chunker_version: &str,
+    base_split_key: Option<u32>,
+) -> Result<()> {
+    let n_lines = (line_end - line_start + 1).max(1);
+    let cv = ChunkerVersion(chunker_version.to_string());
+    let base_policy_hash = policy_hash(policy);
+
+    if n_lines <= AST_CHUNK_MAX_LINES {
+        out.push(build_chunk(
+            doc,
+            &cv,
+            &base_policy_hash,
+            text,
+            line_start,
+            line_end,
+            symbol,
+            lang,
+            base_split_key,
+        ));
+        return Ok(());
+    }
+
+    let lines: Vec<&str> = text.lines().collect();
+    let total = lines.len();
+    let mut window_start = line_start;
+    let mut i = 0usize;
+    while i < total {
+        let take = (AST_CHUNK_MAX_LINES as usize).min(total - i);
+        let window_text = lines[i..i + take].join("\n");
+        let window_end = window_start + take as u32 - 1;
+        out.push(build_chunk(
+            doc,
+            &cv,
+            &base_policy_hash,
+            &window_text,
+            window_start,
+            window_end,
+            symbol,
+            lang,
+            Some(window_start),
+        ));
+        i += take;
+        window_start = window_end + 1;
+    }
+    Ok(())
+}
+
+/// Build a single `Chunk`, mirroring `make_chunk` in `code_rust_ast_v1.rs`
+/// exactly (same id recipe, same token estimate, same field set).
+///
+/// `split_key` is `Some(line_start_of_window)` for oversize splits, `None`
+/// for normal single-chunk emission.  Mirrors the `Some(part_ls)` / `None`
+/// split_key pattern in 1A-2.
+#[allow(clippy::too_many_arguments)]
+pub(crate) fn build_chunk(
+    doc: &CanonicalDocument,
+    chunker_version: &ChunkerVersion,
+    base_policy_hash: &str,
+    text: &str,
+    line_start: u32,
+    line_end: u32,
+    symbol: &str,
+    lang: &str,
+    split_key: Option<u32>,
+) -> Chunk {
+    let span = SourceSpan::Code {
+        line_start,
+        line_end,
+        symbol: Some(symbol.to_string()),
+        lang: Some(lang.to_string()),
+    };
+    build_chunk_from_span(doc, chunker_version, base_policy_hash, text, span, split_key)
+}
+
+/// Like `build_chunk` but emits `symbol: None`. Used by Tier 3 (per spec §9.3).
+///
+/// Accepts `policy: &ChunkPolicy` and `chunker_version: &str` (string slice)
+/// so callers don't need to pre-compute the hash and version wrapper.
+/// `split_key` is `Some(window_start)` for oversize line-window splits.
+#[allow(clippy::too_many_arguments)]
+pub(crate) fn build_chunk_no_symbol(
+    doc: &CanonicalDocument,
+    policy: &ChunkPolicy,
+    text: &str,
+    line_start: u32,
+    line_end: u32,
+    lang: &str,
+    chunker_version: &str,
+    split_key: Option<u32>,
+) -> Chunk {
+    let cv = ChunkerVersion(chunker_version.to_string());
+    let base_policy_hash = policy_hash(policy);
+    let span = SourceSpan::Code {
+        line_start,
+        line_end,
+        symbol: None,
+        lang: Some(lang.to_string()),
+    };
+    build_chunk_from_span(doc, &cv, &base_policy_hash, text, span, split_key)
+}
+
+/// Core chunk-building logic shared by `build_chunk` and `build_chunk_no_symbol`.
+///
+/// Takes a pre-built `SourceSpan` so the only difference between the two
+/// public helpers is whether `symbol` is `Some` or `None`.  All id/hash/
+/// token mechanics are identical.
+fn build_chunk_from_span(
+    doc: &CanonicalDocument,
+    chunker_version: &ChunkerVersion,
+    base_policy_hash: &str,
+    text: &str,
+    span: SourceSpan,
+    split_key: Option<u32>,
+) -> Chunk {
+    // id_hash mirrors code_rust_ast_v1's make_chunk logic:
+    //   split_key Some(k) => "{base_policy_hash}#L{k}"
+    //   split_key None    => base_policy_hash
+    let id_hash = match split_key {
+        Some(k) => format!("{base_policy_hash}#L{k}"),
+        None => base_policy_hash.to_string(),
+    };
+
+    // block_ids: Tier 2/3 chunkers have no per-block structure (the whole file
+    // is one Block::Code), so we pass an empty slice — same as using the doc-
+    // level slice without explicit block granularity.
+    let block_ids: Vec<BlockId> = vec![];
+
+    let chunk_id = id_for_chunk(
+        &DocumentId(doc.doc_id.0.clone()),
+        chunker_version,
+        &block_ids,
+        &id_hash,
+    );
+
+    let token_estimate = text.len().div_ceil(BYTES_PER_TOKEN);
+
+    Chunk {
+        chunk_id,
+        doc_id: DocumentId(doc.doc_id.0.clone()),
+        block_ids,
+        text: text.to_string(),
+        heading_path: Vec::new(),
+        source_spans: vec![span],
+        token_estimate,
+        chunker_version: chunker_version.clone(),
+        policy_hash: base_policy_hash.to_string(),
+    }
+}
--- a/crates/kebab-chunk/tests/code_c_ast_snapshot.rs
+++ b/crates/kebab-chunk/tests/code_c_ast_snapshot.rs
@@ -0,0 +1,196 @@
+//! Snapshot test pinning the `Vec<Chunk>` JSON for a
+//! representative C code `CanonicalDocument`.
+//!
+//! This is an integration test. `kebab-parse-code` is intentionally NOT
+//! a dev-dep (design §6.3 / §8 boundary: AST extraction is parser-side).
+//! The `CanonicalDocument` is built inline from hand-crafted `Block::Code`
+//! units, which is the same pattern used in `code_go_ast_v1.rs`'s
+//! internal `code_doc` test helper.
+//!
+//! Set `UPDATE_SNAPSHOTS=1` to re-bake the baseline.
+
+use std::path::PathBuf;
+
+use kebab_chunk::CodeCAstV1Chunker;
+use kebab_core::{
+    AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+    Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath,
+    id_for_block, id_for_doc,
+};
+use serde_json::Value;
+use time::OffsetDateTime;
+
+fn fixtures_dir() -> PathBuf {
+    PathBuf::from(env!("CARGO_MANIFEST_DIR"))
+        .join("tests")
+        .join("fixtures")
+}
+
+fn fixed_doc() -> CanonicalDocument {
+    let wp = WorkspacePath("projects/record.c".into());
+    let aid = AssetId("c".repeat(64));
+    // Pin parser_version so doc_id / block_ids are reproducible.
+    let pv = ParserVersion("code-c-v1".into());
+    let doc_id = id_for_doc(&wp, &aid, &pv);
+
+    // Representative units:
+    //  0. imports + defines              (lines 1–4,   ≤200)
+    //  1. status_t enum typedef          (lines 6–9,   ≤200)
+    //  2. record_t struct typedef        (lines 11–16, ≤200)
+    //  3. static counter decl glue       (line 18,     ≤200)
+    //  4. parse_record fn                (lines 20–23, ≤200)
+    //  5. print_record fn                (lines 25–27, ≤200)
+    //  6. main fn                        (lines 29–33, ≤200)
+    let raw_units: Vec<(&str, u32, u32, String)> = vec![
+        (
+            "<top-level>",
+            1,
+            18,
+            "#include <stdio.h>\n#include <stdlib.h>\n\n#define MAX_BUF 4096\n\ntypedef enum {\n    OK = 0,\n    ERR_PARSE,\n    ERR_IO,\n} status_t;\n\ntypedef struct {\n    int id;\n    char name[64];\n    status_t status;\n} record_t;\n\nstatic int counter = 0;".to_string(),
+        ),
+        (
+            "parse_record",
+            20,
+            23,
+            "int parse_record(const char *line, record_t *out) {\n    if (line == NULL || out == NULL) return ERR_PARSE;\n    return OK;\n}".to_string(),
+        ),
+        (
+            "print_record",
+            25,
+            27,
+            "void print_record(const record_t *r) {\n    printf(\"[%d] %s (status=%d)\\n\", r->id, r->name, r->status);\n}".to_string(),
+        ),
+        (
+            "main",
+            29,
+            33,
+            "int main(void) {\n    record_t r = { .id = 1, .name = \"foo\", .status = OK };\n    print_record(&r);\n    return 0;\n}".to_string(),
+        ),
+    ];
+
+    let blocks: Vec<Block> = raw_units
+        .iter()
+        .enumerate()
+        .map(|(i, (sym, ls, le, code))| {
+            let span = SourceSpan::Code {
+                line_start: *ls,
+                line_end: *le,
+                symbol: Some((*sym).to_string()),
+                lang: Some("c".into()),
+            };
+            let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+            Block::Code(CodeBlock {
+                common: CommonBlock {
+                    block_id: bid,
+                    heading_path: vec![],
+                    source_span: span,
+                },
+                lang: Some("c".into()),
+                code: code.clone(),
+            })
+        })
+        .collect();
+
+    CanonicalDocument {
+        doc_id,
+        source_asset_id: aid,
+        workspace_path: wp,
+        title: "record.c".into(),
+        lang: Lang("und".into()),
+        blocks,
+        metadata: Metadata {
+            aliases: vec![],
+            tags: vec![],
+            created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            source_type: SourceType::Note,
+            trust_level: TrustLevel::Primary,
+            user_id_alias: None,
+            user: Default::default(),
+            repo: Some("kebab".into()),
+            git_branch: Some("main".into()),
+            git_commit: Some("0".repeat(40)),
+            code_lang: Some("c".into()),
+        },
+        provenance: Provenance { events: vec![] },
+        parser_version: pv,
+        schema_version: 1,
+        doc_version: 1,
+        last_chunker_version: None,
+        last_embedding_version: None,
+    }
+}
+
+fn fixed_policy() -> ChunkPolicy {
+    ChunkPolicy {
+        target_tokens: 500,
+        overlap_tokens: 80,
+        respect_markdown_headings: false,
+        chunker_version: ChunkerVersion("code-c-ast-v1".into()),
+    }
+}
+
+#[test]
+fn code_c_ast_chunks_snapshot() {
+    let doc = fixed_doc();
+    let policy = fixed_policy();
+
+    let chunks = CodeCAstV1Chunker.chunk(&doc, &policy).expect("chunk");
+    let actual = serde_json::to_value(&chunks).unwrap();
+
+    let dir = fixtures_dir();
+    let baseline_path = dir.join("code-sample.c.chunks.snapshot.json");
+    let baseline_text = match std::fs::read_to_string(&baseline_path) {
+        Ok(s) => s,
+        Err(_) if std::env::var("UPDATE_SNAPSHOTS").is_ok() => {
+            std::fs::create_dir_all(&dir).unwrap();
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            return;
+        }
+        Err(e) => panic!(
+            "missing baseline {}; run with UPDATE_SNAPSHOTS=1 to create: {e}",
+            baseline_path.display()
+        ),
+    };
+    let expected: Value = serde_json::from_str(&baseline_text).expect("baseline parses as json");
+
+    if actual != expected {
+        if std::env::var("UPDATE_SNAPSHOTS").is_ok() {
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            eprintln!("updated baseline {}", baseline_path.display());
+            return;
+        }
+        let pretty = serde_json::to_string_pretty(&actual).unwrap();
+        panic!(
+            "code-c-ast-v1 chunks snapshot drift\n\
+             --- expected ({}) ---\n{baseline_text}\n\
+             --- actual ---\n{pretty}\n\
+             If intentional, re-run with UPDATE_SNAPSHOTS=1.",
+            baseline_path.display()
+        );
+    }
+}
+
+/// Determinism cross-check: re-running the same pipeline yields the same
+/// chunk_ids byte-for-byte.
+#[test]
+fn code_c_ast_chunks_are_deterministic() {
+    let policy = fixed_policy();
+    let baseline: Vec<String> = CodeCAstV1Chunker
+        .chunk(&fixed_doc(), &policy)
+        .unwrap()
+        .into_iter()
+        .map(|c| c.chunk_id.0)
+        .collect();
+    for _ in 0..5 {
+        let again: Vec<String> = CodeCAstV1Chunker
+            .chunk(&fixed_doc(), &policy)
+            .unwrap()
+            .into_iter()
+            .map(|c| c.chunk_id.0)
+            .collect();
+        assert_eq!(again, baseline);
+    }
+}
--- a/crates/kebab-chunk/tests/code_cpp_ast_snapshot.rs
+++ b/crates/kebab-chunk/tests/code_cpp_ast_snapshot.rs
@@ -0,0 +1,325 @@
+//! Snapshot test pinning the `Vec<Chunk>` JSON for a
+//! representative C++ code `CanonicalDocument`.
+//!
+//! Two complementary tests:
+//! 1. `code_cpp_ast_chunks_snapshot` — hand-built `fixed_doc()` validates the
+//!    chunker's 1:1 mapping (design §6.3 / §8 boundary: no parse-code dep needed).
+//! 2. `code_cpp_ast_extractor_snapshot` — invokes `CppAstExtractor` against the
+//!    real `tests/fixtures/sample.cpp` fixture, validating the extractor → chunker
+//!    end-to-end pipeline. `kebab-parse-code` is a dev-dep (same pattern as
+//!    `kebab-parse-md` in Markdown snapshot tests).
+//!
+//! Set `UPDATE_SNAPSHOTS=1` to re-bake the baseline.
+
+use std::path::PathBuf;
+
+use kebab_chunk::CodeCppAstV1Chunker;
+use kebab_core::{
+    AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+    Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath,
+    id_for_block, id_for_doc,
+};
+use kebab_parse_code::CppAstExtractor;
+use serde_json::Value;
+use time::OffsetDateTime;
+
+fn fixtures_dir() -> PathBuf {
+    PathBuf::from(env!("CARGO_MANIFEST_DIR"))
+        .join("tests")
+        .join("fixtures")
+}
+
+fn fixed_doc() -> CanonicalDocument {
+    let wp = WorkspacePath("projects/record.cpp".into());
+    let aid = AssetId("c".repeat(64));
+    // Pin parser_version so doc_id / block_ids are reproducible.
+    let pv = ParserVersion("code-cpp-v1".into());
+    let doc_id = id_for_doc(&wp, &aid, &pv);
+
+    // Representative units (C++ specific):
+    //  0. includes + namespace opening  (lines 1–4,   ≤200)
+    //  1. class definition              (lines 6–20,  ≤200)
+    //  2. template function             (lines 22–25, ≤200)
+    //  3. namespace closing + free fn   (lines 27–29, ≤200)
+    //  4. main fn                       (lines 31–34, ≤200)
+    let raw_units: Vec<(&str, u32, u32, String)> = vec![
+        (
+            "<top-level>",
+            1,
+            4,
+            "#include <string>\n#include <vector>\n\nnamespace kebab {".to_string(),
+        ),
+        (
+            "kebab::chunk::MdHeadingV1Chunker",
+            6,
+            20,
+            "class MdHeadingV1Chunker {\npublic:\n    MdHeadingV1Chunker() = default;\n    ~MdHeadingV1Chunker() = default;\n\n    std::string chunk_doc(const std::string& doc) {\n        return doc;\n    }\n\n    int operator()(int x) const {\n        return x * 2;\n    }\n\nprivate:\n    int counter_ = 0;\n};".to_string(),
+        ),
+        (
+            "kebab::identity",
+            22,
+            25,
+            "template <typename T>\nT identity(T value) {\n    return value;\n}".to_string(),
+        ),
+        (
+            "kebab::global_helper",
+            27,
+            29,
+            "void global_helper() {\n    // free function in kebab namespace\n}".to_string(),
+        ),
+        (
+            "main",
+            31,
+            34,
+            "int main() {\n    kebab::chunk::MdHeadingV1Chunker c;\n    return 0;\n}".to_string(),
+        ),
+    ];
+
+    let blocks: Vec<Block> = raw_units
+        .iter()
+        .enumerate()
+        .map(|(i, (sym, ls, le, code))| {
+            let span = SourceSpan::Code {
+                line_start: *ls,
+                line_end: *le,
+                symbol: Some((*sym).to_string()),
+                lang: Some("cpp".into()),
+            };
+            let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+            Block::Code(CodeBlock {
+                common: CommonBlock {
+                    block_id: bid,
+                    heading_path: vec![],
+                    source_span: span,
+                },
+                lang: Some("cpp".into()),
+                code: code.clone(),
+            })
+        })
+        .collect();
+
+    CanonicalDocument {
+        doc_id,
+        source_asset_id: aid,
+        workspace_path: wp,
+        title: "record.cpp".into(),
+        lang: Lang("und".into()),
+        blocks,
+        metadata: Metadata {
+            aliases: vec![],
+            tags: vec![],
+            created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            source_type: SourceType::Note,
+            trust_level: TrustLevel::Primary,
+            user_id_alias: None,
+            user: Default::default(),
+            repo: Some("kebab".into()),
+            git_branch: Some("main".into()),
+            git_commit: Some("0".repeat(40)),
+            code_lang: Some("cpp".into()),
+        },
+        provenance: Provenance { events: vec![] },
+        parser_version: pv,
+        schema_version: 1,
+        doc_version: 1,
+        last_chunker_version: None,
+        last_embedding_version: None,
+    }
+}
+
+fn fixed_policy() -> ChunkPolicy {
+    ChunkPolicy {
+        target_tokens: 500,
+        overlap_tokens: 80,
+        respect_markdown_headings: false,
+        chunker_version: ChunkerVersion("code-cpp-ast-v1".into()),
+    }
+}
+
+// ---------------------------------------------------------------------------
+// Helper: run the real CppAstExtractor against tests/fixtures/sample.cpp
+// ---------------------------------------------------------------------------
+
+fn extract_cpp_fixture() -> CanonicalDocument {
+    use kebab_core::{
+        AssetId, AssetStorage, Checksum, ExtractConfig, ExtractContext, Extractor, RawAsset,
+        SourceUri, WorkspacePath,
+    };
+    use std::path::PathBuf;
+
+    let bytes = std::fs::read(fixtures_dir().join("sample.cpp")).expect("read sample.cpp fixture");
+    let src = String::from_utf8(bytes).expect("fixture is valid UTF-8");
+    let wp = WorkspacePath("tests/fixtures/sample.cpp".to_string());
+    let asset = RawAsset {
+        asset_id: AssetId("e".repeat(64)),
+        source_uri: SourceUri::File(PathBuf::from("tests/fixtures/sample.cpp")),
+        workspace_path: wp,
+        media_type: kebab_core::MediaType::Code("cpp".to_string()),
+        byte_len: src.len() as u64,
+        checksum: Checksum("f".repeat(64)),
+        discovered_at: time::OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+        stored: AssetStorage::Reference {
+            path: PathBuf::from("tests/fixtures/sample.cpp"),
+            sha: Checksum("f".repeat(64)),
+        },
+    };
+    let cfg = ExtractConfig::default();
+    let root = PathBuf::from("/tmp");
+    let ctx = ExtractContext {
+        asset: &asset,
+        workspace_root: &root,
+        config: &cfg,
+    };
+    CppAstExtractor::new().extract(&ctx, src.as_bytes()).unwrap()
+}
+
+// ---------------------------------------------------------------------------
+// Test 1 (hand-built): chunker-only 1:1 mapping validation
+// ---------------------------------------------------------------------------
+
+#[test]
+fn code_cpp_ast_chunks_snapshot() {
+    let doc = fixed_doc();
+    let policy = fixed_policy();
+
+    let chunks = CodeCppAstV1Chunker.chunk(&doc, &policy).expect("chunk");
+    let actual = serde_json::to_value(&chunks).unwrap();
+
+    let dir = fixtures_dir();
+    let baseline_path = dir.join("code-sample.cpp.chunks.snapshot.json");
+    let baseline_text = match std::fs::read_to_string(&baseline_path) {
+        Ok(s) => s,
+        Err(_) if std::env::var("UPDATE_SNAPSHOTS").is_ok() => {
+            std::fs::create_dir_all(&dir).unwrap();
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            return;
+        }
+        Err(e) => panic!(
+            "missing baseline {}; run with UPDATE_SNAPSHOTS=1 to create: {e}",
+            baseline_path.display()
+        ),
+    };
+    let expected: Value = serde_json::from_str(&baseline_text).expect("baseline parses as json");
+
+    if actual != expected {
+        if std::env::var("UPDATE_SNAPSHOTS").is_ok() {
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            eprintln!("updated baseline {}", baseline_path.display());
+            return;
+        }
+        let pretty = serde_json::to_string_pretty(&actual).unwrap();
+        panic!(
+            "code-cpp-ast-v1 chunks snapshot drift\n\
+             --- expected ({}) ---\n{baseline_text}\n\
+             --- actual ---\n{pretty}\n\
+             If intentional, re-run with UPDATE_SNAPSHOTS=1.",
+            baseline_path.display()
+        );
+    }
+}
+
+/// Determinism cross-check: re-running the same pipeline yields the same
+/// chunk_ids byte-for-byte.
+#[test]
+fn code_cpp_ast_chunks_are_deterministic() {
+    let policy = fixed_policy();
+    let baseline: Vec<String> = CodeCppAstV1Chunker
+        .chunk(&fixed_doc(), &policy)
+        .unwrap()
+        .into_iter()
+        .map(|c| c.chunk_id.0)
+        .collect();
+    for _ in 0..5 {
+        let again: Vec<String> = CodeCppAstV1Chunker
+            .chunk(&fixed_doc(), &policy)
+            .unwrap()
+            .into_iter()
+            .map(|c| c.chunk_id.0)
+            .collect();
+        assert_eq!(again, baseline);
+    }
+}
+
+// ---------------------------------------------------------------------------
+// Test 2 (real extractor): end-to-end extractor → chunker pipeline
+// ---------------------------------------------------------------------------
+
+/// Validates that the real `CppAstExtractor` processes `sample.cpp` and
+/// emits the expected set of symbols through the full chunker pipeline.
+///
+/// `sample.cpp` contains:
+/// - `#include` directives + nested namespace `kebab::chunk` → glue + struct unit
+/// - `class MdHeadingV1Chunker` with methods (ctor, dtor, chunk_doc, operator())
+/// - `template <typename T> T identity(T value)` (template fn)
+/// - `void kebab::global_helper()` (free fn in namespace)
+/// - `int main()` (global free fn)
+#[test]
+fn code_cpp_ast_extractor_snapshot() {
+    let doc = extract_cpp_fixture();
+
+    // Verify the extractor emits all expected named units.
+    let block_syms: Vec<Option<String>> = doc.blocks.iter().filter_map(|b| match b {
+        Block::Code(c) => match &c.common.source_span {
+            SourceSpan::Code { symbol, .. } => Some(symbol.clone()),
+            _ => None,
+        },
+        _ => None,
+    }).collect();
+
+    // Must include namespace-qualified class and its methods
+    assert!(
+        block_syms.iter().any(|s| s.as_deref() == Some("kebab::chunk::MdHeadingV1Chunker")),
+        "class unit missing: {block_syms:?}"
+    );
+    assert!(
+        block_syms.iter().any(|s| s.as_deref() == Some("kebab::chunk::MdHeadingV1Chunker::MdHeadingV1Chunker")),
+        "ctor unit missing: {block_syms:?}"
+    );
+    assert!(
+        block_syms.iter().any(|s| s.as_deref() == Some("kebab::chunk::MdHeadingV1Chunker::~MdHeadingV1Chunker")),
+        "dtor unit missing: {block_syms:?}"
+    );
+    assert!(
+        block_syms.iter().any(|s| s.as_deref() == Some("kebab::chunk::MdHeadingV1Chunker::chunk_doc")),
+        "chunk_doc unit missing: {block_syms:?}"
+    );
+    assert!(
+        block_syms.iter().any(|s| s.as_deref() == Some("kebab::chunk::MdHeadingV1Chunker::operator()")),
+        "operator() unit missing: {block_syms:?}"
+    );
+    // Template function (inside kebab::chunk namespace in the fixture)
+    assert!(
+        block_syms.iter().any(|s| s.as_deref() == Some("kebab::chunk::identity")),
+        "identity template fn unit missing: {block_syms:?}"
+    );
+    // Free function in outer namespace
+    assert!(
+        block_syms.iter().any(|s| s.as_deref() == Some("kebab::global_helper")),
+        "global_helper unit missing: {block_syms:?}"
+    );
+    // Global main
+    assert!(
+        block_syms.iter().any(|s| s.as_deref() == Some("main")),
+        "main unit missing: {block_syms:?}"
+    );
+}
+
+/// End-to-end chunker output from real extractor is deterministic.
+#[test]
+fn code_cpp_ast_extractor_chunks_deterministic() {
+    let doc1 = extract_cpp_fixture();
+    let doc2 = extract_cpp_fixture();
+    assert_eq!(doc1.blocks, doc2.blocks, "extractor output non-deterministic");
+
+    let policy = fixed_policy();
+    let chunks1 = CodeCppAstV1Chunker.chunk(&doc1, &policy).unwrap();
+    let chunks2 = CodeCppAstV1Chunker.chunk(&doc2, &policy).unwrap();
+    assert_eq!(
+        chunks1.iter().map(|c| c.chunk_id.0.clone()).collect::<Vec<_>>(),
+        chunks2.iter().map(|c| c.chunk_id.0.clone()).collect::<Vec<_>>(),
+        "chunker output non-deterministic"
+    );
+}
--- a/crates/kebab-chunk/tests/code_go_ast_snapshot.rs
+++ b/crates/kebab-chunk/tests/code_go_ast_snapshot.rs
@@ -0,0 +1,221 @@
+//! Snapshot test pinning the `Vec<Chunk>` JSON for a
+//! representative Go code `CanonicalDocument`.
+//!
+//! This is an integration test. `kebab-parse-code` is intentionally NOT
+//! a dev-dep (design §6.3 / §8 boundary: AST extraction is parser-side).
+//! The `CanonicalDocument` is built inline from hand-crafted `Block::Code`
+//! units, which is the same pattern used in `code_rust_ast_v1.rs`'s
+//! internal `code_doc` test helper.
+//!
+//! Set `UPDATE_SNAPSHOTS=1` to re-bake the baseline.
+
+use std::path::PathBuf;
+
+use kebab_chunk::CodeGoAstV1Chunker;
+use kebab_core::{
+    AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+    Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath,
+    id_for_block, id_for_doc,
+};
+use serde_json::Value;
+use time::OffsetDateTime;
+
+fn fixtures_dir() -> PathBuf {
+    PathBuf::from(env!("CARGO_MANIFEST_DIR"))
+        .join("tests")
+        .join("fixtures")
+}
+
+fn fixed_doc() -> CanonicalDocument {
+    let wp = WorkspacePath("kebab_eval/metrics.go".into());
+    let aid = AssetId("b".repeat(64));
+    // Pin parser_version so doc_id / block_ids are reproducible.
+    let pv = ParserVersion("code-go-v1".into());
+    let doc_id = id_for_doc(&wp, &aid, &pv);
+
+    // Build a >200-line function body to force split_oversize.
+    let big_body: String = {
+        let header = "func BigCompute(data []int) int {\n";
+        let body: String = (0..210u32)
+            .map(|i| format!("\tv{i} := 0\n\tif {i} < len(data) {{\n\t\tv{i} = data[{i}]\n\t}}\n"))
+            .collect();
+        let footer = "\treturn len(data)\n}";
+        format!("{header}{body}{footer}")
+    };
+    let big_line_count = big_body.lines().count() as u32;
+    let big_line_end = 48 + big_line_count - 1;
+
+    // Representative units:
+    //  0. import block                    (lines 1–5,   ≤200)
+    //  1. free fn `ComputeMRR`            (lines 7–12,  ≤200)
+    //  2. struct `MetricsCollector`       (lines 14–20, ≤200)
+    //  3. struct `BaseEvaluator`          (lines 22–30, ≤200)
+    //  4. method `Run`                    (lines 32–38, ≤200)
+    //  5. method `Report`                 (lines 40–46, ≤200)
+    //  6. BigCompute (>200 lines)         to force split_oversize
+    let raw_units: Vec<(&str, u32, u32, String)> = vec![
+        (
+            "imports",
+            1,
+            5,
+            "import (\n\t\"fmt\"\n\t\"os\"\n\t\"strings\"\n)".to_string(),
+        ),
+        (
+            "ComputeMRR",
+            7,
+            12,
+            "func ComputeMRR(scores []float64) float64 {\n\tif len(scores) == 0 {\n\t\treturn 0.0\n\t}\n\t_ = fmt.Sprintf(\"%v\", scores)\n\treturn 1.0 / float64(len(scores))\n}".to_string(),
+        ),
+        (
+            "MetricsCollector",
+            14,
+            20,
+            "type MetricsCollector struct {\n\tScores []float64\n\tLabels []string\n\tCounts map[string]int\n\tTotals map[string]float64\n\tTags   []string\n}".to_string(),
+        ),
+        (
+            "BaseEvaluator",
+            22,
+            30,
+            "type BaseEvaluator struct {\n\tName string\n}\n\nfunc (e *BaseEvaluator) Evaluate(data []string) error {\n\t_ = os.Stderr\n\t_ = strings.Join(data, \",\")\n\treturn nil\n}".to_string(),
+        ),
+        (
+            "MetricsCollector.Run",
+            32,
+            38,
+            "func (m *MetricsCollector) Run(inputs []float64) {\n\tfor _, inp := range inputs {\n\t\tm.Scores = append(\n\t\t\tm.Scores,\n\t\t\tinp,\n\t\t)\n\t}\n}".to_string(),
+        ),
+        (
+            "MetricsCollector.Report",
+            40,
+            46,
+            "func (m *MetricsCollector) Report() map[string]interface{} {\n\treturn map[string]interface{}{\n\t\t\"mean\":  0.0,\n\t\t\"count\": len(m.Scores),\n\t\t\"tags\":  m.Tags,\n\t}\n}".to_string(),
+        ),
+        ("BigCompute", 48, big_line_end, big_body),
+    ];
+
+    let blocks: Vec<Block> = raw_units
+        .iter()
+        .enumerate()
+        .map(|(i, (sym, ls, le, code))| {
+            let span = SourceSpan::Code {
+                line_start: *ls,
+                line_end: *le,
+                symbol: Some((*sym).to_string()),
+                lang: Some("go".into()),
+            };
+            let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+            Block::Code(CodeBlock {
+                common: CommonBlock {
+                    block_id: bid,
+                    heading_path: vec![],
+                    source_span: span,
+                },
+                lang: Some("go".into()),
+                code: code.clone(),
+            })
+        })
+        .collect();
+
+    CanonicalDocument {
+        doc_id,
+        source_asset_id: aid,
+        workspace_path: wp,
+        title: "metrics.go".into(),
+        lang: Lang("und".into()),
+        blocks,
+        metadata: Metadata {
+            aliases: vec![],
+            tags: vec![],
+            created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            source_type: SourceType::Note,
+            trust_level: TrustLevel::Primary,
+            user_id_alias: None,
+            user: Default::default(),
+            repo: Some("kebab".into()),
+            git_branch: Some("main".into()),
+            git_commit: Some("0".repeat(40)),
+            code_lang: Some("go".into()),
+        },
+        provenance: Provenance { events: vec![] },
+        parser_version: pv,
+        schema_version: 1,
+        doc_version: 1,
+        last_chunker_version: None,
+        last_embedding_version: None,
+    }
+}
+
+fn fixed_policy() -> ChunkPolicy {
+    ChunkPolicy {
+        target_tokens: 500,
+        overlap_tokens: 80,
+        respect_markdown_headings: false,
+        chunker_version: ChunkerVersion("code-go-ast-v1".into()),
+    }
+}
+
+#[test]
+fn code_go_ast_chunks_snapshot() {
+    let doc = fixed_doc();
+    let policy = fixed_policy();
+
+    let chunks = CodeGoAstV1Chunker.chunk(&doc, &policy).expect("chunk");
+    let actual = serde_json::to_value(&chunks).unwrap();
+
+    let dir = fixtures_dir();
+    let baseline_path = dir.join("code-sample.go.chunks.snapshot.json");
+    let baseline_text = match std::fs::read_to_string(&baseline_path) {
+        Ok(s) => s,
+        Err(_) if std::env::var("UPDATE_SNAPSHOTS").is_ok() => {
+            std::fs::create_dir_all(&dir).unwrap();
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            return;
+        }
+        Err(e) => panic!(
+            "missing baseline {}; run with UPDATE_SNAPSHOTS=1 to create: {e}",
+            baseline_path.display()
+        ),
+    };
+    let expected: Value = serde_json::from_str(&baseline_text).expect("baseline parses as json");
+
+    if actual != expected {
+        if std::env::var("UPDATE_SNAPSHOTS").is_ok() {
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            eprintln!("updated baseline {}", baseline_path.display());
+            return;
+        }
+        let pretty = serde_json::to_string_pretty(&actual).unwrap();
+        panic!(
+            "code-go-ast-v1 chunks snapshot drift\n\
+             --- expected ({}) ---\n{baseline_text}\n\
+             --- actual ---\n{pretty}\n\
+             If intentional, re-run with UPDATE_SNAPSHOTS=1.",
+            baseline_path.display()
+        );
+    }
+}
+
+/// Determinism cross-check: re-running the same pipeline yields the same
+/// chunk_ids byte-for-byte.
+#[test]
+fn code_go_ast_chunks_are_deterministic() {
+    let policy = fixed_policy();
+    let baseline: Vec<String> = CodeGoAstV1Chunker
+        .chunk(&fixed_doc(), &policy)
+        .unwrap()
+        .into_iter()
+        .map(|c| c.chunk_id.0)
+        .collect();
+    for _ in 0..5 {
+        let again: Vec<String> = CodeGoAstV1Chunker
+            .chunk(&fixed_doc(), &policy)
+            .unwrap()
+            .into_iter()
+            .map(|c| c.chunk_id.0)
+            .collect();
+        assert_eq!(again, baseline);
+    }
+}
--- a/crates/kebab-chunk/tests/code_java_ast_snapshot.rs
+++ b/crates/kebab-chunk/tests/code_java_ast_snapshot.rs
@@ -0,0 +1,221 @@
+//! Snapshot test pinning the `Vec<Chunk>` JSON for a
+//! representative Java code `CanonicalDocument`.
+//!
+//! This is an integration test. `kebab-parse-code` is intentionally NOT
+//! a dev-dep (design §6.3 / §8 boundary: AST extraction is parser-side).
+//! The `CanonicalDocument` is built inline from hand-crafted `Block::Code`
+//! units, which is the same pattern used in `code_rust_ast_v1.rs`'s
+//! internal `code_doc` test helper.
+//!
+//! Set `UPDATE_SNAPSHOTS=1` to re-bake the baseline.
+
+use std::path::PathBuf;
+
+use kebab_chunk::CodeJavaAstV1Chunker;
+use kebab_core::{
+    AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+    Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath,
+    id_for_block, id_for_doc,
+};
+use serde_json::Value;
+use time::OffsetDateTime;
+
+fn fixtures_dir() -> PathBuf {
+    PathBuf::from(env!("CARGO_MANIFEST_DIR"))
+        .join("tests")
+        .join("fixtures")
+}
+
+fn fixed_doc() -> CanonicalDocument {
+    let wp = WorkspacePath("src/main/java/com/example/Metrics.java".into());
+    let aid = AssetId("b".repeat(64));
+    // Pin parser_version so doc_id / block_ids are reproducible.
+    let pv = ParserVersion("code-java-v1".into());
+    let doc_id = id_for_doc(&wp, &aid, &pv);
+
+    // Build a >200-line method body to force split_oversize.
+    let big_body: String = {
+        let header = "public class BigCompute {\n    public int compute(int[] data) {\n";
+        let body: String = (0..210u32)
+            .map(|i| format!("        int v{i} = {i} < data.length ? data[{i}] : 0;\n"))
+            .collect();
+        let footer = "        return data.length;\n    }\n}";
+        format!("{header}{body}{footer}")
+    };
+    let big_line_count = big_body.lines().count() as u32;
+    let big_line_end = 48 + big_line_count - 1;
+
+    // Representative units:
+    //  0. import block                      (lines 1–5,   ≤200)
+    //  1. free method `computeMRR`          (lines 7–12,  ≤200)
+    //  2. class `MetricsCollector`          (lines 14–20, ≤200)
+    //  3. class `BaseEvaluator`             (lines 22–30, ≤200)
+    //  4. method `MetricsCollector.run`     (lines 32–38, ≤200)
+    //  5. method `MetricsCollector.report`  (lines 40–46, ≤200)
+    //  6. BigCompute (>200 lines)           to force split_oversize
+    let raw_units: Vec<(&str, u32, u32, String)> = vec![
+        (
+            "imports",
+            1,
+            5,
+            "import java.util.List;\nimport java.util.Map;\nimport java.util.ArrayList;\nimport java.util.HashMap;\nimport java.util.stream.Collectors;".to_string(),
+        ),
+        (
+            "computeMRR",
+            7,
+            12,
+            "public static double computeMRR(List<Double> scores) {\n    if (scores.isEmpty()) {\n        return 0.0;\n    }\n    return 1.0 / scores.size();\n}".to_string(),
+        ),
+        (
+            "MetricsCollector",
+            14,
+            20,
+            "public class MetricsCollector {\n    private List<Double> scores;\n    private List<String> labels;\n    private Map<String, Integer> counts;\n    private Map<String, Double> totals;\n    private List<String> tags;\n}".to_string(),
+        ),
+        (
+            "BaseEvaluator",
+            22,
+            30,
+            "public class BaseEvaluator {\n    private String name;\n\n    public BaseEvaluator(String name) {\n        this.name = name;\n    }\n\n    public void evaluate(List<String> data) throws Exception {\n        String joined = String.join(\",\", data);\n    }\n}".to_string(),
+        ),
+        (
+            "MetricsCollector.run",
+            32,
+            38,
+            "public void run(List<Double> inputs) {\n    for (Double inp : inputs) {\n        scores.add(\n            inp\n        );\n    }\n}".to_string(),
+        ),
+        (
+            "MetricsCollector.report",
+            40,
+            46,
+            "public Map<String, Object> report() {\n    Map<String, Object> result = new HashMap<>();\n    result.put(\"mean\", 0.0);\n    result.put(\"count\", scores.size());\n    result.put(\"tags\", tags);\n    return result;\n}".to_string(),
+        ),
+        ("BigCompute", 48, big_line_end, big_body),
+    ];
+
+    let blocks: Vec<Block> = raw_units
+        .iter()
+        .enumerate()
+        .map(|(i, (sym, ls, le, code))| {
+            let span = SourceSpan::Code {
+                line_start: *ls,
+                line_end: *le,
+                symbol: Some((*sym).to_string()),
+                lang: Some("java".into()),
+            };
+            let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+            Block::Code(CodeBlock {
+                common: CommonBlock {
+                    block_id: bid,
+                    heading_path: vec![],
+                    source_span: span,
+                },
+                lang: Some("java".into()),
+                code: code.clone(),
+            })
+        })
+        .collect();
+
+    CanonicalDocument {
+        doc_id,
+        source_asset_id: aid,
+        workspace_path: wp,
+        title: "Metrics.java".into(),
+        lang: Lang("und".into()),
+        blocks,
+        metadata: Metadata {
+            aliases: vec![],
+            tags: vec![],
+            created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            source_type: SourceType::Note,
+            trust_level: TrustLevel::Primary,
+            user_id_alias: None,
+            user: Default::default(),
+            repo: Some("kebab".into()),
+            git_branch: Some("main".into()),
+            git_commit: Some("0".repeat(40)),
+            code_lang: Some("java".into()),
+        },
+        provenance: Provenance { events: vec![] },
+        parser_version: pv,
+        schema_version: 1,
+        doc_version: 1,
+        last_chunker_version: None,
+        last_embedding_version: None,
+    }
+}
+
+fn fixed_policy() -> ChunkPolicy {
+    ChunkPolicy {
+        target_tokens: 500,
+        overlap_tokens: 80,
+        respect_markdown_headings: false,
+        chunker_version: ChunkerVersion("code-java-ast-v1".into()),
+    }
+}
+
+#[test]
+fn code_java_ast_chunks_snapshot() {
+    let doc = fixed_doc();
+    let policy = fixed_policy();
+
+    let chunks = CodeJavaAstV1Chunker.chunk(&doc, &policy).expect("chunk");
+    let actual = serde_json::to_value(&chunks).unwrap();
+
+    let dir = fixtures_dir();
+    let baseline_path = dir.join("code-sample.java.chunks.snapshot.json");
+    let baseline_text = match std::fs::read_to_string(&baseline_path) {
+        Ok(s) => s,
+        Err(_) if std::env::var("UPDATE_SNAPSHOTS").is_ok() => {
+            std::fs::create_dir_all(&dir).unwrap();
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            return;
+        }
+        Err(e) => panic!(
+            "missing baseline {}; run with UPDATE_SNAPSHOTS=1 to create: {e}",
+            baseline_path.display()
+        ),
+    };
+    let expected: Value = serde_json::from_str(&baseline_text).expect("baseline parses as json");
+
+    if actual != expected {
+        if std::env::var("UPDATE_SNAPSHOTS").is_ok() {
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            eprintln!("updated baseline {}", baseline_path.display());
+            return;
+        }
+        let pretty = serde_json::to_string_pretty(&actual).unwrap();
+        panic!(
+            "code-java-ast-v1 chunks snapshot drift\n\
+             --- expected ({}) ---\n{baseline_text}\n\
+             --- actual ---\n{pretty}\n\
+             If intentional, re-run with UPDATE_SNAPSHOTS=1.",
+            baseline_path.display()
+        );
+    }
+}
+
+/// Determinism cross-check: re-running the same pipeline yields the same
+/// chunk_ids byte-for-byte.
+#[test]
+fn code_java_ast_chunks_are_deterministic() {
+    let policy = fixed_policy();
+    let baseline: Vec<String> = CodeJavaAstV1Chunker
+        .chunk(&fixed_doc(), &policy)
+        .unwrap()
+        .into_iter()
+        .map(|c| c.chunk_id.0)
+        .collect();
+    for _ in 0..5 {
+        let again: Vec<String> = CodeJavaAstV1Chunker
+            .chunk(&fixed_doc(), &policy)
+            .unwrap()
+            .into_iter()
+            .map(|c| c.chunk_id.0)
+            .collect();
+        assert_eq!(again, baseline);
+    }
+}
--- a/crates/kebab-chunk/tests/code_js_ast_snapshot.rs
+++ b/crates/kebab-chunk/tests/code_js_ast_snapshot.rs
@@ -0,0 +1,221 @@
+//! Snapshot test pinning the `Vec<Chunk>` JSON for a
+//! representative JavaScript code `CanonicalDocument`.
+//!
+//! This is an integration test. `kebab-parse-code` is intentionally NOT
+//! a dev-dep (design §6.3 / §8 boundary: AST extraction is parser-side).
+//! The `CanonicalDocument` is built inline from hand-crafted `Block::Code`
+//! units, which is the same pattern used in `code_rust_ast_v1.rs`'s
+//! internal `code_doc` test helper.
+//!
+//! Set `UPDATE_SNAPSHOTS=1` to re-bake the baseline.
+
+use std::path::PathBuf;
+
+use kebab_chunk::CodeJsAstV1Chunker;
+use kebab_core::{
+    AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+    Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath,
+    id_for_block, id_for_doc,
+};
+use serde_json::Value;
+use time::OffsetDateTime;
+
+fn fixtures_dir() -> PathBuf {
+    PathBuf::from(env!("CARGO_MANIFEST_DIR"))
+        .join("tests")
+        .join("fixtures")
+}
+
+fn fixed_doc() -> CanonicalDocument {
+    let wp = WorkspacePath("src/bar.js".into());
+    let aid = AssetId("b".repeat(64));
+    // Pin parser_version so doc_id / block_ids are reproducible.
+    let pv = ParserVersion("code-js-v1".into());
+    let doc_id = id_for_doc(&wp, &aid, &pv);
+
+    // Build a >200-line function body to force split_oversize.
+    let big_body: String = {
+        let header = "function bigTransform(items) {\n";
+        let body: String = (0..210u32)
+            .map(|i| format!("  const v{i} = items[{i}] !== undefined ? items[{i}] : null;\n"))
+            .collect();
+        let footer = "  return items;\n}";
+        format!("{header}{body}{footer}")
+    };
+    let big_line_count = big_body.lines().count() as u32;
+    let big_line_end = 48 + big_line_count - 1;
+
+    // Representative units:
+    //  0. require/import block       (lines 1–5,   ≤200)
+    //  1. free fn `add`              (lines 7–12,  ≤200)
+    //  2. class `EventBus`           (lines 14–20, ≤200)
+    //  3. class `BaseHandler`        (lines 22–30, ≤200)
+    //  4. method `EventBus.emit`     (lines 32–38, ≤200)
+    //  5. method `EventBus.on`       (lines 40–46, ≤200)
+    //  6. bigTransform (>200 lines)  to force split_oversize
+    let raw_units: Vec<(&str, u32, u32, String)> = vec![
+        (
+            "requires",
+            1,
+            5,
+            "const fs = require('fs');\nconst path = require('path');\nconst { EventEmitter } = require('events');\nconst assert = require('assert');\nconst crypto = require('crypto');".to_string(),
+        ),
+        (
+            "add",
+            7,
+            12,
+            "export function add(a, b) {\n  if (typeof a !== 'number') throw new TypeError('a');\n  if (typeof b !== 'number') throw new TypeError('b');\n  const result = a + b;\n  assert(isFinite(result));\n  return result;\n}".to_string(),
+        ),
+        (
+            "EventBus",
+            14,
+            20,
+            "class EventBus {\n  constructor() {\n    this._handlers = new Map();\n    this._history = [];\n    this._maxHistory = 100;\n    this._seq = 0;\n  }\n}".to_string(),
+        ),
+        (
+            "BaseHandler",
+            22,
+            30,
+            "class BaseHandler {\n  handle(event) {\n    throw new Error('not implemented');\n  }\n  batchHandle(events) {\n    const results = [];\n    for (const ev of events) {\n      results.push(this.handle(ev));\n    }\n    return results;\n  }\n}".to_string(),
+        ),
+        (
+            "EventBus.emit",
+            32,
+            38,
+            "class EventBus {\n  emit(name, payload) {\n    const handlers = this._handlers.get(name) ?? [];\n    for (const h of handlers) {\n      h(payload);\n    }\n    return this;\n  }\n}".to_string(),
+        ),
+        (
+            "EventBus.on",
+            40,
+            46,
+            "class EventBus {\n  on(name, handler) {\n    if (!this._handlers.has(name)) {\n      this._handlers.set(name, []);\n    }\n    this._handlers.get(name).push(handler);\n    return this;\n  }\n}".to_string(),
+        ),
+        ("bigTransform", 48, big_line_end, big_body),
+    ];
+
+    let blocks: Vec<Block> = raw_units
+        .iter()
+        .enumerate()
+        .map(|(i, (sym, ls, le, code))| {
+            let span = SourceSpan::Code {
+                line_start: *ls,
+                line_end: *le,
+                symbol: Some((*sym).to_string()),
+                lang: Some("javascript".into()),
+            };
+            let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+            Block::Code(CodeBlock {
+                common: CommonBlock {
+                    block_id: bid,
+                    heading_path: vec![],
+                    source_span: span,
+                },
+                lang: Some("javascript".into()),
+                code: code.clone(),
+            })
+        })
+        .collect();
+
+    CanonicalDocument {
+        doc_id,
+        source_asset_id: aid,
+        workspace_path: wp,
+        title: "bar.js".into(),
+        lang: Lang("und".into()),
+        blocks,
+        metadata: Metadata {
+            aliases: vec![],
+            tags: vec![],
+            created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            source_type: SourceType::Note,
+            trust_level: TrustLevel::Primary,
+            user_id_alias: None,
+            user: Default::default(),
+            repo: Some("kebab".into()),
+            git_branch: Some("main".into()),
+            git_commit: Some("0".repeat(40)),
+            code_lang: Some("javascript".into()),
+        },
+        provenance: Provenance { events: vec![] },
+        parser_version: pv,
+        schema_version: 1,
+        doc_version: 1,
+        last_chunker_version: None,
+        last_embedding_version: None,
+    }
+}
+
+fn fixed_policy() -> ChunkPolicy {
+    ChunkPolicy {
+        target_tokens: 500,
+        overlap_tokens: 80,
+        respect_markdown_headings: false,
+        chunker_version: ChunkerVersion("code-js-ast-v1".into()),
+    }
+}
+
+#[test]
+fn code_js_ast_chunks_snapshot() {
+    let doc = fixed_doc();
+    let policy = fixed_policy();
+
+    let chunks = CodeJsAstV1Chunker.chunk(&doc, &policy).expect("chunk");
+    let actual = serde_json::to_value(&chunks).unwrap();
+
+    let dir = fixtures_dir();
+    let baseline_path = dir.join("code-sample.js.chunks.snapshot.json");
+    let baseline_text = match std::fs::read_to_string(&baseline_path) {
+        Ok(s) => s,
+        Err(_) if std::env::var("UPDATE_SNAPSHOTS").is_ok() => {
+            std::fs::create_dir_all(&dir).unwrap();
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            return;
+        }
+        Err(e) => panic!(
+            "missing baseline {}; run with UPDATE_SNAPSHOTS=1 to create: {e}",
+            baseline_path.display()
+        ),
+    };
+    let expected: Value = serde_json::from_str(&baseline_text).expect("baseline parses as json");
+
+    if actual != expected {
+        if std::env::var("UPDATE_SNAPSHOTS").is_ok() {
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            eprintln!("updated baseline {}", baseline_path.display());
+            return;
+        }
+        let pretty = serde_json::to_string_pretty(&actual).unwrap();
+        panic!(
+            "code-js-ast-v1 chunks snapshot drift\n\
+             --- expected ({}) ---\n{baseline_text}\n\
+             --- actual ---\n{pretty}\n\
+             If intentional, re-run with UPDATE_SNAPSHOTS=1.",
+            baseline_path.display()
+        );
+    }
+}
+
+/// Determinism cross-check: re-running the same pipeline yields the same
+/// chunk_ids byte-for-byte.
+#[test]
+fn code_js_ast_chunks_are_deterministic() {
+    let policy = fixed_policy();
+    let baseline: Vec<String> = CodeJsAstV1Chunker
+        .chunk(&fixed_doc(), &policy)
+        .unwrap()
+        .into_iter()
+        .map(|c| c.chunk_id.0)
+        .collect();
+    for _ in 0..5 {
+        let again: Vec<String> = CodeJsAstV1Chunker
+            .chunk(&fixed_doc(), &policy)
+            .unwrap()
+            .into_iter()
+            .map(|c| c.chunk_id.0)
+            .collect();
+        assert_eq!(again, baseline);
+    }
+}
--- a/crates/kebab-chunk/tests/code_kotlin_ast_snapshot.rs
+++ b/crates/kebab-chunk/tests/code_kotlin_ast_snapshot.rs
@@ -0,0 +1,221 @@
+//! Snapshot test pinning the `Vec<Chunk>` JSON for a
+//! representative Kotlin code `CanonicalDocument`.
+//!
+//! This is an integration test. `kebab-parse-code` is intentionally NOT
+//! a dev-dep (design §6.3 / §8 boundary: AST extraction is parser-side).
+//! The `CanonicalDocument` is built inline from hand-crafted `Block::Code`
+//! units, which is the same pattern used in `code_rust_ast_v1.rs`'s
+//! internal `code_doc` test helper.
+//!
+//! Set `UPDATE_SNAPSHOTS=1` to re-bake the baseline.
+
+use std::path::PathBuf;
+
+use kebab_chunk::CodeKotlinAstV1Chunker;
+use kebab_core::{
+    AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+    Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath,
+    id_for_block, id_for_doc,
+};
+use serde_json::Value;
+use time::OffsetDateTime;
+
+fn fixtures_dir() -> PathBuf {
+    PathBuf::from(env!("CARGO_MANIFEST_DIR"))
+        .join("tests")
+        .join("fixtures")
+}
+
+fn fixed_doc() -> CanonicalDocument {
+    let wp = WorkspacePath("src/main/kotlin/com/example/Metrics.kt".into());
+    let aid = AssetId("b".repeat(64));
+    // Pin parser_version so doc_id / block_ids are reproducible.
+    let pv = ParserVersion("code-kotlin-v1".into());
+    let doc_id = id_for_doc(&wp, &aid, &pv);
+
+    // Build a >200-line function body to force split_oversize.
+    let big_body: String = {
+        let header = "class BigCompute {\n    fun compute(data: IntArray): Int {\n";
+        let body: String = (0..210u32)
+            .map(|i| format!("        val v{i} = if ({i} < data.size) data[{i}] else 0\n"))
+            .collect();
+        let footer = "        return data.size\n    }\n}";
+        format!("{header}{body}{footer}")
+    };
+    let big_line_count = big_body.lines().count() as u32;
+    let big_line_end = 48 + big_line_count - 1;
+
+    // Representative units:
+    //  0. import block                      (lines 1–5,   ≤200)
+    //  1. top-level fn `computeMRR`         (lines 7–12,  ≤200)
+    //  2. data class `MetricsCollector`     (lines 14–20, ≤200)
+    //  3. class `BaseEvaluator`             (lines 22–30, ≤200)
+    //  4. method `MetricsCollector.run`     (lines 32–38, ≤200)
+    //  5. method `MetricsCollector.report`  (lines 40–46, ≤200)
+    //  6. BigCompute (>200 lines)           to force split_oversize
+    let raw_units: Vec<(&str, u32, u32, String)> = vec![
+        (
+            "imports",
+            1,
+            5,
+            "import kotlin.collections.List\nimport kotlin.collections.Map\nimport kotlin.collections.MutableList\nimport kotlin.collections.MutableMap\nimport kotlin.collections.mutableListOf".to_string(),
+        ),
+        (
+            "computeMRR",
+            7,
+            12,
+            "fun computeMRR(scores: List<Double>): Double {\n    if (scores.isEmpty()) {\n        return 0.0\n    }\n    return 1.0 / scores.size\n}".to_string(),
+        ),
+        (
+            "MetricsCollector",
+            14,
+            20,
+            "data class MetricsCollector(\n    val scores: MutableList<Double> = mutableListOf(),\n    val labels: MutableList<String> = mutableListOf(),\n    val counts: MutableMap<String, Int> = mutableMapOf(),\n    val totals: MutableMap<String, Double> = mutableMapOf(),\n    val tags: MutableList<String> = mutableListOf(),\n)".to_string(),
+        ),
+        (
+            "BaseEvaluator",
+            22,
+            30,
+            "open class BaseEvaluator(val name: String) {\n\n    fun evaluate(data: List<String>) {\n        val joined = data.joinToString(\",\")\n        println(joined)\n    }\n\n    open fun describe(): String = name\n}".to_string(),
+        ),
+        (
+            "MetricsCollector.run",
+            32,
+            38,
+            "fun MetricsCollector.run(inputs: List<Double>) {\n    for (inp in inputs) {\n        scores.add(\n            inp\n        )\n    }\n}".to_string(),
+        ),
+        (
+            "MetricsCollector.report",
+            40,
+            46,
+            "fun MetricsCollector.report(): Map<String, Any> {\n    return mapOf(\n        \"mean\" to 0.0,\n        \"count\" to scores.size,\n        \"tags\" to tags,\n    )\n}".to_string(),
+        ),
+        ("BigCompute", 48, big_line_end, big_body),
+    ];
+
+    let blocks: Vec<Block> = raw_units
+        .iter()
+        .enumerate()
+        .map(|(i, (sym, ls, le, code))| {
+            let span = SourceSpan::Code {
+                line_start: *ls,
+                line_end: *le,
+                symbol: Some((*sym).to_string()),
+                lang: Some("kotlin".into()),
+            };
+            let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+            Block::Code(CodeBlock {
+                common: CommonBlock {
+                    block_id: bid,
+                    heading_path: vec![],
+                    source_span: span,
+                },
+                lang: Some("kotlin".into()),
+                code: code.clone(),
+            })
+        })
+        .collect();
+
+    CanonicalDocument {
+        doc_id,
+        source_asset_id: aid,
+        workspace_path: wp,
+        title: "Metrics.kt".into(),
+        lang: Lang("und".into()),
+        blocks,
+        metadata: Metadata {
+            aliases: vec![],
+            tags: vec![],
+            created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            source_type: SourceType::Note,
+            trust_level: TrustLevel::Primary,
+            user_id_alias: None,
+            user: Default::default(),
+            repo: Some("kebab".into()),
+            git_branch: Some("main".into()),
+            git_commit: Some("0".repeat(40)),
+            code_lang: Some("kotlin".into()),
+        },
+        provenance: Provenance { events: vec![] },
+        parser_version: pv,
+        schema_version: 1,
+        doc_version: 1,
+        last_chunker_version: None,
+        last_embedding_version: None,
+    }
+}
+
+fn fixed_policy() -> ChunkPolicy {
+    ChunkPolicy {
+        target_tokens: 500,
+        overlap_tokens: 80,
+        respect_markdown_headings: false,
+        chunker_version: ChunkerVersion("code-kotlin-ast-v1".into()),
+    }
+}
+
+#[test]
+fn code_kotlin_ast_chunks_snapshot() {
+    let doc = fixed_doc();
+    let policy = fixed_policy();
+
+    let chunks = CodeKotlinAstV1Chunker.chunk(&doc, &policy).expect("chunk");
+    let actual = serde_json::to_value(&chunks).unwrap();
+
+    let dir = fixtures_dir();
+    let baseline_path = dir.join("code-sample.kt.chunks.snapshot.json");
+    let baseline_text = match std::fs::read_to_string(&baseline_path) {
+        Ok(s) => s,
+        Err(_) if std::env::var("UPDATE_SNAPSHOTS").is_ok() => {
+            std::fs::create_dir_all(&dir).unwrap();
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            return;
+        }
+        Err(e) => panic!(
+            "missing baseline {}; run with UPDATE_SNAPSHOTS=1 to create: {e}",
+            baseline_path.display()
+        ),
+    };
+    let expected: Value = serde_json::from_str(&baseline_text).expect("baseline parses as json");
+
+    if actual != expected {
+        if std::env::var("UPDATE_SNAPSHOTS").is_ok() {
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            eprintln!("updated baseline {}", baseline_path.display());
+            return;
+        }
+        let pretty = serde_json::to_string_pretty(&actual).unwrap();
+        panic!(
+            "code-kotlin-ast-v1 chunks snapshot drift\n\
+             --- expected ({}) ---\n{baseline_text}\n\
+             --- actual ---\n{pretty}\n\
+             If intentional, re-run with UPDATE_SNAPSHOTS=1.",
+            baseline_path.display()
+        );
+    }
+}
+
+/// Determinism cross-check: re-running the same pipeline yields the same
+/// chunk_ids byte-for-byte.
+#[test]
+fn code_kotlin_ast_chunks_are_deterministic() {
+    let policy = fixed_policy();
+    let baseline: Vec<String> = CodeKotlinAstV1Chunker
+        .chunk(&fixed_doc(), &policy)
+        .unwrap()
+        .into_iter()
+        .map(|c| c.chunk_id.0)
+        .collect();
+    for _ in 0..5 {
+        let again: Vec<String> = CodeKotlinAstV1Chunker
+            .chunk(&fixed_doc(), &policy)
+            .unwrap()
+            .into_iter()
+            .map(|c| c.chunk_id.0)
+            .collect();
+        assert_eq!(again, baseline);
+    }
+}
--- a/crates/kebab-chunk/tests/code_python_ast_snapshot.rs
+++ b/crates/kebab-chunk/tests/code_python_ast_snapshot.rs
@@ -0,0 +1,221 @@
+//! Snapshot test pinning the `Vec<Chunk>` JSON for a
+//! representative Python code `CanonicalDocument`.
+//!
+//! This is an integration test. `kebab-parse-code` is intentionally NOT
+//! a dev-dep (design §6.3 / §8 boundary: AST extraction is parser-side).
+//! The `CanonicalDocument` is built inline from hand-crafted `Block::Code`
+//! units, which is the same pattern used in `code_rust_ast_v1.rs`'s
+//! internal `code_doc` test helper.
+//!
+//! Set `UPDATE_SNAPSHOTS=1` to re-bake the baseline.
+
+use std::path::PathBuf;
+
+use kebab_chunk::CodePythonAstV1Chunker;
+use kebab_core::{
+    AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+    Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath,
+    id_for_block, id_for_doc,
+};
+use serde_json::Value;
+use time::OffsetDateTime;
+
+fn fixtures_dir() -> PathBuf {
+    PathBuf::from(env!("CARGO_MANIFEST_DIR"))
+        .join("tests")
+        .join("fixtures")
+}
+
+fn fixed_doc() -> CanonicalDocument {
+    let wp = WorkspacePath("kebab_eval/metrics.py".into());
+    let aid = AssetId("b".repeat(64));
+    // Pin parser_version so doc_id / block_ids are reproducible.
+    let pv = ParserVersion("code-python-v1".into());
+    let doc_id = id_for_doc(&wp, &aid, &pv);
+
+    // Build a >200-line function body to force split_oversize.
+    let big_body: String = {
+        let header = "def big_compute(data):\n";
+        let body: String = (0..210u32)
+            .map(|i| format!("    v{i} = data[{i}] if {i} < len(data) else 0\n"))
+            .collect();
+        let footer = "    return sum(data)";
+        format!("{header}{body}{footer}")
+    };
+    let big_line_count = big_body.lines().count() as u32;
+    let big_line_end = 48 + big_line_count - 1;
+
+    // Representative units:
+    //  0. import block               (lines 1–5,   ≤200)
+    //  1. free fn `compute_mrr`      (lines 7–12,  ≤200)
+    //  2. class `MetricsCollector`   (lines 14–20, ≤200)
+    //  3. class `BaseEvaluator`      (lines 22–30, ≤200)
+    //  4. method `run`               (lines 32–38, ≤200)
+    //  5. method `report`            (lines 40–46, ≤200)
+    //  6. big_compute (>200 lines)   to force split_oversize
+    let raw_units: Vec<(&str, u32, u32, String)> = vec![
+        (
+            "imports",
+            1,
+            5,
+            "import os\nimport sys\nfrom typing import List\nfrom pathlib import Path\nfrom collections import defaultdict".to_string(),
+        ),
+        (
+            "compute_mrr",
+            7,
+            12,
+            "def compute_mrr(scores):\n    if not scores:\n        return 0.0\n    return sum(\n        1.0 / r for r in scores\n    ) / len(scores)".to_string(),
+        ),
+        (
+            "MetricsCollector",
+            14,
+            20,
+            "class MetricsCollector:\n    def __init__(self):\n        self.scores = []\n        self.labels = []\n        self.counts = defaultdict(int)\n        self.totals = defaultdict(float)\n        self.tags = []".to_string(),
+        ),
+        (
+            "BaseEvaluator",
+            22,
+            30,
+            "class BaseEvaluator:\n    def evaluate(self, data):\n        raise NotImplementedError\n    def batch_evaluate(self, items):\n        results = []\n        for item in items:\n            results.append(self.evaluate(item))\n        return results\n    def name(self):\n        return type(self).__name__".to_string(),
+        ),
+        (
+            "MetricsCollector.run",
+            32,
+            38,
+            "class MetricsCollector:\n    def run(self, inputs):\n        for inp in inputs:\n            score = self._score(inp)\n            self.scores.append(\n                score\n            )".to_string(),
+        ),
+        (
+            "MetricsCollector.report",
+            40,
+            46,
+            "class MetricsCollector:\n    def report(self):\n        return {\n            'mean': sum(self.scores) / max(len(self.scores), 1),\n            'count': len(self.scores),\n            'tags': self.tags,\n        }".to_string(),
+        ),
+        ("big_compute", 48, big_line_end, big_body),
+    ];
+
+    let blocks: Vec<Block> = raw_units
+        .iter()
+        .enumerate()
+        .map(|(i, (sym, ls, le, code))| {
+            let span = SourceSpan::Code {
+                line_start: *ls,
+                line_end: *le,
+                symbol: Some((*sym).to_string()),
+                lang: Some("python".into()),
+            };
+            let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+            Block::Code(CodeBlock {
+                common: CommonBlock {
+                    block_id: bid,
+                    heading_path: vec![],
+                    source_span: span,
+                },
+                lang: Some("python".into()),
+                code: code.clone(),
+            })
+        })
+        .collect();
+
+    CanonicalDocument {
+        doc_id,
+        source_asset_id: aid,
+        workspace_path: wp,
+        title: "metrics.py".into(),
+        lang: Lang("und".into()),
+        blocks,
+        metadata: Metadata {
+            aliases: vec![],
+            tags: vec![],
+            created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            source_type: SourceType::Note,
+            trust_level: TrustLevel::Primary,
+            user_id_alias: None,
+            user: Default::default(),
+            repo: Some("kebab".into()),
+            git_branch: Some("main".into()),
+            git_commit: Some("0".repeat(40)),
+            code_lang: Some("python".into()),
+        },
+        provenance: Provenance { events: vec![] },
+        parser_version: pv,
+        schema_version: 1,
+        doc_version: 1,
+        last_chunker_version: None,
+        last_embedding_version: None,
+    }
+}
+
+fn fixed_policy() -> ChunkPolicy {
+    ChunkPolicy {
+        target_tokens: 500,
+        overlap_tokens: 80,
+        respect_markdown_headings: false,
+        chunker_version: ChunkerVersion("code-python-ast-v1".into()),
+    }
+}
+
+#[test]
+fn code_python_ast_chunks_snapshot() {
+    let doc = fixed_doc();
+    let policy = fixed_policy();
+
+    let chunks = CodePythonAstV1Chunker.chunk(&doc, &policy).expect("chunk");
+    let actual = serde_json::to_value(&chunks).unwrap();
+
+    let dir = fixtures_dir();
+    let baseline_path = dir.join("code-sample.py.chunks.snapshot.json");
+    let baseline_text = match std::fs::read_to_string(&baseline_path) {
+        Ok(s) => s,
+        Err(_) if std::env::var("UPDATE_SNAPSHOTS").is_ok() => {
+            std::fs::create_dir_all(&dir).unwrap();
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            return;
+        }
+        Err(e) => panic!(
+            "missing baseline {}; run with UPDATE_SNAPSHOTS=1 to create: {e}",
+            baseline_path.display()
+        ),
+    };
+    let expected: Value = serde_json::from_str(&baseline_text).expect("baseline parses as json");
+
+    if actual != expected {
+        if std::env::var("UPDATE_SNAPSHOTS").is_ok() {
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            eprintln!("updated baseline {}", baseline_path.display());
+            return;
+        }
+        let pretty = serde_json::to_string_pretty(&actual).unwrap();
+        panic!(
+            "code-python-ast-v1 chunks snapshot drift\n\
+             --- expected ({}) ---\n{baseline_text}\n\
+             --- actual ---\n{pretty}\n\
+             If intentional, re-run with UPDATE_SNAPSHOTS=1.",
+            baseline_path.display()
+        );
+    }
+}
+
+/// Determinism cross-check: re-running the same pipeline yields the same
+/// chunk_ids byte-for-byte.
+#[test]
+fn code_python_ast_chunks_are_deterministic() {
+    let policy = fixed_policy();
+    let baseline: Vec<String> = CodePythonAstV1Chunker
+        .chunk(&fixed_doc(), &policy)
+        .unwrap()
+        .into_iter()
+        .map(|c| c.chunk_id.0)
+        .collect();
+    for _ in 0..5 {
+        let again: Vec<String> = CodePythonAstV1Chunker
+            .chunk(&fixed_doc(), &policy)
+            .unwrap()
+            .into_iter()
+            .map(|c| c.chunk_id.0)
+            .collect();
+        assert_eq!(again, baseline);
+    }
+}
--- a/crates/kebab-chunk/tests/code_rust_ast_snapshot.rs
+++ b/crates/kebab-chunk/tests/code_rust_ast_snapshot.rs
@@ -0,0 +1,221 @@
+//! Snapshot test pinning the `Vec<Chunk>` JSON for a
+//! representative Rust code `CanonicalDocument`.
+//!
+//! This is an integration test. `kebab-parse-code` is intentionally NOT
+//! a dev-dep (design §6.3 / §8 boundary: AST extraction is parser-side).
+//! The `CanonicalDocument` is built inline from hand-crafted `Block::Code`
+//! units, which is the same pattern used in `code_rust_ast_v1.rs`'s
+//! internal `code_doc` test helper.
+//!
+//! Set `UPDATE_SNAPSHOTS=1` to re-bake the baseline.
+
+use std::path::PathBuf;
+
+use kebab_chunk::CodeRustAstV1Chunker;
+use kebab_core::{
+    AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+    Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath,
+    id_for_block, id_for_doc,
+};
+use serde_json::Value;
+use time::OffsetDateTime;
+
+fn fixtures_dir() -> PathBuf {
+    PathBuf::from(env!("CARGO_MANIFEST_DIR"))
+        .join("tests")
+        .join("fixtures")
+}
+
+fn fixed_doc() -> CanonicalDocument {
+    let wp = WorkspacePath("crates/kebab-chunk/src/code_rust_ast_v1.rs".into());
+    let aid = AssetId("b".repeat(64));
+    // Pin parser_version so doc_id / block_ids are reproducible.
+    let pv = ParserVersion("code-rust-v1".into());
+    let doc_id = id_for_doc(&wp, &aid, &pv);
+
+    // Build a >200-line function body to force split_oversize.
+    let big_body: String = {
+        let header = "pub fn big_fn(input: &[u8]) -> Vec<u8> {\n";
+        let body: String = (0..210u32)
+            .map(|i| format!("    let v{i} = input.get({i} as usize).copied().unwrap_or(0);\n"))
+            .collect();
+        let footer = "    vec![0u8]\n}";
+        format!("{header}{body}{footer}")
+    };
+    let big_line_count = big_body.lines().count() as u32;
+    let big_line_end = 48 + big_line_count - 1;
+
+    // Representative units:
+    //  0. top-level use+const block  (lines 1–5,   ≤200)
+    //  1. free fn `parse`            (lines 7–12,  ≤200)
+    //  2. struct `Foo`               (lines 14–20, ≤200)
+    //  3. trait `Frobable`           (lines 22–30, ≤200)
+    //  4. impl Foo::double           (lines 32–38, ≤200)
+    //  5. impl Foo::triple           (lines 40–46, ≤200)
+    //  6. big_fn (>200 lines)        to force split_oversize
+    let raw_units: Vec<(&str, u32, u32, String)> = vec![
+        (
+            "use+const",
+            1,
+            5,
+            "use std::collections::HashMap;\nuse std::fmt;\n\nconst MAX: usize = 1024;\nconst MIN: usize = 0;".to_string(),
+        ),
+        (
+            "parse",
+            7,
+            12,
+            "pub fn parse(input: &str) -> Option<u32> {\n    input\n        .trim()\n        .parse()\n        .ok()\n}".to_string(),
+        ),
+        (
+            "Foo",
+            14,
+            20,
+            "pub struct Foo {\n    pub name: String,\n    pub value: u32,\n    pub tags: Vec<String>,\n    pub meta: Option<String>,\n    pub count: usize,\n}".to_string(),
+        ),
+        (
+            "Frobable",
+            22,
+            30,
+            "pub trait Frobable {\n    fn frob(&self) -> String;\n    fn frob_twice(&self) -> String {\n        let a = self.frob();\n        let b = self.frob();\n        format!(\"{a}{b}\")\n    }\n    fn name(&self) -> &str;\n}".to_string(),
+        ),
+        (
+            "Foo::double",
+            32,
+            38,
+            "impl Foo {\n    pub fn double(&self) -> u32 {\n        self.value\n            .checked_mul(2)\n            .unwrap_or(u32::MAX)\n    }\n}".to_string(),
+        ),
+        (
+            "Foo::triple",
+            40,
+            46,
+            "impl Foo {\n    pub fn triple(&self) -> u32 {\n        self.value\n            .checked_mul(3)\n            .unwrap_or(u32::MAX)\n    }\n}".to_string(),
+        ),
+        ("big_fn", 48, big_line_end, big_body),
+    ];
+
+    let blocks: Vec<Block> = raw_units
+        .iter()
+        .enumerate()
+        .map(|(i, (sym, ls, le, code))| {
+            let span = SourceSpan::Code {
+                line_start: *ls,
+                line_end: *le,
+                symbol: Some((*sym).to_string()),
+                lang: Some("rust".into()),
+            };
+            let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+            Block::Code(CodeBlock {
+                common: CommonBlock {
+                    block_id: bid,
+                    heading_path: vec![],
+                    source_span: span,
+                },
+                lang: Some("rust".into()),
+                code: code.clone(),
+            })
+        })
+        .collect();
+
+    CanonicalDocument {
+        doc_id,
+        source_asset_id: aid,
+        workspace_path: wp,
+        title: "code_rust_ast_v1.rs".into(),
+        lang: Lang("und".into()),
+        blocks,
+        metadata: Metadata {
+            aliases: vec![],
+            tags: vec![],
+            created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            source_type: SourceType::Note,
+            trust_level: TrustLevel::Primary,
+            user_id_alias: None,
+            user: Default::default(),
+            repo: Some("kebab".into()),
+            git_branch: Some("main".into()),
+            git_commit: Some("0".repeat(40)),
+            code_lang: Some("rust".into()),
+        },
+        provenance: Provenance { events: vec![] },
+        parser_version: pv,
+        schema_version: 1,
+        doc_version: 1,
+        last_chunker_version: None,
+        last_embedding_version: None,
+    }
+}
+
+fn fixed_policy() -> ChunkPolicy {
+    ChunkPolicy {
+        target_tokens: 500,
+        overlap_tokens: 80,
+        respect_markdown_headings: false,
+        chunker_version: ChunkerVersion("code-rust-ast-v1".into()),
+    }
+}
+
+#[test]
+fn code_rust_ast_chunks_snapshot() {
+    let doc = fixed_doc();
+    let policy = fixed_policy();
+
+    let chunks = CodeRustAstV1Chunker.chunk(&doc, &policy).expect("chunk");
+    let actual = serde_json::to_value(&chunks).unwrap();
+
+    let dir = fixtures_dir();
+    let baseline_path = dir.join("code-sample.chunks.snapshot.json");
+    let baseline_text = match std::fs::read_to_string(&baseline_path) {
+        Ok(s) => s,
+        Err(_) if std::env::var("UPDATE_SNAPSHOTS").is_ok() => {
+            std::fs::create_dir_all(&dir).unwrap();
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            return;
+        }
+        Err(e) => panic!(
+            "missing baseline {}; run with UPDATE_SNAPSHOTS=1 to create: {e}",
+            baseline_path.display()
+        ),
+    };
+    let expected: Value = serde_json::from_str(&baseline_text).expect("baseline parses as json");
+
+    if actual != expected {
+        if std::env::var("UPDATE_SNAPSHOTS").is_ok() {
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            eprintln!("updated baseline {}", baseline_path.display());
+            return;
+        }
+        let pretty = serde_json::to_string_pretty(&actual).unwrap();
+        panic!(
+            "code-rust-ast-v1 chunks snapshot drift\n\
+             --- expected ({}) ---\n{baseline_text}\n\
+             --- actual ---\n{pretty}\n\
+             If intentional, re-run with UPDATE_SNAPSHOTS=1.",
+            baseline_path.display()
+        );
+    }
+}
+
+/// Determinism cross-check: re-running the same pipeline yields the same
+/// chunk_ids byte-for-byte.
+#[test]
+fn code_rust_ast_chunks_are_deterministic() {
+    let policy = fixed_policy();
+    let baseline: Vec<String> = CodeRustAstV1Chunker
+        .chunk(&fixed_doc(), &policy)
+        .unwrap()
+        .into_iter()
+        .map(|c| c.chunk_id.0)
+        .collect();
+    for _ in 0..5 {
+        let again: Vec<String> = CodeRustAstV1Chunker
+            .chunk(&fixed_doc(), &policy)
+            .unwrap()
+            .into_iter()
+            .map(|c| c.chunk_id.0)
+            .collect();
+        assert_eq!(again, baseline);
+    }
+}
--- a/crates/kebab-chunk/tests/code_text_paragraph_v1.rs
+++ b/crates/kebab-chunk/tests/code_text_paragraph_v1.rs
@@ -0,0 +1,270 @@
+//! Behavioural tests for `CodeTextParagraphV1Chunker`.
+//!
+//! Documents are constructed manually (no kebab-parse-code dependency) by
+//! placing raw text into a single `Block::Code`, mirroring the pattern used
+//! in `k8s_manifest_resource_v1.rs`.
+
+use std::path::PathBuf;
+
+use kebab_chunk::CodeTextParagraphV1Chunker;
+use kebab_core::{
+    AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
+    CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
+    WorkspacePath, id_for_block, id_for_doc,
+};
+use time::OffsetDateTime;
+
+// ── helpers ──────────────────────────────────────────────────────────────────
+
+fn fixtures_dir() -> PathBuf {
+    PathBuf::from(env!("CARGO_MANIFEST_DIR"))
+        .join("tests")
+        .join("fixtures")
+}
+
+/// Build a `CanonicalDocument` with a single `Block::Code` containing `text`
+/// and the supplied `lang` label.
+fn text_doc(lang: &str, text: &str) -> CanonicalDocument {
+    let wp = WorkspacePath("scripts/sample.sh".into());
+    let aid = AssetId("d".repeat(64));
+    let pv = ParserVersion("code-text-paragraph-v1".into());
+    let doc_id = id_for_doc(&wp, &aid, &pv);
+
+    let line_count = text.lines().count() as u32;
+    let span = SourceSpan::Code {
+        line_start: 1,
+        line_end: line_count.max(1),
+        symbol: None,
+        lang: Some(lang.into()),
+    };
+    let bid = id_for_block(&doc_id, "code", &[], 0, &span);
+    let block = Block::Code(CodeBlock {
+        common: CommonBlock {
+            block_id: bid,
+            heading_path: vec![],
+            source_span: span,
+        },
+        lang: Some(lang.into()),
+        code: text.to_string(),
+    });
+
+    CanonicalDocument {
+        doc_id,
+        source_asset_id: aid,
+        workspace_path: wp,
+        title: "sample.sh".into(),
+        lang: Lang("und".into()),
+        blocks: vec![block],
+        metadata: Metadata {
+            aliases: vec![],
+            tags: vec![],
+            created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            source_type: SourceType::Note,
+            trust_level: TrustLevel::Primary,
+            user_id_alias: None,
+            user: Default::default(),
+            repo: Some("kebab".into()),
+            git_branch: Some("main".into()),
+            git_commit: Some("0".repeat(40)),
+            code_lang: Some(lang.into()),
+        },
+        provenance: Provenance { events: vec![] },
+        parser_version: pv,
+        schema_version: 1,
+        doc_version: 1,
+        last_chunker_version: None,
+        last_embedding_version: None,
+    }
+}
+
+fn policy() -> ChunkPolicy {
+    ChunkPolicy {
+        target_tokens: 500,
+        overlap_tokens: 80,
+        respect_markdown_headings: false,
+        chunker_version: ChunkerVersion("code-text-paragraph-v1".into()),
+    }
+}
+
+// ── tests ─────────────────────────────────────────────────────────────────────
+
+/// `sample_shell.sh` has 4 paragraphs separated by 3 blank lines:
+///   - paragraph 1: lines 1-2  (shebang + set -euo pipefail)
+///   - paragraph 2: lines 4-7  (env setup block)
+///   - paragraph 3: lines 9-11 (ingest block)
+///   - paragraph 4: lines 13-15 (report block)
+///
+/// We assert:
+///   - exactly 4 chunks (one per paragraph)
+///   - all symbols are None (Tier 3 spec §9.3)
+///   - all langs are "shell"
+///   - line ranges are strictly ascending and do NOT include the blank lines
+///     (lines 3, 8, 12 must not appear in any range)
+#[test]
+fn shell_multi_paragraph_splits_on_blank_lines() {
+    let fixture_path = fixtures_dir().join("sample_shell.sh");
+    let text = std::fs::read_to_string(&fixture_path)
+        .unwrap_or_else(|e| panic!("cannot read fixture {}: {e}", fixture_path.display()));
+
+    let doc = text_doc("shell", &text);
+    let chunks = CodeTextParagraphV1Chunker
+        .chunk(&doc, &policy())
+        .expect("chunk");
+
+    assert_eq!(
+        chunks.len(),
+        4,
+        "expected 4 chunks (one per paragraph), got {}: {chunks:#?}",
+        chunks.len()
+    );
+
+    // All symbols must be None (Tier 3 requirement).
+    for (i, chunk) in chunks.iter().enumerate() {
+        match &chunk.source_spans[0] {
+            SourceSpan::Code { symbol, .. } => {
+                assert!(
+                    symbol.is_none(),
+                    "chunk[{i}] symbol must be None for Tier 3 chunker, got {symbol:?}"
+                );
+            }
+            other => panic!("chunk[{i}]: expected Code span, got {other:?}"),
+        }
+    }
+
+    // All langs must be "shell".
+    for (i, chunk) in chunks.iter().enumerate() {
+        match &chunk.source_spans[0] {
+            SourceSpan::Code { lang, .. } => {
+                assert_eq!(
+                    lang.as_deref(),
+                    Some("shell"),
+                    "chunk[{i}] lang must be 'shell', got {lang:?}"
+                );
+            }
+            other => panic!("chunk[{i}]: expected Code span, got {other:?}"),
+        }
+    }
+
+    // Line ranges must be strictly ascending with no overlap,
+    // and blank lines (3, 8, 12) must not be included in any range.
+    let expected_ranges: &[(u32, u32)] = &[(1, 2), (4, 7), (9, 11), (13, 15)];
+    let actual_ranges: Vec<(u32, u32)> = chunks
+        .iter()
+        .map(|c| match &c.source_spans[0] {
+            SourceSpan::Code {
+                line_start,
+                line_end,
+                ..
+            } => (*line_start, *line_end),
+            other => panic!("expected Code span, got {other:?}"),
+        })
+        .collect();
+
+    assert_eq!(
+        actual_ranges, expected_ranges,
+        "line ranges mismatch: got {actual_ranges:?}, expected {expected_ranges:?}"
+    );
+}
+
+/// `sample_long_paragraph.txt` has exactly 200 non-blank lines and no blank
+/// lines, so the entire file is one paragraph.  200 > 80 (FALLBACK_LINES_PER_CHUNK),
+/// so the oversize window split fires with stride 60:
+///   - window 1: lines 1-80
+///   - window 2: lines 61-140
+///   - window 3: lines 121-200
+///
+/// All chunk_ids must be distinct (the #L{window_start} split_key suffix).
+#[test]
+fn single_long_paragraph_line_window_split() {
+    let fixture_path = fixtures_dir().join("sample_long_paragraph.txt");
+    let text = std::fs::read_to_string(&fixture_path)
+        .unwrap_or_else(|e| panic!("cannot read fixture {}: {e}", fixture_path.display()));
+
+    assert_eq!(
+        text.lines().count(),
+        200,
+        "fixture must have exactly 200 lines"
+    );
+
+    let doc = text_doc("shell", &text);
+    let chunks = CodeTextParagraphV1Chunker
+        .chunk(&doc, &policy())
+        .expect("chunk");
+
+    assert_eq!(
+        chunks.len(),
+        3,
+        "expected 3 window chunks for 200-line paragraph, got {}: {chunks:#?}",
+        chunks.len()
+    );
+
+    let expected_ranges: &[(u32, u32)] = &[(1, 80), (61, 140), (121, 200)];
+    let actual_ranges: Vec<(u32, u32)> = chunks
+        .iter()
+        .map(|c| match &c.source_spans[0] {
+            SourceSpan::Code {
+                line_start,
+                line_end,
+                ..
+            } => (*line_start, *line_end),
+            other => panic!("expected Code span, got {other:?}"),
+        })
+        .collect();
+
+    assert_eq!(
+        actual_ranges, expected_ranges,
+        "window ranges mismatch: got {actual_ranges:?}, expected {expected_ranges:?}"
+    );
+
+    // All chunk_ids must be distinct (#L{window_start} suffix differentiates them).
+    let ids: std::collections::HashSet<_> = chunks.iter().map(|c| c.chunk_id.clone()).collect();
+    assert_eq!(
+        ids.len(),
+        chunks.len(),
+        "oversize window chunks must have distinct chunk_ids"
+    );
+}
+
+/// An empty source file (no non-blank lines) must yield zero chunks.
+#[test]
+fn empty_file_emits_zero_chunks() {
+    let doc = text_doc("shell", "");
+    let chunks = CodeTextParagraphV1Chunker
+        .chunk(&doc, &policy())
+        .expect("chunk");
+
+    assert_eq!(
+        chunks.len(),
+        0,
+        "empty file must yield 0 chunks, got {}: {chunks:#?}",
+        chunks.len()
+    );
+}
+
+/// The `lang` field on each emitted chunk must match the `lang` passed to
+/// `text_doc`, regardless of content.  `symbol` must be `None` (Tier 3 spec).
+#[test]
+fn lang_field_preserved_from_input_doc() {
+    let doc = text_doc("yaml", "key1: value1\nkey2: value2\n");
+    let chunks = CodeTextParagraphV1Chunker
+        .chunk(&doc, &policy())
+        .expect("chunk");
+
+    assert!(!chunks.is_empty(), "expected at least one chunk");
+
+    match &chunks[0].source_spans[0] {
+        SourceSpan::Code { lang, symbol, .. } => {
+            assert_eq!(
+                lang.as_deref(),
+                Some("yaml"),
+                "lang must be 'yaml', got {lang:?}"
+            );
+            assert!(
+                symbol.is_none(),
+                "symbol must be None for Tier 3 chunker, got {symbol:?}"
+            );
+        }
+        other => panic!("expected Code span, got {other:?}"),
+    }
+}
--- a/crates/kebab-chunk/tests/code_ts_ast_snapshot.rs
+++ b/crates/kebab-chunk/tests/code_ts_ast_snapshot.rs
@@ -0,0 +1,221 @@
+//! Snapshot test pinning the `Vec<Chunk>` JSON for a
+//! representative TypeScript code `CanonicalDocument`.
+//!
+//! This is an integration test. `kebab-parse-code` is intentionally NOT
+//! a dev-dep (design §6.3 / §8 boundary: AST extraction is parser-side).
+//! The `CanonicalDocument` is built inline from hand-crafted `Block::Code`
+//! units, which is the same pattern used in `code_rust_ast_v1.rs`'s
+//! internal `code_doc` test helper.
+//!
+//! Set `UPDATE_SNAPSHOTS=1` to re-bake the baseline.
+
+use std::path::PathBuf;
+
+use kebab_chunk::CodeTsAstV1Chunker;
+use kebab_core::{
+    AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+    Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath,
+    id_for_block, id_for_doc,
+};
+use serde_json::Value;
+use time::OffsetDateTime;
+
+fn fixtures_dir() -> PathBuf {
+    PathBuf::from(env!("CARGO_MANIFEST_DIR"))
+        .join("tests")
+        .join("fixtures")
+}
+
+fn fixed_doc() -> CanonicalDocument {
+    let wp = WorkspacePath("src/Foo.ts".into());
+    let aid = AssetId("b".repeat(64));
+    // Pin parser_version so doc_id / block_ids are reproducible.
+    let pv = ParserVersion("code-ts-v1".into());
+    let doc_id = id_for_doc(&wp, &aid, &pv);
+
+    // Build a >200-line method body to force split_oversize.
+    let big_body: String = {
+        let header = "export class BigProcessor {\n  process(items: string[]): string[] {\n";
+        let body: String = (0..210u32)
+            .map(|i| format!("    const v{i} = items[{i}] ?? '';\n"))
+            .collect();
+        let footer = "    return items;\n  }\n}";
+        format!("{header}{body}{footer}")
+    };
+    let big_line_count = big_body.lines().count() as u32;
+    let big_line_end = 48 + big_line_count - 1;
+
+    // Representative units:
+    //  0. import block               (lines 1–5,   ≤200)
+    //  1. free fn `parseInput`       (lines 7–12,  ≤200)
+    //  2. interface `Frobable`       (lines 14–20, ≤200)
+    //  3. class `Foo`                (lines 22–30, ≤200)
+    //  4. method `Foo.double`        (lines 32–38, ≤200)
+    //  5. method `Foo.triple`        (lines 40–46, ≤200)
+    //  6. BigProcessor (>200 lines)  to force split_oversize
+    let raw_units: Vec<(&str, u32, u32, String)> = vec![
+        (
+            "imports",
+            1,
+            5,
+            "import { readFileSync } from 'fs';\nimport { join } from 'path';\nimport type { Config } from './config';\nimport { Logger } from './logger';\nimport { EventEmitter } from 'events';".to_string(),
+        ),
+        (
+            "parseInput",
+            7,
+            12,
+            "export function parseInput(raw: string): number | null {\n  const trimmed = raw.trim();\n  const n = Number(trimmed);\n  if (isNaN(n)) return null;\n  return n;\n}".to_string(),
+        ),
+        (
+            "Frobable",
+            14,
+            20,
+            "export interface Frobable {\n  frob(): string;\n  frobTwice(): string;\n  readonly name: string;\n  readonly tags: string[];\n  count: number;\n  reset(): void;\n}".to_string(),
+        ),
+        (
+            "Foo",
+            22,
+            30,
+            "export class Foo implements Frobable {\n  constructor(\n    public readonly name: string,\n    public value: number,\n    public tags: string[] = [],\n  ) {}\n  frob(): string { return this.name; }\n  frobTwice(): string { return this.name.repeat(2); }\n  reset(): void { this.value = 0; }\n}".to_string(),
+        ),
+        (
+            "Foo.double",
+            32,
+            38,
+            "export class Foo {\n  double(): number {\n    const result = this.value * 2;\n    if (result > Number.MAX_SAFE_INTEGER) {\n      return Number.MAX_SAFE_INTEGER;\n    }\n    return result;\n  }\n}".to_string(),
+        ),
+        (
+            "Foo.triple",
+            40,
+            46,
+            "export class Foo {\n  triple(): number {\n    const result = this.value * 3;\n    if (result > Number.MAX_SAFE_INTEGER) {\n      return Number.MAX_SAFE_INTEGER;\n    }\n    return result;\n  }\n}".to_string(),
+        ),
+        ("BigProcessor", 48, big_line_end, big_body),
+    ];
+
+    let blocks: Vec<Block> = raw_units
+        .iter()
+        .enumerate()
+        .map(|(i, (sym, ls, le, code))| {
+            let span = SourceSpan::Code {
+                line_start: *ls,
+                line_end: *le,
+                symbol: Some((*sym).to_string()),
+                lang: Some("typescript".into()),
+            };
+            let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+            Block::Code(CodeBlock {
+                common: CommonBlock {
+                    block_id: bid,
+                    heading_path: vec![],
+                    source_span: span,
+                },
+                lang: Some("typescript".into()),
+                code: code.clone(),
+            })
+        })
+        .collect();
+
+    CanonicalDocument {
+        doc_id,
+        source_asset_id: aid,
+        workspace_path: wp,
+        title: "Foo.ts".into(),
+        lang: Lang("und".into()),
+        blocks,
+        metadata: Metadata {
+            aliases: vec![],
+            tags: vec![],
+            created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            source_type: SourceType::Note,
+            trust_level: TrustLevel::Primary,
+            user_id_alias: None,
+            user: Default::default(),
+            repo: Some("kebab".into()),
+            git_branch: Some("main".into()),
+            git_commit: Some("0".repeat(40)),
+            code_lang: Some("typescript".into()),
+        },
+        provenance: Provenance { events: vec![] },
+        parser_version: pv,
+        schema_version: 1,
+        doc_version: 1,
+        last_chunker_version: None,
+        last_embedding_version: None,
+    }
+}
+
+fn fixed_policy() -> ChunkPolicy {
+    ChunkPolicy {
+        target_tokens: 500,
+        overlap_tokens: 80,
+        respect_markdown_headings: false,
+        chunker_version: ChunkerVersion("code-ts-ast-v1".into()),
+    }
+}
+
+#[test]
+fn code_ts_ast_chunks_snapshot() {
+    let doc = fixed_doc();
+    let policy = fixed_policy();
+
+    let chunks = CodeTsAstV1Chunker.chunk(&doc, &policy).expect("chunk");
+    let actual = serde_json::to_value(&chunks).unwrap();
+
+    let dir = fixtures_dir();
+    let baseline_path = dir.join("code-sample.ts.chunks.snapshot.json");
+    let baseline_text = match std::fs::read_to_string(&baseline_path) {
+        Ok(s) => s,
+        Err(_) if std::env::var("UPDATE_SNAPSHOTS").is_ok() => {
+            std::fs::create_dir_all(&dir).unwrap();
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            return;
+        }
+        Err(e) => panic!(
+            "missing baseline {}; run with UPDATE_SNAPSHOTS=1 to create: {e}",
+            baseline_path.display()
+        ),
+    };
+    let expected: Value = serde_json::from_str(&baseline_text).expect("baseline parses as json");
+
+    if actual != expected {
+        if std::env::var("UPDATE_SNAPSHOTS").is_ok() {
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            eprintln!("updated baseline {}", baseline_path.display());
+            return;
+        }
+        let pretty = serde_json::to_string_pretty(&actual).unwrap();
+        panic!(
+            "code-ts-ast-v1 chunks snapshot drift\n\
+             --- expected ({}) ---\n{baseline_text}\n\
+             --- actual ---\n{pretty}\n\
+             If intentional, re-run with UPDATE_SNAPSHOTS=1.",
+            baseline_path.display()
+        );
+    }
+}
+
+/// Determinism cross-check: re-running the same pipeline yields the same
+/// chunk_ids byte-for-byte.
+#[test]
+fn code_ts_ast_chunks_are_deterministic() {
+    let policy = fixed_policy();
+    let baseline: Vec<String> = CodeTsAstV1Chunker
+        .chunk(&fixed_doc(), &policy)
+        .unwrap()
+        .into_iter()
+        .map(|c| c.chunk_id.0)
+        .collect();
+    for _ in 0..5 {
+        let again: Vec<String> = CodeTsAstV1Chunker
+            .chunk(&fixed_doc(), &policy)
+            .unwrap()
+            .into_iter()
+            .map(|c| c.chunk_id.0)
+            .collect();
+        assert_eq!(again, baseline);
+    }
+}
--- a/crates/kebab-chunk/tests/dockerfile_file_v1.rs
+++ b/crates/kebab-chunk/tests/dockerfile_file_v1.rs
@@ -0,0 +1,134 @@
+//! Behavioural tests for `DockerfileFileV1Chunker`.
+//!
+//! Documents are constructed manually (no kebab-parse-code dependency) by
+//! placing the raw Dockerfile text into a single `Block::Code`, mirroring the
+//! pattern used in `k8s_manifest_resource_v1.rs`.
+
+use std::path::PathBuf;
+
+use kebab_chunk::DockerfileFileV1Chunker;
+use kebab_core::{
+    AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
+    CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
+    WorkspacePath, id_for_block, id_for_doc,
+};
+use time::OffsetDateTime;
+
+// ── helpers ──────────────────────────────────────────────────────────────────
+
+fn fixtures_dir() -> PathBuf {
+    PathBuf::from(env!("CARGO_MANIFEST_DIR"))
+        .join("tests")
+        .join("fixtures")
+}
+
+/// Build a `CanonicalDocument` with a single `Block::Code` containing `dockerfile_text`.
+fn dockerfile_doc(dockerfile_text: &str) -> CanonicalDocument {
+    let wp = WorkspacePath("build/Dockerfile".into());
+    let aid = AssetId("d".repeat(64));
+    let pv = ParserVersion("code-dockerfile-v1".into());
+    let doc_id = id_for_doc(&wp, &aid, &pv);
+
+    let line_count = dockerfile_text.lines().count() as u32;
+    let span = SourceSpan::Code {
+        line_start: 1,
+        line_end: line_count.max(1),
+        symbol: None,
+        lang: Some("dockerfile".into()),
+    };
+    let bid = id_for_block(&doc_id, "code", &[], 0, &span);
+    let block = Block::Code(CodeBlock {
+        common: CommonBlock {
+            block_id: bid,
+            heading_path: vec![],
+            source_span: span,
+        },
+        lang: Some("dockerfile".into()),
+        code: dockerfile_text.to_string(),
+    });
+
+    CanonicalDocument {
+        doc_id,
+        source_asset_id: aid,
+        workspace_path: wp,
+        title: "Dockerfile".into(),
+        lang: Lang("und".into()),
+        blocks: vec![block],
+        metadata: Metadata {
+            aliases: vec![],
+            tags: vec![],
+            created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            source_type: SourceType::Note,
+            trust_level: TrustLevel::Primary,
+            user_id_alias: None,
+            user: Default::default(),
+            repo: Some("kebab".into()),
+            git_branch: Some("main".into()),
+            git_commit: Some("0".repeat(40)),
+            code_lang: Some("dockerfile".into()),
+        },
+        provenance: Provenance { events: vec![] },
+        parser_version: pv,
+        schema_version: 1,
+        doc_version: 1,
+        last_chunker_version: None,
+        last_embedding_version: None,
+    }
+}
+
+fn policy() -> ChunkPolicy {
+    ChunkPolicy {
+        target_tokens: 500,
+        overlap_tokens: 80,
+        respect_markdown_headings: false,
+        chunker_version: ChunkerVersion("dockerfile-file-v1".into()),
+    }
+}
+
+// ── tests ─────────────────────────────────────────────────────────────────────
+
+/// A simple 5-line Dockerfile fixture must emit exactly 1 chunk with the
+/// correct symbol, lang, and line range.
+#[test]
+fn dockerfile_emits_single_chunk() {
+    let fixture_path = fixtures_dir().join("sample.dockerfile");
+    let text = std::fs::read_to_string(&fixture_path)
+        .unwrap_or_else(|e| panic!("cannot read fixture {}: {e}", fixture_path.display()));
+
+    let doc = dockerfile_doc(&text);
+    let chunks = DockerfileFileV1Chunker
+        .chunk(&doc, &policy())
+        .expect("chunk");
+
+    assert_eq!(
+        chunks.len(),
+        1,
+        "expected 1 chunk, got {}: {chunks:#?}",
+        chunks.len()
+    );
+
+    // Inspect the Chunk's source_spans for symbol / lang / line range.
+    let span = chunks[0].source_spans.first().expect("at least one span");
+    match span {
+        SourceSpan::Code {
+            line_start,
+            line_end,
+            symbol,
+            lang,
+        } => {
+            assert_eq!(*line_start, 1, "line_start must be 1");
+            assert_eq!(*line_end, 5, "line_end must be 5 (5-line fixture)");
+            assert_eq!(
+                symbol.as_deref(),
+                Some("<dockerfile>"),
+                "symbol must be '<dockerfile>'"
+            );
+            assert_eq!(lang.as_deref(), Some("dockerfile"), "lang must be 'dockerfile'");
+        }
+        other => panic!("expected SourceSpan::Code, got {other:?}"),
+    }
+
+    // Verify chunker_version label.
+    assert_eq!(chunks[0].chunker_version.0, "dockerfile-file-v1");
+}
--- a/crates/kebab-chunk/tests/fixtures/code-sample.c.chunks.snapshot.json
+++ b/crates/kebab-chunk/tests/fixtures/code-sample.c.chunks.snapshot.json
@@ -0,0 +1,86 @@
+[
+  {
+    "block_ids": [
+      "8149e12ca002489acb4a0f74c97a061a"
+    ],
+    "chunk_id": "ec3cf06ae56c8e9796bbc9196438b7c5",
+    "chunker_version": "code-c-ast-v1",
+    "doc_id": "6bec42dd593920a060541db16c4e8e45",
+    "heading_path": [],
+    "policy_hash": "ecfad2ec1223662d",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "c",
+        "line_end": 18,
+        "line_start": 1,
+        "symbol": "<top-level>"
+      }
+    ],
+    "text": "#include <stdio.h>\n#include <stdlib.h>\n\n#define MAX_BUF 4096\n\ntypedef enum {\n    OK = 0,\n    ERR_PARSE,\n    ERR_IO,\n} status_t;\n\ntypedef struct {\n    int id;\n    char name[64];\n    status_t status;\n} record_t;\n\nstatic int counter = 0;",
+    "token_estimate": 78
+  },
+  {
+    "block_ids": [
+      "1baaa89f21a47b2f32d6396a24a85454"
+    ],
+    "chunk_id": "c2d7a81c898106733ef2e703774a6a4a",
+    "chunker_version": "code-c-ast-v1",
+    "doc_id": "6bec42dd593920a060541db16c4e8e45",
+    "heading_path": [],
+    "policy_hash": "ecfad2ec1223662d",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "c",
+        "line_end": 23,
+        "line_start": 20,
+        "symbol": "parse_record"
+      }
+    ],
+    "text": "int parse_record(const char *line, record_t *out) {\n    if (line == NULL || out == NULL) return ERR_PARSE;\n    return OK;\n}",
+    "token_estimate": 41
+  },
+  {
+    "block_ids": [
+      "8d0e14cbcc6d1e92d7878ab796ea68b8"
+    ],
+    "chunk_id": "0e4d7b131ab64eba03b51903b5d8f96d",
+    "chunker_version": "code-c-ast-v1",
+    "doc_id": "6bec42dd593920a060541db16c4e8e45",
+    "heading_path": [],
+    "policy_hash": "ecfad2ec1223662d",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "c",
+        "line_end": 27,
+        "line_start": 25,
+        "symbol": "print_record"
+      }
+    ],
+    "text": "void print_record(const record_t *r) {\n    printf(\"[%d] %s (status=%d)\\n\", r->id, r->name, r->status);\n}",
+    "token_estimate": 35
+  },
+  {
+    "block_ids": [
+      "9c2ede84423871b615d48c38fefb1853"
+    ],
+    "chunk_id": "e076f8edb2ff141d7e99b4106bb95157",
+    "chunker_version": "code-c-ast-v1",
+    "doc_id": "6bec42dd593920a060541db16c4e8e45",
+    "heading_path": [],
+    "policy_hash": "ecfad2ec1223662d",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "c",
+        "line_end": 33,
+        "line_start": 29,
+        "symbol": "main"
+      }
+    ],
+    "text": "int main(void) {\n    record_t r = { .id = 1, .name = \"foo\", .status = OK };\n    print_record(&r);\n    return 0;\n}",
+    "token_estimate": 38
+  }
+]
--- a/crates/kebab-chunk/tests/fixtures/code-sample.chunks.snapshot.json
+++ b/crates/kebab-chunk/tests/fixtures/code-sample.chunks.snapshot.json
--- a/crates/kebab-chunk/tests/fixtures/code-sample.cpp.chunks.snapshot.json
+++ b/crates/kebab-chunk/tests/fixtures/code-sample.cpp.chunks.snapshot.json
@@ -0,0 +1,107 @@
+[
+  {
+    "block_ids": [
+      "53292605459065d170cd36c118e20546"
+    ],
+    "chunk_id": "50a5b324300d9082eac4ce2a422810e1",
+    "chunker_version": "code-cpp-ast-v1",
+    "doc_id": "fff1e1f0a7ff70ef682937470e5d1d28",
+    "heading_path": [],
+    "policy_hash": "71f3c07bb9ec1d09",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "cpp",
+        "line_end": 4,
+        "line_start": 1,
+        "symbol": "<top-level>"
+      }
+    ],
+    "text": "#include <string>\n#include <vector>\n\nnamespace kebab {",
+    "token_estimate": 18
+  },
+  {
+    "block_ids": [
+      "f349acad94c9fa4cf9ad1c0a93e83610"
+    ],
+    "chunk_id": "0e6bc7c522665af8a4b0f66afb9d29c8",
+    "chunker_version": "code-cpp-ast-v1",
+    "doc_id": "fff1e1f0a7ff70ef682937470e5d1d28",
+    "heading_path": [],
+    "policy_hash": "71f3c07bb9ec1d09",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "cpp",
+        "line_end": 20,
+        "line_start": 6,
+        "symbol": "kebab::chunk::MdHeadingV1Chunker"
+      }
+    ],
+    "text": "class MdHeadingV1Chunker {\npublic:\n    MdHeadingV1Chunker() = default;\n    ~MdHeadingV1Chunker() = default;\n\n    std::string chunk_doc(const std::string& doc) {\n        return doc;\n    }\n\n    int operator()(int x) const {\n        return x * 2;\n    }\n\nprivate:\n    int counter_ = 0;\n};",
+    "token_estimate": 95
+  },
+  {
+    "block_ids": [
+      "8b9811387717d0bd4abf84abcc35b8b1"
+    ],
+    "chunk_id": "d9326d252905b665b2adb9a416c20451",
+    "chunker_version": "code-cpp-ast-v1",
+    "doc_id": "fff1e1f0a7ff70ef682937470e5d1d28",
+    "heading_path": [],
+    "policy_hash": "71f3c07bb9ec1d09",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "cpp",
+        "line_end": 25,
+        "line_start": 22,
+        "symbol": "kebab::identity"
+      }
+    ],
+    "text": "template <typename T>\nT identity(T value) {\n    return value;\n}",
+    "token_estimate": 21
+  },
+  {
+    "block_ids": [
+      "1754cb6b971f6a4cb292f144a4f0570b"
+    ],
+    "chunk_id": "56ee5f991de4a413c016da8dc4acfc35",
+    "chunker_version": "code-cpp-ast-v1",
+    "doc_id": "fff1e1f0a7ff70ef682937470e5d1d28",
+    "heading_path": [],
+    "policy_hash": "71f3c07bb9ec1d09",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "cpp",
+        "line_end": 29,
+        "line_start": 27,
+        "symbol": "kebab::global_helper"
+      }
+    ],
+    "text": "void global_helper() {\n    // free function in kebab namespace\n}",
+    "token_estimate": 22
+  },
+  {
+    "block_ids": [
+      "14b5f3393d6d25f822f5b70763d24acd"
+    ],
+    "chunk_id": "c0d7c043cdd575c530db3909b54cc906",
+    "chunker_version": "code-cpp-ast-v1",
+    "doc_id": "fff1e1f0a7ff70ef682937470e5d1d28",
+    "heading_path": [],
+    "policy_hash": "71f3c07bb9ec1d09",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "cpp",
+        "line_end": 34,
+        "line_start": 31,
+        "symbol": "main"
+      }
+    ],
+    "text": "int main() {\n    kebab::chunk::MdHeadingV1Chunker c;\n    return 0;\n}",
+    "token_estimate": 23
+  }
+]
--- a/crates/kebab-chunk/tests/fixtures/code-sample.go.chunks.snapshot.json
+++ b/crates/kebab-chunk/tests/fixtures/code-sample.go.chunks.snapshot.json
@@ -0,0 +1,233 @@
+[
+  {
+    "block_ids": [
+      "c182bf37e32c7fc1b868bd617f8eaf66"
+    ],
+    "chunk_id": "43de518d946dc18ec040ae20d74e0cff",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 5,
+        "line_start": 1,
+        "symbol": "imports"
+      }
+    ],
+    "text": "import (\n\t\"fmt\"\n\t\"os\"\n\t\"strings\"\n)",
+    "token_estimate": 12
+  },
+  {
+    "block_ids": [
+      "c9992cdcfdf3c2a7700a4abc4782a8a4"
+    ],
+    "chunk_id": "af4c382a83f1e8cdea495d8b33c11abc",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 12,
+        "line_start": 7,
+        "symbol": "ComputeMRR"
+      }
+    ],
+    "text": "func ComputeMRR(scores []float64) float64 {\n\tif len(scores) == 0 {\n\t\treturn 0.0\n\t}\n\t_ = fmt.Sprintf(\"%v\", scores)\n\treturn 1.0 / float64(len(scores))\n}",
+    "token_estimate": 50
+  },
+  {
+    "block_ids": [
+      "5f18dc3e79fe946ba05d32c3bfc00684"
+    ],
+    "chunk_id": "4be6d8f180bc19b8651877e5264852ac",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 20,
+        "line_start": 14,
+        "symbol": "MetricsCollector"
+      }
+    ],
+    "text": "type MetricsCollector struct {\n\tScores []float64\n\tLabels []string\n\tCounts map[string]int\n\tTotals map[string]float64\n\tTags   []string\n}",
+    "token_estimate": 45
+  },
+  {
+    "block_ids": [
+      "3009cc022ca832c323393e4f9bcdb388"
+    ],
+    "chunk_id": "3ae182f4c6d304ee7f0aaf447142f948",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 30,
+        "line_start": 22,
+        "symbol": "BaseEvaluator"
+      }
+    ],
+    "text": "type BaseEvaluator struct {\n\tName string\n}\n\nfunc (e *BaseEvaluator) Evaluate(data []string) error {\n\t_ = os.Stderr\n\t_ = strings.Join(data, \",\")\n\treturn nil\n}",
+    "token_estimate": 53
+  },
+  {
+    "block_ids": [
+      "e0e83d1d7f9327a1902ae9a8f67c1f1c"
+    ],
+    "chunk_id": "b962f14980e756bb8ba514e2282756cd",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 38,
+        "line_start": 32,
+        "symbol": "MetricsCollector.Run"
+      }
+    ],
+    "text": "func (m *MetricsCollector) Run(inputs []float64) {\n\tfor _, inp := range inputs {\n\t\tm.Scores = append(\n\t\t\tm.Scores,\n\t\t\tinp,\n\t\t)\n\t}\n}",
+    "token_estimate": 44
+  },
+  {
+    "block_ids": [
+      "0e6a572bc3fe2bd6d173fe614bd1b763"
+    ],
+    "chunk_id": "441c695e990e7f49188068433e313e87",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 46,
+        "line_start": 40,
+        "symbol": "MetricsCollector.Report"
+      }
+    ],
+    "text": "func (m *MetricsCollector) Report() map[string]interface{} {\n\treturn map[string]interface{}{\n\t\t\"mean\":  0.0,\n\t\t\"count\": len(m.Scores),\n\t\t\"tags\":  m.Tags,\n\t}\n}",
+    "token_estimate": 53
+  },
+  {
+    "block_ids": [
+      "5d269745b2e5dbdcbef0c09ba54b0bd6"
+    ],
+    "chunk_id": "7a942d871c588ec69426290561f05179",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 247,
+        "line_start": 48,
+        "symbol": "BigCompute [part 1/5]"
+      }
+    ],
+    "text": "func BigCompute(data []int) int {\n\tv0 := 0\n\tif 0 < len(data) {\n\t\tv0 = data[0]\n\t}\n\tv1 := 0\n\tif 1 < len(data) {\n\t\tv1 = data[1]\n\t}\n\tv2 := 0\n\tif 2 < len(data) {\n\t\tv2 = data[2]\n\t}\n\tv3 := 0\n\tif 3 < len(data) {\n\t\tv3 = data[3]\n\t}\n\tv4 := 0\n\tif 4 < len(data) {\n\t\tv4 = data[4]\n\t}\n\tv5 := 0\n\tif 5 < len(data) {\n\t\tv5 = data[5]\n\t}\n\tv6 := 0\n\tif 6 < len(data) {\n\t\tv6 = data[6]\n\t}\n\tv7 := 0\n\tif 7 < len(data) {\n\t\tv7 = data[7]\n\t}\n\tv8 := 0\n\tif 8 < len(data) {\n\t\tv8 = data[8]\n\t}\n\tv9 := 0\n\tif 9 < len(data) {\n\t\tv9 = data[9]\n\t}\n\tv10 := 0\n\tif 10 < len(data) {\n\t\tv10 = data[10]\n\t}\n\tv11 := 0\n\tif 11 < len(data) {\n\t\tv11 = data[11]\n\t}\n\tv12 := 0\n\tif 12 < len(data) {\n\t\tv12 = data[12]\n\t}\n\tv13 := 0\n\tif 13 < len(data) {\n\t\tv13 = data[13]\n\t}\n\tv14 := 0\n\tif 14 < len(data) {\n\t\tv14 = data[14]\n\t}\n\tv15 := 0\n\tif 15 < len(data) {\n\t\tv15 = data[15]\n\t}\n\tv16 := 0\n\tif 16 < len(data) {\n\t\tv16 = data[16]\n\t}\n\tv17 := 0\n\tif 17 < len(data) {\n\t\tv17 = data[17]\n\t}\n\tv18 := 0\n\tif 18 < len(data) {\n\t\tv18 = data[18]\n\t}\n\tv19 := 0\n\tif 19 < len(data) {\n\t\tv19 = data[19]\n\t}\n\tv20 := 0\n\tif 20 < len(data) {\n\t\tv20 = data[20]\n\t}\n\tv21 := 0\n\tif 21 < len(data) {\n\t\tv21 = data[21]\n\t}\n\tv22 := 0\n\tif 22 < len(data) {\n\t\tv22 = data[22]\n\t}\n\tv23 := 0\n\tif 23 < len(data) {\n\t\tv23 = data[23]\n\t}\n\tv24 := 0\n\tif 24 < len(data) {\n\t\tv24 = data[24]\n\t}\n\tv25 := 0\n\tif 25 < len(data) {\n\t\tv25 = data[25]\n\t}\n\tv26 := 0\n\tif 26 < len(data) {\n\t\tv26 = data[26]\n\t}\n\tv27 := 0\n\tif 27 < len(data) {\n\t\tv27 = data[27]\n\t}\n\tv28 := 0\n\tif 28 < len(data) {\n\t\tv28 = data[28]\n\t}\n\tv29 := 0\n\tif 29 < len(data) {\n\t\tv29 = data[29]\n\t}\n\tv30 := 0\n\tif 30 < len(data) {\n\t\tv30 = data[30]\n\t}\n\tv31 := 0\n\tif 31 < len(data) {\n\t\tv31 = data[31]\n\t}\n\tv32 := 0\n\tif 32 < len(data) {\n\t\tv32 = data[32]\n\t}\n\tv33 := 0\n\tif 33 < len(data) {\n\t\tv33 = data[33]\n\t}\n\tv34 := 0\n\tif 34 < len(data) {\n\t\tv34 = data[34]\n\t}\n\tv35 := 0\n\tif 35 < len(data) {\n\t\tv35 = data[35]\n\t}\n\tv36 := 0\n\tif 36 < len(data) {\n\t\tv36 = data[36]\n\t}\n\tv37 := 0\n\tif 37 < len(data) {\n\t\tv37 = data[37]\n\t}\n\tv38 := 0\n\tif 38 < len(data) {\n\t\tv38 = data[38]\n\t}\n\tv39 := 0\n\tif 39 < len(data) {\n\t\tv39 = data[39]\n\t}\n\tv40 := 0\n\tif 40 < len(data) {\n\t\tv40 = data[40]\n\t}\n\tv41 := 0\n\tif 41 < len(data) {\n\t\tv41 = data[41]\n\t}\n\tv42 := 0\n\tif 42 < len(data) {\n\t\tv42 = data[42]\n\t}\n\tv43 := 0\n\tif 43 < len(data) {\n\t\tv43 = data[43]\n\t}\n\tv44 := 0\n\tif 44 < len(data) {\n\t\tv44 = data[44]\n\t}\n\tv45 := 0\n\tif 45 < len(data) {\n\t\tv45 = data[45]\n\t}\n\tv46 := 0\n\tif 46 < len(data) {\n\t\tv46 = data[46]\n\t}\n\tv47 := 0\n\tif 47 < len(data) {\n\t\tv47 = data[47]\n\t}\n\tv48 := 0\n\tif 48 < len(data) {\n\t\tv48 = data[48]\n\t}\n\tv49 := 0\n\tif 49 < len(data) {\n\t\tv49 = data[49]",
+    "token_estimate": 847
+  },
+  {
+    "block_ids": [
+      "5d269745b2e5dbdcbef0c09ba54b0bd6"
+    ],
+    "chunk_id": "3f44ba43c9415652e2705bb667776e76",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 447,
+        "line_start": 248,
+        "symbol": "BigCompute [part 2/5]"
+      }
+    ],
+    "text": "\t}\n\tv50 := 0\n\tif 50 < len(data) {\n\t\tv50 = data[50]\n\t}\n\tv51 := 0\n\tif 51 < len(data) {\n\t\tv51 = data[51]\n\t}\n\tv52 := 0\n\tif 52 < len(data) {\n\t\tv52 = data[52]\n\t}\n\tv53 := 0\n\tif 53 < len(data) {\n\t\tv53 = data[53]\n\t}\n\tv54 := 0\n\tif 54 < len(data) {\n\t\tv54 = data[54]\n\t}\n\tv55 := 0\n\tif 55 < len(data) {\n\t\tv55 = data[55]\n\t}\n\tv56 := 0\n\tif 56 < len(data) {\n\t\tv56 = data[56]\n\t}\n\tv57 := 0\n\tif 57 < len(data) {\n\t\tv57 = data[57]\n\t}\n\tv58 := 0\n\tif 58 < len(data) {\n\t\tv58 = data[58]\n\t}\n\tv59 := 0\n\tif 59 < len(data) {\n\t\tv59 = data[59]\n\t}\n\tv60 := 0\n\tif 60 < len(data) {\n\t\tv60 = data[60]\n\t}\n\tv61 := 0\n\tif 61 < len(data) {\n\t\tv61 = data[61]\n\t}\n\tv62 := 0\n\tif 62 < len(data) {\n\t\tv62 = data[62]\n\t}\n\tv63 := 0\n\tif 63 < len(data) {\n\t\tv63 = data[63]\n\t}\n\tv64 := 0\n\tif 64 < len(data) {\n\t\tv64 = data[64]\n\t}\n\tv65 := 0\n\tif 65 < len(data) {\n\t\tv65 = data[65]\n\t}\n\tv66 := 0\n\tif 66 < len(data) {\n\t\tv66 = data[66]\n\t}\n\tv67 := 0\n\tif 67 < len(data) {\n\t\tv67 = data[67]\n\t}\n\tv68 := 0\n\tif 68 < len(data) {\n\t\tv68 = data[68]\n\t}\n\tv69 := 0\n\tif 69 < len(data) {\n\t\tv69 = data[69]\n\t}\n\tv70 := 0\n\tif 70 < len(data) {\n\t\tv70 = data[70]\n\t}\n\tv71 := 0\n\tif 71 < len(data) {\n\t\tv71 = data[71]\n\t}\n\tv72 := 0\n\tif 72 < len(data) {\n\t\tv72 = data[72]\n\t}\n\tv73 := 0\n\tif 73 < len(data) {\n\t\tv73 = data[73]\n\t}\n\tv74 := 0\n\tif 74 < len(data) {\n\t\tv74 = data[74]\n\t}\n\tv75 := 0\n\tif 75 < len(data) {\n\t\tv75 = data[75]\n\t}\n\tv76 := 0\n\tif 76 < len(data) {\n\t\tv76 = data[76]\n\t}\n\tv77 := 0\n\tif 77 < len(data) {\n\t\tv77 = data[77]\n\t}\n\tv78 := 0\n\tif 78 < len(data) {\n\t\tv78 = data[78]\n\t}\n\tv79 := 0\n\tif 79 < len(data) {\n\t\tv79 = data[79]\n\t}\n\tv80 := 0\n\tif 80 < len(data) {\n\t\tv80 = data[80]\n\t}\n\tv81 := 0\n\tif 81 < len(data) {\n\t\tv81 = data[81]\n\t}\n\tv82 := 0\n\tif 82 < len(data) {\n\t\tv82 = data[82]\n\t}\n\tv83 := 0\n\tif 83 < len(data) {\n\t\tv83 = data[83]\n\t}\n\tv84 := 0\n\tif 84 < len(data) {\n\t\tv84 = data[84]\n\t}\n\tv85 := 0\n\tif 85 < len(data) {\n\t\tv85 = data[85]\n\t}\n\tv86 := 0\n\tif 86 < len(data) {\n\t\tv86 = data[86]\n\t}\n\tv87 := 0\n\tif 87 < len(data) {\n\t\tv87 = data[87]\n\t}\n\tv88 := 0\n\tif 88 < len(data) {\n\t\tv88 = data[88]\n\t}\n\tv89 := 0\n\tif 89 < len(data) {\n\t\tv89 = data[89]\n\t}\n\tv90 := 0\n\tif 90 < len(data) {\n\t\tv90 = data[90]\n\t}\n\tv91 := 0\n\tif 91 < len(data) {\n\t\tv91 = data[91]\n\t}\n\tv92 := 0\n\tif 92 < len(data) {\n\t\tv92 = data[92]\n\t}\n\tv93 := 0\n\tif 93 < len(data) {\n\t\tv93 = data[93]\n\t}\n\tv94 := 0\n\tif 94 < len(data) {\n\t\tv94 = data[94]\n\t}\n\tv95 := 0\n\tif 95 < len(data) {\n\t\tv95 = data[95]\n\t}\n\tv96 := 0\n\tif 96 < len(data) {\n\t\tv96 = data[96]\n\t}\n\tv97 := 0\n\tif 97 < len(data) {\n\t\tv97 = data[97]\n\t}\n\tv98 := 0\n\tif 98 < len(data) {\n\t\tv98 = data[98]\n\t}\n\tv99 := 0\n\tif 99 < len(data) {\n\t\tv99 = data[99]",
+    "token_estimate": 850
+  },
+  {
+    "block_ids": [
+      "5d269745b2e5dbdcbef0c09ba54b0bd6"
+    ],
+    "chunk_id": "e4763e10f059d97f40c2932761b56c3e",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 647,
+        "line_start": 448,
+        "symbol": "BigCompute [part 3/5]"
+      }
+    ],
+    "text": "\t}\n\tv100 := 0\n\tif 100 < len(data) {\n\t\tv100 = data[100]\n\t}\n\tv101 := 0\n\tif 101 < len(data) {\n\t\tv101 = data[101]\n\t}\n\tv102 := 0\n\tif 102 < len(data) {\n\t\tv102 = data[102]\n\t}\n\tv103 := 0\n\tif 103 < len(data) {\n\t\tv103 = data[103]\n\t}\n\tv104 := 0\n\tif 104 < len(data) {\n\t\tv104 = data[104]\n\t}\n\tv105 := 0\n\tif 105 < len(data) {\n\t\tv105 = data[105]\n\t}\n\tv106 := 0\n\tif 106 < len(data) {\n\t\tv106 = data[106]\n\t}\n\tv107 := 0\n\tif 107 < len(data) {\n\t\tv107 = data[107]\n\t}\n\tv108 := 0\n\tif 108 < len(data) {\n\t\tv108 = data[108]\n\t}\n\tv109 := 0\n\tif 109 < len(data) {\n\t\tv109 = data[109]\n\t}\n\tv110 := 0\n\tif 110 < len(data) {\n\t\tv110 = data[110]\n\t}\n\tv111 := 0\n\tif 111 < len(data) {\n\t\tv111 = data[111]\n\t}\n\tv112 := 0\n\tif 112 < len(data) {\n\t\tv112 = data[112]\n\t}\n\tv113 := 0\n\tif 113 < len(data) {\n\t\tv113 = data[113]\n\t}\n\tv114 := 0\n\tif 114 < len(data) {\n\t\tv114 = data[114]\n\t}\n\tv115 := 0\n\tif 115 < len(data) {\n\t\tv115 = data[115]\n\t}\n\tv116 := 0\n\tif 116 < len(data) {\n\t\tv116 = data[116]\n\t}\n\tv117 := 0\n\tif 117 < len(data) {\n\t\tv117 = data[117]\n\t}\n\tv118 := 0\n\tif 118 < len(data) {\n\t\tv118 = data[118]\n\t}\n\tv119 := 0\n\tif 119 < len(data) {\n\t\tv119 = data[119]\n\t}\n\tv120 := 0\n\tif 120 < len(data) {\n\t\tv120 = data[120]\n\t}\n\tv121 := 0\n\tif 121 < len(data) {\n\t\tv121 = data[121]\n\t}\n\tv122 := 0\n\tif 122 < len(data) {\n\t\tv122 = data[122]\n\t}\n\tv123 := 0\n\tif 123 < len(data) {\n\t\tv123 = data[123]\n\t}\n\tv124 := 0\n\tif 124 < len(data) {\n\t\tv124 = data[124]\n\t}\n\tv125 := 0\n\tif 125 < len(data) {\n\t\tv125 = data[125]\n\t}\n\tv126 := 0\n\tif 126 < len(data) {\n\t\tv126 = data[126]\n\t}\n\tv127 := 0\n\tif 127 < len(data) {\n\t\tv127 = data[127]\n\t}\n\tv128 := 0\n\tif 128 < len(data) {\n\t\tv128 = data[128]\n\t}\n\tv129 := 0\n\tif 129 < len(data) {\n\t\tv129 = data[129]\n\t}\n\tv130 := 0\n\tif 130 < len(data) {\n\t\tv130 = data[130]\n\t}\n\tv131 := 0\n\tif 131 < len(data) {\n\t\tv131 = data[131]\n\t}\n\tv132 := 0\n\tif 132 < len(data) {\n\t\tv132 = data[132]\n\t}\n\tv133 := 0\n\tif 133 < len(data) {\n\t\tv133 = data[133]\n\t}\n\tv134 := 0\n\tif 134 < len(data) {\n\t\tv134 = data[134]\n\t}\n\tv135 := 0\n\tif 135 < len(data) {\n\t\tv135 = data[135]\n\t}\n\tv136 := 0\n\tif 136 < len(data) {\n\t\tv136 = data[136]\n\t}\n\tv137 := 0\n\tif 137 < len(data) {\n\t\tv137 = data[137]\n\t}\n\tv138 := 0\n\tif 138 < len(data) {\n\t\tv138 = data[138]\n\t}\n\tv139 := 0\n\tif 139 < len(data) {\n\t\tv139 = data[139]\n\t}\n\tv140 := 0\n\tif 140 < len(data) {\n\t\tv140 = data[140]\n\t}\n\tv141 := 0\n\tif 141 < len(data) {\n\t\tv141 = data[141]\n\t}\n\tv142 := 0\n\tif 142 < len(data) {\n\t\tv142 = data[142]\n\t}\n\tv143 := 0\n\tif 143 < len(data) {\n\t\tv143 = data[143]\n\t}\n\tv144 := 0\n\tif 144 < len(data) {\n\t\tv144 = data[144]\n\t}\n\tv145 := 0\n\tif 145 < len(data) {\n\t\tv145 = data[145]\n\t}\n\tv146 := 0\n\tif 146 < len(data) {\n\t\tv146 = data[146]\n\t}\n\tv147 := 0\n\tif 147 < len(data) {\n\t\tv147 = data[147]\n\t}\n\tv148 := 0\n\tif 148 < len(data) {\n\t\tv148 = data[148]\n\t}\n\tv149 := 0\n\tif 149 < len(data) {\n\t\tv149 = data[149]",
+    "token_estimate": 917
+  },
+  {
+    "block_ids": [
+      "5d269745b2e5dbdcbef0c09ba54b0bd6"
+    ],
+    "chunk_id": "24176c911d0bacf9a29fa7f8251f5036",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 847,
+        "line_start": 648,
+        "symbol": "BigCompute [part 4/5]"
+      }
+    ],
+    "text": "\t}\n\tv150 := 0\n\tif 150 < len(data) {\n\t\tv150 = data[150]\n\t}\n\tv151 := 0\n\tif 151 < len(data) {\n\t\tv151 = data[151]\n\t}\n\tv152 := 0\n\tif 152 < len(data) {\n\t\tv152 = data[152]\n\t}\n\tv153 := 0\n\tif 153 < len(data) {\n\t\tv153 = data[153]\n\t}\n\tv154 := 0\n\tif 154 < len(data) {\n\t\tv154 = data[154]\n\t}\n\tv155 := 0\n\tif 155 < len(data) {\n\t\tv155 = data[155]\n\t}\n\tv156 := 0\n\tif 156 < len(data) {\n\t\tv156 = data[156]\n\t}\n\tv157 := 0\n\tif 157 < len(data) {\n\t\tv157 = data[157]\n\t}\n\tv158 := 0\n\tif 158 < len(data) {\n\t\tv158 = data[158]\n\t}\n\tv159 := 0\n\tif 159 < len(data) {\n\t\tv159 = data[159]\n\t}\n\tv160 := 0\n\tif 160 < len(data) {\n\t\tv160 = data[160]\n\t}\n\tv161 := 0\n\tif 161 < len(data) {\n\t\tv161 = data[161]\n\t}\n\tv162 := 0\n\tif 162 < len(data) {\n\t\tv162 = data[162]\n\t}\n\tv163 := 0\n\tif 163 < len(data) {\n\t\tv163 = data[163]\n\t}\n\tv164 := 0\n\tif 164 < len(data) {\n\t\tv164 = data[164]\n\t}\n\tv165 := 0\n\tif 165 < len(data) {\n\t\tv165 = data[165]\n\t}\n\tv166 := 0\n\tif 166 < len(data) {\n\t\tv166 = data[166]\n\t}\n\tv167 := 0\n\tif 167 < len(data) {\n\t\tv167 = data[167]\n\t}\n\tv168 := 0\n\tif 168 < len(data) {\n\t\tv168 = data[168]\n\t}\n\tv169 := 0\n\tif 169 < len(data) {\n\t\tv169 = data[169]\n\t}\n\tv170 := 0\n\tif 170 < len(data) {\n\t\tv170 = data[170]\n\t}\n\tv171 := 0\n\tif 171 < len(data) {\n\t\tv171 = data[171]\n\t}\n\tv172 := 0\n\tif 172 < len(data) {\n\t\tv172 = data[172]\n\t}\n\tv173 := 0\n\tif 173 < len(data) {\n\t\tv173 = data[173]\n\t}\n\tv174 := 0\n\tif 174 < len(data) {\n\t\tv174 = data[174]\n\t}\n\tv175 := 0\n\tif 175 < len(data) {\n\t\tv175 = data[175]\n\t}\n\tv176 := 0\n\tif 176 < len(data) {\n\t\tv176 = data[176]\n\t}\n\tv177 := 0\n\tif 177 < len(data) {\n\t\tv177 = data[177]\n\t}\n\tv178 := 0\n\tif 178 < len(data) {\n\t\tv178 = data[178]\n\t}\n\tv179 := 0\n\tif 179 < len(data) {\n\t\tv179 = data[179]\n\t}\n\tv180 := 0\n\tif 180 < len(data) {\n\t\tv180 = data[180]\n\t}\n\tv181 := 0\n\tif 181 < len(data) {\n\t\tv181 = data[181]\n\t}\n\tv182 := 0\n\tif 182 < len(data) {\n\t\tv182 = data[182]\n\t}\n\tv183 := 0\n\tif 183 < len(data) {\n\t\tv183 = data[183]\n\t}\n\tv184 := 0\n\tif 184 < len(data) {\n\t\tv184 = data[184]\n\t}\n\tv185 := 0\n\tif 185 < len(data) {\n\t\tv185 = data[185]\n\t}\n\tv186 := 0\n\tif 186 < len(data) {\n\t\tv186 = data[186]\n\t}\n\tv187 := 0\n\tif 187 < len(data) {\n\t\tv187 = data[187]\n\t}\n\tv188 := 0\n\tif 188 < len(data) {\n\t\tv188 = data[188]\n\t}\n\tv189 := 0\n\tif 189 < len(data) {\n\t\tv189 = data[189]\n\t}\n\tv190 := 0\n\tif 190 < len(data) {\n\t\tv190 = data[190]\n\t}\n\tv191 := 0\n\tif 191 < len(data) {\n\t\tv191 = data[191]\n\t}\n\tv192 := 0\n\tif 192 < len(data) {\n\t\tv192 = data[192]\n\t}\n\tv193 := 0\n\tif 193 < len(data) {\n\t\tv193 = data[193]\n\t}\n\tv194 := 0\n\tif 194 < len(data) {\n\t\tv194 = data[194]\n\t}\n\tv195 := 0\n\tif 195 < len(data) {\n\t\tv195 = data[195]\n\t}\n\tv196 := 0\n\tif 196 < len(data) {\n\t\tv196 = data[196]\n\t}\n\tv197 := 0\n\tif 197 < len(data) {\n\t\tv197 = data[197]\n\t}\n\tv198 := 0\n\tif 198 < len(data) {\n\t\tv198 = data[198]\n\t}\n\tv199 := 0\n\tif 199 < len(data) {\n\t\tv199 = data[199]",
+    "token_estimate": 917
+  },
+  {
+    "block_ids": [
+      "5d269745b2e5dbdcbef0c09ba54b0bd6"
+    ],
+    "chunk_id": "438127626378632c03780d10603de32c",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 890,
+        "line_start": 848,
+        "symbol": "BigCompute [part 5/5]"
+      }
+    ],
+    "text": "\t}\n\tv200 := 0\n\tif 200 < len(data) {\n\t\tv200 = data[200]\n\t}\n\tv201 := 0\n\tif 201 < len(data) {\n\t\tv201 = data[201]\n\t}\n\tv202 := 0\n\tif 202 < len(data) {\n\t\tv202 = data[202]\n\t}\n\tv203 := 0\n\tif 203 < len(data) {\n\t\tv203 = data[203]\n\t}\n\tv204 := 0\n\tif 204 < len(data) {\n\t\tv204 = data[204]\n\t}\n\tv205 := 0\n\tif 205 < len(data) {\n\t\tv205 = data[205]\n\t}\n\tv206 := 0\n\tif 206 < len(data) {\n\t\tv206 = data[206]\n\t}\n\tv207 := 0\n\tif 207 < len(data) {\n\t\tv207 = data[207]\n\t}\n\tv208 := 0\n\tif 208 < len(data) {\n\t\tv208 = data[208]\n\t}\n\tv209 := 0\n\tif 209 < len(data) {\n\t\tv209 = data[209]\n\t}\n\treturn len(data)\n}",
+    "token_estimate": 191
+  }
+]
--- a/crates/kebab-chunk/tests/fixtures/code-sample.java.chunks.snapshot.json
+++ b/crates/kebab-chunk/tests/fixtures/code-sample.java.chunks.snapshot.json
--- a/crates/kebab-chunk/tests/fixtures/code-sample.js.chunks.snapshot.json
+++ b/crates/kebab-chunk/tests/fixtures/code-sample.js.chunks.snapshot.json
--- a/crates/kebab-chunk/tests/fixtures/code-sample.kt.chunks.snapshot.json
+++ b/crates/kebab-chunk/tests/fixtures/code-sample.kt.chunks.snapshot.json
--- a/crates/kebab-chunk/tests/fixtures/code-sample.py.chunks.snapshot.json
+++ b/crates/kebab-chunk/tests/fixtures/code-sample.py.chunks.snapshot.json
--- a/crates/kebab-chunk/tests/fixtures/code-sample.ts.chunks.snapshot.json
+++ b/crates/kebab-chunk/tests/fixtures/code-sample.ts.chunks.snapshot.json
--- a/crates/kebab-chunk/tests/fixtures/sample.c
+++ b/crates/kebab-chunk/tests/fixtures/sample.c
@@ -0,0 +1,33 @@
+#include <stdio.h>
+#include <stdlib.h>
+
+#define MAX_BUF 4096
+
+typedef enum {
+    OK = 0,
+    ERR_PARSE,
+    ERR_IO,
+} status_t;
+
+typedef struct {
+    int id;
+    char name[64];
+    status_t status;
+} record_t;
+
+static int counter = 0;
+
+int parse_record(const char *line, record_t *out) {
+    if (line == NULL || out == NULL) return ERR_PARSE;
+    return OK;
+}
+
+void print_record(const record_t *r) {
+    printf("[%d] %s (status=%d)\n", r->id, r->name, r->status);
+}
+
+int main(void) {
+    record_t r = { .id = 1, .name = "foo", .status = OK };
+    print_record(&r);
+    return 0;
+}
--- a/crates/kebab-chunk/tests/fixtures/sample.cpp
+++ b/crates/kebab-chunk/tests/fixtures/sample.cpp
@@ -0,0 +1,40 @@
+#include <string>
+#include <vector>
+
+namespace kebab {
+namespace chunk {
+
+class MdHeadingV1Chunker {
+public:
+    MdHeadingV1Chunker() = default;
+    ~MdHeadingV1Chunker() = default;
+
+    std::string chunk_doc(const std::string& doc) {
+        return doc;
+    }
+
+    int operator()(int x) const {
+        return x * 2;
+    }
+
+private:
+    int counter_ = 0;
+};
+
+template <typename T>
+T identity(T value) {
+    return value;
+}
+
+}  // namespace chunk
+
+void global_helper() {
+    // free function in kebab namespace
+}
+
+}  // namespace kebab
+
+int main() {
+    kebab::chunk::MdHeadingV1Chunker c;
+    return 0;
+}
--- a/crates/kebab-chunk/tests/fixtures/sample.dockerfile
+++ b/crates/kebab-chunk/tests/fixtures/sample.dockerfile
@@ -0,0 +1,5 @@
+FROM rust:1.94-slim AS builder
+WORKDIR /app
+COPY . .
+RUN cargo build --release
+CMD ["/app/target/release/kebab"]
--- a/crates/kebab-chunk/tests/fixtures/sample_cargo.toml
+++ b/crates/kebab-chunk/tests/fixtures/sample_cargo.toml
@@ -0,0 +1,7 @@
+[package]
+name = "demo"
+version = "0.1.0"
+edition = "2021"
+
+[dependencies]
+serde = "1"
--- a/crates/kebab-chunk/tests/fixtures/sample_go.mod
+++ b/crates/kebab-chunk/tests/fixtures/sample_go.mod
@@ -0,0 +1,5 @@
+module example.com/demo
+
+go 1.22
+
+require github.com/spf13/cobra v1.8.0
--- a/crates/kebab-chunk/tests/fixtures/sample_k8s.yaml
+++ b/crates/kebab-chunk/tests/fixtures/sample_k8s.yaml
@@ -0,0 +1,34 @@
+apiVersion: apps/v1
+kind: Deployment
+metadata:
+  name: api-server
+  namespace: prod
+spec:
+  replicas: 3
+  selector:
+    matchLabels:
+      app: api-server
+  template:
+    metadata:
+      labels:
+        app: api-server
+    spec:
+      containers:
+      - name: api
+        image: example/api:1.2.3
+---
+apiVersion: v1
+kind: Service
+metadata:
+  name: api-server
+  namespace: prod
+spec:
+  selector:
+    app: api-server
+  ports:
+  - port: 80
+    targetPort: 8080
+---
+# Non-k8s document — apiVersion missing
+kind: ClusterIP
+foo: bar
--- a/crates/kebab-chunk/tests/fixtures/sample_long_paragraph.txt
+++ b/crates/kebab-chunk/tests/fixtures/sample_long_paragraph.txt
@@ -0,0 +1,200 @@
+line 001
+line 002
+line 003
+line 004
+line 005
+line 006
+line 007
+line 008
+line 009
+line 010
+line 011
+line 012
+line 013
+line 014
+line 015
+line 016
+line 017
+line 018
+line 019
+line 020
+line 021
+line 022
+line 023
+line 024
+line 025
+line 026
+line 027
+line 028
+line 029
+line 030
+line 031
+line 032
+line 033
+line 034
+line 035
+line 036
+line 037
+line 038
+line 039
+line 040
+line 041
+line 042
+line 043
+line 044
+line 045
+line 046
+line 047
+line 048
+line 049
+line 050
+line 051
+line 052
+line 053
+line 054
+line 055
+line 056
+line 057
+line 058
+line 059
+line 060
+line 061
+line 062
+line 063
+line 064
+line 065
+line 066
+line 067
+line 068
+line 069
+line 070
+line 071
+line 072
+line 073
+line 074
+line 075
+line 076
+line 077
+line 078
+line 079
+line 080
+line 081
+line 082
+line 083
+line 084
+line 085
+line 086
+line 087
+line 088
+line 089
+line 090
+line 091
+line 092
+line 093
+line 094
+line 095
+line 096
+line 097
+line 098
+line 099
+line 100
+line 101
+line 102
+line 103
+line 104
+line 105
+line 106
+line 107
+line 108
+line 109
+line 110
+line 111
+line 112
+line 113
+line 114
+line 115
+line 116
+line 117
+line 118
+line 119
+line 120
+line 121
+line 122
+line 123
+line 124
+line 125
+line 126
+line 127
+line 128
+line 129
+line 130
+line 131
+line 132
+line 133
+line 134
+line 135
+line 136
+line 137
+line 138
+line 139
+line 140
+line 141
+line 142
+line 143
+line 144
+line 145
+line 146
+line 147
+line 148
+line 149
+line 150
+line 151
+line 152
+line 153
+line 154
+line 155
+line 156
+line 157
+line 158
+line 159
+line 160
+line 161
+line 162
+line 163
+line 164
+line 165
+line 166
+line 167
+line 168
+line 169
+line 170
+line 171
+line 172
+line 173
+line 174
+line 175
+line 176
+line 177
+line 178
+line 179
+line 180
+line 181
+line 182
+line 183
+line 184
+line 185
+line 186
+line 187
+line 188
+line 189
+line 190
+line 191
+line 192
+line 193
+line 194
+line 195
+line 196
+line 197
+line 198
+line 199
+line 200
--- a/crates/kebab-chunk/tests/fixtures/sample_package.json
+++ b/crates/kebab-chunk/tests/fixtures/sample_package.json
@@ -0,0 +1,7 @@
+{
+  "name": "demo",
+  "version": "0.1.0",
+  "dependencies": {
+    "react": "^18.0.0"
+  }
+}
--- a/crates/kebab-chunk/tests/fixtures/sample_pom.xml
+++ b/crates/kebab-chunk/tests/fixtures/sample_pom.xml
@@ -0,0 +1,7 @@
+<?xml version="1.0" encoding="UTF-8"?>
+<project xmlns="http://maven.apache.org/POM/4.0.0">
+  <modelVersion>4.0.0</modelVersion>
+  <groupId>com.demo</groupId>
+  <artifactId>demo</artifactId>
+  <version>0.1.0</version>
+</project>
--- a/crates/kebab-chunk/tests/fixtures/sample_shell.sh
+++ b/crates/kebab-chunk/tests/fixtures/sample_shell.sh
@@ -0,0 +1,15 @@
+#!/usr/bin/env bash
+set -euo pipefail
+
+# First paragraph: env setup
+export KEBAB_HOME="${KEBAB_HOME:-$HOME/.local/share/kebab}"
+mkdir -p "$KEBAB_HOME"
+cd "$KEBAB_HOME"
+
+# Second paragraph: ingest
+echo "ingesting workspace..."
+kebab ingest --config /etc/kebab/config.toml
+
+# Third paragraph: report
+echo "done"
+kebab schema --json | jq '.stats'
--- a/crates/kebab-chunk/tests/k8s_manifest_resource_v1.rs
+++ b/crates/kebab-chunk/tests/k8s_manifest_resource_v1.rs
@@ -0,0 +1,286 @@
+//! Behavioural tests for `K8sManifestResourceV1Chunker`.
+//!
+//! Documents are constructed manually (no kebab-parse-code dependency) by
+//! placing the raw YAML text into a single `Block::Code`, mirroring the
+//! pattern used in `code_rust_ast_snapshot.rs`.
+
+use std::path::PathBuf;
+
+use kebab_chunk::K8sManifestResourceV1Chunker;
+use kebab_core::{
+    AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
+    CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
+    WorkspacePath, id_for_block, id_for_doc,
+};
+use time::OffsetDateTime;
+
+// ── helpers ──────────────────────────────────────────────────────────────────
+
+fn fixtures_dir() -> PathBuf {
+    PathBuf::from(env!("CARGO_MANIFEST_DIR"))
+        .join("tests")
+        .join("fixtures")
+}
+
+/// Build a `CanonicalDocument` with a single `Block::Code` containing `yaml_text`.
+fn yaml_doc(yaml_text: &str) -> CanonicalDocument {
+    let wp = WorkspacePath("manifests/deploy.yaml".into());
+    let aid = AssetId("c".repeat(64));
+    let pv = ParserVersion("code-yaml-v1".into());
+    let doc_id = id_for_doc(&wp, &aid, &pv);
+
+    let line_count = yaml_text.lines().count() as u32;
+    let span = SourceSpan::Code {
+        line_start: 1,
+        line_end: line_count.max(1),
+        symbol: None,
+        lang: Some("yaml".into()),
+    };
+    let bid = id_for_block(&doc_id, "code", &[], 0, &span);
+    let block = Block::Code(CodeBlock {
+        common: CommonBlock {
+            block_id: bid,
+            heading_path: vec![],
+            source_span: span,
+        },
+        lang: Some("yaml".into()),
+        code: yaml_text.to_string(),
+    });
+
+    CanonicalDocument {
+        doc_id,
+        source_asset_id: aid,
+        workspace_path: wp,
+        title: "deploy.yaml".into(),
+        lang: Lang("und".into()),
+        blocks: vec![block],
+        metadata: Metadata {
+            aliases: vec![],
+            tags: vec![],
+            created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            source_type: SourceType::Note,
+            trust_level: TrustLevel::Primary,
+            user_id_alias: None,
+            user: Default::default(),
+            repo: Some("kebab".into()),
+            git_branch: Some("main".into()),
+            git_commit: Some("0".repeat(40)),
+            code_lang: Some("yaml".into()),
+        },
+        provenance: Provenance { events: vec![] },
+        parser_version: pv,
+        schema_version: 1,
+        doc_version: 1,
+        last_chunker_version: None,
+        last_embedding_version: None,
+    }
+}
+
+fn policy() -> ChunkPolicy {
+    ChunkPolicy {
+        target_tokens: 500,
+        overlap_tokens: 80,
+        respect_markdown_headings: false,
+        chunker_version: ChunkerVersion("k8s-manifest-resource-v1".into()),
+    }
+}
+
+// ── tests ─────────────────────────────────────────────────────────────────────
+
+/// Three YAML documents: 2 valid k8s resources + 1 non-k8s (no apiVersion).
+/// The chunker must emit exactly 2 chunks with the correct symbols and lang.
+#[test]
+fn k8s_multi_doc_emits_one_chunk_per_resource() {
+    let fixture_path = fixtures_dir().join("sample_k8s.yaml");
+    let text = std::fs::read_to_string(&fixture_path)
+        .unwrap_or_else(|e| panic!("cannot read fixture {}: {e}", fixture_path.display()));
+
+    let doc = yaml_doc(&text);
+    let chunks = K8sManifestResourceV1Chunker
+        .chunk(&doc, &policy())
+        .expect("chunk");
+
+    assert_eq!(
+        chunks.len(),
+        2,
+        "expected 2 k8s chunks, got {}: {chunks:#?}",
+        chunks.len()
+    );
+
+    let symbols: Vec<&str> = chunks
+        .iter()
+        .map(|c| {
+            match &c.source_spans[0] {
+                SourceSpan::Code { symbol, .. } => {
+                    symbol.as_deref().expect("symbol must be Some for k8s chunks")
+                }
+                other => panic!("expected Code span, got {other:?}"),
+            }
+        })
+        .collect();
+
+    assert_eq!(
+        symbols,
+        vec!["Deployment/prod/api-server", "Service/prod/api-server"],
+        "symbols mismatch: {symbols:?}"
+    );
+
+    // Verify lang = "yaml" on every chunk.
+    for chunk in &chunks {
+        match &chunk.source_spans[0] {
+            SourceSpan::Code { lang, .. } => {
+                assert_eq!(lang.as_deref(), Some("yaml"), "lang must be 'yaml'");
+            }
+            other => panic!("expected Code span, got {other:?}"),
+        }
+    }
+
+    // Verify chunker_version label.
+    for chunk in &chunks {
+        assert_eq!(chunk.chunker_version.0, "k8s-manifest-resource-v1");
+    }
+
+    // Every chunk from a multi-resource file must have a distinct chunk_id.
+    // Without the fix, all non-oversize resources get split_key=None which
+    // collapses to the same id_hash (= base_policy_hash) → UNIQUE constraint
+    // violation on the second resource.
+    let ids: std::collections::HashSet<_> = chunks.iter().map(|c| c.chunk_id.clone()).collect();
+    assert_eq!(
+        ids.len(),
+        chunks.len(),
+        "every k8s resource chunk must have a distinct chunk_id (multi-resource collision regression)"
+    );
+}
+
+/// A YAML document with an indentation error (tab in a space-indented context)
+/// must cause the chunker to return 0 chunks for the entire file.
+#[test]
+fn k8s_invalid_yaml_emits_zero_chunks() {
+    // serde_yaml 0.9 is lenient about duplicate keys (last wins), so use a
+    // genuine YAML structural error (unclosed flow sequence) to force a parse
+    // failure.
+    let actually_bad = "apiVersion: v1\nkind: Service\nfoo: [\nbar\n";
+
+    let doc = yaml_doc(actually_bad);
+    let chunks = K8sManifestResourceV1Chunker
+        .chunk(&doc, &policy())
+        .expect("chunk should not error — return Ok(vec![]) for invalid yaml");
+
+    assert_eq!(
+        chunks.len(),
+        0,
+        "invalid YAML must yield 0 chunks, got {}: {chunks:#?}",
+        chunks.len()
+    );
+}
+
+/// A cluster-scoped resource (no `metadata.namespace`) must produce a symbol
+/// of the form `<Kind>/<name>` (two components, no namespace segment).
+#[test]
+fn k8s_cluster_scoped_resource_symbol() {
+    let yaml = "\
+apiVersion: rbac.authorization.k8s.io/v1
+kind: ClusterRole
+metadata:
+  name: cluster-admin
+rules:
+- apiGroups: [\"*\"]
+  resources: [\"*\"]
+  verbs: [\"*\"]
+";
+
+    let doc = yaml_doc(yaml);
+    let chunks = K8sManifestResourceV1Chunker
+        .chunk(&doc, &policy())
+        .expect("chunk");
+
+    assert_eq!(
+        chunks.len(),
+        1,
+        "expected 1 chunk for cluster-scoped resource, got {}: {chunks:#?}",
+        chunks.len()
+    );
+
+    match &chunks[0].source_spans[0] {
+        SourceSpan::Code { symbol, lang, .. } => {
+            assert_eq!(
+                symbol.as_deref(),
+                Some("ClusterRole/cluster-admin"),
+                "cluster-scoped symbol must be <Kind>/<name>"
+            );
+            assert_eq!(lang.as_deref(), Some("yaml"));
+        }
+        other => panic!("expected Code span, got {other:?}"),
+    }
+}
+
+/// 200+ line resource exercises `tier2_shared::push_chunks_with_oversize`'s
+/// line-window split branch. All chunks must share the same symbol
+/// (`<Kind>/<ns>/<name>`); their line ranges must form a contiguous
+/// partition; chunk_ids must all differ (the `#L{k}` suffix on `id_for_chunk`
+/// ensures uniqueness across windows). Spec p10-2 risks section explicitly
+/// flags "거대 ConfigMap" — this test covers that path.
+#[test]
+fn k8s_oversize_splits_into_line_windows_sharing_symbol() {
+    // ConfigMap with 250 data keys → ~256 total lines, > AST_CHUNK_MAX_LINES (200).
+    let mut yaml = String::from(
+        "apiVersion: v1\nkind: ConfigMap\nmetadata:\n  name: big\n  namespace: prod\ndata:\n",
+    );
+    for i in 0..250 {
+        yaml.push_str(&format!("  key{i}: value{i}\n"));
+    }
+
+    let doc = yaml_doc(&yaml);
+    let chunks = K8sManifestResourceV1Chunker
+        .chunk(&doc, &policy())
+        .expect("chunk");
+
+    assert!(
+        chunks.len() >= 2,
+        "expected ≥2 chunks for oversize resource, got {}",
+        chunks.len()
+    );
+
+    // Every chunk must share the same symbol + lang.
+    let expected_symbol = "ConfigMap/prod/big";
+    for (i, c) in chunks.iter().enumerate() {
+        match &c.source_spans[0] {
+            SourceSpan::Code { symbol, lang, .. } => {
+                assert_eq!(
+                    symbol.as_deref(),
+                    Some(expected_symbol),
+                    "chunk[{i}] symbol must equal `{expected_symbol}`"
+                );
+                assert_eq!(lang.as_deref(), Some("yaml"));
+            }
+            other => panic!("chunk[{i}]: expected Code span, got {other:?}"),
+        }
+    }
+
+    // chunk_ids must all be distinct (oversize fallback's #L{k} suffix).
+    let ids: std::collections::HashSet<_> = chunks.iter().map(|c| c.chunk_id.clone()).collect();
+    assert_eq!(
+        ids.len(),
+        chunks.len(),
+        "oversize chunks must have distinct chunk_ids (the #L{{k}} suffix should disambiguate)"
+    );
+
+    // Line ranges must form a contiguous partition: chunk[i].line_end + 1 == chunk[i+1].line_start.
+    let ranges: Vec<(u32, u32)> = chunks
+        .iter()
+        .map(|c| match &c.source_spans[0] {
+            SourceSpan::Code { line_start, line_end, .. } => (*line_start, *line_end),
+            other => panic!("expected Code span, got {other:?}"),
+        })
+        .collect();
+    for w in ranges.windows(2) {
+        let (_, prev_end) = w[0];
+        let (next_start, _) = w[1];
+        assert_eq!(
+            prev_end + 1,
+            next_start,
+            "line ranges must be contiguous: {prev_end} → {next_start} (got gap or overlap)"
+        );
+    }
+}
--- a/crates/kebab-chunk/tests/long_section_snapshot.rs
+++ b/crates/kebab-chunk/tests/long_section_snapshot.rs
@@ -18,8 +18,7 @@ use kebab_core::{
    AssetId, AssetStorage, Checksum, ChunkPolicy, ChunkerVersion, Chunker, MediaType,
    ParserVersion, RawAsset, SourceUri, WorkspacePath,
 };
-use kebab_normalize::build_canonical_document;
-use kebab_parse_md::{BodyHints, parse_blocks, parse_frontmatter};
+use kebab_parse_md::{BodyHints, build_canonical_document, parse_blocks, parse_frontmatter};
 use serde_json::Value;
 use time::OffsetDateTime;

--- a/crates/kebab-chunk/tests/manifest_file_v1.rs
+++ b/crates/kebab-chunk/tests/manifest_file_v1.rs
@@ -0,0 +1,267 @@
+//! Behavioural tests for `ManifestFileV1Chunker`.
+//!
+//! Documents are constructed manually (no kebab-parse-code dependency) by
+//! placing the raw manifest text into a single `Block::Code`, mirroring the
+//! pattern used in `dockerfile_file_v1.rs`.
+
+use std::path::PathBuf;
+
+use kebab_chunk::ManifestFileV1Chunker;
+use kebab_core::{
+    AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock,
+    CommonBlock, Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel,
+    WorkspacePath, id_for_block, id_for_doc,
+};
+use time::OffsetDateTime;
+
+// ── helpers ──────────────────────────────────────────────────────────────────
+
+fn fixtures_dir() -> PathBuf {
+    PathBuf::from(env!("CARGO_MANIFEST_DIR"))
+        .join("tests")
+        .join("fixtures")
+}
+
+/// Build a `CanonicalDocument` with a single `Block::Code` containing manifest text.
+fn manifest_doc(lang: &str, manifest_text: &str) -> CanonicalDocument {
+    let wp = WorkspacePath(format!("build/{}", manifest_filename(lang)));
+    let aid = AssetId("m".repeat(64));
+    let pv = ParserVersion("code-manifest-v1".into());
+    let doc_id = id_for_doc(&wp, &aid, &pv);
+
+    let line_count = manifest_text.lines().count() as u32;
+    let span = SourceSpan::Code {
+        line_start: 1,
+        line_end: line_count.max(1),
+        symbol: None,
+        lang: Some(lang.into()),
+    };
+    let bid = id_for_block(&doc_id, "code", &[], 0, &span);
+    let block = Block::Code(CodeBlock {
+        common: CommonBlock {
+            block_id: bid,
+            heading_path: vec![],
+            source_span: span,
+        },
+        lang: Some(lang.into()),
+        code: manifest_text.to_string(),
+    });
+
+    CanonicalDocument {
+        doc_id,
+        source_asset_id: aid,
+        workspace_path: wp,
+        title: format!("Manifest ({lang})"),
+        lang: Lang("und".into()),
+        blocks: vec![block],
+        metadata: Metadata {
+            aliases: vec![],
+            tags: vec![],
+            created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            source_type: SourceType::Note,
+            trust_level: TrustLevel::Primary,
+            user_id_alias: None,
+            user: Default::default(),
+            repo: Some("kebab".into()),
+            git_branch: Some("main".into()),
+            git_commit: Some("0".repeat(40)),
+            code_lang: Some(lang.into()),
+        },
+        provenance: Provenance { events: vec![] },
+        parser_version: pv,
+        schema_version: 1,
+        doc_version: 1,
+        last_chunker_version: None,
+        last_embedding_version: None,
+    }
+}
+
+fn manifest_filename(lang: &str) -> &'static str {
+    match lang {
+        "toml" => "Cargo.toml",
+        "json" => "package.json",
+        "xml" => "pom.xml",
+        "go-mod" => "go.mod",
+        _ => "manifest",
+    }
+}
+
+fn policy() -> ChunkPolicy {
+    ChunkPolicy {
+        target_tokens: 500,
+        overlap_tokens: 80,
+        respect_markdown_headings: false,
+        chunker_version: ChunkerVersion("manifest-file-v1".into()),
+    }
+}
+
+// ── tests ─────────────────────────────────────────────────────────────────────
+
+/// A Cargo.toml fixture must emit exactly 1 chunk with the correct symbol,
+/// lang, and line range.
+#[test]
+fn cargo_toml_single_chunk_with_toml_lang() {
+    let fixture_path = fixtures_dir().join("sample_cargo.toml");
+    let text = std::fs::read_to_string(&fixture_path)
+        .unwrap_or_else(|e| panic!("cannot read fixture {}: {e}", fixture_path.display()));
+
+    let doc = manifest_doc("toml", &text);
+    let chunks = ManifestFileV1Chunker
+        .chunk(&doc, &policy())
+        .expect("chunk");
+
+    assert_eq!(
+        chunks.len(),
+        1,
+        "expected 1 chunk, got {}: {chunks:#?}",
+        chunks.len()
+    );
+
+    let span = chunks[0].source_spans.first().expect("at least one span");
+    match span {
+        SourceSpan::Code {
+            line_start,
+            line_end: _,
+            symbol,
+            lang,
+        } => {
+            assert_eq!(*line_start, 1, "line_start must be 1");
+            assert_eq!(
+                symbol.as_deref(),
+                Some("<manifest>"),
+                "symbol must be '<manifest>'"
+            );
+            assert_eq!(lang.as_deref(), Some("toml"), "lang must be 'toml'");
+        }
+        other => panic!("expected SourceSpan::Code, got {other:?}"),
+    }
+
+    assert_eq!(chunks[0].chunker_version.0, "manifest-file-v1");
+}
+
+/// A package.json fixture must emit exactly 1 chunk with the correct symbol,
+/// lang, and line range.
+#[test]
+fn package_json_single_chunk_with_json_lang() {
+    let fixture_path = fixtures_dir().join("sample_package.json");
+    let text = std::fs::read_to_string(&fixture_path)
+        .unwrap_or_else(|e| panic!("cannot read fixture {}: {e}", fixture_path.display()));
+
+    let doc = manifest_doc("json", &text);
+    let chunks = ManifestFileV1Chunker
+        .chunk(&doc, &policy())
+        .expect("chunk");
+
+    assert_eq!(
+        chunks.len(),
+        1,
+        "expected 1 chunk, got {}: {chunks:#?}",
+        chunks.len()
+    );
+
+    let span = chunks[0].source_spans.first().expect("at least one span");
+    match span {
+        SourceSpan::Code {
+            line_start,
+            line_end: _,
+            symbol,
+            lang,
+        } => {
+            assert_eq!(*line_start, 1, "line_start must be 1");
+            assert_eq!(
+                symbol.as_deref(),
+                Some("<manifest>"),
+                "symbol must be '<manifest>'"
+            );
+            assert_eq!(lang.as_deref(), Some("json"), "lang must be 'json'");
+        }
+        other => panic!("expected SourceSpan::Code, got {other:?}"),
+    }
+
+    assert_eq!(chunks[0].chunker_version.0, "manifest-file-v1");
+}
+
+/// A pom.xml fixture must emit exactly 1 chunk with the correct symbol,
+/// lang, and line range.
+#[test]
+fn pom_xml_single_chunk_with_xml_lang() {
+    let fixture_path = fixtures_dir().join("sample_pom.xml");
+    let text = std::fs::read_to_string(&fixture_path)
+        .unwrap_or_else(|e| panic!("cannot read fixture {}: {e}", fixture_path.display()));
+
+    let doc = manifest_doc("xml", &text);
+    let chunks = ManifestFileV1Chunker
+        .chunk(&doc, &policy())
+        .expect("chunk");
+
+    assert_eq!(
+        chunks.len(),
+        1,
+        "expected 1 chunk, got {}: {chunks:#?}",
+        chunks.len()
+    );
+
+    let span = chunks[0].source_spans.first().expect("at least one span");
+    match span {
+        SourceSpan::Code {
+            line_start,
+            line_end: _,
+            symbol,
+            lang,
+        } => {
+            assert_eq!(*line_start, 1, "line_start must be 1");
+            assert_eq!(
+                symbol.as_deref(),
+                Some("<manifest>"),
+                "symbol must be '<manifest>'"
+            );
+            assert_eq!(lang.as_deref(), Some("xml"), "lang must be 'xml'");
+        }
+        other => panic!("expected SourceSpan::Code, got {other:?}"),
+    }
+
+    assert_eq!(chunks[0].chunker_version.0, "manifest-file-v1");
+}
+
+/// A go.mod fixture must emit exactly 1 chunk with the correct symbol,
+/// lang, and line range.
+#[test]
+fn go_mod_single_chunk_with_go_mod_lang() {
+    let fixture_path = fixtures_dir().join("sample_go.mod");
+    let text = std::fs::read_to_string(&fixture_path)
+        .unwrap_or_else(|e| panic!("cannot read fixture {}: {e}", fixture_path.display()));
+
+    let doc = manifest_doc("go-mod", &text);
+    let chunks = ManifestFileV1Chunker
+        .chunk(&doc, &policy())
+        .expect("chunk");
+
+    assert_eq!(
+        chunks.len(),
+        1,
+        "expected 1 chunk, got {}: {chunks:#?}",
+        chunks.len()
+    );
+
+    let span = chunks[0].source_spans.first().expect("at least one span");
+    match span {
+        SourceSpan::Code {
+            line_start,
+            line_end: _,
+            symbol,
+            lang,
+        } => {
+            assert_eq!(*line_start, 1, "line_start must be 1");
+            assert_eq!(
+                symbol.as_deref(),
+                Some("<manifest>"),
+                "symbol must be '<manifest>'"
+            );
+            assert_eq!(lang.as_deref(), Some("go-mod"), "lang must be 'go-mod'");
+        }
+        other => panic!("expected SourceSpan::Code, got {other:?}"),
+    }
+
+    assert_eq!(chunks[0].chunker_version.0, "manifest-file-v1");
+}
--- a/crates/kebab-cli/Cargo.toml
+++ b/crates/kebab-cli/Cargo.toml
@@ -50,3 +50,6 @@ tempfile     = { workspace = true }
 # to simulate stale docs. `time` is the formatter used by the helper.
 rusqlite     = { workspace = true }
 time         = { workspace = true }
+
+[lints]
+workspace = true
--- a/crates/kebab-cli/src/main.rs
+++ b/crates/kebab-cli/src/main.rs
@@ -46,6 +46,11 @@ struct Cli {
    command: Cmd,
 }

+// p10-1A-1: adding `repo` and `code_lang` Vec<String> fields pushed `Cmd`
+// over clippy's large_enum_variant threshold. The enum is short-lived
+// (parsed once at startup, never cloned in a hot path) — boxing would add
+// noise with no real benefit.
+#[allow(clippy::large_enum_variant)]
 #[derive(Subcommand, Debug)]
 enum Cmd {
    /// Initialise XDG dirs + workspace + `config.toml`.
@@ -165,6 +170,18 @@ enum Cmd {
        #[arg(long)]
        doc_id: Option<String>,

+        /// p10-1A-1: filter by repo name (`metadata.repo`). Repeatable;
+        /// multi-value = OR.  Empty = no filter (all repos returned).
+        #[arg(long = "repo", value_name = "NAME", num_args = 1)]
+        repo: Vec<String>,
+
+        /// p10-1A-1: filter by code language identifier (lowercase
+        /// canonical).  Repeatable or comma-separated.
+        /// Examples: `rust`, `python`, `typescript`.
+        /// Unknown values produce empty hits.
+        #[arg(long = "code-lang", value_name = "LANG", num_args = 1, value_delimiter = ',')]
+        code_lang: Vec<String>,
+
        /// p9-fb-37: emit pre-fusion lexical / vector / RRF candidate
        /// lists + per-stage timing in the response. Bypasses cache
        /// (debug intent — fresh run guaranteed). Requires embeddings
@@ -233,6 +250,18 @@ enum Cmd {
        /// `answer.v1`. Off by default to preserve final-only behavior.
        #[arg(long)]
        stream: bool,
+
+        /// p9-fb-41: route this ask through the multi-hop pipeline
+        /// — the query is decomposed into sub-questions, each
+        /// retrieved independently, then synthesized over the
+        /// merged chunk pool. Cost trade-off: 2–5× LLM calls
+        /// (decompose + 0..N decide + synthesize) vs. single-pass.
+        /// Worth it for compound questions (X 와 Y 의 관계, prereq
+        /// chain, cross-doc reasoning); single-pass is faster for
+        /// simple fact lookups. The full per-hop trace is exposed
+        /// on `Answer.hops` in `--json` mode.
+        #[arg(long)]
+        multi_hop: bool,
    },

    /// Wipe XDG data dirs (and optionally the Lance vector store) so the
@@ -258,6 +287,14 @@ enum Cmd {
        #[arg(long, group = "reset_scope")]
        config_only: bool,

+        /// Purge stored docs that are outside the current walker scope
+        /// (config narrowing / removed sub-directory). No filesystem paths
+        /// are removed — this is purely a store-level reconciliation.
+        /// Filesystem existence is NOT checked; anything the current walker
+        /// would not visit is considered an orphan and removed from the store.
+        #[arg(long, group = "reset_scope")]
+        orphans_only: bool,
+
        /// Skip the interactive confirm. Required in non-interactive
        /// contexts (CI, pipes).
        #[arg(long)]
@@ -578,14 +615,20 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
                println!("{}", serde_json::to_string(&wire::wire_ingest(&report))?);
            } else {
                let skipped_breakdown = kebab_app::render_skipped_breakdown(&report.skipped_by_extension);
+                let purged_suffix = if report.purged_deleted_files > 0 {
+                    format!("  purged {}", report.purged_deleted_files)
+                } else {
+                    String::new()
+                };
                println!(
-                    "scanned {}  new {}  updated {}  skipped {}{}  errors {}  ({} ms)",
+                    "scanned {}  new {}  updated {}  skipped {}{}  errors {}{}  ({} ms)",
                    report.scanned,
                    report.new,
                    report.updated,
                    report.skipped,
                    skipped_breakdown,
                    report.errors,
+                    purged_suffix,
                    report.duration_ms
                );
            }
@@ -688,6 +731,8 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
            media,
            ingested_after,
            doc_id,
+            repo,
+            code_lang,
            trace,
            bulk,
        } => {
@@ -752,7 +797,7 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
                            serde_json::to_string(&item.query)?,
                        )?;
                        if let Some(err) = &item.error {
-                            writeln!(stdout, "error: {}", err)?;
+                            writeln!(stdout, "error: {err}")?;
                        } else if let Some(resp) = &item.response {
                            writeln!(
                                stdout,
@@ -819,7 +864,7 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
                    None => None,
                };

-            // p9-fb-36: build SearchFilters from the 7 new flags.
+            // p9-fb-36 + p10-1A-1: build SearchFilters from CLI flags.
            let filters = kebab_core::SearchFilters {
                tags_any: tag.clone(),
                lang: lang.as_ref().map(|s| kebab_core::Lang(s.clone())),
@@ -828,6 +873,8 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
                media: media_norm,
                ingested_after: ingested_after_parsed,
                doc_id: doc_id.as_ref().map(|s| kebab_core::DocumentId(s.clone())),
+                repo: repo.clone(),
+                code_lang: code_lang.clone(),
            };

            let q = kebab_core::SearchQuery {
@@ -898,6 +945,15 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
                    let next = resp.next_cursor.as_deref().unwrap_or("(none)");
                    eprintln!("[truncated; use --cursor {next} for the next page]");
                }
+                // v0.17.0 A5 Step 4: short-query advisory. `resp.hint`
+                // is `Some` only when the result list is empty and the
+                // trimmed query is shorter than the trigram tokenizer
+                // can resolve (raw FTS5 mode opts out). stderr so it
+                // doesn't pollute the stdout hit list. `--json` skips
+                // this branch entirely; the field rides the wire.
+                if let Some(hint) = &resp.hint {
+                    eprintln!("[hint] {hint}");
+                }
                if *trace {
                    if let Some(t) = &resp.trace {
                        eprintln!();
@@ -929,6 +985,7 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
            hide_citations,
            session,
            stream,
+            multi_hop,
        } => {
            let cfg = kebab_config::Config::load(cli.config.as_deref())?;
            if *stream {
@@ -955,6 +1012,7 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
                    history: Vec::new(),
                    conversation_id: None,
                    turn_index: None,
+                    multi_hop: *multi_hop,
                };
                let cfg2 = cfg.clone();
                let q = query.clone();
@@ -1030,6 +1088,7 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
                    history: Vec::new(),
                    conversation_id: None,
                    turn_index: None,
+                    multi_hop: *multi_hop,
                };
                let ans = match session.as_deref() {
                    Some(sid) => kebab_app::ask_with_session_with_config(cfg, sid, query, opts)?,
@@ -1067,6 +1126,7 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
            data_only: _,
            vector_only,
            config_only,
+            orphans_only,
            yes,
        } => {
            use kebab_app::ResetScope;
@@ -1080,11 +1140,48 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
                ResetScope::VectorOnly
            } else if *config_only {
                ResetScope::ConfigOnly
+            } else if *orphans_only {
+                ResetScope::OrphansOnly
            } else {
                ResetScope::DataOnly
            };

            let cfg = kebab_config::Config::load(cli.config.as_deref())?;
+
+            if matches!(scope, ResetScope::OrphansOnly) {
+                // OrphansOnly: confirm UI shows orphan count + sample paths
+                // rather than on-disk directory sizes.
+                let orphan_paths = kebab_app::enumerate_orphans(&cfg)?;
+
+                if !*yes {
+                    use std::io::IsTerminal;
+                    if !std::io::stdin().is_terminal() {
+                        anyhow::bail!(
+                            "reset --orphans-only is destructive and stdin is non-interactive — pass --yes to proceed"
+                        );
+                    }
+                    if !confirm_orphans_only(&orphan_paths)? {
+                        if !cli.quiet {
+                            eprintln!("aborted.");
+                        }
+                        return Ok(());
+                    }
+                }
+
+                let report = kebab_app::reset::execute(scope, &cfg)?;
+                if cli.json {
+                    println!("{}", serde_json::to_string(&wire::wire_reset(&report))?);
+                } else if report.orphans_purged > 0 {
+                    println!("orphans purged: {}", report.orphans_purged);
+                    for p in &report.purged_paths {
+                        println!("  - {}", p.0);
+                    }
+                } else {
+                    println!("no orphaned docs found — store is already in sync with walker scope");
+                }
+                return Ok(());
+            }
+
            let paths = kebab_app::reset::enumerate_paths(scope, &cfg);
            let bytes = kebab_app::reset::estimate_size_bytes(&paths);

@@ -1409,11 +1506,11 @@ fn confirm_destructive(
 ) -> anyhow::Result<bool> {
    use std::io::Write;
    let mut out = std::io::stderr().lock();
-    writeln!(out, "kebab reset ({:?}): about to remove", scope)?;
+    writeln!(out, "kebab reset ({scope:?}): about to remove")?;
    for p in paths {
        writeln!(out, "  - {}", p.display())?;
    }
-    writeln!(out, "estimated total: {} bytes", bytes)?;
+    writeln!(out, "estimated total: {bytes} bytes")?;
    write!(out, "Proceed? [y/N] ")?;
    out.flush()?;

@@ -1423,6 +1520,46 @@ fn confirm_destructive(
    Ok(matches!(s.as_str(), "y" | "yes"))
 }

+/// Confirm prompt for `--orphans-only`: shows the orphan count + a
+/// sample of up to 5 paths so the user knows what will be purged before
+/// committing. No filesystem paths are removed — only store records.
+fn confirm_orphans_only(
+    orphan_paths: &[kebab_core::WorkspacePath],
+) -> anyhow::Result<bool> {
+    use std::io::Write;
+    let n = orphan_paths.len();
+    let mut out = std::io::stderr().lock();
+
+    if n == 0 {
+        writeln!(out, "no orphaned docs found — nothing to purge.")?;
+        out.flush()?;
+        // Nothing to do; treat as confirmed so the caller can emit the
+        // "no orphans" report without prompting.
+        return Ok(true);
+    }
+
+    let sample: Vec<&str> = orphan_paths
+        .iter()
+        .take(5)
+        .map(|p| p.0.as_str())
+        .collect();
+    let sample_str = sample.join(", ");
+    let ellipsis = if n > 5 { ", …" } else { "" };
+
+    writeln!(
+        out,
+        "Purge {n} stored doc(s) outside the current walker scope? (no filesystem paths removed)"
+    )?;
+    writeln!(out, "  sample: {sample_str}{ellipsis}")?;
+    write!(out, "[y/N] ")?;
+    out.flush()?;
+
+    let mut line = String::new();
+    std::io::stdin().read_line(&mut line)?;
+    let s = line.trim().to_ascii_lowercase();
+    Ok(matches!(s.as_str(), "y" | "yes"))
+}
+
 /// p9-fb-35: human-friendly plain output for `kebab fetch`.
 fn render_fetch_plain(r: &kebab_core::FetchResult) {
    println!("# {} ({})", r.doc_path.0, format_kind(r.kind));
@@ -1434,19 +1571,19 @@ fn render_fetch_plain(r: &kebab_core::FetchResult) {
            if !r.context_before.is_empty() {
                println!("\n=== before ===");
                for c in &r.context_before {
-                    let heading = c.heading_path.last().map(|s| s.as_str()).unwrap_or("");
+                    let heading = c.heading_path.last().map_or("", std::string::String::as_str);
                    println!("[{} § {}]\n{}\n", c.chunk_id.0, heading, c.text);
                }
            }
            if let Some(c) = &r.chunk {
                println!("\n=== target ===");
-                let heading = c.heading_path.last().map(|s| s.as_str()).unwrap_or("");
+                let heading = c.heading_path.last().map_or("", std::string::String::as_str);
                println!("[{} § {}]\n{}\n", c.chunk_id.0, heading, c.text);
            }
            if !r.context_after.is_empty() {
                println!("\n=== after ===");
                for c in &r.context_after {
-                    let heading = c.heading_path.last().map(|s| s.as_str()).unwrap_or("");
+                    let heading = c.heading_path.last().map_or("", std::string::String::as_str);
                    println!("[{} § {}]\n{}\n", c.chunk_id.0, heading, c.text);
                }
            }
@@ -1513,6 +1650,8 @@ mod tests {
            created_at: OffsetDateTime::now_utc(),
            conversation_id: None,
            turn_index: None,
+            hops: None,
+            verification: None,
        }
    }

--- a/crates/kebab-cli/src/wire.rs
+++ b/crates/kebab-cli/src/wire.rs
@@ -92,6 +92,14 @@ pub fn wire_search_response(r: &kebab_app::SearchResponse) -> Value {
            map.insert("trace".to_string(), trace_v);
        }
    }
+    // v0.17.0 A5 Step 4b: emit `hint` only when set. Keeps responses
+    // that don't carry a hint backward-compatible with v0 consumers
+    // that don't know the field.
+    if let Some(hint) = &r.hint {
+        if let Value::Object(ref mut map) = v {
+            map.insert("hint".to_string(), Value::String(hint.clone()));
+        }
+    }
    tag_object(v, "search_response.v1")
 }

@@ -239,7 +247,7 @@ mod tests {

    #[test]
    fn ingest_wrapper_tags_schema_version() {
-        use kebab_core::SourceScope;
+        use kebab_core::{SkipExamples, SourceScope};
        let r = IngestReport {
            scope: SourceScope {
                root: std::path::PathBuf::from("/tmp"),
@@ -254,6 +262,13 @@ mod tests {
            errors: 0,
            duration_ms: 0,
            skipped_by_extension: std::collections::BTreeMap::new(),
+            skipped_gitignore: 0,
+            skipped_kebabignore: 0,
+            skipped_builtin_blacklist: 0,
+            skipped_generated: 0,
+            skipped_size_exceeded: 0,
+            skip_examples: SkipExamples::default(),
+            purged_deleted_files: 0,
            items: None,
        };
        let v = wire_ingest(&r);
@@ -285,6 +300,7 @@ mod tests {
            next_cursor: Some("opaque-cursor-abc".to_string()),
            truncated: true,
            trace: None,
+            hint: None,
        };
        let v = wire_search_response(&r);
        assert_eq!(schema_of(&v), Some("search_response.v1"));
@@ -297,7 +313,7 @@ mod tests {
            v.get("next_cursor").and_then(|c| c.as_str()),
            Some("opaque-cursor-abc")
        );
-        assert_eq!(v.get("truncated").and_then(|t| t.as_bool()), Some(true));
+        assert_eq!(v.get("truncated").and_then(serde_json::Value::as_bool), Some(true));
    }

    #[test]
@@ -328,6 +344,8 @@ mod tests {
                lang_breakdown: Default::default(),
                index_bytes: Default::default(),
                stale_doc_count: 0,
+                // p10-1A-1: new fields added to Stats; use Default for the test fixture.
+                ..Default::default()
            },
        };
        let v = wire_schema(&schema);
@@ -356,6 +374,8 @@ mod tests {
            scope: kebab_app::ResetScope::DataOnly,
            removed_paths: vec![std::path::PathBuf::from("/tmp/x")],
            embedding_rows_truncated: 0,
+            orphans_purged: 0,
+            purged_paths: vec![],
        };
        let v = wire_reset(&r);
        assert_eq!(schema_of(&v), Some("reset_report.v1"));
@@ -394,6 +414,7 @@ mod tests {
                }],
                timing: TraceTiming { lexical_ms: 5, vector_ms: 0, fusion_ms: 1, total_ms: 7 },
            }),
+            hint: None,
        };
        let v = wire_search_response(&r);
        assert_eq!(schema_of(&v), Some("search_response.v1"));
@@ -409,6 +430,7 @@ mod tests {
            next_cursor: None,
            truncated: false,
            trace: None,
+            hint: None,
        };
        let v = wire_search_response(&r);
        assert!(v.get("trace").is_none(), "trace field absent when None");
--- a/crates/kebab-cli/tests/cli_ingest_file.rs
+++ b/crates/kebab-cli/tests/cli_ingest_file.rs
@@ -88,5 +88,5 @@ max_context_tokens = 8000
    let stdout = String::from_utf8_lossy(&out.stdout);
    let v: serde_json::Value = serde_json::from_str(stdout.trim()).unwrap();
    assert_eq!(v.get("schema_version").and_then(|s| s.as_str()), Some("ingest_report.v1"));
-    assert_eq!(v.get("new").and_then(|n| n.as_u64()), Some(1));
+    assert_eq!(v.get("new").and_then(serde_json::Value::as_u64), Some(1));
 }
--- a/crates/kebab-cli/tests/cli_ingest_stdin.rs
+++ b/crates/kebab-cli/tests/cli_ingest_stdin.rs
@@ -96,5 +96,5 @@ max_context_tokens = 8000
    let stdout = String::from_utf8_lossy(&out.stdout);
    let v: serde_json::Value = serde_json::from_str(stdout.trim()).unwrap();
    assert_eq!(v.get("schema_version").and_then(|s| s.as_str()), Some("ingest_report.v1"));
-    assert_eq!(v.get("new").and_then(|n| n.as_u64()), Some(1));
+    assert_eq!(v.get("new").and_then(serde_json::Value::as_u64), Some(1));
 }
--- a/crates/kebab-cli/tests/cli_mcp_smoke.rs
+++ b/crates/kebab-cli/tests/cli_mcp_smoke.rs
@@ -43,7 +43,7 @@ fn cli_mcp_initialize_then_tools_list() {
    reader.read_line(&mut line).unwrap();
    let init: serde_json::Value = serde_json::from_str(line.trim()).unwrap();
    assert_eq!(
-        init.get("id").and_then(|i| i.as_i64()),
+        init.get("id").and_then(serde_json::Value::as_i64),
        Some(1),
        "unexpected id in initialize response: {init}"
    );
@@ -57,7 +57,7 @@ fn cli_mcp_initialize_then_tools_list() {
    reader.read_line(&mut line).unwrap();
    let list: serde_json::Value = serde_json::from_str(line.trim()).unwrap();
    assert_eq!(
-        list.get("id").and_then(|i| i.as_i64()),
+        list.get("id").and_then(serde_json::Value::as_i64),
        Some(2),
        "unexpected id in tools/list response: {list}"
    );
--- a/crates/kebab-cli/tests/cli_schema.rs
+++ b/crates/kebab-cli/tests/cli_schema.rs
@@ -76,8 +76,7 @@ fn cli_schema_json_emits_schema_v1() {
    assert!(
        v.get("kebab_version")
            .and_then(|s| s.as_str())
-            .map(|s| !s.is_empty())
-            .unwrap_or(false),
+            .is_some_and(|s| !s.is_empty()),
        "kebab_version must be a non-empty string"
    );

@@ -86,12 +85,12 @@ fn cli_schema_json_emits_schema_v1() {
        .and_then(|c| c.as_object())
        .expect("capabilities must be a JSON object");
    assert_eq!(
-        caps.get("json_mode").and_then(|b| b.as_bool()),
+        caps.get("json_mode").and_then(serde_json::Value::as_bool),
        Some(true),
        "capabilities.json_mode must be true"
    );
    assert_eq!(
-        caps.get("mcp_server").and_then(|b| b.as_bool()),
+        caps.get("mcp_server").and_then(serde_json::Value::as_bool),
        Some(true),
        "capabilities.mcp_server must be true (fb-30)"
    );
--- a/crates/kebab-cli/tests/ingest_progress_cli.rs
+++ b/crates/kebab-cli/tests/ingest_progress_cli.rs
@@ -155,8 +155,8 @@ fn ingest_json_progress_lines_carry_kind_and_ts() {
            saw_completed = true;
            // Counts mirror the report.
            let counts = v.get("counts").unwrap();
-            assert_eq!(counts.get("scanned").and_then(|n| n.as_u64()), Some(2));
-            assert_eq!(counts.get("new").and_then(|n| n.as_u64()), Some(2));
+            assert_eq!(counts.get("scanned").and_then(serde_json::Value::as_u64), Some(2));
+            assert_eq!(counts.get("new").and_then(serde_json::Value::as_u64), Some(2));
        }
    }
    assert!(saw_scan_started, "missing scan_started event");
--- a/crates/kebab-cli/tests/wire_ask_multi_hop.rs
+++ b/crates/kebab-cli/tests/wire_ask_multi_hop.rs
@@ -0,0 +1,254 @@
+//! p9-fb-41 PR-4: CLI `--multi-hop` flag wiring + answer.v1 / error.v1
+//! schema additivity.
+//!
+//! Four Ollama-free pins:
+//!
+//! 1. `--multi-hop` is exposed on `kebab ask --help` so users can
+//!    discover the flag at the CLI surface (clap-level smoke).
+//! 2. `answer.schema.json` parses as valid JSON and declares a
+//!    `hops` property with a `HopRecord` `$defs` entry — guards
+//!    against accidental schema deletion / typo in future edits.
+//! 3. `answer.schema.json`'s `refusal_reason` enum lists
+//!    `multi_hop_decompose_failed` — agents validating against
+//!    the schema accept the new variant on refusal answers.
+//! 4. `error.schema.json`'s `code` enum lists
+//!    `multi_hop_decompose_failed` — forward-looking enum extension
+//!    documented in PR-4.
+//!
+//! End-to-end multi-hop ask against a live Ollama lands in a
+//! follow-up `#[ignore]` test (same pattern as `wire_ask_stale.rs`).
+
+use std::path::PathBuf;
+use std::process::Command;
+
+fn schema_path(name: &str) -> PathBuf {
+    PathBuf::from(env!("CARGO_MANIFEST_DIR"))
+        .join("..")
+        .join("..")
+        .join("docs")
+        .join("wire-schema")
+        .join("v1")
+        .join(name)
+}
+
+fn parse_schema(name: &str) -> serde_json::Value {
+    let text = std::fs::read_to_string(schema_path(name))
+        .unwrap_or_else(|e| panic!("read {name}: {e}"));
+    serde_json::from_str(&text)
+        .unwrap_or_else(|e| panic!("{name} must parse as valid JSON: {e}"))
+}
+
+#[test]
+fn cli_ask_help_advertises_multi_hop_flag() {
+    let bin = env!("CARGO_BIN_EXE_kebab");
+    let out = Command::new(bin).args(["ask", "--help"]).output().unwrap();
+    let stdout = String::from_utf8_lossy(&out.stdout);
+    assert!(
+        stdout.contains("--multi-hop"),
+        "`kebab ask --help` must advertise --multi-hop so users can discover it:\n{stdout}"
+    );
+}
+
+#[test]
+fn answer_schema_declares_hops_property_with_hop_record_defs() {
+    let schema = parse_schema("answer.schema.json");
+    assert!(
+        schema["properties"]["hops"].is_object(),
+        "`hops` property must be declared on answer.v1"
+    );
+    // `hops` allows array-or-null (single-pass omits the field;
+    // multi-hop emits a non-empty array).
+    let hops_any_of = schema["properties"]["hops"]["anyOf"]
+        .as_array()
+        .expect("hops must declare anyOf (array | null)");
+    assert!(
+        hops_any_of.iter().any(|v| v["type"] == "array"),
+        "hops anyOf must include array shape"
+    );
+    assert!(
+        hops_any_of.iter().any(|v| v["type"] == "null"),
+        "hops anyOf must include null (single-pass omits the field)"
+    );
+
+    // HopRecord $defs entry — guards against accidental deletion or
+    // structural drift in future schema edits.
+    let hop_record = &schema["$defs"]["HopRecord"];
+    assert!(
+        hop_record.is_object(),
+        "$defs.HopRecord must be declared so `hops.items` can $ref it"
+    );
+    let kind_enum = hop_record["properties"]["kind"]["enum"]
+        .as_array()
+        .expect("HopRecord.kind must be an enum");
+    let kinds: Vec<&str> = kind_enum.iter().filter_map(|v| v.as_str()).collect();
+    for needed in ["decompose", "decide", "synthesize"] {
+        assert!(
+            kinds.contains(&needed),
+            "HopRecord.kind enum must include {needed:?}, got {kinds:?}"
+        );
+    }
+}
+
+#[test]
+fn answer_schema_refusal_reason_enum_includes_multi_hop_decompose_failed() {
+    let schema = parse_schema("answer.schema.json");
+    let refusal_any_of = schema["properties"]["refusal_reason"]["anyOf"]
+        .as_array()
+        .expect("refusal_reason must declare anyOf");
+    let enum_arr = refusal_any_of
+        .iter()
+        .find_map(|v| v["enum"].as_array())
+        .expect("one of refusal_reason.anyOf entries must declare an enum");
+    let values: Vec<&str> = enum_arr.iter().filter_map(|v| v.as_str()).collect();
+    assert!(
+        values.contains(&"multi_hop_decompose_failed"),
+        "refusal_reason enum must include `multi_hop_decompose_failed`, got {values:?}"
+    );
+    // All earlier RefusalReason wire values remain on the enum —
+    // guards against an accidental rewrite dropping old variants.
+    for needed in [
+        "score_gate",
+        "llm_self_judge",
+        "no_index",
+        "no_chunks",
+        "llm_stream_aborted",
+    ] {
+        assert!(
+            values.contains(&needed),
+            "refusal_reason enum must keep prior variant {needed:?}, got {values:?}"
+        );
+    }
+}
+
+#[test]
+fn error_schema_code_enum_includes_multi_hop_decompose_failed() {
+    let schema = parse_schema("error.schema.json");
+    let code_enum = schema["properties"]["code"]["enum"]
+        .as_array()
+        .expect("error.v1 must declare code.enum");
+    let values: Vec<&str> = code_enum.iter().filter_map(|v| v.as_str()).collect();
+    assert!(
+        values.contains(&"multi_hop_decompose_failed"),
+        "error.v1 code enum must include forward-looking `multi_hop_decompose_failed`, got {values:?}"
+    );
+    // Existing codes remain — guards against accidental deletion.
+    for needed in [
+        "config_invalid",
+        "not_indexed",
+        "model_unreachable",
+        "generic",
+    ] {
+        assert!(
+            values.contains(&needed),
+            "error.v1 code enum must keep prior code {needed:?}, got {values:?}"
+        );
+    }
+}
+
+// ── p9-fb-41 PR-9c-1: NLI verification surface pins ─────────────────────
+
+/// answer.v1 must declare a `verification` property AND a
+/// `$defs.VerificationSummary` entry with all three required fields.
+/// Guards against accidental schema deletion / typo in future edits.
+#[test]
+fn answer_schema_declares_verification_field_and_defs() {
+    let schema = parse_schema("answer.schema.json");
+    assert!(
+        schema["properties"]["verification"].is_object(),
+        "`verification` property must be declared on answer.v1"
+    );
+    // `verification` allows object-or-null (multi-hop with threshold>0
+    // emits an object; everything else omits the field).
+    let v_any_of = schema["properties"]["verification"]["anyOf"]
+        .as_array()
+        .expect("verification must declare anyOf (object | null)");
+    assert!(
+        v_any_of.iter().any(|v| v["type"] == "null"),
+        "verification anyOf must include null (single-pass / disabled gate omits the field)"
+    );
+    assert!(
+        v_any_of
+            .iter()
+            .any(|v| v["$ref"].as_str() == Some("#/$defs/VerificationSummary")),
+        "verification anyOf must $ref VerificationSummary"
+    );
+
+    // VerificationSummary $defs entry + required fields.
+    let vs = &schema["$defs"]["VerificationSummary"];
+    assert!(
+        vs.is_object(),
+        "$defs.VerificationSummary must be declared so verification.anyOf can $ref it"
+    );
+    let required: Vec<&str> = vs["required"]
+        .as_array()
+        .expect("VerificationSummary.required must be an array")
+        .iter()
+        .filter_map(|v| v.as_str())
+        .collect();
+    for needed in ["nli_score", "nli_threshold", "nli_passed"] {
+        assert!(
+            required.contains(&needed),
+            "VerificationSummary.required must include {needed:?}, got {required:?}"
+        );
+    }
+}
+
+#[test]
+fn answer_schema_refusal_reason_enum_includes_nli_verification_failed() {
+    let schema = parse_schema("answer.schema.json");
+    let refusal_any_of = schema["properties"]["refusal_reason"]["anyOf"]
+        .as_array()
+        .expect("refusal_reason must declare anyOf");
+    let enum_arr = refusal_any_of
+        .iter()
+        .find_map(|v| v["enum"].as_array())
+        .expect("one of refusal_reason.anyOf entries must declare an enum");
+    let values: Vec<&str> = enum_arr.iter().filter_map(|v| v.as_str()).collect();
+    assert!(
+        values.contains(&"nli_verification_failed"),
+        "refusal_reason enum must include `nli_verification_failed`, got {values:?}"
+    );
+}
+
+#[test]
+fn answer_schema_refusal_reason_enum_includes_nli_model_unavailable() {
+    let schema = parse_schema("answer.schema.json");
+    let refusal_any_of = schema["properties"]["refusal_reason"]["anyOf"]
+        .as_array()
+        .expect("refusal_reason must declare anyOf");
+    let enum_arr = refusal_any_of
+        .iter()
+        .find_map(|v| v["enum"].as_array())
+        .expect("one of refusal_reason.anyOf entries must declare an enum");
+    let values: Vec<&str> = enum_arr.iter().filter_map(|v| v.as_str()).collect();
+    assert!(
+        values.contains(&"nli_model_unavailable"),
+        "refusal_reason enum must include `nli_model_unavailable`, got {values:?}"
+    );
+}
+
+#[test]
+fn error_schema_code_enum_includes_nli_verification_failed() {
+    let schema = parse_schema("error.schema.json");
+    let code_enum = schema["properties"]["code"]["enum"]
+        .as_array()
+        .expect("error.v1 must declare code.enum");
+    let values: Vec<&str> = code_enum.iter().filter_map(|v| v.as_str()).collect();
+    assert!(
+        values.contains(&"nli_verification_failed"),
+        "error.v1 code enum must include forward-looking `nli_verification_failed`, got {values:?}"
+    );
+}
+
+#[test]
+fn error_schema_code_enum_includes_nli_model_unavailable() {
+    let schema = parse_schema("error.schema.json");
+    let code_enum = schema["properties"]["code"]["enum"]
+        .as_array()
+        .expect("error.v1 must declare code.enum");
+    let values: Vec<&str> = code_enum.iter().filter_map(|v| v.as_str()).collect();
+    assert!(
+        values.contains(&"nli_model_unavailable"),
+        "error.v1 code enum must include forward-looking `nli_model_unavailable`, got {values:?}"
+    );
+}
--- a/crates/kebab-cli/tests/wire_citation_5_variants_unchanged.rs
+++ b/crates/kebab-cli/tests/wire_citation_5_variants_unchanged.rs
@@ -0,0 +1,100 @@
+//! p10-1A-1 Task 13: regression — the 5 original Citation variants
+//! (Line, Page, Region, Caption, Time) serialize byte-identically to
+//! pre-Task-1 form.  No spurious `code`, `line_start`, or `symbol` keys
+//! must leak into these variants.
+
+use kebab_core::{Citation, WorkspacePath};
+
+#[test]
+fn line_variant_serialization_unchanged() {
+    let c = Citation::Line {
+        path: WorkspacePath::new("a.md".into()).unwrap(),
+        start: 1,
+        end: 2,
+        section: Some("§14".into()),
+    };
+    let v = serde_json::to_value(&c).unwrap();
+    assert_eq!(v["kind"], "line");
+    assert_eq!(v["start"], 1);
+    assert_eq!(v["end"], 2);
+    assert_eq!(v["section"], "§14");
+    // Must not bleed Code-variant keys.
+    assert!(v.get("line_start").is_none(), "line_start must be absent: {v}");
+    assert!(v.get("symbol").is_none(), "symbol must be absent: {v}");
+    assert!(v.get("code").is_none(), "code must be absent: {v}");
+}
+
+#[test]
+fn line_variant_null_section_omitted() {
+    let c = Citation::Line {
+        path: WorkspacePath::new("b.md".into()).unwrap(),
+        start: 5,
+        end: 10,
+        section: None,
+    };
+    let v = serde_json::to_value(&c).unwrap();
+    assert_eq!(v["kind"], "line");
+    // `section` with None should be omitted (skip_serializing_if = is_none).
+    assert!(v.get("section").is_none() || v["section"].is_null());
+}
+
+#[test]
+fn page_variant_serialization_unchanged() {
+    let c = Citation::Page {
+        path: WorkspacePath::new("a.pdf".into()).unwrap(),
+        page: 13,
+        section: None,
+    };
+    let v = serde_json::to_value(&c).unwrap();
+    assert_eq!(v["kind"], "page");
+    assert_eq!(v["page"], 13);
+    assert!(v.get("line_start").is_none(), "line_start must be absent: {v}");
+    assert!(v.get("symbol").is_none(), "symbol must be absent: {v}");
+}
+
+#[test]
+fn region_variant_serialization_unchanged() {
+    let c = Citation::Region {
+        path: WorkspacePath::new("img.png".into()).unwrap(),
+        x: 10,
+        y: 20,
+        w: 100,
+        h: 200,
+    };
+    let v = serde_json::to_value(&c).unwrap();
+    assert_eq!(v["kind"], "region");
+    assert_eq!(v["x"], 10);
+    assert_eq!(v["y"], 20);
+    assert_eq!(v["w"], 100);
+    assert_eq!(v["h"], 200);
+    assert!(v.get("line_start").is_none(), "line_start must be absent: {v}");
+}
+
+#[test]
+fn caption_variant_serialization_unchanged() {
+    let c = Citation::Caption {
+        path: WorkspacePath::new("a.png".into()).unwrap(),
+        model: "qwen2.5-vl:7b".into(),
+    };
+    let v = serde_json::to_value(&c).unwrap();
+    assert_eq!(v["kind"], "caption");
+    assert_eq!(v["model"], "qwen2.5-vl:7b");
+    assert!(v.get("line_start").is_none(), "line_start must be absent: {v}");
+}
+
+#[test]
+fn time_variant_serialization_unchanged() {
+    let c = Citation::Time {
+        path: WorkspacePath::new("audio.mp3".into()).unwrap(),
+        start_ms: 1000,
+        end_ms: 5000,
+        speaker: Some("Alice".into()),
+    };
+    let v = serde_json::to_value(&c).unwrap();
+    assert_eq!(v["kind"], "time");
+    assert_eq!(v["start_ms"], 1000);
+    assert_eq!(v["end_ms"], 5000);
+    assert_eq!(v["speaker"], "Alice");
+    assert!(v.get("line_start").is_none(), "line_start must be absent: {v}");
+    assert!(v.get("symbol").is_none(), "symbol must be absent: {v}");
+}
--- a/crates/kebab-cli/tests/wire_search_filters_code.rs
+++ b/crates/kebab-cli/tests/wire_search_filters_code.rs
@@ -0,0 +1,72 @@
+//! p10-1A-1 Task 15: CLI accepts --repo and --code-lang flags.
+//!
+//! These tests verify that clap parses the new flags without error.
+//! They drive `kebab search --help` (which exercises flag parsing
+//! via clap's help generation path, exiting 0) or use a minimal
+//! config + `--json` round-trip to verify the flags reach the wire.
+
+use std::process::Command;
+
+fn kebab() -> Command {
+    Command::new(env!("CARGO_BIN_EXE_kebab"))
+}
+
+/// `kebab search --help` must exit 0 and mention `--repo`.
+#[test]
+fn cli_search_help_mentions_repo_flag() {
+    let out = kebab()
+        .args(["search", "--help"])
+        .output()
+        .expect("failed to run kebab");
+    // clap help exits 0.
+    assert!(
+        out.status.success(),
+        "kebab search --help exited non-zero: {:?}",
+        out.status
+    );
+    let stdout = String::from_utf8_lossy(&out.stdout);
+    assert!(
+        stdout.contains("--repo"),
+        "--repo flag must appear in search help output:\n{stdout}"
+    );
+}
+
+/// `kebab search --help` must exit 0 and mention `--code-lang`.
+#[test]
+fn cli_search_help_mentions_code_lang_flag() {
+    let out = kebab()
+        .args(["search", "--help"])
+        .output()
+        .expect("failed to run kebab");
+    assert!(
+        out.status.success(),
+        "kebab search --help exited non-zero: {:?}",
+        out.status
+    );
+    let stdout = String::from_utf8_lossy(&out.stdout);
+    assert!(
+        stdout.contains("--code-lang"),
+        "--code-lang flag must appear in search help output:\n{stdout}"
+    );
+}
+
+/// `kebab search --help` must exit 0 and mention `--media`.
+/// Confirms `--media code` value pathway is available (media is
+/// a free-form Vec<String> that already accepted arbitrary values).
+#[test]
+fn cli_search_help_mentions_media_flag() {
+    let out = kebab()
+        .args(["search", "--help"])
+        .output()
+        .expect("failed to run kebab");
+    assert!(
+        out.status.success(),
+        "kebab search --help exited non-zero: {:?}",
+        out.status
+    );
+    let stdout = String::from_utf8_lossy(&out.stdout);
+    assert!(
+        stdout.contains("--media"),
+        "--media flag must appear in search help output:\n{stdout}"
+    );
+}
--- a/crates/kebab-cli/tests/wire_search_hit_no_code_fields.rs
+++ b/crates/kebab-cli/tests/wire_search_hit_no_code_fields.rs
@@ -0,0 +1,47 @@
+//! p10-1A-1 Task 13: regression — markdown SearchHit omits `repo` and
+//! `code_lang` from JSON when both are `None`.
+//!
+//! Proves that adding optional fields to SearchHit does not silently
+//! inject spurious keys into the existing markdown corpus wire shape.
+
+use kebab_core::{
+    Citation, ChunkId, ChunkerVersion, DocumentId, IndexVersion, RetrievalDetail, ScoreKind,
+    SearchHit, WorkspacePath,
+};
+
+#[test]
+fn markdown_hit_omits_repo_and_code_lang() {
+    let hit = SearchHit {
+        rank: 1,
+        chunk_id: ChunkId("c1".into()),
+        doc_id: DocumentId("d1".into()),
+        doc_path: WorkspacePath::new("notes/foo.md".into()).unwrap(),
+        heading_path: vec!["A".into(), "B".into()],
+        section_label: Some("B".into()),
+        snippet: "hi".into(),
+        citation: Citation::Line {
+            path: WorkspacePath::new("notes/foo.md".into()).unwrap(),
+            start: 1,
+            end: 2,
+            section: None,
+        },
+        retrieval: RetrievalDetail::default(),
+        index_version: IndexVersion("v1".into()),
+        embedding_model: None,
+        chunker_version: ChunkerVersion("md-heading-v1".into()),
+        indexed_at: time::OffsetDateTime::UNIX_EPOCH,
+        stale: false,
+        score_kind: ScoreKind::Rrf,
+        repo: None,
+        code_lang: None,
+    };
+    let s = serde_json::to_string(&hit).unwrap();
+    assert!(
+        !s.contains("\"repo\""),
+        "repo should be absent from markdown hit JSON: {s}"
+    );
+    assert!(
+        !s.contains("\"code_lang\""),
+        "code_lang should be absent from markdown hit JSON: {s}"
+    );
+}
--- a/crates/kebab-cli/tests/wire_search_response.rs
+++ b/crates/kebab-cli/tests/wire_search_response.rs
@@ -47,8 +47,20 @@ fn search_json_emits_search_response_v1_wrapper() {
 fn search_json_truncates_with_max_tokens() {
    let dir = tempfile::tempdir().unwrap();
    let (cfg, workspace, _data) = common::write_config(dir.path(), 30);
-    let body: String = "rust ownership is a memory model. ".repeat(10);
-    fs::write(workspace.join("a.md"), format!("# T\n\n{body}\n")).unwrap();
+    // v0.17.0 trigram tokenizer makes FTS5 snippet() tokens 3-char wide
+    // (was full words under unicode61), so an individual snippet stays
+    // around ~60 chars — too short to ever exceed the snippet-shorten
+    // budget cap on a single-hit fixture. To still exercise the budget
+    // loop deterministically, we ingest multiple hits and pick a budget
+    // small enough that the loop has to *pop* hits, which flips
+    // truncated=true regardless of snippet length.
+    for i in 0..5 {
+        fs::write(
+            workspace.join(format!("d{i}.md")),
+            format!("# T{i}\n\nrust ownership is a memory model.\n"),
+        )
+        .unwrap();
+    }
    common::ingest(&cfg, &workspace);

    let (stdout, _stderr) = common::run_search_with_args(
@@ -211,8 +223,15 @@ fn search_stale_cursor_returns_error_v1_with_stale_cursor_code() {
 fn search_plain_emits_truncated_hint_to_stderr() {
    let dir = tempfile::tempdir().unwrap();
    let (cfg, workspace, _data) = common::write_config(dir.path(), 30);
-    let body: String = "rust ownership is a memory model. ".repeat(10);
-    fs::write(workspace.join("a.md"), format!("# T\n\n{body}\n")).unwrap();
+    // v0.17.0 trigram tokenizer — same multi-doc rationale as
+    // `search_json_truncates_with_max_tokens` above.
+    for i in 0..5 {
+        fs::write(
+            workspace.join(format!("d{i}.md")),
+            format!("# T{i}\n\nrust ownership is a memory model.\n"),
+        )
+        .unwrap();
+    }
    common::ingest(&cfg, &workspace);

    let (_stdout, stderr) = common::run_search_with_args(
@@ -224,3 +243,76 @@ fn search_plain_emits_truncated_hint_to_stderr() {
        "stderr must carry truncated hint: {stderr:?}"
    );
 }
+
+#[test]
+fn search_plain_emits_short_query_hint_to_stderr() {
+    // v0.17.0 A5 Step 6: 2-char query under trigram tokenizer emits
+    // empty hits + stderr `[hint]` advisory. Empty workspace is enough
+    // — hits are always empty so the hint condition depends only on
+    // query length (<3 chars trimmed) + non-raw mode + hits.is_empty.
+    let dir = tempfile::tempdir().unwrap();
+    let (cfg, workspace, _data) = common::write_config(dir.path(), 30);
+    common::ingest(&cfg, &workspace);
+
+    let (_stdout, stderr) = common::run_search_with_args(
+        &cfg,
+        &["--mode", "lexical", "ab"],
+    );
+    assert!(
+        stderr.contains("[hint]"),
+        "stderr must carry short-query hint: {stderr:?}"
+    );
+    assert!(
+        stderr.contains("3자 이상"),
+        "hint message must mention '3자 이상' (Korean advisory): {stderr:?}"
+    );
+}
+
+#[test]
+fn search_json_emits_hint_field_for_short_query() {
+    // v0.17.0 A5 Step 6: --json mode carries the same advisory on the
+    // `search_response.v1.hint` additive field. Empty hits + 2-char
+    // query + non-raw mode trips the helper. Verifies the MCP-visible
+    // surface (agents read the field instead of parsing stderr).
+    let dir = tempfile::tempdir().unwrap();
+    let (cfg, workspace, _data) = common::write_config(dir.path(), 30);
+    common::ingest(&cfg, &workspace);
+
+    let (stdout, _stderr) = common::run_search_with_args(
+        &cfg,
+        &["--json", "--mode", "lexical", "ab"],
+    );
+    let v: Value = serde_json::from_str(stdout.trim())
+        .unwrap_or_else(|e| panic!("not JSON: {stdout:?}: {e}"));
+    assert!(
+        v["hits"].as_array().unwrap().is_empty(),
+        "empty hits expected for short query in empty KB: {v}"
+    );
+    assert_eq!(
+        v["hint"].as_str().expect("hint field set on short empty result"),
+        "3자 이상 키워드 권장 (trigram tokenizer 제약)",
+        "hint must carry the standard advisory: {v}"
+    );
+}
+
+#[test]
+fn search_json_omits_hint_field_when_query_is_long_enough() {
+    // v0.17.0 A5 Step 6 (negative case): 3+ char query never trips
+    // hint, even on an empty KB. Verifies `serialize_search_response`
+    // omits the additive `hint` field when `None` so existing wire
+    // consumers stay backward-compatible.
+    let dir = tempfile::tempdir().unwrap();
+    let (cfg, workspace, _data) = common::write_config(dir.path(), 30);
+    common::ingest(&cfg, &workspace);
+
+    let (stdout, _stderr) = common::run_search_with_args(
+        &cfg,
+        &["--json", "--mode", "lexical", "abc"],
+    );
+    let v: Value = serde_json::from_str(stdout.trim())
+        .unwrap_or_else(|e| panic!("not JSON: {stdout:?}: {e}"));
+    assert!(
+        v.get("hint").is_none(),
+        "hint must be absent for ≥3-char queries: {v}"
+    );
+}
--- a/crates/kebab-config/Cargo.toml
+++ b/crates/kebab-config/Cargo.toml
@@ -22,3 +22,6 @@ tracing      = { workspace = true }

 [dev-dependencies]
 tempfile     = { workspace = true }
+
+[lints]
+workspace = true
--- a/crates/kebab-config/src/lib.rs
+++ b/crates/kebab-config/src/lib.rs
@@ -45,6 +45,11 @@ pub struct Config {
    /// `dark`).
    #[serde(default = "UiCfg::defaults")]
    pub ui: UiCfg,
+    /// p10-1A-1: code ingest settings. `#[serde(default)]` so existing
+    /// config files without an `[ingest]` / `[ingest.code]` section
+    /// load cleanly with built-in defaults.
+    #[serde(default)]
+    pub ingest: IngestCfg,
    /// p9-fb-05: directory of the on-disk config file this `Config`
    /// was loaded from, if any. Populated by `Config::from_file` /
    /// `Config::load` — never serialized (`#[serde(skip)]`). Used by
@@ -98,6 +103,34 @@ pub struct ChunkingCfg {
 pub struct ModelsCfg {
    pub embedding: EmbeddingModelCfg,
    pub llm: LlmCfg,
+    /// p9-fb-41 PR-9c-1: NLI verifier model + provider knob.
+    /// `#[serde(default)]` so pre-v0.18 config files that predate the
+    /// `[models.nli]` section still load with built-in defaults
+    /// (`Xenova/mDeBERTa-v3-base-xnli-multilingual-nli-2mil7` / `onnx`).
+    /// The verifier itself is gated by `[rag].nli_threshold` — even
+    /// with a model configured here, threshold `0.0` (the default)
+    /// skips the verification step entirely.
+    #[serde(default = "NliCfg::defaults")]
+    pub nli: NliCfg,
+}
+
+/// p9-fb-41 PR-9c-1: NLI verifier configuration. The model id flows to
+/// `OnnxNliVerifier::new` via `kebab-nli` (PR-9c-2 wiring); the provider
+/// is reserved for future verifier swap-in (currently only `"onnx"` is
+/// recognized — anything else falls back to the same path).
+#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
+pub struct NliCfg {
+    pub model: String,
+    pub provider: String,
+}
+
+impl NliCfg {
+    pub fn defaults() -> Self {
+        Self {
+            model: "Xenova/mDeBERTa-v3-base-xnli-multilingual-nli-2mil7".to_string(),
+            provider: "onnx".to_string(),
+        }
+    }
 }

 #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
@@ -117,6 +150,23 @@ pub struct LlmCfg {
    pub endpoint: String,
    pub temperature: f32,
    pub seed: u64,
+    /// v0.17.0 post-dogfood: Hard ceiling on a single HTTP exchange to
+    /// the LLM endpoint (Ollama, etc.). Cold-loading an 8B+ model on
+    /// CPU-only hosts can spend 60-90s on model load + several minutes
+    /// on a first inference, blowing past the old hard-coded 300s cap
+    /// and surfacing as `error: kb-rag: llm.generate_stream` to the
+    /// user. Config-driven so 16-GB / CPU-only deployments using small
+    /// (≤4B) models can keep the original 300s and large-model dogfood
+    /// can dial it up (e.g. 1200s) without rebuilding.
+    ///
+    /// **Edge case — `0` is NOT a disable sentinel.**
+    /// `reqwest::ClientBuilder::timeout(Duration::from_secs(0))` sets a
+    /// 0-second read timeout, so every request fails *immediately* with
+    /// `error: kb-rag: ollama timeout`. To approximate "no cap", use a
+    /// large finite value (e.g. `u64::MAX` ≈ 5.8 × 10¹¹ years, or
+    /// just a generous number like `86400`).
+    #[serde(default = "default_llm_request_timeout_secs")]
+    pub request_timeout_secs: u64,
 }

 #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
@@ -142,6 +192,13 @@ fn default_cache_capacity() -> usize {
    256
 }

+/// v0.17.0 post-dogfood: matches the legacy hard-coded ceiling so
+/// existing configs that omit the field keep behaving identically.
+/// Overridable per config / `KEBAB_MODELS_LLM_REQUEST_TIMEOUT_SECS`.
+fn default_llm_request_timeout_secs() -> u64 {
+    300
+}
+
 fn default_stale_threshold_days() -> u32 {
    30
 }
@@ -152,6 +209,73 @@ pub struct RagCfg {
    pub score_gate: f32,
    pub explain_default: bool,
    pub max_context_tokens: usize,
+    /// p9-fb-41: hard ceiling on the number of multi-hop iterations
+    /// (decompose iter + decide iters). When the LLM keeps returning
+    /// `continue` past this depth the pipeline cuts to `synthesize`
+    /// with `HopRecord.forced_stop = true`. Default `3` — enough for
+    /// most cross-doc reasoning, low enough to bound LLM cost.
+    #[serde(default = "default_multi_hop_max_depth")]
+    pub multi_hop_max_depth: u32,
+    /// p9-fb-41: cap on how many sub-queries the LLM may emit in a
+    /// single decompose / decide call. This is the *prompt-side
+    /// soft hint* — the value the pipeline injects into the
+    /// decompose / decide prompts so the LLM knows what to aim for.
+    /// kebab-rag enforces a separate compile-time hard ceiling
+    /// (`MULTI_HOP_MAX_SUB_QUERIES_HARD_CAP`, currently 10) as a
+    /// safety net against misbehaving models — if you raise this
+    /// knob above the hard cap, bump the const in the same PR.
+    /// Default `5`.
+    #[serde(default = "default_multi_hop_max_sub_queries_per_iter")]
+    pub multi_hop_max_sub_queries_per_iter: u32,
+    /// p9-fb-41: hard ceiling on the deduped chunk pool. When the
+    /// accumulated pool would exceed this many chunks the pipeline
+    /// stops accepting new retrieval results and forces synthesize
+    /// with `forced_stop = true`.
+    ///
+    /// Default `15` — tuned down from the original 30 in the v0.18
+    /// pre-cut dogfood (`tasks/HOTFIXES.md` 2026-05-25 fb-41 entry,
+    /// "post-PR-7 dogfood retest + PR-8 partial mitigation" sub-section).
+    /// With 30 chunks the synthesize prompt was large enough for
+    /// gemma3:4b to lose the citation rule + drift into unrelated
+    /// chunks; 15 keeps the prompt tight while still allowing 3-iter
+    /// cross-doc reasoning over ~5 chunks per iter.
+    #[serde(default = "default_multi_hop_max_pool_chunks")]
+    pub multi_hop_max_pool_chunks: u32,
+    /// p9-fb-41 PR-9c-1: minimum NLI entailment score required for the
+    /// multi-hop synthesize answer to be returned as `grounded=true`
+    /// (spec §2.6 single gate). When the post-synthesize NLI verifier
+    /// returns `NliScores::faithfulness() < nli_threshold` the
+    /// pipeline refuses with `RefusalReason::NliVerificationFailed`.
+    ///
+    /// Default `0.0` = verification disabled — no NLI call, multi-hop
+    /// matches its PR-3b behavior exactly. Set to e.g. `0.5` to
+    /// activate the gate. Knob lives on `[rag]` (the gate is a RAG
+    /// policy, not a model property); the model itself comes from
+    /// `[models.nli].model`.
+    ///
+    /// Single-pass `ask` ignores this knob entirely — only multi-hop
+    /// runs through the verification step (PR-9c-2 wires it).
+    #[serde(default = "default_nli_threshold")]
+    pub nli_threshold: f32,
+}
+
+fn default_multi_hop_max_depth() -> u32 {
+    3
+}
+
+fn default_multi_hop_max_sub_queries_per_iter() -> u32 {
+    5
+}
+
+fn default_multi_hop_max_pool_chunks() -> u32 {
+    15
+}
+
+/// p9-fb-41 PR-9c-1: NLI gate disabled by default per spec §2.6
+/// (verification opt-in — users explicitly raise the threshold once
+/// they're ready to trade refusal-rate for groundedness).
+fn default_nli_threshold() -> f32 {
+    0.0
 }

 /// Settings for the image ingest pipeline (P6). `ocr` controls OCR
@@ -199,6 +323,22 @@ pub struct OcrCfg {
    /// Cap the long edge of the image (in pixels) before sending. Larger
    /// images bloat prompt cost. Default `1600`.
    pub max_pixels: u32,
+    /// v0.17.2 post-dogfood: Hard ceiling on a single HTTP exchange to
+    /// the OCR endpoint. Sister knob to [`LlmCfg::request_timeout_secs`]
+    /// — kept separate because OCR latency is typically shorter than
+    /// chat-LLM cold start, and large vision models on CPU-only hosts
+    /// occasionally need a different budget. See HOTFIXES 2026-05-25
+    /// for the rationale.
+    ///
+    /// **Edge case — `0` is NOT a disable sentinel.** Same semantics as
+    /// [`LlmCfg::request_timeout_secs`]: `Duration::from_secs(0)` means
+    /// "every request fails immediately" (reqwest 0.12.x — the read
+    /// timeout is applied as a 0-second deadline), not "no timeout".
+    /// To approximate "no cap", use a large finite value (e.g.
+    /// `u64::MAX` ≈ 5.8 × 10¹¹ years, or just a generous number like
+    /// `86400`).
+    #[serde(default = "default_ocr_request_timeout_secs")]
+    pub request_timeout_secs: u64,
 }

 impl OcrCfg {
@@ -210,10 +350,18 @@ impl OcrCfg {
            endpoint: None,
            languages: vec!["eng".to_string(), "kor".to_string()],
            max_pixels: 1600,
+            request_timeout_secs: default_ocr_request_timeout_secs(),
        }
    }
 }

+/// v0.17.2 post-dogfood: matches the legacy hard-coded ceiling so
+/// existing configs that omit the field keep behaving identically.
+/// Overridable per config / `KEBAB_IMAGE_OCR_REQUEST_TIMEOUT_SECS`.
+fn default_ocr_request_timeout_secs() -> u64 {
+    300
+}
+
 /// Caption settings (P6-3). Caption uses the same Ollama-vision /
 /// `LanguageModel` pipeline as the rest of the workspace; the trait
 /// abstraction is the part the spec demands. `enabled` defaults to
@@ -265,6 +413,52 @@ impl UiCfg {
    }
 }

+/// p10-1A-1: top-level ingest configuration wrapper. Contains per-media-type
+/// sub-sections; currently only `code` is defined.
+#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
+#[serde(default)]
+pub struct IngestCfg {
+    pub code: IngestCodeCfg,
+}
+
+/// p10-1A-1: settings for the code ingest pipeline. All fields have
+/// reasonable defaults so the user need not set anything in `config.toml`
+/// to get working code ingest.
+#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
+#[serde(default)]
+pub struct IngestCodeCfg {
+    /// Generated header sniff. Reads first ~512 bytes, checks 7 markers.
+    pub skip_generated_header: bool,
+    /// Max byte size per file. Bigger files skipped.
+    pub max_file_bytes: u64,
+    /// Max line count per file. Bigger files skipped (byte cap checked first).
+    pub max_file_lines: u32,
+    /// User extra skip globs (gitignore syntax). Applied on top of built-in
+    /// + `.gitignore` + `.kebabignore`.
+    pub extra_skip_globs: Vec<String>,
+    /// AST chunk size cap. Functions/classes longer than this fall back to
+    /// paragraph-based split (1A-2 and later).
+    pub ast_chunk_max_lines: u32,
+    /// Tier 3 fallback chunker: lines per chunk.
+    pub fallback_lines_per_chunk: u32,
+    /// Tier 3 fallback chunker: line overlap between adjacent chunks.
+    pub fallback_lines_overlap: u32,
+}
+
+impl Default for IngestCodeCfg {
+    fn default() -> Self {
+        Self {
+            skip_generated_header: true,
+            max_file_bytes: 262_144,
+            max_file_lines: 5_000,
+            extra_skip_globs: vec![],
+            ast_chunk_max_lines: 200,
+            fallback_lines_per_chunk: 80,
+            fallback_lines_overlap: 20,
+        }
+    }
+}
+
 impl Config {
    /// Defaults per design §6.4.
    pub fn defaults() -> Self {
@@ -312,13 +506,16 @@ impl Config {
                    // gemma4 계열 통일 — OCR (P6-2) + caption (P6-3)
                    // 어댑터가 같은 family 사용. 사용자가 더 큰
                    // variant (gemma4:26b 등) 원하면 자기 config.toml
-                    // 에서 override.
+                    // 에서 override. CPU-only / ≤16 GB RAM 환경이면
+                    // gemma3:4b 같은 ≤4B Q4 모델 권장 (README 참조).
                    model: "gemma4:e4b".to_string(),
                    context_tokens: 32768,
                    endpoint: "http://127.0.0.1:11434".to_string(),
                    temperature: 0.0,
                    seed: 0,
+                    request_timeout_secs: default_llm_request_timeout_secs(),
                },
+                nli: NliCfg::defaults(),
            },
            search: SearchCfg {
                default_k: 10,
@@ -333,9 +530,15 @@ impl Config {
                score_gate: 0.30,
                explain_default: false,
                max_context_tokens: 8000,
+                multi_hop_max_depth: default_multi_hop_max_depth(),
+                multi_hop_max_sub_queries_per_iter:
+                    default_multi_hop_max_sub_queries_per_iter(),
+                multi_hop_max_pool_chunks: default_multi_hop_max_pool_chunks(),
+                nli_threshold: default_nli_threshold(),
            },
            image: ImageCfg::defaults(),
            ui: UiCfg::defaults(),
+            ingest: IngestCfg::default(),
            // p9-fb-05: defaults are not loaded from disk, so no
            // source_dir. Relative `workspace.root` (rare with
            // defaults) falls back to caller `cwd` via the
@@ -569,6 +772,15 @@ impl Config {
                        self.models.llm.seed = n;
                    }
                }
+                "KEBAB_MODELS_LLM_REQUEST_TIMEOUT_SECS" => {
+                    if let Ok(n) = v.parse::<u64>() {
+                        self.models.llm.request_timeout_secs = n;
+                    }
+                }
+
+                // models.nli (p9-fb-41 PR-9c-1)
+                "KEBAB_MODELS_NLI_MODEL" => self.models.nli.model = v.clone(),
+                "KEBAB_MODELS_NLI_PROVIDER" => self.models.nli.provider = v.clone(),

                // search
                "KEBAB_SEARCH_DEFAULT_K" => {
@@ -610,6 +822,39 @@ impl Config {
                        self.rag.max_context_tokens = n;
                    }
                }
+                "KEBAB_RAG_MULTI_HOP_MAX_DEPTH" => {
+                    if let Ok(n) = v.parse::<u32>() {
+                        self.rag.multi_hop_max_depth = n;
+                    }
+                }
+                "KEBAB_RAG_MULTI_HOP_MAX_SUB_QUERIES_PER_ITER" => {
+                    if let Ok(n) = v.parse::<u32>() {
+                        self.rag.multi_hop_max_sub_queries_per_iter = n;
+                    }
+                }
+                "KEBAB_RAG_MULTI_HOP_MAX_POOL_CHUNKS" => {
+                    if let Ok(n) = v.parse::<u32>() {
+                        self.rag.multi_hop_max_pool_chunks = n;
+                    }
+                }
+                // p9-fb-41 PR-9c-1: NLI gate threshold. Parse failure
+                // emits a `tracing::warn!` (not silent like the other
+                // numeric env overrides) because this knob gates the
+                // NLI verification entirely — a malformed env value
+                // would silently disable a security-flavored gate the
+                // user thought they enabled, which is the failure mode
+                // most worth surfacing. The default (`0.0`) survives
+                // on parse failure so behaviour stays well-defined.
+                "KEBAB_RAG_NLI_THRESHOLD" => match v.parse::<f32>() {
+                    Ok(f) => self.rag.nli_threshold = f,
+                    Err(e) => tracing::warn!(
+                        target: "kebab-config",
+                        env_key = "KEBAB_RAG_NLI_THRESHOLD",
+                        env_value = %v,
+                        error = %e,
+                        "invalid KEBAB_RAG_NLI_THRESHOLD; keeping prior value (0.0 = NLI gate disabled)"
+                    ),
+                },

                // image.ocr
                "KEBAB_IMAGE_OCR_ENABLED" => {
@@ -639,6 +884,11 @@ impl Config {
                        self.image.ocr.max_pixels = n;
                    }
                }
+                "KEBAB_IMAGE_OCR_REQUEST_TIMEOUT_SECS" => {
+                    if let Ok(n) = v.parse::<u64>() {
+                        self.image.ocr.request_timeout_secs = n;
+                    }
+                }

                // image.caption (P6-3)
                "KEBAB_IMAGE_CAPTION_ENABLED" => {
@@ -751,6 +1001,83 @@ fn parse_bool(s: &str) -> bool {
 mod tests {
    use super::*;

+    /// Legacy TOML fixture written before the `request_timeout_secs`
+    /// knobs (LLM in v0.17.1, OCR follow-up) existed. Shared by
+    /// `legacy_config_without_request_timeout_secs_uses_default`
+    /// (LLM-side) and `legacy_config_without_ocr_request_timeout_secs_uses_default`
+    /// (OCR-side) so both invariants pin against the same on-disk
+    /// shape — schema drift in the legacy form only needs one edit.
+    const LEGACY_PRE_TIMEOUT_TOML: &str = r#"
+schema_version = 1
+
+[workspace]
+root = "/tmp/x"
+exclude = []
+
+[storage]
+data_dir = "/tmp/x"
+sqlite = "/tmp/x/kebab.sqlite"
+vector_dir = "/tmp/x/lancedb"
+asset_dir = "/tmp/x/assets"
+artifact_dir = "/tmp/x/artifacts"
+model_dir = "/tmp/x/models"
+runs_dir = "/tmp/x/runs"
+copy_threshold_mb = 100
+
+[indexing]
+max_parallel_extractors = 2
+max_parallel_embeddings = 1
+watch_filesystem = false
+
+[chunking]
+target_tokens = 500
+overlap_tokens = 80
+respect_markdown_headings = true
+chunker_version = "md-heading-v1"
+
+[models.embedding]
+provider = "fastembed"
+model = "multilingual-e5-large"
+version = "v1"
+dimensions = 1024
+batch_size = 64
+
+[models.llm]
+provider = "ollama"
+model = "gemma3:4b"
+context_tokens = 4096
+endpoint = "http://127.0.0.1:11434"
+temperature = 0.0
+seed = 0
+
+[search]
+default_k = 10
+hybrid_fusion = "rrf"
+rrf_k = 60
+snippet_chars = 220
+
+[rag]
+prompt_template_version = "rag-v2"
+score_gate = 0.3
+explain_default = false
+max_context_tokens = 8000
+
+[image.ocr]
+enabled = false
+engine = "ollama-vision"
+model = "gemma3:4b"
+languages = ["eng"]
+max_pixels = 1600
+
+[image.caption]
+enabled = false
+max_pixels = 768
+prompt_template_version = "caption-v1"
+
+[ui]
+theme = "dark"
+"#;
+
    #[test]
    fn defaults_are_serde_roundtrip_stable() {
        let c = Config::defaults();
@@ -821,6 +1148,35 @@ mod tests {
        assert!((c.models.llm.temperature - 0.7).abs() < 1e-6);
    }

+    /// v0.17.0 post-dogfood: matches the legacy hard-coded 300s cap so
+    /// existing configs that omit the new field are not affected.
+    #[test]
+    fn default_llm_request_timeout_secs_is_300() {
+        assert_eq!(Config::defaults().models.llm.request_timeout_secs, 300);
+    }
+
+    #[test]
+    fn env_overrides_models_llm_request_timeout_secs() {
+        let mut env = HashMap::new();
+        env.insert(
+            "KEBAB_MODELS_LLM_REQUEST_TIMEOUT_SECS".to_string(),
+            "1200".to_string(),
+        );
+        let c = Config::defaults().apply_env(&env);
+        assert_eq!(c.models.llm.request_timeout_secs, 1200);
+    }
+
+    /// v0.17.0 post-dogfood: a config file written before the field
+    /// existed (no `request_timeout_secs` key) must still parse and fall
+    /// back to the 300s default — backwards-compat invariant. Fixture
+    /// shared with the OCR-side invariant via [`LEGACY_PRE_TIMEOUT_TOML`].
+    #[test]
+    fn legacy_config_without_request_timeout_secs_uses_default() {
+        let c: Config = toml::from_str(LEGACY_PRE_TIMEOUT_TOML)
+            .expect("parse legacy config");
+        assert_eq!(c.models.llm.request_timeout_secs, 300);
+    }
+
    #[test]
    fn env_overrides_indexing_watch_filesystem_bool() {
        let mut env = HashMap::new();
@@ -842,6 +1198,175 @@ mod tests {
        assert_eq!(c.image.ocr.max_pixels, 1600);
    }

+    /// v0.17.2 post-dogfood: matches the legacy hard-coded 300s cap so
+    /// existing configs that omit the new field keep behaving identically.
+    #[test]
+    fn default_ocr_request_timeout_secs_is_300() {
+        assert_eq!(
+            Config::defaults().image.ocr.request_timeout_secs,
+            300
+        );
+    }
+
+    #[test]
+    fn env_overrides_image_ocr_request_timeout_secs() {
+        let mut env = HashMap::new();
+        env.insert(
+            "KEBAB_IMAGE_OCR_REQUEST_TIMEOUT_SECS".to_string(),
+            "900".to_string(),
+        );
+        let c = Config::defaults().apply_env(&env);
+        assert_eq!(c.image.ocr.request_timeout_secs, 900);
+    }
+
+    /// post-v0.17.1 dogfood: a config file written before the OCR
+    /// timeout field existed must still parse and fall back to the
+    /// 300s default — backwards-compat invariant. Fixture shared
+    /// with the LLM-side invariant via [`LEGACY_PRE_TIMEOUT_TOML`].
+    #[test]
+    fn legacy_config_without_ocr_request_timeout_secs_uses_default() {
+        let c: Config = toml::from_str(LEGACY_PRE_TIMEOUT_TOML)
+            .expect("parse legacy config");
+        assert_eq!(c.image.ocr.request_timeout_secs, 300);
+    }
+
+    // ── p9-fb-41: multi-hop RAG knobs ────────────────────────────────────
+
+    #[test]
+    fn default_multi_hop_max_depth_is_3() {
+        assert_eq!(Config::defaults().rag.multi_hop_max_depth, 3);
+    }
+
+    #[test]
+    fn default_multi_hop_max_sub_queries_per_iter_is_5() {
+        assert_eq!(
+            Config::defaults().rag.multi_hop_max_sub_queries_per_iter,
+            5
+        );
+    }
+
+    #[test]
+    fn default_multi_hop_max_pool_chunks_is_15() {
+        // v0.18 dogfood (HOTFIXES 2026-05-25 fb-41 post-PR-7) tuned
+        // this down from 30 → 15 to keep the synthesize prompt tight
+        // enough for gemma3:4b to follow the citation rule.
+        assert_eq!(Config::defaults().rag.multi_hop_max_pool_chunks, 15);
+    }
+
+    #[test]
+    fn env_overrides_multi_hop_knobs() {
+        let mut env = HashMap::new();
+        env.insert(
+            "KEBAB_RAG_MULTI_HOP_MAX_DEPTH".to_string(),
+            "5".to_string(),
+        );
+        env.insert(
+            "KEBAB_RAG_MULTI_HOP_MAX_SUB_QUERIES_PER_ITER".to_string(),
+            "7".to_string(),
+        );
+        env.insert(
+            "KEBAB_RAG_MULTI_HOP_MAX_POOL_CHUNKS".to_string(),
+            "50".to_string(),
+        );
+        let c = Config::defaults().apply_env(&env);
+        assert_eq!(c.rag.multi_hop_max_depth, 5);
+        assert_eq!(c.rag.multi_hop_max_sub_queries_per_iter, 7);
+        assert_eq!(c.rag.multi_hop_max_pool_chunks, 50);
+    }
+
+    /// post-PR-3 fb-41: a config file written before the multi-hop
+    /// knobs existed must still parse and fall back to the documented
+    /// defaults — backwards-compat invariant. Fixture shared with the
+    /// LLM / OCR timeout invariants via [`LEGACY_PRE_TIMEOUT_TOML`]
+    /// (that fixture also predates the multi_hop_* fields).
+    #[test]
+    fn legacy_config_without_multi_hop_knobs_uses_defaults() {
+        let c: Config = toml::from_str(LEGACY_PRE_TIMEOUT_TOML)
+            .expect("parse legacy config");
+        assert_eq!(c.rag.multi_hop_max_depth, 3);
+        assert_eq!(c.rag.multi_hop_max_sub_queries_per_iter, 5);
+        // v0.18 dogfood (post-PR-7): pool default 30 → 15.
+        assert_eq!(c.rag.multi_hop_max_pool_chunks, 15);
+    }
+
+    // ── p9-fb-41 PR-9c-1: NLI verification knobs ─────────────────────────
+
+    #[test]
+    fn default_nli_threshold_is_zero() {
+        // Spec §2.6: NLI gate disabled by default — verification is
+        // opt-in. `0.0` keeps multi-hop behavior identical to PR-3b.
+        assert_eq!(Config::defaults().rag.nli_threshold, 0.0);
+    }
+
+    #[test]
+    fn default_nli_model_is_xenova_mdeberta() {
+        // Pin the default model id so a refactor that touches NliCfg
+        // can't silently flip to a different verifier model.
+        assert_eq!(
+            Config::defaults().models.nli.model,
+            "Xenova/mDeBERTa-v3-base-xnli-multilingual-nli-2mil7"
+        );
+        assert_eq!(Config::defaults().models.nli.provider, "onnx");
+    }
+
+    /// A config file written before the `[models.nli]` / `nli_threshold`
+    /// keys existed must still parse and fall back to the documented
+    /// defaults. Fixture shared via [`LEGACY_PRE_TIMEOUT_TOML`] (predates
+    /// all PR-9c-1 fields).
+    #[test]
+    fn legacy_config_without_nli_uses_defaults() {
+        let c: Config = toml::from_str(LEGACY_PRE_TIMEOUT_TOML)
+            .expect("parse legacy config");
+        assert_eq!(c.rag.nli_threshold, 0.0);
+        assert_eq!(
+            c.models.nli.model,
+            "Xenova/mDeBERTa-v3-base-xnli-multilingual-nli-2mil7"
+        );
+        assert_eq!(c.models.nli.provider, "onnx");
+    }
+
+    #[test]
+    fn env_override_nli_threshold() {
+        let mut env = HashMap::new();
+        env.insert("KEBAB_RAG_NLI_THRESHOLD".to_string(), "0.5".to_string());
+        let c = Config::defaults().apply_env(&env);
+        assert!((c.rag.nli_threshold - 0.5).abs() < 1e-6);
+    }
+
+    #[test]
+    fn env_override_nli_model_and_provider() {
+        let mut env = HashMap::new();
+        env.insert(
+            "KEBAB_MODELS_NLI_MODEL".to_string(),
+            "user/custom-nli-model".to_string(),
+        );
+        env.insert(
+            "KEBAB_MODELS_NLI_PROVIDER".to_string(),
+            "candle".to_string(),
+        );
+        let c = Config::defaults().apply_env(&env);
+        assert_eq!(c.models.nli.model, "user/custom-nli-model");
+        assert_eq!(c.models.nli.provider, "candle");
+    }
+
+    /// Malformed `KEBAB_RAG_NLI_THRESHOLD` keeps the prior value (does
+    /// NOT silently disable nor crash). The `tracing::warn!` surface
+    /// is observable only when the user has tracing wired; the
+    /// behavior contract is "default survives".
+    #[test]
+    fn env_malformed_nli_threshold_keeps_prior_value() {
+        let mut env = HashMap::new();
+        env.insert(
+            "KEBAB_RAG_NLI_THRESHOLD".to_string(),
+            "not-a-float".to_string(),
+        );
+        let c = Config::defaults().apply_env(&env);
+        assert_eq!(
+            c.rag.nli_threshold, 0.0,
+            "malformed env value must keep the default unchanged"
+        );
+    }
+
    #[test]
    fn image_ocr_env_overrides() {
        let mut env = HashMap::new();
@@ -1060,6 +1585,49 @@ max_context_tokens = 8000
            }
        }
    }
+
+    #[test]
+    fn ingest_code_cfg_defaults() {
+        let cfg: IngestCodeCfg = toml::from_str("").unwrap();
+        assert_eq!(cfg.max_file_bytes, 262_144);
+        assert_eq!(cfg.max_file_lines, 5_000);
+        assert!(cfg.skip_generated_header);
+        assert!(cfg.extra_skip_globs.is_empty());
+        assert_eq!(cfg.ast_chunk_max_lines, 200);
+        assert_eq!(cfg.fallback_lines_per_chunk, 80);
+        assert_eq!(cfg.fallback_lines_overlap, 20);
+    }
+
+    #[test]
+    fn ingest_code_cfg_user_override() {
+        let toml = r#"
+            max_file_bytes = 1048576
+            max_file_lines = 20000
+            skip_generated_header = false
+            extra_skip_globs = ["**/fixtures/**", "**/snapshots/**"]
+        "#;
+        let cfg: IngestCodeCfg = toml::from_str(toml).unwrap();
+        assert_eq!(cfg.max_file_bytes, 1_048_576);
+        assert_eq!(cfg.max_file_lines, 20_000);
+        assert!(!cfg.skip_generated_header);
+        assert_eq!(cfg.extra_skip_globs.len(), 2);
+    }
+
+    #[test]
+    fn config_with_ingest_code_section() {
+        // Build a full valid Config serialization and patch only the
+        // [ingest.code] field we care about — avoids having to enumerate
+        // every required Config field in the test fixture.
+        let base = Config::defaults();
+        let mut toml_text = toml::to_string(&base).unwrap();
+        // Inject max_file_bytes override into the [ingest.code] table.
+        toml_text = toml_text.replace(
+            "max_file_bytes = 262144",
+            "max_file_bytes = 524288",
+        );
+        let cfg: Config = toml::from_str(&toml_text).unwrap();
+        assert_eq!(cfg.ingest.code.max_file_bytes, 524_288);
+    }
 }

 #[cfg(test)]
--- a/crates/kebab-config/src/paths.rs
+++ b/crates/kebab-config/src/paths.rs
@@ -157,7 +157,7 @@ mod tests {

    #[test]
    fn xdg_data_home_set_replaces_var() {
-        let _lock = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
+        let _lock = ENV_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
        let _guard = XdgGuard::capture();
        // SAFETY: lock held for the duration of this test.
        unsafe { std::env::set_var("XDG_DATA_HOME", "/custom/path") };
@@ -168,7 +168,7 @@ mod tests {

    #[test]
    fn xdg_data_home_unset_uses_default() {
-        let _lock = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
+        let _lock = ENV_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
        let _guard = XdgGuard::capture();
        // SAFETY: lock held for the duration of this test.
        unsafe { std::env::remove_var("XDG_DATA_HOME") };
@@ -181,7 +181,7 @@ mod tests {

    #[test]
    fn xdg_with_no_default_resolves_to_empty_when_unset() {
-        let _lock = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
+        let _lock = ENV_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
        let _guard = XdgGuard::capture();
        // SAFETY: lock held for the duration of this test.
        unsafe { std::env::remove_var("XDG_DATA_HOME") };
@@ -193,7 +193,7 @@ mod tests {

    #[test]
    fn leading_tilde_expands_to_home() {
-        let _lock = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
+        let _lock = ENV_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
        let home = std::env::var("HOME").expect("HOME must be set in tests");
        let p = expand_path("~/runs", "");
        assert_eq!(p, PathBuf::from(home).join("runs"));
@@ -229,7 +229,7 @@ mod tests {

    #[test]
    fn tilde_path_ignores_base_dir() {
-        let _lock = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
+        let _lock = ENV_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
        let home = std::env::var("HOME").expect("HOME must be set in tests");
        let base = Path::new("/tmp/ignored-cfg");
        let p = expand_path_with_base("~/x", "", base);
@@ -238,7 +238,7 @@ mod tests {

    #[test]
    fn xdg_var_path_ignores_base_dir() {
-        let _lock = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
+        let _lock = ENV_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
        let _guard = XdgGuard::capture();
        // SAFETY: lock held for the duration of this test.
        unsafe { std::env::set_var("XDG_DATA_HOME", "/xdg/data") };
@@ -255,7 +255,7 @@ mod tests {
        // Order matters: substitute `{data_dir}` (which itself contains
        // an unexpanded `${XDG_DATA_HOME}` and `~`), then the other two
        // resolve the result.
-        let _lock = ENV_LOCK.lock().unwrap_or_else(|p| p.into_inner());
+        let _lock = ENV_LOCK.lock().unwrap_or_else(std::sync::PoisonError::into_inner);
        let _guard = XdgGuard::capture();
        // SAFETY: lock held for the duration of this test.
        unsafe { std::env::set_var("XDG_DATA_HOME", "/xdg/data") };
--- a/crates/kebab-core/Cargo.toml
+++ b/crates/kebab-core/Cargo.toml
@@ -16,3 +16,6 @@ time                      = { workspace = true }
 blake3                    = { workspace = true }
 serde_json_canonicalizer  = "0.3"
 unicode-normalization     = "0.1"
+
+[lints]
+workspace = true
--- a/crates/kebab-core/src/answer.rs
+++ b/crates/kebab-core/src/answer.rs
@@ -29,6 +29,37 @@ pub struct Answer {
    /// 이면 single-shot.
    #[serde(default, skip_serializing_if = "Option::is_none")]
    pub turn_index: Option<u32>,
+    /// p9-fb-41: multi-hop hop trace. `None` for single-pass asks.
+    /// Each entry records one hop (`decompose` / `decide` / `synthesize`)
+    /// — the LLM call category, the sub-queries emitted, retrieval
+    /// counts, and a `forced_stop` flag for cap-driven termination.
+    /// Wire-additive: `answer.v1` schema_version unchanged; consumers
+    /// reading older single-pass answers see `hops: None` (or absent).
+    #[serde(default, skip_serializing_if = "Option::is_none")]
+    pub hops: Option<Vec<HopRecord>>,
+    /// p9-fb-41 PR-9c-1: NLI-based post-synthesis verification summary.
+    /// `None` for single-pass asks and for multi-hop runs with
+    /// `[rag].nli_threshold == 0` (verification disabled — the default).
+    /// Present only when the multi-hop pipeline reached the post-
+    /// synthesize verification step (PR-9c-2 wires step 8.5). Wire-
+    /// additive: `answer.v1` schema_version unchanged; consumers
+    /// reading pre-v0.18 answers see `verification: None` (or absent).
+    #[serde(default, skip_serializing_if = "Option::is_none")]
+    pub verification: Option<VerificationSummary>,
+}
+
+/// p9-fb-41 PR-9c-1: post-synthesize NLI verification summary stamped
+/// onto [`Answer::verification`] when multi-hop runs reach step 8.5
+/// (NLI gate). Three required fields ride together on every wire emit:
+/// `nli_score` is the entailment channel of the XNLI verifier,
+/// `nli_threshold` mirrors `[rag].nli_threshold` for audit, and
+/// `nli_passed` is `nli_score >= nli_threshold`. The whole struct is
+/// omitted (serde skip) when no verification ran.
+#[derive(Clone, Copy, Debug, Default, PartialEq, Serialize, Deserialize)]
+pub struct VerificationSummary {
+    pub nli_score: f32,
+    pub nli_threshold: f32,
+    pub nli_passed: bool,
 }

 #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
@@ -55,6 +86,79 @@ pub struct Turn {
    pub created_at: OffsetDateTime,
 }

+/// p9-fb-41: one entry in [`Answer::hops`] — the per-iteration trace
+/// of a multi-hop ask. The pipeline appends a `HopRecord` per LLM
+/// call (decompose / decide / synthesize) so a `--multi-hop` user
+/// can see what sub-queries the LLM emitted, how many chunks each
+/// hop contributed, whether the iter stopped on the model's own
+/// signal or hit a cap, and the per-hop LLM latency.
+///
+/// Wire-additive — every field uses `#[serde(default)]` where it
+/// could plausibly be omitted by a future schema reader.
+#[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
+pub struct HopRecord {
+    /// 0-based hop index within this ask. `iter=0` is always the
+    /// initial decompose call; subsequent iters are decide calls;
+    /// the final iter is the synthesize call.
+    pub iter: u32,
+    pub kind: HopKind,
+    /// Sub-queries associated with this hop. The meaning depends on
+    /// `kind`:
+    ///
+    /// - [`HopKind::Decompose`]: the initial sub-queries the LLM
+    ///   broke the original user query into. These drive the
+    ///   `iter=1` retrieval round.
+    /// - [`HopKind::Decide`]: the *new* sub-queries the LLM
+    ///   emitted to drive the next retrieval round. Empty when the
+    ///   LLM signalled stop OR when `forced_stop = true` (cap hit
+    ///   or parse-degraded).
+    /// - [`HopKind::Synthesize`]: always empty — the final hop
+    ///   produces the user-visible answer, not more sub-queries.
+    #[serde(default)]
+    pub sub_queries: Vec<String>,
+    /// Number of *new* chunks the retrieval round contributed to the
+    /// pool (dedup'd by `chunk_id` — repeated hits from a previous
+    /// iter do not count). `0` for the decompose hop (no retrieval
+    /// yet) and the synthesize hop.
+    pub context_chunks_added: u32,
+    /// `true` when the pipeline cut the iter loop short because a
+    /// safety cap fired (`max_depth` / `max_total_sub_queries` /
+    /// `max_pool_chunks`) rather than because the LLM signalled
+    /// stop. The user-visible answer still reflects all chunks
+    /// accumulated up to that point — `forced_stop` is a tracing
+    /// signal, not a refusal.
+    pub forced_stop: bool,
+    /// Wall-clock latency of the LLM call for this hop, in
+    /// milliseconds. Useful for cost / latency analysis when a
+    /// `kebab eval` run records `Answer.hops`.
+    ///
+    /// `0` is overloaded: it means "no LLM call happened at this
+    /// hop" when (a) the hop was a Decide skipped due to
+    /// `forced_stop` (depth-cap or pool-cap fired before the LLM
+    /// was asked) or (b) the pool was empty before any decide
+    /// could run. Treat `0` as "absent or instantaneous" rather
+    /// than as a genuine measurement.
+    pub llm_call_ms: u32,
+}
+
+/// p9-fb-41: which stage of the multi-hop pipeline a [`HopRecord`]
+/// describes. The serde tag matches the wire shape so agents /
+/// CLIs can branch on the snake_case string without referencing
+/// the Rust enum.
+#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq, Serialize, Deserialize)]
+#[serde(rename_all = "snake_case")]
+pub enum HopKind {
+    /// First hop — LLM decomposed the user query into sub-queries.
+    Decompose,
+    /// Subsequent hop — LLM was asked whether more retrieval is
+    /// needed and either emitted new sub-queries (`continue`) or
+    /// returned an empty array (`stop`).
+    Decide,
+    /// Terminal hop — LLM produced the final user-visible answer
+    /// over the accumulated chunk pool.
+    Synthesize,
+}
+
 #[derive(Clone, Copy, Debug, Eq, Hash, PartialEq, Serialize, Deserialize)]
 #[serde(rename_all = "snake_case")]
 pub enum RefusalReason {
@@ -66,6 +170,24 @@ pub enum RefusalReason {
    /// 가 채워져 있을 수 있음 (사용자가 본 부분까지). RAG retrieval
    /// 자체는 정상 — 모델 generation 단계에서만 중단.
    LlmStreamAborted,
+    /// p9-fb-41: multi-hop pipeline 의 decompose LLM call 이 JSON
+    /// parse 가능한 sub-question array 를 반환하지 못함 (parse
+    /// error, 빈 응답, 또는 잘못된 형식). retrieval / synthesize
+    /// 단계 진입 못 함. CLI / MCP / TUI 가 받는 wire error code
+    /// = `"multi_hop_decompose_failed"` (PR-4 의 error_wire 매핑).
+    MultiHopDecomposeFailed,
+    /// p9-fb-41 PR-9c-1: post-synthesize NLI verification gate fired —
+    /// `NliScores::faithfulness()` (entailment channel) fell below
+    /// `[rag].nli_threshold`. Wire string = `"nli_verification_failed"`
+    /// (single source of truth: also the matching `error.v1.code`).
+    /// Multi-hop only; behavior wiring lands in PR-9c-2.
+    NliVerificationFailed,
+    /// p9-fb-41 PR-9c-1: NLI verifier was configured (threshold > 0)
+    /// but the model / runtime is unavailable (download failure,
+    /// missing tokenizer, ONNX session init error). Treated as a soft
+    /// refusal — the user sees an unverified-answer outcome rather
+    /// than crashing the ask. Wire string = `"nli_model_unavailable"`.
+    NliModelUnavailable,
 }

 #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
@@ -103,6 +225,81 @@ mod tests {
    use crate::citation::Citation;
    use time::macros::datetime;

+    /// p9-fb-41 PR-9c-1: pin the wire-side spelling of the new
+    /// `RefusalReason` variants. The strings here must match
+    /// `answer.schema.json::refusal_reason.enum` AND
+    /// `error.schema.json::code.enum` byte-for-byte (single source of
+    /// truth per spec §2.4).
+    #[test]
+    fn refusal_reason_nli_variants_serialize_to_snake_case() {
+        assert_eq!(
+            serde_json::to_string(&RefusalReason::NliVerificationFailed).unwrap(),
+            "\"nli_verification_failed\""
+        );
+        assert_eq!(
+            serde_json::to_string(&RefusalReason::NliModelUnavailable).unwrap(),
+            "\"nli_model_unavailable\""
+        );
+    }
+
+    /// p9-fb-41 PR-9c-1: `Answer.verification` is `Option<...>` with
+    /// `skip_serializing_if = None`. A `verification: None` answer
+    /// must NOT emit a `"verification"` key on the wire — the field
+    /// is additive and pre-v0.18 readers see no new key.
+    #[test]
+    fn answer_omits_verification_field_when_none() {
+        let ans = Answer {
+            answer: "x".into(),
+            citations: vec![],
+            grounded: true,
+            refusal_reason: None,
+            model: ModelRef {
+                id: "m".into(),
+                provider: "p".into(),
+                dimensions: None,
+            },
+            embedding: None,
+            prompt_template_version: PromptTemplateVersion("rag-v2".into()),
+            retrieval: AnswerRetrievalSummary {
+                trace_id: TraceId("t".into()),
+                mode: crate::SearchMode::Lexical,
+                k: 1,
+                score_gate: 0.0,
+                top_score: 0.0,
+                chunks_returned: 0,
+                chunks_used: 0,
+            },
+            usage: TokenUsage {
+                prompt_tokens: 0,
+                completion_tokens: 0,
+                latency_ms: 0,
+            },
+            created_at: datetime!(2026-05-09 12:00:00 UTC),
+            conversation_id: None,
+            turn_index: None,
+            hops: None,
+            verification: None,
+        };
+        let v = serde_json::to_value(&ans).unwrap();
+        assert!(
+            v.get("verification").is_none(),
+            "verification: None must be omitted from wire output, got: {v}"
+        );
+    }
+
+    #[test]
+    fn verification_summary_serializes_all_three_required_fields() {
+        let vs = VerificationSummary {
+            nli_score: 0.87,
+            nli_threshold: 0.5,
+            nli_passed: true,
+        };
+        let v = serde_json::to_value(vs).unwrap();
+        assert!((v["nli_score"].as_f64().unwrap() - 0.87).abs() < 1e-5);
+        assert!((v["nli_threshold"].as_f64().unwrap() - 0.5).abs() < 1e-5);
+        assert_eq!(v["nli_passed"], true);
+    }
+
    #[test]
    fn answer_citation_serializes_indexed_at_and_stale() {
        let ac = AnswerCitation {
--- a/crates/kebab-core/src/citation.rs
+++ b/crates/kebab-core/src/citation.rs
@@ -37,6 +37,13 @@ pub enum Citation {
        end_ms: u64,
        speaker: Option<String>,
    },
+    Code {
+        path: WorkspacePath,
+        line_start: u32,
+        line_end: u32,
+        symbol: Option<String>,
+        lang: Option<String>,
+    },
 }

 impl Citation {
@@ -46,7 +53,8 @@ impl Citation {
            | Citation::Page { path, .. }
            | Citation::Region { path, .. }
            | Citation::Caption { path, .. }
-            | Citation::Time { path, .. } => path,
+            | Citation::Time { path, .. }
+            | Citation::Code { path, .. } => path,
        }
    }

@@ -80,6 +88,18 @@ impl Citation {
                    None => format!("{}#t={},{}", path.0, s, e),
                }
            }
+            Citation::Code {
+                path,
+                line_start,
+                line_end,
+                ..
+            } => {
+                if line_start == line_end {
+                    format!("{}#L{}", path.0, line_start)
+                } else {
+                    format!("{}#L{}-L{}", path.0, line_start, line_end)
+                }
+            }
        }
    }

@@ -206,28 +226,25 @@ fn parse_hms_ms(s: &str) -> Result<u64> {
    let m: u64 = parts[1]
        .parse()
        .map_err(|_| anyhow::anyhow!("bad minutes in {:?} (input {s:?})", parts[1]))?;
-    let (sec, ms) = match parts[2].split_once('.') {
-        Some((s_part, ms_part)) => {
-            let sec: u64 = s_part
-                .parse()
-                .map_err(|_| anyhow::anyhow!("bad seconds in {s_part:?} (input {s:?})"))?;
-            // Pad/truncate to exactly 3 digits.
-            let mut ms_str = ms_part.to_owned();
-            while ms_str.len() < 3 {
-                ms_str.push('0');
-            }
-            ms_str.truncate(3);
-            let ms: u64 = ms_str
-                .parse()
-                .map_err(|_| anyhow::anyhow!("bad milliseconds in {ms_part:?} (input {s:?})"))?;
-            (sec, ms)
-        }
-        None => {
-            let sec: u64 = parts[2]
-                .parse()
-                .map_err(|_| anyhow::anyhow!("bad seconds in {:?} (input {s:?})", parts[2]))?;
-            (sec, 0)
+    let (sec, ms) = if let Some((s_part, ms_part)) = parts[2].split_once('.') {
+        let sec: u64 = s_part
+            .parse()
+            .map_err(|_| anyhow::anyhow!("bad seconds in {s_part:?} (input {s:?})"))?;
+        // Pad/truncate to exactly 3 digits.
+        let mut ms_str = ms_part.to_owned();
+        while ms_str.len() < 3 {
+            ms_str.push('0');
        }
+        ms_str.truncate(3);
+        let ms: u64 = ms_str
+            .parse()
+            .map_err(|_| anyhow::anyhow!("bad milliseconds in {ms_part:?} (input {s:?})"))?;
+        (sec, ms)
+    } else {
+        let sec: u64 = parts[2]
+            .parse()
+            .map_err(|_| anyhow::anyhow!("bad seconds in {:?} (input {s:?})", parts[2]))?;
+        (sec, 0)
    };
    Ok(h * 3_600_000 + m * 60_000 + sec * 1000 + ms)
 }
@@ -354,4 +371,64 @@ mod tests {
        let r = Citation::parse("notes/x#evil.md#L7");
        assert!(r.is_err(), "path with embedded '#' must be rejected");
    }
+
+    #[test]
+    fn citation_code_variant_serializes_with_kind_tag() {
+        let c = Citation::Code {
+            path: WorkspacePath("crates/kebab-chunk/src/md_heading_v1.rs".into()),
+            line_start: 142,
+            line_end: 168,
+            symbol: Some("MdHeadingV1Chunker::chunk_doc".into()),
+            lang: Some("rust".into()),
+        };
+        let v = serde_json::to_value(&c).unwrap();
+        assert_eq!(v["kind"], "code");
+        assert_eq!(v["line_start"], 142);
+        assert_eq!(v["line_end"], 168);
+        assert_eq!(v["symbol"], "MdHeadingV1Chunker::chunk_doc");
+        assert_eq!(v["lang"], "rust");
+        // Existing 5 variants must NOT pick up these fields.
+        let line = Citation::Line {
+            path: WorkspacePath("notes/foo.md".into()),
+            start: 1,
+            end: 10,
+            section: None,
+        };
+        let lv = serde_json::to_value(&line).unwrap();
+        assert!(lv.get("line_start").is_none());
+        assert!(lv.get("symbol").is_none());
+    }
+
+    #[test]
+    fn citation_code_uri_format() {
+        let c = Citation::Code {
+            path: WorkspacePath("a/b.rs".into()),
+            line_start: 10,
+            line_end: 20,
+            symbol: None,
+            lang: Some("rust".into()),
+        };
+        assert_eq!(c.to_uri(), "a/b.rs#L10-L20");
+        // Single-line uses `#L10`.
+        let single = Citation::Code {
+            path: WorkspacePath("a/b.rs".into()),
+            line_start: 5,
+            line_end: 5,
+            symbol: None,
+            lang: None,
+        };
+        assert_eq!(single.to_uri(), "a/b.rs#L5");
+    }
+
+    #[test]
+    fn citation_code_path_accessor() {
+        let c = Citation::Code {
+            path: WorkspacePath("x.rs".into()),
+            line_start: 1,
+            line_end: 1,
+            symbol: None,
+            lang: None,
+        };
+        assert_eq!(c.path().0, "x.rs");
+    }
 }
--- a/crates/kebab-core/src/document.rs
+++ b/crates/kebab-core/src/document.rs
@@ -142,6 +142,18 @@ pub enum SourceSpan {
        start_ms: u64,
        end_ms: u64,
    },
+    /// p10-1A-2: AST-unit span for code ingest. Internal storage shape
+    /// (chunks.source_spans_json) — `citation_helper` maps this to the
+    /// wire `Citation::Code` (added 1A-1). `symbol` is the per-language
+    /// self-reference path (design §3.4); `<top-level>` / `<module>` for
+    /// glue regions, never null for an identified unit. `lang` is the
+    /// canonical code_lang.
+    Code {
+        line_start: u32,
+        line_end: u32,
+        symbol: Option<String>,
+        lang: Option<String>,
+    },
 }

 // ── Forward-declared stubs (§3.7a). Bodies are final per design. ────────
@@ -195,6 +207,24 @@ mod tests {
    /// previously failed at serde runtime because `tag = "kind"` cannot
    /// describe a newtype carrying a non-struct value. The struct-variant
    /// shape used here is the §9 schema migration.
+    #[test]
+    fn source_span_code_round_trips_and_tags_lowercase() {
+        let s = SourceSpan::Code {
+            line_start: 10,
+            line_end: 42,
+            symbol: Some("foo::Bar::baz".to_string()),
+            lang: Some("rust".to_string()),
+        };
+        let v = serde_json::to_value(&s).unwrap();
+        assert_eq!(v["kind"], "code");
+        assert_eq!(v["line_start"], 10);
+        assert_eq!(v["line_end"], 42);
+        assert_eq!(v["symbol"], "foo::Bar::baz");
+        assert_eq!(v["lang"], "rust");
+        let back: SourceSpan = serde_json::from_value(v).unwrap();
+        assert_eq!(back, s);
+    }
+
    #[test]
    fn inline_serde_round_trip() {
        let cases = vec![
--- a/crates/kebab-core/src/ingest.rs
+++ b/crates/kebab-core/src/ingest.rs
@@ -25,10 +25,52 @@ pub struct IngestReport {
    /// extension key under "<no-ext>". `BTreeMap` so the wire JSON
    /// has stable key order across runs.
    pub skipped_by_extension: std::collections::BTreeMap<String, u32>,
+    /// p10-1A-1: files skipped because they matched a repo-local `.gitignore`.
+    #[serde(default)]
+    pub skipped_gitignore: u32,
+    /// p10-1A-1: files skipped because they matched a `.kebabignore` entry.
+    #[serde(default)]
+    pub skipped_kebabignore: u32,
+    /// p10-1A-1: files skipped because they matched the built-in safety-net
+    /// blacklist (`node_modules/`, `target/`, `__pycache__/`, `.venv/`,
+    /// `venv/`, `env/`).
+    #[serde(default)]
+    pub skipped_builtin_blacklist: u32,
+    /// p10-1A-1: files skipped because their first ~512 bytes contained a
+    /// generated-file marker (`@generated`, `do not edit`, …).
+    #[serde(default)]
+    pub skipped_generated: u32,
+    /// p10-1A-1: files skipped because they exceeded `max_file_bytes` or
+    /// `max_file_lines` in `[ingest.code]`.
+    #[serde(default)]
+    pub skipped_size_exceeded: u32,
+    /// p10-1A-1: sample file paths per skip category (≤ 5 each).
+    #[serde(default)]
+    pub skip_examples: SkipExamples,
+    /// Dogfood: docs whose on-disk file was deleted since the last ingest
+    /// and were therefore removed from the store. Additive field — older
+    /// wire consumers that pre-date this field read it as 0 via
+    /// `#[serde(default)]`.
+    #[serde(default)]
+    pub purged_deleted_files: u32,
    /// `None` ↔ wire `items: null` (`--summary-only`).
    pub items: Option<Vec<IngestItem>>,
 }

+/// p10-1A-1: per-category sample of skipped file paths. Each category caps at
+/// 5 entries (oldest-first). Used for debugging "why was X not indexed?"
+#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
+pub struct SkipExamples {
+    #[serde(default)]
+    pub generated: Vec<String>,
+    #[serde(default)]
+    pub size_exceeded: Vec<String>,
+    #[serde(default)]
+    pub builtin_blacklist: Vec<String>,
+    #[serde(default)]
+    pub gitignore: Vec<String>,
+}
+
 #[derive(Clone, Debug, PartialEq, Serialize, Deserialize)]
 pub struct IngestItem {
    pub kind: IngestItemKind,
@@ -58,3 +100,56 @@ pub enum IngestItemKind {
    Unchanged,
    Error,
 }
+
+#[cfg(test)]
+mod tests {
+    use super::*;
+    use crate::traits::SourceScope;
+
+    #[test]
+    fn skip_examples_default_is_empty() {
+        let s = SkipExamples::default();
+        assert!(s.generated.is_empty());
+        assert!(s.size_exceeded.is_empty());
+        assert!(s.builtin_blacklist.is_empty());
+        assert!(s.gitignore.is_empty());
+    }
+
+    #[test]
+    fn ingest_report_skip_counters_serialize() {
+        let r = IngestReport {
+            scope: SourceScope {
+                root: std::path::PathBuf::from("/tmp"),
+                include: vec![],
+                exclude: vec![],
+            },
+            scanned: 100,
+            new: 50,
+            updated: 0,
+            skipped: 0,
+            unchanged: 0,
+            errors: 0,
+            duration_ms: 1234,
+            skipped_by_extension: Default::default(),
+            skipped_gitignore: 30,
+            skipped_kebabignore: 5,
+            skipped_builtin_blacklist: 10,
+            skipped_generated: 3,
+            skipped_size_exceeded: 2,
+            skip_examples: SkipExamples {
+                generated: vec!["a/b.pb.rs".into()],
+                size_exceeded: vec![],
+                builtin_blacklist: vec!["node_modules/x.js".into()],
+                gitignore: vec![],
+            },
+            purged_deleted_files: 0,
+            items: None,
+        };
+        let v = serde_json::to_value(&r).unwrap();
+        assert_eq!(v["skipped_gitignore"], 30);
+        assert_eq!(v["skipped_builtin_blacklist"], 10);
+        assert_eq!(v["skipped_generated"], 3);
+        assert_eq!(v["skipped_size_exceeded"], 2);
+        assert_eq!(v["skip_examples"]["generated"][0], "a/b.pb.rs");
+    }
+}
--- a/Show More
+++ b/Show More