Merge pull request 'feat(p10-1C-Go): tree-sitter-go AST extractor + chunker — Go 코드 색인 활성화' (#151 ) from feat/p10-1c-go into main

docs(p10-1c-go): README/HANDOFF/ARCHITECTURE/SMOKE/INDEX + design §10.1; chore: bump version 0.11.1 → 0.12.0
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-20 10:16:09 +00:00 · 2026-05-20 10:02:21 +00:00 · 2026-05-20 09:54:17 +00:00 · 2026-05-20 09:13:47 +00:00 · 2026-05-20 09:11:14 +00:00 · 2026-05-20 09:08:46 +00:00
46 changed files with 3684 additions and 116 deletions
--- a/Cargo.lock
+++ b/Cargo.lock
@@ -4127,7 +4127,7 @@ dependencies = [

 [[package]]
 name = "kebab-app"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "base64 0.22.1",
@@ -4172,7 +4172,7 @@ dependencies = [

 [[package]]
 name = "kebab-chunk"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "blake3",
@@ -4187,7 +4187,7 @@ dependencies = [

 [[package]]
 name = "kebab-cli"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "clap",
@@ -4208,7 +4208,7 @@ dependencies = [

 [[package]]
 name = "kebab-config"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "dirs 5.0.1",
@@ -4223,7 +4223,7 @@ dependencies = [

 [[package]]
 name = "kebab-core"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "blake3",
@@ -4237,7 +4237,7 @@ dependencies = [

 [[package]]
 name = "kebab-embed"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "blake3",
@@ -4251,7 +4251,7 @@ dependencies = [

 [[package]]
 name = "kebab-embed-local"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "fastembed",
@@ -4264,7 +4264,7 @@ dependencies = [

 [[package]]
 name = "kebab-eval"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "kebab-app",
@@ -4283,7 +4283,7 @@ dependencies = [

 [[package]]
 name = "kebab-llm"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "kebab-core",
@@ -4292,7 +4292,7 @@ dependencies = [

 [[package]]
 name = "kebab-llm-local"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "kebab-config",
@@ -4309,7 +4309,7 @@ dependencies = [

 [[package]]
 name = "kebab-mcp"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "kebab-app",
@@ -4327,7 +4327,7 @@ dependencies = [

 [[package]]
 name = "kebab-normalize"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "kebab-core",
@@ -4342,7 +4342,7 @@ dependencies = [

 [[package]]
 name = "kebab-parse-code"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "gix",
@@ -4352,6 +4352,7 @@ dependencies = [
 "time",
 "tracing",
 "tree-sitter",
+ "tree-sitter-go",
 "tree-sitter-javascript",
 "tree-sitter-python",
 "tree-sitter-rust",
@@ -4360,7 +4361,7 @@ dependencies = [

 [[package]]
 name = "kebab-parse-image"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "ab_glyph",
 "anyhow",
@@ -4384,7 +4385,7 @@ dependencies = [

 [[package]]
 name = "kebab-parse-md"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "kebab-core",
@@ -4401,7 +4402,7 @@ dependencies = [

 [[package]]
 name = "kebab-parse-pdf"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "blake3",
@@ -4414,7 +4415,7 @@ dependencies = [

 [[package]]
 name = "kebab-parse-types"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "kebab-core",
 "serde",
@@ -4422,7 +4423,7 @@ dependencies = [

 [[package]]
 name = "kebab-rag"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "blake3",
@@ -4443,7 +4444,7 @@ dependencies = [

 [[package]]
 name = "kebab-search"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "globset",
@@ -4462,10 +4463,11 @@ dependencies = [

 [[package]]
 name = "kebab-source-fs"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "blake3",
+ "globset",
 "ignore",
 "kebab-config",
 "kebab-core",
@@ -4480,7 +4482,7 @@ dependencies = [

 [[package]]
 name = "kebab-store-sqlite"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "blake3",
@@ -4501,7 +4503,7 @@ dependencies = [

 [[package]]
 name = "kebab-store-vector"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "arrow",
@@ -4525,7 +4527,7 @@ dependencies = [

 [[package]]
 name = "kebab-tui"
-version = "0.8.1"
+version = "0.12.0"
 dependencies = [
 "anyhow",
 "crossterm",
@@ -8526,6 +8528,16 @@ dependencies = [
 "tree-sitter-language",
 ]

+[[package]]
+name = "tree-sitter-go"
+version = "0.25.0"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "c8560a4d2f835cc0d4d2c2e03cbd0dde2f6114b43bc491164238d333e28b16ea"
+dependencies = [
+ "cc",
+ "tree-sitter-language",
+]
+
 [[package]]
 name = "tree-sitter-javascript"
 version = "0.25.0"
--- a/Cargo.toml
+++ b/Cargo.toml
@@ -31,7 +31,7 @@ edition       = "2024"
 rust-version  = "1.85"
 license       = "MIT OR Apache-2.0"
 repository    = "https://github.com/altair823/kebab"
-version       = "0.8.1"
+version       = "0.12.0"

 [workspace.dependencies]
 anyhow       = "1"
@@ -94,6 +94,8 @@ tree-sitter-rust = "0.24"
 tree-sitter-python     = "0.25.0"
 tree-sitter-typescript = "0.23.2"
 tree-sitter-javascript = "0.25.0"
+# Go grammar for code ingest (kebab-parse-code, p10-1C-Go).
+tree-sitter-go         = "0.25.0"

 # Disk-footprint trim for dev / test builds. Codegen, opt-level, and
 # behavior are unchanged — only DWARF debug info is reduced (line
--- a/HANDOFF.md
+++ b/HANDOFF.md
@@ -4,7 +4,7 @@

 ## 한 줄 요약

-P0–P5 + P6 + P7 + P9-1/2/3/4 (Library / Search / Ask / Inspect) 머지 완료. `kebab ingest` 가 markdown / image / PDF 모두 처리. `kebab search` / `kebab ask` 가 매체 가로질러 결과 + page citation 반환. `kebab tui` 가 4 패널 (Library + Search + Ask + Inspect) 제공 — 사용자가 `?` 로 ask, `/` 로 search, Library Enter / Search `i` 로 inspect, Search `g` 로 editor jump. 다음 후보 = P9-5 (desktop tauri) 또는 보류 중인 P8 (audio) 의 시스템 dep brainstorm.
+P0–P5 + P6 + P7 + P9-1/2/3/4 (Library / Search / Ask / Inspect) 머지 완료. `kebab ingest` 가 markdown / image / PDF / 소스코드 (Rust / Python / TS / JS / Go) 처리. `kebab search` / `kebab ask` 가 매체 가로질러 결과 + page / code citation 반환. `kebab tui` 가 4 패널 (Library + Search + Ask + Inspect) 제공. 다음 후보 = P10-1C-JavaKotlin 또는 P9-5 (desktop tauri) 또는 보류 중인 P8 (audio).

 ## Phase 로드맵

@@ -20,7 +20,7 @@ P0–P5 + P6 + P7 + P9-1/2/3/4 (Library / Search / Ask / Inspect) 머지 완료.
 | **P7** | PDF text + page citation | `kebab-parse-pdf` | P5 | ✅ 완료 (3/3 component, page-level chunker + ingest wiring) |
 | **P8** | 음성 transcription + timestamp citation | `kebab-parse-audio` | P5 | ⏸ 보류 (whisper-rs 시스템 dep brainstorm 필요) |
 | **P9** | TUI + desktop app | `kebab-tui`, `kebab-desktop` | P5 | 🟡 진행 (4/5 component — P9-1/2/3/4 완료 [Library / Search / Ask / Inspect], P9-5 desktop 예정 · 도그푸딩 피드백 **20/20 ✅**) |
-| **P10** | code ingest framework | `kebab-parse-code` | P5 | 🟡 진행 중 — 1A-1 ✅ (wire schema + parse-code skeleton + filter flags), 1A-2 ✅ (Rust AST chunker, tree-sitter-rust, `code-rust-ast-v1` — v0.7.0), **1B 🟡 PR 오픈** (Python `code-python-ast-v1` + TypeScript `code-ts-ast-v1` + JavaScript `code-js-ast-v1` — 3 언어 dogfooding 가능, v0.8.0 대기) |
+| **P10** | code ingest framework | `kebab-parse-code` | P5 | 🟡 진행 중 — 1A-1 ✅ (wire schema + parse-code skeleton + filter flags), 1A-2 ✅ (Rust AST chunker, `code-rust-ast-v1` — v0.7.0), 1B ✅ (Python/TS/JS AST chunkers — v0.8.0 이후), **1C-Go ✅ (Go AST chunker, `code-go-ast-v1` — v0.12.0)**, 1C-JavaKotlin ⏳ (후속 PR) |

 P0~P5 직렬. P6~P9 P5 이후 병렬 가능.

--- a/README.md
+++ b/README.md
@@ -34,7 +34,7 @@ cargo install --git https://gitea.altair823.xyz/altair823-org/kebab.git --bin ke

 업데이트는 `git pull && cargo install --path crates/kebab-cli --locked --force` 또는 git URL 형식의 경우 `cargo install --git ... --force`.

-제거는 `cargo uninstall kebab-cli`. 이 명령은 binary 만 지우고 워크스페이스 데이터는 그대로 남는다. 데이터까지 정리하려면 `kebab reset --all --yes` (config + data + cache + state 4 개 XDG 경로 모두 wipe — **irreversible**, 재시작 시 `kebab init` 다시 실행). 부분 wipe 는 `kebab reset --data-only` (config 보존), `kebab reset --vector-only` (Lance + `embedding_records` 만, 다음 ingest 가 re-embed) 등.
+제거는 `cargo uninstall kebab-cli`. 이 명령은 binary 만 지우고 워크스페이스 데이터는 그대로 남는다. 데이터까지 정리하려면 `kebab reset --all --yes` (config + data + cache + state 4 개 XDG 경로 모두 wipe — **irreversible**, 재시작 시 `kebab init` 다시 실행). 부분 wipe 는 `kebab reset --data-only` (config 보존), `kebab reset --vector-only` (Lance + `embedding_records` 만, 다음 ingest 가 re-embed), **`kebab reset --orphans-only`** (현재 walker scope 밖에 있는 stored doc 만 정리 — `config.workspace.include` 좁히거나 sub-dir 옮긴 후 explicit reconcile; fs 의 file 은 건드리지 않음) 등.

 ## Quick start

@@ -42,7 +42,7 @@ cargo install --git https://gitea.altair823.xyz/altair823-org/kebab.git --bin ke
 # 첫 실행 — XDG 경로에 데이터 디렉토리 + config.toml 생성
 kebab init

-# config 손보고 — workspace.root, 모델 endpoint 등 설정 (지원 형식: md / png / jpg / pdf / rs / py / ts / js)
+# config 손보고 — workspace.root, 모델 endpoint 등 설정 (지원 형식: md / png / jpg / pdf / rs / py / ts / js / go)
 ${EDITOR:-vi} ~/.config/kebab/config.toml

 # 색인 (Markdown / 이미지 / PDF 모두 한 번에)
@@ -70,7 +70,7 @@ kebab doctor
 | 명령 | 동작 |
 |------|------|
 | `kebab init` | XDG 경로에 데이터 디렉토리 + config.toml 생성 |
-| `kebab ingest [<path>]` | Markdown / 이미지 / PDF / Rust 소스코드 색인 (idempotent). TTY 에서는 stderr 진행 바, non-TTY (CI / pipe) 는 stderr 한 줄씩, `--json` 은 stdout 에 `ingest_progress.v1` 라인 streaming 후 마지막에 `ingest_report.v1`. Ctrl-C 한 번이면 현재 asset 마무리 후 abort (부분 commit 보존, idempotent re-run), 두 번째 Ctrl-C 는 hard exit. Markdown title 이 frontmatter 에 없어도 첫 H1 → H2 → 첫 paragraph 80 자 → 파일명 순으로 자동 채움 (parser_version `md-frontmatter-v2`) — 기존 색인된 doc 도 다음 ingest 에서 새 title 로 갱신. **Incremental** (p9-fb-23): 두 번째 이후의 ingest 는 변하지 않은 doc (blake3 + parser/chunker/embedder version 모두 동일) 의 parse/chunk/embed/vector upsert 를 자동 스킵. final summary 에 `N unchanged` 카운트 표시. `--force-reingest` 로 skip 무시 강제 재처리. **지원 형식** (extractor 자동 결정 — config 에 명시 불가): Markdown (`.md`), 이미지 (`.png` / `.jpg` / `.jpeg`, OCR + caption), PDF (`.pdf`), **소스코드** (`.rs` → `code-rust-ast-v1`, `.py` → `code-python-ast-v1`, `.ts`/`.tsx` → `code-ts-ast-v1`, `.js`/`.mjs`/`.cjs`/`.jsx` → `code-js-ast-v1` — 모두 tree-sitter AST chunker). 다른 확장자는 자동 skip — `IngestItem.warnings` 에 사유 (`"unsupported media type: .docx"` 등), `IngestReport.skipped_by_extension` 에 카운트 분류, CLI / TUI summary 에 breakdown 표시. 코드 chunk 는 `citation.kind = "code"` 에 `citation.lang = "<lang>"` + `symbol` + line range 를 담고, SearchHit top-level 에 `code_lang` + `repo` (`.git/` walk-up 의 디렉토리 이름) 가 backfill 됨. `--code-lang rust` / `--code-lang python` / `--code-lang typescript` / `--code-lang javascript` / `--media code` filter 로 언어별·코드 전용 검색 가능 (p10-1A-1 filter flags). Python symbol 은 workspace 경로 → dotted module path prefix (예: `kebab_eval.metrics.compute_mrr`), TS/JS symbol 은 slash-style module path prefix (예: `src/Foo.Foo.search`). |
+| `kebab ingest [<path>]` | Markdown / 이미지 / PDF / Rust 소스코드 색인 (idempotent). TTY 에서는 stderr 진행 바, non-TTY (CI / pipe) 는 stderr 한 줄씩, `--json` 은 stdout 에 `ingest_progress.v1` 라인 streaming 후 마지막에 `ingest_report.v1`. Ctrl-C 한 번이면 현재 asset 마무리 후 abort (부분 commit 보존, idempotent re-run), 두 번째 Ctrl-C 는 hard exit. Markdown title 이 frontmatter 에 없어도 첫 H1 → H2 → 첫 paragraph 80 자 → 파일명 순으로 자동 채움 (parser_version `md-frontmatter-v2`) — 기존 색인된 doc 도 다음 ingest 에서 새 title 로 갱신. **Incremental** (p9-fb-23): 두 번째 이후의 ingest 는 변하지 않은 doc (blake3 + parser/chunker/embedder version 모두 동일) 의 parse/chunk/embed/vector upsert 를 자동 스킵. final summary 에 `N unchanged` 카운트 표시. `--force-reingest` 로 skip 무시 강제 재처리. **지원 형식** (extractor 자동 결정 — config 에 명시 불가): Markdown (`.md`), 이미지 (`.png` / `.jpg` / `.jpeg`, OCR + caption), PDF (`.pdf`), **소스코드** (`.rs` → `code-rust-ast-v1`, `.py` → `code-python-ast-v1`, `.ts`/`.tsx` → `code-ts-ast-v1`, `.js`/`.mjs`/`.cjs`/`.jsx` → `code-js-ast-v1`, `.go` → `code-go-ast-v1` — 모두 tree-sitter AST chunker). 다른 확장자는 자동 skip — `IngestItem.warnings` 에 사유 (`"unsupported media type: .docx"` 등), `IngestReport.skipped_by_extension` 에 카운트 분류, CLI / TUI summary 에 breakdown 표시. 코드 chunk 는 `citation.kind = "code"` 에 `citation.lang = "<lang>"` + `symbol` + line range 를 담고, SearchHit top-level 에 `code_lang` + `repo` (`.git/` walk-up 의 디렉토리 이름) 가 backfill 됨. `--code-lang rust` / `--code-lang python` / `--code-lang typescript` / `--code-lang javascript` / `--code-lang go` / `--media code` filter 로 언어별·코드 전용 검색 가능 (p10-1A-1 filter flags). Python symbol 은 workspace 경로 → dotted module path prefix (예: `kebab_eval.metrics.compute_mrr`), TS/JS symbol 은 slash-style module path prefix (예: `src/Foo.Foo.search`), Go symbol 은 `package.Func` / `package.(*Receiver).Method` 형식. |
 | `kebab search --mode {lexical,vector,hybrid} "<query>" [--no-cache] [--max-tokens N] [--snippet-chars N] [--cursor <opaque>] [--tag T] [--lang L] [--path-glob G] [--trust-min LEVEL] [--media TYPE] [--ingested-after RFC3339] [--doc-id ID] [--trace] [--bulk] [--repo NAME ...] [--code-lang LIST]` | 검색. hybrid는 RRF fusion, citation 포함. 같은 process 안에서 동일 query (NFKC + trim + lowercase 정규화) 반복 시 in-process LRU 캐시 hit (capacity = `[search] cache_capacity`, default 256). `--no-cache` 로 강제 bypass — 디버깅용. ingest commit 발생 시 `kv['corpus_revision']` bump 으로 모든 entry 자동 stale. **`--max-tokens` / `--snippet-chars` / `--cursor` (p9-fb-34)** — agent budget controls. `--json` 출력은 `search_response.v1` wrapper (`{hits, next_cursor, truncated}`) — pre-fb-34 의 bare array 와 호환 안 됨. mismatched cursor → `error.v1.code = stale_cursor`. **filter flags (p9-fb-36):** `--tag` 는 반복 가능 flag (`--tag rust --tag async`) 로 OR 매칭, `--media` 는 `,` 구분 다중 값 OR 매칭, 나머지 flags 간은 AND 조합. `--trust-min` 은 `primary\|secondary\|generated` 중 하나 (해당 level 이상 포함). `--ingested-after` 는 RFC3339 UTC — 파싱 실패 시 `error.v1.code = config_invalid` (exit 2). `--media md` 는 `markdown` alias 로 정규화. 알 수 없는 `--media` 값은 무조건 empty hits (오류 아님). **`--trace` (p9-fb-37)** — `search_response.v1.trace` 에 lexical / vector pre-fusion 후보 + RRF union + per-stage timing (`lexical_ms` / `vector_ms` / `fusion_ms` / `total_ms`) 노출. trace 요청은 캐시 우회 (`--no-cache` 없이도 항상 cold). **`--bulk` (p9-fb-42)** — stdin ndjson 으로 N query 한 번에 실행. `--json` 면 stdout per-query ndjson (`bulk_search_item.v1`) + stderr summary (`bulk_summary: total=N succeeded=S failed=F`). Cap 100. agent 가 query decomposition 후 sub-query 일괄 실행 시 single round-trip — App instance 재사용으로 캐시 / embedder cold-start 비용 한 번만. Per-query failure 는 item 의 `error` (error.v1) 에 격리, 다른 query 계속 진행. **code corpus filters (p10-1A-1):** `--repo` 는 반복 가능 (`--repo kebab --repo other`) OR 매칭. `--code-lang` 는 반복 또는 comma 다중 값 (`--code-lang rust,python`), 알 수 없는 값은 빈 hits. `--media code` 는 Tier 1/2/3 모든 code chunk 포함. 1A-1 시점에서는 indexed 된 code chunk 가 없어 filter 가 항상 빈 결과 — 1A-2 (Rust AST chunker) 머지 이후 실효. |
 | `kebab list docs` | 색인된 문서 목록 |
 | `kebab inspect doc <id>` / `kebab inspect chunk <id>` | raw record 보기 |
@@ -132,7 +132,7 @@ flowchart TB

    subgraph Pipeline["도메인 + 파이프라인"]
        parse["parse-md / parse-pdf / parse-image / parse-code"]
-        chunker["chunker (md-heading-v1, pdf-page-v1, code-rust-ast-v1, code-python-ast-v1, code-ts-ast-v1, code-js-ast-v1)"]
+        chunker["chunker (md-heading-v1, pdf-page-v1, code-rust-ast-v1, code-python-ast-v1, code-ts-ast-v1, code-js-ast-v1, code-go-ast-v1)"]
        embedder["embedder (fastembed multilingual-e5-large)"]
        retriever["retriever (lexical / vector / hybrid RRF)"]
        rag["RAG pipeline"]
--- a/crates/kebab-app/src/fetch.rs
+++ b/crates/kebab-app/src/fetch.rs
@@ -189,10 +189,12 @@ fn fetch_span(
    // (markdown / note / paper / reference / inbox) is the *user-facing*
    // category, not the rendering format — the actual byte-level format
    // lives on the source `RawAsset.media_type`. Look it up via
-    // workspace_path (unique key per asset).
-    if let Some(asset) = <kebab_store_sqlite::SqliteStore as DocumentStore>::get_asset_by_workspace_path(
+    // doc.source_asset_id (PRIMARY KEY) so twin files (identical content
+    // at different paths) always read *this* document's own asset row,
+    // not whichever twin last wrote `assets.workspace_path`.
+    if let Some(asset) = <kebab_store_sqlite::SqliteStore as DocumentStore>::get_asset(
        &app.sqlite,
-        &doc.workspace_path,
+        &doc.source_asset_id,
    )? {
        if matches!(
            asset.media_type,
--- a/crates/kebab-app/src/lib.rs
+++ b/crates/kebab-app/src/lib.rs
@@ -39,7 +39,7 @@ use std::sync::Arc;
 use anyhow::{Context, anyhow};
 use serde::{Deserialize, Serialize};

-use kebab_chunk::{CodeJsAstV1Chunker, CodePythonAstV1Chunker, CodeRustAstV1Chunker, CodeTsAstV1Chunker, MdHeadingV1Chunker, PdfPageV1Chunker};
+use kebab_chunk::{CodeGoAstV1Chunker, CodeJsAstV1Chunker, CodePythonAstV1Chunker, CodeRustAstV1Chunker, CodeTsAstV1Chunker, MdHeadingV1Chunker, PdfPageV1Chunker};
 use kebab_core::{
    Answer, Block, CanonicalDocument, Chunk, ChunkId, ChunkPolicy, ChunkerVersion, Chunker,
    DocFilter, DocSummary, DocumentId, DocumentStore, Embedder, EmbeddingInput,
@@ -50,7 +50,7 @@ use kebab_core::{
 use kebab_llm_local::OllamaLanguageModel;
 use kebab_normalize::build_canonical_document;
 use kebab_parse_image::{ImageExtractor, OllamaVisionOcr, apply_caption, apply_ocr};
-use kebab_parse_code::{JavascriptAstExtractor, PythonAstExtractor, RustAstExtractor, TypescriptAstExtractor};
+use kebab_parse_code::{GoAstExtractor, JavascriptAstExtractor, PythonAstExtractor, RustAstExtractor, TypescriptAstExtractor};
 use kebab_parse_pdf::PdfTextExtractor;
 use kebab_parse_md::{BodyHints, parse_blocks, parse_frontmatter};
 use kebab_source_fs::FsSourceConnector;
@@ -71,7 +71,7 @@ mod staleness;

 pub use app::{App, SearchResponse};
 pub use ingest_progress::{AggregateCounts, IngestEvent, render_skipped_breakdown};
-pub use reset::{ResetReport, ResetScope};
+pub use reset::{ResetReport, ResetScope, enumerate_orphans};
 pub use error_wire::{ERROR_V1_ID, ErrorV1, StructuredError, classify};
 pub use fetch::fetch_with_config;
 #[doc(hidden)]
@@ -375,6 +375,28 @@ pub fn ingest_with_config_opts(
        .map(|d| d.doc_id.0)
        .collect();

+    // Dogfood: post-walker sweep to remove stored docs whose source
+    // file has been deleted from the filesystem. Must run BEFORE the
+    // per-asset loop so the loop's New/Updated labelling is based on
+    // the post-purge store state (the purged doc_ids won't be in
+    // `existing_doc_ids` above — they were already removed, OR the
+    // sweep here removes them before we start counting).
+    //
+    // Critical design invariant: only purge when the file is TRULY
+    // absent from disk. A file that is still on disk but outside the
+    // current walker scope (config narrowing / include-glob change) is
+    // NOT purged — we leave it in place to protect against accidental
+    // data loss via config edits.
+    let scanned_paths: std::collections::HashSet<kebab_core::WorkspacePath> = assets
+        .iter()
+        .map(|a| a.workspace_path.clone())
+        .collect();
+    let purged_deleted_files = sweep_deleted_files(
+        &app,
+        &scanned_paths,
+        vector_store.as_ref().map(|v| v.as_ref()),
+    )?;
+
    let started_at = time::OffsetDateTime::now_utc();

    let mut items: Vec<kebab_core::IngestItem> = Vec::new();
@@ -647,11 +669,11 @@ pub fn ingest_with_config_opts(
    crate::ingest_progress::emit(progress, terminal_event);

    // p9-fb-19: bump the persistent corpus_revision counter when a
-    // commit landed (any new / updated). This invalidates every
+    // commit landed (any new / updated / purged). This invalidates every
    // entry in any in-process LRU search cache (in this process or
    // a sibling) on the next lookup. No-op when nothing changed
    // (skipped-only run) — the cache stays valid.
-    if new_count > 0 || updated_count > 0 {
+    if new_count > 0 || updated_count > 0 || purged_deleted_files > 0 {
        match app.sqlite.bump_corpus_revision() {
            Ok(rev) => tracing::debug!(
                target: "kebab-app",
@@ -682,6 +704,7 @@ pub fn ingest_with_config_opts(
        skipped_generated: fs_skips.skipped_generated,
        skipped_size_exceeded: fs_skips.skipped_size_exceeded,
        skip_examples: fs_skips.skip_examples,
+        purged_deleted_files,
        items: if summary_only { None } else { Some(items) },
    })
 }
@@ -748,15 +771,18 @@ struct ImagePipeline<'a> {
 /// hold (per design §9 cascade rule):
 ///
 /// 1. `force_reingest == false` — caller hasn't asked to bypass skip.
-/// 2. The freshly-scanned asset's blake3 checksum equals what the
-///    existing `assets` row stores at the same `workspace_path`.
-/// 3. The doc keyed on `(workspace_path, asset_id, current_parser_version)`
-///    exists. If the parser_version changed, `id_for_doc` produces a
-///    different `doc_id` so the lookup misses → no skip → re-process.
-/// 4. The existing doc's stamped `last_chunker_version` AND
-///    `last_embedding_version` match the values the caller is about
-///    to use (`Some(v) == Some(v)` and `None == None` — see design
-///    doc for the `None == None` rule when no embedder is configured).
+/// 2. A document already exists at this `workspace_path`
+///    (`get_document_by_workspace_path`). The lookup is document-side, not
+///    asset-side, so twin files (identical content at different paths) each
+///    hit their own stable doc row — `documents.workspace_path` is UNIQUE
+///    while `assets` may dedupe content into a single row with a flip-flop
+///    `workspace_path` column (dogfood bug #4, see `tasks/HOTFIXES.md`).
+/// 3. The existing doc's `source_asset_id` equals the freshly-scanned
+///    asset's blake3 checksum (content unchanged).
+/// 4. The existing doc's `parser_version` matches the current extractor's
+///    `parser_version` (extractor not upgraded). Combined with `chunker_version`
+///    and `last_embedding_version` checks immediately below — full cascade
+///    per design §9.
 ///
 /// Returns `Ok(None)` (proceed with full re-process) when any check
 /// fails or any DB read errors out — the skip path is opportunistic;
@@ -773,31 +799,19 @@ fn try_skip_unchanged(
    if force_reingest {
        return Ok(None);
    }
-    let existing_asset = match app
+    // Document-centric skip: look up the existing document row by
+    // workspace_path directly. This avoids the twin-file flip-flop
+    // that the old asset-side lookup suffers from — multiple files
+    // with identical content share one `assets` row whose
+    // `workspace_path` is overwritten on every UPSERT, so
+    // `get_asset_by_workspace_path(path1)` could return the OTHER
+    // twin's path (or None) after any ingest of the twin. The
+    // `documents` table has a UNIQUE index on `workspace_path` (V001),
+    // so each twin has its own stable row regardless of asset de-dup.
+    let existing_doc = match app
        .sqlite
-        .get_asset_by_workspace_path(&asset.workspace_path)
+        .get_document_by_workspace_path(&asset.workspace_path)
    {
-        Ok(Some(a)) => a,
-        Ok(None) => return Ok(None),
-        Err(e) => {
-            tracing::debug!(
-                target: "kebab-app",
-                path = %asset.workspace_path.0,
-                error = %e,
-                "skip-check: get_asset_by_workspace_path failed; falling through to re-process"
-            );
-            return Ok(None);
-        }
-    };
-    if existing_asset.checksum != asset.checksum {
-        return Ok(None);
-    }
-    let candidate_doc_id = kebab_core::id_for_doc(
-        &asset.workspace_path,
-        &asset.asset_id,
-        current_parser_version,
-    );
-    let existing_doc = match app.sqlite.get_document(&candidate_doc_id) {
        Ok(Some(d)) => d,
        Ok(None) => return Ok(None),
        Err(e) => {
@@ -805,21 +819,37 @@ fn try_skip_unchanged(
                target: "kebab-app",
                path = %asset.workspace_path.0,
                error = %e,
-                "skip-check: get_document failed; falling through to re-process"
+                "skip-check: get_document_by_workspace_path failed; falling through to re-process"
            );
            return Ok(None);
        }
    };
+    // 1. Content unchanged: the freshly-computed asset_id (blake3
+    //    content hash) must match what this document was ingested from.
+    if existing_doc.source_asset_id != asset.asset_id {
+        return Ok(None);
+    }
+    // 2. Parser unchanged: parser_version is baked into id_for_doc so
+    //    a version bump yields a different doc_id and the row above
+    //    would have been missing. Checking here explicitly keeps the
+    //    logic self-documenting and guards against future id_for_doc
+    //    changes.
+    if existing_doc.parser_version != *current_parser_version {
+        return Ok(None);
+    }
+    // 3. Chunker unchanged.
    let chunker_match = existing_doc.last_chunker_version.as_ref()
        == Some(current_chunker_version);
    if !chunker_match {
        return Ok(None);
    }
+    // 4. Embedder unchanged.
    let embedder_match = existing_doc.last_embedding_version.as_ref()
        == current_embedding_version;
    if !embedder_match {
        return Ok(None);
    }
+    let candidate_doc_id = existing_doc.doc_id.clone();
    tracing::debug!(
        target: "kebab-app::ingest",
        path = %asset.workspace_path.0,
@@ -920,7 +950,7 @@ fn ingest_one_asset(
        }
        // p10-1A-2 / 1B: code ingest dispatch.
        MediaType::Code(lang)
-            if matches!(lang.as_str(), "rust" | "python" | "typescript" | "javascript") =>
+            if matches!(lang.as_str(), "rust" | "python" | "typescript" | "javascript" | "go") =>
        {
            return ingest_one_code_asset(
                app,
@@ -1446,6 +1476,120 @@ fn purge_vector_orphans_for_workspace_path(
    Ok(())
 }

+/// Dogfood: post-walker sweep that purges stored documents whose source
+/// file has been physically deleted from the filesystem.
+///
+/// Algorithm:
+/// 1. Query `documents` for every `workspace_path` currently stored.
+/// 2. Compute `orphan_candidates = stored_paths - scanned_paths`.
+/// 3. For each candidate: resolve to an absolute path and call
+///    `Path::try_exists().unwrap_or(true)` — transient FS errors
+///    (EACCES, NFS hiccup, ownership change) conservatively count as
+///    "still present" so we never purge on uncertain signal. If the
+///    file still exists on disk it was merely out-of-scope this run
+///    (config narrowing / include-glob change) — leave it alone. Only
+///    files that are truly absent trigger a purge.
+/// 4. For absent files: call `purge_deleted_workspace_path` (SQLite
+///    cascade delete + optional copied-asset file removal) and, if a
+///    vector store is present, delete the associated vectors.
+///
+/// Returns the number of documents purged.
+///
+/// Non-fatal design: individual purge failures are logged and counted
+/// as errors on the per-file level but do NOT abort the sweep — a
+/// partial failure is preferable to blocking the rest of ingest. The
+/// return value only counts successful purges.
+fn sweep_deleted_files(
+    app: &App,
+    scanned_paths: &std::collections::HashSet<kebab_core::WorkspacePath>,
+    vector_store: Option<&kebab_store_vector::LanceVectorStore>,
+) -> anyhow::Result<u32> {
+    use kebab_core::DocumentStore as _;
+
+    let stored_paths = app
+        .sqlite
+        .all_workspace_paths()
+        .context("sweep_deleted_files: all_workspace_paths")?;
+
+    if stored_paths.is_empty() {
+        return Ok(0);
+    }
+
+    let workspace_root = app.config.resolve_workspace_root();
+    let mut purged: u32 = 0;
+
+    for stored_path in stored_paths {
+        if scanned_paths.contains(&stored_path) {
+            continue; // still in scope — skip
+        }
+
+        // Resolve to an absolute path and check existence on disk.
+        // Use `try_exists` + `unwrap_or(true)` so transient FS errors
+        // (EACCES on a path we lack read on, NFS hiccups, ownership
+        // change) are CONSERVATIVELY treated as "file still present" —
+        // never purge on uncertain signal (data-safety: PR #148 review).
+        // `exists()` would return false on Err and trigger a wrongful
+        // purge. Files whose path cannot be joined (theoretically
+        // impossible for non-empty workspace_path strings, but
+        // defense-in-depth) are likewise treated as still present.
+        let abs = workspace_root.join(&stored_path.0);
+        if abs.try_exists().unwrap_or(true) {
+            // File is on disk but not in this scan's scope (config
+            // narrowing). DO NOT purge — critical design constraint.
+            tracing::debug!(
+                target: "kebab-app",
+                path = %stored_path.0,
+                "sweep_deleted_files: file on disk but out of scope — leaving in store"
+            );
+            continue;
+        }
+
+        // File is truly absent → purge.
+        let chunk_ids = match kebab_store_sqlite::purge_deleted_workspace_path(
+            &app.sqlite,
+            &stored_path,
+        ) {
+            Ok(ids) => ids,
+            Err(e) => {
+                tracing::warn!(
+                    target: "kebab-app",
+                    path = %stored_path.0,
+                    error = %e,
+                    "sweep_deleted_files: purge failed; skipping this path"
+                );
+                continue;
+            }
+        };
+
+        // Purge associated vectors (best-effort; partial failure
+        // acceptable — orphan vectors get cleaned by `kebab reset
+        // --vector-only` if they accumulate).
+        if let Some(vec) = vector_store {
+            if !chunk_ids.is_empty() {
+                use kebab_core::VectorStore as _;
+                if let Err(e) = vec.delete_by_chunk_ids(&chunk_ids) {
+                    tracing::warn!(
+                        target: "kebab-app",
+                        path = %stored_path.0,
+                        count = chunk_ids.len(),
+                        error = %e,
+                        "sweep_deleted_files: vector delete failed; SQLite side already cleaned"
+                    );
+                }
+            }
+        }
+
+        tracing::info!(
+            target: "kebab-app",
+            path = %stored_path.0,
+            "sweep_deleted_files: purged document for deleted file"
+        );
+        purged = purged.saturating_add(1);
+    }
+
+    Ok(purged)
+}
+
 /// P7-3: process one `MediaType::Pdf` asset end-to-end.
 ///
 /// - Reads bytes from disk.
@@ -1683,6 +1827,7 @@ fn ingest_one_code_asset(
        "python"     => ParserVersion(kebab_parse_code::PYTHON_PARSER_VERSION.to_string()),
        "typescript" => ParserVersion(kebab_parse_code::TS_PARSER_VERSION.to_string()),
        "javascript" => ParserVersion(kebab_parse_code::JS_PARSER_VERSION.to_string()),
+        "go" => ParserVersion(kebab_parse_code::GO_PARSER_VERSION.to_string()),
        other => anyhow::bail!("unsupported code_lang: {other}"),
    };

@@ -1692,6 +1837,7 @@ fn ingest_one_code_asset(
        "python"     => CodePythonAstV1Chunker.chunker_version(),
        "typescript" => CodeTsAstV1Chunker.chunker_version(),
        "javascript" => CodeJsAstV1Chunker.chunker_version(),
+        "go" => CodeGoAstV1Chunker.chunker_version(),
        other => anyhow::bail!("unreachable chunker_version: {other}"),
    };

@@ -1730,6 +1876,9 @@ fn ingest_one_code_asset(
        "javascript" => JavascriptAstExtractor::new()
            .extract(&ctx, &bytes)
            .context("kb-parse-code::JavascriptAstExtractor::extract (code:javascript)")?,
+        "go" => GoAstExtractor::new()
+            .extract(&ctx, &bytes)
+            .context("kb-parse-code::GoAstExtractor::extract (code:go)")?,
        other => anyhow::bail!("unreachable (extract): {other}"),
    };

@@ -1747,6 +1896,9 @@ fn ingest_one_code_asset(
        "javascript" => CodeJsAstV1Chunker
            .chunk(&canonical, chunk_policy)
            .context("kb-chunk::CodeJsAstV1Chunker::chunk (code:javascript)")?,
+        "go" => CodeGoAstV1Chunker
+            .chunk(&canonical, chunk_policy)
+            .context("kb-chunk::CodeGoAstV1Chunker::chunk (code:go)")?,
        other => anyhow::bail!("unreachable (chunk): {other}"),
    };

--- a/crates/kebab-app/src/reset.rs
+++ b/crates/kebab-app/src/reset.rs
@@ -9,13 +9,19 @@
 //!
 //! `--vector-only` additionally truncates `embedding_records` in SQLite
 //! so the next `kebab ingest` re-embeds cleanly without orphan rows.
+//!
+//! `--orphans-only` purges stored docs that are outside the current walker
+//! scope (config narrowing / removed sub-directory). No filesystem paths are
+//! removed — this is purely a store-level reconciliation.

+use std::collections::HashSet;
 use std::path::PathBuf;

 use anyhow::{Context, Result};
 use serde::{Deserialize, Serialize};

 use kebab_config::{Config, expand_path};
+use kebab_core::WorkspacePath;

 /// What the user asked to remove. Mutually exclusive — picked by the CLI
 /// from a clap `ArgGroup`.
@@ -32,6 +38,13 @@ pub enum ResetScope {
    VectorOnly,
    /// Wipe only the config dir.
    ConfigOnly,
+    /// Purge stored docs that are outside the current walker scope (no
+    /// filesystem paths are removed). Filesystem existence is NOT checked —
+    /// anything the current walker would not visit is considered an orphan.
+    /// The explicit complement to the conservative `sweep_deleted_files`
+    /// that runs during ingest (which leaves on-disk-but-out-of-scope docs
+    /// alone for data safety).
+    OrphansOnly,
 }

 /// Result of a successful wipe — emitted as `reset_report.v1` by the
@@ -41,6 +54,16 @@ pub struct ResetReport {
    pub scope: ResetScope,
    pub removed_paths: Vec<PathBuf>,
    pub embedding_rows_truncated: u64,
+    /// Number of stored docs purged because they are outside the current
+    /// walker scope. Non-zero only when `scope == OrphansOnly`.
+    /// `#[serde(default)]` preserves back-compat with older callers that
+    /// do not include this field.
+    #[serde(default)]
+    pub orphans_purged: u32,
+    /// Paths of the orphaned docs that were purged. Sorted for deterministic
+    /// output. Non-empty only when `scope == OrphansOnly`.
+    #[serde(default)]
+    pub purged_paths: Vec<WorkspacePath>,
 }

 /// Compute the absolute on-disk paths a given scope will wipe, given a
@@ -67,6 +90,10 @@ pub fn enumerate_paths(scope: ResetScope, cfg: &Config) -> Vec<PathBuf> {
            vec![vector_dir]
        }
        ResetScope::ConfigOnly => vec![cfg_dir],
+        // OrphansOnly operates purely at the store level — no filesystem paths
+        // are removed. Return empty so `estimate_size_bytes` stays zero and
+        // the existing confirm UI path for directory wipes is skipped.
+        ResetScope::OrphansOnly => vec![],
    }
 }

@@ -96,16 +123,82 @@ pub fn estimate_size_bytes(paths: &[PathBuf]) -> u64 {
    paths.iter().map(|p| walk(p)).sum()
 }

+/// Compute the workspace paths stored in SQLite that are NOT visited by
+/// the current walker scope (i.e. they are "orphans" — on disk but
+/// outside the configured include/exclude rules, or from a sub-directory
+/// that has since been removed from the workspace).
+///
+/// Does NOT check filesystem existence — `OrphansOnly` is the explicit
+/// "I know what I'm doing" variant; callers that want the conservative
+/// fs-aware sweep should use `sweep_deleted_files` inside ingest.
+///
+/// Returns the list sorted for deterministic output. Called twice by the
+/// CLI path (once for the confirm UI preview, once inside `execute`);
+/// the double scan is acceptable for a rare destructive operation.
+pub fn enumerate_orphans(cfg: &Config) -> Result<Vec<WorkspacePath>> {
+    use kebab_core::DocumentStore as _;
+    use kebab_source_fs::FsSourceConnector;
+    use kebab_core::SourceScope;
+
+    let store = kebab_store_sqlite::SqliteStore::open(cfg)
+        .context("enumerate_orphans: open SqliteStore")?;
+
+    let stored = store
+        .all_workspace_paths()
+        .context("enumerate_orphans: all_workspace_paths")?;
+
+    if stored.is_empty() {
+        return Ok(Vec::new());
+    }
+
+    // Build the same SourceScope the CLI's ingest path uses: root from
+    // config, exclude list from config, no include override (full scope).
+    let root = cfg.resolve_workspace_root();
+    let scope = SourceScope {
+        root: root.clone(),
+        exclude: cfg.workspace.exclude.clone(),
+        ..Default::default()
+    };
+
+    let connector = FsSourceConnector::new(cfg)
+        .context("enumerate_orphans: build FsSourceConnector")?;
+    let (assets, _skips) = connector
+        .scan_with_skips(&scope)
+        .context("enumerate_orphans: scan workspace")?;
+
+    let scanned: HashSet<WorkspacePath> = assets
+        .into_iter()
+        .map(|a| a.workspace_path)
+        .collect();
+
+    let mut orphans: Vec<WorkspacePath> = stored
+        .into_iter()
+        .filter(|p| !scanned.contains(p))
+        .collect();
+    orphans.sort_by(|a, b| a.0.cmp(&b.0));
+    Ok(orphans)
+}
+
 /// Wipe every path from `enumerate_paths(scope, cfg)`. For
 /// `ResetScope::VectorOnly`, also truncates the SQLite
 /// `embedding_records` table so the store doesn't point at the Lance
 /// rows we just removed off-disk.
 ///
+/// For `ResetScope::OrphansOnly`, no filesystem directories are removed.
+/// Instead the store is reconciled: stored docs outside the current walker
+/// scope are purged from SQLite (+ vector store when configured). The
+/// caller is expected to have already shown the confirm UI using
+/// `enumerate_orphans`.
+///
 /// Idempotent: a missing path is treated as already-removed (success).
 /// Returns a `ResetReport` listing exactly what was removed (paths that
 /// existed before the call) so `--json` callers see the truth, not the
 /// request.
 pub fn execute(scope: ResetScope, cfg: &Config) -> Result<ResetReport> {
+    if matches!(scope, ResetScope::OrphansOnly) {
+        return execute_orphans_only(cfg);
+    }
+
    let paths = enumerate_paths(scope, cfg);
    let mut removed = Vec::new();

@@ -128,9 +221,100 @@ pub fn execute(scope: ResetScope, cfg: &Config) -> Result<ResetReport> {
        scope,
        removed_paths: removed,
        embedding_rows_truncated,
+        orphans_purged: 0,
+        purged_paths: Vec::new(),
    })
 }

+/// Execute the `OrphansOnly` variant: reconcile stored docs against the
+/// current walker scope without touching any filesystem directory.
+fn execute_orphans_only(cfg: &Config) -> Result<ResetReport> {
+    let orphans = enumerate_orphans(cfg)
+        .context("execute_orphans_only: enumerate orphans")?;
+
+    if orphans.is_empty() {
+        return Ok(ResetReport {
+            scope: ResetScope::OrphansOnly,
+            removed_paths: Vec::new(),
+            embedding_rows_truncated: 0,
+            orphans_purged: 0,
+            purged_paths: Vec::new(),
+        });
+    }
+
+    let store = std::sync::Arc::new(
+        kebab_store_sqlite::SqliteStore::open(cfg)
+            .context("execute_orphans_only: open SqliteStore")?,
+    );
+
+    // Open vector store if configured. Mirror the same guard the ingest
+    // path uses: only construct when the provider is not "none" / dims > 0.
+    let vector_store: Option<kebab_store_vector::LanceVectorStore> =
+        open_vector_store_if_configured(cfg, store.clone())?;
+
+    let mut purged_paths: Vec<WorkspacePath> = Vec::new();
+
+    for path in &orphans {
+        let chunk_ids = kebab_store_sqlite::purge_deleted_workspace_path(&store, path)
+            .with_context(|| format!("execute_orphans_only: purge {}", path.0))?;
+
+        if let Some(ref vs) = vector_store {
+            if !chunk_ids.is_empty() {
+                use kebab_core::VectorStore as _;
+                if let Err(e) = vs.delete_by_chunk_ids(&chunk_ids) {
+                    tracing::warn!(
+                        target: "kebab-app",
+                        path = %path.0,
+                        count = chunk_ids.len(),
+                        error = %e,
+                        "reset --orphans-only: vector delete failed; SQLite side already cleaned"
+                    );
+                }
+            }
+        }
+
+        tracing::info!(
+            target: "kebab-app",
+            path = %path.0,
+            "reset --orphans-only: purged orphan document"
+        );
+        purged_paths.push(path.clone());
+    }
+
+    let orphans_purged = u32::try_from(purged_paths.len()).unwrap_or(u32::MAX);
+
+    Ok(ResetReport {
+        scope: ResetScope::OrphansOnly,
+        removed_paths: Vec::new(),
+        embedding_rows_truncated: 0,
+        orphans_purged,
+        purged_paths,
+    })
+}
+
+/// Open the Lance vector store if the configured embedding provider is
+/// active (non-"none", dimensions > 0). Returns `None` for lexical-only
+/// configs. Mirrors the guard in `App::vector`.
+fn open_vector_store_if_configured(
+    cfg: &Config,
+    store: std::sync::Arc<kebab_store_sqlite::SqliteStore>,
+) -> Result<Option<kebab_store_vector::LanceVectorStore>> {
+    if cfg.models.embedding.provider == "none" || cfg.models.embedding.dimensions == 0 {
+        return Ok(None);
+    }
+    match kebab_store_vector::LanceVectorStore::new(cfg, store) {
+        Ok(vs) => Ok(Some(vs)),
+        Err(e) => {
+            tracing::warn!(
+                target: "kebab-app",
+                error = %e,
+                "reset --orphans-only: could not open vector store; skipping vector delete"
+            );
+            Ok(None)
+        }
+    }
+}
+
 /// Open the SQLite store at the configured path and run
 /// `truncate_embedding_records`. Returns the count of truncated rows
 /// (the helper itself reports `DELETE` rowcount). If the SQLite file
@@ -200,4 +384,14 @@ mod tests {
        let bytes = estimate_size_bytes(&[dir.path().to_path_buf()]);
        assert_eq!(bytes, 5 + 6);
    }
+
+    #[test]
+    fn enumerate_orphans_only_returns_empty_paths() {
+        let cfg = Config::defaults();
+        let paths = enumerate_paths(ResetScope::OrphansOnly, &cfg);
+        assert!(
+            paths.is_empty(),
+            "OrphansOnly must return empty vec from enumerate_paths"
+        );
+    }
 }
--- a/crates/kebab-app/src/schema.rs
+++ b/crates/kebab-app/src/schema.rs
@@ -168,7 +168,9 @@ fn collect_stats(
        stale_doc_count: counts.stale_doc_count,
        // p10-1A-2: populated by the store query added in this task.
        code_lang_breakdown: store.code_lang_breakdown()?,
-        repo_breakdown: std::collections::BTreeMap::new(),
+        // p10-1A-2 follow-up: dogfooding (2026-05-20) revealed this was a
+        // placeholder — mirror of code_lang_breakdown for the repo field.
+        repo_breakdown: store.repo_breakdown()?,
    })
 }

--- a/crates/kebab-app/tests/code_ingest_smoke.rs
+++ b/crates/kebab-app/tests/code_ingest_smoke.rs
@@ -390,6 +390,77 @@ fn javascript_file_ingests_and_searches_as_code_citation() {
    );
 }

+/// p10-1c-go Task F: a `.go` file in a sub-directory is ingested and the
+/// resulting `Citation::Code` hit must carry `lang="go"`,
+/// `symbol="chunk.ParseDoc"`, and `line_start >= 1`.
+/// The sub-directory (`chunk/`) ensures the Go package-prefix wiring
+/// produces a non-empty module prefix so the fully-qualified symbol assertion
+/// exercises that path end-to-end.
+#[test]
+fn go_file_ingests_and_searches_as_code_citation() {
+    let env = TestEnv::lexical_only();
+
+    let pkg_dir = env.workspace_root.join("chunk");
+    std::fs::create_dir_all(&pkg_dir).unwrap();
+    std::fs::write(
+        pkg_dir.join("ast.go"),
+        "package chunk\n\nfunc ParseDoc(input string) string {\n    return input\n}\n",
+    )
+    .unwrap();
+
+    let report = kebab_app::ingest_with_config(env.config.clone(), env.scope(), false)
+        .expect("ingest must succeed");
+    assert_eq!(report.errors, 0);
+    assert!(report.new >= 1);
+
+    let go_item = report
+        .items
+        .as_ref()
+        .expect("items present")
+        .iter()
+        .find(|i| i.doc_path.0.ends_with("ast.go"))
+        .expect("ast.go item present");
+    assert_eq!(
+        go_item.parser_version.as_ref().map(|p| p.0.as_str()),
+        Some("code-go-v1"),
+        "parser_version must be code-go-v1"
+    );
+    assert_eq!(
+        go_item.chunker_version.as_ref().map(|c| c.0.as_str()),
+        Some("code-go-ast-v1"),
+        "chunker_version must be code-go-ast-v1"
+    );
+
+    let hits = kebab_app::search_with_config(env.config.clone(), lexical_query("ParseDoc"))
+        .expect("search must succeed");
+    let h = hits
+        .iter()
+        .find(|h| matches!(&h.citation, kebab_core::Citation::Code { .. }))
+        .expect("Citation::Code hit");
+    match &h.citation {
+        kebab_core::Citation::Code {
+            lang,
+            symbol,
+            line_start,
+            ..
+        } => {
+            assert_eq!(lang.as_deref(), Some("go"), "citation.lang must be 'go'");
+            assert_eq!(
+                symbol.as_deref(),
+                Some("chunk.ParseDoc"),
+                "citation.symbol must be 'chunk.ParseDoc'"
+            );
+            assert!(*line_start >= 1, "line_start must be >=1");
+        }
+        _ => unreachable!(),
+    }
+    assert_eq!(
+        h.code_lang.as_deref(),
+        Some("go"),
+        "SearchHit.code_lang must be 'go'"
+    );
+}
+
 /// Re-ingesting the same `.rs` file without changes must report
 /// `Unchanged` (incremental-skip path exercised).
 #[test]
--- a/crates/kebab-app/tests/file_deletion_auto_purge.rs
+++ b/crates/kebab-app/tests/file_deletion_auto_purge.rs
@@ -0,0 +1,178 @@
+//! Dogfood: auto-purge stored docs for filesystem-deleted files.
+//!
+//! Two tests:
+//!
+//! 1. `file_deletion_auto_purge` — ingest 2 files, delete one, re-ingest.
+//!    The re-ingest must report `purged_deleted_files = 1`, the deleted
+//!    file must no longer appear in `list_docs`, and lexical search for
+//!    its unique content must return no hits.
+//!
+//! 2. `include_scope_narrowing_does_not_purge` — ingest 2 files under a
+//!    wide glob, narrow the walker scope to only one file, re-ingest.
+//!    The narrowed ingest must NOT purge the out-of-scope file because
+//!    the file is still on disk (just excluded from this run). Protects
+//!    users against accidental data loss via config edits.
+
+mod common;
+
+use common::TestEnv;
+use kebab_app::ingest_with_config_opts;
+use kebab_app::IngestOpts;
+use kebab_core::{DocFilter, DocumentStore, SearchMode, SearchQuery, SourceScope};
+
+/// Helper: open the store via `TestEnv` and run `list_documents`.
+fn list_doc_paths(env: &TestEnv) -> Vec<String> {
+    use kebab_store_sqlite::SqliteStore;
+    let store = SqliteStore::open(&env.config).unwrap();
+    store.run_migrations().unwrap();
+    store
+        .list_documents(&DocFilter::default())
+        .unwrap()
+        .into_iter()
+        .map(|d| d.doc_path.0)
+        .collect()
+}
+
+#[test]
+fn file_deletion_auto_purge() {
+    let env = TestEnv::lexical_only();
+
+    // Write two .rs files into the workspace.
+    let a_path = env.workspace_root.join("a.rs");
+    let b_path = env.workspace_root.join("b.rs");
+    std::fs::write(&a_path, "// file a\nfn alpha() {}\n").unwrap();
+    std::fs::write(&b_path, "// file b\nfn bravo() {}\n").unwrap();
+
+    // First ingest — both must be New.
+    let first = ingest_with_config_opts(
+        env.config.clone(),
+        env.scope(),
+        false,
+        IngestOpts::default(),
+    )
+    .expect("first ingest must succeed");
+    // Only count the .rs files we added (there may be fixture files too).
+    let first_new = first.new;
+    assert!(first_new >= 2, "expected at least 2 new docs: {first:?}");
+    assert_eq!(
+        first.purged_deleted_files, 0,
+        "no purges on first ingest: {first:?}"
+    );
+    assert_eq!(first.errors, 0, "no errors on first ingest: {first:?}");
+
+    // Delete one file from the filesystem.
+    std::fs::remove_file(&b_path).expect("remove b.rs");
+
+    // Second ingest — scanned count drops by 1; b.rs should be purged.
+    let second = ingest_with_config_opts(
+        env.config.clone(),
+        env.scope(),
+        false,
+        IngestOpts::default(),
+    )
+    .expect("second ingest must succeed");
+
+    assert_eq!(
+        second.purged_deleted_files, 1,
+        "exactly 1 file should be purged: {second:?}"
+    );
+    assert_eq!(second.new, 0, "no new docs after deletion: {second:?}");
+    assert_eq!(second.updated, 0, "no updated docs: {second:?}");
+    assert_eq!(second.errors, 0, "no errors: {second:?}");
+
+    // b.rs must no longer appear in list_docs.
+    let doc_paths = list_doc_paths(&env);
+    let b_ws_path = "b.rs";
+    assert!(
+        !doc_paths.iter().any(|p| p == b_ws_path),
+        "b.rs must be gone from list_docs; got: {doc_paths:?}"
+    );
+    // a.rs must still be present.
+    let a_ws_path = "a.rs";
+    assert!(
+        doc_paths.iter().any(|p| p == a_ws_path),
+        "a.rs must still be in list_docs; got: {doc_paths:?}"
+    );
+
+    // Lexical search for b.rs's unique content returns no hits.
+    let app = env.app();
+    let query = SearchQuery {
+        text: "bravo".to_string(),
+        mode: SearchMode::Lexical,
+        k: 10,
+        filters: kebab_core::SearchFilters::default(),
+    };
+    let hits = app.search(query).expect("search must not error");
+    assert!(
+        hits.is_empty(),
+        "search for deleted file's content must return no hits; got: {hits:?}"
+    );
+}
+
+#[test]
+fn include_scope_narrowing_does_not_purge() {
+    let env = TestEnv::lexical_only();
+
+    // Write two .rs files.
+    let a_path = env.workspace_root.join("a_narrow.rs");
+    let b_path = env.workspace_root.join("b_narrow.rs");
+    std::fs::write(&a_path, "// narrow a\nfn alpha_narrow() {}\n").unwrap();
+    std::fs::write(&b_path, "// narrow b\nfn bravo_narrow() {}\n").unwrap();
+
+    // Wide scope: first ingest — both must be New.
+    let wide_scope = SourceScope {
+        root: env.workspace_root.clone(),
+        include: vec!["**/*.rs".to_string()],
+        exclude: env.config.workspace.exclude.clone(),
+    };
+    let first = ingest_with_config_opts(
+        env.config.clone(),
+        wide_scope,
+        false,
+        IngestOpts::default(),
+    )
+    .expect("first ingest (wide) must succeed");
+    assert!(
+        first.new >= 2,
+        "expected at least 2 new docs: {first:?}"
+    );
+    assert_eq!(
+        first.purged_deleted_files, 0,
+        "no purges on first ingest: {first:?}"
+    );
+
+    // Narrow scope: only a_narrow.rs in include — b_narrow.rs is still
+    // on disk but excluded from the walker scope.
+    let narrow_scope = SourceScope {
+        root: env.workspace_root.clone(),
+        include: vec!["a_narrow.rs".to_string()],
+        exclude: env.config.workspace.exclude.clone(),
+    };
+    let second = ingest_with_config_opts(
+        env.config.clone(),
+        narrow_scope,
+        false,
+        IngestOpts::default(),
+    )
+    .expect("second ingest (narrow) must succeed");
+
+    // CRITICAL: b_narrow.rs is still on disk — must NOT be purged.
+    assert_eq!(
+        second.purged_deleted_files, 0,
+        "scope-narrowing must NOT purge on-disk files; got: {second:?}"
+    );
+    assert_eq!(second.errors, 0, "no errors: {second:?}");
+
+    // b_narrow.rs must still exist in the store.
+    let doc_paths = list_doc_paths(&env);
+    let b_ws_path = "b_narrow.rs";
+    assert!(
+        doc_paths.iter().any(|p| p == b_ws_path),
+        "b_narrow.rs must still be in list_docs after scope narrowing; got: {doc_paths:?}"
+    );
+    // And the file must still be on disk.
+    assert!(
+        b_path.exists(),
+        "b_narrow.rs must still be on disk (we didn't delete it)"
+    );
+}
--- a/crates/kebab-app/tests/reset_orphans.rs
+++ b/crates/kebab-app/tests/reset_orphans.rs
@@ -0,0 +1,141 @@
+//! Integration test for `kebab reset --orphans-only`.
+//!
+//! Verifies that stored docs outside the current walker scope are purged
+//! from the store without removing any files from the filesystem.
+//!
+//! Test outline:
+//! 1. Ingest 3 .rs files (a.rs, b.rs, c.rs) — all New.
+//! 2. Narrow the config `include` to `["a.rs"]` only; b.rs and c.rs are
+//!    still on disk but outside the walker scope.
+//! 3. Run `execute(ResetScope::OrphansOnly, &cfg)` — report must show
+//!    `orphans_purged == 2` and `purged_paths` contains b.rs + c.rs.
+//! 4. `list docs` must show only a.rs.
+//! 5. b.rs and c.rs must still exist on disk (no filesystem removal).
+//! 6. Second reset → `orphans_purged == 0` (idempotent).
+
+mod common;
+
+use common::TestEnv;
+use kebab_app::IngestOpts;
+use kebab_app::reset::{ResetScope, execute};
+use kebab_core::{DocFilter, DocumentStore, SourceScope};
+
+/// Open the SqliteStore and list all `workspace_path` values.
+fn list_doc_paths(env: &TestEnv) -> Vec<String> {
+    use kebab_store_sqlite::SqliteStore;
+    let store = SqliteStore::open(&env.config).unwrap();
+    store.run_migrations().unwrap();
+    store
+        .list_documents(&DocFilter::default())
+        .unwrap()
+        .into_iter()
+        .map(|d| d.doc_path.0)
+        .collect()
+}
+
+#[test]
+fn reset_orphans_only_purges_out_of_scope_docs() {
+    let env = TestEnv::lexical_only();
+
+    // Write three .rs files into the workspace.
+    let a_path = env.workspace_root.join("a.rs");
+    let b_path = env.workspace_root.join("b.rs");
+    let c_path = env.workspace_root.join("c.rs");
+    std::fs::write(&a_path, "// file a\nfn alpha() {}\n").unwrap();
+    std::fs::write(&b_path, "// file b\nfn bravo() {}\n").unwrap();
+    std::fs::write(&c_path, "// file c\nfn charlie() {}\n").unwrap();
+
+    // Ingest all three with a wide scope.
+    let wide_scope = SourceScope {
+        root: env.workspace_root.clone(),
+        include: vec!["**/*.rs".to_string()],
+        exclude: env.config.workspace.exclude.clone(),
+    };
+    let first = kebab_app::ingest_with_config_opts(
+        env.config.clone(),
+        wide_scope,
+        false,
+        IngestOpts::default(),
+    )
+    .expect("first ingest must succeed");
+    // The fixture workspace may contain other .rs files — just assert we
+    // got at least 3 new docs (our a.rs, b.rs, c.rs).
+    assert!(first.new >= 3, "expected at least 3 new docs: {first:?}");
+    assert_eq!(first.errors, 0, "no errors on first ingest");
+
+    // Narrow config to include only a.rs; b.rs + c.rs are still on disk.
+    let mut narrow_cfg = env.config.clone();
+    narrow_cfg.workspace.exclude.clear();
+    // Re-point workspace root (already correct) and restrict include via
+    // the SourceScope in the connector. The config's `workspace.root` is
+    // used by `enumerate_orphans` to build its scope — we keep that
+    // pointing at the workspace root. We simulate narrowing by setting a
+    // glob that only matches a.rs.
+    //
+    // NOTE: `kebab_config::WorkspaceCfg` does not have an `include` field
+    // (it was removed in p9-fb-25). We narrow the scope via the walker
+    // exclude list: exclude b.rs and c.rs explicitly.
+    narrow_cfg.workspace.exclude = vec!["b.rs".to_string(), "c.rs".to_string()];
+
+    // Run orphans-only reset.
+    let report = execute(ResetScope::OrphansOnly, &narrow_cfg)
+        .expect("orphans-only reset must succeed");
+
+    assert_eq!(
+        report.orphans_purged, 2,
+        "expected 2 orphans purged (b.rs + c.rs): {report:?}"
+    );
+
+    let mut purged: Vec<String> = report
+        .purged_paths
+        .iter()
+        .map(|p| p.0.clone())
+        .collect();
+    purged.sort();
+    assert_eq!(
+        purged,
+        vec!["b.rs".to_string(), "c.rs".to_string()],
+        "purged_paths must list b.rs and c.rs in sorted order: {purged:?}"
+    );
+
+    // list docs must show only a.rs (and any pre-existing fixture files
+    // that are not excluded by the narrow config).
+    let doc_paths = list_doc_paths(&env);
+    // The narrow_cfg excludes b.rs + c.rs — they must no longer be in store.
+    assert!(
+        !doc_paths.iter().any(|p| p == "b.rs"),
+        "b.rs must be gone from store after orphans-only reset; got: {doc_paths:?}"
+    );
+    assert!(
+        !doc_paths.iter().any(|p| p == "c.rs"),
+        "c.rs must be gone from store after orphans-only reset; got: {doc_paths:?}"
+    );
+    assert!(
+        doc_paths.iter().any(|p| p == "a.rs"),
+        "a.rs must still be in store; got: {doc_paths:?}"
+    );
+
+    // Both b.rs and c.rs must still exist on the filesystem — no file
+    // removal is performed by orphans-only.
+    assert!(
+        b_path.exists(),
+        "b.rs must still be on disk after orphans-only reset"
+    );
+    assert!(
+        c_path.exists(),
+        "c.rs must still be on disk after orphans-only reset"
+    );
+
+    // Second reset must be idempotent: nothing left to purge.
+    let second = execute(ResetScope::OrphansOnly, &narrow_cfg)
+        .expect("second orphans-only reset must succeed");
+    assert_eq!(
+        second.orphans_purged, 0,
+        "second reset must be idempotent (orphans_purged == 0): {second:?}"
+    );
+    assert!(
+        second.purged_paths.is_empty(),
+        "second reset purged_paths must be empty: {:?}",
+        second.purged_paths
+    );
+}
--- a/crates/kebab-app/tests/twin_files_fetch_span.rs
+++ b/crates/kebab-app/tests/twin_files_fetch_span.rs
@@ -0,0 +1,176 @@
+//! Regression test for the twin-file fetch_span media-type lookup bug.
+//!
+//! Twin files (identical content at different workspace paths) share one
+//! `assets` row whose PRIMARY KEY is the blake3 content hash. The old
+//! `fetch_span` implementation called
+//! `get_asset_by_workspace_path(&doc.workspace_path)` to check whether the
+//! media type was PDF/audio (and therefore reject span fetch). For a twin
+//! file that lookup could silently return the *other* twin's asset row if
+//! `assets.workspace_path` had been overwritten on the most recent ingest of
+//! the sibling — making the media-type branch decision incorrect.
+//!
+//! Fix: `fetch_span` now uses the 2-step lookup
+//!   `get_document_by_workspace_path` → `doc.source_asset_id` → `get_asset`
+//! so the result is always anchored to the requesting document, not
+//! whichever twin last updated `assets.workspace_path`.
+//!
+//! This test builds a twin-file scenario (two .md files at different paths
+//! with identical content), ingests both, then calls `fetch_span` on each
+//! twin's `doc_id` and asserts it succeeds. Before the fix, if the asset
+//! row's workspace_path happened to point at the wrong twin the span could
+//! return an incorrect `span_not_supported` for a non-PDF/audio file, or
+//! conversely allow span on a PDF twin by accident. After the fix, the
+//! lookup is always doc-specific.
+
+mod common;
+
+use common::TestEnv;
+use kebab_app::ingest_with_config;
+use kebab_core::{DocumentStore, FetchKind, FetchOpts, FetchQuery, IngestItemKind};
+
+#[test]
+fn twin_files_fetch_span_uses_correct_asset() {
+    let env = TestEnv::lexical_only();
+
+    // Write two markdown files with identical content at different paths.
+    let dir_a = env.workspace_root.join("src_a");
+    let dir_b = env.workspace_root.join("src_b");
+    std::fs::create_dir_all(&dir_a).unwrap();
+    std::fs::create_dir_all(&dir_b).unwrap();
+
+    // The content must produce at least 1 line so span fetch is non-trivial.
+    let content = "# Twin\n\nLine one.\n\nLine two.\n\nLine three.\n";
+    std::fs::write(dir_a.join("note.md"), content).unwrap();
+    std::fs::write(dir_b.join("note.md"), content).unwrap();
+
+    // Ingest all files (fixture workspace + our two new twins).
+    let report = ingest_with_config(env.config.clone(), env.scope(), false)
+        .expect("ingest must succeed");
+    assert_eq!(report.errors, 0, "no ingest errors; report={report:?}");
+
+    // Both twin paths must appear as New in the report.
+    let items = report.items.as_ref().expect("items must be present");
+    let twin_items: Vec<_> = items
+        .iter()
+        .filter(|i| {
+            i.doc_path.0.ends_with("src_a/note.md")
+                || i.doc_path.0.ends_with("src_b/note.md")
+        })
+        .collect();
+    assert_eq!(
+        twin_items.len(),
+        2,
+        "exactly 2 twin items expected; items={items:?}"
+    );
+    for item in &twin_items {
+        assert_eq!(
+            item.kind,
+            IngestItemKind::New,
+            "each twin must be New; item={item:?}"
+        );
+    }
+
+    // Resolve doc_ids for both workspace paths.
+    // The ingest layer normalises workspace_path to the path relative to
+    // workspace_root (e.g. "src_a/note.md"), so we look up by that form.
+    let store = kebab_store_sqlite::SqliteStore::open(&env.config).unwrap();
+    store.run_migrations().unwrap();
+
+    // Find the twin items by matching on suffix so the test is robust to
+    // however the workspace root is represented.
+    let items = report.items.as_ref().expect("items must be present");
+    let path_a_str = items
+        .iter()
+        .find(|i| i.doc_path.0.ends_with("src_a/note.md"))
+        .map(|i| i.doc_path.0.clone())
+        .expect("src_a/note.md must appear in ingest report");
+    let path_b_str = items
+        .iter()
+        .find(|i| i.doc_path.0.ends_with("src_b/note.md"))
+        .map(|i| i.doc_path.0.clone())
+        .expect("src_b/note.md must appear in ingest report");
+
+    let path_a = kebab_core::WorkspacePath(path_a_str);
+    let path_b = kebab_core::WorkspacePath(path_b_str);
+
+    let doc_a = store
+        .get_document_by_workspace_path(&path_a)
+        .expect("get_document_by_workspace_path path_a")
+        .expect("doc_a must exist after ingest");
+    let doc_b = store
+        .get_document_by_workspace_path(&path_b)
+        .expect("get_document_by_workspace_path path_b")
+        .expect("doc_b must exist after ingest");
+
+    // Both twins share one asset_id (same content hash).
+    assert_eq!(
+        doc_a.source_asset_id, doc_b.source_asset_id,
+        "twin files must share one asset_id"
+    );
+
+    // Open App and issue span fetch on each twin's doc_id.
+    let app = env.app();
+
+    let result_a = app
+        .fetch(
+            FetchQuery::Span {
+                doc_id: doc_a.doc_id.clone(),
+                line_start: 1,
+                line_end: 2,
+            },
+            FetchOpts::default(),
+        )
+        .expect("fetch_span on twin A must succeed for a markdown file");
+    assert_eq!(result_a.kind, FetchKind::Span);
+    assert!(
+        result_a.text.as_deref().is_some_and(|t| !t.is_empty()),
+        "span text for twin A must not be empty"
+    );
+
+    let result_b = app
+        .fetch(
+            FetchQuery::Span {
+                doc_id: doc_b.doc_id.clone(),
+                line_start: 1,
+                line_end: 2,
+            },
+            FetchOpts::default(),
+        )
+        .expect("fetch_span on twin B must succeed for a markdown file");
+    assert_eq!(result_b.kind, FetchKind::Span);
+    assert!(
+        result_b.text.as_deref().is_some_and(|t| !t.is_empty()),
+        "span text for twin B must not be empty"
+    );
+
+    // Ingest again to force the asset.workspace_path flip-flop, then
+    // re-check. Pre-fix this was the scenario that triggered the bug:
+    // after the second ingest the asset row's workspace_path could point
+    // at either twin, making one twin's span fetch behave incorrectly.
+    let report2 = ingest_with_config(env.config.clone(), env.scope(), false)
+        .expect("second ingest must succeed");
+    assert_eq!(report2.errors, 0, "no ingest errors on second run; report={report2:?}");
+
+    // Re-open app after second ingest and verify span still works on both.
+    let app2 = env.app();
+
+    app2.fetch(
+        FetchQuery::Span {
+            doc_id: doc_a.doc_id.clone(),
+            line_start: 1,
+            line_end: 3,
+        },
+        FetchOpts::default(),
+    )
+    .expect("fetch_span on twin A after flip-flop must still succeed");
+
+    app2.fetch(
+        FetchQuery::Span {
+            doc_id: doc_b.doc_id.clone(),
+            line_start: 1,
+            line_end: 3,
+        },
+        FetchOpts::default(),
+    )
+    .expect("fetch_span on twin B after flip-flop must still succeed");
+}
--- a/crates/kebab-app/tests/twin_files_idempotent.rs
+++ b/crates/kebab-app/tests/twin_files_idempotent.rs
@@ -0,0 +1,90 @@
+//! Regression test for the twin-file idempotency bug.
+//!
+//! Identical-content files at different workspace paths share one
+//! `assets` row (`asset_id` = blake3 content hash, PRIMARY KEY). The
+//! old UPSERT `ON CONFLICT(asset_id) DO UPDATE SET workspace_path =
+//! excluded.workspace_path` made each twin overwrite the other's path
+//! on every ingest, so `get_asset_by_workspace_path(path1)` returned
+//! None (or the wrong twin) → re-process every time.
+//!
+//! Fix: `try_skip_unchanged` now uses `get_document_by_workspace_path`
+//! instead.  `documents.workspace_path` is UNIQUE (V001) so each twin
+//! has its own stable document row.
+//!
+//! Assertion contract:
+//!   1st ingest → 2 New (one per twin)
+//!   2nd ingest → 0 New, 0 Updated, 2 Unchanged
+
+mod common;
+
+use common::TestEnv;
+use kebab_app::ingest_with_config;
+use kebab_core::IngestItemKind;
+
+#[test]
+fn twin_files_second_ingest_is_unchanged() {
+    let env = TestEnv::lexical_only();
+
+    // Write two files with identical content at different paths.
+    let pkg_a = env.workspace_root.join("pkg_a");
+    let pkg_b = env.workspace_root.join("pkg_b");
+    std::fs::create_dir_all(&pkg_a).unwrap();
+    std::fs::create_dir_all(&pkg_b).unwrap();
+
+    let content = b"# shared\nThis content is identical in both files.\n";
+    std::fs::write(pkg_a.join("__init__.py"), content).unwrap();
+    std::fs::write(pkg_b.join("__init__.py"), content).unwrap();
+
+    // First ingest — both files must be New.
+    let first = ingest_with_config(env.config.clone(), env.scope(), false)
+        .expect("first ingest must succeed");
+    assert_eq!(first.errors, 0, "first ingest: no errors; report={first:?}");
+
+    let items = first.items.as_ref().expect("items must be present");
+    let twin_items: Vec<_> = items
+        .iter()
+        .filter(|i| {
+            i.doc_path.0.ends_with("__init__.py")
+        })
+        .collect();
+    assert_eq!(
+        twin_items.len(),
+        2,
+        "first ingest: expected exactly 2 __init__.py items; items={items:?}"
+    );
+    for item in &twin_items {
+        assert_eq!(
+            item.kind,
+            IngestItemKind::New,
+            "first ingest: each twin must be New; item={item:?}"
+        );
+    }
+
+    // Second ingest — same files, same content → both must be Unchanged.
+    let second = ingest_with_config(env.config.clone(), env.scope(), false)
+        .expect("second ingest must succeed");
+    assert_eq!(second.errors, 0, "second ingest: no errors; report={second:?}");
+    assert_eq!(second.new, 0, "second ingest: no new docs; report={second:?}");
+    assert_eq!(
+        second.updated, 0,
+        "second ingest: no updated docs (twin-file bug would set this to 2); report={second:?}"
+    );
+
+    let second_items = second.items.as_ref().expect("items must be present");
+    let twin_items2: Vec<_> = second_items
+        .iter()
+        .filter(|i| i.doc_path.0.ends_with("__init__.py"))
+        .collect();
+    assert_eq!(
+        twin_items2.len(),
+        2,
+        "second ingest: expected exactly 2 __init__.py items; items={second_items:?}"
+    );
+    for item in &twin_items2 {
+        assert_eq!(
+            item.kind,
+            IngestItemKind::Unchanged,
+            "second ingest: each twin must be Unchanged; item={item:?}"
+        );
+    }
+}
--- a/crates/kebab-chunk/src/code_go_ast_v1.rs
+++ b/crates/kebab-chunk/src/code_go_ast_v1.rs
@@ -0,0 +1,322 @@
+//! `code-go-ast-v1` — maps a tree-sitter-derived Go AST
+//! `CanonicalDocument` (one `Block::Code` per semantic unit, each with
+//! `SourceSpan::Code`) to chunks 1:1. A unit longer than
+//! `AST_CHUNK_MAX_LINES` is split into `<symbol> [part i/N]` sub-chunks
+//! at blank-line paragraph boundaries (design §9.1 oversize fallback).
+//!
+//! tree-sitter is intentionally NOT a dependency here: AST work is
+//! parser-side (`kebab-parse-code`, design §6.3). This chunker only
+//! consumes the `CanonicalDocument`.
+//!
+//! `AST_CHUNK_MAX_LINES` is a constant matching
+//! `IngestCodeCfg::default().ast_chunk_max_lines` (200). Per-medium
+//! config threading needs a chunker registry (P+); same deviation
+//! pattern as `pdf-page-v1`'s pinned `chunker_version`
+//! (`tasks/HOTFIXES.md`).
+
+use kebab_core::{
+    Block, BlockId, CanonicalDocument, Chunk, ChunkPolicy, Chunker, ChunkerVersion, DocumentId,
+    SourceSpan, id_for_chunk,
+};
+
+const VERSION_LABEL: &str = "code-go-ast-v1";
+const BYTES_PER_TOKEN: usize = 3;
+const POLICY_HASH_HEX_LEN: usize = 16;
+const AST_CHUNK_MAX_LINES: u32 = 200;
+
+#[derive(Clone, Copy, Debug, Default)]
+pub struct CodeGoAstV1Chunker;
+
+impl Chunker for CodeGoAstV1Chunker {
+    fn chunker_version(&self) -> ChunkerVersion {
+        ChunkerVersion(VERSION_LABEL.to_string())
+    }
+
+    fn policy_hash(&self, policy: &ChunkPolicy) -> String {
+        let bytes = serde_json_canonicalizer::to_vec(policy)
+            .expect("canonical JSON serialization of ChunkPolicy must not fail");
+        let hex = blake3::hash(&bytes).to_hex().to_string();
+        hex[..POLICY_HASH_HEX_LEN].to_string()
+    }
+
+    fn chunk(
+        &self,
+        doc: &CanonicalDocument,
+        policy: &ChunkPolicy,
+    ) -> anyhow::Result<Vec<Chunk>> {
+        for b in &doc.blocks {
+            let c = match b {
+                Block::Code(c) => c,
+                _ => anyhow::bail!(
+                    "CodeGoAstV1Chunker only handles code docs (got non-Code block)"
+                ),
+            };
+            if !matches!(c.common.source_span, SourceSpan::Code { .. }) {
+                anyhow::bail!(
+                    "CodeGoAstV1Chunker only handles code docs (got non-Code source_span)"
+                );
+            }
+        }
+
+        let base_policy_hash = self.policy_hash(policy);
+        let chunker_version = self.chunker_version();
+        let mut out: Vec<Chunk> = Vec::new();
+
+        for b in &doc.blocks {
+            let cb = match b {
+                Block::Code(c) => c,
+                _ => unreachable!("validated above"),
+            };
+            let (ls, le, symbol, lang) = match &cb.common.source_span {
+                SourceSpan::Code { line_start, line_end, symbol, lang } => {
+                    (*line_start, *line_end, symbol.clone(), lang.clone())
+                }
+                _ => unreachable!("validated above"),
+            };
+            let block_ids: Vec<BlockId> = vec![cb.common.block_id.clone()];
+            let span_lines = le.saturating_sub(ls) + 1;
+
+            if span_lines <= AST_CHUNK_MAX_LINES {
+                let span = SourceSpan::Code {
+                    line_start: ls,
+                    line_end: le,
+                    symbol: symbol.clone(),
+                    lang: lang.clone(),
+                };
+                out.push(make_chunk(
+                    doc, &chunker_version, &block_ids, &base_policy_hash,
+                    None, span, cb.code.clone(),
+                ));
+            } else {
+                let parts = split_oversize(&cb.code);
+                let n = parts.len();
+                for (i, (off_start, off_end, text)) in parts.into_iter().enumerate() {
+                    let part_ls = ls + off_start;
+                    let part_le = ls + off_end;
+                    let part_sym = symbol
+                        .as_ref()
+                        .map(|s| format!("{s} [part {}/{n}]", i + 1));
+                    let span = SourceSpan::Code {
+                        line_start: part_ls,
+                        line_end: part_le,
+                        symbol: part_sym,
+                        lang: lang.clone(),
+                    };
+                    out.push(make_chunk(
+                        doc, &chunker_version, &block_ids, &base_policy_hash,
+                        Some(part_ls), span, text,
+                    ));
+                }
+            }
+        }
+
+        tracing::debug!(
+            target: "kebab-chunk",
+            doc_id = %doc.doc_id,
+            chunks = out.len(),
+            "code-go-ast-v1 chunked",
+        );
+        Ok(out)
+    }
+}
+
+#[allow(clippy::too_many_arguments)]
+fn make_chunk(
+    doc: &CanonicalDocument,
+    chunker_version: &ChunkerVersion,
+    block_ids: &[BlockId],
+    base_policy_hash: &str,
+    split_key: Option<u32>,
+    span: SourceSpan,
+    text: String,
+) -> Chunk {
+    let id_hash = match split_key {
+        Some(k) => format!("{base_policy_hash}#L{k}"),
+        None => base_policy_hash.to_string(),
+    };
+    let chunk_id = id_for_chunk(&doc.doc_id, chunker_version, block_ids, &id_hash);
+    let token_estimate = text.len().div_ceil(BYTES_PER_TOKEN);
+    Chunk {
+        chunk_id,
+        doc_id: DocumentId(doc.doc_id.0.clone()),
+        block_ids: block_ids.to_vec(),
+        text,
+        heading_path: Vec::new(),
+        source_spans: vec![span],
+        token_estimate,
+        chunker_version: chunker_version.clone(),
+        policy_hash: base_policy_hash.to_string(),
+    }
+}
+
+/// Split an oversize unit at blank-line paragraph boundaries, greedily
+/// gluing paragraphs until ~`AST_CHUNK_MAX_LINES` lines accumulate.
+/// Returns `(line_offset_start, line_offset_end, text)` where offsets are
+/// 0-based within the unit (caller adds the unit's absolute `line_start`).
+fn split_oversize(code: &str) -> Vec<(u32, u32, String)> {
+    let lines: Vec<&str> = code.split('\n').collect();
+    let total = lines.len() as u32;
+    let mut out: Vec<(u32, u32, String)> = Vec::new();
+    let mut start: u32 = 0;
+    while start < total {
+        let mut end = (start + AST_CHUNK_MAX_LINES).min(total);
+        let floor = start + (AST_CHUNK_MAX_LINES * 4 / 5);
+        if end < total {
+            if let Some(b) = (floor.min(end)..end)
+                .rev()
+                .find(|&i| lines[i as usize].trim().is_empty())
+            {
+                end = b + 1;
+            }
+        }
+        let text = lines[start as usize..end as usize].join("\n");
+        out.push((start, end.saturating_sub(1), text));
+        start = end;
+    }
+    if out.is_empty() {
+        out.push((0, total.saturating_sub(1), code.to_string()));
+    }
+    out
+}
+
+#[cfg(test)]
+mod tests {
+    use super::*;
+    use kebab_core::{
+        Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+        SourceSpan, id_for_block, id_for_doc, AssetId, Lang, Metadata, ParserVersion, Provenance,
+        SourceType, TrustLevel, WorkspacePath,
+    };
+    use time::OffsetDateTime;
+
+    fn code_doc(units: &[(&str, u32, u32, &str)]) -> CanonicalDocument {
+        let wp = WorkspacePath("crates/x/src/a.go".into());
+        let aid = AssetId("a".repeat(64));
+        let pv = ParserVersion("code-go-v1".into());
+        let doc_id = id_for_doc(&wp, &aid, &pv);
+        let blocks = units
+            .iter()
+            .enumerate()
+            .map(|(i, (sym, ls, le, code))| {
+                let span = SourceSpan::Code {
+                    line_start: *ls,
+                    line_end: *le,
+                    symbol: Some((*sym).to_string()),
+                    lang: Some("go".into()),
+                };
+                let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+                Block::Code(CodeBlock {
+                    common: CommonBlock { block_id: bid, heading_path: vec![], source_span: span },
+                    lang: Some("go".into()),
+                    code: (*code).to_string(),
+                })
+            })
+            .collect();
+        CanonicalDocument {
+            doc_id, source_asset_id: aid, workspace_path: wp, title: "a".into(),
+            lang: Lang("und".into()), blocks,
+            metadata: Metadata {
+                aliases: vec![], tags: vec![],
+                created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+                source_type: SourceType::Note, trust_level: TrustLevel::Primary,
+                user_id_alias: None, user: Default::default(),
+                repo: Some("kebab".into()), git_branch: Some("main".into()),
+                git_commit: Some("0".repeat(40)), code_lang: Some("go".into()),
+            },
+            provenance: Provenance { events: vec![] },
+            parser_version: pv, schema_version: 1, doc_version: 1,
+            last_chunker_version: None, last_embedding_version: None,
+        }
+    }
+    fn policy() -> ChunkPolicy {
+        ChunkPolicy { target_tokens: 500, overlap_tokens: 80,
+            respect_markdown_headings: false,
+            chunker_version: ChunkerVersion(VERSION_LABEL.into()) }
+    }
+
+    #[test]
+    fn chunker_version_is_code_go_ast_v1() {
+        assert_eq!(CodeGoAstV1Chunker.chunker_version(),
+            ChunkerVersion("code-go-ast-v1".into()));
+    }
+
+    #[test]
+    fn one_chunk_per_unit_preserves_code_span() {
+        let doc = code_doc(&[
+            ("parse", 1, 3, "func parse() {\n\t// x\n}"),
+            ("Foo.double", 5, 7, "func double() int {\n\t//\n\treturn 0\n}"),
+        ]);
+        let chunks = CodeGoAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert_eq!(chunks.len(), 2);
+        for c in &chunks {
+            assert_eq!(c.source_spans.len(), 1);
+            assert!(matches!(c.source_spans[0], SourceSpan::Code { .. }));
+            assert_eq!(c.heading_path, Vec::<String>::new());
+            assert_eq!(c.chunker_version.0, "code-go-ast-v1");
+        }
+        match &chunks[0].source_spans[0] {
+            SourceSpan::Code { symbol, line_start, line_end, .. } => {
+                assert_eq!(symbol.as_deref(), Some("parse"));
+                assert_eq!((*line_start, *line_end), (1, 3));
+            }
+            _ => unreachable!(),
+        }
+    }
+
+    #[test]
+    fn oversize_unit_splits_into_parts_with_unique_ids() {
+        let body = (0..500).map(|i| format!("\tx{i} := {i}")).collect::<Vec<_>>().join("\n");
+        let code = format!("func big() {{\n{body}\n}}");
+        let doc = code_doc(&[("big", 1, 502, &code)]);
+        let chunks = CodeGoAstV1Chunker.chunk(&doc, &policy()).unwrap();
+        assert!(chunks.len() >= 2, "oversize unit must split, got {}", chunks.len());
+        for c in &chunks {
+            match &c.source_spans[0] {
+                SourceSpan::Code { symbol, .. } => {
+                    assert!(symbol.as_deref().unwrap().starts_with("big [part "),
+                        "part-numbered symbol, got {symbol:?}");
+                }
+                _ => unreachable!(),
+            }
+        }
+        let mut ids: Vec<&str> = chunks.iter().map(|c| c.chunk_id.0.as_str()).collect();
+        let n = ids.len(); ids.sort(); ids.dedup();
+        assert_eq!(ids.len(), n, "chunk_ids unique across split parts");
+    }
+
+    #[test]
+    fn non_code_doc_errors() {
+        use kebab_core::TextBlock;
+        let mut doc = code_doc(&[("parse", 1, 1, "func parse() {}")]);
+        doc.blocks = vec![Block::Paragraph(TextBlock {
+            common: CommonBlock {
+                block_id: kebab_core::BlockId("b".into()),
+                heading_path: vec![],
+                source_span: SourceSpan::Line { start: 1, end: 1 },
+            },
+            text: "x".into(), inlines: vec![],
+        })];
+        let err = CodeGoAstV1Chunker.chunk(&doc, &policy()).unwrap_err();
+        assert!(err.to_string().contains("CodeGoAstV1Chunker"));
+    }
+
+    #[test]
+    fn deterministic_chunk_ids_1000() {
+        let doc = code_doc(&[("parse", 1, 2, "func parse() {}\n")]);
+        let base: Vec<String> = CodeGoAstV1Chunker.chunk(&doc, &policy())
+            .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+        for _ in 0..1000 {
+            let again: Vec<String> = CodeGoAstV1Chunker.chunk(&doc, &policy())
+                .unwrap().into_iter().map(|c| c.chunk_id.0).collect();
+            assert_eq!(again, base);
+        }
+    }
+
+    #[test]
+    fn policy_hash_matches_md_heading_v1() {
+        let p = policy();
+        assert_eq!(CodeGoAstV1Chunker.policy_hash(&p),
+            crate::MdHeadingV1Chunker.policy_hash(&p));
+    }
+}
--- a/crates/kebab-chunk/src/lib.rs
+++ b/crates/kebab-chunk/src/lib.rs
@@ -15,6 +15,7 @@
 //! embedder, the retriever, the LLM, the RAG layer, or the UI layers.
 //! It consumes `CanonicalDocument` purely through `kb-core` types.

+mod code_go_ast_v1;
 mod code_js_ast_v1;
 mod code_python_ast_v1;
 mod code_rust_ast_v1;
@@ -22,6 +23,7 @@ mod code_ts_ast_v1;
 mod md_heading_v1;
 mod pdf_page_v1;

+pub use code_go_ast_v1::CodeGoAstV1Chunker;
 pub use code_js_ast_v1::CodeJsAstV1Chunker;
 pub use code_python_ast_v1::CodePythonAstV1Chunker;
 pub use code_rust_ast_v1::CodeRustAstV1Chunker;
--- a/crates/kebab-chunk/tests/code_go_ast_snapshot.rs
+++ b/crates/kebab-chunk/tests/code_go_ast_snapshot.rs
@@ -0,0 +1,221 @@
+//! Snapshot test pinning the `Vec<Chunk>` JSON for a
+//! representative Go code `CanonicalDocument`.
+//!
+//! This is an integration test. `kebab-parse-code` is intentionally NOT
+//! a dev-dep (design §6.3 / §8 boundary: AST extraction is parser-side).
+//! The `CanonicalDocument` is built inline from hand-crafted `Block::Code`
+//! units, which is the same pattern used in `code_rust_ast_v1.rs`'s
+//! internal `code_doc` test helper.
+//!
+//! Set `UPDATE_SNAPSHOTS=1` to re-bake the baseline.
+
+use std::path::PathBuf;
+
+use kebab_chunk::CodeGoAstV1Chunker;
+use kebab_core::{
+    AssetId, Block, CanonicalDocument, ChunkPolicy, Chunker, ChunkerVersion, CodeBlock, CommonBlock,
+    Lang, Metadata, ParserVersion, Provenance, SourceSpan, SourceType, TrustLevel, WorkspacePath,
+    id_for_block, id_for_doc,
+};
+use serde_json::Value;
+use time::OffsetDateTime;
+
+fn fixtures_dir() -> PathBuf {
+    PathBuf::from(env!("CARGO_MANIFEST_DIR"))
+        .join("tests")
+        .join("fixtures")
+}
+
+fn fixed_doc() -> CanonicalDocument {
+    let wp = WorkspacePath("kebab_eval/metrics.go".into());
+    let aid = AssetId("b".repeat(64));
+    // Pin parser_version so doc_id / block_ids are reproducible.
+    let pv = ParserVersion("code-go-v1".into());
+    let doc_id = id_for_doc(&wp, &aid, &pv);
+
+    // Build a >200-line function body to force split_oversize.
+    let big_body: String = {
+        let header = "func BigCompute(data []int) int {\n";
+        let body: String = (0..210u32)
+            .map(|i| format!("\tv{i} := 0\n\tif {i} < len(data) {{\n\t\tv{i} = data[{i}]\n\t}}\n"))
+            .collect();
+        let footer = "\treturn len(data)\n}";
+        format!("{header}{body}{footer}")
+    };
+    let big_line_count = big_body.lines().count() as u32;
+    let big_line_end = 48 + big_line_count - 1;
+
+    // Representative units:
+    //  0. import block                    (lines 1–5,   ≤200)
+    //  1. free fn `ComputeMRR`            (lines 7–12,  ≤200)
+    //  2. struct `MetricsCollector`       (lines 14–20, ≤200)
+    //  3. struct `BaseEvaluator`          (lines 22–30, ≤200)
+    //  4. method `Run`                    (lines 32–38, ≤200)
+    //  5. method `Report`                 (lines 40–46, ≤200)
+    //  6. BigCompute (>200 lines)         to force split_oversize
+    let raw_units: Vec<(&str, u32, u32, String)> = vec![
+        (
+            "imports",
+            1,
+            5,
+            "import (\n\t\"fmt\"\n\t\"os\"\n\t\"strings\"\n)".to_string(),
+        ),
+        (
+            "ComputeMRR",
+            7,
+            12,
+            "func ComputeMRR(scores []float64) float64 {\n\tif len(scores) == 0 {\n\t\treturn 0.0\n\t}\n\t_ = fmt.Sprintf(\"%v\", scores)\n\treturn 1.0 / float64(len(scores))\n}".to_string(),
+        ),
+        (
+            "MetricsCollector",
+            14,
+            20,
+            "type MetricsCollector struct {\n\tScores []float64\n\tLabels []string\n\tCounts map[string]int\n\tTotals map[string]float64\n\tTags   []string\n}".to_string(),
+        ),
+        (
+            "BaseEvaluator",
+            22,
+            30,
+            "type BaseEvaluator struct {\n\tName string\n}\n\nfunc (e *BaseEvaluator) Evaluate(data []string) error {\n\t_ = os.Stderr\n\t_ = strings.Join(data, \",\")\n\treturn nil\n}".to_string(),
+        ),
+        (
+            "MetricsCollector.Run",
+            32,
+            38,
+            "func (m *MetricsCollector) Run(inputs []float64) {\n\tfor _, inp := range inputs {\n\t\tm.Scores = append(\n\t\t\tm.Scores,\n\t\t\tinp,\n\t\t)\n\t}\n}".to_string(),
+        ),
+        (
+            "MetricsCollector.Report",
+            40,
+            46,
+            "func (m *MetricsCollector) Report() map[string]interface{} {\n\treturn map[string]interface{}{\n\t\t\"mean\":  0.0,\n\t\t\"count\": len(m.Scores),\n\t\t\"tags\":  m.Tags,\n\t}\n}".to_string(),
+        ),
+        ("BigCompute", 48, big_line_end, big_body),
+    ];
+
+    let blocks: Vec<Block> = raw_units
+        .iter()
+        .enumerate()
+        .map(|(i, (sym, ls, le, code))| {
+            let span = SourceSpan::Code {
+                line_start: *ls,
+                line_end: *le,
+                symbol: Some((*sym).to_string()),
+                lang: Some("go".into()),
+            };
+            let bid = id_for_block(&doc_id, "code", &[], i as u32, &span);
+            Block::Code(CodeBlock {
+                common: CommonBlock {
+                    block_id: bid,
+                    heading_path: vec![],
+                    source_span: span,
+                },
+                lang: Some("go".into()),
+                code: code.clone(),
+            })
+        })
+        .collect();
+
+    CanonicalDocument {
+        doc_id,
+        source_asset_id: aid,
+        workspace_path: wp,
+        title: "metrics.go".into(),
+        lang: Lang("und".into()),
+        blocks,
+        metadata: Metadata {
+            aliases: vec![],
+            tags: vec![],
+            created_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            updated_at: OffsetDateTime::from_unix_timestamp(1_700_000_000).unwrap(),
+            source_type: SourceType::Note,
+            trust_level: TrustLevel::Primary,
+            user_id_alias: None,
+            user: Default::default(),
+            repo: Some("kebab".into()),
+            git_branch: Some("main".into()),
+            git_commit: Some("0".repeat(40)),
+            code_lang: Some("go".into()),
+        },
+        provenance: Provenance { events: vec![] },
+        parser_version: pv,
+        schema_version: 1,
+        doc_version: 1,
+        last_chunker_version: None,
+        last_embedding_version: None,
+    }
+}
+
+fn fixed_policy() -> ChunkPolicy {
+    ChunkPolicy {
+        target_tokens: 500,
+        overlap_tokens: 80,
+        respect_markdown_headings: false,
+        chunker_version: ChunkerVersion("code-go-ast-v1".into()),
+    }
+}
+
+#[test]
+fn code_go_ast_chunks_snapshot() {
+    let doc = fixed_doc();
+    let policy = fixed_policy();
+
+    let chunks = CodeGoAstV1Chunker.chunk(&doc, &policy).expect("chunk");
+    let actual = serde_json::to_value(&chunks).unwrap();
+
+    let dir = fixtures_dir();
+    let baseline_path = dir.join("code-sample.go.chunks.snapshot.json");
+    let baseline_text = match std::fs::read_to_string(&baseline_path) {
+        Ok(s) => s,
+        Err(_) if std::env::var("UPDATE_SNAPSHOTS").is_ok() => {
+            std::fs::create_dir_all(&dir).unwrap();
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            return;
+        }
+        Err(e) => panic!(
+            "missing baseline {}; run with UPDATE_SNAPSHOTS=1 to create: {e}",
+            baseline_path.display()
+        ),
+    };
+    let expected: Value = serde_json::from_str(&baseline_text).expect("baseline parses as json");
+
+    if actual != expected {
+        if std::env::var("UPDATE_SNAPSHOTS").is_ok() {
+            let pretty = serde_json::to_string_pretty(&actual).unwrap();
+            std::fs::write(&baseline_path, format!("{pretty}\n")).unwrap();
+            eprintln!("updated baseline {}", baseline_path.display());
+            return;
+        }
+        let pretty = serde_json::to_string_pretty(&actual).unwrap();
+        panic!(
+            "code-go-ast-v1 chunks snapshot drift\n\
+             --- expected ({}) ---\n{baseline_text}\n\
+             --- actual ---\n{pretty}\n\
+             If intentional, re-run with UPDATE_SNAPSHOTS=1.",
+            baseline_path.display()
+        );
+    }
+}
+
+/// Determinism cross-check: re-running the same pipeline yields the same
+/// chunk_ids byte-for-byte.
+#[test]
+fn code_go_ast_chunks_are_deterministic() {
+    let policy = fixed_policy();
+    let baseline: Vec<String> = CodeGoAstV1Chunker
+        .chunk(&fixed_doc(), &policy)
+        .unwrap()
+        .into_iter()
+        .map(|c| c.chunk_id.0)
+        .collect();
+    for _ in 0..5 {
+        let again: Vec<String> = CodeGoAstV1Chunker
+            .chunk(&fixed_doc(), &policy)
+            .unwrap()
+            .into_iter()
+            .map(|c| c.chunk_id.0)
+            .collect();
+        assert_eq!(again, baseline);
+    }
+}
--- a/crates/kebab-chunk/tests/fixtures/code-sample.go.chunks.snapshot.json
+++ b/crates/kebab-chunk/tests/fixtures/code-sample.go.chunks.snapshot.json
@@ -0,0 +1,233 @@
+[
+  {
+    "block_ids": [
+      "c182bf37e32c7fc1b868bd617f8eaf66"
+    ],
+    "chunk_id": "43de518d946dc18ec040ae20d74e0cff",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 5,
+        "line_start": 1,
+        "symbol": "imports"
+      }
+    ],
+    "text": "import (\n\t\"fmt\"\n\t\"os\"\n\t\"strings\"\n)",
+    "token_estimate": 12
+  },
+  {
+    "block_ids": [
+      "c9992cdcfdf3c2a7700a4abc4782a8a4"
+    ],
+    "chunk_id": "af4c382a83f1e8cdea495d8b33c11abc",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 12,
+        "line_start": 7,
+        "symbol": "ComputeMRR"
+      }
+    ],
+    "text": "func ComputeMRR(scores []float64) float64 {\n\tif len(scores) == 0 {\n\t\treturn 0.0\n\t}\n\t_ = fmt.Sprintf(\"%v\", scores)\n\treturn 1.0 / float64(len(scores))\n}",
+    "token_estimate": 50
+  },
+  {
+    "block_ids": [
+      "5f18dc3e79fe946ba05d32c3bfc00684"
+    ],
+    "chunk_id": "4be6d8f180bc19b8651877e5264852ac",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 20,
+        "line_start": 14,
+        "symbol": "MetricsCollector"
+      }
+    ],
+    "text": "type MetricsCollector struct {\n\tScores []float64\n\tLabels []string\n\tCounts map[string]int\n\tTotals map[string]float64\n\tTags   []string\n}",
+    "token_estimate": 45
+  },
+  {
+    "block_ids": [
+      "3009cc022ca832c323393e4f9bcdb388"
+    ],
+    "chunk_id": "3ae182f4c6d304ee7f0aaf447142f948",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 30,
+        "line_start": 22,
+        "symbol": "BaseEvaluator"
+      }
+    ],
+    "text": "type BaseEvaluator struct {\n\tName string\n}\n\nfunc (e *BaseEvaluator) Evaluate(data []string) error {\n\t_ = os.Stderr\n\t_ = strings.Join(data, \",\")\n\treturn nil\n}",
+    "token_estimate": 53
+  },
+  {
+    "block_ids": [
+      "e0e83d1d7f9327a1902ae9a8f67c1f1c"
+    ],
+    "chunk_id": "b962f14980e756bb8ba514e2282756cd",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 38,
+        "line_start": 32,
+        "symbol": "MetricsCollector.Run"
+      }
+    ],
+    "text": "func (m *MetricsCollector) Run(inputs []float64) {\n\tfor _, inp := range inputs {\n\t\tm.Scores = append(\n\t\t\tm.Scores,\n\t\t\tinp,\n\t\t)\n\t}\n}",
+    "token_estimate": 44
+  },
+  {
+    "block_ids": [
+      "0e6a572bc3fe2bd6d173fe614bd1b763"
+    ],
+    "chunk_id": "441c695e990e7f49188068433e313e87",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 46,
+        "line_start": 40,
+        "symbol": "MetricsCollector.Report"
+      }
+    ],
+    "text": "func (m *MetricsCollector) Report() map[string]interface{} {\n\treturn map[string]interface{}{\n\t\t\"mean\":  0.0,\n\t\t\"count\": len(m.Scores),\n\t\t\"tags\":  m.Tags,\n\t}\n}",
+    "token_estimate": 53
+  },
+  {
+    "block_ids": [
+      "5d269745b2e5dbdcbef0c09ba54b0bd6"
+    ],
+    "chunk_id": "7a942d871c588ec69426290561f05179",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 247,
+        "line_start": 48,
+        "symbol": "BigCompute [part 1/5]"
+      }
+    ],
+    "text": "func BigCompute(data []int) int {\n\tv0 := 0\n\tif 0 < len(data) {\n\t\tv0 = data[0]\n\t}\n\tv1 := 0\n\tif 1 < len(data) {\n\t\tv1 = data[1]\n\t}\n\tv2 := 0\n\tif 2 < len(data) {\n\t\tv2 = data[2]\n\t}\n\tv3 := 0\n\tif 3 < len(data) {\n\t\tv3 = data[3]\n\t}\n\tv4 := 0\n\tif 4 < len(data) {\n\t\tv4 = data[4]\n\t}\n\tv5 := 0\n\tif 5 < len(data) {\n\t\tv5 = data[5]\n\t}\n\tv6 := 0\n\tif 6 < len(data) {\n\t\tv6 = data[6]\n\t}\n\tv7 := 0\n\tif 7 < len(data) {\n\t\tv7 = data[7]\n\t}\n\tv8 := 0\n\tif 8 < len(data) {\n\t\tv8 = data[8]\n\t}\n\tv9 := 0\n\tif 9 < len(data) {\n\t\tv9 = data[9]\n\t}\n\tv10 := 0\n\tif 10 < len(data) {\n\t\tv10 = data[10]\n\t}\n\tv11 := 0\n\tif 11 < len(data) {\n\t\tv11 = data[11]\n\t}\n\tv12 := 0\n\tif 12 < len(data) {\n\t\tv12 = data[12]\n\t}\n\tv13 := 0\n\tif 13 < len(data) {\n\t\tv13 = data[13]\n\t}\n\tv14 := 0\n\tif 14 < len(data) {\n\t\tv14 = data[14]\n\t}\n\tv15 := 0\n\tif 15 < len(data) {\n\t\tv15 = data[15]\n\t}\n\tv16 := 0\n\tif 16 < len(data) {\n\t\tv16 = data[16]\n\t}\n\tv17 := 0\n\tif 17 < len(data) {\n\t\tv17 = data[17]\n\t}\n\tv18 := 0\n\tif 18 < len(data) {\n\t\tv18 = data[18]\n\t}\n\tv19 := 0\n\tif 19 < len(data) {\n\t\tv19 = data[19]\n\t}\n\tv20 := 0\n\tif 20 < len(data) {\n\t\tv20 = data[20]\n\t}\n\tv21 := 0\n\tif 21 < len(data) {\n\t\tv21 = data[21]\n\t}\n\tv22 := 0\n\tif 22 < len(data) {\n\t\tv22 = data[22]\n\t}\n\tv23 := 0\n\tif 23 < len(data) {\n\t\tv23 = data[23]\n\t}\n\tv24 := 0\n\tif 24 < len(data) {\n\t\tv24 = data[24]\n\t}\n\tv25 := 0\n\tif 25 < len(data) {\n\t\tv25 = data[25]\n\t}\n\tv26 := 0\n\tif 26 < len(data) {\n\t\tv26 = data[26]\n\t}\n\tv27 := 0\n\tif 27 < len(data) {\n\t\tv27 = data[27]\n\t}\n\tv28 := 0\n\tif 28 < len(data) {\n\t\tv28 = data[28]\n\t}\n\tv29 := 0\n\tif 29 < len(data) {\n\t\tv29 = data[29]\n\t}\n\tv30 := 0\n\tif 30 < len(data) {\n\t\tv30 = data[30]\n\t}\n\tv31 := 0\n\tif 31 < len(data) {\n\t\tv31 = data[31]\n\t}\n\tv32 := 0\n\tif 32 < len(data) {\n\t\tv32 = data[32]\n\t}\n\tv33 := 0\n\tif 33 < len(data) {\n\t\tv33 = data[33]\n\t}\n\tv34 := 0\n\tif 34 < len(data) {\n\t\tv34 = data[34]\n\t}\n\tv35 := 0\n\tif 35 < len(data) {\n\t\tv35 = data[35]\n\t}\n\tv36 := 0\n\tif 36 < len(data) {\n\t\tv36 = data[36]\n\t}\n\tv37 := 0\n\tif 37 < len(data) {\n\t\tv37 = data[37]\n\t}\n\tv38 := 0\n\tif 38 < len(data) {\n\t\tv38 = data[38]\n\t}\n\tv39 := 0\n\tif 39 < len(data) {\n\t\tv39 = data[39]\n\t}\n\tv40 := 0\n\tif 40 < len(data) {\n\t\tv40 = data[40]\n\t}\n\tv41 := 0\n\tif 41 < len(data) {\n\t\tv41 = data[41]\n\t}\n\tv42 := 0\n\tif 42 < len(data) {\n\t\tv42 = data[42]\n\t}\n\tv43 := 0\n\tif 43 < len(data) {\n\t\tv43 = data[43]\n\t}\n\tv44 := 0\n\tif 44 < len(data) {\n\t\tv44 = data[44]\n\t}\n\tv45 := 0\n\tif 45 < len(data) {\n\t\tv45 = data[45]\n\t}\n\tv46 := 0\n\tif 46 < len(data) {\n\t\tv46 = data[46]\n\t}\n\tv47 := 0\n\tif 47 < len(data) {\n\t\tv47 = data[47]\n\t}\n\tv48 := 0\n\tif 48 < len(data) {\n\t\tv48 = data[48]\n\t}\n\tv49 := 0\n\tif 49 < len(data) {\n\t\tv49 = data[49]",
+    "token_estimate": 847
+  },
+  {
+    "block_ids": [
+      "5d269745b2e5dbdcbef0c09ba54b0bd6"
+    ],
+    "chunk_id": "3f44ba43c9415652e2705bb667776e76",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 447,
+        "line_start": 248,
+        "symbol": "BigCompute [part 2/5]"
+      }
+    ],
+    "text": "\t}\n\tv50 := 0\n\tif 50 < len(data) {\n\t\tv50 = data[50]\n\t}\n\tv51 := 0\n\tif 51 < len(data) {\n\t\tv51 = data[51]\n\t}\n\tv52 := 0\n\tif 52 < len(data) {\n\t\tv52 = data[52]\n\t}\n\tv53 := 0\n\tif 53 < len(data) {\n\t\tv53 = data[53]\n\t}\n\tv54 := 0\n\tif 54 < len(data) {\n\t\tv54 = data[54]\n\t}\n\tv55 := 0\n\tif 55 < len(data) {\n\t\tv55 = data[55]\n\t}\n\tv56 := 0\n\tif 56 < len(data) {\n\t\tv56 = data[56]\n\t}\n\tv57 := 0\n\tif 57 < len(data) {\n\t\tv57 = data[57]\n\t}\n\tv58 := 0\n\tif 58 < len(data) {\n\t\tv58 = data[58]\n\t}\n\tv59 := 0\n\tif 59 < len(data) {\n\t\tv59 = data[59]\n\t}\n\tv60 := 0\n\tif 60 < len(data) {\n\t\tv60 = data[60]\n\t}\n\tv61 := 0\n\tif 61 < len(data) {\n\t\tv61 = data[61]\n\t}\n\tv62 := 0\n\tif 62 < len(data) {\n\t\tv62 = data[62]\n\t}\n\tv63 := 0\n\tif 63 < len(data) {\n\t\tv63 = data[63]\n\t}\n\tv64 := 0\n\tif 64 < len(data) {\n\t\tv64 = data[64]\n\t}\n\tv65 := 0\n\tif 65 < len(data) {\n\t\tv65 = data[65]\n\t}\n\tv66 := 0\n\tif 66 < len(data) {\n\t\tv66 = data[66]\n\t}\n\tv67 := 0\n\tif 67 < len(data) {\n\t\tv67 = data[67]\n\t}\n\tv68 := 0\n\tif 68 < len(data) {\n\t\tv68 = data[68]\n\t}\n\tv69 := 0\n\tif 69 < len(data) {\n\t\tv69 = data[69]\n\t}\n\tv70 := 0\n\tif 70 < len(data) {\n\t\tv70 = data[70]\n\t}\n\tv71 := 0\n\tif 71 < len(data) {\n\t\tv71 = data[71]\n\t}\n\tv72 := 0\n\tif 72 < len(data) {\n\t\tv72 = data[72]\n\t}\n\tv73 := 0\n\tif 73 < len(data) {\n\t\tv73 = data[73]\n\t}\n\tv74 := 0\n\tif 74 < len(data) {\n\t\tv74 = data[74]\n\t}\n\tv75 := 0\n\tif 75 < len(data) {\n\t\tv75 = data[75]\n\t}\n\tv76 := 0\n\tif 76 < len(data) {\n\t\tv76 = data[76]\n\t}\n\tv77 := 0\n\tif 77 < len(data) {\n\t\tv77 = data[77]\n\t}\n\tv78 := 0\n\tif 78 < len(data) {\n\t\tv78 = data[78]\n\t}\n\tv79 := 0\n\tif 79 < len(data) {\n\t\tv79 = data[79]\n\t}\n\tv80 := 0\n\tif 80 < len(data) {\n\t\tv80 = data[80]\n\t}\n\tv81 := 0\n\tif 81 < len(data) {\n\t\tv81 = data[81]\n\t}\n\tv82 := 0\n\tif 82 < len(data) {\n\t\tv82 = data[82]\n\t}\n\tv83 := 0\n\tif 83 < len(data) {\n\t\tv83 = data[83]\n\t}\n\tv84 := 0\n\tif 84 < len(data) {\n\t\tv84 = data[84]\n\t}\n\tv85 := 0\n\tif 85 < len(data) {\n\t\tv85 = data[85]\n\t}\n\tv86 := 0\n\tif 86 < len(data) {\n\t\tv86 = data[86]\n\t}\n\tv87 := 0\n\tif 87 < len(data) {\n\t\tv87 = data[87]\n\t}\n\tv88 := 0\n\tif 88 < len(data) {\n\t\tv88 = data[88]\n\t}\n\tv89 := 0\n\tif 89 < len(data) {\n\t\tv89 = data[89]\n\t}\n\tv90 := 0\n\tif 90 < len(data) {\n\t\tv90 = data[90]\n\t}\n\tv91 := 0\n\tif 91 < len(data) {\n\t\tv91 = data[91]\n\t}\n\tv92 := 0\n\tif 92 < len(data) {\n\t\tv92 = data[92]\n\t}\n\tv93 := 0\n\tif 93 < len(data) {\n\t\tv93 = data[93]\n\t}\n\tv94 := 0\n\tif 94 < len(data) {\n\t\tv94 = data[94]\n\t}\n\tv95 := 0\n\tif 95 < len(data) {\n\t\tv95 = data[95]\n\t}\n\tv96 := 0\n\tif 96 < len(data) {\n\t\tv96 = data[96]\n\t}\n\tv97 := 0\n\tif 97 < len(data) {\n\t\tv97 = data[97]\n\t}\n\tv98 := 0\n\tif 98 < len(data) {\n\t\tv98 = data[98]\n\t}\n\tv99 := 0\n\tif 99 < len(data) {\n\t\tv99 = data[99]",
+    "token_estimate": 850
+  },
+  {
+    "block_ids": [
+      "5d269745b2e5dbdcbef0c09ba54b0bd6"
+    ],
+    "chunk_id": "e4763e10f059d97f40c2932761b56c3e",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 647,
+        "line_start": 448,
+        "symbol": "BigCompute [part 3/5]"
+      }
+    ],
+    "text": "\t}\n\tv100 := 0\n\tif 100 < len(data) {\n\t\tv100 = data[100]\n\t}\n\tv101 := 0\n\tif 101 < len(data) {\n\t\tv101 = data[101]\n\t}\n\tv102 := 0\n\tif 102 < len(data) {\n\t\tv102 = data[102]\n\t}\n\tv103 := 0\n\tif 103 < len(data) {\n\t\tv103 = data[103]\n\t}\n\tv104 := 0\n\tif 104 < len(data) {\n\t\tv104 = data[104]\n\t}\n\tv105 := 0\n\tif 105 < len(data) {\n\t\tv105 = data[105]\n\t}\n\tv106 := 0\n\tif 106 < len(data) {\n\t\tv106 = data[106]\n\t}\n\tv107 := 0\n\tif 107 < len(data) {\n\t\tv107 = data[107]\n\t}\n\tv108 := 0\n\tif 108 < len(data) {\n\t\tv108 = data[108]\n\t}\n\tv109 := 0\n\tif 109 < len(data) {\n\t\tv109 = data[109]\n\t}\n\tv110 := 0\n\tif 110 < len(data) {\n\t\tv110 = data[110]\n\t}\n\tv111 := 0\n\tif 111 < len(data) {\n\t\tv111 = data[111]\n\t}\n\tv112 := 0\n\tif 112 < len(data) {\n\t\tv112 = data[112]\n\t}\n\tv113 := 0\n\tif 113 < len(data) {\n\t\tv113 = data[113]\n\t}\n\tv114 := 0\n\tif 114 < len(data) {\n\t\tv114 = data[114]\n\t}\n\tv115 := 0\n\tif 115 < len(data) {\n\t\tv115 = data[115]\n\t}\n\tv116 := 0\n\tif 116 < len(data) {\n\t\tv116 = data[116]\n\t}\n\tv117 := 0\n\tif 117 < len(data) {\n\t\tv117 = data[117]\n\t}\n\tv118 := 0\n\tif 118 < len(data) {\n\t\tv118 = data[118]\n\t}\n\tv119 := 0\n\tif 119 < len(data) {\n\t\tv119 = data[119]\n\t}\n\tv120 := 0\n\tif 120 < len(data) {\n\t\tv120 = data[120]\n\t}\n\tv121 := 0\n\tif 121 < len(data) {\n\t\tv121 = data[121]\n\t}\n\tv122 := 0\n\tif 122 < len(data) {\n\t\tv122 = data[122]\n\t}\n\tv123 := 0\n\tif 123 < len(data) {\n\t\tv123 = data[123]\n\t}\n\tv124 := 0\n\tif 124 < len(data) {\n\t\tv124 = data[124]\n\t}\n\tv125 := 0\n\tif 125 < len(data) {\n\t\tv125 = data[125]\n\t}\n\tv126 := 0\n\tif 126 < len(data) {\n\t\tv126 = data[126]\n\t}\n\tv127 := 0\n\tif 127 < len(data) {\n\t\tv127 = data[127]\n\t}\n\tv128 := 0\n\tif 128 < len(data) {\n\t\tv128 = data[128]\n\t}\n\tv129 := 0\n\tif 129 < len(data) {\n\t\tv129 = data[129]\n\t}\n\tv130 := 0\n\tif 130 < len(data) {\n\t\tv130 = data[130]\n\t}\n\tv131 := 0\n\tif 131 < len(data) {\n\t\tv131 = data[131]\n\t}\n\tv132 := 0\n\tif 132 < len(data) {\n\t\tv132 = data[132]\n\t}\n\tv133 := 0\n\tif 133 < len(data) {\n\t\tv133 = data[133]\n\t}\n\tv134 := 0\n\tif 134 < len(data) {\n\t\tv134 = data[134]\n\t}\n\tv135 := 0\n\tif 135 < len(data) {\n\t\tv135 = data[135]\n\t}\n\tv136 := 0\n\tif 136 < len(data) {\n\t\tv136 = data[136]\n\t}\n\tv137 := 0\n\tif 137 < len(data) {\n\t\tv137 = data[137]\n\t}\n\tv138 := 0\n\tif 138 < len(data) {\n\t\tv138 = data[138]\n\t}\n\tv139 := 0\n\tif 139 < len(data) {\n\t\tv139 = data[139]\n\t}\n\tv140 := 0\n\tif 140 < len(data) {\n\t\tv140 = data[140]\n\t}\n\tv141 := 0\n\tif 141 < len(data) {\n\t\tv141 = data[141]\n\t}\n\tv142 := 0\n\tif 142 < len(data) {\n\t\tv142 = data[142]\n\t}\n\tv143 := 0\n\tif 143 < len(data) {\n\t\tv143 = data[143]\n\t}\n\tv144 := 0\n\tif 144 < len(data) {\n\t\tv144 = data[144]\n\t}\n\tv145 := 0\n\tif 145 < len(data) {\n\t\tv145 = data[145]\n\t}\n\tv146 := 0\n\tif 146 < len(data) {\n\t\tv146 = data[146]\n\t}\n\tv147 := 0\n\tif 147 < len(data) {\n\t\tv147 = data[147]\n\t}\n\tv148 := 0\n\tif 148 < len(data) {\n\t\tv148 = data[148]\n\t}\n\tv149 := 0\n\tif 149 < len(data) {\n\t\tv149 = data[149]",
+    "token_estimate": 917
+  },
+  {
+    "block_ids": [
+      "5d269745b2e5dbdcbef0c09ba54b0bd6"
+    ],
+    "chunk_id": "24176c911d0bacf9a29fa7f8251f5036",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 847,
+        "line_start": 648,
+        "symbol": "BigCompute [part 4/5]"
+      }
+    ],
+    "text": "\t}\n\tv150 := 0\n\tif 150 < len(data) {\n\t\tv150 = data[150]\n\t}\n\tv151 := 0\n\tif 151 < len(data) {\n\t\tv151 = data[151]\n\t}\n\tv152 := 0\n\tif 152 < len(data) {\n\t\tv152 = data[152]\n\t}\n\tv153 := 0\n\tif 153 < len(data) {\n\t\tv153 = data[153]\n\t}\n\tv154 := 0\n\tif 154 < len(data) {\n\t\tv154 = data[154]\n\t}\n\tv155 := 0\n\tif 155 < len(data) {\n\t\tv155 = data[155]\n\t}\n\tv156 := 0\n\tif 156 < len(data) {\n\t\tv156 = data[156]\n\t}\n\tv157 := 0\n\tif 157 < len(data) {\n\t\tv157 = data[157]\n\t}\n\tv158 := 0\n\tif 158 < len(data) {\n\t\tv158 = data[158]\n\t}\n\tv159 := 0\n\tif 159 < len(data) {\n\t\tv159 = data[159]\n\t}\n\tv160 := 0\n\tif 160 < len(data) {\n\t\tv160 = data[160]\n\t}\n\tv161 := 0\n\tif 161 < len(data) {\n\t\tv161 = data[161]\n\t}\n\tv162 := 0\n\tif 162 < len(data) {\n\t\tv162 = data[162]\n\t}\n\tv163 := 0\n\tif 163 < len(data) {\n\t\tv163 = data[163]\n\t}\n\tv164 := 0\n\tif 164 < len(data) {\n\t\tv164 = data[164]\n\t}\n\tv165 := 0\n\tif 165 < len(data) {\n\t\tv165 = data[165]\n\t}\n\tv166 := 0\n\tif 166 < len(data) {\n\t\tv166 = data[166]\n\t}\n\tv167 := 0\n\tif 167 < len(data) {\n\t\tv167 = data[167]\n\t}\n\tv168 := 0\n\tif 168 < len(data) {\n\t\tv168 = data[168]\n\t}\n\tv169 := 0\n\tif 169 < len(data) {\n\t\tv169 = data[169]\n\t}\n\tv170 := 0\n\tif 170 < len(data) {\n\t\tv170 = data[170]\n\t}\n\tv171 := 0\n\tif 171 < len(data) {\n\t\tv171 = data[171]\n\t}\n\tv172 := 0\n\tif 172 < len(data) {\n\t\tv172 = data[172]\n\t}\n\tv173 := 0\n\tif 173 < len(data) {\n\t\tv173 = data[173]\n\t}\n\tv174 := 0\n\tif 174 < len(data) {\n\t\tv174 = data[174]\n\t}\n\tv175 := 0\n\tif 175 < len(data) {\n\t\tv175 = data[175]\n\t}\n\tv176 := 0\n\tif 176 < len(data) {\n\t\tv176 = data[176]\n\t}\n\tv177 := 0\n\tif 177 < len(data) {\n\t\tv177 = data[177]\n\t}\n\tv178 := 0\n\tif 178 < len(data) {\n\t\tv178 = data[178]\n\t}\n\tv179 := 0\n\tif 179 < len(data) {\n\t\tv179 = data[179]\n\t}\n\tv180 := 0\n\tif 180 < len(data) {\n\t\tv180 = data[180]\n\t}\n\tv181 := 0\n\tif 181 < len(data) {\n\t\tv181 = data[181]\n\t}\n\tv182 := 0\n\tif 182 < len(data) {\n\t\tv182 = data[182]\n\t}\n\tv183 := 0\n\tif 183 < len(data) {\n\t\tv183 = data[183]\n\t}\n\tv184 := 0\n\tif 184 < len(data) {\n\t\tv184 = data[184]\n\t}\n\tv185 := 0\n\tif 185 < len(data) {\n\t\tv185 = data[185]\n\t}\n\tv186 := 0\n\tif 186 < len(data) {\n\t\tv186 = data[186]\n\t}\n\tv187 := 0\n\tif 187 < len(data) {\n\t\tv187 = data[187]\n\t}\n\tv188 := 0\n\tif 188 < len(data) {\n\t\tv188 = data[188]\n\t}\n\tv189 := 0\n\tif 189 < len(data) {\n\t\tv189 = data[189]\n\t}\n\tv190 := 0\n\tif 190 < len(data) {\n\t\tv190 = data[190]\n\t}\n\tv191 := 0\n\tif 191 < len(data) {\n\t\tv191 = data[191]\n\t}\n\tv192 := 0\n\tif 192 < len(data) {\n\t\tv192 = data[192]\n\t}\n\tv193 := 0\n\tif 193 < len(data) {\n\t\tv193 = data[193]\n\t}\n\tv194 := 0\n\tif 194 < len(data) {\n\t\tv194 = data[194]\n\t}\n\tv195 := 0\n\tif 195 < len(data) {\n\t\tv195 = data[195]\n\t}\n\tv196 := 0\n\tif 196 < len(data) {\n\t\tv196 = data[196]\n\t}\n\tv197 := 0\n\tif 197 < len(data) {\n\t\tv197 = data[197]\n\t}\n\tv198 := 0\n\tif 198 < len(data) {\n\t\tv198 = data[198]\n\t}\n\tv199 := 0\n\tif 199 < len(data) {\n\t\tv199 = data[199]",
+    "token_estimate": 917
+  },
+  {
+    "block_ids": [
+      "5d269745b2e5dbdcbef0c09ba54b0bd6"
+    ],
+    "chunk_id": "438127626378632c03780d10603de32c",
+    "chunker_version": "code-go-ast-v1",
+    "doc_id": "83daba5fbb026e7a400d68a1c4bd36db",
+    "heading_path": [],
+    "policy_hash": "6cfe77abe2b0e5c3",
+    "source_spans": [
+      {
+        "kind": "code",
+        "lang": "go",
+        "line_end": 890,
+        "line_start": 848,
+        "symbol": "BigCompute [part 5/5]"
+      }
+    ],
+    "text": "\t}\n\tv200 := 0\n\tif 200 < len(data) {\n\t\tv200 = data[200]\n\t}\n\tv201 := 0\n\tif 201 < len(data) {\n\t\tv201 = data[201]\n\t}\n\tv202 := 0\n\tif 202 < len(data) {\n\t\tv202 = data[202]\n\t}\n\tv203 := 0\n\tif 203 < len(data) {\n\t\tv203 = data[203]\n\t}\n\tv204 := 0\n\tif 204 < len(data) {\n\t\tv204 = data[204]\n\t}\n\tv205 := 0\n\tif 205 < len(data) {\n\t\tv205 = data[205]\n\t}\n\tv206 := 0\n\tif 206 < len(data) {\n\t\tv206 = data[206]\n\t}\n\tv207 := 0\n\tif 207 < len(data) {\n\t\tv207 = data[207]\n\t}\n\tv208 := 0\n\tif 208 < len(data) {\n\t\tv208 = data[208]\n\t}\n\tv209 := 0\n\tif 209 < len(data) {\n\t\tv209 = data[209]\n\t}\n\treturn len(data)\n}",
+    "token_estimate": 191
+  }
+]
--- a/crates/kebab-cli/src/main.rs
+++ b/crates/kebab-cli/src/main.rs
@@ -275,6 +275,14 @@ enum Cmd {
        #[arg(long, group = "reset_scope")]
        config_only: bool,

+        /// Purge stored docs that are outside the current walker scope
+        /// (config narrowing / removed sub-directory). No filesystem paths
+        /// are removed — this is purely a store-level reconciliation.
+        /// Filesystem existence is NOT checked; anything the current walker
+        /// would not visit is considered an orphan and removed from the store.
+        #[arg(long, group = "reset_scope")]
+        orphans_only: bool,
+
        /// Skip the interactive confirm. Required in non-interactive
        /// contexts (CI, pipes).
        #[arg(long)]
@@ -595,14 +603,20 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
                println!("{}", serde_json::to_string(&wire::wire_ingest(&report))?);
            } else {
                let skipped_breakdown = kebab_app::render_skipped_breakdown(&report.skipped_by_extension);
+                let purged_suffix = if report.purged_deleted_files > 0 {
+                    format!("  purged {}", report.purged_deleted_files)
+                } else {
+                    String::new()
+                };
                println!(
-                    "scanned {}  new {}  updated {}  skipped {}{}  errors {}  ({} ms)",
+                    "scanned {}  new {}  updated {}  skipped {}{}  errors {}{}  ({} ms)",
                    report.scanned,
                    report.new,
                    report.updated,
                    report.skipped,
                    skipped_breakdown,
                    report.errors,
+                    purged_suffix,
                    report.duration_ms
                );
            }
@@ -1088,6 +1102,7 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
            data_only: _,
            vector_only,
            config_only,
+            orphans_only,
            yes,
        } => {
            use kebab_app::ResetScope;
@@ -1101,11 +1116,50 @@ fn run(cli: &Cli) -> anyhow::Result<()> {
                ResetScope::VectorOnly
            } else if *config_only {
                ResetScope::ConfigOnly
+            } else if *orphans_only {
+                ResetScope::OrphansOnly
            } else {
                ResetScope::DataOnly
            };

            let cfg = kebab_config::Config::load(cli.config.as_deref())?;
+
+            if matches!(scope, ResetScope::OrphansOnly) {
+                // OrphansOnly: confirm UI shows orphan count + sample paths
+                // rather than on-disk directory sizes.
+                let orphan_paths = kebab_app::enumerate_orphans(&cfg)?;
+
+                if !*yes {
+                    use std::io::IsTerminal;
+                    if !std::io::stdin().is_terminal() {
+                        anyhow::bail!(
+                            "reset --orphans-only is destructive and stdin is non-interactive — pass --yes to proceed"
+                        );
+                    }
+                    if !confirm_orphans_only(&orphan_paths)? {
+                        if !cli.quiet {
+                            eprintln!("aborted.");
+                        }
+                        return Ok(());
+                    }
+                }
+
+                let report = kebab_app::reset::execute(scope, &cfg)?;
+                if cli.json {
+                    println!("{}", serde_json::to_string(&wire::wire_reset(&report))?);
+                } else {
+                    if report.orphans_purged > 0 {
+                        println!("orphans purged: {}", report.orphans_purged);
+                        for p in &report.purged_paths {
+                            println!("  - {}", p.0);
+                        }
+                    } else {
+                        println!("no orphaned docs found — store is already in sync with walker scope");
+                    }
+                }
+                return Ok(());
+            }
+
            let paths = kebab_app::reset::enumerate_paths(scope, &cfg);
            let bytes = kebab_app::reset::estimate_size_bytes(&paths);

@@ -1444,6 +1498,46 @@ fn confirm_destructive(
    Ok(matches!(s.as_str(), "y" | "yes"))
 }

+/// Confirm prompt for `--orphans-only`: shows the orphan count + a
+/// sample of up to 5 paths so the user knows what will be purged before
+/// committing. No filesystem paths are removed — only store records.
+fn confirm_orphans_only(
+    orphan_paths: &[kebab_core::WorkspacePath],
+) -> anyhow::Result<bool> {
+    use std::io::Write;
+    let n = orphan_paths.len();
+    let mut out = std::io::stderr().lock();
+
+    if n == 0 {
+        writeln!(out, "no orphaned docs found — nothing to purge.")?;
+        out.flush()?;
+        // Nothing to do; treat as confirmed so the caller can emit the
+        // "no orphans" report without prompting.
+        return Ok(true);
+    }
+
+    let sample: Vec<&str> = orphan_paths
+        .iter()
+        .take(5)
+        .map(|p| p.0.as_str())
+        .collect();
+    let sample_str = sample.join(", ");
+    let ellipsis = if n > 5 { ", …" } else { "" };
+
+    writeln!(
+        out,
+        "Purge {n} stored doc(s) outside the current walker scope? (no filesystem paths removed)"
+    )?;
+    writeln!(out, "  sample: {sample_str}{ellipsis}")?;
+    write!(out, "[y/N] ")?;
+    out.flush()?;
+
+    let mut line = String::new();
+    std::io::stdin().read_line(&mut line)?;
+    let s = line.trim().to_ascii_lowercase();
+    Ok(matches!(s.as_str(), "y" | "yes"))
+}
+
 /// p9-fb-35: human-friendly plain output for `kebab fetch`.
 fn render_fetch_plain(r: &kebab_core::FetchResult) {
    println!("# {} ({})", r.doc_path.0, format_kind(r.kind));
--- a/crates/kebab-cli/src/wire.rs
+++ b/crates/kebab-cli/src/wire.rs
@@ -260,6 +260,7 @@ mod tests {
            skipped_generated: 0,
            skipped_size_exceeded: 0,
            skip_examples: SkipExamples::default(),
+            purged_deleted_files: 0,
            items: None,
        };
        let v = wire_ingest(&r);
@@ -364,6 +365,8 @@ mod tests {
            scope: kebab_app::ResetScope::DataOnly,
            removed_paths: vec![std::path::PathBuf::from("/tmp/x")],
            embedding_rows_truncated: 0,
+            orphans_purged: 0,
+            purged_paths: vec![],
        };
        let v = wire_reset(&r);
        assert_eq!(schema_of(&v), Some("reset_report.v1"));
--- a/crates/kebab-core/src/ingest.rs
+++ b/crates/kebab-core/src/ingest.rs
@@ -47,6 +47,12 @@ pub struct IngestReport {
    /// p10-1A-1: sample file paths per skip category (≤ 5 each).
    #[serde(default)]
    pub skip_examples: SkipExamples,
+    /// Dogfood: docs whose on-disk file was deleted since the last ingest
+    /// and were therefore removed from the store. Additive field — older
+    /// wire consumers that pre-date this field read it as 0 via
+    /// `#[serde(default)]`.
+    #[serde(default)]
+    pub purged_deleted_files: u32,
    /// `None` ↔ wire `items: null` (`--summary-only`).
    pub items: Option<Vec<IngestItem>>,
 }
@@ -136,6 +142,7 @@ mod tests {
                builtin_blacklist: vec!["node_modules/x.js".into()],
                gitignore: vec![],
            },
+            purged_deleted_files: 0,
            items: None,
        };
        let v = serde_json::to_value(&r).unwrap();
--- a/crates/kebab-core/src/traits.rs
+++ b/crates/kebab-core/src/traits.rs
@@ -8,7 +8,7 @@ use serde_json::Value;
 use crate::asset::{RawAsset, WorkspacePath};
 use crate::chunk::Chunk;
 use crate::document::{Block, CanonicalDocument};
-use crate::ids::{ChunkId, DocumentId};
+use crate::ids::{AssetId, ChunkId, DocumentId};
 use crate::jobs::{JobFilter, JobId, JobKind, JobRow, JobStatus};
 use crate::media::MediaType;
 use crate::search::{DocFilter, DocSummary, SearchFilters, SearchHit, SearchQuery};
@@ -161,14 +161,51 @@ pub trait DocumentStore {
    fn get_document(&self, id: &DocumentId) -> anyhow::Result<Option<CanonicalDocument>>;
    fn get_chunk(&self, id: &ChunkId) -> anyhow::Result<Option<Chunk>>;
    fn list_documents(&self, filter: &DocFilter) -> anyhow::Result<Vec<DocSummary>>;
+    /// Look up an asset row by its `asset_id` (PRIMARY KEY = blake3
+    /// content hash). Twin-file safe: asset_id is PK so there is
+    /// exactly one row per unique content hash, regardless of how many
+    /// `documents` rows share it. Use this instead of
+    /// `get_asset_by_workspace_path` when you already have a
+    /// `CanonicalDocument` (which carries `source_asset_id`).
+    fn get_asset(&self, id: &AssetId) -> anyhow::Result<Option<RawAsset>>;
+
    /// p9-fb-23: look up an asset row by its workspace path. Used by
    /// the incremental-ingest skip path to compare the freshly
    /// computed blake3 checksum against what's already in SQLite. The
    /// schema enforces a unique workspace_path per asset.
+    ///
+    /// NOTE: for twin files (identical content at different paths),
+    /// `assets.workspace_path` is "last-registered path" — it
+    /// flip-flops on every ingest. Prefer `get_asset` (by asset_id)
+    /// when you have a `CanonicalDocument.source_asset_id`.
    fn get_asset_by_workspace_path(
        &self,
        path: &WorkspacePath,
    ) -> anyhow::Result<Option<RawAsset>>;
+
+    /// Look up a document row by its workspace path. Used by the
+    /// document-centric skip path in `try_skip_unchanged` to avoid the
+    /// twin-file flip-flop that the asset-side lookup suffers from
+    /// (multiple files with identical content share one `assets` row
+    /// whose `workspace_path` is overwritten on every UPSERT, so
+    /// `get_asset_by_workspace_path` returns the wrong twin's path).
+    ///
+    /// `documents.workspace_path` is UNIQUE (V001), so each twin has
+    /// its own stable document row regardless of the asset de-dup.
+    fn get_document_by_workspace_path(
+        &self,
+        path: &WorkspacePath,
+    ) -> anyhow::Result<Option<CanonicalDocument>>;
+
+    /// Return every `workspace_path` stored in the `documents` table.
+    ///
+    /// Used by the post-walker sweep in `kebab-app::ingest` to detect
+    /// documents whose source file has been deleted from the filesystem.
+    /// The set difference `(stored - scanned)` yields orphan candidates;
+    /// each candidate is then existence-checked on disk so that
+    /// out-of-scope files (config narrowing) are NOT purged — only truly
+    /// absent files trigger the purge.
+    fn all_workspace_paths(&self) -> anyhow::Result<Vec<WorkspacePath>>;
 }

 pub trait VectorStore {
--- a/crates/kebab-parse-code/Cargo.toml
+++ b/crates/kebab-parse-code/Cargo.toml
@@ -19,6 +19,7 @@ tree-sitter-rust = { workspace = true }
 tree-sitter-python     = { workspace = true }
 tree-sitter-typescript = { workspace = true }
 tree-sitter-javascript = { workspace = true }
+tree-sitter-go         = { workspace = true }

 [dev-dependencies]
 tempfile = { workspace = true }
--- a/crates/kebab-parse-code/src/go.rs
+++ b/crates/kebab-parse-code/src/go.rs
@@ -0,0 +1,451 @@
+//! `kebab-parse-code::go` — tree-sitter Go AST extractor (P10-1C-Go Task D).
+//!
+//! Implements [`kebab_core::Extractor`] for [`MediaType::Code("go")`].
+//! Walks the tree-sitter parse tree and emits one [`Block::Code`] per
+//! top-level AST semantic unit (free fn, method, each type spec) carrying
+//! [`SourceSpan::Code`] with the unit's self-reference symbol path
+//! (design §3.4 Go row). Glue declarations (`import` / `const` / `var`)
+//! collapse into one grouped `<top-level>` (or `<module>`) unit.
+//!
+//! Unlike the Python/TS/JS extractors which path-derive their module
+//! prefix from the workspace file path, Go's package identity comes from
+//! the source itself (the leading `package` clause) — `extract_package`
+//! reads it from the AST. If the `package_clause` is missing (invalid Go
+//! in practice) the prefix falls back to `"<unknown>"`.
+//!
+//! Doc comments immediately preceding an item are folded into that
+//! item's line range via `unit_start` (1B pattern). Go has no separate
+//! attribute/decorator AST nodes.
+//!
+//! Per design §3.4 / §9.1 / §9 versioning.
+
+use anyhow::Result;
+use kebab_core::{
+    Block, CanonicalDocument, CodeBlock, CommonBlock, Extractor, Lang, MediaType, Metadata,
+    ParserVersion, Provenance, ProvenanceEvent, ProvenanceKind, SourceSpan, SourceType, TrustLevel,
+    id_for_block, id_for_doc,
+};
+use serde_json::Map;
+use time::OffsetDateTime;
+
+use crate::scaffold::{filename_from_workspace_path, join_symbol, strip_extension};
+
+pub const PARSER_VERSION: &str = "code-go-v1";
+
+/// Go AST extractor. Per-unit blocks via tree-sitter-go 0.25
+/// (`LANGUAGE: LanguageFn`) parsed by tree-sitter 0.26.
+pub struct GoAstExtractor;
+
+impl GoAstExtractor {
+    pub fn new() -> Self {
+        Self
+    }
+}
+
+impl Default for GoAstExtractor {
+    fn default() -> Self {
+        Self::new()
+    }
+}
+
+impl Extractor for GoAstExtractor {
+    fn supports(&self, m: &MediaType) -> bool {
+        matches!(m, MediaType::Code(l) if l == "go")
+    }
+
+    fn parser_version(&self) -> ParserVersion {
+        ParserVersion(PARSER_VERSION.to_string())
+    }
+
+    fn extract(
+        &self,
+        ctx: &kebab_core::ExtractContext<'_>,
+        bytes: &[u8],
+    ) -> Result<CanonicalDocument> {
+        let asset = ctx.asset;
+        if !self.supports(&asset.media_type) {
+            anyhow::bail!(
+                "kebab-parse-code: unsupported media_type for GoAstExtractor: {:?}",
+                asset.media_type
+            );
+        }
+
+        let parser_version = self.parser_version();
+        let doc_id = id_for_doc(&asset.workspace_path, &asset.asset_id, &parser_version);
+
+        let source = String::from_utf8(bytes.to_vec())
+            .map_err(|e| anyhow::anyhow!("kebab-parse-code: Go source is not valid UTF-8: {e}"))?;
+
+        let blocks = build_blocks(&source, &doc_id)?;
+        let unit_count = blocks.len() as u32;
+
+        let now = OffsetDateTime::now_utc();
+        let mut events: Vec<ProvenanceEvent> = Vec::with_capacity(2);
+        events.push(ProvenanceEvent {
+            at: asset.discovered_at,
+            agent: "kb-source-fs".to_string(),
+            kind: ProvenanceKind::Discovered,
+            note: None,
+        });
+        events.push(ProvenanceEvent {
+            at: now,
+            agent: "kb-parse-code".to_string(),
+            kind: ProvenanceKind::Parsed,
+            note: Some(format!(
+                "parser_version={}; unit_count={}",
+                parser_version.0, unit_count
+            )),
+        });
+
+        let title = {
+            let fname = filename_from_workspace_path(&asset.workspace_path.0);
+            strip_extension(&fname)
+        };
+
+        // Resolve the file's absolute path for repo detection. If the
+        // source URI carries a relative path, anchor it at the workspace
+        // root so the `.git/` walk-up starts from the right place.
+        let abs_path = match &asset.source_uri {
+            kebab_core::SourceUri::File(p) => {
+                if p.is_absolute() {
+                    p.clone()
+                } else {
+                    ctx.workspace_root.join(p)
+                }
+            }
+            kebab_core::SourceUri::Kb(_) => ctx.workspace_root.to_path_buf(),
+        };
+        let (repo, git_branch, git_commit) = match crate::repo::detect_repo(&abs_path) {
+            Some(r) => (Some(r.name), r.branch, r.commit),
+            None => (None, None, None),
+        };
+
+        let metadata = Metadata {
+            aliases: Vec::new(),
+            tags: Vec::new(),
+            created_at: asset.discovered_at,
+            updated_at: asset.discovered_at,
+            source_type: SourceType::Note,
+            trust_level: TrustLevel::Primary,
+            user_id_alias: None,
+            user: Map::new(),
+            repo,
+            git_branch,
+            git_commit,
+            code_lang: Some("go".to_string()),
+        };
+
+        tracing::debug!(
+            target: "kebab-parse-code",
+            "extracted Go doc_id={} workspace_path={} units={}",
+            doc_id.0,
+            asset.workspace_path.0,
+            unit_count
+        );
+
+        Ok(CanonicalDocument {
+            doc_id,
+            source_asset_id: asset.asset_id.clone(),
+            workspace_path: asset.workspace_path.clone(),
+            title,
+            lang: Lang("und".to_string()),
+            blocks,
+            metadata,
+            provenance: Provenance { events },
+            parser_version,
+            schema_version: 1,
+            doc_version: 1,
+            last_chunker_version: None,
+            last_embedding_version: None,
+        })
+    }
+}
+
+/// p10-1C-Go: extract `package` declaration text from a tree-sitter-go
+/// `source_file`. Returns `None` if no `package_clause` (invalid Go in
+/// practice but defense-in-depth). Per design §3.4 Go row.
+fn extract_package(root: tree_sitter::Node, src: &str) -> Option<String> {
+    let mut cur = root.walk();
+    for child in root.named_children(&mut cur) {
+        if child.kind() == "package_clause" {
+            let mut c2 = child.walk();
+            for sub in child.named_children(&mut c2) {
+                if sub.kind() == "package_identifier" {
+                    return Some(src[sub.start_byte()..sub.end_byte()].to_string());
+                }
+            }
+        }
+    }
+    None
+}
+
+fn build_blocks(
+    source: &str,
+    doc_id: &kebab_core::DocumentId,
+) -> anyhow::Result<Vec<kebab_core::Block>> {
+    let mut parser = tree_sitter::Parser::new();
+    parser
+        .set_language(&tree_sitter_go::LANGUAGE.into())
+        .map_err(|e| anyhow::anyhow!("set tree-sitter-go language: {e}"))?;
+    let tree = parser
+        .parse(source.as_bytes(), None)
+        .ok_or_else(|| anyhow::anyhow!("tree-sitter failed to parse Go source"))?;
+    let lines: Vec<&str> = source.split('\n').collect();
+
+    let root = tree.root_node();
+    let mod_prefix = extract_package(root, source).unwrap_or_else(|| "<unknown>".to_string());
+
+    // units: (symbol, line_start, line_end, is_real_semantic_unit).
+    // Glue groups are pushed with a sentinel symbol + is_real=false so a
+    // post-pass can decide `<module>` vs `<top-level>` (1B post-pass
+    // mirror).
+    let mut units: Vec<(String, u32, u32, bool)> = Vec::new();
+    // (is_import 0/1, s, e). `is_import` flags `import_declaration` —
+    // used by the glue flush to pick `<module>` vs `<top-level>`
+    // provisional label.
+    let mut glue: Vec<(usize, u32, u32)> = Vec::new();
+
+    fn node_name_text<'a>(n: &tree_sitter::Node, src: &'a str) -> Option<&'a str> {
+        n.child_by_field_name("name")
+            .map(|c| &src[c.start_byte()..c.end_byte()])
+    }
+    /// Walk preceding `comment` siblings to extend the unit's line range
+    /// upward, folding leading doc / line comments into the unit. Go has
+    /// no decorator/attribute nodes — doc comments are simply preceding
+    /// `comment` siblings (the 1B pattern).
+    fn unit_start(n: &tree_sitter::Node) -> u32 {
+        let mut start = n.start_position().row as u32 + 1;
+        let mut prev = n.prev_sibling();
+        while let Some(p) = prev {
+            if p.kind() == "comment" {
+                start = p.start_position().row as u32 + 1;
+                prev = p.prev_sibling();
+            } else {
+                break;
+            }
+        }
+        start
+    }
+
+    /// Extract the receiver type text for a `method_declaration`. The
+    /// returned slice INCLUDES the leading `*` for pointer receivers
+    /// (`(*Foo).Bar`) per design §3.4 Go row example. Returns `None` if
+    /// the receiver is malformed (defense in depth).
+    fn receiver_type_text<'a>(method_node: &tree_sitter::Node, src: &'a str) -> Option<&'a str> {
+        let recv = method_node.child_by_field_name("receiver")?;
+        let mut cw = recv.walk();
+        for p in recv.named_children(&mut cw) {
+            if p.kind() == "parameter_declaration" {
+                if let Some(ty) = p.child_by_field_name("type") {
+                    return Some(&src[ty.start_byte()..ty.end_byte()]);
+                }
+            }
+        }
+        None
+    }
+
+    let mut cur = root.walk();
+    for child in root.named_children(&mut cur) {
+        let s = unit_start(&child);
+        let e = child.end_position().row as u32 + 1;
+        match child.kind() {
+            "function_declaration" => {
+                if let Some(name) = node_name_text(&child, source) {
+                    glue.retain(|(_, gs, _)| *gs < s);
+                    flush_glue(&mut glue, &mut units, &mod_prefix);
+                    let sym = join_symbol(&mod_prefix, &[], name);
+                    units.push((sym, s, e, true));
+                }
+            }
+            "method_declaration" => {
+                if let Some(name_node) = child.child_by_field_name("name") {
+                    glue.retain(|(_, gs, _)| *gs < s);
+                    flush_glue(&mut glue, &mut units, &mod_prefix);
+                    let owner = receiver_type_text(&child, source).unwrap_or("<unknown>");
+                    let method_name = &source[name_node.start_byte()..name_node.end_byte()];
+                    let sym = format!("{mod_prefix}.({owner}).{method_name}");
+                    units.push((sym, s, e, true));
+                }
+            }
+            "type_declaration" => {
+                // One unit per inner `type_spec`. Each type_spec gets
+                // the type_declaration's whole upward-folded `s` range
+                // start so doc comments are attached to the first spec;
+                // subsequent specs use their own start. Match 1B
+                // pattern: keep the outer `s` only when there's a single
+                // spec; otherwise use the spec's own start.
+                let mut tcur = child.walk();
+                let specs: Vec<tree_sitter::Node> = child
+                    .named_children(&mut tcur)
+                    .filter(|c| c.kind() == "type_spec")
+                    .collect();
+                let single = specs.len() == 1;
+                for spec in specs {
+                    let name_node = match spec.child_by_field_name("name") {
+                        Some(n) => n,
+                        None => continue,
+                    };
+                    let spec_s = if single {
+                        s
+                    } else {
+                        spec.start_position().row as u32 + 1
+                    };
+                    let spec_e = spec.end_position().row as u32 + 1;
+                    glue.retain(|(_, gs, _)| *gs < spec_s);
+                    flush_glue(&mut glue, &mut units, &mod_prefix);
+                    let name = &source[name_node.start_byte()..name_node.end_byte()];
+                    let sym = join_symbol(&mod_prefix, &[], name);
+                    units.push((sym, spec_s, spec_e, true));
+                }
+            }
+            "import_declaration" => {
+                glue.push((1, s, e));
+            }
+            "const_declaration" | "var_declaration" => {
+                glue.push((0, s, e));
+            }
+            _ => {}
+        }
+    }
+    flush_glue(&mut glue, &mut units, &mod_prefix);
+
+    // `<module>` is correct only when the file produced no real unit.
+    // Otherwise the import/const/var-only group becomes `<top-level>`
+    // (same post-pass as 1B). Match on the suffix so the demotion stays
+    // mod-prefix-agnostic.
+    let has_real_unit = units.iter().any(|(_, _, _, is_real)| *is_real);
+    if has_real_unit {
+        for (sym, _, _, is_real) in units.iter_mut() {
+            if !*is_real && sym.ends_with("<module>") {
+                let pre = &sym[..sym.len() - "<module>".len()];
+                *sym = format!("{pre}<top-level>");
+            }
+        }
+    }
+
+    let total_lines = lines.len() as u32;
+    let mut blocks = Vec::with_capacity(units.len());
+    for (ordinal, (symbol, ls, le, _is_real)) in units.into_iter().enumerate() {
+        let line_start = ls.max(1);
+        let line_end = le.min(total_lines.max(1));
+        let span = SourceSpan::Code {
+            line_start,
+            line_end,
+            symbol: Some(symbol),
+            lang: Some("go".to_string()),
+        };
+        let block_id = id_for_block(doc_id, "code", &[], ordinal as u32, &span);
+        let code = lines[(line_start as usize - 1)..=(line_end as usize - 1)].join("\n");
+        blocks.push(Block::Code(CodeBlock {
+            common: CommonBlock {
+                block_id,
+                heading_path: Vec::new(),
+                source_span: span,
+            },
+            lang: Some("go".to_string()),
+            code,
+        }));
+    }
+    Ok(blocks)
+}
+
+fn flush_glue(
+    glue: &mut Vec<(usize, u32, u32)>,
+    units: &mut Vec<(String, u32, u32, bool)>,
+    mod_prefix: &str,
+) {
+    if glue.is_empty() {
+        return;
+    }
+    let s = glue.iter().map(|(_, a, _)| *a).min().unwrap();
+    let e = glue.iter().map(|(_, _, b)| *b).max().unwrap();
+    // Provisional label: `<module>` only if the group is exclusively
+    // imports (1A's `only_mod_decls` analog). The post-pass demotes any
+    // `<module>` to `<top-level>` if the file produced any real unit.
+    let only_imports = glue.iter().all(|(is_import, _, _)| *is_import == 1);
+    let label = if only_imports { "<module>" } else { "<top-level>" };
+    units.push((join_symbol(mod_prefix, &[], label), s, e, false));
+    glue.clear();
+}
+
+#[cfg(test)]
+mod tests {
+    use super::*;
+    use kebab_core::{Block, MediaType, SourceSpan};
+
+    fn extract_fixture() -> kebab_core::CanonicalDocument {
+        let bytes = std::fs::read(concat!(
+            env!("CARGO_MANIFEST_DIR"),
+            "/tests/fixtures/sample.go"
+        ))
+        .unwrap();
+        // Reuse the cross-language test-support helper promoted in 1B.
+        let asset = crate::rust::tests_support::fixed_code_asset("crates/x/src/sample.go", "go");
+        let cfg = kebab_core::ExtractConfig::default();
+        let root = std::path::PathBuf::from("/tmp");
+        let ctx = kebab_core::ExtractContext {
+            asset: &asset,
+            workspace_root: &root,
+            config: &cfg,
+        };
+        GoAstExtractor::new().extract(&ctx, &bytes).unwrap()
+    }
+
+    #[test]
+    fn extractor_supports_only_media_code_go() {
+        let e = GoAstExtractor::new();
+        assert!(e.supports(&MediaType::Code("go".into())));
+        assert!(!e.supports(&MediaType::Code("rust".into())));
+        assert!(!e.supports(&MediaType::Markdown));
+    }
+
+    #[test]
+    fn go_units_match_design_3_4_symbols() {
+        let doc = extract_fixture();
+        let mut syms: Vec<String> = doc
+            .blocks
+            .iter()
+            .filter_map(|b| match b {
+                Block::Code(c) => match &c.common.source_span {
+                    SourceSpan::Code { symbol, lang, .. } => {
+                        assert_eq!(lang.as_deref(), Some("go"));
+                        symbol.clone()
+                    }
+                    _ => None,
+                },
+                _ => None,
+            })
+            .collect();
+        syms.sort();
+        assert!(syms.iter().any(|s| s == "chunk.Free"), "got {syms:?}");
+        assert!(syms.iter().any(|s| s == "chunk.init"), "got {syms:?}");
+        assert!(
+            syms.iter().any(|s| s == "chunk.MdHeadingV1Chunker"),
+            "got {syms:?}"
+        );
+        assert!(
+            syms.iter()
+                .any(|s| s == "chunk.(*MdHeadingV1Chunker).ChunkDoc"),
+            "got {syms:?}"
+        );
+        assert!(
+            syms.iter()
+                .any(|s| s == "chunk.(MdHeadingV1Chunker).Name2"),
+            "got {syms:?}"
+        );
+        assert!(syms.iter().any(|s| s == "chunk.Stringer"), "got {syms:?}");
+        // import + const grouped into one glue unit (no isolated `<module>`).
+        assert!(
+            syms.iter().any(|s| s == "chunk.<top-level>"),
+            "got {syms:?}"
+        );
+    }
+
+    #[test]
+    fn deterministic_across_runs() {
+        let a = extract_fixture();
+        for _ in 0..50 {
+            assert_eq!(extract_fixture().blocks, a.blocks);
+        }
+    }
+}
--- a/crates/kebab-parse-code/src/lang.rs
+++ b/crates/kebab-parse-code/src/lang.rs
@@ -24,7 +24,7 @@ pub fn code_lang_for_path(path: &Path) -> Option<&'static str> {
    match ext.as_str() {
        "rs" => Some("rust"),
        "py" | "pyi" => Some("python"),
-        "ts" | "tsx" => Some("typescript"),
+        "ts" | "tsx" | "mts" | "cts" => Some("typescript"),
        "js" | "mjs" | "cjs" | "jsx" => Some("javascript"),
        "go" => Some("go"),
        "java" => Some("java"),
@@ -82,7 +82,7 @@ pub fn module_path_for_python(workspace_path: &str) -> String {
 /// (no slash replacement, no source-root strip). See plan §Task C.
 pub fn module_path_for_tsjs(workspace_path: &str) -> String {
    let p = workspace_path;
-    for ext in [".tsx", ".ts", ".jsx", ".mjs", ".cjs", ".js"] {
+    for ext in [".tsx", ".mts", ".cts", ".ts", ".jsx", ".mjs", ".cjs", ".js"] {
        if let Some(stripped) = p.strip_suffix(ext) {
            return stripped.to_string();
        }
@@ -110,7 +110,7 @@ mod tests {

    #[test]
    fn module_path_for_tsjs_keeps_slashes_and_strips_ext() {
-        for ext in ["ts", "tsx", "js", "jsx", "mjs", "cjs"] {
+        for ext in ["ts", "tsx", "mts", "cts", "js", "jsx", "mjs", "cjs"] {
            let p = format!("src/search/retriever/Retriever.{ext}");
            assert_eq!(module_path_for_tsjs(&p), "src/search/retriever/Retriever");
        }
--- a/crates/kebab-parse-code/src/lib.rs
+++ b/crates/kebab-parse-code/src/lib.rs
@@ -13,6 +13,7 @@
 //! `kebab-parse-*` crates per design §8: must NOT depend on store / embed
 //! / llm / rag.

+pub mod go;
 pub mod javascript;
 pub mod lang;
 pub mod python;
@@ -22,6 +23,7 @@ pub(crate) mod scaffold;
 pub mod skip;
 pub mod typescript;

+pub use go::{PARSER_VERSION as GO_PARSER_VERSION, GoAstExtractor};
 pub use javascript::{PARSER_VERSION as JS_PARSER_VERSION, JavascriptAstExtractor};
 pub use lang::{code_lang_for_path, module_path_for_python, module_path_for_tsjs};
 pub use python::{PARSER_VERSION as PYTHON_PARSER_VERSION, PythonAstExtractor};
--- a/crates/kebab-parse-code/src/typescript.rs
+++ b/crates/kebab-parse-code/src/typescript.rs
@@ -173,8 +173,9 @@ impl Extractor for TypescriptAstExtractor {
 }

 /// Select the tree-sitter grammar based on the workspace path's
-/// extension. `.tsx` → TSX grammar; everything else (`.ts`, `.d.ts`,
-/// missing extension) → TypeScript grammar.
+/// extension. `.tsx` → TSX grammar; everything else (`.ts`, `.mts`,
+/// `.cts`, `.d.ts`, missing extension) → TypeScript grammar (the JSX-
+/// agnostic variants all share one grammar in tree-sitter-typescript 0.23).
 fn select_grammar(workspace_path: &str) -> tree_sitter::Language {
    if workspace_path.ends_with(".tsx") {
        tree_sitter_typescript::LANGUAGE_TSX.into()
--- a/crates/kebab-parse-code/tests/fixtures/sample.go
+++ b/crates/kebab-parse-code/tests/fixtures/sample.go
@@ -0,0 +1,34 @@
+// sample.go
+package chunk
+
+import (
+	"fmt"
+	"strings"
+)
+
+const Version = "v1"
+
+type MdHeadingV1Chunker struct {
+	Name string
+}
+
+// ChunkDoc returns a stub list of strings.
+func (m *MdHeadingV1Chunker) ChunkDoc(input string) []string {
+	return []string{m.Name}
+}
+
+func (m MdHeadingV1Chunker) Name2() string {
+	return m.Name
+}
+
+type Stringer interface {
+	String() string
+}
+
+func Free(x int) int {
+	return x + 1
+}
+
+func init() {
+	fmt.Println(strings.ToUpper("init"))
+}
--- a/crates/kebab-parse-code/tests/lang.rs
+++ b/crates/kebab-parse-code/tests/lang.rs
@@ -9,6 +9,8 @@ fn known_extensions_map_to_canonical_identifiers() {
        ("foo.pyi", Some("python")),
        ("foo.ts", Some("typescript")),
        ("foo.tsx", Some("typescript")),
+        ("foo.mts", Some("typescript")),  // ESM TS — same grammar
+        ("foo.cts", Some("typescript")),  // CommonJS TS — same grammar
        ("foo.js", Some("javascript")),
        ("foo.mjs", Some("javascript")),
        ("foo.cjs", Some("javascript")),
--- a/crates/kebab-source-fs/Cargo.toml
+++ b/crates/kebab-source-fs/Cargo.toml
@@ -18,6 +18,7 @@ blake3       = { workspace = true }
 tracing      = { workspace = true }
 walkdir      = "2"
 ignore       = "0.4"
+globset      = "0.4"

 [dev-dependencies]
 serde_json   = { workspace = true }
--- a/crates/kebab-source-fs/src/connector.rs
+++ b/crates/kebab-source-fs/src/connector.rs
@@ -86,7 +86,7 @@ impl FsSourceConnector {
        excludes.extend(scope.exclude.iter().cloned());
        let kbignore = read_kbignore(&root)?;

-        let overrides = build_overrides(&root, &excludes, &kbignore)?;
+        let overrides = build_overrides(&root, &excludes, &kbignore, &scope.include)?;
        Ok((root, overrides))
    }

@@ -103,8 +103,6 @@ impl FsSourceConnector {
    ) -> Result<(Vec<RawAsset>, FsScanSkips)> {
        let (root, overrides) = self.resolve_scan_params(scope)?;

-        log_scope_include_warning(scope);
-
        let (files, skipped_entries) = walk_files_with_skips(&root, &overrides)?;

        // Accumulate per-category skip counts and sample paths.
@@ -284,14 +282,6 @@ fn build_assets(
    Ok(assets)
 }

-fn log_scope_include_warning(scope: &SourceScope) {
-    if !scope.include.is_empty() {
-        tracing::debug!(
-            count = scope.include.len(),
-            "FsSourceConnector ignores scope.include — handled by extractor router"
-        );
-    }
-}

 impl SourceConnector for FsSourceConnector {
    fn scan(&self, scope: &SourceScope) -> Result<Vec<RawAsset>> {
--- a/crates/kebab-source-fs/src/media.rs
+++ b/crates/kebab-source-fs/src/media.rs
@@ -19,7 +19,9 @@ pub(crate) fn media_type_for(path: &Path) -> MediaType {
        .unwrap_or_default();

    match ext.as_str() {
-        "md" => MediaType::Markdown,
+        // Markdown + MDX (markdown + JSX, treated as plain markdown — the
+        // JSX islands are folded into raw passthrough by the md parser).
+        "md" | "mdx" => MediaType::Markdown,
        "pdf" => MediaType::Pdf,

        "png" => MediaType::Image(ImageType::Png),
@@ -40,9 +42,13 @@ pub(crate) fn media_type_for(path: &Path) -> MediaType {

        // p10-1B: Python / TS / JS AST chunkers active.
        "py" | "pyi"               => MediaType::Code("python".into()),
-        "ts" | "tsx"               => MediaType::Code("typescript".into()),
+        // .mts / .cts are TypeScript ESM / CommonJS variants — same grammar.
+        "ts" | "tsx" | "mts" | "cts" => MediaType::Code("typescript".into()),
        "js" | "mjs" | "cjs" | "jsx" => MediaType::Code("javascript".into()),

+        // p10-1C-Go: Go ingest activated.
+        "go" => MediaType::Code("go".into()),
+
        // Empty string (no extension) and any other extension: bucket as
        // Other and let downstream extractors decide if they support it.
        _ => MediaType::Other(ext),
@@ -102,6 +108,25 @@ mod tests {
        assert_eq!(media_type_for(Path::new("a/b.rs")),    MediaType::Code("rust".into()));
    }

+    #[test]
+    fn ts_variants_mts_cts() {
+        // .mts / .cts are TypeScript ESM / CommonJS — same grammar as .ts.
+        assert_eq!(media_type_for(Path::new("a/b.mts")), MediaType::Code("typescript".into()));
+        assert_eq!(media_type_for(Path::new("a/b.cts")), MediaType::Code("typescript".into()));
+    }
+
+    #[test]
+    fn mdx_routes_to_markdown() {
+        // MDX is markdown with JSX islands; the md parser folds the JSX
+        // through as raw passthrough.
+        assert_eq!(media_type_for(Path::new("docs/page.mdx")), MediaType::Markdown);
+    }
+
+    #[test]
+    fn go_files_map_to_media_code_go() {
+        assert_eq!(media_type_for(Path::new("a/b.go")), MediaType::Code("go".into()));
+    }
+
    #[test]
    fn unknown_and_missing_extension() {
        assert_eq!(
--- a/crates/kebab-source-fs/src/walker.rs
+++ b/crates/kebab-source-fs/src/walker.rs
@@ -44,6 +44,7 @@ use std::collections::HashSet;
 use std::path::{Path, PathBuf};

 use anyhow::{Context, Result};
+use globset::{GlobBuilder, GlobSet, GlobSetBuilder};
 use ignore::overrides::{Override, OverrideBuilder};
 use walkdir::WalkDir;

@@ -69,6 +70,11 @@ const DEFAULT_EXCLUDES: &[&str] = &[
 ///
 /// `default_and_config` covers DEFAULT_EXCLUDES + `config.workspace.exclude`
 /// — these do NOT map to any of the three named `IngestReport` counters.
+///
+/// `include` is the compiled `scope.include` allow-list. When the set is
+/// empty (no patterns) every file passes; when non-empty a file must match
+/// at least one pattern to be accepted (directories always pass, so the
+/// walker can still descend into them).
 pub(crate) struct WalkOverrides {
    /// Merged matcher — same as today's `Override`; used for the walk decision.
    pub combined: Override,
@@ -78,6 +84,8 @@ pub(crate) struct WalkOverrides {
    pub kebabignore: Override,
    /// Matcher built from `kebab_parse_code::BUILTIN_BLACKLIST` only.
    pub builtin: Override,
+    /// Compiled allow-list from `scope.include`. Empty set = pass all.
+    pub include: GlobSet,
 }

 /// Skip attribution category. Used by the connector when counting per-source
@@ -161,10 +169,15 @@ fn build_single_matcher_owned(root: &Path, patterns: &[String]) -> Result<Overri
 /// The three per-source matchers (`gitignore`, `kebabignore`, `builtin`) are
 /// built in addition to the combined one so the connector can attribute skips
 /// to the correct `IngestReport` counter without a second walker pass.
+///
+/// `include_patterns` (from `scope.include`) are compiled into an allow-list
+/// `GlobSet`. Empty slice → pass-all (backward-compat); non-empty → file
+/// must match at least one pattern to be accepted.
 pub(crate) fn build_overrides(
    root: &Path,
    config_exclude: &[String],
    kbignore_patterns: &[String],
+    include_patterns: &[String],
 ) -> Result<WalkOverrides> {
    let gitignore_patterns = read_gitignore(root)?;

@@ -209,14 +222,41 @@ pub(crate) fn build_overrides(
        .build()
        .context("failed to compile combined override set")?;

+    // Allow-list GlobSet: empty Vec → matches nothing (= pass all); non-empty
+    // → file must match at least one glob to be accepted. We compile with
+    // `case_insensitive=false` to keep the semantics consistent with the
+    // OverrideBuilder exclude patterns above.
+    let include = build_include_globset(include_patterns)?;
+
    Ok(WalkOverrides {
        combined,
        gitignore,
        kebabignore,
        builtin,
+        include,
    })
 }

+/// Compile `scope.include` patterns into a `GlobSet` allow-list.
+///
+/// Each pattern uses `GlobBuilder` with `literal_separator = true` so that
+/// `**` can cross directory boundaries while `*` stops at `/`, matching the
+/// gitignore convention used throughout the rest of the walker.
+///
+/// An empty slice produces an empty `GlobSet` — callers interpret that as
+/// "pass all files" (no allow-list constraint).
+fn build_include_globset(patterns: &[String]) -> Result<GlobSet> {
+    let mut builder = GlobSetBuilder::new();
+    for pat in patterns {
+        let glob = GlobBuilder::new(pat)
+            .literal_separator(true)
+            .build()
+            .with_context(|| format!("invalid include pattern: {pat}"))?;
+        builder.add(glob);
+    }
+    builder.build().context("failed to compile include globset")
+}
+
 /// Classify why a path was excluded, using per-source matchers in spec §5.2
 /// priority order: built-in > gitignore > kebabignore > other.
 ///
@@ -391,6 +431,13 @@ pub(crate) fn walk_files_with_skips(
        }

        if entry.file_type().is_file() {
+            // Apply include allow-list: if non-empty, the file's path
+            // relative to root must match at least one pattern.
+            if !overrides.include.is_empty() && !overrides.include.is_match(rel) {
+                // Not in the allow-list — silently drop (no skip counter;
+                // the include filter is not a "skip" source in IngestReport).
+                continue;
+            }
            accepted.push(path.to_path_buf());
        }
    }
@@ -406,7 +453,7 @@ mod tests {
    #[test]
    fn empty_inputs_compile_into_an_override() {
        let dir = tempfile::tempdir().unwrap();
-        let ov = build_overrides(dir.path(), &[], &[]).unwrap();
+        let ov = build_overrides(dir.path(), &[], &[], &[]).unwrap();
        // Default-excludes only; non-special files should not match.
        let m = ov.combined.matched(Path::new("notes/alpha.md"), false);
        assert!(!m.is_ignore());
@@ -415,7 +462,7 @@ mod tests {
    #[test]
    fn default_excludes_ds_store_and_resource_forks() {
        let dir = tempfile::tempdir().unwrap();
-        let ov = build_overrides(dir.path(), &[], &[]).unwrap();
+        let ov = build_overrides(dir.path(), &[], &[], &[]).unwrap();
        assert!(ov.combined.matched(Path::new(".DS_Store"), false).is_ignore());
        assert!(
            ov.combined.matched(Path::new("notes/.DS_Store"), false).is_ignore()
@@ -433,6 +480,7 @@ mod tests {
            dir.path(),
            &["*.tmp".to_string(), "node_modules/**".to_string()],
            &[],
+            &[],
        )
        .unwrap();
        assert!(ov.combined.matched(Path::new("a.tmp"), false).is_ignore());
@@ -452,6 +500,7 @@ mod tests {
            dir.path(),
            &["*.tmp".to_string()],
            &["secret/**".to_string()],
+            &[],
        )
        .unwrap();
        assert!(ov.combined.matched(Path::new("a.tmp"), false).is_ignore());
@@ -491,7 +540,7 @@ mod tests {
        fs::write(root.join("src/main.rs"), "x").unwrap();
        fs::write(root.join("node_modules/foo/bar.js"), "x").unwrap();

-        let overrides = build_overrides(root, &[], &[]).unwrap();
+        let overrides = build_overrides(root, &[], &[], &[]).unwrap();
        // Override::matched expects paths relative to the builder's root.
        let m_in = overrides.combined.matched(Path::new("src/main.rs"), false);
        let m_out = overrides.combined.matched(Path::new("node_modules/foo/bar.js"), false);
@@ -514,7 +563,7 @@ mod tests {
        fs::create_dir_all(root.join("ok")).unwrap();
        fs::write(root.join("ok/z.txt"), "z").unwrap();

-        let overrides = build_overrides(root, &[], &[]).unwrap();
+        let overrides = build_overrides(root, &[], &[], &[]).unwrap();
        // Override::matched expects paths relative to the builder's root.
        for blacklisted in [
            "target/x/y.txt",
@@ -544,7 +593,7 @@ mod tests {
        fs::create_dir_all(root.join("dist")).unwrap();
        fs::write(root.join("dist/bundle.js"), "x").unwrap();

-        let overrides = build_overrides(root, &[], &[]).unwrap();
+        let overrides = build_overrides(root, &[], &[], &[]).unwrap();
        assert!(overrides.combined.matched(Path::new("a.log"), false).is_ignore());
        assert!(overrides.combined.matched(Path::new("dist/bundle.js"), false).is_ignore());
        assert!(!overrides.combined.matched(Path::new("src/main.rs"), false).is_ignore());
@@ -562,7 +611,7 @@ mod tests {
        fs::write(root.join("src/main.rs"), "x").unwrap();

        // No .gitignore present — patterns from .gitignore should not affect overrides.
-        let overrides = build_overrides(root, &[], &[]).unwrap();
+        let overrides = build_overrides(root, &[], &[], &[]).unwrap();
        assert!(!overrides.combined.matched(Path::new("a.log"), false).is_ignore());
        assert!(!overrides.combined.matched(Path::new("src/main.rs"), false).is_ignore());
    }
@@ -577,7 +626,7 @@ mod tests {
        // semantics, but at minimum it must not produce double-`!` corruption.
        fs::write(root.join(".gitignore"), "!keep/\n").unwrap();
        // Just verify build_overrides doesn't error.
-        let result = build_overrides(root, &[], &[]);
+        let result = build_overrides(root, &[], &[], &[]);
        assert!(result.is_ok(), "should not error on negation pattern: {:?}", result.err());
    }

@@ -594,7 +643,7 @@ mod tests {
        // .gitignore entry. Builtin must win (priority order §5.2).
        fs::write(root.join(".gitignore"), "node_modules/\n").unwrap();

-        let ov = build_overrides(root, &[], &[]).unwrap();
+        let ov = build_overrides(root, &[], &[], &[]).unwrap();
        // node_modules/ dir itself
        let cat = classify_skip(Path::new("node_modules"), true, &ov);
        assert_eq!(cat, SkipCategory::BuiltinBlacklist, "builtin must have priority");
@@ -609,7 +658,7 @@ mod tests {
        let root = tmp.path();
        fs::write(root.join(".gitignore"), "*.log\n").unwrap();

-        let ov = build_overrides(root, &[], &[]).unwrap();
+        let ov = build_overrides(root, &[], &[], &[]).unwrap();
        let cat = classify_skip(Path::new("app.log"), false, &ov);
        assert_eq!(cat, SkipCategory::Gitignore);
    }
@@ -621,7 +670,7 @@ mod tests {
        let tmp = TempDir::new().unwrap();
        let root = tmp.path();

-        let ov = build_overrides(root, &[], &["*.secret".to_string()]).unwrap();
+        let ov = build_overrides(root, &[], &["*.secret".to_string()], &[]).unwrap();
        let cat = classify_skip(Path::new("creds.secret"), false, &ov);
        assert_eq!(cat, SkipCategory::Kebabignore);
    }
@@ -637,7 +686,7 @@ mod tests {
        fs::write(root.join("ok.md"), "# ok").unwrap();
        fs::write(root.join("skipme.log"), "x").unwrap();

-        let ov = build_overrides(root, &[], &[]).unwrap();
+        let ov = build_overrides(root, &[], &[], &[]).unwrap();
        let (accepted, skipped_entries) = walk_files_with_skips(root, &ov).unwrap();

        let accepted_names: Vec<_> = accepted
@@ -677,7 +726,7 @@ mod tests {
        fs::write(root.join("node_modules/foo/bar.js"), "x").unwrap();
        fs::write(root.join("ok.md"), "# ok").unwrap();

-        let ov = build_overrides(root, &[], &[]).unwrap();
+        let ov = build_overrides(root, &[], &[], &[]).unwrap();
        let (accepted, skipped_entries) = walk_files_with_skips(root, &ov).unwrap();

        let accepted_names: Vec<_> = accepted
--- a/crates/kebab-source-fs/tests/include_allowlist.rs
+++ b/crates/kebab-source-fs/tests/include_allowlist.rs
@@ -0,0 +1,111 @@
+//! Integration test: `scope.include` enforces an allow-list.
+//!
+//! Semantics (gitignore convention):
+//!   - `include` is empty Vec → all files pass through (backward-compat).
+//!   - `include` is non-empty → only files matching at least one pattern
+//!     are accepted. `exclude` rules still apply after include.
+//!
+//! Layout (built per-test in a TempDir):
+//!   root/
+//!   ├── a.md
+//!   ├── b.py
+//!   ├── c.png
+//!   └── d.pdf
+
+use std::fs;
+
+use kebab_config::Config;
+use kebab_core::{SourceConnector, SourceScope};
+use kebab_source_fs::FsSourceConnector;
+
+fn cfg_with_root(root: &str) -> Config {
+    let mut c = Config::defaults();
+    c.workspace.root = root.to_string();
+    c.workspace.exclude.clear();
+    // Disable size / generated caps so small test files always pass.
+    c.ingest.code.max_file_bytes = u64::MAX;
+    c.ingest.code.max_file_lines = u32::MAX;
+    c.ingest.code.skip_generated_header = false;
+    c
+}
+
+fn setup_mixed_dir() -> tempfile::TempDir {
+    let dir = tempfile::tempdir().unwrap();
+    let root = dir.path();
+    fs::write(root.join("a.md"), b"md").unwrap();
+    fs::write(root.join("b.py"), b"py").unwrap();
+    fs::write(root.join("c.png"), b"\x89PNG").unwrap();
+    fs::write(root.join("d.pdf"), b"%PDF").unwrap();
+    dir
+}
+
+/// Empty include → all 4 files pass (backward-compat).
+#[test]
+fn include_empty_accepts_all_files() {
+    let dir = setup_mixed_dir();
+    let conn = FsSourceConnector::new(&cfg_with_root(dir.path().to_str().unwrap())).unwrap();
+    let scope = SourceScope {
+        include: vec![],
+        ..SourceScope::default()
+    };
+    let assets = conn.scan(&scope).unwrap();
+    let names: Vec<_> = assets.iter().map(|a| a.workspace_path.0.clone()).collect();
+    assert!(names.contains(&"a.md".to_string()), "a.md missing; got: {names:?}");
+    assert!(names.contains(&"b.py".to_string()), "b.py missing; got: {names:?}");
+    assert!(names.contains(&"c.png".to_string()), "c.png missing; got: {names:?}");
+    assert!(names.contains(&"d.pdf".to_string()), "d.pdf missing; got: {names:?}");
+    assert_eq!(names.len(), 4, "expected exactly 4 files; got: {names:?}");
+}
+
+/// Non-empty include → only md + py come back; png + pdf are excluded.
+#[test]
+fn include_nonempty_is_allowlist() {
+    let dir = setup_mixed_dir();
+    let conn = FsSourceConnector::new(&cfg_with_root(dir.path().to_str().unwrap())).unwrap();
+    let scope = SourceScope {
+        include: vec!["**/*.md".to_string(), "**/*.py".to_string()],
+        ..SourceScope::default()
+    };
+    let assets = conn.scan(&scope).unwrap();
+    let names: Vec<_> = assets.iter().map(|a| a.workspace_path.0.clone()).collect();
+    assert!(names.contains(&"a.md".to_string()), "a.md should be accepted; got: {names:?}");
+    assert!(names.contains(&"b.py".to_string()), "b.py should be accepted; got: {names:?}");
+    assert!(
+        !names.contains(&"c.png".to_string()),
+        "c.png must be rejected by include allowlist; got: {names:?}"
+    );
+    assert!(
+        !names.contains(&"d.pdf".to_string()),
+        "d.pdf must be rejected by include allowlist; got: {names:?}"
+    );
+    assert_eq!(names.len(), 2, "expected exactly 2 files; got: {names:?}");
+}
+
+/// include + exclude are ANDed: a file matching include but also matching
+/// exclude must be rejected.
+#[test]
+fn include_and_exclude_are_anded() {
+    let dir = tempfile::tempdir().unwrap();
+    let root = dir.path();
+    fs::write(root.join("keep.md"), b"keep").unwrap();
+    fs::write(root.join("drop.md"), b"drop").unwrap();
+    fs::write(root.join("other.py"), b"py").unwrap();
+
+    let conn = FsSourceConnector::new(&cfg_with_root(root.to_str().unwrap())).unwrap();
+    let scope = SourceScope {
+        include: vec!["**/*.md".to_string()],
+        exclude: vec!["drop.md".to_string()],
+        ..SourceScope::default()
+    };
+    let assets = conn.scan(&scope).unwrap();
+    let names: Vec<_> = assets.iter().map(|a| a.workspace_path.0.clone()).collect();
+    assert!(names.contains(&"keep.md".to_string()), "keep.md should be accepted; got: {names:?}");
+    assert!(
+        !names.contains(&"drop.md".to_string()),
+        "drop.md should be excluded (matched exclude); got: {names:?}"
+    );
+    assert!(
+        !names.contains(&"other.py".to_string()),
+        "other.py should be excluded (not in include); got: {names:?}"
+    );
+}
--- a/crates/kebab-store-sqlite/snapshots/ingest_report.snapshot.json
+++ b/crates/kebab-store-sqlite/snapshots/ingest_report.snapshot.json
@@ -56,5 +56,6 @@
  "skipped_kebabignore": 0,
  "skipped_size_exceeded": 0,
  "unchanged": 0,
+  "purged_deleted_files": 0,
  "updated": 1
 }
--- a/crates/kebab-store-sqlite/src/documents.rs
+++ b/crates/kebab-store-sqlite/src/documents.rs
@@ -264,6 +264,28 @@ impl kebab_core::DocumentStore for SqliteStore {
        }))
    }

+    fn get_asset(
+        &self,
+        id: &kebab_core::AssetId,
+    ) -> Result<Option<kebab_core::RawAsset>> {
+        let conn = self.lock_conn();
+        let result = conn.query_row(
+            r#"SELECT
+                asset_id, source_uri, workspace_path, media_type,
+                byte_len, checksum, storage_kind, storage_path,
+                discovered_at
+            FROM assets
+            WHERE asset_id = ?"#,
+            rusqlite::params![id.0.as_str()],
+            asset_from_row,
+        );
+        match result {
+            Ok(asset) => Ok(Some(asset)),
+            Err(rusqlite::Error::QueryReturnedNoRows) => Ok(None),
+            Err(e) => Err(e.into()),
+        }
+    }
+
    fn get_asset_by_workspace_path(
        &self,
        path: &kebab_core::WorkspacePath,
@@ -286,6 +308,88 @@ impl kebab_core::DocumentStore for SqliteStore {
        }
    }

+    fn get_document_by_workspace_path(
+        &self,
+        path: &kebab_core::WorkspacePath,
+    ) -> Result<Option<kebab_core::CanonicalDocument>> {
+        let conn = self.lock_conn();
+        let row: Option<DocumentRow> = conn
+            .query_row(
+                "SELECT
+                    doc_id, asset_id, workspace_path, title, lang,
+                    source_type, trust_level, parser_version,
+                    doc_version, schema_version, metadata_json,
+                    provenance_json, created_at, updated_at,
+                    last_chunker_version, last_embedding_version
+                FROM documents WHERE workspace_path = ?",
+                params![path.0],
+                document_row_from_sql,
+            )
+            .map(Some)
+            .or_else(rows_optional)
+            .map_err(StoreError::from)?;
+        let Some(row) = row else { return Ok(None) };
+
+        let doc_id = kebab_core::DocumentId(row.doc_id.clone());
+        let mut blocks_stmt = conn
+            .prepare(
+                "SELECT payload_json FROM blocks
+                 WHERE doc_id = ? ORDER BY ordinal ASC",
+            )
+            .map_err(StoreError::from)?;
+        let block_rows = blocks_stmt
+            .query_map(params![row.doc_id], |r| {
+                let payload_json: String = r.get(0)?;
+                Ok(payload_json)
+            })
+            .map_err(StoreError::from)?;
+        let mut blocks: Vec<kebab_core::Block> = Vec::new();
+        for block_row in block_rows {
+            let payload_json = block_row.map_err(StoreError::from)?;
+            let block: kebab_core::Block = serde_json::from_str(&payload_json)
+                .context("deserialize block payload_json")?;
+            blocks.push(block);
+        }
+
+        let metadata: kebab_core::Metadata = serde_json::from_str(&row.metadata_json)
+            .context("deserialize metadata_json")?;
+        let provenance: kebab_core::Provenance =
+            serde_json::from_str(&row.provenance_json)
+                .context("deserialize provenance_json")?;
+
+        Ok(Some(kebab_core::CanonicalDocument {
+            doc_id,
+            source_asset_id: kebab_core::AssetId(row.asset_id),
+            workspace_path: kebab_core::WorkspacePath(row.workspace_path),
+            title: row.title.unwrap_or_default(),
+            lang: kebab_core::Lang(row.lang.unwrap_or_default()),
+            blocks,
+            metadata,
+            provenance,
+            parser_version: kebab_core::ParserVersion(row.parser_version),
+            schema_version: row.schema_version as u32,
+            doc_version: row.doc_version as u32,
+            last_chunker_version: row.last_chunker_version.map(kebab_core::ChunkerVersion),
+            last_embedding_version: row.last_embedding_version.map(kebab_core::EmbeddingVersion),
+        }))
+    }
+
+    fn all_workspace_paths(&self) -> Result<Vec<kebab_core::WorkspacePath>> {
+        let conn = self.lock_conn();
+        let mut stmt = conn
+            .prepare("SELECT workspace_path FROM documents")
+            .map_err(StoreError::from)?;
+        let rows = stmt
+            .query_map([], |r| r.get::<_, String>(0))
+            .map_err(StoreError::from)?;
+        let mut out = Vec::new();
+        for row in rows {
+            let path = row.map_err(StoreError::from)?;
+            out.push(kebab_core::WorkspacePath(path));
+        }
+        Ok(out)
+    }
+
    fn list_documents(
        &self,
        filter: &kebab_core::DocFilter,
@@ -550,7 +654,8 @@ fn rows_optional<T>(err: rusqlite::Error) -> rusqlite::Result<Option<T>> {

 /// Reconstruct a [`kebab_core::RawAsset`] from one `assets` row.
 /// Row mapper for `RawAsset`. Column names are self-documenting; the
-/// SELECT in [`DocumentStore::get_asset_by_workspace_path`] must include
+/// SELECTs in [`DocumentStore::get_asset`] and
+/// [`DocumentStore::get_asset_by_workspace_path`] must both include
 /// all nine columns by their schema names.
 fn asset_from_row(row: &rusqlite::Row<'_>) -> rusqlite::Result<kebab_core::RawAsset> {
    use std::path::PathBuf;
--- a/crates/kebab-store-sqlite/src/lib.rs
+++ b/crates/kebab-store-sqlite/src/lib.rs
@@ -35,4 +35,4 @@ pub use error::StoreError;
 pub use eval::{EvalQueryResultRecord, EvalRunRecord, EvalRunRow};
 pub use fts::rebuild_chunks_fts;
 pub use jobs::IngestRunRow;
-pub use store::{CountSummary, NotIndexed, SqliteStore};
+pub use store::{CountSummary, NotIndexed, SqliteStore, purge_deleted_workspace_path};
--- a/crates/kebab-store-sqlite/src/store.rs
+++ b/crates/kebab-store-sqlite/src/store.rs
@@ -540,10 +540,132 @@ pub(crate) fn purge_orphan_at_workspace_path(
    Ok(())
 }

+/// Purge all stored data for a document whose on-disk file has been
+/// deleted (as opposed to content-changed, which is handled by
+/// `purge_orphan_at_workspace_path`).
+///
+/// Returns the `chunk_id`s that were associated with the document so
+/// the caller can issue a matching `VectorStore::delete_by_chunk_ids`
+/// on the LanceDB side.
+///
+/// Deletion order:
+/// 1. Collect chunk_ids (before cascade removes them).
+/// 2. DELETE the `documents` row → CASCADE clears `blocks`, `chunks`,
+///    `embedding_records`.
+/// 3. DELETE the `assets` row **only if no other document still
+///    references it** (twin-file protection — `assets` can be shared
+///    across identical-content files via the blake3 PK).
+/// 4. If the asset was `storage_kind = 'copied'`, best-effort delete
+///    the on-disk byte file at `storage_path`.
+///
+/// Returns `Ok(vec![])` when no document exists at `workspace_path`
+/// (idempotent — caller doesn't need to pre-check).
+pub fn purge_deleted_workspace_path(
+    store: &SqliteStore,
+    workspace_path: &kebab_core::WorkspacePath,
+) -> anyhow::Result<Vec<kebab_core::ChunkId>> {
+    let conn = store.lock_conn();
+
+    // Look up the document + its asset_id.
+    let doc_row: Option<(String, String)> = conn
+        .query_row(
+            "SELECT doc_id, asset_id FROM documents WHERE workspace_path = ?",
+            rusqlite::params![workspace_path.0],
+            |r| Ok((r.get(0)?, r.get(1)?)),
+        )
+        .optional()
+        .map_err(StoreError::from)?;
+
+    let Some((doc_id, asset_id)) = doc_row else {
+        return Ok(Vec::new());
+    };
+
+    // 1. Collect chunk_ids before CASCADE removes them.
+    let mut stmt = conn
+        .prepare("SELECT chunk_id FROM chunks WHERE doc_id = ?")
+        .map_err(StoreError::from)?;
+    let rows = stmt
+        .query_map(rusqlite::params![doc_id], |r| r.get::<_, String>(0))
+        .map_err(StoreError::from)?;
+    let chunk_ids: Vec<kebab_core::ChunkId> = rows
+        .map(|r| r.map(kebab_core::ChunkId))
+        .collect::<rusqlite::Result<Vec<_>>>()
+        .map_err(StoreError::from)?;
+    drop(stmt);
+
+    // 2. DELETE the document row (CASCADE clears blocks / chunks /
+    //    embedding_records via the FK constraints in V001).
+    conn.execute(
+        "DELETE FROM documents WHERE doc_id = ?",
+        rusqlite::params![doc_id],
+    )
+    .map_err(StoreError::from)?;
+
+    // 3. Delete the asset row only when no other document still
+    //    references it (twin-file safety: two files with identical
+    //    bytes share a single asset row via the blake3 PK).
+    let remaining_refs: i64 = conn
+        .query_row(
+            "SELECT COUNT(*) FROM documents WHERE asset_id = ?",
+            rusqlite::params![asset_id],
+            |r| r.get(0),
+        )
+        .map_err(StoreError::from)?;
+
+    if remaining_refs == 0 {
+        // 4. Capture storage details before deleting the row.
+        let asset_storage: Option<(String, String)> = conn
+            .query_row(
+                "SELECT storage_kind, storage_path FROM assets WHERE asset_id = ?",
+                rusqlite::params![asset_id],
+                |r| Ok((r.get(0)?, r.get(1)?)),
+            )
+            .optional()
+            .map_err(StoreError::from)?;
+
+        conn.execute(
+            "DELETE FROM assets WHERE asset_id = ?",
+            rusqlite::params![asset_id],
+        )
+        .map_err(StoreError::from)?;
+
+        // 5. Best-effort: remove the on-disk copied asset file.
+        if let Some((storage_kind, storage_path)) = asset_storage {
+            if storage_kind == "copied" {
+                let _ = std::fs::remove_file(&storage_path);
+            }
+        }
+    }
+
+    tracing::debug!(
+        target: "kebab-store-sqlite",
+        workspace_path = %workspace_path.0,
+        doc_id = %doc_id,
+        chunk_count = chunk_ids.len(),
+        "purged deleted-file document from store"
+    );
+
+    Ok(chunk_ids)
+}
+
 /// UPSERT a row into `assets`. Used by both the `put_asset_with_bytes`
 /// path (which has bytes + computed `storage_kind/path`) and the
 /// `DocumentStore::put_asset` path (which only has the `RawAsset` and
 /// reads `storage_kind/path` from `asset.stored`).
+///
+/// **`assets.workspace_path` is "last-registered path" semantics for
+/// twin files** (two source files with identical content share one
+/// `assets` row keyed on `asset_id = blake3(content)`). Each ingest
+/// of either twin overwrites `workspace_path` with whichever path was
+/// seen most recently — this is intentional and correct after PR #146
+/// made `try_skip_unchanged` document-centric (uses
+/// `get_document_by_workspace_path`, not `get_asset_by_workspace_path`)
+/// and PR #149 made `reset --orphans-only` document-centric too.
+/// Do NOT "fix" the flip-flop by adding a UNIQUE constraint on
+/// `workspace_path` in the `assets` table — twin de-dup is load-bearing.
+/// When you need media_type for a known document, use the 2-step lookup
+/// `get_document_by_workspace_path` → `doc.source_asset_id` →
+/// `get_asset(asset_id)` so the result is twin-safe.
 pub(crate) fn upsert_asset_row(
    conn: &Connection,
    asset: &kebab_core::RawAsset,
@@ -701,6 +823,39 @@ impl SqliteStore {
        }
        Ok(out)
    }
+
+    /// p10-1A-2 follow-up (dogfooding 2026-05-20): per-repo doc count for
+    /// `schema.v1`.
+    ///
+    /// Reads `metadata_json->'$.repo'`, groups by the value, and skips rows
+    /// where `repo` is NULL (documents without an explicit repo tag).
+    /// Returns `BTreeMap<String, u32>` — key is the repo name as stored in
+    /// frontmatter, value is the doc count.
+    pub fn repo_breakdown(
+        &self,
+    ) -> anyhow::Result<std::collections::BTreeMap<String, u32>> {
+        use anyhow::Context;
+        let conn = self.read_conn();
+        let mut stmt = conn
+            .prepare(
+                "SELECT json_extract(metadata_json, '$.repo') AS rp, COUNT(*) \
+                 FROM documents \
+                 WHERE rp IS NOT NULL \
+                 GROUP BY rp",
+            )
+            .context("prepare repo_breakdown")?;
+        let rows = stmt
+            .query_map([], |r| {
+                Ok((r.get::<_, String>(0)?, r.get::<_, i64>(1)? as u32))
+            })
+            .context("query repo_breakdown")?;
+        let mut out = std::collections::BTreeMap::new();
+        for row in rows {
+            let (k, v) = row.context("read repo_breakdown row")?;
+            out.insert(k, v);
+        }
+        Ok(out)
+    }
 }

 /// Apply the design §5 / task-spec pragmas. Called once per connection.
@@ -817,5 +972,79 @@ mod tests {
        // only one key total
        assert_eq!(bd.len(), 1, "expected exactly 1 entry, got: {bd:?}");
    }
+
+    /// p10-1A-2 follow-up: `repo_breakdown` counts docs by
+    /// `metadata_json.repo`.
+    ///
+    /// Inserts:
+    /// - one doc with `repo = "my-repo"` → must appear with count 1
+    /// - one doc with `repo = null`       → must NOT appear (NULL skipped)
+    ///
+    /// Uses a side rusqlite connection that bypasses the `assets` FK via
+    /// `PRAGMA foreign_keys = OFF` so the test is self-contained.
+    #[test]
+    fn repo_breakdown_counts_by_repo() {
+        let (dir, store) = open_fresh_store();
+
+        let db_path = dir.path().join("kebab.sqlite");
+        let conn = rusqlite::Connection::open(&db_path).unwrap();
+        conn.pragma_update(None, "foreign_keys", "OFF").unwrap();
+
+        // Doc 1: doc with repo = "my-repo"
+        conn.execute(
+            "INSERT INTO documents (
+                doc_id, asset_id, workspace_path,
+                source_type, trust_level, parser_version,
+                doc_version, schema_version,
+                metadata_json, provenance_json,
+                created_at, updated_at
+            ) VALUES (
+                'doc-repo-1', 'asset-r1', 'my-repo/README.md',
+                'markdown', 'primary', 'test-v1',
+                1, 1,
+                '{\"repo\":\"my-repo\"}', '{}',
+                '2024-01-01T00:00:00Z', '2024-01-01T00:00:00Z'
+            )",
+            [],
+        )
+        .unwrap();
+
+        // Doc 2: doc with repo absent (null in JSON)
+        conn.execute(
+            "INSERT INTO documents (
+                doc_id, asset_id, workspace_path,
+                source_type, trust_level, parser_version,
+                doc_version, schema_version,
+                metadata_json, provenance_json,
+                created_at, updated_at
+            ) VALUES (
+                'doc-norepo-1', 'asset-r2', 'standalone/notes.md',
+                'markdown', 'primary', 'test-v1',
+                1, 1,
+                '{\"repo\":null}', '{}',
+                '2024-01-01T00:00:00Z', '2024-01-01T00:00:00Z'
+            )",
+            [],
+        )
+        .unwrap();
+
+        drop(conn); // release side connection before querying via store
+
+        let bd = store.repo_breakdown().unwrap();
+
+        // "my-repo" must appear with count 1
+        assert_eq!(
+            bd.get("my-repo"),
+            Some(&1u32),
+            "expected my-repo=1 in repo_breakdown, got: {bd:?}"
+        );
+        // null repo must NOT appear as any key
+        assert!(
+            !bd.contains_key("null"),
+            "null repo must not appear in breakdown, got: {bd:?}"
+        );
+        // only one key total
+        assert_eq!(bd.len(), 1, "expected exactly 1 entry, got: {bd:?}");
+    }
 }

--- a/crates/kebab-store-sqlite/tests/ingest_report_snapshot.rs
+++ b/crates/kebab-store-sqlite/tests/ingest_report_snapshot.rs
@@ -41,6 +41,7 @@ fn fixture_report() -> IngestReport {
        skipped_generated: 0,
        skipped_size_exceeded: 0,
        skip_examples: kebab_core::SkipExamples::default(),
+        purged_deleted_files: 0,
        items: Some(vec![
            IngestItem {
                kind: IngestItemKind::New,
--- a/docs/ARCHITECTURE.md
+++ b/docs/ARCHITECTURE.md
@@ -22,7 +22,7 @@ Cargo workspace, 함수 호출 기반 모듈러 모놀리스. UI binary (`kebab-
 | OCR | Ollama vision LM (default `gemma4:e4b`) — `OcrEngine` trait 으로 Tesseract / Apple Vision 등 future swap (HOTFIXES P6-2) |
 | Image caption | Ollama vision LM, runtime gate `image.caption.enabled` (default OFF) |
 | PDF parser | `lopdf` per-page 텍스트, `chunker_version = "pdf-page-v1"` 가 PDF 자산에 하드코딩 (HOTFIXES P7-3) |
-| code parser | `tree-sitter` + `tree-sitter-rust` / `tree-sitter-python` / `tree-sitter-typescript` / `tree-sitter-javascript` — **parser-side** (`kebab-parse-code`), chunker-side 아님 (design §6.3). chunker versions: Rust = `code-rust-ast-v1`, Python = `code-python-ast-v1`, TypeScript = `code-ts-ast-v1`, JavaScript = `code-js-ast-v1`. `ast_chunk_max_lines = 200` 상수 고정 (HOTFIXES 2026-05-19 — Chunker trait 이 per-medium config 미노출). |
+| code parser | `tree-sitter` + `tree-sitter-rust` / `tree-sitter-python` / `tree-sitter-typescript` / `tree-sitter-javascript` / `tree-sitter-go` — **parser-side** (`kebab-parse-code`), chunker-side 아님 (design §6.3). chunker versions: Rust = `code-rust-ast-v1`, Python = `code-python-ast-v1`, TypeScript = `code-ts-ast-v1`, JavaScript = `code-js-ast-v1`, Go = `code-go-ast-v1`. `ast_chunk_max_lines = 200` 상수 고정 (HOTFIXES 2026-05-19 — Chunker trait 이 per-medium config 미노출). |
 | 1B symbol path | workspace path → module path: Python = dotted prefix (`kebab_eval.metrics.compute_mrr`), TypeScript/JavaScript = slash-style prefix (`src/Foo.Foo.search`). Rust 1A-2 는 file-scope nesting 만 (workspace prefix 없음, 비일관 수용 — HOTFIXES 2026-05-20). |
 | TUI | Ratatui + crossterm — P9-1 Library 패널, P9-2/3/4 진행 예정 |
 | Desktop | Tauri 2 + `pdfjs-dist` (native PDF render backend 금지) — P9-5 |
@@ -52,7 +52,7 @@ flowchart TB
        ppdf["kebab-parse-pdf"]
        pimg["kebab-parse-image"]
        paud["kebab-parse-audio<br/>(P8 보류)"]
-        pcode["kebab-parse-code<br/>(P10-1A-2 + P10-1B)"]
+        pcode["kebab-parse-code<br/>(P10-1A-2 + P10-1B + P10-1C-Go)"]
        ptypes["kebab-parse-types"]
        norm["kebab-normalize"]
        chunk["kebab-chunk"]
@@ -127,7 +127,7 @@ flowchart TB

 UI → store/llm/parse 직접 의존 금지. 모든 user-facing 진입은 `kebab-app` facade 만 통한다 (frozen 설계 §8). `kebab-cli` 가 `--config <path>` flag 를 honor 하려면 `kebab_app::*_with_config(cfg, …)` companion 을 통해 Config 을 명시적으로 thread 하는 패턴 — 자세한 이유는 [tasks/HOTFIXES.md](../tasks/HOTFIXES.md) 의 `--config` 항목.

-`kebab-parse-code` 의 외부 tree-sitter grammar crate 의존: P10-1A-2 에서 `tree-sitter-rust` 추가, P10-1B 에서 `tree-sitter-python` / `tree-sitter-typescript` / `tree-sitter-javascript` 추가. 모두 `kebab-parse-code` 에만 격리 (facade 룰 — UI crate / chunker 가 직접 import 금지).
+`kebab-parse-code` 의 외부 tree-sitter grammar crate 의존: P10-1A-2 에서 `tree-sitter-rust` 추가, P10-1B 에서 `tree-sitter-python` / `tree-sitter-typescript` / `tree-sitter-javascript` 추가, P10-1C-Go 에서 `tree-sitter-go` 추가. 모두 `kebab-parse-code` 에만 격리 (facade 룰 — UI crate / chunker 가 직접 import 금지).

 ## 디렉토리 구조

@@ -165,7 +165,7 @@ kebab/
 │   ├── kebab-source-fs/                               # 워크스페이스 walk + checksum (P1-1)
 │   ├── kebab-parse-md/                                # Markdown frontmatter + blocks (P1-2/3)
 │   ├── kebab-normalize/                               # ParsedBlock → CanonicalDocument (P1-4)
-│   ├── kebab-chunk/                                   # heading-aware + pdf-page-v1 + code-rust-ast-v1 + code-python-ast-v1 + code-ts-ast-v1 + code-js-ast-v1 chunker (P1-5, P7-2, P10-1A-2, P10-1B)
+│   ├── kebab-chunk/                                   # heading-aware + pdf-page-v1 + code-rust-ast-v1 + code-python-ast-v1 + code-ts-ast-v1 + code-js-ast-v1 + code-go-ast-v1 chunker (P1-5, P7-2, P10-1A-2, P10-1B, P10-1C-Go)
 │   ├── kebab-store-sqlite/                            # SQLite + FTS5 (V001/V002/V003) (P1-6, P2-1, P3-3)
 │   ├── kebab-search/                                  # Lexical + Vector + Hybrid retriever (P2-2, P3-4)
 │   ├── kebab-embed/  kebab-embed-local/                  # Embedder trait + fastembed adapter (P3-1, P3-2)
@@ -175,7 +175,7 @@ kebab/
 │   ├── kebab-eval/                                    # golden query runner + metrics (P5-1, P5-2)
 │   ├── kebab-parse-image/                             # ImageExtractor + Ollama OCR + caption (P6)
 │   ├── kebab-parse-pdf/                               # lopdf per-page text extractor (P7-1)
-│   ├── kebab-parse-code/                              # tree-sitter AST extractors: Rust (P10-1A-2), Python + TypeScript + JavaScript (P10-1B); chunker lives in kebab-chunk
+│   ├── kebab-parse-code/                              # tree-sitter AST extractors: Rust (P10-1A-2), Python + TypeScript + JavaScript (P10-1B), Go (P10-1C-Go); chunker lives in kebab-chunk
 │   ├── kebab-app/                                     # facade (P0 시그니처 + P3-5/P6-4/P7-3 본체)
 │   ├── kebab-tui/                                     # Ratatui shell + Library 패널 (P9-1)
 │   ├── kebab-mcp/                                     # stdio MCP server — tools: schema, doctor, search, ask (P9-FB-30)
--- a/docs/SMOKE.md
+++ b/docs/SMOKE.md
@@ -401,6 +401,27 @@ KB --json schema | jq '.stats.code_lang_breakdown'
 - `const foo = () => {...}` 같은 expression-level 함수는 `<top-level>` glue 로 잡힘 (declaration-level 단위만 1B 1차 범위). 자세한 내용: `tasks/HOTFIXES.md` (2026-05-20).
 - `.gitignore` honor — `node_modules/` / `__pycache__/` / `.venv/` 등 built-in 안전망 자동 skip.

+## P10-1C-Go Go 코드 색인
+
+P10-1B 와 동일한 격리 KB 설정. `.go` 파일을 워크스페이스에 두고 ingest 하면 `code-go-ast-v1` chunker 가 package 단위 AST 로 처리한다.
+
+```bash
+cat > /tmp/kebab-smoke/workspace/sample_code/hello.go <<'EOF'
+package main
+
+import "fmt"
+
+func Hello(name string) string {
+    return fmt.Sprintf("Hello, %s!", name)
+}
+EOF
+
+KB ingest
+KB search --mode hybrid "Hello" --code-lang go --json | \
+  jq '{hits: [.hits[] | {symbol: .citation.symbol, lang: .citation.lang}]}'
+# 기대: symbol = "main.Hello", lang = "go"
+```
+
 ## 검증 체크리스트

 - `kebab doctor` 가 `--config` path 를 honor 하고 그 안의 `storage.data_dir` 를 출력 (XDG default 가 아님).
@@ -433,6 +454,7 @@ rm -rf /tmp/kebab-smoke              # 통째로 정리
 - (P7-3) 한 PDF 가 N 페이지면 `kebab ingest` 가 N 개 (또는 그 이상의, 페이지 길면 multi-chunk) 의 chunk 를 한 transaction 안에서 commit. 500 페이지 책 → 500+ chunk 한 번에 → embedding throughput 가 bottleneck. 임베딩 활성 워크스페이스에서 큰 PDF 를 처음 ingest 하면 분-단위 시간 + WAL 크기 증가 가능 — P+ 스케일 hardening task 까지 정상 동작이지만 비용은 측정 가능.
 - (P10-1A-2) `.rs` 파일을 워크스페이스에 두면 `kebab ingest` 결과에 `new` 카운터에 포함. `kebab search --mode hybrid "<함수명>" --code-lang rust --json` 가 `citation.kind = "code"`, `citation.lang = "rust"` (SearchHit top-level `code_lang` 도 동일), `citation.symbol` (함수/타입 이름), `citation.line_start` / `citation.line_end` 를 반환하면 wiring 정상. `kebab schema --json | jq .stats.code_lang_breakdown` 에 `"rust": N` 이 나오면 chunk 가 색인됨.
 - (P10-1B) `.py` / `.ts` / `.tsx` / `.js` / `.mjs` / `.cjs` / `.jsx` 파일을 워크스페이스에 두면 `kebab ingest` 결과에 `new` 카운터에 포함. `--code-lang python` / `--code-lang typescript` / `--code-lang javascript` 검색이 `citation.symbol` 에 module path prefix 를 포함한 결과를 반환하면 wiring 정상. `kebab schema --json | jq .stats.code_lang_breakdown` 에 해당 언어 카운트 등장 확인.
+- (P10-1C-Go) `.go` 파일을 워크스페이스에 두면 `kebab ingest` 가 `code-go-ast-v1` 로 처리. `--code-lang go` 검색이 `citation.symbol` 에 `<package>.<Func>` / `<package>.(*Receiver).<Method>` 형식 결과를 반환하면 wiring 정상. `kebab schema --json | jq .stats.code_lang_breakdown` 에 `"go": N` 등장 확인.
 - (P7-3 + follow-up) 동일 path 에 byte 가 다른 PDF 를 두 번째 ingest 하면 `purge_vector_orphans_for_workspace_path` 가 옛 chunk_id 를 LanceDB 에서 먼저 삭제, 이어서 `purge_orphan_at_workspace_path` 가 옛 doc / chunks / embedding_records 를 SQLite 에서 sweep. 새 byte 가 새 `doc_id` 로 색인됨. `IngestReport` 에 그 자산만 `new+=1` (다른 자산은 `updated`). 두 store 모두 정합 — 옛 본문 검색 시 옛 chunks 가 더 이상 surface 되지 않음.

 ### Embedding upgrade (fb-39b)
--- a/docs/superpowers/plans/2026-05-20-p10-1c-go-ast-chunker.md
+++ b/docs/superpowers/plans/2026-05-20-p10-1c-go-ast-chunker.md
@@ -0,0 +1,540 @@
+# p10-1C-Go Implementation Plan
+
+> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development to implement this plan task-by-task.
+
+**Goal:** Activate Go code ingest end-to-end on top of 1A-2 (Rust) + 1B (Python/TS/JS) infrastructure. Add `tree-sitter-go` grammar + `GoAstExtractor` + `code-go-ast-v1` chunker + media routing + app dispatch arm.
+
+**Architecture:** Mirror 1A-2 / 1B exactly. `kebab-parse-code/src/go.rs` walks tree-sitter-go parse tree; emits one `Block::Code` per top-level AST semantic unit with `SourceSpan::Code { symbol, lang: Some("go") }`. Symbol prefix = **source-extracted package name** (from `package_clause` AST node — design §3.4 Go row). `kebab-chunk/src/code_go_ast_v1.rs` is a near-duplicate of `code-rust-ast-v1`. App dispatch's `ingest_one_code_asset` (PR #142 generalized 4-arm match) gets a 5th arm.
+
+**Tech Stack:** Rust 2024 workspace, `tree-sitter` 0.26 (already in workspace), `tree-sitter-go` (NEW), 1A-2/1B infrastructure unchanged.
+
+**Memory note:** Host has been OOM-killed previously. Use `cargo test -p <crate>` and `cargo check -p <crate>` only. ONE full-suite invocation reserved for Task G gate.
+
+---
+
+## Pre-flight
+
+Branch `feat/p10-1c-go` already exists.
+
+- [ ] **Disk hygiene**: `cargo clean` if previous artifacts are bloated. Skip if disk is comfortable (`df -h /`).
+
+Reference files:
+- 1A-2 Rust extractor: `crates/kebab-parse-code/src/rust.rs` — closest single-language scaffold template.
+- 1B Python extractor (closest analog for "class-nesting recursion" — Go doesn't have classes but has package as the single prefix): `crates/kebab-parse-code/src/python.rs`.
+- 1A-2 chunker scaffold: `crates/kebab-chunk/src/code_rust_ast_v1.rs`.
+- 1B dispatch generalization: `crates/kebab-app/src/lib.rs::ingest_one_code_asset` (~L1645, 4-arm match).
+- 1A-2 source-fs routing: `crates/kebab-source-fs/src/media.rs` `"rs" =>` arm.
+
+---
+
+## Task A: Workspace dep `tree-sitter-go`
+
+**Files:**
+- Modify: `Cargo.toml` (workspace `[workspace.dependencies]`, after `tree-sitter-javascript` line)
+- Modify: `crates/kebab-parse-code/Cargo.toml`
+
+- [ ] **Step 1**: `cargo add tree-sitter-go -p kebab-parse-code` to resolve version.
+
+- [ ] **Step 2**: Lift the resolved version into `[workspace.dependencies]` after `tree-sitter-javascript`:
+
+```toml
+# Go grammar for code ingest (kebab-parse-code, p10-1C).
+tree-sitter-go         = "<resolved>"
+```
+
+Switch the crate's entry to `{ workspace = true }` matching existing tree-sitter-* style.
+
+- [ ] **Step 3**: `cargo build -p kebab-parse-code` → clean. Unused dep warning is fine.
+
+- [ ] **Step 4**: Commit:
+
+```bash
+git add Cargo.toml Cargo.lock crates/kebab-parse-code/Cargo.toml
+git commit -m "build(p10-1c-go): add tree-sitter-go workspace dep
+
+Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>"
+```
+
+---
+
+## Task B: source-fs media routing `.go` → `MediaType::Code("go")`
+
+**Files:**
+- Modify: `crates/kebab-source-fs/src/media.rs` (add arm after the existing JS arm at ~L44)
+- Test: same file's test module
+
+- [ ] **Step 1 (failing test)** — add to existing tests near `py_ts_js_files_map_to_media_code`:
+
+```rust
+#[test]
+fn go_files_map_to_media_code_go() {
+    assert_eq!(media_type_for(Path::new("a/b.go")), MediaType::Code("go".into()));
+}
+```
+
+- [ ] **Step 2**: Run → FAIL.
+
+- [ ] **Step 3**: Add the arm before the catch-all `_ => MediaType::Other(ext)`:
+
+```rust
+        // p10-1C-Go: Go ingest activated.
+        "go" => MediaType::Code("go".into()),
+```
+
+- [ ] **Step 4**: Run → PASS. `cargo test -p kebab-source-fs` → no regression.
+
+- [ ] **Step 5**: clippy clean, commit:
+
+```bash
+git add crates/kebab-source-fs/
+git commit -m "feat(p10-1c-go): route .go to MediaType::Code(go)
+
+Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>"
+```
+
+---
+
+## Task C: App dispatch allowlist + bail arm for "go"
+
+**Files:**
+- Modify: `crates/kebab-app/src/lib.rs` (dispatch match guard + 4 internal match arms in `ingest_one_code_asset`)
+
+- [ ] **Step 1**: Find the `MediaType::Code(lang) if matches!(lang.as_str(), "rust" | "python" | "typescript" | "javascript")` arm (~L953). Add `"go"` to the allowlist:
+
+```rust
+        MediaType::Code(lang)
+            if matches!(lang.as_str(), "rust" | "python" | "typescript" | "javascript" | "go") =>
+        {
+```
+
+- [ ] **Step 2**: In `ingest_one_code_asset`'s 4 `match code_lang` blocks (parser_version, chunker_version, extract, chunk), add a "go" arm that `bail!()`s for now (extractor + chunker land in Task D/E). Mirror the Python/TS/JS bail-then-activate pattern:
+
+```rust
+let parser_version = match code_lang {
+    // ... existing arms ...
+    "go" => anyhow::bail!("go ingest not yet wired (p10-1c-go Task F)"),
+    other => anyhow::bail!("unsupported code_lang: {other}"),
+};
+// similar for chunker_version / extract / chunk matches
+```
+
+- [ ] **Step 3**: `cargo test -p kebab-app --lib` → existing 52 lib tests stay green. `cargo test -p kebab-app --test code_ingest_smoke` → 6 stay green (Rust path unaffected).
+
+- [ ] **Step 4**: clippy clean, commit:
+
+```bash
+git add crates/kebab-app/
+git commit -m "refactor(p10-1c-go): add go to ingest dispatch allowlist (bail until Task F)
+
+Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>"
+```
+
+---
+
+## Task D: `GoAstExtractor` (`kebab-parse-code/src/go.rs`)
+
+**Files:**
+- Create: `crates/kebab-parse-code/src/go.rs`
+- Modify: `crates/kebab-parse-code/src/lib.rs` (`pub mod go;` + re-exports `GO_PARSER_VERSION`, `GoAstExtractor`)
+- Create: `crates/kebab-parse-code/tests/fixtures/sample.go`
+
+Scaffold mirrors `crates/kebab-parse-code/src/rust.rs` line-for-line for the `CanonicalDocument` skeleton (Extractor trait impl, `id_for_doc`, ProvenanceEvent, final `CanonicalDocument` literal). The novel parts:
+
+### Constants
+
+```rust
+pub const PARSER_VERSION: &str = "code-go-v1";
+
+pub struct GoAstExtractor;
+// new() + Default
+// supports: matches!(m, MediaType::Code(l) if l == "go")
+// agent = "kb-parse-code"
+// metadata.code_lang = Some("go")
+// SourceType::Note (no SourceType::Code variant)
+// repo/git_branch/git_commit via detect_repo
+```
+
+### Package extraction
+
+Unlike 1B's path-based `module_path_for_python` / `_for_tsjs`, the Go package prefix comes from the **source code's `package` declaration** (design §3.4). tree-sitter-go's grammar:
+
+- Root: `source_file`
+- First named child is typically `package_clause` → contains `package_identifier` child whose text is the package name.
+
+Helper (local to `go.rs`):
+
+```rust
+/// Returns the package name from a tree-sitter-go `source_file`, or
+/// `None` if the file has no `package_clause` (invalid Go in practice,
+/// but be defensive).
+fn extract_package(root: tree_sitter::Node, src: &str) -> Option<String> {
+    let mut cur = root.walk();
+    for child in root.named_children(&mut cur) {
+        if child.kind() == "package_clause" {
+            // `package_clause` has a `package_identifier` named child.
+            let mut c2 = child.walk();
+            for sub in child.named_children(&mut c2) {
+                if sub.kind() == "package_identifier" {
+                    return Some(src[sub.start_byte()..sub.end_byte()].to_string());
+                }
+            }
+        }
+    }
+    None
+}
+```
+
+### Semantic-unit rules
+
+| node kind | unit | symbol |
+|-----------|------|--------|
+| `function_declaration` (name field) | 1 | `<pkg>.<fn_name>` |
+| `method_declaration` | 1 | `<pkg>.(<TypeText>).<MethodName>` where `<TypeText>` includes a leading `*` if the receiver is `pointer_type`. Examples: `chunk.(*MdHeadingV1Chunker).ChunkDoc`, `chunk.(Foo).Bar`. |
+| `type_declaration` (struct / interface / type alias) | 1 per inner `type_spec` | `<pkg>.<TypeName>` |
+| `const_declaration`, `var_declaration`, `import_declaration` (single or block) | glue | `<pkg>.<top-level>` (or `<pkg>.<package>` if file has ZERO real units AND glue is import-only — same `<module>` post-pass pattern as 1B Python, renamed to `<package>` to avoid colliding with Go's `package` keyword? — actually use `<module>` per design §3.4 — see "module / namespace 만 있고 symbol 없는 경우" line) |
+
+`unit_start` walks `comment` siblings (same as 1B). Go doesn't have separate attribute / decorator nodes.
+
+Method receiver pointer detection:
+
+```rust
+// In the method_declaration arm:
+let receiver = child.child_by_field_name("receiver");  // parameter_list
+let receiver_type_text = receiver.and_then(|r| {
+    let mut cw = r.walk();
+    for p in r.named_children(&mut cw) {
+        if p.kind() == "parameter_declaration" {
+            // type field is either type_identifier (value) or pointer_type (ptr)
+            if let Some(ty) = p.child_by_field_name("type") {
+                let s = &src[ty.start_byte()..ty.end_byte()];
+                return Some(s.to_string());  // includes leading "*" if pointer_type
+            }
+        }
+    }
+    None
+});
+// Format: "(*Foo)" or "(Foo)" — wrap in parens, preserve leading "*" if any.
+let owner = receiver_type_text
+    .map(|t| format!("({t})"))
+    .unwrap_or_else(|| "()".to_string());
+let method_name = name_text(&child, src);
+// symbol = format!("{pkg}.{owner}.{method_name}")
+```
+
+Read tree-sitter-go's grammar.json or node-types.json (in the registry source) if any field name above differs in the resolved crate version.
+
+### Fixture `tests/fixtures/sample.go`:
+
+```go
+// sample.go
+package chunk
+
+import (
+	"fmt"
+	"strings"
+)
+
+const Version = "v1"
+
+type MdHeadingV1Chunker struct {
+	Name string
+}
+
+// ChunkDoc returns a stub list of strings.
+func (m *MdHeadingV1Chunker) ChunkDoc(input string) []string {
+	return []string{m.Name}
+}
+
+func (m MdHeadingV1Chunker) Name2() string {
+	return m.Name
+}
+
+type Stringer interface {
+	String() string
+}
+
+func Free(x int) int {
+	return x + 1
+}
+
+func init() {
+	fmt.Println(strings.ToUpper("init"))
+}
+```
+
+### Test module
+
+Mirror Python's test shape (use `crate::rust::tests_support::fixed_code_asset` from 1B):
+
+```rust
+#[cfg(test)]
+mod tests {
+    use super::*;
+    use kebab_core::{Block, MediaType, SourceSpan};
+
+    fn extract_fixture() -> kebab_core::CanonicalDocument {
+        let bytes = std::fs::read(
+            concat!(env!("CARGO_MANIFEST_DIR"), "/tests/fixtures/sample.go"),
+        ).unwrap();
+        let asset = crate::rust::tests_support::fixed_code_asset(
+            "crates/x/src/sample.go", "go",
+        );
+        let cfg = kebab_core::ExtractConfig::default();
+        let root = std::path::PathBuf::from("/tmp");
+        let ctx = kebab_core::ExtractContext { asset: &asset, workspace_root: &root, config: &cfg };
+        GoAstExtractor::new().extract(&ctx, &bytes).unwrap()
+    }
+
+    #[test]
+    fn extractor_supports_only_media_code_go() {
+        let e = GoAstExtractor::new();
+        assert!(e.supports(&MediaType::Code("go".into())));
+        assert!(!e.supports(&MediaType::Code("rust".into())));
+        assert!(!e.supports(&MediaType::Markdown));
+    }
+
+    #[test]
+    fn go_units_match_design_3_4_symbols() {
+        let doc = extract_fixture();
+        let mut syms: Vec<String> = doc.blocks.iter().filter_map(|b| match b {
+            Block::Code(c) => match &c.common.source_span {
+                SourceSpan::Code { symbol, lang, .. } => {
+                    assert_eq!(lang.as_deref(), Some("go"));
+                    symbol.clone()
+                }
+                _ => None,
+            },
+            _ => None,
+        }).collect();
+        syms.sort();
+        assert!(syms.iter().any(|s| s == "chunk.Free"), "got {syms:?}");
+        assert!(syms.iter().any(|s| s == "chunk.init"));
+        assert!(syms.iter().any(|s| s == "chunk.MdHeadingV1Chunker"));
+        assert!(syms.iter().any(|s| s == "chunk.(*MdHeadingV1Chunker).ChunkDoc"));
+        assert!(syms.iter().any(|s| s == "chunk.(MdHeadingV1Chunker).Name2"));
+        assert!(syms.iter().any(|s| s == "chunk.Stringer"));
+        assert!(syms.iter().any(|s| s == "chunk.<top-level>"));  // import + const grouped
+    }
+
+    #[test]
+    fn deterministic_across_runs() {
+        let a = extract_fixture();
+        for _ in 0..50 { assert_eq!(extract_fixture().blocks, a.blocks); }
+    }
+}
+```
+
+### Step list
+
+- [ ] Step 1: create fixture + test module.
+- [ ] Step 2: run → FAIL (`GoAstExtractor` undefined).
+- [ ] Step 3: implement `go.rs`. Scaffold mirrors `python.rs` (Extractor impl + extract scaffold + `build_blocks` returning blocks). `build_blocks` does: extract_package → walk root's named children → branch per node kind per the table above → emit `Block::Code` with `SourceSpan::Code { symbol, lang: Some("go") }`. Use the same `flush_glue` / glue grouping / `<top-level>` vs `<module>` post-pass as Python (rename to `<package>` if user prefers, but spec §3.4 says `<module>` so keep that name for cross-language consistency).
+- [ ] Step 4: wire into `lib.rs`:
+
+```rust
+pub mod go;
+pub use go::{PARSER_VERSION as GO_PARSER_VERSION, GoAstExtractor};
+```
+
+- [ ] Step 5: `cargo test -p kebab-parse-code` → all pass (Rust/Python/TS/JS + new Go). `cargo clippy -p kebab-parse-code --all-targets -- -D warnings` clean.
+- [ ] Step 6: commit:
+
+```bash
+git add crates/kebab-parse-code/
+git commit -m "feat(p10-1c-go): tree-sitter-go AST extractor (GoAstExtractor)
+
+Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>"
+```
+
+---
+
+## Task E: `code-go-ast-v1` chunker
+
+**Files:**
+- Create: `crates/kebab-chunk/src/code_go_ast_v1.rs`
+- Modify: `crates/kebab-chunk/src/lib.rs`
+
+Identical pattern to PR #142 Task I (TS) / Task L (JS) — near-duplicate of `code_rust_ast_v1.rs` with substitutions:
+- `const VERSION_LABEL: &str = "code-go-ast-v1";`
+- struct name `CodeGoAstV1Chunker`
+- error message says `"CodeGoAstV1Chunker only handles..."`
+- module doc-comment prose `Rust` → `Go`, `code-rust-ast-v1` → `code-go-ast-v1`
+
+`split_oversize` / `make_chunk` / `AST_CHUNK_MAX_LINES = 200` / `BYTES_PER_TOKEN = 3` / `POLICY_HASH_HEX_LEN = 16` IDENTICAL (language-agnostic).
+
+Test module: copy from `code_ts_ast_v1.rs` and substitute names. KEEP cross-chunker `policy_hash_matches_md_heading_v1`.
+
+Wire into `crates/kebab-chunk/src/lib.rs`:
+
+```rust
+mod code_go_ast_v1;
+pub use code_go_ast_v1::CodeGoAstV1Chunker;
+```
+
+(Alphabetical placement.)
+
+Verify + commit:
+- `cargo test -p kebab-chunk code_go_ast` PASS (~6 tests)
+- `cargo test -p kebab-chunk` full per-crate green
+- `cargo clippy -p kebab-chunk --all-targets -- -D warnings` clean
+
+```bash
+git add crates/kebab-chunk/
+git commit -m "feat(p10-1c-go): code-go-ast-v1 chunker (1:1 + oversize split)
+
+Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>"
+```
+
+---
+
+## Task F: Activate Go in app dispatch
+
+**Files:**
+- Modify: `crates/kebab-app/src/lib.rs` (replace 4 "go" bail! arms with real calls)
+- Modify: `crates/kebab-app/tests/code_ingest_smoke.rs` (add Go integration test)
+
+Replace the 4 `"go" => anyhow::bail!(...)` arms in `ingest_one_code_asset` (added in Task C) with real:
+
+```rust
+"go" => ParserVersion(kebab_parse_code::GO_PARSER_VERSION.to_string()),
+// ...
+"go" => CodeGoAstV1Chunker.chunker_version(),
+// ...
+"go" => kebab_parse_code::GoAstExtractor::new()
+    .extract(&ctx, &bytes)
+    .context("kb-parse-code::GoAstExtractor::extract (code:go)")?,
+// ...
+"go" => CodeGoAstV1Chunker
+    .chunk(&canonical, chunk_policy)
+    .context("kb-chunk::CodeGoAstV1Chunker::chunk (code:go)")?,
+```
+
+Add imports at top of lib.rs:
+- `kebab_chunk::CodeGoAstV1Chunker`
+- `kebab_parse_code::GoAstExtractor`
+
+Integration test (mirror PR #142's `python_file_ingests_and_searches_as_code_citation`):
+
+```rust
+#[test]
+fn go_file_ingests_and_searches_as_code_citation() {
+    // ... TempDir + Config harness same as Python/TS test ...
+    let pkg_dir = env.workspace_root.join("chunk");
+    std::fs::create_dir_all(&pkg_dir).unwrap();
+    std::fs::write(
+        pkg_dir.join("ast.go"),
+        "package chunk\n\nfunc ParseDoc(input string) string {\n    return input\n}\n",
+    ).unwrap();
+
+    let report = kebab_app::ingest_with_config(/* ... */).unwrap();
+    assert!(report.new >= 1);
+    let go_item = report.items.as_ref().unwrap().iter()
+        .find(|i| i.doc_path.0.ends_with("ast.go")).expect("ast.go item");
+    assert_eq!(go_item.parser_version.as_ref().unwrap().0, "code-go-v1");
+    assert_eq!(go_item.chunker_version.as_ref().unwrap().0, "code-go-ast-v1");
+
+    let hits = kebab_app::search_with_config(/* search "ParseDoc" */).unwrap();
+    let h = hits.iter().find(|h| matches!(h.citation, kebab_core::Citation::Code { .. }))
+        .expect("Citation::Code hit");
+    match &h.citation {
+        kebab_core::Citation::Code { lang, symbol, line_start, .. } => {
+            assert_eq!(lang.as_deref(), Some("go"));
+            assert_eq!(symbol.as_deref(), Some("chunk.ParseDoc"));
+            assert!(*line_start >= 1);
+        }
+        _ => unreachable!(),
+    }
+    assert_eq!(h.code_lang.as_deref(), Some("go"));
+}
+```
+
+Verify:
+- `cargo test -p kebab-app --test code_ingest_smoke` → 7/7 (6 existing + 1 new go)
+- `cargo test -p kebab-app --lib` → 52/52 (no regression)
+- clippy clean
+
+```bash
+git add crates/kebab-app/
+git commit -m "feat(p10-1c-go): activate Go in ingest_one_code_asset dispatch
+
+Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>"
+```
+
+---
+
+## Task G: Snapshot + full-suite gate + manual SMOKE
+
+**Files:**
+- Create: `crates/kebab-chunk/tests/code_go_ast_snapshot.rs` + fixture + baseline (mirror `code_python_ast_snapshot.rs` from PR #142)
+
+- [ ] **Step 1**: Add snapshot integration test. In-memory `CanonicalDocument` (no kebab-parse-code dep — boundary §6.3). Generate baseline: `UPDATE_SNAPSHOTS=1 cargo test -p kebab-chunk code_go_ast_snapshot` → re-run without env → PASS.
+
+- [ ] **Step 2**: Full-suite gate (the ONE invocation allowed this PR):
+
+```bash
+cargo clippy --workspace --all-targets -- -D warnings
+cargo test --workspace --no-fail-fast -j 1
+```
+
+Both must be CLEAN/GREEN.
+
+- [ ] **Step 3**: Manual SMOKE (optional but recommended — mirror PR #142 SMOKE):
+
+```bash
+cargo build --release   # OR debug if RAM-tight
+rm -rf /tmp/kebab-go-smoke && mkdir -p /tmp/kebab-go-smoke/ws/chunk
+echo 'package chunk
+
+func ParseDoc(input string) string { return input }
+' > /tmp/kebab-go-smoke/ws/chunk/ast.go
+# adapt isolated config from docs/SMOKE.md
+./target/release/kebab --config /tmp/kebab-go-smoke/config.toml ingest --json | jq '.items[].parser_version' | sort -u
+./target/release/kebab --config /tmp/kebab-go-smoke/config.toml search "ParseDoc" --code-lang go --json | jq '.hits[0]'
+```
+
+Expected: `code-go-v1` in parser_versions; Citation::Code with symbol `chunk.ParseDoc`.
+
+- [ ] **Step 4**: Commit snapshot only (full-suite + SMOKE are gates, not commit content):
+
+```bash
+git add crates/kebab-chunk/tests/
+git commit -m "test(p10-1c-go): code-go-ast-v1 chunker snapshot + full-suite gate
+
+Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>"
+```
+
+---
+
+## Task H: Docs + version bump
+
+- README: 지원 형식 row — add Go (`.go`, `code-go-ast-v1`).
+- HANDOFF: P10 phase row note 1C-Go merged (Go active). Java/Kotlin remain pending.
+- ARCHITECTURE: directory tree note for kebab-parse-code includes `go.rs` (Java/Kotlin coming in next PR). Decisions table — no new row (1C-Go follows the 1A-2/1B convention).
+- SMOKE: extend the P10 section with a 1-line note for Go (or compact Go example).
+- tasks/INDEX + tasks/p10/INDEX: flip the row for 1C-Go to 🟡 (PR open) → ✅ on merge. The 1C row in p10/INDEX may need a split — `p10-1C-Go ⏳ → 🟡` and `p10-1C-JavaKotlin ⏳ unchanged` (since user split into 2 PRs).
+- frozen design §10.1: add a one-liner — "p10-1C-Go 활성화 (Go)" (Java/Kotlin will get its own line in the next PR).
+- `Cargo.toml`: workspace version `0.11.1 → 0.12.0` (minor — dogfooding surface 확장, 새 chunker + extractor 활성화).
+
+```bash
+git add -A
+git commit -m "docs(p10-1c-go): README/HANDOFF/ARCHITECTURE/SMOKE/INDEX + chore: bump version 0.11.1 → 0.12.0
+
+Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>"
+```
+
+---
+
+## Finalize: PR + review loop + release
+
+Per workflow memory (gitea-pr + review loop, no single-shot):
+
+- [ ] `gitea-pr` → PR title `feat(p10-1C-Go): tree-sitter-go AST extractor + chunker — Go 코드 색인 활성화`
+- [ ] Review loop until APPROVE → merge → main pull → branch cleanup → `cargo clean` → `gitea-release v0.12.0`.
+
+---
+
+## Self-Review (filled by plan author)
+
+- **Spec coverage**: design §1C Go (extractor + chunker + activation) → Tasks D/E/F; §3.3 (`code-go-ast-v1`) → Task E; §3.4 symbol path → Task D (extract_package + method receiver pointer detection); §6.1 (`kebab-parse-code/src/go.rs`) → Task D; §6.2 (`kebab-chunk/src/code_go_ast_v1.rs`) → Task E; §6.3 dep graph (`tree-sitter-go` parser-side) → Task A; §9.1 Tier-1 + oversize fallback → Task E (1A-2 split_oversize reused identically).
+- **No placeholders**: novel logic (`extract_package`, method receiver pointer detection, fixture, test assertions, dispatch arm additions) given concretely. Mechanical mirrors (chunker, integration test, snapshot test) pinned to exact existing files with substitutions.
+- **Type consistency**: `GoAstExtractor` / `GO_PARSER_VERSION = "code-go-v1"` / `CodeGoAstV1Chunker` / `VERSION_LABEL = "code-go-ast-v1"` used consistently across Tasks A-H. `MediaType::Code("go")` in routing + dispatch. `Citation::Code` with `lang: Some("go")` in integration test.
--- a/docs/superpowers/specs/2026-04-27-kebab-final-form-design.md
+++ b/docs/superpowers/specs/2026-04-27-kebab-final-form-design.md
@@ -1545,6 +1545,8 @@ transitional 형태) 의 source of truth.

 **p10-1B 활성화 (Python / TypeScript / JavaScript) (2026-05-20)**: Python (`code-python-ast-v1`, `.py`), TypeScript (`code-ts-ast-v1`, `.ts`/`.tsx`), JavaScript (`code-js-ast-v1`, `.js`/`.mjs`/`.cjs`/`.jsx`) AST chunker 활성화. symbol path 는 workspace 경로 → module path prefix: Python = dotted (예: `kebab_eval.metrics.compute_mrr`), TypeScript/JavaScript = slash-style (예: `src/Foo.Foo.search`). Rust 1A-2 의 file-scope-only symbol 과 비일관 수용 (HOTFIXES 2026-05-20). expression-level 함수 (`const foo = () => {}`) 는 glue 처리 (HOTFIXES 2026-05-20).

+**p10-1C-Go 활성화 (Go) (2026-05-20)**: Go (`code-go-ast-v1`, `.go`) AST chunker 활성화. symbol = `<package>.<Func>` / `<package>.(*Receiver).<Method>` 형식. Java / Kotlin 은 후속 PR (p10-1C-JavaKotlin) 에서 별도 활성화.
+
 ### 10.2 MCP server transport (fb-30)

 `kebab mcp` 가 stdio JSON-RPC server. Rust SDK = `rmcp 1.6`. Tool surface
--- a/docs/wire-schema/v1/reset_report.schema.json
+++ b/docs/wire-schema/v1/reset_report.schema.json
@@ -14,12 +14,18 @@
    "schema_version":           { "const": "reset_report.v1" },
    "scope":                    {
      "type": "string",
-      "enum": ["all", "data_only", "vector_only", "config_only"]
+      "enum": ["all", "data_only", "vector_only", "config_only", "orphans_only"]
    },
    "removed_paths":            {
      "type": "array",
      "items": { "type": "string" }
    },
-    "embedding_rows_truncated": { "type": "integer", "minimum": 0 }
+    "embedding_rows_truncated": { "type": "integer", "minimum": 0 },
+    "orphans_purged":           { "type": "integer", "minimum": 0, "default": 0 },
+    "purged_paths":             {
+      "type": "array",
+      "items": { "type": "string" },
+      "default": []
+    }
  }
 }
--- a/tasks/INDEX.md
+++ b/tasks/INDEX.md
@@ -142,7 +142,8 @@ P0~P5 는 직렬. P6~P9 는 P5 이후 병렬 가능.
  - [p10-1A-1 code ingest framework](p10/p10-1a-1-code-ingest-framework.md) — ✅ 머지
  - [p10-1A-2 Rust AST chunker](p10/p10-1a-2-rust-ast-chunker.md) — ✅ 머지
  - [p10-1B Python + TS/JS AST chunkers](p10/p10-1b-py-ts-js-ast-chunkers.md) — 🟡 PR 오픈 (코드 완성, 머지 대기)
-  - p10-1C Go + Java + Kotlin AST chunkers — ⏳
+  - p10-1C-Go Go AST chunker — 🟡 PR 오픈 (v0.12.0, `code-go-ast-v1`)
+  - p10-1C-JavaKotlin Java + Kotlin AST chunkers — ⏳
  - p10-1D C + C++ AST chunkers — ⏳
  - p10-2 Tier 2 resource-aware — ⏳
  - p10-3 Tier 3 paragraph + line-window fallback — ⏳
--- a/tasks/p10/INDEX.md
+++ b/tasks/p10/INDEX.md
@@ -5,7 +5,8 @@
 | 1A-1 | code ingest framework (wire schema, parse-code crate skeleton, filter flags, skip policy, config 절) | ✅ 머지 |
 | 1A-2 | Rust AST chunker | ✅ 머지 |
 | 1B | Python + TS/JS AST chunkers | 🟡 PR 오픈 (코드 완성, 머지 대기) |
-| 1C | Go + Java + Kotlin AST chunkers | ⏳ |
+| 1C-Go | Go AST chunker (`code-go-ast-v1`) | 🟡 PR 오픈 (v0.12.0) |
+| 1C-JavaKotlin | Java + Kotlin AST chunkers | ⏳ |
 | 1D | C + C++ AST chunkers | ⏳ |
 | 2 | Tier 2 resource-aware (k8s / Dockerfile / manifest) | ⏳ |
 | 3 | Tier 3 paragraph + line-window fallback | ⏳ |
--- a/tasks/p10/p10-1c-go-ast-chunker.md
+++ b/tasks/p10/p10-1c-go-ast-chunker.md
@@ -0,0 +1,54 @@
+# p10-1C-Go — Go AST chunker
+
+**Status:** 🟡 진행 중
+**Contract sections:** §3.3 (chunker_version `code-go-ast-v1`), §3.4 (symbol path — Go `package.Func` / `package.(*Receiver).Method`), §3.5 (code_lang `go`, ext `.go`), §6.1 (`kebab-parse-code/src/go.rs`), §6.2 (`kebab-chunk/src/code_go_ast_v1.rs`), §9.1 (Tier 1 AST per-language + oversize fallback).
+**Design:** [2026-05-15-kebab-code-ingest-design.md](../../docs/superpowers/specs/2026-05-15-kebab-code-ingest-design.md) §1C (Go 부분 — Java + Kotlin 은 후속 PR).
+**Plan:** [2026-05-20-p10-1c-go-ast-chunker.md](../../docs/superpowers/plans/2026-05-20-p10-1c-go-ast-chunker.md).
+
+## Goal
+
+1A-2 / 1B 인프라 위에 Go AST chunker 활성화. 사용자 결정으로 1C 의 3 언어 (Go + Java + Kotlin) 를 2 PR 로 분할 — Go 가 method receiver / package convention 면에서 Java/Kotlin (JVM family) 과 다르므로 별 PR. 본 PR 머지 시점부터 Go 프로젝트 dogfooding 가능.
+
+## 동결된 설계 결정 (이 task 로 확정)
+
+- **Symbol path 의 package prefix = 소스 코드의 `package` 선언에서 추출** (design §3.4 그대로). 1B 의 workspace-path 변환과 다름 — Go 는 언어 자체에 `package` declaration 이 있어 그게 canonical source. tree-sitter-go 의 `source_file` root 의 첫 named child `package_clause` 에서 추출. 빈 경우 (이론상 invalid Go, 실용엔 거의 없음) `<unknown>` 또는 fallback `<package>` (1A `<module>` 패턴과 유사).
+- **Method receiver 표현** (design 예시 그대로): `package.(*Receiver).Method` (포인터 receiver), `package.(Receiver).Method` (value receiver). tree-sitter-go 의 `method_declaration` 의 `receiver` field 에서 type + pointer 여부 추출. 예: `func (m *MdHeadingV1Chunker) ChunkDoc(...)` → symbol `chunk.(*MdHeadingV1Chunker).ChunkDoc`.
+- **Top-level unit 종류**:
+  - `function_declaration` → 1 unit, symbol `package.Func`
+  - `method_declaration` → 1 unit, symbol `package.(*Receiver).Method` / `package.(Receiver).Method`
+  - `type_declaration` (struct / interface / type alias) → 1 unit each, symbol `package.TypeName`
+  - `const_declaration`, `var_declaration`, `import_declaration` (블록 또는 단일) → glue, grouped → `package.<top-level>` (1A/1B 패턴)
+- **Go 의 generic 처리**: `func Foo[T any](...)` 또는 `type Foo[T any] struct{}` 의 type parameter 는 symbol 에 미포함 (Go 자체도 보통 symbol 에 안 적음). 단순 `package.Foo` 만.
+- **Test detection**: Go 의 `func TestXxx(t *testing.T)` 는 *일반 fn 으로 emit*. test 감지 boost/penalty 등 ranking 영향은 본 task 범위 밖 (ranking brainstorm 보류 메모리 따름).
+- frozen design 자체는 변경 없음 (§3.4 의 Go 행이 이미 본 결정과 일치). §10.1 에 1C-Go 활성화 한 줄 추가.
+
+## Acceptance criteria
+
+- `cargo test --workspace --no-fail-fast -j 1` passes (memory-conscious: per-crate 위주, full-suite gate 는 docs task 직전 1회).
+- `cargo clippy --workspace --all-targets -- -D warnings` passes.
+- Go fixture (`tests/fixtures/sample.go`) ingest → chunk snapshot 안정 + `Citation::Code` 의 symbol 이 §3.4 컨벤션 일치 (`pkg.Func` / `pkg.(*Receiver).Method`).
+- 격리 TempDir KB 에 Go 파일 두고 `kebab search --code-lang go --json` 가 `Citation::Code { lang: "go", symbol: "...", ... }` 반환.
+- `kebab schema --json | jq .stats.code_lang_breakdown` 에 `"go"` 카운트.
+- README + HANDOFF + ARCHITECTURE + SMOKE + tasks/INDEX + tasks/p10/INDEX 갱신.
+- frozen design §10.1 한 줄 추가.
+- workspace `Cargo.toml` minor bump (0.11.1 → 0.12.0).
+
+## Allowed dependencies
+
+- `kebab-parse-code` 에 `tree-sitter-go` 추가 (workspace deps). 기존 deps 유지.
+- `kebab-chunk` 의 새 모듈 `code_go_ast_v1.rs` — kebab-core + serde_json_canonicalizer + blake3 + anyhow + tracing. tree-sitter 절대 import 금지.
+- `kebab-app`, `kebab-source-fs` 변경 — 새 crate dep 없음.
+
+## Forbidden dependencies
+
+- `kebab-chunk` 가 `tree-sitter-go` 직접 import 금지.
+- UI crate 가 `kebab-parse-code` 직접 import 금지.
+- `kebab-parse-code` 가 store / embed / llm / rag 직접 import 금지.
+
+## Risks / notes
+
+- tree-sitter-go 의 `package_clause` node 가 root 의 첫 named child 인지 grammar 버전에 따라 다를 수 있음 — extractor 가 `source_file` 전체를 named_children iterate 하면서 첫 `package_clause` 잡는 방식이 안전.
+- `method_declaration` 의 receiver pointer 여부: tree-sitter-go AST 에서 receiver type 이 `pointer_type` 노드면 `*Receiver`, 그냥 `type_identifier` 면 `Receiver`. 정확한 텍스트 추출 필요.
+- Generic type parameter (`[T any]`) 가 method_declaration / function_declaration 의 name field 와 별도 child — name 만 추출하면 generic 부분 자동 제외.
+- 1B Python/TS/JS 패턴 (helpers from lang.rs) 와 *다른* 모델 — 본 task 의 mod_prefix 는 source-side AST 에서 추출, helper fn 불필요.
+- 머지 후 deviation 은 `tasks/HOTFIXES.md` 에 dated 로그 + 본 spec `Risks / notes` 에 one-line cross-link.
Author	SHA1	Message	Date
altair823	9c9e391b15	Merge pull request 'feat(p10-1C-Go): tree-sitter-go AST extractor + chunker — Go 코드 색인 활성화' (#151 ) from feat/p10-1c-go into main	2026-05-20 10:16:09 +00:00
altair823	f95cd55484	docs(p10-1c-go): README/HANDOFF/ARCHITECTURE/SMOKE/INDEX + design §10.1; chore: bump version 0.11.1 → 0.12.0 Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 10:02:21 +00:00
altair823	ab288135e9	test(p10-1c-go): code-go-ast-v1 chunker snapshot + full-suite gate Mirrors code_python_ast_snapshot / code_ts_ast_snapshot patterns. In-memory CanonicalDocument (no kebab-parse-code dep — boundary §6.3 respected). verify: - cargo test -p kebab-chunk --test code_go_ast_snapshot → 2/2 - cargo test --workspace --no-fail-fast -j 1 → 0 failures (all green) - cargo clippy --workspace --all-targets -- -D warnings → clean - SMOKE: chunk.ParseDoc symbol + code_lang_breakdown {"go": 1} 확인 Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 09:54:17 +00:00
altair823	c19aa006d0	feat(p10-1c-go): activate Go in ingest_one_code_asset dispatch Replaces Go bail! arms with GoAstExtractor + CodeGoAstV1Chunker. Adds go_file_ingests_and_searches_as_code_citation integration test — asserts citation.lang=go, symbol=chunk.ParseDoc, code_lang=go. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 09:13:47 +00:00
altair823	f1a4f67e12	feat(p10-1c-go): code-go-ast-v1 chunker (1:1 + oversize split) Duplicate of code-rust-ast-v1 / code-{python,ts,js}-ast-v1 with language-agnostic body unchanged. Cross-chunker policy_hash identity asserted vs md-heading-v1. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 09:11:14 +00:00
altair823	6463c52827	feat(p10-1c-go): tree-sitter-go AST extractor (GoAstExtractor) Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 09:08:46 +00:00
altair823	2559d0d95a	refactor(p10-1c-go): add go to ingest dispatch allowlist (bail until Task F) Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 09:03:28 +00:00
altair823	4524830306	feat(p10-1c-go): route .go to MediaType::Code(go) Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 09:01:29 +00:00
altair823	8cdd3903c7	build(p10-1c-go): add tree-sitter-go workspace dep Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 09:00:04 +00:00
altair823	8b89961ada	docs(p10-1c-go): task spec + implementation plan Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 08:58:45 +00:00
altair823	eec90996aa	chore: bump version 0.11.0 → 0.11.1 dogfood semantic cleanup (PR #150) lands: document-centric fetch_span + assets.workspace_path 'last-registered' semantic explicitly documented. patch bump 사유: 외부 wire / CLI / config surface 변경 없음. 새 internal trait method (get_asset) + caller refactor + doc-comment 갱신. twin file 의 fetch_span 잘못 분기 가능성 fix (rare). Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 08:09:46 +00:00
altair823	ce1c778b4a	Merge pull request 'fix(dogfood): document-centric fetch_span + assets.workspace_path semantic doc' (#150 ) from fix/dogfood-asset-flip-flop-cleanup into main	2026-05-20 08:08:55 +00:00
altair823	453ec15df4	fix(dogfood): document-centric fetch_span + remove get_asset_by_workspace_path assets.workspace_path is INTENTIONALLY 'last-registered path' for twin files (identical content at different paths share one asset row PK'd by blake3 content hash). PR #146 made try_skip_unchanged document-centric; PR #149 made reset --orphans-only document-centric; this PR removes the last caller of get_asset_by_workspace_path (fetch.rs:193 in fetch_span, which used it to reject PDF/audio media — for twins this could read the wrong asset's media_type and pick the wrong branch). Replaced with the natural 2-step lookup: get_document_by_workspace_path (PR #146) → doc.source_asset_id → get_asset (NEW trait method, asset_id is PRIMARY KEY so flip-flop-immune by construction). Then removed get_asset_by_workspace_path trait method + SqliteStore impl — 0 callers after the refactor. UPSERT doc-comment refreshed in store.rs to make the 'last-registered' semantics explicit so future readers don't try to 'fix' the flip-flop. Dogfood follow-up (PR #142 1B + multi-root corpus). Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 08:03:38 +00:00
altair823	1e6de9fe9f	chore: bump version 0.10.0 → 0.11.0 dogfood follow-up (PR #149) lands: kebab reset --orphans-only explicit complement to PR #148's conservative sweep. minor bump 사유: 새 CLI flag (--orphans-only) + 새 ResetScope variant + ResetReport additive 필드 = surface 확장. design §10.4 트리거 충족. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 07:53:55 +00:00
altair823	9fa2a1ebac	Merge pull request 'feat(dogfood): kebab reset --orphans-only — explicit complement to PR #148 sweep' (#149 ) from feat/dogfood-reset-orphans-only into main	2026-05-20 07:50:43 +00:00
altair823	749c6ae240	docs(dogfood): sync reset_report schema + README for --orphans-only (PR #149 review) Round 1 review found 2 doc gaps: - docs/wire-schema/v1/reset_report.schema.json: 'orphans_only' missing from scope enum; orphans_purged/purged_paths properties absent - README: --orphans-only not listed in the reset prose Schema additions are additive minor (default values keep back-compat). Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 07:47:44 +00:00
altair823	5f2bd9e97e	feat(dogfood): kebab reset --orphans-only — purge stored docs outside walker scope PR #148 auto-purges only filesystem-missing files (conservative — leaves on-disk-but-out-of-scope docs alone for data safety). This is the explicit complement: when the user has narrowed include / widened exclude / removed a sub-directory from the workspace and WANTS the stored docs reconciled, they invoke 'kebab reset --orphans-only'. Confirm prompt with orphan count + sample paths; --yes required in non-TTY. SQLite purge via existing purge_deleted_workspace_path (PR #148) + vector store delete_by_chunk_ids when configured. No fs existence check — orphans-only is the explicit 'I know what I'm doing' variant. dogfood follow-up to PR #148 (file deletion auto-purge). Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 07:38:10 +00:00
altair823	1ce06c1e2d	chore: bump version 0.9.0 → 0.10.0 dogfood-discovered file-deletion auto-purge (PR #148) lands. minor bump 사유: additive wire field IngestReport.purged_deleted_files + 새 CLI summary surface (purged N) + 새 사용자-가시 동작 (rm a.md 후 ingest 시 자동 정리). design §10.4 도그푸딩-ready surface 확장 트리거. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 07:12:58 +00:00
altair823	d26efe167f	Merge pull request 'fix(dogfood): auto-purge stored docs for filesystem-deleted files' (#148 ) from fix/dogfood-file-deletion-auto-purge into main	2026-05-20 07:10:33 +00:00
altair823	d6d165df01	docs(dogfood): sync sweep_deleted_files algorithm doc with try_exists (PR #148 nit) Round 2 review found the function-level doc-comment still referenced the old fs::exists() (now replaced by try_exists().unwrap_or(true) in commit `2baa846`). One-line clarification — describes the conservative-on-Err semantics so future readers don't reintroduce the data-safety bug. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 07:10:27 +00:00
altair823	2baa846c6b	fix(dogfood): conservative try_exists() in sweep_deleted_files (PR #148 review) Round 1 review found a data-safety bug: fs::exists() returns false on errors like EACCES / EPERM / NFS-hiccup / ownership-change, which would trigger purge on a file that is in fact still on disk (just unreadable this moment). Switched to try_exists().unwrap_or(true) so transient FS errors are CONSERVATIVELY treated as 'file present' — never purge on uncertain signal. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 07:04:03 +00:00
altair823	27baec82ea	fix(dogfood): auto-purge stored docs for filesystem-deleted files Files deleted from disk (rm a.md) were leaving stale documents + chunks + embeddings in the store, surfacing as ghost citations in search/ask. Existing purge_orphan_at_workspace_path only handled content-changed stale (WHERE workspace_path=? AND asset_id != ?) — file deletion has no new asset_id. Fix: post-walker-scan sweep. Compute (stored_paths - scanned_paths), for each candidate check filesystem existence — only purge when the file is TRULY missing. Scope-narrowing case (file on disk but outside include glob) is explicitly NOT purged to protect users from accidental data loss via config edits. Adds: - DocumentStore::all_workspace_paths trait method + SqliteStore impl - purge_deleted_workspace_path in store-sqlite (returns chunk_ids for vector delete; deletes doc CASCADE + asset row + copied storage file) - sweep_deleted_files in kebab-app::ingest path; called once per ingest before the per-asset loop - IngestReport.purged_deleted_files counter (additive, serde default) - CLI ingest summary mentions purge count when > 0 - 2 integration tests: file_deletion_auto_purge + include_scope_narrowing_does_NOT_purge dogfood discovery (PR #142 1B + multi-root: kebab-docs + httpx + zod + lodash). Per user decision: only filesystem deletion auto-purges; scope narrowing requires explicit kebab reset. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 06:51:07 +00:00
altair823	acf8cf3be2	chore: bump version 0.8.3 → 0.9.0 dogfood-discovered routing additions (PR #147) land: - .mts / .cts → MediaType::Code(typescript) - .mdx → MediaType::Markdown minor bump 사유: 사용자 도그푸딩 surface 확장 — 이전에 skip 되던 28+ 파일이 이제 색인됨. design §10.4 dogfooding-ready surface 확장 = minor trigger. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 06:29:27 +00:00
altair823	ea5f7b22c8	Merge pull request 'feat(dogfood): route .mts/.cts → typescript + .mdx → markdown' (#147 ) from feat/dogfood-routing-cts-mts-mdx into main	2026-05-20 06:28:41 +00:00
altair823	5497c6e7b5	feat(dogfood): route .mts/.cts to typescript + .mdx to markdown Dogfood (PR #142 1B + multi-root: kebab-docs + httpx + zod + lodash) showed 28 files skipped by extension that are routable to existing extractors: - .mts (ESM TypeScript) / .cts (CommonJS TypeScript) — same grammar as .ts in tree-sitter-typescript 0.23 (LANGUAGE_TYPESCRIPT covers JSX- agnostic variants; LANGUAGE_TSX stays for .tsx only) - .mdx (Markdown + JSX) — routed as MediaType::Markdown; the md parser folds JSX islands through as raw passthrough Changes: - crates/kebab-source-fs/src/media.rs: 'mts'\|'cts' → Code(typescript), 'mdx' → Markdown. +2 unit tests. - crates/kebab-parse-code/src/lang.rs: code_lang_for_path matches mts/cts; module_path_for_tsjs strips .mts/.cts as well. Test cases extended. - crates/kebab-parse-code/src/typescript.rs: doc comment on select_grammar refreshed to mention .mts/.cts. - crates/kebab-parse-code/tests/lang.rs: 2 new assertions. verify: kebab-source-fs 44 / kebab-parse-code lib 20 + lang 4 all pass; clippy clean. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 06:24:21 +00:00
altair823	5a90940f1c	chore: bump version 0.8.2 → 0.8.3 dogfood-discovered fix (PR #146) lands: idempotent re-ingest now correctly returns Unchanged for twin files (identical content at different paths) via document-centric try_skip_unchanged lookup. patch bump 사유: advertised idempotency 의 정상 동작 복원. 새 wire / config / surface 변경 없음. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 06:20:34 +00:00
altair823	4389b887f0	Merge pull request 'fix(dogfood): document-centric try_skip_unchanged for twin-file idempotency' (#146 ) from fix/dogfood-bug4-idempotent-twin-files into main	2026-05-20 06:16:28 +00:00
altair823	360f825f3a	docs(dogfood): refresh try_skip_unchanged doc-comment to match new flow (PR #146 review) Round 1 review found the function-level doc-comment still described the old asset-side algorithm (item 2 asset-row checksum, item 3 id_for_doc miss). Updated to the document-centric flow. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 05:35:17 +00:00
altair823	641b92af7d	fix(dogfood): document-centric try_skip_unchanged for twin-file idempotency Identical-content files at different workspace paths share one assets row (assets.asset_id = blake3 content hash, PRIMARY KEY). The UPSERT `ON CONFLICT(asset_id) DO UPDATE SET workspace_path = excluded` made twin files overwrite each other's workspace_path on every ingest, so `get_asset_by_workspace_path(path1)` returned the OTHER twin's row (or None) — break idempotent unchanged-detection for both files. Fix: switch try_skip_unchanged to document-centric lookup. `documents. workspace_path` is already UNIQUE (V001) and `id_for_doc(path, ...)` includes path, so each twin has its own stable document row. Compare `doc.source_asset_id` with the new asset's checksum instead of going through the assets table. Dogfood (multi-root: kebab-docs + httpx + zod + lodash) showed 27 of 726 docs marked Updated on every idempotent re-ingest — all 27 are twin-file victims (empty `__init__.py` ×3, AGENTS.md ↔ CLAUDE.md same content, duplicate logo PDFs/JPGs). After: re-ingest reports 0 new / 0 updated / 726 unchanged. No schema migration needed. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 05:27:21 +00:00
altair823	08fb743598	chore: bump version 0.8.1 → 0.8.2 dogfood-discovered fixes (PR #145) land in production: - schema.v1.repo_breakdown 가 실제로 채워짐 (이전: 항상 빈 BTreeMap) - workspace.include glob 가 walker 에서 enforce 됨 (이전: 완전 무시) patch bump 사유: 둘 다 advertised surface 의 정상 동작 복원. 새 wire / config / surface 변경 없음. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 05:20:48 +00:00
altair823	0a2a7ae214	Merge pull request 'fix(dogfood): schema.repo_breakdown + workspace.include walker enforcement (dogfood-discovered)' (#145 ) from fix/dogfood-bugs-schema-walker-incremental into main	2026-05-20 05:18:59 +00:00
altair823	803d02b68b	fix(dogfood): enforce workspace.include in walker (allow-list semantics) config.workspace.include was completely ignored by the walker — connector.rs log_scope_include_warning literally said "handled by extractor router" but no extractor router exists. Dogfooding (PR #142 1B + multi-root corpus kebab-docs + httpx + zod + lodash) showed user-set include of code+md still ingested 84 .png + 8 .pdf files. Fix: walker treats scope.include as an allow-list — empty Vec preserves backward-compat (all files pass), non-empty requires file path to match at least one pattern (AND with the existing exclude rules). Removed the misleading debug log. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 05:15:04 +00:00
altair823	4e8b84c4e0	fix(dogfood): populate schema.v1.repo_breakdown (Task 9 follow-up) Dogfooding (PR #142 1B + multi-root corpus: kebab-docs + httpx + zod + lodash) revealed schema.v1.repo_breakdown is always {} despite the 1A-2 Task 9 having added the code_lang_breakdown sibling. The schema.rs:171 placeholder `BTreeMap::new()` was left in place. Mirror Task 9's code_lang_breakdown query for the repo field — same metadata_json JSON-path pattern. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-20 05:09:19 +00:00