fix(parse-image): #239 얇은 검출 박스가 이미지 OCR 전체를 날리던 문제 #240

Merged
altair823 merged 5 commits from fix/paddle-onnx-rec-min-width into main 2026-08-29 01:52:14 +00:00
16 changed files with 288 additions and 31 deletions

View File

@@ -1629,7 +1629,7 @@ fn ingest_one_image_asset(
}
};
// p9-fb-23 task 7: incremental-ingest early-skip for the image flow.
// Image docs use the `image-meta-v1` parser_version + the same
// Image docs use the `image-meta-v2` parser_version + the same
// MdHeadingV2Chunker as the markdown flow (single-block doc). The
// embedding-version check matches the markdown path: when the
// active embedder's model_version equals what was stamped on the
@@ -1676,7 +1676,7 @@ fn ingest_one_image_asset(
.extract_for(&asset.media_type, &ctx, &bytes)
.context("kb-app::extract_for (image)")?;
// v0.26.2: store the composite parser_version (extractor baked the base
// `image-meta-v1`, which already fixed doc_id). Skip compare + stored
// `image-meta-v2`, which already fixed doc_id). Skip compare + stored
// field must agree for next-run detection.
canonical.parser_version = eff_parser_version.clone();
// `[[workspace.sources]]`: stamp the owning source id (image extractor
@@ -2541,7 +2541,7 @@ fn ingest_one_pdf_asset(
}
};
// p9-fb-23 task 7: incremental-ingest early-skip for the PDF flow.
// PDF docs use `pdf-text-v2` as the parser_version and `PdfPageV1Chunker`
// PDF docs use `pdf-text-v3` as the parser_version and `PdfPageV1Chunker`
// as the chunker — both pinned per-medium today (no config knob).
// v0.26.2: composite parser_version folds pdf.ocr (enabled/always_on/
// model) + chunking, so enabling scanned-PDF OCR auto-re-indexes PDFs.
@@ -2578,7 +2578,7 @@ fn ingest_one_pdf_asset(
let mut canonical = app
.extract_for(&asset.media_type, &ctx, &bytes)
.context("kb-app::extract_for (pdf)")?;
// v0.26.2: store the composite parser_version (base `pdf-text-v2` already
// v0.26.2: store the composite parser_version (base `pdf-text-v3` already
// fixed doc_id) so the next run's skip compare matches.
canonical.parser_version = eff_parser_version.clone();
// `[[workspace.sources]]`: stamp the owning source id (pdf extractor

View File

@@ -447,7 +447,10 @@ where
Err(e) => {
// OCR failure: warning event + skip (text-detect block 그대로).
let note = format!(
"page={} OCR failed engine={} version={} err={}",
// `{e:#}` (not `{e}`): the ORT detail lives under a `.context`, and
// the plain Display drops it — so a note written with `{e}` reads
// `err=rec session run` and cannot be grepped for the real cause.
"page={} OCR failed engine={} version={} err={:#}",
page_num,
engine.engine_name(),
engine.engine_version(),

View File

@@ -1,6 +1,6 @@
use std::sync::Mutex;
use anyhow::Result;
use anyhow::{Context, Result};
use kebab_core::{Lang, OcrText};
use kebab_parse_image::OcrEngine;
@@ -52,7 +52,12 @@ impl OcrEngine for MockOcrEngine {
fn recognize(&self, _img: &[u8], _hint: Option<&Lang>) -> Result<OcrText> {
if self.fail {
anyhow::bail!("mock failure");
// Layered on purpose: the real paddle-onnx failure arrives as an
// ORT message under a `.context("rec session run")`, and anyhow's
// plain Display shows only the outer layer. A single-layer error
// would render identically under `{}` and `{:#}`, so it could not
// tell whether the provenance note keeps the cause (issue #239).
return Err(anyhow::anyhow!("mock inner cause")).context("mock failure");
}
let mut idx = self.call_index.lock().unwrap();
let text = self

View File

@@ -324,6 +324,18 @@ fn ocr_engine_failure_surfaces_as_warning() {
warning_with_failure,
"OCR failure 의 error message 가 warning event 의 note 안"
);
// issue #239: the note must carry the whole error chain, not just the
// outermost layer. The real cause (ORT's "Invalid input shape") sits under
// a `.context`, so a note formatted with `{e}` instead of `{e:#}` drops it
// and the KB can no longer be searched for which documents were hit.
let warning_with_cause = canonical.provenance.events.iter().any(|e| {
e.kind == kebab_core::ProvenanceKind::Warning
&& e.note.as_deref().unwrap_or("").contains("mock inner cause")
});
assert!(
warning_with_cause,
"provenance note 가 error chain 의 안쪽 원인까지 담아야 한다 (`{{e:#}}`)"
);
}
// Test 9: dual-block ordinals are deterministic and unique

View File

@@ -165,7 +165,7 @@ fn ingest_3_page_pdf_produces_one_doc_and_per_page_chunks() {
pdf_item.parser_version
.as_ref()
.map(|p| p.0.split('|').next().unwrap()),
Some("pdf-text-v2")
Some("pdf-text-v3")
);
assert_eq!(
pdf_item.chunker_version.as_ref().map(|c| c.0.as_str()),
@@ -479,10 +479,10 @@ fn inspect_doc_surfaces_page_spans() {
.find(|i| i.doc_path.0.ends_with("inspect.pdf"))
.unwrap();
let doc = kebab_app::inspect_doc_with_config(cfg, pdf_item.doc_id.as_ref().unwrap()).unwrap();
// v0.26.2: stored parser_version is now `pdf-text-v2|<ingest-config-sig>`
// v0.26.2: stored parser_version is now `pdf-text-v3|<ingest-config-sig>`
// (the signature folds chunking / pdf.ocr settings for skip detection).
// Assert the base identity by taking the prefix before the first '|'.
assert_eq!(doc.parser_version.0.split('|').next().unwrap(), "pdf-text-v2");
assert_eq!(doc.parser_version.0.split('|').next().unwrap(), "pdf-text-v3");
assert_eq!(doc.blocks.len(), 3);
for block in &doc.blocks {
match block {

View File

@@ -400,7 +400,10 @@ mod tests {
fn make_pdf_doc(pages: &[&str]) -> CanonicalDocument {
let workspace_path = WorkspacePath::new("docs/test.pdf".into()).unwrap();
let asset_id = AssetId("a".repeat(64));
let parser_version = ParserVersion("pdf-text-v2".into());
// Version-neutral on purpose: `kebab-chunk` cannot import
// `kebab-parse-pdf` (design §8), so a real version string here would
// go stale at every bump. The chunker only feeds it to `id_for_doc`.
let parser_version = ParserVersion("test-parser-v1".into());
let doc_id = id_for_doc(&workspace_path, &asset_id, &parser_version);
let mut blocks: Vec<Block> = Vec::new();

View File

@@ -49,7 +49,19 @@ use serde_json::{Map, Value};
use time::OffsetDateTime;
/// Parser version label for the image extractor (§9 versioning).
pub const PARSER_VERSION: &str = "image-meta-v1";
///
/// Bumped to v2 for issue #239 (2026-08-28): a detection box narrower than
/// the rec graph's floor used to abort the ONNX session and discard OCR for
/// the *whole* image, so an affected image was indexed with its filename and
/// nothing else. `REC_MIN_WIDTH` in `paddle_onnx` fixes the extraction; this
/// bump is what makes the fix reach stores that were already indexed.
/// `try_skip_unchanged` calls an asset Unchanged when its content hash *and*
/// version inputs match — the image file on disk did not change, and neither
/// did the OCR model assets that feed `ingest_config_signature`, so without
/// this every already-indexed image would keep its empty extraction until the
/// user thought to pass `--force-reingest`. Per CLAUDE.md §Versioning
/// cascade, changing this invalidates downstream image records.
pub const PARSER_VERSION: &str = "image-meta-v2";
/// Maximum decode dimension (per axis) before we refuse to read the image.
/// Matches the §9.1 "cap decode at ~16k" policy in the design doc.

View File

@@ -51,6 +51,28 @@ const REC_CLASSES: usize = 11947;
const DET_LIMIT_SIDE_LEN: u32 = 960;
/// rec input height (PP-OCRv5 mobile).
const REC_HEIGHT: u32 = 48;
/// Narrowest rec input the graph survives. The backbone emits
/// `T = ceil((w - 4) / 8)` CTC timesteps, so `w <= 4` collapses the feature
/// map to zero columns and ORT aborts the whole session run with
/// `Invalid input shape: {1,0}` — taking every already-recognized box on the
/// image down with it (issue #239). Measured against the bundled
/// `korean_ppocrv5_mobile_rec.onnx`: 1..=4 always fail, 5.. always succeed.
/// `rec_min_width_is_the_graph_floor` holds the constant from below — set it
/// under the graph's floor and that test errors. The upward direction cannot
/// be a test (both of its assertions pass *better* as the constant grows), so
/// it is the `const _` ceiling right below instead.
const REC_MIN_WIDTH: u32 = 5;
/// Raising `REC_MIN_WIDTH` is only known to be free while it stays inside the
/// band that was actually measured: widths 5..=16 come back empty even with
/// real ink in them, while width 40 reads glyphs at 0.97+ confidence (issue
/// #239). 17..=39 was never swept, so a ceiling above 16 is not measurement
/// any more — and by 40 the guard would be discarding crops the graph reads,
/// which is the silent loss this constant exists to prevent. Enforced at
/// compile time rather than left to review.
const _: () = assert!(
REC_MIN_WIDTH <= 16,
"REC_MIN_WIDTH is past the measured no-loss ceiling (16)"
);
/// DBNet probability-map binarization threshold. Looser than Paddle's default
/// `box_thresh` (0.6) to keep recall high on low-contrast Korean text.
const DET_BIN_THRESH: f32 = 0.3;
@@ -356,6 +378,13 @@ impl OnnxPaddleOcr {
// resize keep-aspect to height 48, then this single crop is its own batch
let (cw, ch) = (crop.width().max(1), crop.height().max(1));
let new_w = ((REC_HEIGHT as f32 / ch as f32) * cw as f32).round().max(1.0) as u32;
if new_w < REC_MIN_WIDTH {
// A sliver this thin holds no glyph, and feeding it to the graph
// would abort recognition for the entire image. Report it the way
// an undecodable box is already reported: the caller's
// `text.is_empty()` arm drops this box and keeps the rest.
return Ok((String::new(), 0.0));
}
let resized = image::imageops::resize(
crop,
new_w,
@@ -1002,4 +1031,57 @@ mod tests {
);
}
}
/// Issue #239: a detection box narrower than the rec graph's floor used to
/// abort the ORT session, and `recognize`'s `?` turned that into "this
/// image has no OCR text at all" — 36 of 240 corpus images.
///
/// Both runtime directions are asserted here. Below
/// `REC_MIN_WIDTH` the guard must short-circuit (delete it and this
/// errors). At exactly `REC_MIN_WIDTH` the real session must run, which is
/// what keeps the constant pinned to the shipped model rather than being a
/// number nobody rechecks (set it below the graph's floor, e.g. after a
/// model swap, and this errors). The upward direction — a constant raised
/// past the width band that was measured to decode nothing anyway, which
/// would silently drop crops the graph can read — is pinned by the
/// `const _` assertion next to `REC_MIN_WIDTH` itself.
#[test]
fn rec_min_width_is_the_graph_floor() {
// The upward direction is pinned by the `const _` next to
// REC_MIN_WIDTH — a raise past the measured ceiling fails the build,
// not just this test.
//
// Pin to the *bundled* assets: `ModelPaths::from_default_dir` honors
// `KEBAB_IMAGE_OCR_MODEL_DIR`, and a developer with that exported would
// otherwise measure their own model here.
let dir = Path::new(env!("CARGO_MANIFEST_DIR")).join("assets/paddleocr-onnx");
let engine = OnnxPaddleOcr::from_paths(
&ModelPaths {
det: dir.join("ppocrv5_mobile_det.onnx"),
rec: dir.join("korean_ppocrv5_mobile_rec.onnx"),
dict: dir.join("korean_dict.txt"),
},
0.3,
1.5,
1000,
1600,
)
.expect("bundled OCR assets must load");
// A crop already at REC_HEIGHT makes run_rec's keep-aspect resize the
// identity, so the crop width *is* the rec input width — no rounding
// to reason about.
let crop = |w| image::RgbImage::from_pixel(w, REC_HEIGHT, image::Rgb([255, 255, 255]));
for w in 1..REC_MIN_WIDTH {
let (text, conf) = engine
.run_rec(&crop(w))
.unwrap_or_else(|e| panic!("w={w} must never reach the rec session: {e:#}"));
assert!(text.is_empty(), "w={w} decoded {text:?}");
assert_eq!(conf, 0.0, "w={w}");
}
engine.run_rec(&crop(REC_MIN_WIDTH)).unwrap_or_else(|e| {
panic!("REC_MIN_WIDTH={REC_MIN_WIDTH} must survive the rec session: {e:#}")
});
}
}

View File

@@ -36,6 +36,13 @@ use kebab_core::{
use serde_json::{Map, Value};
use time::OffsetDateTime;
/// Bumped to v3 for issue #239 (2026-08-28): scanned pages share the
/// paddle-onnx rec path with image OCR, so a page whose raster produced one
/// over-thin detection box lost that page's OCR text entirely. The fix lives
/// in `kebab-parse-image`; this bump is what re-processes scans that were
/// already indexed under v2 (same reasoning as the v2 bump below — the PDF
/// bytes have not changed, so nothing else would invalidate them).
///
/// Bumped to v2 for issue #232 (2026-08-17): scanned pages are now
/// rasterized by rendering rather than by pulling out an embedded JPEG,
/// so pages encoded with CCITTFax / JBIG2 / Flate / JPX — previously
@@ -47,7 +54,7 @@ use time::OffsetDateTime;
/// every already-indexed scan would keep its empty extraction until the
/// user thought to pass `--force-reingest`. Per CLAUDE.md §Versioning
/// cascade, changing this invalidates downstream PDF records.
pub const PARSER_VERSION: &str = "pdf-text-v2";
pub const PARSER_VERSION: &str = "pdf-text-v3";
/// Text-PDF extractor. Per-page text via `lopdf::Document::extract_text`
/// (the only stable per-page API in the lopdf / pdf-extract pair —

View File

@@ -267,7 +267,7 @@ fn snapshot_three_page_canonical_document_stable() {
// golden file (the full JSON contains BLAKE3 ids that would
// change if `id_from(...)`'s tuple shape ever shifts — that would
// be a separate, intentional break).
assert_eq!(json["parser_version"], Value::String("pdf-text-v2".into()));
assert_eq!(json["parser_version"], Value::String("pdf-text-v3".into()));
assert_eq!(json["lang"], Value::String("und".into()));
assert_eq!(json["schema_version"], Value::Number(1.into()));
assert_eq!(json["doc_version"], Value::Number(1.into()));

View File

@@ -1,5 +1,5 @@
{
"doc_id": "bd04fa013899592afcf671013404ed68",
"doc_id": "960e4c445f1365a633c9647b42ed27ba",
"source_asset_id": "babe9824b6b28237c0898575a40ba48d",
"workspace_path": "mojibake.pdf",
"title": "untitled",
@@ -8,7 +8,7 @@
{
"kind": "paragraph",
"common": {
"block_id": "964162ec3cf0c191849f5a5cf8a3b675",
"block_id": "b85b8a6d2462bb396112f45cfcea6c44",
"heading_path": [],
"source_span": {
"kind": "page",
@@ -54,11 +54,11 @@
"at": "1970-01-01T00:00:00Z",
"agent": "kb-parse-pdf",
"kind": "parsed",
"note": "parser_version=pdf-text-v2; page_count=1"
"note": "parser_version=pdf-text-v3; page_count=1"
}
]
},
"parser_version": "pdf-text-v2",
"parser_version": "pdf-text-v3",
"schema_version": 1,
"doc_version": 1,
"last_chunker_version": null,

View File

@@ -20,11 +20,11 @@ Cargo workspace, 함수 호출 기반 모듈러 모놀리스. UI binary (`kebab-
| 한국어 형태소분석 | `lindera-ko-dic` (FTS5 외부 tokenizer, v0.20.1) — 2자 이상 한국어 query 지원 |
| LLM | Ollama HTTP (default `gemma4:e4b` ─ OCR / caption 와 family 통일. 사용자가 더 큰 variant `gemma4:26b` 등으로 override 가능) |
| 음성 ASR | `whisper.cpp` (via `whisper-rs`) — P8 보류, 시스템 dep brainstorm 후 |
| OCR (image) | `OcrEngine` trait, 2 백엔드: **`ollama-vision`** (default, `gemma4:e4b`) / **`paddle-onnx`** (v0.27.0 — PP-OCRv5 ONNX in-process via `ort` =2.0.0-rc.9, DBNet det + CTC rec, 후처리 min-area rect/unclip pure-Rust, Python 런타임 0). engine 선택은 `[image.ocr] engine`, 팩토리는 `kebab-app::build_image_ocr_engine`. e2e CER 0.005 / 큰 페이지 <4초. (HOTFIXES P6-2, 2026-06-04) |
| OCR (image) | `OcrEngine` trait, 2 백엔드: **`ollama-vision`** (default, `gemma4:e4b`) / **`paddle-onnx`** (v0.27.0 — PP-OCRv5 ONNX in-process via `ort` =2.0.0-rc.9, DBNet det + CTC rec, 후처리 min-area rect/unclip pure-Rust, Python 런타임 0). engine 선택은 `[ingest.image.ocr] engine`, 팩토리는 `kebab-app::build_image_ocr_engine`. e2e CER 0.005 / 큰 페이지 <4초. (HOTFIXES P6-2, 2026-06-04) **불변식**: rec 세션 입력 폭은 `REC_MIN_WIDTH = 5` 이상이어야 한다 — 백본이 `T = ceil((w-4)/8)` 개의 CTC 타임스텝을 내므로 `w <= 4` 는 특징맵을 0 열로 접어 ORT 가 세션 전체를 실패시킨다. 그 실패는 `recognize` 의 박스 루프를 뚫고 나가 이미 인식한 박스까지 전부 버리므로, 얇은 크롭은 세션에 넣지 말고 빈 문자열로 돌려보내야 한다 (issue #239). `parser_version = "image-meta-v2"` (issue #239 에서 v1 → v2, 기존 색인 이미지 재처리 유발). |
| OCR (PDF, v0.20.0+) | Ollama vision LM (default `qwen2.5vl:3b`) — post-extract enrichment via `kebab-app::pdf_ocr_apply` (H-1 resolution). DCTDecode-only v1 (FlateDecode/CCITTFax skip + warning). family asymmetry vs image OCR: PoC alnum 94.79% (qwen2.5vl) >> 27% (gemma4:e4b 받침), 본 단계에서 PDF OCR 만 qwen2.5vl. |
| Image caption | Ollama vision LM, runtime gate `image.caption.enabled` (default OFF) |
| RAG groundedness 검증 | `kebab-nli` 의 mDeBERTa-v3 XNLI 가 `(packed_chunks, generated_answer)` entailment 검사 (fb-41). `[rag] nli_threshold > 0` (default 0 = disabled, production 권장 0.5) 일 때 활성 — 미달 시 `refusal_reason = nli_verification_failed` (LLM self-judge ceiling 보완). 첫 호출 시 ~280 MB ONNX 자동 다운로드 |
| PDF parser | `lopdf` per-page 텍스트 + 스캔 페이지 래스터화. 래스터는 **pdfium 페이지 렌더링**(`page_render::PageRenderer`, issue #232) 이 1순위 — 필터·XObject 구성과 무관하게 페이지를 그린다. pdfium 이 없으면 `page_image::extract_dctdecode_page_image` 로 떨어지며 그 경우 단일 DCTDecode 이미지 페이지만 OCR 된다. pdfium 은 공유 라이브러리로만 배포돼 링크하면 단일 바이너리 원칙이 깨지므로 **런타임 바인딩**이고, `[ingest.pdf.ocr] render_library` 로 경로를 지정하거나 로더 경로에 두면 된다. `kebab doctor` 의 `pdf_render` 가 어느 쪽인지 보고한다. `chunker_version = "pdf-page-v1"` 하드코딩 (HOTFIXES P7-3). `parser_version = "pdf-text-v2"` (issue #232 에서 v1 → v2, 기존 색인 스캔본 재처리 유발). |
| PDF parser | `lopdf` per-page 텍스트 + 스캔 페이지 래스터화. 래스터는 **pdfium 페이지 렌더링**(`page_render::PageRenderer`, issue #232) 이 1순위 — 필터·XObject 구성과 무관하게 페이지를 그린다. pdfium 이 없으면 `page_image::extract_dctdecode_page_image` 로 떨어지며 그 경우 단일 DCTDecode 이미지 페이지만 OCR 된다. pdfium 은 공유 라이브러리로만 배포돼 링크하면 단일 바이너리 원칙이 깨지므로 **런타임 바인딩**이고, `[ingest.pdf.ocr] render_library` 로 경로를 지정하거나 로더 경로에 두면 된다. `kebab doctor` 의 `pdf_render` 가 어느 쪽인지 보고한다. `chunker_version = "pdf-page-v1"` 하드코딩 (HOTFIXES P7-3). `parser_version = "pdf-text-v3"` (issue #232 에서 v1 → v2, issue #239 에서 v2 → v3). 고친 것은 스캔본 경로지만 **재처리 대상은 기존 색인 PDF 전부** 다 — base `parser_version` 이 `id_for_doc` 에 접히므로 doc_id 가 바뀌고 store 가 다시 쓰인다 (HOTFIXES 2026-08-28). |
| code parser | `tree-sitter` + `tree-sitter-rust` / `tree-sitter-python` / `tree-sitter-typescript` / `tree-sitter-javascript` / `tree-sitter-go` / `tree-sitter-java` / `tree-sitter-kotlin-ng` — **parser-side** (`kebab-parse-code`), chunker-side 아님 (design §6.3). chunker versions: Rust = `code-rust-ast-v1`, Python = `code-python-ast-v1`, TypeScript = `code-ts-ast-v1`, JavaScript = `code-js-ast-v1`, Go = `code-go-ast-v1`, Java = `code-java-ast-v1`, Kotlin = `code-kotlin-ast-v1`. (v0.32.0 #220: 9개 언어 chunker 가 단일 `CodeAstV1Chunker` 로 통합 — `for_lang(lang)` 가 per-lang `chunker_version` 라벨을 verbatim 매핑. chunker 는 tree-sitter 미사용·`lang` 은 `SourceSpan::Code` 데이터에서 흐르므로 9개 struct 차이는 `VERSION_LABEL` 문자열뿐이었음 → chunk_id byte-identical.) `ast_chunk_max_lines = 200` 상수 고정 (HOTFIXES 2026-05-19 — Chunker trait 이 per-medium config 미노출). Kotlin grammar 은 `tree-sitter-kotlin-ng` 사용 — bare `tree-sitter-kotlin` 은 tree-sitter 0.21–0.23 에 고착되어 있어 사용 불가. **Tier 2 (p10-2)**: YAML/k8s → `serde_yaml_ng` + `k8s-manifest-resource-v1` (apiVersion+kind per resource), Dockerfile → `dockerfile-file-v1` (whole-file), Cargo.toml/go.mod/.json/.xml/.groovy → `manifest-file-v1` (whole-file). Tier 2 chunkers live in `kebab-chunk`; no tree-sitter grammar needed (structure from file type, not AST). **Tier 3 (p10-3)**: shell scripts (`.sh`/`.bash`/`.zsh`) direct → `code-text-paragraph-v1` (blank-line paragraph segmentation + 80-line / 20-overlap line-window for oversize). Same chunker also serves as fallback when Tier 1/2 emit 0 chunks or Err — non-k8s YAML / invalid YAML / AST extractor failures all picked up. symbol = None; lang preserved from input doc. **Tier 1 family complete (p10-1D)**: C (`tree-sitter-c`, `code-c-ast-v1`, `.c`/`.h`) + C++ (`tree-sitter-cpp`, `code-cpp-ast-v1`, `.cpp`/`.cc`/`.cxx`/`.hpp`/`.hh`/`.hxx`). C symbol = function name only; C++ symbol = `namespace::Class::method` (recursive nesting). `.h` 가 C++ syntax 만나면 tree-sitter-c parse 실패 → Tier 3 fallback. |
| symbol path 형식 | workspace path → module path: Python = dotted prefix (`kebab_eval.metrics.compute_mrr`), TypeScript/JavaScript = slash-style prefix (`src/Foo.Foo.search`), Go = `package.Func` / `package.(*Receiver).Method`, Java/Kotlin = `com.foo.Foo.bar` (패키지+클래스+메서드/필드), C = 함수명, C++ = `namespace::Class::method`. Rust 1A-2 는 file-scope nesting 만 (workspace prefix 없음, 비일관 수용 — HOTFIXES 2026-05-20). code chunk 은 `citation.kind = "code"` + `citation.lang` + `symbol` + line range, SearchHit 에 `code_lang` + `repo`(`.git` walk-up 디렉토리명) backfill. |
| Desktop | Tauri 2 + `pdfjs-dist` (native PDF render backend 금지) — P9-5 |

View File

@@ -122,10 +122,18 @@ endpoint = "http://192.168.0.47:11434"
enabled = false # opt-in
```
**Config (v0.28.0~)**: 위 블록은 `[ingest.image.ocr]` / `[ingest.image.caption]` 로 옮겨졌다. 옛 키를 그대로 쓰면 **`schema_version` 이 5 보다 낮은 파일에서만** 로드 시 자동 이관된다 — 이미 `schema_version = 5` 인 config 에 `[image.ocr]` 를 붙여 넣으면 경고 없이 통째로 무시되고 OCR 이 꺼진 채로 돈다. `paddle-onnx` 백엔드 (v0.27.0~, in-process ONNX — Ollama 없이 돈다):
```toml
[ingest.image.ocr]
enabled = true
engine = "paddle-onnx"
```
**verify**:
- `*.png` / `*.jpg` / `*.jpeg` 만 ingest target.
- OCR text 가 `Block::ImageRef.ocr.joined` 안.
- `[image.caption].enabled=true` 시 caption 도.
- caption 토글을 켜면 caption 도.
- (issue #239) 한 장도 **통째로** 비지 않는다. 얇은 검출 박스 하나가 rec 세션을 실패시키면 그 이미지의 인식 결과가 전량 버려지던 버그였다. 색인은 성공으로 끝나므로 `documents.provenance_json` 에 `Invalid input shape` 이 남았는지로 확인한다. (PDF 경로는 노트 형식이 달라 문자열이 다르다 — HOTFIXES 2026-08-28 참고.)
**scenarios**:
- 1.2.a Korean OCR (한국어 scan PNG) → OCR text + search hit.
@@ -133,6 +141,7 @@ enabled = false # opt-in
- 1.2.c photo (자연 사진, OCR 없음) → empty OCR or warning.
- 1.2.d corrupt image → graceful error.
- 1.2.e oversized image (> max_pixels) → downscale.
- 1.2.f (issue #239) `[ingest.image.ocr] engine = "paddle-onnx"` 로 §13.4 이미지 코퍼스 전량 → OCR 오류 0 건. 먼저 진행 출력에 `ocr(ppocrv5-mobile-kor…)` 단계가 실제로 찍히는지 본다 — 안 찍히면 OCR 이 꺼진 것이고, 그 상태에서는 모든 이미지가 본문 0 자로 나와 시나리오가 통과한 것처럼 보인다. 본문이 파일명뿐인 문서가 남으면 provenance 를 확인한다. 글자가 없는 사진이라 0 자인 것과 이 버그로 통째로 버려진 것은 다르다 — 후자만 provenance 에 오류가 남는다.
### §1.3 PDF text ingest (P7-1)
@@ -141,7 +150,7 @@ enabled = false # opt-in
- 1 `Block::Paragraph` per page (P7-1 invariant).
**verify**:
- `parser_version = "pdf-text-v2"`.
- `parser_version = "pdf-text-v3"`.
- `chunker_version = "pdf-page-v1"` (또는 `"pdf-page-v1.1"` from v0.20.1).
- `block_count` ≥ page count.
@@ -975,7 +984,16 @@ bug 발견 시:
### §13.4 Image corpus
(P6 dogfood — 향후 추가).
도그푸딩 스토어의 `corpus/images/` — 4 분류 240 개 파일 (jpg 172 · png 63 · jpeg 4 · tif 1). `kebab ingest` 가 집는 것은 tif 를 뺀 239 개다.
| 분류 | 장수 | 쓰임 |
|---|---|---|
| `charts/` | 50 | 도표·다이어그램. 박스가 많고 얇은 조각이 잘 생긴다. #239 실패 8 장. |
| `english-text/` | 70 | 스크린샷·표지판 등 영문 위주. 1.2.b 판정용. #239 실패 10 장. |
| `korean-text/` | 70 | 한국어 스캔·필기. 1.2.a. #239 실패 9 장. |
| `photos/` | 50 | 글자가 거의 없는 자연 사진. 1.2.c 의 "정상적으로 0 자" 대조군. #239 실패 9 장 — 글자가 없어도 얇은 박스는 검출되므로 사진도 걸렸다. |
issue #239 실측 근거가 이 코퍼스 전량 스윕이다 (수정 전 36/240 실패 → 수정 후 0). 상세는 tasks/HOTFIXES.md 2026-08-28 항목.
---

View File

@@ -713,7 +713,7 @@ KB --json schema | jq '.stats.code_lang_breakdown'
- 코퍼스에 없는 주제로 `kebab ask` → `refusal_reason: "llm_self_judge"` (또는 `no_chunks` / `score_gate`) + `grounded: false`.
- (P6-4) `image.ocr.enabled = true` 로 PNG 자산을 ingest 하면 `kebab list docs` 가 markdown 옆에 image doc 도 출력 (`workspace_path` 가 `*.png`). `kebab inspect doc <image_doc_id>` 의 `block.ocr.joined` 가 vision LM 의 OCR 결과 (예: 스크린샷 안의 텍스트). `kebab search --mode lexical "<OCR text>"` 가 그 image chunk 를 반환하면 wiring 정상.
- OCR / caption 부분 실패는 `errors` 카운터 미증가 — `kebab inspect doc <id>` 의 Provenance Warning 이벤트 또는 `--debug` 로그에서만 확인.
- (P7-3) `*.pdf` 자산을 워크스페이스에 두면 `kebab ingest` 출력에 PDF 도 `new` 카운터에 포함. `kebab inspect doc <pdf_doc_id>` 가 `parser_version = "pdf-text-v2"` + 페이지마다 `Block::Paragraph` + `SourceSpan::Page { page, char_start, char_end }`. 본문에 등장하는 단어로 `kebab search --mode hybrid` 시 PDF chunk 가 결과에 포함되고 `source_span.kind = "page"` 면 wiring 정상. 암호화 PDF 는 `errors+=1` 로 분류되며 `error` 필드에 `qpdf --decrypt` 안내 보존. 빈/스캔 페이지 (PDF 가 텍스트를 추출하지 못한 페이지) 는 0 chunk + `Provenance::Warning` ("scanned candidate") 로 표시 — P+ scanned-PDF OCR fallback 까지는 검색 불가.
- (P7-3) `*.pdf` 자산을 워크스페이스에 두면 `kebab ingest` 출력에 PDF 도 `new` 카운터에 포함. `kebab inspect doc <pdf_doc_id>` 가 `parser_version = "pdf-text-v3"` + 페이지마다 `Block::Paragraph` + `SourceSpan::Page { page, char_start, char_end }`. 본문에 등장하는 단어로 `kebab search --mode hybrid` 시 PDF chunk 가 결과에 포함되고 `source_span.kind = "page"` 면 wiring 정상. 암호화 PDF 는 `errors+=1` 로 분류되며 `error` 필드에 `qpdf --decrypt` 안내 보존. 빈/스캔 페이지 (PDF 가 텍스트를 추출하지 못한 페이지) 는 0 chunk + `Provenance::Warning` ("scanned candidate") 로 표시 — P+ scanned-PDF OCR fallback 까지는 검색 불가.
## config migrate (마이그레이션)

View File

@@ -26,22 +26,27 @@ classDiagram
parse_blocks(body) (Vec~ParsedBlock~, Warnings)
}
class PdfTextExtractor {
PARSER_VERSION = "pdf-text-v2"
PARSER_VERSION = "pdf-text-v3"
new() Self
}
class ImageExtractor {
PARSER_VERSION = "image-meta-v1"
PARSER_VERSION = "image-meta-v2"
MAX_DECODE_DIM = 16384
new() Self
}
class OcrEngine {
<<trait kebab-parse-image>>
engine_id() str
run(image_bytes, langs) OcrText
engine_name() str
engine_version() String
recognize(image_bytes, lang_hint) Result~OcrText~
}
class OllamaVisionOcr {
endpoint, model, max_pixels
}
class OnnxPaddleOcr {
REC_HEIGHT = 48
REC_MIN_WIDTH = 5
}
class CaptionFns {
caption_image(lm, prep, opts) ModelCaption
apply_caption(block, lm, opts)
@@ -49,6 +54,7 @@ classDiagram
Extractor <|.. PdfTextExtractor
Extractor <|.. ImageExtractor
OcrEngine <|.. OllamaVisionOcr
OcrEngine <|.. OnnxPaddleOcr
ImageExtractor ..> OcrEngine : applied via apply_ocr
ImageExtractor ..> CaptionFns : applied via apply_caption
```
@@ -101,13 +107,15 @@ flowchart LR
**PDF** (`kebab-parse-pdf`):
- `PdfTextExtractor` — `Extractor` 구현체. `lopdf::Document::load_mem` 로 한 번 파싱, encrypted 면 즉시 bail.
- `PARSER_VERSION = "pdf-text-v2"` — version cascade entry (issue #232 에서 v1 → v2, 페이지 렌더링 도입으로 기존 색인 스캔본 재처리 유발). (HOTFIXES P7-2 의 chunker_version `pdf-page-v1` 와 별개.)
- `PARSER_VERSION = "pdf-text-v3"` — version cascade entry (issue #232 에서 v1 → v2, 페이지 렌더링 도입; issue #239 에서 v2 → v3, 얇은 검출 박스가 페이지 OCR 을 통째로 날리던 것을 고치면서). 고친 것은 둘 다 스캔본 경로지만 **재처리 대상은 기존 색인 PDF 전부** — base 가 `id_for_doc` 에 접혀 doc_id 가 바뀐다. (HOTFIXES P7-2 의 chunker_version `pdf-page-v1` 와 별개.)
- 빈 페이지 / extract 실패 → `Block::Paragraph` 빈 inlines + `ProvenanceKind::Warning("scanned candidate")`. OCR fallback 미구현.
**Image** (`kebab-parse-image`):
- `ImageExtractor` — `Extractor` 구현체. `MAX_DECODE_DIM = 16384` 초과 거부 (decode bomb 방어).
- `OcrEngine` (trait) — `engine_id() / run(...) -> OcrText`. `OcrText.engine` 필드로 trust level 분기.
- `OllamaVisionOcr { endpoint, model, max_pixels }` — v1 유일 구현. `apply_ocr(block, engine, langs)` 가 `ImageRefBlock.ocr` 슬롯 채움.
- `PARSER_VERSION = "image-meta-v2"` — version cascade entry (issue #239 에서 v1 → v2, 얇은 검출 박스가 이미지 OCR 을 통째로 날리던 것을 고치면서). PDF 와 마찬가지로 기존 색인 이미지 **전부** 가 재처리 대상이다.
- `OcrEngine` (trait) — `engine_name() -> &'static str` / `engine_version() -> String` / `model() -> &str` / `recognize(&[u8], Option<&Lang>) -> Result<OcrText>`. `OcrText.engine` 필드로 trust level 분기.
- `OllamaVisionOcr { endpoint, model, max_pixels }` — `ollama-vision` 백엔드 (기본값). `apply_ocr(block, engine, langs)` 가 `ImageRefBlock.ocr` 슬롯 채움.
- `OnnxPaddleOcr` — `paddle-onnx` 백엔드 (v0.27.0, PP-OCRv5 ONNX in-process). rec 세션 입력 폭 하한은 `REC_MIN_WIDTH = 5` — 그 아래는 세션에 넣지 않고 빈 문자열을 돌려준다 (issue #239).
- `caption_image(lm: &dyn LanguageModel, prep, opts) -> Result<ModelCaption>` — `LanguageModel.generate_stream` 의 vision 입력 (`GenerateRequest.images`) 사용. `apply_caption` 이 block 에 in-place 주입.
## 외부 의존

View File

@@ -14,6 +14,113 @@ historical contract that was implemented; this file accumulates the
deltas so phase 5+ readers can find the live behavior without diffing
git history.
## 2026-08-28 — #239 얇은 검출 박스 하나가 이미지 OCR 전체를 날림 (paddle-onnx rec 폭 하한)
### 무엇이 문제였나
`OnnxPaddleOcr::run_rec` 은 검출된 박스를 높이 48 로 리사이즈하면서 **폭을 1 이상으로만** 보장했다. 그런데 PP-OCRv5 rec 백본은 폭을 반복해서 줄이기 때문에 폭이 한 자릿수인 입력은 도중에 폭 0 인 특징맵이 되고, ORT 가 Conv 에서 세션 실행 전체를 실패시킨다.
그 실패가 `recognize` 의 `?` 를 타고 나가면서, **이미 인식해 둔 나머지 박스가 전부 함께 버려졌다.** 바로 아래 줄(`if text.is_empty() { continue; }`)이 "박스 하나가 비면 나머지는 살린다"는 의도를 담고 있는데 오류 경로에만 그 방어가 없었던 것이다. 손실이 얇은 조각 하나에서 그치지 않고 그 이미지 전체로 번졌다.
색인 자체는 성공으로 끝나므로 **조용한 손실**이었다. 이미지 문서는 본문에 파일명만 남고, 스캔 PDF 페이지는 청크가 0 이 된다. 검색해서 0 건이 나올 때까지 드러나지 않는다.
### 하한은 16 이 아니라 5 였다 — 실측
이슈는 `REC_MIN_WIDTH = 16` 을 제안했지만, 번들된 `korean_ppocrv5_mobile_rec.onnx` 를 직접 스윕해 보면 **실패하는 폭은 4 까지이고, 그래서 하한은 5** 다. 서로 다른 경로로 세 번 재 봤고 같은 표가 나왔다.
| rec 입력 폭 (높이 48) | 결과 |
|---|---|
| 1 ~ 4 | 전부 실패 — `Invalid input shape: {1,0}` |
| 5 ~ 48 | 전부 성공 |
경계는 흐릿하지 않고 딱 떨어지며, 픽셀 내용과 무관하다(흰 이미지·체커보드·글자 모양 렌더 모두 동일). 이유도 유도된다: rec 출력의 CTC 타임스텝 수가 `T = ceil((w - 4) / 8)` 이라서 `w <= 4` 에서 `T = 0` 이 되고, 오류 메시지의 `{1,0}` 이 바로 그 0 이다. 실측한 44 개 폭 전부가 이 식에 맞았다(`w=12 → T=1`, `w=13 → T=2`, `w=45 → T=6`).
그래서 상수를 **5** 로 잡았다. 16 을 쓰면 폭 5~15 구간을 새로 버리게 되는데, 그 구간은 지금 정상 동작하는 범위다. 다만 그 구간이 실제로 글자를 뱉는지도 따로 재 봤고 — 폭 40 대조군은 `"1"`(0.97) / `"3"`(0.999) / `"7"`(0.998) 을 제대로 읽는데 5~16 구간은 잉크가 있어도 전부 빈 문자열이었다 — 즉 **16 을 써도 잃는 건 사실상 없다.** 두 값의 실질 차이는 없고, 5 를 고른 이유는 그것이 그래프의 진짜 하한이라서 상수의 이름과 주석이 거짓이 되지 않기 때문이다.
`crop.width()` 가 아니라 리사이즈 **이후**의 `new_w` 를 검사한다. 3×200 짜리 크롭은 폭이 3 이지만 `new_w` 가 1 이고, 3×10 크롭은 같은 폭 3 인데 `new_w` 가 14 다. 네트워크가 보는 값은 `new_w` 뿐이다.
### 회귀 테스트는 상수를 위아래로 다 가둔다
아래쪽은 `rec_min_width_is_the_graph_floor` (`paddle_onnx.rs` 의 `mod tests`) 가 두 가지로 잡는다.
- `1..REC_MIN_WIDTH` 는 세션에 닿지 않고 빈 문자열로 빠져야 한다 → **가드를 지우면 실패**한다. 가드만 지우고 돌려 확인했다: `w=1 must never reach the rec session: rec session run: ... Invalid input shape: {1,0}`.
- 정확히 `REC_MIN_WIDTH` 는 **진짜 세션을 통과**해야 한다 → 상수가 모델의 실제 하한보다 낮으면(예: 모델 교체 후) 실패한다.
위쪽은 상수 옆의 `const _: () = assert!(REC_MIN_WIDTH <= 16, …)` 가 잡는다. **컴파일이 안 된다** — 100 으로 바꾸면 `error[E0080]: evaluation panicked: REC_MIN_WIDTH is past the measured no-loss ceiling (16)`.
위쪽 단언이 없으면 테스트가 한 방향으로만 샌다. 초안이 그랬다 — 상수를 100 으로 바꿔도 통과했다. 위의 두 단언은 상수가 커질수록 **더 잘** 통과하기 때문이다(가드가 더 많이 잡아 주고, 세션은 폭이 클수록 잘 돈다). 그 상태면 높이 48 기준 폭 100 미만 크롭, 즉 글자 한두 개짜리 박스가 전부 조용히 버려진다 — 이 항목이 없애려는 손실과 같은 종류다. 상한 16 은 위에서 잰 "5~16 은 잉크가 있어도 빈 문자열, 40 은 제대로 읽음" 에서 온 숫자이고, 테스트가 아니라 컴파일 시점에 건 이유는 상수를 만지는 사람이 테스트를 돌리기 전에 막히는 편이 낫기 때문이다.
테스트는 `ModelPaths::from_default_dir()` 대신 `CARGO_MANIFEST_DIR` 에서 경로를 직접 만든다. 전자는 `KEBAB_IMAGE_OCR_MODEL_DIR` 을 타므로, 자기 모델 디렉토리를 export 해 둔 사람이 `cargo test` 를 돌리면 상수를 엉뚱한 모델에 대고 재게 된다. 이 테스트의 일은 상수를 **배포되는** 모델에 붙들어 두는 것이다.
모델 에셋이 in-tree 로 커밋돼 있으므로(`git ls-files crates/kebab-parse-image/assets/`) skip 가드는 붙이지 않았다 — 조용히 안 도는 테스트가 이 이슈가 경고하는 바로 그 함정이다.
PDF 노트의 `{e:#}` 도 `ocr_engine_failure_surfaces_as_warning` 이 고정한다. 원래 이 테스트는 mock 이 단층 오류를 내서 `{}` 로 되돌려도 통과했다 — anyhow 는 원인이 없는 오류를 두 형식에서 똑같이 찍기 때문이다. mock 을 실제와 같은 두 층 오류로 바꾸고 안쪽 원인까지 단언하도록 했다. `{}` 로 되돌리면 실패한다.
### 실측 (도그푸딩 말뭉치 이미지 240 개 파일)
`corpus/images/` 전량(jpg 172 · png 63 · jpeg 4 · tif 1 = 240 개 파일. `kebab ingest` 가 집는 것은 tif 를 뺀 239 개지만, 여기서는 OCR 엔진을 파일 목록에 직접 물렸다)을 수정 전후로 같은 엔진(`ppocrv5-mobile-kor-1b55f062d055`)·같은 설정(score 0.3 / unclip 1.5 / max_boxes 1000 / max_pixels 2048)으로 돌렸다.
| | 수정 전 | 수정 후 |
|---|---|---|
| OCR 성공 | 204 / 240 | **240 / 240** |
| OCR 실패 | 36 (15.0%) | **0** |
실패 36 장은 charts 8 · english-text 10 · korean-text 9 · photos 9 로, 특정 종류에 몰려 있지 않았다. 되살아난 본문은 합계 11,747 자(중앙값 37 자, 최대 3,950 자)다. 한 장이 얼마나 크게 손해 보고 있었는지가 드러나는 예:
| 이미지 | 수정 후 |
|---|---|
| `charts/Bassano_Politi___1505___Questio_de_modalibus___diagrams.jpg` | 3,950 자 |
| `charts/Clickpath_Analysis.png` | 1,116 자 / 87 영역 |
| `korean-text/연고한2.jpg` | 1,088 자 |
`Clickpath_Analysis.png` 은 얇은 조각 하나 때문에 **이미 인식된 87 개 영역**을 통째로 잃고 있었다.
부작용이 없다는 것도 확인했다: 원래 성공하던 204 장의 인식 글자 수가 **한 장도 변하지 않았다**. 이 수정은 순수 가산이다.
36 장 중 4 장은 수정 후에도 0 자인데, 글자가 없는 사진(예: `photos/Aphid_2007_1.jpg`, 진딧물 접사)이라 정상이다. 이 이슈로 인한 손실과 "원래 글자가 없어서 비는 문서"를 혼동하면 안 된다 — 성공한 204 장 중에서도 31 장은 det 가 박스를 못 찾아 정상적으로 0 자다. KB 쪽에서 영향 문서를 셀 때는 본문 길이가 아니라 `provenance_json LIKE '%Invalid input shape%'` 로 걸러야 한다. 다만 이 쿼리에는 조건이 둘 붙는다.
**언제 세느냐.** 새 바이너리로 `kebab ingest` 를 돌리기 **전에** 세야 한다. 아래 §재색인 의 `parser_version` bump 가 해당 경로의 documents 행을 지우고 다시 쓰므로, 한 번 색인한 뒤에는 이 쿼리가 0 을 돌려준다. "업그레이드 → 색인 → 릴리스 노트 읽기" 순서로 가면 안 당한 것처럼 보인다. 이미 색인해 버렸다면 PDF 쪽은 `SELECT count(DISTINCT doc_id) FROM pdf_ocr_events WHERE success = 0 AND reason = 'ocr_error' AND ocr_engine = 'paddle-onnx'` 로 아직 셀 수 있다 — `pdf_ocr_events` 는 documents 에 FK 가 없어 purge 를 넘겨 살아남는다(`logging.retention_days` 기본 30 일 prune 만 받는다). `count(*)` 가 아니라 `count(DISTINCT doc_id)` 인 이유는 이 표가 **페이지마다** 한 행이고 유니크 제약도 없어서, 40 쪽을 잃은 문서 하나가 40 을 더하고 수정 전 색인을 두 번 돌렸으면 또 두 배가 되기 때문이다. `ocr_engine` 을 거는 이유는 `'ocr_error'` 가 `recognize()` 의 모든 실패를 받는 통칭이라, 기본 엔진인 ollama-vision 을 쓰는 KB 에서도 #239 와 무관한 행이 쌓이기 때문이다. 이렇게 걸러도 paddle-onnx 의 다른 실패까지 포함하는 상한이라는 점은 남는다.
**스캔 PDF 는 이 필터로 안 걸린다.** 이번 수정 이전에 나간 **모든** 릴리스에서 — v0.33.0 을 포함해서 — PDF 경로는 노트를 `err={}` 로 찍었고, anyhow 의 기본 Display 는 가장 바깥 context 하나(`rec session run`)만 내보낸다. 노출 구간은 이렇다 — PDF OCR 엔진으로 paddle-onnx 를 고를 수 있게 된 것이 **v0.28.0** 이므로 그때부터 스캔 PDF 도 같은 손실을 겪을 수 있었고, 다만 v0.32.0 까지는 단일 DCTDecode 이미지 페이지만 OCR 대상이었다. (태그별로 확인할 때 `crates/kebab-app/src/ingest.rs` 만 훑으면 v0.31.0 처럼 보인다. 그 파일이 v0.31.0 에서 생겼을 뿐이고, `build_pdf_ocr_engine` 은 v0.28.0~v0.30.1 에서 `crates/kebab-app/src/lib.rs` 에 있었다. 파일부터 찾은 뒤 세야 한다.) 페이지 렌더링(#232)이 임의 인코딩까지 넓힌 v0.33.0 에서 대상이 크게 늘었으니 기록의 대부분은 그 릴리스가 만든 것이겠지만, 전부는 아니다. 그래서 스캔본까지 세려면 `OR provenance_json LIKE '%err=rec session run%'` 을 반드시 함께 걸어야 한다. 이미지 경로는 처음부터 `{err:#}` 라 체인 전체가 들어갔고, 이번에 PDF 쪽도 `{e:#}` 로 맞췄으므로 앞으로 색인되는 것은 양쪽 다 첫 필터에 걸린다.
### 재색인: 버전 두 개를 올렸다
코드만 고치면 이미 색인된 문서에는 닿지 않는다. 이미지 문서의 실효 `parser_version` 은 `image-meta-v1|chunk:…|ocr:1:paddle-onnx:<모델 blake3>` 인데, 이미지 파일도 모델 에셋도 안 바뀌었으니 서명이 동일하고 `try_skip_unchanged` 가 Unchanged 로 건너뛴다. #232 에서 세운 원칙 그대로다 — 사용자가 `--force-reingest` 를 떠올려야만 고쳐지는 수정은 고쳐진 게 아니다.
- `image-meta-v1` → **`image-meta-v2`**
- `pdf-text-v2` → **`pdf-text-v3`** (스캔 PDF 도 같은 `run_rec` 을 타므로 같은 손실을 겪었다)
**비싼 OCR 은 대부분 다시 안 돈다.** OCR 산출물은 `derivation_cache` 에 **소스 바이트** 키로 들어가 있어서 `parser_version` 캐스케이드와 분리돼 있다(`docs/ARCHITECTURE.md:35` 의 derivation_cache 행, v0.31.0 #217). 그리고 실패한 OCR 은 캐시에 **저장되지 않는다** — `Err` 분기가 `derivation_cache_put` 앞에서 빠져나간다(이미지는 `ingest.rs` 의 `ingest_one_image_asset`, PDF 는 `pdf_ocr_apply.rs` 의 `apply_ocr_to_pdf_pages`). 줄 번호를 안 적은 건 이 항목이 처음 썼던 두 참조가 같은 PR 의 후속 커밋에 밀려 둘 다 어긋났기 때문이다. 그래서 이미 성공했던 문서는 캐시에 히트해 엔진 호출을 건너뛰고, **실제로 다시 OCR 되는 건 이 버그로 실패했던 문서뿐**이다.
**하지만 나머지 비용은 전부 든다.** `id_for_doc` 이 접는 것은 composite 가 아니라 **base** PARSER_VERSION 이므로(`kebab-parse-image/src/lib.rs`, `kebab-parse-pdf/src/lib.rs` 의 `extract`; composite 는 그 뒤에 `canonical.parser_version` 에만 찍힌다) 이 bump 는 **모든 이미지·PDF 문서의 doc_id 를 바꾼다**. 그러면 `ingest.rs` 의 `purge_workspace_path_for_parser_bump` 가 돌아 documents 행이 지워지고(blocks·chunks·embedding_records CASCADE) 해당 chunk_id 의 Lance 벡터도 전량 삭제된 뒤, 재파싱·재청킹·재임베딩·재삽입이 이어진다. 임베딩은 파생물 캐시에 히트하지만 행은 다시 쓴다. doc_id 는 wire 필수 필드이자 `kebab inspect doc <id>` 의 핸들이라, 파일을 하나도 안 고쳤는데 전부 한꺼번에 바뀐다.
그리고 이 비용은 **버그가 닿지 않는 KB 에도** 걸린다. 기본 OCR 엔진은 `ollama-vision` 이고 이미지 OCR 은 기본 off, `PdfOcrCfg::defaults()` 도 `enabled: false, always_on: false` 라, paddle-onnx 를 안 쓰는 KB 는 얻는 것 없이 이미지·PDF 를 다시 색인하게 된다.
**그래도 base 를 올리는 쪽을 택했다.** 대안은 `ingest_config_signature` 의 `ocr_engine_version_for_sig` 에 paddle-onnx 전용 revision 토큰을 넣어 영향 문서만 무효화하는 것이고, 그러면 doc_id 도 유지되고 ollama-vision·OCR off KB 도 안 건드린다. 더 정확하지만 #232 가 세운 선례와 다른 새 무효화 경로를 하나 더 만드는 일이고, 이 저장소는 단일 사용자용이라 실제로 손해 보는 KB 가 사용자 자신의 것 하나다. 캐스케이드 규칙이 이미 있는데 그 옆에 두 번째 규칙을 세우는 값을 치를 만큼은 아니라고 봤다. 다중 사용자 배포로 가면 다시 볼 결정이다.
### 스냅샷도 함께 움직였다
`pdf-text-v3` 는 `vector_pdf_canonical.json` 을 움직인다. #232 때와 같은 형태임을 확인했다 — 바뀐 것은 **파생 식별자와 버전 문자열뿐**이고 본문 텍스트·inlines·`source_span`·metadata 는 동일하다.
| 필드 | v2 | v3 |
|---|---|---|
| `doc_id` | `bd04fa01…` | `960e4c44…` |
| `block_id` | `964162ec…` | `b85b8a6d…` |
| `parser_version` | `pdf-text-v2` | `pdf-text-v3` |
| provenance note | `parser_version=pdf-text-v2; …` | `parser_version=pdf-text-v3; …` |
`kebab-parse-pdf/tests/extractor.rs` 와 `kebab-app/tests/pdf_pipeline.rs` 의 버전 단언 세 곳도 함께 옮겼다. 이미지 쪽은 스냅샷이 상수에서 값을 유도하고 있어 손댈 것이 없었다.
### 고치지 않고 남긴 것
`recognize` 의 박스 루프 안 `self.run_rec(&crop)?` 는 그대로 뒀다. 폭 가드가 알려진 유일한 방아쇠를 닫았고, `T = ceil((w-4)/8) >= 1` 이 `w >= 5` 에서 항상 성립하므로 폭 때문에 이 경로가 다시 터지는 일은 없다.
박스 단위로 살려 두려면 그 `?` 를 `continue` 로 바꿔야 하는데, 그러면 `run_rec` 안의 클래스 수 검사(`rec output has {c} classes`)까지 함께 삼킨다. 그건 입력과 무관한 **설정 오류**(rec 모델과 dict 짝이 안 맞음)라 즉시 실패하는 편이 맞다. 다만 이게 `?` 를 유지할 결정적 이유는 아니다 — 클래스 차원은 정적 그래프 메타데이터라(번들 rec 출력 shape 가 `[-1, -1, 11947]`) 세션 로드 시점에 읽을 수 있고, 그렇다면 그 검사는 `from_paths` 의 `dict.len() != DICT_LINES` bail 옆으로 옮기는 편이 더 낫다. 잘못된 모델이 박스가 검출되는 이미지를 기다릴 것 없이 엔진 생성에서 바로 터지기 때문이다. 이번에 안 한 건 #239 의 방아쇠와 무관한 별개 정리이기 때문이고, 옮길 때는 `continue` 에 `tracing::warn!` 을 반드시 함께 달아야 한다. 맨 `continue` 는 이 항목이 없애려는 조용한 손실을 다른 자리로 옮기는 것에 지나지 않는다.
### 이슈 본문과 다른 점 하나
이슈는 오류의 노드 이름을 `Conv.33` 으로 적었는데, 이 머신에서는 같은 실패가 `p2o.pd_op.batch_norm_.1.0_nchwc` 로 나온다. ORT 가 CPU 명령어 집합에 맞춰 그래프를 최적화(NCHWc 레이아웃 변환)하면서 노드 이름을 다시 붙이기 때문이고, 상태 메시지와 `{1,0}` 은 동일하다. 다른 버그가 아니다.
## 2026-08-17 — #232 PDF OCR 이 DCTDecode 아닌 스캔본을 전량 건너뜀 (페이지 렌더링)
### 무엇이 문제였나