diff --git a/crates/kebab-app/src/ingest.rs b/crates/kebab-app/src/ingest.rs index bb12d78..8c42740 100644 --- a/crates/kebab-app/src/ingest.rs +++ b/crates/kebab-app/src/ingest.rs @@ -1629,7 +1629,7 @@ fn ingest_one_image_asset( } }; // p9-fb-23 task 7: incremental-ingest early-skip for the image flow. - // Image docs use the `image-meta-v1` parser_version + the same + // Image docs use the `image-meta-v2` parser_version + the same // MdHeadingV2Chunker as the markdown flow (single-block doc). The // embedding-version check matches the markdown path: when the // active embedder's model_version equals what was stamped on the @@ -1676,7 +1676,7 @@ fn ingest_one_image_asset( .extract_for(&asset.media_type, &ctx, &bytes) .context("kb-app::extract_for (image)")?; // v0.26.2: store the composite parser_version (extractor baked the base - // `image-meta-v1`, which already fixed doc_id). Skip compare + stored + // `image-meta-v2`, which already fixed doc_id). Skip compare + stored // field must agree for next-run detection. canonical.parser_version = eff_parser_version.clone(); // `[[workspace.sources]]`: stamp the owning source id (image extractor @@ -2541,7 +2541,7 @@ fn ingest_one_pdf_asset( } }; // p9-fb-23 task 7: incremental-ingest early-skip for the PDF flow. - // PDF docs use `pdf-text-v2` as the parser_version and `PdfPageV1Chunker` + // PDF docs use `pdf-text-v3` as the parser_version and `PdfPageV1Chunker` // as the chunker — both pinned per-medium today (no config knob). // v0.26.2: composite parser_version folds pdf.ocr (enabled/always_on/ // model) + chunking, so enabling scanned-PDF OCR auto-re-indexes PDFs. @@ -2578,7 +2578,7 @@ fn ingest_one_pdf_asset( let mut canonical = app .extract_for(&asset.media_type, &ctx, &bytes) .context("kb-app::extract_for (pdf)")?; - // v0.26.2: store the composite parser_version (base `pdf-text-v2` already + // v0.26.2: store the composite parser_version (base `pdf-text-v3` already // fixed doc_id) so the next run's skip compare matches. canonical.parser_version = eff_parser_version.clone(); // `[[workspace.sources]]`: stamp the owning source id (pdf extractor diff --git a/crates/kebab-app/src/pdf_ocr_apply.rs b/crates/kebab-app/src/pdf_ocr_apply.rs index afaebeb..7b8f8eb 100644 --- a/crates/kebab-app/src/pdf_ocr_apply.rs +++ b/crates/kebab-app/src/pdf_ocr_apply.rs @@ -447,7 +447,10 @@ where Err(e) => { // OCR failure: warning event + skip (text-detect block 그대로). let note = format!( - "page={} OCR failed engine={} version={} err={}", + // `{e:#}` (not `{e}`): the ORT detail lives under a `.context`, and + // the plain Display drops it — so a note written with `{e}` reads + // `err=rec session run` and cannot be grepped for the real cause. + "page={} OCR failed engine={} version={} err={:#}", page_num, engine.engine_name(), engine.engine_version(), diff --git a/crates/kebab-app/tests/common/mock_ocr.rs b/crates/kebab-app/tests/common/mock_ocr.rs index b1fb330..f687271 100644 --- a/crates/kebab-app/tests/common/mock_ocr.rs +++ b/crates/kebab-app/tests/common/mock_ocr.rs @@ -1,6 +1,6 @@ use std::sync::Mutex; -use anyhow::Result; +use anyhow::{Context, Result}; use kebab_core::{Lang, OcrText}; use kebab_parse_image::OcrEngine; @@ -52,7 +52,12 @@ impl OcrEngine for MockOcrEngine { fn recognize(&self, _img: &[u8], _hint: Option<&Lang>) -> Result { if self.fail { - anyhow::bail!("mock failure"); + // Layered on purpose: the real paddle-onnx failure arrives as an + // ORT message under a `.context("rec session run")`, and anyhow's + // plain Display shows only the outer layer. A single-layer error + // would render identically under `{}` and `{:#}`, so it could not + // tell whether the provenance note keeps the cause (issue #239). + return Err(anyhow::anyhow!("mock inner cause")).context("mock failure"); } let mut idx = self.call_index.lock().unwrap(); let text = self diff --git a/crates/kebab-app/tests/pdf_ocr_apply.rs b/crates/kebab-app/tests/pdf_ocr_apply.rs index a6e6848..9342375 100644 --- a/crates/kebab-app/tests/pdf_ocr_apply.rs +++ b/crates/kebab-app/tests/pdf_ocr_apply.rs @@ -324,6 +324,18 @@ fn ocr_engine_failure_surfaces_as_warning() { warning_with_failure, "OCR failure 의 error message 가 warning event 의 note 안" ); + // issue #239: the note must carry the whole error chain, not just the + // outermost layer. The real cause (ORT's "Invalid input shape") sits under + // a `.context`, so a note formatted with `{e}` instead of `{e:#}` drops it + // and the KB can no longer be searched for which documents were hit. + let warning_with_cause = canonical.provenance.events.iter().any(|e| { + e.kind == kebab_core::ProvenanceKind::Warning + && e.note.as_deref().unwrap_or("").contains("mock inner cause") + }); + assert!( + warning_with_cause, + "provenance note 가 error chain 의 안쪽 원인까지 담아야 한다 (`{{e:#}}`)" + ); } // Test 9: dual-block ordinals are deterministic and unique diff --git a/crates/kebab-app/tests/pdf_pipeline.rs b/crates/kebab-app/tests/pdf_pipeline.rs index ad7ca67..b629a87 100644 --- a/crates/kebab-app/tests/pdf_pipeline.rs +++ b/crates/kebab-app/tests/pdf_pipeline.rs @@ -165,7 +165,7 @@ fn ingest_3_page_pdf_produces_one_doc_and_per_page_chunks() { pdf_item.parser_version .as_ref() .map(|p| p.0.split('|').next().unwrap()), - Some("pdf-text-v2") + Some("pdf-text-v3") ); assert_eq!( pdf_item.chunker_version.as_ref().map(|c| c.0.as_str()), @@ -479,10 +479,10 @@ fn inspect_doc_surfaces_page_spans() { .find(|i| i.doc_path.0.ends_with("inspect.pdf")) .unwrap(); let doc = kebab_app::inspect_doc_with_config(cfg, pdf_item.doc_id.as_ref().unwrap()).unwrap(); - // v0.26.2: stored parser_version is now `pdf-text-v2|` + // v0.26.2: stored parser_version is now `pdf-text-v3|` // (the signature folds chunking / pdf.ocr settings for skip detection). // Assert the base identity by taking the prefix before the first '|'. - assert_eq!(doc.parser_version.0.split('|').next().unwrap(), "pdf-text-v2"); + assert_eq!(doc.parser_version.0.split('|').next().unwrap(), "pdf-text-v3"); assert_eq!(doc.blocks.len(), 3); for block in &doc.blocks { match block { diff --git a/crates/kebab-chunk/src/pdf_page_v1.rs b/crates/kebab-chunk/src/pdf_page_v1.rs index 601c789..7d6b28a 100644 --- a/crates/kebab-chunk/src/pdf_page_v1.rs +++ b/crates/kebab-chunk/src/pdf_page_v1.rs @@ -400,7 +400,10 @@ mod tests { fn make_pdf_doc(pages: &[&str]) -> CanonicalDocument { let workspace_path = WorkspacePath::new("docs/test.pdf".into()).unwrap(); let asset_id = AssetId("a".repeat(64)); - let parser_version = ParserVersion("pdf-text-v2".into()); + // Version-neutral on purpose: `kebab-chunk` cannot import + // `kebab-parse-pdf` (design §8), so a real version string here would + // go stale at every bump. The chunker only feeds it to `id_for_doc`. + let parser_version = ParserVersion("test-parser-v1".into()); let doc_id = id_for_doc(&workspace_path, &asset_id, &parser_version); let mut blocks: Vec = Vec::new(); diff --git a/crates/kebab-parse-image/src/lib.rs b/crates/kebab-parse-image/src/lib.rs index b310a18..8b91eb1 100644 --- a/crates/kebab-parse-image/src/lib.rs +++ b/crates/kebab-parse-image/src/lib.rs @@ -49,7 +49,19 @@ use serde_json::{Map, Value}; use time::OffsetDateTime; /// Parser version label for the image extractor (§9 versioning). -pub const PARSER_VERSION: &str = "image-meta-v1"; +/// +/// Bumped to v2 for issue #239 (2026-08-28): a detection box narrower than +/// the rec graph's floor used to abort the ONNX session and discard OCR for +/// the *whole* image, so an affected image was indexed with its filename and +/// nothing else. `REC_MIN_WIDTH` in `paddle_onnx` fixes the extraction; this +/// bump is what makes the fix reach stores that were already indexed. +/// `try_skip_unchanged` calls an asset Unchanged when its content hash *and* +/// version inputs match — the image file on disk did not change, and neither +/// did the OCR model assets that feed `ingest_config_signature`, so without +/// this every already-indexed image would keep its empty extraction until the +/// user thought to pass `--force-reingest`. Per CLAUDE.md §Versioning +/// cascade, changing this invalidates downstream image records. +pub const PARSER_VERSION: &str = "image-meta-v2"; /// Maximum decode dimension (per axis) before we refuse to read the image. /// Matches the §9.1 "cap decode at ~16k" policy in the design doc. diff --git a/crates/kebab-parse-image/src/paddle_onnx.rs b/crates/kebab-parse-image/src/paddle_onnx.rs index 86ab114..9202703 100644 --- a/crates/kebab-parse-image/src/paddle_onnx.rs +++ b/crates/kebab-parse-image/src/paddle_onnx.rs @@ -51,6 +51,28 @@ const REC_CLASSES: usize = 11947; const DET_LIMIT_SIDE_LEN: u32 = 960; /// rec input height (PP-OCRv5 mobile). const REC_HEIGHT: u32 = 48; +/// Narrowest rec input the graph survives. The backbone emits +/// `T = ceil((w - 4) / 8)` CTC timesteps, so `w <= 4` collapses the feature +/// map to zero columns and ORT aborts the whole session run with +/// `Invalid input shape: {1,0}` — taking every already-recognized box on the +/// image down with it (issue #239). Measured against the bundled +/// `korean_ppocrv5_mobile_rec.onnx`: 1..=4 always fail, 5.. always succeed. +/// `rec_min_width_is_the_graph_floor` holds the constant from below — set it +/// under the graph's floor and that test errors. The upward direction cannot +/// be a test (both of its assertions pass *better* as the constant grows), so +/// it is the `const _` ceiling right below instead. +const REC_MIN_WIDTH: u32 = 5; +/// Raising `REC_MIN_WIDTH` is only known to be free while it stays inside the +/// band that was actually measured: widths 5..=16 come back empty even with +/// real ink in them, while width 40 reads glyphs at 0.97+ confidence (issue +/// #239). 17..=39 was never swept, so a ceiling above 16 is not measurement +/// any more — and by 40 the guard would be discarding crops the graph reads, +/// which is the silent loss this constant exists to prevent. Enforced at +/// compile time rather than left to review. +const _: () = assert!( + REC_MIN_WIDTH <= 16, + "REC_MIN_WIDTH is past the measured no-loss ceiling (16)" +); /// DBNet probability-map binarization threshold. Looser than Paddle's default /// `box_thresh` (0.6) to keep recall high on low-contrast Korean text. const DET_BIN_THRESH: f32 = 0.3; @@ -356,6 +378,13 @@ impl OnnxPaddleOcr { // resize keep-aspect to height 48, then this single crop is its own batch let (cw, ch) = (crop.width().max(1), crop.height().max(1)); let new_w = ((REC_HEIGHT as f32 / ch as f32) * cw as f32).round().max(1.0) as u32; + if new_w < REC_MIN_WIDTH { + // A sliver this thin holds no glyph, and feeding it to the graph + // would abort recognition for the entire image. Report it the way + // an undecodable box is already reported: the caller's + // `text.is_empty()` arm drops this box and keeps the rest. + return Ok((String::new(), 0.0)); + } let resized = image::imageops::resize( crop, new_w, @@ -1002,4 +1031,57 @@ mod tests { ); } } + + /// Issue #239: a detection box narrower than the rec graph's floor used to + /// abort the ORT session, and `recognize`'s `?` turned that into "this + /// image has no OCR text at all" — 36 of 240 corpus images. + /// + /// Both runtime directions are asserted here. Below + /// `REC_MIN_WIDTH` the guard must short-circuit (delete it and this + /// errors). At exactly `REC_MIN_WIDTH` the real session must run, which is + /// what keeps the constant pinned to the shipped model rather than being a + /// number nobody rechecks (set it below the graph's floor, e.g. after a + /// model swap, and this errors). The upward direction — a constant raised + /// past the width band that was measured to decode nothing anyway, which + /// would silently drop crops the graph can read — is pinned by the + /// `const _` assertion next to `REC_MIN_WIDTH` itself. + #[test] + fn rec_min_width_is_the_graph_floor() { + // The upward direction is pinned by the `const _` next to + // REC_MIN_WIDTH — a raise past the measured ceiling fails the build, + // not just this test. + // + // Pin to the *bundled* assets: `ModelPaths::from_default_dir` honors + // `KEBAB_IMAGE_OCR_MODEL_DIR`, and a developer with that exported would + // otherwise measure their own model here. + let dir = Path::new(env!("CARGO_MANIFEST_DIR")).join("assets/paddleocr-onnx"); + let engine = OnnxPaddleOcr::from_paths( + &ModelPaths { + det: dir.join("ppocrv5_mobile_det.onnx"), + rec: dir.join("korean_ppocrv5_mobile_rec.onnx"), + dict: dir.join("korean_dict.txt"), + }, + 0.3, + 1.5, + 1000, + 1600, + ) + .expect("bundled OCR assets must load"); + // A crop already at REC_HEIGHT makes run_rec's keep-aspect resize the + // identity, so the crop width *is* the rec input width — no rounding + // to reason about. + let crop = |w| image::RgbImage::from_pixel(w, REC_HEIGHT, image::Rgb([255, 255, 255])); + + for w in 1..REC_MIN_WIDTH { + let (text, conf) = engine + .run_rec(&crop(w)) + .unwrap_or_else(|e| panic!("w={w} must never reach the rec session: {e:#}")); + assert!(text.is_empty(), "w={w} decoded {text:?}"); + assert_eq!(conf, 0.0, "w={w}"); + } + + engine.run_rec(&crop(REC_MIN_WIDTH)).unwrap_or_else(|e| { + panic!("REC_MIN_WIDTH={REC_MIN_WIDTH} must survive the rec session: {e:#}") + }); + } } diff --git a/crates/kebab-parse-pdf/src/lib.rs b/crates/kebab-parse-pdf/src/lib.rs index 986897a..b18cac5 100644 --- a/crates/kebab-parse-pdf/src/lib.rs +++ b/crates/kebab-parse-pdf/src/lib.rs @@ -36,6 +36,13 @@ use kebab_core::{ use serde_json::{Map, Value}; use time::OffsetDateTime; +/// Bumped to v3 for issue #239 (2026-08-28): scanned pages share the +/// paddle-onnx rec path with image OCR, so a page whose raster produced one +/// over-thin detection box lost that page's OCR text entirely. The fix lives +/// in `kebab-parse-image`; this bump is what re-processes scans that were +/// already indexed under v2 (same reasoning as the v2 bump below — the PDF +/// bytes have not changed, so nothing else would invalidate them). +/// /// Bumped to v2 for issue #232 (2026-08-17): scanned pages are now /// rasterized by rendering rather than by pulling out an embedded JPEG, /// so pages encoded with CCITTFax / JBIG2 / Flate / JPX — previously @@ -47,7 +54,7 @@ use time::OffsetDateTime; /// every already-indexed scan would keep its empty extraction until the /// user thought to pass `--force-reingest`. Per CLAUDE.md §Versioning /// cascade, changing this invalidates downstream PDF records. -pub const PARSER_VERSION: &str = "pdf-text-v2"; +pub const PARSER_VERSION: &str = "pdf-text-v3"; /// Text-PDF extractor. Per-page text via `lopdf::Document::extract_text` /// (the only stable per-page API in the lopdf / pdf-extract pair — diff --git a/crates/kebab-parse-pdf/tests/extractor.rs b/crates/kebab-parse-pdf/tests/extractor.rs index 1645533..8dfc26a 100644 --- a/crates/kebab-parse-pdf/tests/extractor.rs +++ b/crates/kebab-parse-pdf/tests/extractor.rs @@ -267,7 +267,7 @@ fn snapshot_three_page_canonical_document_stable() { // golden file (the full JSON contains BLAKE3 ids that would // change if `id_from(...)`'s tuple shape ever shifts — that would // be a separate, intentional break). - assert_eq!(json["parser_version"], Value::String("pdf-text-v2".into())); + assert_eq!(json["parser_version"], Value::String("pdf-text-v3".into())); assert_eq!(json["lang"], Value::String("und".into())); assert_eq!(json["schema_version"], Value::Number(1.into())); assert_eq!(json["doc_version"], Value::Number(1.into())); diff --git a/crates/kebab-parse-pdf/tests/snapshots/vector_pdf_canonical.json b/crates/kebab-parse-pdf/tests/snapshots/vector_pdf_canonical.json index e970c5d..a9bb9a3 100644 --- a/crates/kebab-parse-pdf/tests/snapshots/vector_pdf_canonical.json +++ b/crates/kebab-parse-pdf/tests/snapshots/vector_pdf_canonical.json @@ -1,5 +1,5 @@ { - "doc_id": "bd04fa013899592afcf671013404ed68", + "doc_id": "960e4c445f1365a633c9647b42ed27ba", "source_asset_id": "babe9824b6b28237c0898575a40ba48d", "workspace_path": "mojibake.pdf", "title": "untitled", @@ -8,7 +8,7 @@ { "kind": "paragraph", "common": { - "block_id": "964162ec3cf0c191849f5a5cf8a3b675", + "block_id": "b85b8a6d2462bb396112f45cfcea6c44", "heading_path": [], "source_span": { "kind": "page", @@ -54,11 +54,11 @@ "at": "1970-01-01T00:00:00Z", "agent": "kb-parse-pdf", "kind": "parsed", - "note": "parser_version=pdf-text-v2; page_count=1" + "note": "parser_version=pdf-text-v3; page_count=1" } ] }, - "parser_version": "pdf-text-v2", + "parser_version": "pdf-text-v3", "schema_version": 1, "doc_version": 1, "last_chunker_version": null, diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 3c6b644..12c2380 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -20,11 +20,11 @@ Cargo workspace, 함수 호출 기반 모듈러 모놀리스. UI binary (`kebab- | 한국어 형태소분석 | `lindera-ko-dic` (FTS5 외부 tokenizer, v0.20.1) — 2자 이상 한국어 query 지원 | | LLM | Ollama HTTP (default `gemma4:e4b` ─ OCR / caption 와 family 통일. 사용자가 더 큰 variant `gemma4:26b` 등으로 override 가능) | | 음성 ASR | `whisper.cpp` (via `whisper-rs`) — P8 보류, 시스템 dep brainstorm 후 | -| OCR (image) | `OcrEngine` trait, 2 백엔드: **`ollama-vision`** (default, `gemma4:e4b`) / **`paddle-onnx`** (v0.27.0 — PP-OCRv5 ONNX in-process via `ort` =2.0.0-rc.9, DBNet det + CTC rec, 후처리 min-area rect/unclip pure-Rust, Python 런타임 0). engine 선택은 `[image.ocr] engine`, 팩토리는 `kebab-app::build_image_ocr_engine`. e2e CER 0.005 / 큰 페이지 <4초. (HOTFIXES P6-2, 2026-06-04) | +| OCR (image) | `OcrEngine` trait, 2 백엔드: **`ollama-vision`** (default, `gemma4:e4b`) / **`paddle-onnx`** (v0.27.0 — PP-OCRv5 ONNX in-process via `ort` =2.0.0-rc.9, DBNet det + CTC rec, 후처리 min-area rect/unclip pure-Rust, Python 런타임 0). engine 선택은 `[ingest.image.ocr] engine`, 팩토리는 `kebab-app::build_image_ocr_engine`. e2e CER 0.005 / 큰 페이지 <4초. (HOTFIXES P6-2, 2026-06-04) **불변식**: rec 세션 입력 폭은 `REC_MIN_WIDTH = 5` 이상이어야 한다 — 백본이 `T = ceil((w-4)/8)` 개의 CTC 타임스텝을 내므로 `w <= 4` 는 특징맵을 0 열로 접어 ORT 가 세션 전체를 실패시킨다. 그 실패는 `recognize` 의 박스 루프를 뚫고 나가 이미 인식한 박스까지 전부 버리므로, 얇은 크롭은 세션에 넣지 말고 빈 문자열로 돌려보내야 한다 (issue #239). `parser_version = "image-meta-v2"` (issue #239 에서 v1 → v2, 기존 색인 이미지 재처리 유발). | | OCR (PDF, v0.20.0+) | Ollama vision LM (default `qwen2.5vl:3b`) — post-extract enrichment via `kebab-app::pdf_ocr_apply` (H-1 resolution). DCTDecode-only v1 (FlateDecode/CCITTFax skip + warning). family asymmetry vs image OCR: PoC alnum 94.79% (qwen2.5vl) >> 27% (gemma4:e4b 받침), 본 단계에서 PDF OCR 만 qwen2.5vl. | | Image caption | Ollama vision LM, runtime gate `image.caption.enabled` (default OFF) | | RAG groundedness 검증 | `kebab-nli` 의 mDeBERTa-v3 XNLI 가 `(packed_chunks, generated_answer)` entailment 검사 (fb-41). `[rag] nli_threshold > 0` (default 0 = disabled, production 권장 0.5) 일 때 활성 — 미달 시 `refusal_reason = nli_verification_failed` (LLM self-judge ceiling 보완). 첫 호출 시 ~280 MB ONNX 자동 다운로드 | -| PDF parser | `lopdf` per-page 텍스트 + 스캔 페이지 래스터화. 래스터는 **pdfium 페이지 렌더링**(`page_render::PageRenderer`, issue #232) 이 1순위 — 필터·XObject 구성과 무관하게 페이지를 그린다. pdfium 이 없으면 `page_image::extract_dctdecode_page_image` 로 떨어지며 그 경우 단일 DCTDecode 이미지 페이지만 OCR 된다. pdfium 은 공유 라이브러리로만 배포돼 링크하면 단일 바이너리 원칙이 깨지므로 **런타임 바인딩**이고, `[ingest.pdf.ocr] render_library` 로 경로를 지정하거나 로더 경로에 두면 된다. `kebab doctor` 의 `pdf_render` 가 어느 쪽인지 보고한다. `chunker_version = "pdf-page-v1"` 하드코딩 (HOTFIXES P7-3). `parser_version = "pdf-text-v2"` (issue #232 에서 v1 → v2, 기존 색인 스캔본 재처리 유발). | +| PDF parser | `lopdf` per-page 텍스트 + 스캔 페이지 래스터화. 래스터는 **pdfium 페이지 렌더링**(`page_render::PageRenderer`, issue #232) 이 1순위 — 필터·XObject 구성과 무관하게 페이지를 그린다. pdfium 이 없으면 `page_image::extract_dctdecode_page_image` 로 떨어지며 그 경우 단일 DCTDecode 이미지 페이지만 OCR 된다. pdfium 은 공유 라이브러리로만 배포돼 링크하면 단일 바이너리 원칙이 깨지므로 **런타임 바인딩**이고, `[ingest.pdf.ocr] render_library` 로 경로를 지정하거나 로더 경로에 두면 된다. `kebab doctor` 의 `pdf_render` 가 어느 쪽인지 보고한다. `chunker_version = "pdf-page-v1"` 하드코딩 (HOTFIXES P7-3). `parser_version = "pdf-text-v3"` (issue #232 에서 v1 → v2, issue #239 에서 v2 → v3). 고친 것은 스캔본 경로지만 **재처리 대상은 기존 색인 PDF 전부** 다 — base `parser_version` 이 `id_for_doc` 에 접히므로 doc_id 가 바뀌고 store 가 다시 쓰인다 (HOTFIXES 2026-08-28). | | code parser | `tree-sitter` + `tree-sitter-rust` / `tree-sitter-python` / `tree-sitter-typescript` / `tree-sitter-javascript` / `tree-sitter-go` / `tree-sitter-java` / `tree-sitter-kotlin-ng` — **parser-side** (`kebab-parse-code`), chunker-side 아님 (design §6.3). chunker versions: Rust = `code-rust-ast-v1`, Python = `code-python-ast-v1`, TypeScript = `code-ts-ast-v1`, JavaScript = `code-js-ast-v1`, Go = `code-go-ast-v1`, Java = `code-java-ast-v1`, Kotlin = `code-kotlin-ast-v1`. (v0.32.0 #220: 9개 언어 chunker 가 단일 `CodeAstV1Chunker` 로 통합 — `for_lang(lang)` 가 per-lang `chunker_version` 라벨을 verbatim 매핑. chunker 는 tree-sitter 미사용·`lang` 은 `SourceSpan::Code` 데이터에서 흐르므로 9개 struct 차이는 `VERSION_LABEL` 문자열뿐이었음 → chunk_id byte-identical.) `ast_chunk_max_lines = 200` 상수 고정 (HOTFIXES 2026-05-19 — Chunker trait 이 per-medium config 미노출). Kotlin grammar 은 `tree-sitter-kotlin-ng` 사용 — bare `tree-sitter-kotlin` 은 tree-sitter 0.21–0.23 에 고착되어 있어 사용 불가. **Tier 2 (p10-2)**: YAML/k8s → `serde_yaml_ng` + `k8s-manifest-resource-v1` (apiVersion+kind per resource), Dockerfile → `dockerfile-file-v1` (whole-file), Cargo.toml/go.mod/.json/.xml/.groovy → `manifest-file-v1` (whole-file). Tier 2 chunkers live in `kebab-chunk`; no tree-sitter grammar needed (structure from file type, not AST). **Tier 3 (p10-3)**: shell scripts (`.sh`/`.bash`/`.zsh`) direct → `code-text-paragraph-v1` (blank-line paragraph segmentation + 80-line / 20-overlap line-window for oversize). Same chunker also serves as fallback when Tier 1/2 emit 0 chunks or Err — non-k8s YAML / invalid YAML / AST extractor failures all picked up. symbol = None; lang preserved from input doc. **Tier 1 family complete (p10-1D)**: C (`tree-sitter-c`, `code-c-ast-v1`, `.c`/`.h`) + C++ (`tree-sitter-cpp`, `code-cpp-ast-v1`, `.cpp`/`.cc`/`.cxx`/`.hpp`/`.hh`/`.hxx`). C symbol = function name only; C++ symbol = `namespace::Class::method` (recursive nesting). `.h` 가 C++ syntax 만나면 tree-sitter-c parse 실패 → Tier 3 fallback. | | symbol path 형식 | workspace path → module path: Python = dotted prefix (`kebab_eval.metrics.compute_mrr`), TypeScript/JavaScript = slash-style prefix (`src/Foo.Foo.search`), Go = `package.Func` / `package.(*Receiver).Method`, Java/Kotlin = `com.foo.Foo.bar` (패키지+클래스+메서드/필드), C = 함수명, C++ = `namespace::Class::method`. Rust 1A-2 는 file-scope nesting 만 (workspace prefix 없음, 비일관 수용 — HOTFIXES 2026-05-20). code chunk 은 `citation.kind = "code"` + `citation.lang` + `symbol` + line range, SearchHit 에 `code_lang` + `repo`(`.git` walk-up 디렉토리명) backfill. | | Desktop | Tauri 2 + `pdfjs-dist` (native PDF render backend 금지) — P9-5 | diff --git a/docs/DOGFOOD.md b/docs/DOGFOOD.md index 61ea286..55b4ad6 100644 --- a/docs/DOGFOOD.md +++ b/docs/DOGFOOD.md @@ -122,10 +122,18 @@ endpoint = "http://192.168.0.47:11434" enabled = false # opt-in ``` +**Config (v0.28.0~)**: 위 블록은 `[ingest.image.ocr]` / `[ingest.image.caption]` 로 옮겨졌다. 옛 키를 그대로 쓰면 **`schema_version` 이 5 보다 낮은 파일에서만** 로드 시 자동 이관된다 — 이미 `schema_version = 5` 인 config 에 `[image.ocr]` 를 붙여 넣으면 경고 없이 통째로 무시되고 OCR 이 꺼진 채로 돈다. `paddle-onnx` 백엔드 (v0.27.0~, in-process ONNX — Ollama 없이 돈다): +```toml +[ingest.image.ocr] +enabled = true +engine = "paddle-onnx" +``` + **verify**: - `*.png` / `*.jpg` / `*.jpeg` 만 ingest target. - OCR text 가 `Block::ImageRef.ocr.joined` 안. -- `[image.caption].enabled=true` 시 caption 도. +- caption 토글을 켜면 caption 도. +- (issue #239) 한 장도 **통째로** 비지 않는다. 얇은 검출 박스 하나가 rec 세션을 실패시키면 그 이미지의 인식 결과가 전량 버려지던 버그였다. 색인은 성공으로 끝나므로 `documents.provenance_json` 에 `Invalid input shape` 이 남았는지로 확인한다. (PDF 경로는 노트 형식이 달라 문자열이 다르다 — HOTFIXES 2026-08-28 참고.) **scenarios**: - 1.2.a Korean OCR (한국어 scan PNG) → OCR text + search hit. @@ -133,6 +141,7 @@ enabled = false # opt-in - 1.2.c photo (자연 사진, OCR 없음) → empty OCR or warning. - 1.2.d corrupt image → graceful error. - 1.2.e oversized image (> max_pixels) → downscale. +- 1.2.f (issue #239) `[ingest.image.ocr] engine = "paddle-onnx"` 로 §13.4 이미지 코퍼스 전량 → OCR 오류 0 건. 먼저 진행 출력에 `ocr(ppocrv5-mobile-kor…)` 단계가 실제로 찍히는지 본다 — 안 찍히면 OCR 이 꺼진 것이고, 그 상태에서는 모든 이미지가 본문 0 자로 나와 시나리오가 통과한 것처럼 보인다. 본문이 파일명뿐인 문서가 남으면 provenance 를 확인한다. 글자가 없는 사진이라 0 자인 것과 이 버그로 통째로 버려진 것은 다르다 — 후자만 provenance 에 오류가 남는다. ### §1.3 PDF text ingest (P7-1) @@ -141,7 +150,7 @@ enabled = false # opt-in - 1 `Block::Paragraph` per page (P7-1 invariant). **verify**: -- `parser_version = "pdf-text-v2"`. +- `parser_version = "pdf-text-v3"`. - `chunker_version = "pdf-page-v1"` (또는 `"pdf-page-v1.1"` from v0.20.1). - `block_count` ≥ page count. @@ -975,7 +984,16 @@ bug 발견 시: ### §13.4 Image corpus -(P6 dogfood — 향후 추가). +도그푸딩 스토어의 `corpus/images/` — 4 분류 240 개 파일 (jpg 172 · png 63 · jpeg 4 · tif 1). `kebab ingest` 가 집는 것은 tif 를 뺀 239 개다. + +| 분류 | 장수 | 쓰임 | +|---|---|---| +| `charts/` | 50 | 도표·다이어그램. 박스가 많고 얇은 조각이 잘 생긴다. #239 실패 8 장. | +| `english-text/` | 70 | 스크린샷·표지판 등 영문 위주. 1.2.b 판정용. #239 실패 10 장. | +| `korean-text/` | 70 | 한국어 스캔·필기. 1.2.a. #239 실패 9 장. | +| `photos/` | 50 | 글자가 거의 없는 자연 사진. 1.2.c 의 "정상적으로 0 자" 대조군. #239 실패 9 장 — 글자가 없어도 얇은 박스는 검출되므로 사진도 걸렸다. | + +issue #239 실측 근거가 이 코퍼스 전량 스윕이다 (수정 전 36/240 실패 → 수정 후 0). 상세는 tasks/HOTFIXES.md 2026-08-28 항목. --- diff --git a/docs/SMOKE.md b/docs/SMOKE.md index 778402a..7b58b15 100644 --- a/docs/SMOKE.md +++ b/docs/SMOKE.md @@ -713,7 +713,7 @@ KB --json schema | jq '.stats.code_lang_breakdown' - 코퍼스에 없는 주제로 `kebab ask` → `refusal_reason: "llm_self_judge"` (또는 `no_chunks` / `score_gate`) + `grounded: false`. - (P6-4) `image.ocr.enabled = true` 로 PNG 자산을 ingest 하면 `kebab list docs` 가 markdown 옆에 image doc 도 출력 (`workspace_path` 가 `*.png`). `kebab inspect doc ` 의 `block.ocr.joined` 가 vision LM 의 OCR 결과 (예: 스크린샷 안의 텍스트). `kebab search --mode lexical ""` 가 그 image chunk 를 반환하면 wiring 정상. - OCR / caption 부분 실패는 `errors` 카운터 미증가 — `kebab inspect doc ` 의 Provenance Warning 이벤트 또는 `--debug` 로그에서만 확인. -- (P7-3) `*.pdf` 자산을 워크스페이스에 두면 `kebab ingest` 출력에 PDF 도 `new` 카운터에 포함. `kebab inspect doc ` 가 `parser_version = "pdf-text-v2"` + 페이지마다 `Block::Paragraph` + `SourceSpan::Page { page, char_start, char_end }`. 본문에 등장하는 단어로 `kebab search --mode hybrid` 시 PDF chunk 가 결과에 포함되고 `source_span.kind = "page"` 면 wiring 정상. 암호화 PDF 는 `errors+=1` 로 분류되며 `error` 필드에 `qpdf --decrypt` 안내 보존. 빈/스캔 페이지 (PDF 가 텍스트를 추출하지 못한 페이지) 는 0 chunk + `Provenance::Warning` ("scanned candidate") 로 표시 — P+ scanned-PDF OCR fallback 까지는 검색 불가. +- (P7-3) `*.pdf` 자산을 워크스페이스에 두면 `kebab ingest` 출력에 PDF 도 `new` 카운터에 포함. `kebab inspect doc ` 가 `parser_version = "pdf-text-v3"` + 페이지마다 `Block::Paragraph` + `SourceSpan::Page { page, char_start, char_end }`. 본문에 등장하는 단어로 `kebab search --mode hybrid` 시 PDF chunk 가 결과에 포함되고 `source_span.kind = "page"` 면 wiring 정상. 암호화 PDF 는 `errors+=1` 로 분류되며 `error` 필드에 `qpdf --decrypt` 안내 보존. 빈/스캔 페이지 (PDF 가 텍스트를 추출하지 못한 페이지) 는 0 chunk + `Provenance::Warning` ("scanned candidate") 로 표시 — P+ scanned-PDF OCR fallback 까지는 검색 불가. ## config migrate (마이그레이션) diff --git a/docs/components/parse/README.md b/docs/components/parse/README.md index 67d47f0..7ba891a 100644 --- a/docs/components/parse/README.md +++ b/docs/components/parse/README.md @@ -26,22 +26,27 @@ classDiagram parse_blocks(body) (Vec~ParsedBlock~, Warnings) } class PdfTextExtractor { - PARSER_VERSION = "pdf-text-v2" + PARSER_VERSION = "pdf-text-v3" new() Self } class ImageExtractor { - PARSER_VERSION = "image-meta-v1" + PARSER_VERSION = "image-meta-v2" MAX_DECODE_DIM = 16384 new() Self } class OcrEngine { <> - engine_id() str - run(image_bytes, langs) OcrText + engine_name() str + engine_version() String + recognize(image_bytes, lang_hint) Result~OcrText~ } class OllamaVisionOcr { endpoint, model, max_pixels } + class OnnxPaddleOcr { + REC_HEIGHT = 48 + REC_MIN_WIDTH = 5 + } class CaptionFns { caption_image(lm, prep, opts) ModelCaption apply_caption(block, lm, opts) @@ -49,6 +54,7 @@ classDiagram Extractor <|.. PdfTextExtractor Extractor <|.. ImageExtractor OcrEngine <|.. OllamaVisionOcr + OcrEngine <|.. OnnxPaddleOcr ImageExtractor ..> OcrEngine : applied via apply_ocr ImageExtractor ..> CaptionFns : applied via apply_caption ``` @@ -101,13 +107,15 @@ flowchart LR **PDF** (`kebab-parse-pdf`): - `PdfTextExtractor` — `Extractor` 구현체. `lopdf::Document::load_mem` 로 한 번 파싱, encrypted 면 즉시 bail. -- `PARSER_VERSION = "pdf-text-v2"` — version cascade entry (issue #232 에서 v1 → v2, 페이지 렌더링 도입으로 기존 색인 스캔본 재처리 유발). (HOTFIXES P7-2 의 chunker_version `pdf-page-v1` 와 별개.) +- `PARSER_VERSION = "pdf-text-v3"` — version cascade entry (issue #232 에서 v1 → v2, 페이지 렌더링 도입; issue #239 에서 v2 → v3, 얇은 검출 박스가 페이지 OCR 을 통째로 날리던 것을 고치면서). 고친 것은 둘 다 스캔본 경로지만 **재처리 대상은 기존 색인 PDF 전부** — base 가 `id_for_doc` 에 접혀 doc_id 가 바뀐다. (HOTFIXES P7-2 의 chunker_version `pdf-page-v1` 와 별개.) - 빈 페이지 / extract 실패 → `Block::Paragraph` 빈 inlines + `ProvenanceKind::Warning("scanned candidate")`. OCR fallback 미구현. **Image** (`kebab-parse-image`): - `ImageExtractor` — `Extractor` 구현체. `MAX_DECODE_DIM = 16384` 초과 거부 (decode bomb 방어). -- `OcrEngine` (trait) — `engine_id() / run(...) -> OcrText`. `OcrText.engine` 필드로 trust level 분기. -- `OllamaVisionOcr { endpoint, model, max_pixels }` — v1 유일 구현. `apply_ocr(block, engine, langs)` 가 `ImageRefBlock.ocr` 슬롯 채움. +- `PARSER_VERSION = "image-meta-v2"` — version cascade entry (issue #239 에서 v1 → v2, 얇은 검출 박스가 이미지 OCR 을 통째로 날리던 것을 고치면서). PDF 와 마찬가지로 기존 색인 이미지 **전부** 가 재처리 대상이다. +- `OcrEngine` (trait) — `engine_name() -> &'static str` / `engine_version() -> String` / `model() -> &str` / `recognize(&[u8], Option<&Lang>) -> Result`. `OcrText.engine` 필드로 trust level 분기. +- `OllamaVisionOcr { endpoint, model, max_pixels }` — `ollama-vision` 백엔드 (기본값). `apply_ocr(block, engine, langs)` 가 `ImageRefBlock.ocr` 슬롯 채움. +- `OnnxPaddleOcr` — `paddle-onnx` 백엔드 (v0.27.0, PP-OCRv5 ONNX in-process). rec 세션 입력 폭 하한은 `REC_MIN_WIDTH = 5` — 그 아래는 세션에 넣지 않고 빈 문자열을 돌려준다 (issue #239). - `caption_image(lm: &dyn LanguageModel, prep, opts) -> Result` — `LanguageModel.generate_stream` 의 vision 입력 (`GenerateRequest.images`) 사용. `apply_caption` 이 block 에 in-place 주입. ## 외부 의존 diff --git a/tasks/HOTFIXES.md b/tasks/HOTFIXES.md index 236accf..d9320ca 100644 --- a/tasks/HOTFIXES.md +++ b/tasks/HOTFIXES.md @@ -14,6 +14,113 @@ historical contract that was implemented; this file accumulates the deltas so phase 5+ readers can find the live behavior without diffing git history. +## 2026-08-28 — #239 얇은 검출 박스 하나가 이미지 OCR 전체를 날림 (paddle-onnx rec 폭 하한) + +### 무엇이 문제였나 + +`OnnxPaddleOcr::run_rec` 은 검출된 박스를 높이 48 로 리사이즈하면서 **폭을 1 이상으로만** 보장했다. 그런데 PP-OCRv5 rec 백본은 폭을 반복해서 줄이기 때문에 폭이 한 자릿수인 입력은 도중에 폭 0 인 특징맵이 되고, ORT 가 Conv 에서 세션 실행 전체를 실패시킨다. + +그 실패가 `recognize` 의 `?` 를 타고 나가면서, **이미 인식해 둔 나머지 박스가 전부 함께 버려졌다.** 바로 아래 줄(`if text.is_empty() { continue; }`)이 "박스 하나가 비면 나머지는 살린다"는 의도를 담고 있는데 오류 경로에만 그 방어가 없었던 것이다. 손실이 얇은 조각 하나에서 그치지 않고 그 이미지 전체로 번졌다. + +색인 자체는 성공으로 끝나므로 **조용한 손실**이었다. 이미지 문서는 본문에 파일명만 남고, 스캔 PDF 페이지는 청크가 0 이 된다. 검색해서 0 건이 나올 때까지 드러나지 않는다. + +### 하한은 16 이 아니라 5 였다 — 실측 + +이슈는 `REC_MIN_WIDTH = 16` 을 제안했지만, 번들된 `korean_ppocrv5_mobile_rec.onnx` 를 직접 스윕해 보면 **실패하는 폭은 4 까지이고, 그래서 하한은 5** 다. 서로 다른 경로로 세 번 재 봤고 같은 표가 나왔다. + +| rec 입력 폭 (높이 48) | 결과 | +|---|---| +| 1 ~ 4 | 전부 실패 — `Invalid input shape: {1,0}` | +| 5 ~ 48 | 전부 성공 | + +경계는 흐릿하지 않고 딱 떨어지며, 픽셀 내용과 무관하다(흰 이미지·체커보드·글자 모양 렌더 모두 동일). 이유도 유도된다: rec 출력의 CTC 타임스텝 수가 `T = ceil((w - 4) / 8)` 이라서 `w <= 4` 에서 `T = 0` 이 되고, 오류 메시지의 `{1,0}` 이 바로 그 0 이다. 실측한 44 개 폭 전부가 이 식에 맞았다(`w=12 → T=1`, `w=13 → T=2`, `w=45 → T=6`). + +그래서 상수를 **5** 로 잡았다. 16 을 쓰면 폭 5~15 구간을 새로 버리게 되는데, 그 구간은 지금 정상 동작하는 범위다. 다만 그 구간이 실제로 글자를 뱉는지도 따로 재 봤고 — 폭 40 대조군은 `"1"`(0.97) / `"3"`(0.999) / `"7"`(0.998) 을 제대로 읽는데 5~16 구간은 잉크가 있어도 전부 빈 문자열이었다 — 즉 **16 을 써도 잃는 건 사실상 없다.** 두 값의 실질 차이는 없고, 5 를 고른 이유는 그것이 그래프의 진짜 하한이라서 상수의 이름과 주석이 거짓이 되지 않기 때문이다. + +`crop.width()` 가 아니라 리사이즈 **이후**의 `new_w` 를 검사한다. 3×200 짜리 크롭은 폭이 3 이지만 `new_w` 가 1 이고, 3×10 크롭은 같은 폭 3 인데 `new_w` 가 14 다. 네트워크가 보는 값은 `new_w` 뿐이다. + +### 회귀 테스트는 상수를 위아래로 다 가둔다 + +아래쪽은 `rec_min_width_is_the_graph_floor` (`paddle_onnx.rs` 의 `mod tests`) 가 두 가지로 잡는다. + +- `1..REC_MIN_WIDTH` 는 세션에 닿지 않고 빈 문자열로 빠져야 한다 → **가드를 지우면 실패**한다. 가드만 지우고 돌려 확인했다: `w=1 must never reach the rec session: rec session run: ... Invalid input shape: {1,0}`. +- 정확히 `REC_MIN_WIDTH` 는 **진짜 세션을 통과**해야 한다 → 상수가 모델의 실제 하한보다 낮으면(예: 모델 교체 후) 실패한다. + +위쪽은 상수 옆의 `const _: () = assert!(REC_MIN_WIDTH <= 16, …)` 가 잡는다. **컴파일이 안 된다** — 100 으로 바꾸면 `error[E0080]: evaluation panicked: REC_MIN_WIDTH is past the measured no-loss ceiling (16)`. + +위쪽 단언이 없으면 테스트가 한 방향으로만 샌다. 초안이 그랬다 — 상수를 100 으로 바꿔도 통과했다. 위의 두 단언은 상수가 커질수록 **더 잘** 통과하기 때문이다(가드가 더 많이 잡아 주고, 세션은 폭이 클수록 잘 돈다). 그 상태면 높이 48 기준 폭 100 미만 크롭, 즉 글자 한두 개짜리 박스가 전부 조용히 버려진다 — 이 항목이 없애려는 손실과 같은 종류다. 상한 16 은 위에서 잰 "5~16 은 잉크가 있어도 빈 문자열, 40 은 제대로 읽음" 에서 온 숫자이고, 테스트가 아니라 컴파일 시점에 건 이유는 상수를 만지는 사람이 테스트를 돌리기 전에 막히는 편이 낫기 때문이다. + +테스트는 `ModelPaths::from_default_dir()` 대신 `CARGO_MANIFEST_DIR` 에서 경로를 직접 만든다. 전자는 `KEBAB_IMAGE_OCR_MODEL_DIR` 을 타므로, 자기 모델 디렉토리를 export 해 둔 사람이 `cargo test` 를 돌리면 상수를 엉뚱한 모델에 대고 재게 된다. 이 테스트의 일은 상수를 **배포되는** 모델에 붙들어 두는 것이다. + +모델 에셋이 in-tree 로 커밋돼 있으므로(`git ls-files crates/kebab-parse-image/assets/`) skip 가드는 붙이지 않았다 — 조용히 안 도는 테스트가 이 이슈가 경고하는 바로 그 함정이다. + +PDF 노트의 `{e:#}` 도 `ocr_engine_failure_surfaces_as_warning` 이 고정한다. 원래 이 테스트는 mock 이 단층 오류를 내서 `{}` 로 되돌려도 통과했다 — anyhow 는 원인이 없는 오류를 두 형식에서 똑같이 찍기 때문이다. mock 을 실제와 같은 두 층 오류로 바꾸고 안쪽 원인까지 단언하도록 했다. `{}` 로 되돌리면 실패한다. + +### 실측 (도그푸딩 말뭉치 이미지 240 개 파일) + +`corpus/images/` 전량(jpg 172 · png 63 · jpeg 4 · tif 1 = 240 개 파일. `kebab ingest` 가 집는 것은 tif 를 뺀 239 개지만, 여기서는 OCR 엔진을 파일 목록에 직접 물렸다)을 수정 전후로 같은 엔진(`ppocrv5-mobile-kor-1b55f062d055`)·같은 설정(score 0.3 / unclip 1.5 / max_boxes 1000 / max_pixels 2048)으로 돌렸다. + +| | 수정 전 | 수정 후 | +|---|---|---| +| OCR 성공 | 204 / 240 | **240 / 240** | +| OCR 실패 | 36 (15.0%) | **0** | + +실패 36 장은 charts 8 · english-text 10 · korean-text 9 · photos 9 로, 특정 종류에 몰려 있지 않았다. 되살아난 본문은 합계 11,747 자(중앙값 37 자, 최대 3,950 자)다. 한 장이 얼마나 크게 손해 보고 있었는지가 드러나는 예: + +| 이미지 | 수정 후 | +|---|---| +| `charts/Bassano_Politi___1505___Questio_de_modalibus___diagrams.jpg` | 3,950 자 | +| `charts/Clickpath_Analysis.png` | 1,116 자 / 87 영역 | +| `korean-text/연고한2.jpg` | 1,088 자 | + +`Clickpath_Analysis.png` 은 얇은 조각 하나 때문에 **이미 인식된 87 개 영역**을 통째로 잃고 있었다. + +부작용이 없다는 것도 확인했다: 원래 성공하던 204 장의 인식 글자 수가 **한 장도 변하지 않았다**. 이 수정은 순수 가산이다. + +36 장 중 4 장은 수정 후에도 0 자인데, 글자가 없는 사진(예: `photos/Aphid_2007_1.jpg`, 진딧물 접사)이라 정상이다. 이 이슈로 인한 손실과 "원래 글자가 없어서 비는 문서"를 혼동하면 안 된다 — 성공한 204 장 중에서도 31 장은 det 가 박스를 못 찾아 정상적으로 0 자다. KB 쪽에서 영향 문서를 셀 때는 본문 길이가 아니라 `provenance_json LIKE '%Invalid input shape%'` 로 걸러야 한다. 다만 이 쿼리에는 조건이 둘 붙는다. + +**언제 세느냐.** 새 바이너리로 `kebab ingest` 를 돌리기 **전에** 세야 한다. 아래 §재색인 의 `parser_version` bump 가 해당 경로의 documents 행을 지우고 다시 쓰므로, 한 번 색인한 뒤에는 이 쿼리가 0 을 돌려준다. "업그레이드 → 색인 → 릴리스 노트 읽기" 순서로 가면 안 당한 것처럼 보인다. 이미 색인해 버렸다면 PDF 쪽은 `SELECT count(DISTINCT doc_id) FROM pdf_ocr_events WHERE success = 0 AND reason = 'ocr_error' AND ocr_engine = 'paddle-onnx'` 로 아직 셀 수 있다 — `pdf_ocr_events` 는 documents 에 FK 가 없어 purge 를 넘겨 살아남는다(`logging.retention_days` 기본 30 일 prune 만 받는다). `count(*)` 가 아니라 `count(DISTINCT doc_id)` 인 이유는 이 표가 **페이지마다** 한 행이고 유니크 제약도 없어서, 40 쪽을 잃은 문서 하나가 40 을 더하고 수정 전 색인을 두 번 돌렸으면 또 두 배가 되기 때문이다. `ocr_engine` 을 거는 이유는 `'ocr_error'` 가 `recognize()` 의 모든 실패를 받는 통칭이라, 기본 엔진인 ollama-vision 을 쓰는 KB 에서도 #239 와 무관한 행이 쌓이기 때문이다. 이렇게 걸러도 paddle-onnx 의 다른 실패까지 포함하는 상한이라는 점은 남는다. + +**스캔 PDF 는 이 필터로 안 걸린다.** 이번 수정 이전에 나간 **모든** 릴리스에서 — v0.33.0 을 포함해서 — PDF 경로는 노트를 `err={}` 로 찍었고, anyhow 의 기본 Display 는 가장 바깥 context 하나(`rec session run`)만 내보낸다. 노출 구간은 이렇다 — PDF OCR 엔진으로 paddle-onnx 를 고를 수 있게 된 것이 **v0.28.0** 이므로 그때부터 스캔 PDF 도 같은 손실을 겪을 수 있었고, 다만 v0.32.0 까지는 단일 DCTDecode 이미지 페이지만 OCR 대상이었다. (태그별로 확인할 때 `crates/kebab-app/src/ingest.rs` 만 훑으면 v0.31.0 처럼 보인다. 그 파일이 v0.31.0 에서 생겼을 뿐이고, `build_pdf_ocr_engine` 은 v0.28.0~v0.30.1 에서 `crates/kebab-app/src/lib.rs` 에 있었다. 파일부터 찾은 뒤 세야 한다.) 페이지 렌더링(#232)이 임의 인코딩까지 넓힌 v0.33.0 에서 대상이 크게 늘었으니 기록의 대부분은 그 릴리스가 만든 것이겠지만, 전부는 아니다. 그래서 스캔본까지 세려면 `OR provenance_json LIKE '%err=rec session run%'` 을 반드시 함께 걸어야 한다. 이미지 경로는 처음부터 `{err:#}` 라 체인 전체가 들어갔고, 이번에 PDF 쪽도 `{e:#}` 로 맞췄으므로 앞으로 색인되는 것은 양쪽 다 첫 필터에 걸린다. + +### 재색인: 버전 두 개를 올렸다 + +코드만 고치면 이미 색인된 문서에는 닿지 않는다. 이미지 문서의 실효 `parser_version` 은 `image-meta-v1|chunk:…|ocr:1:paddle-onnx:<모델 blake3>` 인데, 이미지 파일도 모델 에셋도 안 바뀌었으니 서명이 동일하고 `try_skip_unchanged` 가 Unchanged 로 건너뛴다. #232 에서 세운 원칙 그대로다 — 사용자가 `--force-reingest` 를 떠올려야만 고쳐지는 수정은 고쳐진 게 아니다. + +- `image-meta-v1` → **`image-meta-v2`** +- `pdf-text-v2` → **`pdf-text-v3`** (스캔 PDF 도 같은 `run_rec` 을 타므로 같은 손실을 겪었다) + +**비싼 OCR 은 대부분 다시 안 돈다.** OCR 산출물은 `derivation_cache` 에 **소스 바이트** 키로 들어가 있어서 `parser_version` 캐스케이드와 분리돼 있다(`docs/ARCHITECTURE.md:35` 의 derivation_cache 행, v0.31.0 #217). 그리고 실패한 OCR 은 캐시에 **저장되지 않는다** — `Err` 분기가 `derivation_cache_put` 앞에서 빠져나간다(이미지는 `ingest.rs` 의 `ingest_one_image_asset`, PDF 는 `pdf_ocr_apply.rs` 의 `apply_ocr_to_pdf_pages`). 줄 번호를 안 적은 건 이 항목이 처음 썼던 두 참조가 같은 PR 의 후속 커밋에 밀려 둘 다 어긋났기 때문이다. 그래서 이미 성공했던 문서는 캐시에 히트해 엔진 호출을 건너뛰고, **실제로 다시 OCR 되는 건 이 버그로 실패했던 문서뿐**이다. + +**하지만 나머지 비용은 전부 든다.** `id_for_doc` 이 접는 것은 composite 가 아니라 **base** PARSER_VERSION 이므로(`kebab-parse-image/src/lib.rs`, `kebab-parse-pdf/src/lib.rs` 의 `extract`; composite 는 그 뒤에 `canonical.parser_version` 에만 찍힌다) 이 bump 는 **모든 이미지·PDF 문서의 doc_id 를 바꾼다**. 그러면 `ingest.rs` 의 `purge_workspace_path_for_parser_bump` 가 돌아 documents 행이 지워지고(blocks·chunks·embedding_records CASCADE) 해당 chunk_id 의 Lance 벡터도 전량 삭제된 뒤, 재파싱·재청킹·재임베딩·재삽입이 이어진다. 임베딩은 파생물 캐시에 히트하지만 행은 다시 쓴다. doc_id 는 wire 필수 필드이자 `kebab inspect doc ` 의 핸들이라, 파일을 하나도 안 고쳤는데 전부 한꺼번에 바뀐다. + +그리고 이 비용은 **버그가 닿지 않는 KB 에도** 걸린다. 기본 OCR 엔진은 `ollama-vision` 이고 이미지 OCR 은 기본 off, `PdfOcrCfg::defaults()` 도 `enabled: false, always_on: false` 라, paddle-onnx 를 안 쓰는 KB 는 얻는 것 없이 이미지·PDF 를 다시 색인하게 된다. + +**그래도 base 를 올리는 쪽을 택했다.** 대안은 `ingest_config_signature` 의 `ocr_engine_version_for_sig` 에 paddle-onnx 전용 revision 토큰을 넣어 영향 문서만 무효화하는 것이고, 그러면 doc_id 도 유지되고 ollama-vision·OCR off KB 도 안 건드린다. 더 정확하지만 #232 가 세운 선례와 다른 새 무효화 경로를 하나 더 만드는 일이고, 이 저장소는 단일 사용자용이라 실제로 손해 보는 KB 가 사용자 자신의 것 하나다. 캐스케이드 규칙이 이미 있는데 그 옆에 두 번째 규칙을 세우는 값을 치를 만큼은 아니라고 봤다. 다중 사용자 배포로 가면 다시 볼 결정이다. + +### 스냅샷도 함께 움직였다 + +`pdf-text-v3` 는 `vector_pdf_canonical.json` 을 움직인다. #232 때와 같은 형태임을 확인했다 — 바뀐 것은 **파생 식별자와 버전 문자열뿐**이고 본문 텍스트·inlines·`source_span`·metadata 는 동일하다. + +| 필드 | v2 | v3 | +|---|---|---| +| `doc_id` | `bd04fa01…` | `960e4c44…` | +| `block_id` | `964162ec…` | `b85b8a6d…` | +| `parser_version` | `pdf-text-v2` | `pdf-text-v3` | +| provenance note | `parser_version=pdf-text-v2; …` | `parser_version=pdf-text-v3; …` | + +`kebab-parse-pdf/tests/extractor.rs` 와 `kebab-app/tests/pdf_pipeline.rs` 의 버전 단언 세 곳도 함께 옮겼다. 이미지 쪽은 스냅샷이 상수에서 값을 유도하고 있어 손댈 것이 없었다. + +### 고치지 않고 남긴 것 + +`recognize` 의 박스 루프 안 `self.run_rec(&crop)?` 는 그대로 뒀다. 폭 가드가 알려진 유일한 방아쇠를 닫았고, `T = ceil((w-4)/8) >= 1` 이 `w >= 5` 에서 항상 성립하므로 폭 때문에 이 경로가 다시 터지는 일은 없다. + +박스 단위로 살려 두려면 그 `?` 를 `continue` 로 바꿔야 하는데, 그러면 `run_rec` 안의 클래스 수 검사(`rec output has {c} classes`)까지 함께 삼킨다. 그건 입력과 무관한 **설정 오류**(rec 모델과 dict 짝이 안 맞음)라 즉시 실패하는 편이 맞다. 다만 이게 `?` 를 유지할 결정적 이유는 아니다 — 클래스 차원은 정적 그래프 메타데이터라(번들 rec 출력 shape 가 `[-1, -1, 11947]`) 세션 로드 시점에 읽을 수 있고, 그렇다면 그 검사는 `from_paths` 의 `dict.len() != DICT_LINES` bail 옆으로 옮기는 편이 더 낫다. 잘못된 모델이 박스가 검출되는 이미지를 기다릴 것 없이 엔진 생성에서 바로 터지기 때문이다. 이번에 안 한 건 #239 의 방아쇠와 무관한 별개 정리이기 때문이고, 옮길 때는 `continue` 에 `tracing::warn!` 을 반드시 함께 달아야 한다. 맨 `continue` 는 이 항목이 없애려는 조용한 손실을 다른 자리로 옮기는 것에 지나지 않는다. + +### 이슈 본문과 다른 점 하나 + +이슈는 오류의 노드 이름을 `Conv.33` 으로 적었는데, 이 머신에서는 같은 실패가 `p2o.pd_op.batch_norm_.1.0_nchwc` 로 나온다. ORT 가 CPU 명령어 집합에 맞춰 그래프를 최적화(NCHWc 레이아웃 변환)하면서 노드 이름을 다시 붙이기 때문이고, 상태 메시지와 `{1,0}` 은 동일하다. 다른 버그가 아니다. + ## 2026-08-17 — #232 PDF OCR 이 DCTDecode 아닌 스캔본을 전량 건너뜀 (페이지 렌더링) ### 무엇이 문제였나