fix(parse-image): #239 얇은 검출 박스가 이미지 OCR 전체를 날리던 문제
`run_rec` 이 크롭을 높이 48 로 리사이즈하면서 폭을 1 이상으로만 보장했다. PP-OCRv5 rec 백본은 `T = ceil((w-4)/8)` 개의 CTC 타임스텝을 내므로 `w <= 4` 에서 특징맵이 0 열로 접히고, ORT 가 세션 실행 전체를 실패시킨다. 그 오류가 `recognize` 의 `?` 를 타고 나가면서 이미 인식해 둔 나머지 박스까지 전부 버려졌다 — 색인은 성공으로 끝나므로 검색이 0 건 나올 때까지 안 드러나는 조용한 손실이었다. `REC_MIN_WIDTH = 5` 미만이면 세션에 넣지 않고 빈 문자열로 돌려보낸다. 바로 아래 `text.is_empty()` 분기가 그 박스만 버리고 나머지를 살린다. 이슈는 16 을 제안했지만 번들 모델을 직접 스윕하니 실제 하한은 4 였다(1~4 전부 실패, 5~48 전부 성공, 픽셀 내용과 무관하게 경계가 딱 떨어짐). 16 을 써도 잃는 건 사실상 없지만(5~15 구간은 잉크가 있어도 빈 문자열을 낸다), 5 가 그래프의 진짜 하한이라 상수 이름과 주석이 거짓이 되지 않는다. 회귀 테스트는 양방향을 고정한다. `1..REC_MIN_WIDTH` 는 세션에 닿지 않아야 하고(가드를 지우면 실패), 정확히 `REC_MIN_WIDTH` 는 진짜 세션을 통과해야 한다(상수가 모델의 실제 하한보다 낮으면 실패). 모델 에셋이 in-tree 라 skip 가드는 붙이지 않았다. parser_version cascade: 코드만 고치면 이미 색인된 문서에 닿지 않는다. `image-meta-v1` → `image-meta-v2`, `pdf-text-v2` → `pdf-text-v3` (스캔 PDF 도 같은 `run_rec` 을 탄다). 실패한 OCR 은 derivation_cache 에 저장되지 않으므로, 다음 ingest 는 전량 재추출하되 이미 성공한 것은 캐시 히트로 엔진을 건너뛰고 이 버그로 실패했던 문서만 실제로 다시 OCR 된다. 실측 (도그푸딩 말뭉치 이미지 240 장, 같은 엔진·같은 설정): - OCR 실패 36/240 (15.0%) → 0/240 - 되살아난 본문 11,747 자 (최대 3,950 자, `Clickpath_Analysis.png` 은 얇은 조각 하나 때문에 이미 인식된 87 개 영역을 통째로 잃고 있었다) - 원래 성공하던 204 장은 인식 글자 수 변화 0 — 순수 가산 Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -165,7 +165,7 @@ fn ingest_3_page_pdf_produces_one_doc_and_per_page_chunks() {
|
||||
pdf_item.parser_version
|
||||
.as_ref()
|
||||
.map(|p| p.0.split('|').next().unwrap()),
|
||||
Some("pdf-text-v2")
|
||||
Some("pdf-text-v3")
|
||||
);
|
||||
assert_eq!(
|
||||
pdf_item.chunker_version.as_ref().map(|c| c.0.as_str()),
|
||||
@@ -479,10 +479,10 @@ fn inspect_doc_surfaces_page_spans() {
|
||||
.find(|i| i.doc_path.0.ends_with("inspect.pdf"))
|
||||
.unwrap();
|
||||
let doc = kebab_app::inspect_doc_with_config(cfg, pdf_item.doc_id.as_ref().unwrap()).unwrap();
|
||||
// v0.26.2: stored parser_version is now `pdf-text-v2|<ingest-config-sig>`
|
||||
// v0.26.2: stored parser_version is now `pdf-text-v3|<ingest-config-sig>`
|
||||
// (the signature folds chunking / pdf.ocr settings for skip detection).
|
||||
// Assert the base identity by taking the prefix before the first '|'.
|
||||
assert_eq!(doc.parser_version.0.split('|').next().unwrap(), "pdf-text-v2");
|
||||
assert_eq!(doc.parser_version.0.split('|').next().unwrap(), "pdf-text-v3");
|
||||
assert_eq!(doc.blocks.len(), 3);
|
||||
for block in &doc.blocks {
|
||||
match block {
|
||||
|
||||
@@ -49,7 +49,19 @@ use serde_json::{Map, Value};
|
||||
use time::OffsetDateTime;
|
||||
|
||||
/// Parser version label for the image extractor (§9 versioning).
|
||||
pub const PARSER_VERSION: &str = "image-meta-v1";
|
||||
///
|
||||
/// Bumped to v2 for issue #239 (2026-08-28): a detection box narrower than
|
||||
/// the rec graph's floor used to abort the ONNX session and discard OCR for
|
||||
/// the *whole* image, so an affected image was indexed with its filename and
|
||||
/// nothing else. `REC_MIN_WIDTH` in `paddle_onnx` fixes the extraction; this
|
||||
/// bump is what makes the fix reach stores that were already indexed.
|
||||
/// `try_skip_unchanged` calls an asset Unchanged when its content hash *and*
|
||||
/// version inputs match — the image file on disk did not change, and neither
|
||||
/// did the OCR model assets that feed `ingest_config_signature`, so without
|
||||
/// this every already-indexed image would keep its empty extraction until the
|
||||
/// user thought to pass `--force-reingest`. Per CLAUDE.md §Versioning
|
||||
/// cascade, changing this invalidates downstream image records.
|
||||
pub const PARSER_VERSION: &str = "image-meta-v2";
|
||||
|
||||
/// Maximum decode dimension (per axis) before we refuse to read the image.
|
||||
/// Matches the §9.1 "cap decode at ~16k" policy in the design doc.
|
||||
|
||||
@@ -51,6 +51,14 @@ const REC_CLASSES: usize = 11947;
|
||||
const DET_LIMIT_SIDE_LEN: u32 = 960;
|
||||
/// rec input height (PP-OCRv5 mobile).
|
||||
const REC_HEIGHT: u32 = 48;
|
||||
/// Narrowest rec input the graph survives. The backbone emits
|
||||
/// `T = ceil((w - 4) / 8)` CTC timesteps, so `w <= 4` collapses the feature
|
||||
/// map to zero columns and ORT aborts the whole session run with
|
||||
/// `Invalid input shape: {1,0}` — taking every already-recognized box on the
|
||||
/// image down with it (issue #239). Measured against the bundled
|
||||
/// `korean_ppocrv5_mobile_rec.onnx`: 1..=4 always fail, 5.. always succeed.
|
||||
/// `rec_min_width_is_the_graph_floor` pins both sides of that boundary.
|
||||
const REC_MIN_WIDTH: u32 = 5;
|
||||
/// DBNet probability-map binarization threshold. Looser than Paddle's default
|
||||
/// `box_thresh` (0.6) to keep recall high on low-contrast Korean text.
|
||||
const DET_BIN_THRESH: f32 = 0.3;
|
||||
@@ -356,6 +364,13 @@ impl OnnxPaddleOcr {
|
||||
// resize keep-aspect to height 48, then this single crop is its own batch
|
||||
let (cw, ch) = (crop.width().max(1), crop.height().max(1));
|
||||
let new_w = ((REC_HEIGHT as f32 / ch as f32) * cw as f32).round().max(1.0) as u32;
|
||||
if new_w < REC_MIN_WIDTH {
|
||||
// A sliver this thin holds no glyph, and feeding it to the graph
|
||||
// would abort recognition for the entire image. Report it the way
|
||||
// an undecodable box is already reported: the caller's
|
||||
// `text.is_empty()` arm drops this box and keeps the rest.
|
||||
return Ok((String::new(), 0.0));
|
||||
}
|
||||
let resized = image::imageops::resize(
|
||||
crop,
|
||||
new_w,
|
||||
@@ -1002,4 +1017,37 @@ mod tests {
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// Issue #239: a detection box narrower than the rec graph's floor used to
|
||||
/// abort the ORT session, and `recognize`'s `?` turned that into "this
|
||||
/// image has no OCR text at all" — 36 of 240 corpus images.
|
||||
///
|
||||
/// Both directions matter, so both are asserted. Below `REC_MIN_WIDTH` the
|
||||
/// guard must short-circuit (delete it and this errors). At exactly
|
||||
/// `REC_MIN_WIDTH` the real session must run (set the constant below the
|
||||
/// graph's true floor, e.g. after a model swap, and this errors) — that
|
||||
/// second half is what keeps the constant pinned to the shipped model
|
||||
/// instead of being a number nobody rechecks.
|
||||
#[test]
|
||||
fn rec_min_width_is_the_graph_floor() {
|
||||
let engine =
|
||||
OnnxPaddleOcr::from_paths(&ModelPaths::from_default_dir(), 0.3, 1.5, 1000, 1600)
|
||||
.expect("bundled OCR assets must load");
|
||||
// A crop already at REC_HEIGHT makes run_rec's keep-aspect resize the
|
||||
// identity, so the crop width *is* the rec input width — no rounding
|
||||
// to reason about.
|
||||
let crop = |w| image::RgbImage::from_pixel(w, REC_HEIGHT, image::Rgb([255, 255, 255]));
|
||||
|
||||
for w in 1..REC_MIN_WIDTH {
|
||||
let (text, conf) = engine
|
||||
.run_rec(&crop(w))
|
||||
.unwrap_or_else(|e| panic!("w={w} must never reach the rec session: {e:#}"));
|
||||
assert!(text.is_empty(), "w={w} decoded {text:?}");
|
||||
assert_eq!(conf, 0.0, "w={w}");
|
||||
}
|
||||
|
||||
engine.run_rec(&crop(REC_MIN_WIDTH)).unwrap_or_else(|e| {
|
||||
panic!("REC_MIN_WIDTH={REC_MIN_WIDTH} must survive the rec session: {e:#}")
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -36,6 +36,13 @@ use kebab_core::{
|
||||
use serde_json::{Map, Value};
|
||||
use time::OffsetDateTime;
|
||||
|
||||
/// Bumped to v3 for issue #239 (2026-08-28): scanned pages share the
|
||||
/// paddle-onnx rec path with image OCR, so a page whose raster produced one
|
||||
/// over-thin detection box lost that page's OCR text entirely. The fix lives
|
||||
/// in `kebab-parse-image`; this bump is what re-processes scans that were
|
||||
/// already indexed under v2 (same reasoning as the v2 bump below — the PDF
|
||||
/// bytes have not changed, so nothing else would invalidate them).
|
||||
///
|
||||
/// Bumped to v2 for issue #232 (2026-08-17): scanned pages are now
|
||||
/// rasterized by rendering rather than by pulling out an embedded JPEG,
|
||||
/// so pages encoded with CCITTFax / JBIG2 / Flate / JPX — previously
|
||||
@@ -47,7 +54,7 @@ use time::OffsetDateTime;
|
||||
/// every already-indexed scan would keep its empty extraction until the
|
||||
/// user thought to pass `--force-reingest`. Per CLAUDE.md §Versioning
|
||||
/// cascade, changing this invalidates downstream PDF records.
|
||||
pub const PARSER_VERSION: &str = "pdf-text-v2";
|
||||
pub const PARSER_VERSION: &str = "pdf-text-v3";
|
||||
|
||||
/// Text-PDF extractor. Per-page text via `lopdf::Document::extract_text`
|
||||
/// (the only stable per-page API in the lopdf / pdf-extract pair —
|
||||
|
||||
@@ -267,7 +267,7 @@ fn snapshot_three_page_canonical_document_stable() {
|
||||
// golden file (the full JSON contains BLAKE3 ids that would
|
||||
// change if `id_from(...)`'s tuple shape ever shifts — that would
|
||||
// be a separate, intentional break).
|
||||
assert_eq!(json["parser_version"], Value::String("pdf-text-v2".into()));
|
||||
assert_eq!(json["parser_version"], Value::String("pdf-text-v3".into()));
|
||||
assert_eq!(json["lang"], Value::String("und".into()));
|
||||
assert_eq!(json["schema_version"], Value::Number(1.into()));
|
||||
assert_eq!(json["doc_version"], Value::Number(1.into()));
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
{
|
||||
"doc_id": "bd04fa013899592afcf671013404ed68",
|
||||
"doc_id": "960e4c445f1365a633c9647b42ed27ba",
|
||||
"source_asset_id": "babe9824b6b28237c0898575a40ba48d",
|
||||
"workspace_path": "mojibake.pdf",
|
||||
"title": "untitled",
|
||||
@@ -8,7 +8,7 @@
|
||||
{
|
||||
"kind": "paragraph",
|
||||
"common": {
|
||||
"block_id": "964162ec3cf0c191849f5a5cf8a3b675",
|
||||
"block_id": "b85b8a6d2462bb396112f45cfcea6c44",
|
||||
"heading_path": [],
|
||||
"source_span": {
|
||||
"kind": "page",
|
||||
@@ -54,11 +54,11 @@
|
||||
"at": "1970-01-01T00:00:00Z",
|
||||
"agent": "kb-parse-pdf",
|
||||
"kind": "parsed",
|
||||
"note": "parser_version=pdf-text-v2; page_count=1"
|
||||
"note": "parser_version=pdf-text-v3; page_count=1"
|
||||
}
|
||||
]
|
||||
},
|
||||
"parser_version": "pdf-text-v2",
|
||||
"parser_version": "pdf-text-v3",
|
||||
"schema_version": 1,
|
||||
"doc_version": 1,
|
||||
"last_chunker_version": null,
|
||||
|
||||
@@ -20,11 +20,11 @@ Cargo workspace, 함수 호출 기반 모듈러 모놀리스. UI binary (`kebab-
|
||||
| 한국어 형태소분석 | `lindera-ko-dic` (FTS5 외부 tokenizer, v0.20.1) — 2자 이상 한국어 query 지원 |
|
||||
| LLM | Ollama HTTP (default `gemma4:e4b` ─ OCR / caption 와 family 통일. 사용자가 더 큰 variant `gemma4:26b` 등으로 override 가능) |
|
||||
| 음성 ASR | `whisper.cpp` (via `whisper-rs`) — P8 보류, 시스템 dep brainstorm 후 |
|
||||
| OCR (image) | `OcrEngine` trait, 2 백엔드: **`ollama-vision`** (default, `gemma4:e4b`) / **`paddle-onnx`** (v0.27.0 — PP-OCRv5 ONNX in-process via `ort` =2.0.0-rc.9, DBNet det + CTC rec, 후처리 min-area rect/unclip pure-Rust, Python 런타임 0). engine 선택은 `[image.ocr] engine`, 팩토리는 `kebab-app::build_image_ocr_engine`. e2e CER 0.005 / 큰 페이지 <4초. (HOTFIXES P6-2, 2026-06-04) |
|
||||
| OCR (image) | `OcrEngine` trait, 2 백엔드: **`ollama-vision`** (default, `gemma4:e4b`) / **`paddle-onnx`** (v0.27.0 — PP-OCRv5 ONNX in-process via `ort` =2.0.0-rc.9, DBNet det + CTC rec, 후처리 min-area rect/unclip pure-Rust, Python 런타임 0). engine 선택은 `[image.ocr] engine`, 팩토리는 `kebab-app::build_image_ocr_engine`. e2e CER 0.005 / 큰 페이지 <4초. (HOTFIXES P6-2, 2026-06-04) **불변식**: rec 세션 입력 폭은 `REC_MIN_WIDTH = 5` 이상이어야 한다 — 백본이 `T = ceil((w-4)/8)` 개의 CTC 타임스텝을 내므로 `w <= 4` 는 특징맵을 0 열로 접어 ORT 가 세션 전체를 실패시킨다. 그 실패는 `recognize` 의 박스 루프를 뚫고 나가 이미 인식한 박스까지 전부 버리므로, 얇은 크롭은 세션에 넣지 말고 빈 문자열로 돌려보내야 한다 (issue #239). `parser_version = "image-meta-v2"` (issue #239 에서 v1 → v2, 기존 색인 이미지 재처리 유발). |
|
||||
| OCR (PDF, v0.20.0+) | Ollama vision LM (default `qwen2.5vl:3b`) — post-extract enrichment via `kebab-app::pdf_ocr_apply` (H-1 resolution). DCTDecode-only v1 (FlateDecode/CCITTFax skip + warning). family asymmetry vs image OCR: PoC alnum 94.79% (qwen2.5vl) >> 27% (gemma4:e4b 받침), 본 단계에서 PDF OCR 만 qwen2.5vl. |
|
||||
| Image caption | Ollama vision LM, runtime gate `image.caption.enabled` (default OFF) |
|
||||
| RAG groundedness 검증 | `kebab-nli` 의 mDeBERTa-v3 XNLI 가 `(packed_chunks, generated_answer)` entailment 검사 (fb-41). `[rag] nli_threshold > 0` (default 0 = disabled, production 권장 0.5) 일 때 활성 — 미달 시 `refusal_reason = nli_verification_failed` (LLM self-judge ceiling 보완). 첫 호출 시 ~280 MB ONNX 자동 다운로드 |
|
||||
| PDF parser | `lopdf` per-page 텍스트 + 스캔 페이지 래스터화. 래스터는 **pdfium 페이지 렌더링**(`page_render::PageRenderer`, issue #232) 이 1순위 — 필터·XObject 구성과 무관하게 페이지를 그린다. pdfium 이 없으면 `page_image::extract_dctdecode_page_image` 로 떨어지며 그 경우 단일 DCTDecode 이미지 페이지만 OCR 된다. pdfium 은 공유 라이브러리로만 배포돼 링크하면 단일 바이너리 원칙이 깨지므로 **런타임 바인딩**이고, `[ingest.pdf.ocr] render_library` 로 경로를 지정하거나 로더 경로에 두면 된다. `kebab doctor` 의 `pdf_render` 가 어느 쪽인지 보고한다. `chunker_version = "pdf-page-v1"` 하드코딩 (HOTFIXES P7-3). `parser_version = "pdf-text-v2"` (issue #232 에서 v1 → v2, 기존 색인 스캔본 재처리 유발). |
|
||||
| PDF parser | `lopdf` per-page 텍스트 + 스캔 페이지 래스터화. 래스터는 **pdfium 페이지 렌더링**(`page_render::PageRenderer`, issue #232) 이 1순위 — 필터·XObject 구성과 무관하게 페이지를 그린다. pdfium 이 없으면 `page_image::extract_dctdecode_page_image` 로 떨어지며 그 경우 단일 DCTDecode 이미지 페이지만 OCR 된다. pdfium 은 공유 라이브러리로만 배포돼 링크하면 단일 바이너리 원칙이 깨지므로 **런타임 바인딩**이고, `[ingest.pdf.ocr] render_library` 로 경로를 지정하거나 로더 경로에 두면 된다. `kebab doctor` 의 `pdf_render` 가 어느 쪽인지 보고한다. `chunker_version = "pdf-page-v1"` 하드코딩 (HOTFIXES P7-3). `parser_version = "pdf-text-v3"` (issue #232 에서 v1 → v2, issue #239 에서 v2 → v3 — 둘 다 기존 색인 스캔본 재처리 유발). |
|
||||
| code parser | `tree-sitter` + `tree-sitter-rust` / `tree-sitter-python` / `tree-sitter-typescript` / `tree-sitter-javascript` / `tree-sitter-go` / `tree-sitter-java` / `tree-sitter-kotlin-ng` — **parser-side** (`kebab-parse-code`), chunker-side 아님 (design §6.3). chunker versions: Rust = `code-rust-ast-v1`, Python = `code-python-ast-v1`, TypeScript = `code-ts-ast-v1`, JavaScript = `code-js-ast-v1`, Go = `code-go-ast-v1`, Java = `code-java-ast-v1`, Kotlin = `code-kotlin-ast-v1`. (v0.32.0 #220: 9개 언어 chunker 가 단일 `CodeAstV1Chunker` 로 통합 — `for_lang(lang)` 가 per-lang `chunker_version` 라벨을 verbatim 매핑. chunker 는 tree-sitter 미사용·`lang` 은 `SourceSpan::Code` 데이터에서 흐르므로 9개 struct 차이는 `VERSION_LABEL` 문자열뿐이었음 → chunk_id byte-identical.) `ast_chunk_max_lines = 200` 상수 고정 (HOTFIXES 2026-05-19 — Chunker trait 이 per-medium config 미노출). Kotlin grammar 은 `tree-sitter-kotlin-ng` 사용 — bare `tree-sitter-kotlin` 은 tree-sitter 0.21–0.23 에 고착되어 있어 사용 불가. **Tier 2 (p10-2)**: YAML/k8s → `serde_yaml_ng` + `k8s-manifest-resource-v1` (apiVersion+kind per resource), Dockerfile → `dockerfile-file-v1` (whole-file), Cargo.toml/go.mod/.json/.xml/.groovy → `manifest-file-v1` (whole-file). Tier 2 chunkers live in `kebab-chunk`; no tree-sitter grammar needed (structure from file type, not AST). **Tier 3 (p10-3)**: shell scripts (`.sh`/`.bash`/`.zsh`) direct → `code-text-paragraph-v1` (blank-line paragraph segmentation + 80-line / 20-overlap line-window for oversize). Same chunker also serves as fallback when Tier 1/2 emit 0 chunks or Err — non-k8s YAML / invalid YAML / AST extractor failures all picked up. symbol = None; lang preserved from input doc. **Tier 1 family complete (p10-1D)**: C (`tree-sitter-c`, `code-c-ast-v1`, `.c`/`.h`) + C++ (`tree-sitter-cpp`, `code-cpp-ast-v1`, `.cpp`/`.cc`/`.cxx`/`.hpp`/`.hh`/`.hxx`). C symbol = function name only; C++ symbol = `namespace::Class::method` (recursive nesting). `.h` 가 C++ syntax 만나면 tree-sitter-c parse 실패 → Tier 3 fallback. |
|
||||
| symbol path 형식 | workspace path → module path: Python = dotted prefix (`kebab_eval.metrics.compute_mrr`), TypeScript/JavaScript = slash-style prefix (`src/Foo.Foo.search`), Go = `package.Func` / `package.(*Receiver).Method`, Java/Kotlin = `com.foo.Foo.bar` (패키지+클래스+메서드/필드), C = 함수명, C++ = `namespace::Class::method`. Rust 1A-2 는 file-scope nesting 만 (workspace prefix 없음, 비일관 수용 — HOTFIXES 2026-05-20). code chunk 은 `citation.kind = "code"` + `citation.lang` + `symbol` + line range, SearchHit 에 `code_lang` + `repo`(`.git` walk-up 디렉토리명) backfill. |
|
||||
| Desktop | Tauri 2 + `pdfjs-dist` (native PDF render backend 금지) — P9-5 |
|
||||
|
||||
@@ -141,7 +141,7 @@ enabled = false # opt-in
|
||||
- 1 `Block::Paragraph` per page (P7-1 invariant).
|
||||
|
||||
**verify**:
|
||||
- `parser_version = "pdf-text-v2"`.
|
||||
- `parser_version = "pdf-text-v3"`.
|
||||
- `chunker_version = "pdf-page-v1"` (또는 `"pdf-page-v1.1"` from v0.20.1).
|
||||
- `block_count` ≥ page count.
|
||||
|
||||
|
||||
@@ -713,7 +713,7 @@ KB --json schema | jq '.stats.code_lang_breakdown'
|
||||
- 코퍼스에 없는 주제로 `kebab ask` → `refusal_reason: "llm_self_judge"` (또는 `no_chunks` / `score_gate`) + `grounded: false`.
|
||||
- (P6-4) `image.ocr.enabled = true` 로 PNG 자산을 ingest 하면 `kebab list docs` 가 markdown 옆에 image doc 도 출력 (`workspace_path` 가 `*.png`). `kebab inspect doc <image_doc_id>` 의 `block.ocr.joined` 가 vision LM 의 OCR 결과 (예: 스크린샷 안의 텍스트). `kebab search --mode lexical "<OCR text>"` 가 그 image chunk 를 반환하면 wiring 정상.
|
||||
- OCR / caption 부분 실패는 `errors` 카운터 미증가 — `kebab inspect doc <id>` 의 Provenance Warning 이벤트 또는 `--debug` 로그에서만 확인.
|
||||
- (P7-3) `*.pdf` 자산을 워크스페이스에 두면 `kebab ingest` 출력에 PDF 도 `new` 카운터에 포함. `kebab inspect doc <pdf_doc_id>` 가 `parser_version = "pdf-text-v2"` + 페이지마다 `Block::Paragraph` + `SourceSpan::Page { page, char_start, char_end }`. 본문에 등장하는 단어로 `kebab search --mode hybrid` 시 PDF chunk 가 결과에 포함되고 `source_span.kind = "page"` 면 wiring 정상. 암호화 PDF 는 `errors+=1` 로 분류되며 `error` 필드에 `qpdf --decrypt` 안내 보존. 빈/스캔 페이지 (PDF 가 텍스트를 추출하지 못한 페이지) 는 0 chunk + `Provenance::Warning` ("scanned candidate") 로 표시 — P+ scanned-PDF OCR fallback 까지는 검색 불가.
|
||||
- (P7-3) `*.pdf` 자산을 워크스페이스에 두면 `kebab ingest` 출력에 PDF 도 `new` 카운터에 포함. `kebab inspect doc <pdf_doc_id>` 가 `parser_version = "pdf-text-v3"` + 페이지마다 `Block::Paragraph` + `SourceSpan::Page { page, char_start, char_end }`. 본문에 등장하는 단어로 `kebab search --mode hybrid` 시 PDF chunk 가 결과에 포함되고 `source_span.kind = "page"` 면 wiring 정상. 암호화 PDF 는 `errors+=1` 로 분류되며 `error` 필드에 `qpdf --decrypt` 안내 보존. 빈/스캔 페이지 (PDF 가 텍스트를 추출하지 못한 페이지) 는 0 chunk + `Provenance::Warning` ("scanned candidate") 로 표시 — P+ scanned-PDF OCR fallback 까지는 검색 불가.
|
||||
|
||||
## config migrate (마이그레이션)
|
||||
|
||||
|
||||
@@ -26,11 +26,11 @@ classDiagram
|
||||
parse_blocks(body) (Vec~ParsedBlock~, Warnings)
|
||||
}
|
||||
class PdfTextExtractor {
|
||||
PARSER_VERSION = "pdf-text-v2"
|
||||
PARSER_VERSION = "pdf-text-v3"
|
||||
new() Self
|
||||
}
|
||||
class ImageExtractor {
|
||||
PARSER_VERSION = "image-meta-v1"
|
||||
PARSER_VERSION = "image-meta-v2"
|
||||
MAX_DECODE_DIM = 16384
|
||||
new() Self
|
||||
}
|
||||
@@ -101,7 +101,7 @@ flowchart LR
|
||||
|
||||
**PDF** (`kebab-parse-pdf`):
|
||||
- `PdfTextExtractor` — `Extractor` 구현체. `lopdf::Document::load_mem` 로 한 번 파싱, encrypted 면 즉시 bail.
|
||||
- `PARSER_VERSION = "pdf-text-v2"` — version cascade entry (issue #232 에서 v1 → v2, 페이지 렌더링 도입으로 기존 색인 스캔본 재처리 유발). (HOTFIXES P7-2 의 chunker_version `pdf-page-v1` 와 별개.)
|
||||
- `PARSER_VERSION = "pdf-text-v3"` — version cascade entry (issue #232 에서 v1 → v2, 페이지 렌더링 도입으로 기존 색인 스캔본 재처리 유발; issue #239 에서 v2 → v3, 얇은 검출 박스가 페이지 OCR 을 통째로 날리던 것을 고치면서 기존 색인 스캔본 재처리 유발). (HOTFIXES P7-2 의 chunker_version `pdf-page-v1` 와 별개.)
|
||||
- 빈 페이지 / extract 실패 → `Block::Paragraph` 빈 inlines + `ProvenanceKind::Warning("scanned candidate")`. OCR fallback 미구현.
|
||||
|
||||
**Image** (`kebab-parse-image`):
|
||||
|
||||
@@ -14,6 +14,93 @@ historical contract that was implemented; this file accumulates the
|
||||
deltas so phase 5+ readers can find the live behavior without diffing
|
||||
git history.
|
||||
|
||||
## 2026-08-28 — #239 얇은 검출 박스 하나가 이미지 OCR 전체를 날림 (paddle-onnx rec 폭 하한)
|
||||
|
||||
### 무엇이 문제였나
|
||||
|
||||
`OnnxPaddleOcr::run_rec` 은 검출된 박스를 높이 48 로 리사이즈하면서 **폭을 1 이상으로만** 보장했다. 그런데 PP-OCRv5 rec 백본은 폭을 반복해서 줄이기 때문에 폭이 한 자릿수인 입력은 도중에 폭 0 인 특징맵이 되고, ORT 가 Conv 에서 세션 실행 전체를 실패시킨다.
|
||||
|
||||
그 실패가 `recognize` 의 `?` 를 타고 나가면서, **이미 인식해 둔 나머지 박스가 전부 함께 버려졌다.** 바로 아래 줄(`if text.is_empty() { continue; }`)이 "박스 하나가 비면 나머지는 살린다"는 의도를 담고 있는데 오류 경로에만 그 방어가 없었던 것이다. 손실이 얇은 조각 하나에서 그치지 않고 그 이미지 전체로 번졌다.
|
||||
|
||||
색인 자체는 성공으로 끝나므로 **조용한 손실**이었다. 이미지 문서는 본문에 파일명만 남고, 스캔 PDF 페이지는 청크가 0 이 된다. 검색해서 0 건이 나올 때까지 드러나지 않는다.
|
||||
|
||||
### 하한은 16 이 아니라 5 였다 — 실측
|
||||
|
||||
이슈는 `REC_MIN_WIDTH = 16` 을 제안했지만, 번들된 `korean_ppocrv5_mobile_rec.onnx` 를 직접 스윕해 보면 **실제 한계는 4** 다. 세 번의 독립 측정(서로 다른 조사 레인 + 검증 레인)이 같은 표를 냈다.
|
||||
|
||||
| rec 입력 폭 (높이 48) | 결과 |
|
||||
|---|---|
|
||||
| 1 ~ 4 | 전부 실패 — `Invalid input shape: {1,0}` |
|
||||
| 5 ~ 48 | 전부 성공 |
|
||||
|
||||
경계는 흐릿하지 않고 딱 떨어지며, 픽셀 내용과 무관하다(흰 이미지·체커보드·글자 모양 렌더 모두 동일). 이유도 유도된다: rec 출력의 CTC 타임스텝 수가 `T = ceil((w - 4) / 8)` 이라서 `w <= 4` 에서 `T = 0` 이 되고, 오류 메시지의 `{1,0}` 이 바로 그 0 이다. 실측한 44 개 폭 전부가 이 식에 맞았다(`w=12 → T=1`, `w=13 → T=2`, `w=45 → T=6`).
|
||||
|
||||
그래서 상수를 **5** 로 잡았다. 16 을 쓰면 폭 5~15 구간을 새로 버리게 되는데, 그 구간은 지금 정상 동작하는 범위다. 다만 그 구간이 실제로 글자를 뱉는지도 따로 재 봤고 — 폭 40 대조군은 `"1"`(0.97) / `"3"`(0.999) / `"7"`(0.998) 을 제대로 읽는데 5~16 구간은 잉크가 있어도 전부 빈 문자열이었다 — 즉 **16 을 써도 잃는 건 사실상 없다.** 두 값의 실질 차이는 없고, 5 를 고른 이유는 그것이 그래프의 진짜 하한이라서 상수의 이름과 주석이 거짓이 되지 않기 때문이다.
|
||||
|
||||
`crop.width()` 가 아니라 리사이즈 **이후**의 `new_w` 를 검사한다. 3×200 짜리 크롭은 폭이 3 이지만 `new_w` 가 1 이고, 3×10 크롭은 같은 폭 3 인데 `new_w` 가 14 다. 네트워크가 보는 값은 `new_w` 뿐이다.
|
||||
|
||||
### 회귀 테스트는 양쪽을 다 고정한다
|
||||
|
||||
`rec_min_width_is_the_graph_floor` (`paddle_onnx.rs` 의 `mod tests`) 는 두 방향을 모두 단언한다.
|
||||
|
||||
- `1..REC_MIN_WIDTH` 는 세션에 닿지 않고 빈 문자열로 빠져야 한다 → **가드를 지우면 실패**한다. 실제로 가드만 지우고 돌려 확인했다: `w=1 must never reach the rec session: rec session run: ... Invalid input shape: {1,0}`.
|
||||
- 정확히 `REC_MIN_WIDTH` 는 **진짜 세션을 통과**해야 한다 → 상수가 모델의 실제 하한보다 낮으면(예: 모델 교체 후) 실패한다.
|
||||
|
||||
아래쪽만 있는 테스트는 상수가 너무 낮아도 통과해 버린다. 위쪽 단언이 상수를 배포된 모델에 붙들어 두는 역할을 한다. 모델 에셋이 in-tree 로 커밋돼 있으므로(`git ls-files crates/kebab-parse-image/assets/`) skip 가드는 붙이지 않았다 — 조용히 안 도는 테스트가 이 이슈가 경고하는 바로 그 함정이다.
|
||||
|
||||
### 실측 (도그푸딩 말뭉치 이미지 240 장)
|
||||
|
||||
`corpus/images/` 전량을 수정 전후로 같은 엔진(`ppocrv5-mobile-kor-1b55f062d055`)·같은 설정(score 0.3 / unclip 1.5 / max_boxes 1000 / max_pixels 2048)으로 돌렸다.
|
||||
|
||||
| | 수정 전 | 수정 후 |
|
||||
|---|---|---|
|
||||
| OCR 성공 | 204 / 240 | **240 / 240** |
|
||||
| OCR 실패 | 36 (15.0%) | **0** |
|
||||
|
||||
실패 36 장은 charts 8 · english-text 10 · korean-text 9 · photos 9 로, 특정 종류에 몰려 있지 않았다. 되살아난 본문은 합계 11,747 자(중앙값 37 자, 최대 3,950 자)다. 한 장이 얼마나 크게 손해 보고 있었는지가 드러나는 예:
|
||||
|
||||
| 이미지 | 수정 후 |
|
||||
|---|---|
|
||||
| `charts/Bassano_Politi___1505___Questio_de_modalibus___diagrams.jpg` | 3,950 자 |
|
||||
| `charts/Clickpath_Analysis.png` | 1,116 자 / 87 영역 |
|
||||
| `korean-text/연고한2.jpg` | 1,088 자 |
|
||||
|
||||
`Clickpath_Analysis.png` 은 얇은 조각 하나 때문에 **이미 인식된 87 개 영역**을 통째로 잃고 있었다.
|
||||
|
||||
부작용이 없다는 것도 확인했다: 원래 성공하던 204 장의 인식 글자 수가 **한 장도 변하지 않았다**. 이 수정은 순수 가산이다.
|
||||
|
||||
36 장 중 4 장은 수정 후에도 0 자인데, 글자가 없는 사진(예: `photos/Aphid_2007_1.jpg`, 진딧물 접사)이라 정상이다. 이 이슈로 인한 손실과 "원래 글자가 없어서 비는 문서"를 혼동하면 안 된다 — 성공한 204 장 중에서도 31 장은 det 가 박스를 못 찾아 정상적으로 0 자다. KB 쪽에서 영향 문서를 셀 때는 본문 길이가 아니라 `provenance_json LIKE '%Invalid input shape%'` 로 걸러야 한다.
|
||||
|
||||
### 재색인: 버전 두 개를 올렸다
|
||||
|
||||
코드만 고치면 이미 색인된 문서에는 닿지 않는다. 이미지 문서의 실효 `parser_version` 은 `image-meta-v1|chunk:…|ocr:1:paddle-onnx:<모델 blake3>` 인데, 이미지 파일도 모델 에셋도 안 바뀌었으니 서명이 동일하고 `try_skip_unchanged` 가 Unchanged 로 건너뛴다. #232 에서 세운 원칙 그대로다 — 사용자가 `--force-reingest` 를 떠올려야만 고쳐지는 수정은 고쳐진 게 아니다.
|
||||
|
||||
- `image-meta-v1` → **`image-meta-v2`**
|
||||
- `pdf-text-v2` → **`pdf-text-v3`** (스캔 PDF 도 같은 `run_rec` 을 타므로 같은 손실을 겪었다)
|
||||
|
||||
**재처리 비용은 생각보다 싸다.** OCR 산출물은 `derivation_cache` 에 **소스 바이트** 키로 들어가 있어서(§3.4, v0.31.0 #217) `parser_version` 캐스케이드와 분리돼 있다. 그리고 실패한 OCR 은 캐시에 **저장되지 않는다** — `Err` 분기가 `derivation_cache_put` 앞에서 빠져나간다(이미지 `ingest.rs:1750`, PDF `pdf_ocr_apply.rs:475`). 그래서 다음 `kebab ingest` 는 모든 이미지·PDF 문서를 다시 추출하되, 이미 성공했던 것들은 OCR 캐시에 히트해 비싼 엔진 호출을 건너뛰고, **실제로 다시 OCR 되는 건 이 버그로 실패했던 문서뿐**이다.
|
||||
|
||||
### 스냅샷 낙수
|
||||
|
||||
`pdf-text-v3` 는 `vector_pdf_canonical.json` 을 움직인다. #232 때와 같은 형태임을 확인했다 — 바뀐 것은 **파생 식별자와 버전 문자열뿐**이고 본문 텍스트·inlines·`source_span`·metadata 는 동일하다.
|
||||
|
||||
| 필드 | v2 | v3 |
|
||||
|---|---|---|
|
||||
| `doc_id` | `bd04fa01…` | `960e4c44…` |
|
||||
| `block_id` | `964162ec…` | `b85b8a6d…` |
|
||||
| `parser_version` | `pdf-text-v2` | `pdf-text-v3` |
|
||||
| provenance note | `parser_version=pdf-text-v2; …` | `parser_version=pdf-text-v3; …` |
|
||||
|
||||
`kebab-parse-pdf/tests/extractor.rs` 와 `kebab-app/tests/pdf_pipeline.rs` 의 버전 단언 세 곳도 함께 옮겼다. 이미지 쪽은 스냅샷이 상수에서 값을 유도하고 있어 손댈 것이 없었다.
|
||||
|
||||
### 고치지 않고 남긴 것
|
||||
|
||||
`recognize` 안의 `self.run_rec(&crop)?` (`paddle_onnx.rs:299`) 는 그대로 뒀다. 폭 가드가 알려진 유일한 방아쇠를 닫았고, `T = ceil((w-4)/8) >= 1` 이 `w >= 5` 에서 항상 성립하므로 폭 때문에 이 경로가 다시 터지는 일은 없다. 여기를 박스 단위 `continue` 로 바꾸면 바로 아래 클래스 수 검사(`rec output has {c} classes`)까지 함께 삼키게 되는데, 그건 입력과 무관한 **설정 오류**(rec 모델과 dict 짝이 안 맞음)라서 지금처럼 즉시 실패하는 편이 맞다. 조용한 빈 OCR 로 바꿀 이유가 없다.
|
||||
|
||||
### 이슈 본문과 다른 점 하나
|
||||
|
||||
이슈는 오류의 노드 이름을 `Conv.33` 으로 적었는데, 이 머신에서는 같은 실패가 `p2o.pd_op.batch_norm_.1.0_nchwc` 로 나온다. ORT 가 CPU 명령어 집합에 맞춰 그래프를 최적화(NCHWc 레이아웃 변환)하면서 노드 이름을 다시 붙이기 때문이고, 상태 메시지와 `{1,0}` 은 동일하다. 다른 버그가 아니다.
|
||||
|
||||
## 2026-08-17 — #232 PDF OCR 이 DCTDecode 아닌 스캔본을 전량 건너뜀 (페이지 렌더링)
|
||||
|
||||
### 무엇이 문제였나
|
||||
|
||||
Reference in New Issue
Block a user