feat(nli): fb-41 PR-9b — OnnxNliVerifier 의 ONNX inference + model download
- OnnxNliVerifier fields: model_id, cache_dir (XDG model_dir/nli/<sanitized>), session/tokenizer OnceLock. - new(): eager cache_dir stamp만 — actual model download + Session::commit_from_file 는 첫 score 호출 시 ensure_loaded() 가 lazy 수행. - score(): ensure_loaded → tokenizer.encode(pair, OnlyFirst truncation max_length=512) → ndarray Array2<i64> → ort::Session::run → logits[1,3] → NliScores::from_xnli_logits. - empty hypothesis edge: defense-in-depth bail (spec §2.3 의 caller-side skip 외 추가). - sanitize_model_id helper: "/" → "_". - 5 #[ignore] integration tests (EN self-entailment, EN unrelated, KR entailment, long premise truncation, empty hypothesis err) — manual smoke 가 PR description 첨부. Cargo.toml: `download-binaries` feature 를 kebab-nli 의 ort dep 에 활성화 (PR-9b prep commit 의 후속). 단독 `cargo test -p kebab-nli` 의 per-crate feature 유니온은 fastembed 없이 ort/download-binaries 가 OFF 되어 ort-sys link 가 실패 — kebab-nli 측에서 명시적으로 켜 줘야 standalone build 가 ONNX 런타임 link 됨. workspace 전체 빌드에서는 fastembed 의 동일 opt-in 과 union 되어 부작용 없음. Verification: - cargo test -p kebab-nli -j 1 — PR-9a 의 6 unit pass (`score_returns_err_in_skeleton` → `score_empty_hypothesis_returns_err` 로 stub→실 path 갱신, 갯수 유지). - cargo clippy -p kebab-nli --all-targets -- -D warnings clean. - cargo build --workspace -j 1 — 회귀 0. - Manual --ignored smoke 결과 PR body 첨부. Wire 영향: 없음 (crate-internal). Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
140
crates/kebab-nli/tests/inference.rs
Normal file
140
crates/kebab-nli/tests/inference.rs
Normal file
@@ -0,0 +1,140 @@
|
||||
//! Integration tests for `OnnxNliVerifier` against the real
|
||||
//! mDeBERTa-v3 XNLI model. Every test is `#[ignore]` — plain
|
||||
//! `cargo test -p kebab-nli` skips them; run explicitly with
|
||||
//! `cargo test -p kebab-nli --test inference -- --ignored` to
|
||||
//! exercise the (slow + network-bound on first run) inference path.
|
||||
//!
|
||||
//! First test in the file triggers the ~280 MB ONNX + ~16 MB
|
||||
//! tokenizer download into `config.storage.model_dir/nli/...`;
|
||||
//! subsequent tests hit the OnceLock cache for free.
|
||||
|
||||
use kebab_config::Config;
|
||||
use kebab_nli::{NliVerifier, OnnxNliVerifier};
|
||||
|
||||
/// Test 1: an English statement entails itself with high confidence.
|
||||
/// Smoke evidence captured for the PR description's `## 검증` section.
|
||||
#[test]
|
||||
#[ignore]
|
||||
fn en_self_entailment_high_score() {
|
||||
let cfg = Config::defaults();
|
||||
let v = OnnxNliVerifier::new(&cfg).expect("verifier construction");
|
||||
let premise = "Caffeine is a stimulant.";
|
||||
let hypothesis = "Caffeine is a stimulant.";
|
||||
let s = v.score(premise, hypothesis).expect("score should succeed");
|
||||
eprintln!(
|
||||
"[test1 en_self_entailment_high_score] premise={premise:?} hypothesis={hypothesis:?} \
|
||||
scores: entailment={:.4}, neutral={:.4}, contradiction={:.4}",
|
||||
s.entailment, s.neutral, s.contradiction
|
||||
);
|
||||
assert!(
|
||||
s.entailment > 0.8,
|
||||
"expected entailment > 0.8, got {:.4} (full scores: {:?})",
|
||||
s.entailment,
|
||||
s
|
||||
);
|
||||
}
|
||||
|
||||
/// Test 2: an unrelated chemistry fact does NOT entail the premise.
|
||||
/// Entailment should be low — neutral / contradiction wins.
|
||||
#[test]
|
||||
#[ignore]
|
||||
fn en_unrelated_low_entailment() {
|
||||
let cfg = Config::defaults();
|
||||
let v = OnnxNliVerifier::new(&cfg).expect("verifier construction");
|
||||
let premise = "Caffeine is a stimulant.";
|
||||
let hypothesis = "The chemical formula of caffeine is C8H10N4O2.";
|
||||
let s = v.score(premise, hypothesis).expect("score should succeed");
|
||||
eprintln!(
|
||||
"[test2 en_unrelated_low_entailment] \
|
||||
scores: entailment={:.4}, neutral={:.4}, contradiction={:.4}",
|
||||
s.entailment, s.neutral, s.contradiction
|
||||
);
|
||||
assert!(
|
||||
s.entailment < 0.3,
|
||||
"expected entailment < 0.3, got {:.4} (full scores: {:?})",
|
||||
s.entailment,
|
||||
s
|
||||
);
|
||||
}
|
||||
|
||||
/// Test 3: Korean entailment. The threshold is intentionally generous
|
||||
/// (> 0.5) because cross-lingual XNLI is noisier than English-only.
|
||||
#[test]
|
||||
#[ignore]
|
||||
fn ko_entailment_high_score() {
|
||||
let cfg = Config::defaults();
|
||||
let v = OnnxNliVerifier::new(&cfg).expect("verifier construction");
|
||||
let premise = "사과는 빨갛다.";
|
||||
let hypothesis = "사과는 색이 있다.";
|
||||
let s = v.score(premise, hypothesis).expect("score should succeed");
|
||||
eprintln!(
|
||||
"[test3 ko_entailment_high_score] \
|
||||
scores: entailment={:.4}, neutral={:.4}, contradiction={:.4}",
|
||||
s.entailment, s.neutral, s.contradiction
|
||||
);
|
||||
assert!(
|
||||
s.entailment > 0.5,
|
||||
"expected entailment > 0.5, got {:.4} (full scores: {:?})",
|
||||
s.entailment,
|
||||
s
|
||||
);
|
||||
}
|
||||
|
||||
/// Test 4: a > 24 000-char premise must not panic. mDeBERTa-v3 is
|
||||
/// trained at 512 tokens; the `OnlyFirst` truncation strategy keeps
|
||||
/// the premise side from blowing the positional embedding cap.
|
||||
#[test]
|
||||
#[ignore]
|
||||
fn long_premise_truncates_without_panic() {
|
||||
let cfg = Config::defaults();
|
||||
let v = OnnxNliVerifier::new(&cfg).expect("verifier construction");
|
||||
let premise = "foo bar baz ".repeat(2000); // ~24 000 chars
|
||||
let hypothesis = "foo";
|
||||
let s = v
|
||||
.score(&premise, hypothesis)
|
||||
.expect("score should succeed on long premise");
|
||||
eprintln!(
|
||||
"[test4 long_premise_truncates_without_panic] premise_len={} \
|
||||
scores: entailment={:.4}, neutral={:.4}, contradiction={:.4}",
|
||||
premise.len(),
|
||||
s.entailment,
|
||||
s.neutral,
|
||||
s.contradiction
|
||||
);
|
||||
// No NaN / infinity in any channel.
|
||||
for (name, x) in [
|
||||
("entailment", s.entailment),
|
||||
("neutral", s.neutral),
|
||||
("contradiction", s.contradiction),
|
||||
] {
|
||||
assert!(
|
||||
x.is_finite(),
|
||||
"channel {name} non-finite: {x} (full scores: {:?})",
|
||||
s
|
||||
);
|
||||
}
|
||||
// Softmax invariant — the three channels sum to ~1.
|
||||
let sum = s.entailment + s.neutral + s.contradiction;
|
||||
assert!(
|
||||
(sum - 1.0).abs() < 1e-3,
|
||||
"softmax channels must sum to ~1, got {sum:.6}"
|
||||
);
|
||||
}
|
||||
|
||||
/// Test 5: an empty hypothesis triggers the defense-in-depth bail
|
||||
/// path BEFORE the tokenizer runs. Hits no network — fast, even on
|
||||
/// a fresh machine.
|
||||
#[test]
|
||||
#[ignore]
|
||||
fn empty_hypothesis_returns_err() {
|
||||
let cfg = Config::defaults();
|
||||
let v = OnnxNliVerifier::new(&cfg).expect("verifier construction");
|
||||
let err = v
|
||||
.score("anything", "")
|
||||
.expect_err("empty hypothesis must error");
|
||||
let msg = err.to_string();
|
||||
assert!(
|
||||
msg.contains("empty hypothesis"),
|
||||
"expected 'empty hypothesis' in error, got: {msg}"
|
||||
);
|
||||
}
|
||||
Reference in New Issue
Block a user