kebab-search/tests/lexical.rs 의 alias 채널 테스트 + insert_chunk_with_aliases 헬퍼 제거(body 회수 회귀 테스트로 대체). Chunk 리터럴 aliases: None 제거 (embedding_records_fk/idempotency/inspect). chunk 스냅샷 fixture 의 aliases 키 제거. config_migrate 는 ingest.code 앵커로, corpus_revision/search_lexical 주석은 V013 비-bump 명시로 갱신. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
91 lines
3.3 KiB
JSON
91 lines
3.3 KiB
JSON
[
|
|
{
|
|
"block_ids": [
|
|
"8149e12ca002489acb4a0f74c97a061a"
|
|
],
|
|
"chunk_id": "ec3cf06ae56c8e9796bbc9196438b7c5",
|
|
"chunker_version": "code-c-ast-v1",
|
|
"doc_id": "6bec42dd593920a060541db16c4e8e45",
|
|
"heading_path": [],
|
|
"policy_hash": "ecfad2ec1223662d",
|
|
"source_spans": [
|
|
{
|
|
"kind": "code",
|
|
"lang": "c",
|
|
"line_end": 18,
|
|
"line_start": 1,
|
|
"symbol": "<top-level>"
|
|
}
|
|
],
|
|
"text": "#include <stdio.h>\n#include <stdlib.h>\n\n#define MAX_BUF 4096\n\ntypedef enum {\n OK = 0,\n ERR_PARSE,\n ERR_IO,\n} status_t;\n\ntypedef struct {\n int id;\n char name[64];\n status_t status;\n} record_t;\n\nstatic int counter = 0;",
|
|
"token_estimate": 78,
|
|
"tokenized_korean_text": "# include < stdio . h > # include < stdlib . h > # define MAX _ BUF 4096 typedef enum { OK = 0 , ERR _ PARSE , ERR _ IO , } status _ t ; typedef struct { int id ; char name [ 64 ]; status _ t status ; } record _ t ; static int counter = 0 ;"
|
|
},
|
|
{
|
|
"block_ids": [
|
|
"1baaa89f21a47b2f32d6396a24a85454"
|
|
],
|
|
"chunk_id": "c2d7a81c898106733ef2e703774a6a4a",
|
|
"chunker_version": "code-c-ast-v1",
|
|
"doc_id": "6bec42dd593920a060541db16c4e8e45",
|
|
"heading_path": [],
|
|
"policy_hash": "ecfad2ec1223662d",
|
|
"source_spans": [
|
|
{
|
|
"kind": "code",
|
|
"lang": "c",
|
|
"line_end": 23,
|
|
"line_start": 20,
|
|
"symbol": "parse_record"
|
|
}
|
|
],
|
|
"text": "int parse_record(const char *line, record_t *out) {\n if (line == NULL || out == NULL) return ERR_PARSE;\n return OK;\n}",
|
|
"token_estimate": 41,
|
|
"tokenized_korean_text": "int parse _ record ( const char * line , record _ t * out ) { if ( line == NULL || out == NULL ) return ERR _ PARSE ; return OK ; }"
|
|
},
|
|
{
|
|
"block_ids": [
|
|
"8d0e14cbcc6d1e92d7878ab796ea68b8"
|
|
],
|
|
"chunk_id": "0e4d7b131ab64eba03b51903b5d8f96d",
|
|
"chunker_version": "code-c-ast-v1",
|
|
"doc_id": "6bec42dd593920a060541db16c4e8e45",
|
|
"heading_path": [],
|
|
"policy_hash": "ecfad2ec1223662d",
|
|
"source_spans": [
|
|
{
|
|
"kind": "code",
|
|
"lang": "c",
|
|
"line_end": 27,
|
|
"line_start": 25,
|
|
"symbol": "print_record"
|
|
}
|
|
],
|
|
"text": "void print_record(const record_t *r) {\n printf(\"[%d] %s (status=%d)\\n\", r->id, r->name, r->status);\n}",
|
|
"token_estimate": 35,
|
|
"tokenized_korean_text": "void print _ record ( const record _ t * r ) { printf (\"[% d ] % s ( status =% d )\\ n \", r -> id , r -> name , r -> status ); }"
|
|
},
|
|
{
|
|
"block_ids": [
|
|
"9c2ede84423871b615d48c38fefb1853"
|
|
],
|
|
"chunk_id": "e076f8edb2ff141d7e99b4106bb95157",
|
|
"chunker_version": "code-c-ast-v1",
|
|
"doc_id": "6bec42dd593920a060541db16c4e8e45",
|
|
"heading_path": [],
|
|
"policy_hash": "ecfad2ec1223662d",
|
|
"source_spans": [
|
|
{
|
|
"kind": "code",
|
|
"lang": "c",
|
|
"line_end": 33,
|
|
"line_start": 29,
|
|
"symbol": "main"
|
|
}
|
|
],
|
|
"text": "int main(void) {\n record_t r = { .id = 1, .name = \"foo\", .status = OK };\n print_record(&r);\n return 0;\n}",
|
|
"token_estimate": 38,
|
|
"tokenized_korean_text": "int main ( void ) { record _ t r = { . id = 1 , . name = \" foo \", . status = OK }; print _ record (& r ); return 0 ; }"
|
|
}
|
|
]
|