diff --git a/docs/corpus-analysis/NIKL_2025_V1.md b/docs/corpus-analysis/NIKL_2025_V1.md new file mode 100644 index 00000000..fb5903d0 --- /dev/null +++ b/docs/corpus-analysis/NIKL_2025_V1.md @@ -0,0 +1,3845 @@ +# NIKL 2025 v1.0 corpus analysis + +> Generated by `cargo run --release -p braillify --example nikl_corpus_analyze`. The tool reads only `input` and `unicode`; it never loads or compares the read-only `world` or `jeomsarang` fields. + +## Current measurement + +| Metric | Count | +|---|---:| +| Total | 83528 | +| Exact | 75785 | +| Mismatch | 7743 | +| Exact accuracy | 90.73% | +| Duplicate records | 0 | +| Inputs with conflicting references | 0 | + +## Classification policy + +Primary classes are evidence gates, not permissions to change the engine. `implementation_defect` is restricted to defects independently confirmed from the PDF (currently the rules 28/29 roman-indicator ordering signature). `unsupported_character_review` contains encoding failures fully explained by one or more singleton characters whose support obligation has not been confirmed from the PDF. `unclassified_encoding_error_review` contains other encoding failures until a PDF-backed implementation obligation or a reproducible comparison/corpus issue is established. `pending_rule_review` contains foreign-text, number, punctuation, and Korean candidates that have not yet been resolved against the PDF. `corpus_suspect` is reserved for independently detectable contradictions: conflicting duplicate references or a localized reference-cell signature that contradicts an explicit PDF/UEB rule. `comparison_method` requires equality after a named normalization. + +| Primary class | Count | +|---|---:| +| `corpus_suspect` | 1234 | +| `exact` | 75785 | +| `pending_rule_review` | 6505 | +| `unsupported_character_review` | 4 | + +| Reproducible reason | Count | +|---|---:| +| `exact` | 75785 | +| `foreign_text_rule_review` | 6456 | +| `number_rule_review` | 45 | +| `punctuation_rule_review` | 4 | +| `roman_ellipsis_uses_korean_cells_in_roman_enclosure` | 1 | +| `rule34_roman_indicator_before_opening_parenthesis` | 1160 | +| `ueb_capitalized_passage_written_as_separate_capital_words` | 3 | +| `ueb_grade1_before_nonstanding_opening_parenthesis` | 70 | +| `unsupported_character_review` | 4 | + +## Pending first-difference cell transitions + +This ranking is a diagnostic selector, not an implementation rule. It counts only current `pending_rule_review` cases whose encoder call succeeded, keyed by the expected and actual cell at the sentence's first differing position. Candidate implementation work must still bind a transition to a localized input structure, exact controls, and independent PDF evidence. + +| Rank | Expected → actual first cell | Cases | +|---:|---|---:| +| 1 | `U+2815 ⠕ -> U+2833 ⠳` | 1509 | +| 2 | `U+2810 ⠐ -> U+2832 ⠲` | 648 | +| 3 | `U+280E ⠎ -> U+280C ⠌` | 473 | +| 4 | `U+2801 ⠁ -> U+281C ⠜` | 413 | +| 5 | `U+2811 ⠑ -> U+282B ⠫` | 344 | +| 6 | `U+281B ⠛ -> U+2823 ⠣` | 147 | +| 7 | `U+2826 ⠦ -> U+2834 ⠴` | 122 | +| 8 | `U+2811 ⠑ -> U+283B ⠻` | 101 | +| 9 | `U+2800 ⠀ -> U+2832 ⠲` | 88 | +| 10 | `U+2826 ⠦ -> U+2810 ⠐` | 84 | +| 11 | `U+280E ⠎ -> U+2829 ⠩` | 79 | +| 12 | `U+2811 ⠑ -> U+2822 ⠢` | 68 | +| 13 | `U+2810 ⠐ -> U+2815 ⠕` | 65 | +| 14 | `U+2820 ⠠ -> U+2830 ⠰` | 64 | +| 15 | `U+2826 ⠦ -> U+2800 ⠀` | 63 | +| 16 | `U+2820 ⠠ -> U+2834 ⠴` | 61 | +| 17 | `U+280A ⠊ -> U+2814 ⠔` | 60 | +| 18 | `U+283C ⠼ -> U+2800 ⠀` | 58 | +| 19 | `U+2834 ⠴ -> U+2800 ⠀` | 55 | +| 20 | `U+2824 ⠤ -> U+2800 ⠀` | 54 | + +### `U+2815 ⠕ -> U+2833 ⠳` + +- `sentence_01.json` #47: 다날은 계열사 ‘제프’가 국내 대체불가토큰(NFT) 거래소를 운영하는 ‘팔라’와 메타버스·NFT 협력 관련 협약(MOU)을 맺고 메타버스 플랫폼 ‘제프월드’의 인프라 확대를 추진한다고 3일 밝혔다. + - expected: `⠜⠁⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠑⠗⠅⠈⠥⠀⠑⠝⠓⠘` + - actual: `⠜⠁⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠑⠗⠅⠈⠥⠀⠑⠝⠓⠘⠎` + - first differing cell (zero-based): 114 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #128: 이번 방문에서 대표단은 우수 외투기업과 투자협약(MOU)을 체결하고 투자 상담, 기업정보 교류 등 적극적인 외자 유치 활동을 펼칠 계획이다. + - expected: `⠜⠁⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠈⠥⠀⠓⠍` + - actual: `⠜⠁⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠈⠥⠀⠓⠍⠨` + - first differing cell (zero-based): 48 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #1: 김 지사는 18일(이하 현지시각) 미국 코네티컷주 댄버리 린데 본사에서 산지브 람바(Sanjiv Lamba) 린데 회장, 성백석 린데코리아 회장, 조일교 아산시 부시장과 투자양해각서(MOU)을 체결했다. + - expected: `⠠⠎⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - actual: `⠠⠎⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - first differing cell (zero-based): 175 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #191: SK온은 23일 서울 종로구 SK서린사옥에서 에코프로머티리얼즈, 중국의 GEM(거린메이)과 전구체 생산을 위한 3자 합작법인인 지이엠코리아뉴에너지머티리얼즈㈜(이하 지이엠코리아) 설립 양해각서(MOU)를 체결했다고 밝혔다. + - expected: `⠠⠎⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠐⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈` + - actual: `⠠⠎⠦⠄⠴⠠⠠⠍⠳⠠⠴⠐⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈⠥` + - first differing cell (zero-based): 191 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `U+2810 ⠐ -> U+2832 ⠲` + +- `sentence_01.json` #57: 아키에이지 워는 원작 아키에이지 개발사 엑스엘게임즈(각자대표 송재경, 최관호)가 개발 중인 신작 PC·모바일 크로스플랫폼 대규모다중접속역할수행게임(MMORPG)이다. 원작의 향수를 자극하는 스토리와 캐릭터, 언리얼 엔진4를 활용한 고퀄리티 그래픽이 특징이다. + - expected: `⠨⠁⠀⠴⠠⠠⠏⠉⠐⠆⠑⠥⠘⠣⠕⠂⠀⠋⠪⠐⠥⠠⠪⠙` + - actual: `⠨⠁⠀⠴⠠⠠⠏⠉⠲⠐⠆⠑⠥⠘⠣⠕⠂⠀⠋⠪⠐⠥⠠⠪` + - first differing cell (zero-based): 92 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #10: 설동호 대전교육감은 “학교 교육과정 및 여건에 맞는 인공지능(AI)교육 선도학교 운영을 통해 AI·SW교육 기반 구축과 학생들과 교원들의 미래 디지털 역량이 함양될 것이다.”라고 말했다. + - expected: `⠚⠗⠀⠴⠠⠠⠁⠊⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕⠘` + - actual: `⠚⠗⠀⠴⠠⠠⠁⠊⠲⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕` + - first differing cell (zero-based): 93 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #91: 지난 13일 브라질에서 우수제조관리기준(BGMP) 인증을 받으며 중남미시장 진출에도 첫발을 내디뎠다. BGMP 인증으로 HA·Ca필러가 브라질을 비롯한 남미 국가들에 진입하는 데 속도가 붙을 것이란 전망이다. + - expected: `⠐⠥⠀⠴⠠⠠⠓⠁⠐⠆⠴⠠⠉⠁⠲⠙⠕⠂⠐⠎⠫⠀⠘⠪` + - actual: `⠐⠥⠀⠴⠠⠠⠓⠁⠲⠐⠆⠴⠠⠉⠁⠲⠙⠕⠂⠐⠎⠫⠀⠘` + - first differing cell (zero-based): 121 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #49: 유럽중앙은행(ECB)이 크레디트스위스(CS)의 유동성 위기에도 ‘빅스텝’(기준금리를 한 번에 0.5%포인트 인상)을 단행하자, 미국 연방준비제도(Fed·연준) 역시 금리 인상을 멈추지 않을 것이라는 전망이 힘을 얻고 있다. + - expected: `⠊⠥⠦⠄⠴⠠⠋⠫⠐⠆⠡⠨⠛⠠⠴⠀⠱⠁⠠⠕⠀⠈⠪⠢` + - actual: `⠊⠥⠦⠄⠴⠠⠋⠫⠲⠐⠆⠡⠨⠛⠠⠴⠀⠱⠁⠠⠕⠀⠈⠪` + - first differing cell (zero-based): 149 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `U+280E ⠎ -> U+280C ⠌` + +- `sentence_01.json` #167: 올해 전 세계 반도체 시장 전망도 암울하다. 세계반도체시장통계기구(WSTS)에 따르면 올해 글로벌 반도체 매출은 5천565억 달러(약 706조5천880억원)로 지난해(5천801억달러) 대비 4.1% 감소할 것으로 조사됐다. + - expected: `⠈⠍⠦⠄⠴⠠⠠⠺⠎⠞⠎⠠⠴⠝⠀⠠⠊⠐⠪⠑⠡⠀⠥⠂` + - actual: `⠈⠍⠦⠄⠴⠠⠠⠺⠌⠎⠠⠴⠝⠀⠠⠊⠐⠪⠑⠡⠀⠥⠂⠚` + - first differing cell (zero-based): 67 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #244: 충남대학교 이진숙 총장을 비롯한 관계자들은 5월 19일, 베트남 최고의 국립대인 하노이과학기술대학(HUST), 베트남국립농업대학(VNUA)을 연이어 방문해 ‘오픈 캠퍼스(open campus)’ 설립을 골자로 한 협정을 체결했다. + - expected: `⠁⠦⠄⠴⠠⠠⠓⠥⠎⠞⠠⠴⠐⠀⠘⠝⠓⠪⠉⠢⠈⠍⠁⠐` + - actual: `⠁⠦⠄⠴⠠⠠⠓⠥⠌⠠⠴⠐⠀⠘⠝⠓⠪⠉⠢⠈⠍⠁⠐⠕` + - first differing cell (zero-based): 101 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #142: 수많은 작품 OST를 책임지며 실력파 프로듀서로 인정받고 있는 필승불패W, 리디아(Lydia), 장석원이 힘을 합쳐 만든 곡으로 웰메이드 OST 탄생을 예감케 한다. + - expected: `⠙⠍⠢⠀⠴⠠⠠⠕⠎⠞⠲⠐⠮⠀⠰⠗⠁⠕⠢⠨⠕⠑⠱⠀` + - actual: `⠙⠍⠢⠀⠴⠠⠠⠕⠌⠲⠐⠮⠀⠰⠗⠁⠕⠢⠨⠕⠑⠱⠀⠠` + - first differing cell (zero-based): 17 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #375: 윤석열 대통령은 30일 캐서린 타이 미국 무역대표부(USTR) 대표를 만나 미 인플레이션감축법(IRA)·반도체지원법(CHIPS Act)과 관련해 “미국에 진출하는 한국 기업들이 어려움을 겪지 않도록 우호적인 방향으로 배려해 달라”고 당부했다. + - expected: `⠘⠍⠦⠄⠴⠠⠠⠥⠎⠞⠗⠠⠴⠀⠊⠗⠙⠬⠐⠮⠀⠑⠒⠉` + - actual: `⠘⠍⠦⠄⠴⠠⠠⠥⠌⠗⠠⠴⠀⠊⠗⠙⠬⠐⠮⠀⠑⠒⠉⠀` + - first differing cell (zero-based): 53 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `U+2801 ⠁ -> U+281C ⠜` + +- `sentence_01.json` #142: ‘메타버스존’에서는 증강현실(AR)·가상현실(VR)에 적용되는 혁신기술이 소개된다. 글라스를 착용하고 최첨단 3D 센싱모듈이 구현하는 가상현실을 관람객이 직접 체험할 수 있도록 했다. + - expected: `⠠⠕⠂⠦⠄⠴⠠⠠⠁⠗⠠⠴⠐⠆⠫⠇⠶⠚⠡⠠⠕⠂⠦⠄` + - actual: `⠠⠕⠂⠦⠄⠴⠠⠠⠜⠠⠴⠐⠆⠫⠇⠶⠚⠡⠠⠕⠂⠦⠄⠴` + - first differing cell (zero-based): 34 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #28: 체납된 세금은 전국 어디서나 은행 현금자동인출기(ATM)를 이용해 고지서 없이 현금과 신용카드로 납부가 가능하며, 자동응답시스템(ARS) 지방세 납부서비스(044-300-7114)를 이용해 가상계좌를 확인해 계좌이체를 하거나 신용카드로 납부할 수도 있다. + - expected: `⠓⠝⠢⠦⠄⠴⠠⠠⠁⠗⠎⠠⠴⠀⠨⠕⠘⠶⠠⠝⠀⠉⠃⠘` + - actual: `⠓⠝⠢⠦⠄⠴⠠⠠⠜⠎⠠⠴⠀⠨⠕⠘⠶⠠⠝⠀⠉⠃⠘⠍` + - first differing cell (zero-based): 129 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #52: 종합식품기업 하림이 공룡 모양으로 생긴 동그랑땡 ‘용가리 땡’을 20일 출시했다. 용가리땡에는 특별 제작한 증강현실(AR) 공룡 카드까지 한 장씩 들어 있어 아이들의 많은 관심이 예상된다. + - expected: `⠠⠕⠂⠦⠄⠴⠠⠠⠁⠗⠠⠴⠀⠈⠿⠐⠬⠶⠀⠋⠊⠪⠠⠫` + - actual: `⠠⠕⠂⠦⠄⠴⠠⠠⠜⠠⠴⠀⠈⠿⠐⠬⠶⠀⠋⠊⠪⠠⠫⠨` + - first differing cell (zero-based): 126 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #187: 주정차단속 ARS 알림서비스는 기존 불법 주정차 문자 알림서비스를 업그레이드 해 문자와 함께 자동응답서비스(ARS)로도 주정차단속을 알려주는 시스템이다. + - expected: `⠊⠒⠠⠭⠀⠴⠠⠠⠁⠗⠎⠲⠀⠣⠂⠐⠕⠢⠠⠎⠘⠕⠠⠪` + - actual: `⠊⠒⠠⠭⠀⠴⠠⠠⠜⠎⠲⠀⠣⠂⠐⠕⠢⠠⠎⠘⠕⠠⠪⠉` + - first differing cell (zero-based): 14 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `U+2811 ⠑ -> U+282B ⠫` + +- `sentence_01.json` #51: 와이캅 픽셀은 와이어·패키지·렌즈가 필요 없는 ‘와이캅’ 기술을 기반으로 한다. 적녹청(RGB) 3개의 마이크로 LED를 수직 방향으로 적층한 세계 최초의 풀 컬러 원칩 기술이다. + - expected: `⠪⠐⠥⠀⠴⠠⠠⠇⠑⠙⠲⠐⠮⠀⠠⠍⠨⠕⠁⠀⠘⠶⠚⠜` + - actual: `⠪⠐⠥⠀⠴⠠⠠⠇⠫⠲⠐⠮⠀⠠⠍⠨⠕⠁⠀⠘⠶⠚⠜⠶` + - first differing cell (zero-based): 107 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #33: 지원은 학생이 영재교육종합데이터베이스(GED)를 통해 원서 접수하고 교사의 추천을 받아 진행하고, 자세한 내용은 학교 및 영재교육 기관별 홈페이지에 탑재되어 있다. + - expected: `⠠⠪⠦⠄⠴⠠⠠⠛⠑⠙⠠⠴⠐⠮⠀⠓⠿⠚⠗⠀⠏⠒⠠⠎` + - actual: `⠠⠪⠦⠄⠴⠠⠠⠛⠫⠠⠴⠐⠮⠀⠓⠿⠚⠗⠀⠏⠒⠠⠎⠀` + - first differing cell (zero-based): 40 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #209: 포드의 글로벌 트럭 디자인 DNA를 토대로 강인하면서도 다양한 사용 목적에 부합하는 실용적인 내·외부를 구현했다. 전면부 시그니처 C클램프 헤드라이트, 매트릭스 LED 헤드라이트, 포드(FORD) 레터링이 특징이다. + - expected: `⠁⠠⠪⠀⠴⠠⠠⠇⠑⠙⠲⠀⠚⠝⠊⠪⠐⠣⠕⠓⠪⠐⠀⠙` + - actual: `⠁⠠⠪⠀⠴⠠⠠⠇⠫⠲⠀⠚⠝⠊⠪⠐⠣⠕⠓⠪⠐⠀⠙⠥` + - first differing cell (zero-based): 159 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #388: 김 신부는 그동안 제작한 스테인드글라스 작품은 물론 회화·LED(발광다이오드)조명작품·도자기 등 60여점의 작품을 전시한다. 그는 “형상을 떠난 자유로움과 원초적인 아름다움에 대한 깊이를 관람객들에게 전달하고 싶다”고 밝혔다. + - expected: `⠚⠧⠐⠆⠴⠠⠠⠇⠑⠙⠦⠄⠘⠂⠈⠧⠶⠊⠣⠕⠥⠊⠪⠠` + - actual: `⠚⠧⠐⠆⠴⠠⠠⠇⠫⠦⠄⠘⠂⠈⠧⠶⠊⠣⠕⠥⠊⠪⠠⠴` + - first differing cell (zero-based): 61 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `U+281B ⠛ -> U+2823 ⠣` + +- `sentence_01.json` #1753: 지난해 말 경기주택도시공사(GH)에서 퇴직한 전씨는 ‘성남FC 불법 후원금 의혹’과 관련해 검찰 조사를 받은 바 있으며 ‘GH 합숙소 의혹’에도 연루된 것으로 알려졌다. + - expected: `⠈⠿⠇⠦⠄⠴⠠⠠⠛⠓⠠⠴⠝⠠⠎⠀⠓⠽⠨⠕⠁⠚⠒⠀` + - actual: `⠈⠿⠇⠦⠄⠴⠠⠠⠣⠠⠴⠝⠠⠎⠀⠓⠽⠨⠕⠁⠚⠒⠀⠨` + - first differing cell (zero-based): 31 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #24: 대전대학교(총장 남상호)는 창업보육센터 입주기업인 ㈜티알(대표 김병수)이 최근 트랜스글로벌헬스케어(TGH)와 세계최초 AI 기반 만성폐쇄성폐질환(COPD) 진단기인 ‘The Spirokit’(더스피로킷)에 대한 물품공급 계약을 체결했다고 지난 14일 밝혔다. + - expected: `⠝⠎⠦⠄⠴⠠⠠⠞⠛⠓⠠⠴⠧⠀⠠⠝⠈⠌⠰⠽⠰⠥⠀⠴` + - actual: `⠝⠎⠦⠄⠴⠠⠠⠞⠣⠠⠴⠧⠀⠠⠝⠈⠌⠰⠽⠰⠥⠀⠴⠠` + - first differing cell (zero-based): 113 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #292: GH 주거 분야 전문인력과 HUG(주택도시보증공사), 변호사, 법무사 등 부동산·금융 전문인력이 상주하며 전세사기 피해자에게 부동산 법률, 긴급 금융지원, 주거지원 등 종합적인 맞춤형 상담을 제공한다. + - expected: `⠴⠠⠠⠛⠓⠲⠀⠨⠍⠈⠎⠀⠘⠛⠜⠀⠨⠾⠑⠛⠟⠐⠱⠁` + - actual: `⠴⠠⠠⠣⠲⠀⠨⠍⠈⠎⠀⠘⠛⠜⠀⠨⠾⠑⠛⠟⠐⠱⠁⠈` + - first differing cell (zero-based): 3 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #1853: 경기주택도시공사(GH)가 층간 소음 등 아파트 품질 사각지대를 일소하고, 건설산업 근로자의 적정 임금을 보장하는 등 경기도 대표 공공기관으로써 사회적 책임 실천에 나선다. + - expected: `⠈⠿⠇⠦⠄⠴⠠⠠⠛⠓⠠⠴⠫⠀⠰⠪⠶⠫⠒⠀⠠⠥⠪⠢` + - actual: `⠈⠿⠇⠦⠄⠴⠠⠠⠣⠠⠴⠫⠀⠰⠪⠶⠫⠒⠀⠠⠥⠪⠢⠀` + - first differing cell (zero-based): 21 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `U+2826 ⠦ -> U+2834 ⠴` + +- `sentence_01.json` #87: 4일 진원생명과학은 자체 개발 흡인작용 피내 접종기 진덤(GeneDerm)을 이용한 코로나19 DNA 백신(GLS-5310)의 안전성·면역원성에 대한 임상1상 결과를 국제감염병학회 (ISID) 학술지인 국제감염질환저널에 발표했다고 4일 밝혔다. + - expected: `⠢⠘⠻⠚⠁⠚⠽⠀⠦⠄⠴⠠⠠⠊⠎⠊⠙⠠⠴⠀⠚⠁⠠⠯` + - actual: `⠢⠘⠻⠚⠁⠚⠽⠀⠴⠐⠣⠠⠠⠊⠎⠊⠙⠐⠜⠲⠀⠚⠁⠠` + - first differing cell (zero-based): 172 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #38: 중국금형전시회(DMC) 및 독일금형전시회 (Moulding Expo)와 함께 세계 3대 규모의 금형산업 전문 전시회이며, 연관산업 전문 전시회를 KOPLAS와 동시 개최를 통한 시너지를 창출했다. + - expected: `⠻⠨⠾⠠⠕⠚⠽⠀⠦⠄⠴⠠⠍⠳⠇⠙⠬⠀⠠⠑⠭⠏⠕⠠` + - actual: `⠻⠨⠾⠠⠕⠚⠽⠀⠴⠐⠣⠠⠍⠳⠇⠙⠬⠀⠠⠑⠭⠏⠕⠐` + - first differing cell (zero-based): 48 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #862: 평소 사진찍기를 꺼리던 작가는 지난 3일 기자간담회에서 직접 오큘러스 메타퀘스트 고글을 머리에 쓰고 걸으면서 VR(가상현실)작품 ‘OP.VR/01’(2022)을 시연하는 모습을 선보였다. + - expected: `⠗⠸⠌⠼⠚⠁⠴⠄⠦⠄⠼⠃⠚⠃⠃⠠⠴⠮⠀⠠⠕⠡⠚⠉` + - actual: `⠗⠸⠌⠼⠚⠁⠴⠄⠴⠐⠣⠼⠃⠚⠃⠃⠴⠐⠜⠲⠮⠀⠠⠕` + - first differing cell (zero-based): 149 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #2544: 관상동맥 내 딱딱하게 쌓인 죽종을 깎아내는 회전죽종절제술(ROTA)을 이용한 관상동맥 중재술 (PCI)을 받은 환자가 시술 후 심근경색을 경험하더라도 예후에 영향이 없는 것으로 확인됐다. + - expected: `⠨⠍⠶⠨⠗⠠⠯⠀⠦⠄⠴⠠⠠⠏⠉⠊⠠⠴⠮⠀⠘⠔⠵⠀` + - actual: `⠨⠍⠶⠨⠗⠠⠯⠀⠴⠐⠣⠠⠠⠏⠉⠊⠐⠜⠲⠮⠀⠘⠔⠵` + - first differing cell (zero-based): 99 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `U+2811 ⠑ -> U+283B ⠻` + +- `sentence_01.json` #1903: 김현용 현대차증권 연구원은 “전사적자원관리(ERP), 생산관리프로그램(MES) 등 기존의 기업향 솔루션에 이번 인수로 구매공급망관리(SRM)이 더해지며 사업 포트폴리오가 한층 단단해진 것으로 판단된다”고 설명했다. + - expected: `⠒⠐⠕⠦⠄⠴⠠⠠⠑⠗⠏⠠⠴⠐⠀⠠⠗⠶⠇⠒⠈⠧⠒⠐` + - actual: `⠒⠐⠕⠦⠄⠴⠠⠠⠻⠏⠠⠴⠐⠀⠠⠗⠶⠇⠒⠈⠧⠒⠐⠕` + - first differing cell (zero-based): 48 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #14: 한국타이어앤테크놀로지㈜(대표이사 이수일, 이하 한국타이어)가 국제자동차연맹(FIA)이 주관하는 ‘FIA 주니어 ERC(FIA Junior ERC, 이하 주니어 ERC)’ 대회에 레이싱 타이어를 독점 공급한다. + - expected: `⠍⠉⠕⠎⠀⠴⠠⠠⠑⠗⠉⠦⠄⠴⠠⠠⠋⠊⠁⠀⠠⠚⠥⠝` + - actual: `⠍⠉⠕⠎⠀⠴⠠⠠⠻⠉⠦⠄⠴⠠⠠⠋⠊⠁⠀⠠⠚⠥⠝⠊` + - first differing cell (zero-based): 114 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #751: 올 하반기 위하고 기반 차세대 ‘스마트A 10’으로 본격적인 고객 전환이 예정된 가운데 세무회계사무소 전용 전사적자원관리 시스템(ERP)인 ‘위하고T’와 개인용 앱 ‘나하고(NAHAGO)’도 업계에서 주목받고 있다. + - expected: `⠓⠝⠢⠦⠄⠴⠠⠠⠑⠗⠏⠠⠴⠟⠀⠠⠦⠍⠗⠚⠈⠥⠴⠠` + - actual: `⠓⠝⠢⠦⠄⠴⠠⠠⠻⠏⠠⠴⠟⠀⠠⠦⠍⠗⠚⠈⠥⠴⠠⠞` + - first differing cell (zero-based): 127 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #4038: 현대건설은 현지 건설사 이알버드(ERBUD), 유니베프(UNIBEP)와 신재생에너지 및 인프라 분야 협력에 관한 업무협약도 체결했다. 이를 통해 신공항, 도심 인프라, 스마트시티 분야 관련 사업을 함께 추진한다. + - expected: `⠎⠊⠪⠦⠄⠴⠠⠠⠑⠗⠃⠥⠙⠠⠴⠐⠀⠩⠉⠕⠘⠝⠙⠪` + - actual: `⠎⠊⠪⠦⠄⠴⠠⠠⠻⠃⠥⠙⠠⠴⠐⠀⠩⠉⠕⠘⠝⠙⠪⠦` + - first differing cell (zero-based): 33 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `U+2800 ⠀ -> U+2832 ⠲` + +- `sentence_01.json` #4520: 방탄소년단(BTS) 10주년 기념 불꽃쇼가 지난 17일 서울 영등포구 여의도 한강공원에서 열린 방탄소년단 데뷔 10주년 FESTA @여의도(BTS 10th Anniversary FESTA @Yeouido)에서 펼쳐지고 있다. + - expected: `⠽⠀⠠⠠⠋⠑⠌⠁⠀⠈⠁⠠⠽⠑⠳⠊⠙⠕⠠⠴⠝⠠⠎⠀` + - actual: `⠽⠀⠠⠠⠋⠑⠌⠁⠲⠀⠈⠁⠴⠠⠽⠑⠳⠊⠙⠕⠠⠴⠝⠠` + - first differing cell (zero-based): 162 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #1841: 한편 현대차그룹은 현대차 아이오닉6와 제네시스 GV70 전동화 모델이 미국 고속도로 안전보험협회(IIHS)가 발표한 충돌평가에서 최고 등급인 ‘톱 세이프티 픽 플러스(TSP +)’를 받았다고 전했다. + - expected: `⠦⠄⠴⠠⠠⠞⠎⠏⠀⠐⠖⠠⠴⠴⠄⠐⠮⠀⠘⠔⠣⠌⠊⠈` + - actual: `⠦⠄⠴⠠⠠⠞⠎⠏⠲⠀⠢⠠⠴⠴⠄⠐⠮⠀⠘⠔⠣⠌⠊⠈` + - first differing cell (zero-based): 174 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #19: 해당 펀드는 전기차와 2차전지 및 2차전지 연관산업인 ESS /VPP(가상발전소)(14%) 등 전세계 친환경 기술 기업에 투자하는 상품이다. + - expected: `⠟⠀⠴⠠⠠⠑⠎⠎⠀⠸⠌⠠⠠⠧⠏⠏⠦⠄⠫⠇⠶⠘⠂⠨` + - actual: `⠟⠀⠴⠠⠠⠑⠎⠎⠲⠀⠸⠌⠴⠠⠠⠧⠏⠏⠦⠄⠫⠇⠶⠘` + - first differing cell (zero-based): 58 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #3324: 삼성전자는 이날 생성형 인공지능(AI) 서버에 적용되는 서버용 SSD ‘PM1743’과 쿼드러플 레벨 셀(QLC) 낸드 기반 256TB SSD도 선보였다. + - expected: `⠶⠀⠴⠠⠠⠎⠎⠙⠀⠠⠦⠠⠠⠏⠍⠼⠁⠛⠙⠉⠠⠴⠲⠈` + - actual: `⠶⠀⠴⠠⠠⠎⠎⠙⠲⠀⠠⠦⠴⠠⠠⠏⠍⠼⠁⠛⠙⠉⠴⠄` + - first differing cell (zero-based): 68 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `U+2826 ⠦ -> U+2810 ⠐` + +- `sentence_01.json` #1318: 비제조업 업황 BSI(73)는 2p 상승했다. 비제조업 업황 BSI가 전월 대비 상승한 것은 지난 2022년 8월 이후 6개월 만에 처음이다. + - expected: `⠶⠀⠴⠠⠠⠃⠎⠊⠦⠄⠼⠛⠉⠠⠴⠉⠵⠀⠼⠃⠴⠏⠲⠀` + - actual: `⠶⠀⠴⠠⠠⠃⠎⠊⠐⠣⠼⠛⠉⠴⠐⠜⠲⠉⠵⠀⠼⠃⠴⠏` + - first differing cell (zero-based): 21 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #1010: 첫 외국인 방문객으로는 말레이시아에서 온 Tan Chen Loon(47)씨 가족이다. 부인과 아들과 함께 공원을 방문했다가 행운의 방문객이 됐다. + - expected: `⠡⠢⠀⠠⠇⠕⠕⠝⠦⠄⠼⠙⠛⠠⠴⠠⠠⠕⠀⠫⠨⠭⠕⠊` + - actual: `⠡⠢⠀⠠⠇⠕⠕⠝⠐⠣⠼⠙⠛⠴⠐⠜⠲⠠⠠⠕⠀⠫⠨⠭` + - first differing cell (zero-based): 52 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #869: 카메라를 총처럼 든 남자가 있다. 피사체는 2019년 칸영화제 심사위원상을 수상한 말리 출신 이민자 영화 감독 래드 리다. 사진작가 JR(40)의 예술 인생은 이 사진으로 바뀌었다. + - expected: `⠁⠫⠀⠴⠠⠠⠚⠗⠦⠄⠼⠙⠚⠠⠴⠺⠀⠌⠠⠯⠀⠟⠠⠗` + - actual: `⠁⠫⠀⠴⠠⠠⠚⠗⠐⠣⠼⠙⠚⠴⠐⠜⠲⠺⠀⠌⠠⠯⠀⠟` + - first differing cell (zero-based): 119 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #1027: 5월 업황에 대한 전망 BSI(74)는 한 달 새 1포인트 올랐다. 제조업(72)에서 3포인트, 비제조업(76)에서 1포인트 상승했다. BSI에 소비자동향지수(CSI)를 반영한 4월 경제심리지수(ESI)는 전월보다 2.3포인트 상승한 93.8을 기록했다. + - expected: `⠶⠀⠴⠠⠠⠃⠎⠊⠦⠄⠼⠛⠙⠠⠴⠉⠵⠀⠚⠒⠀⠊⠂⠀` + - actual: `⠶⠀⠴⠠⠠⠃⠎⠊⠐⠣⠼⠛⠙⠴⠐⠜⠲⠉⠵⠀⠚⠒⠀⠊` + - first differing cell (zero-based): 28 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +## Residual first-difference transitions after localized cohorts + +This ranking removes only cases whose first difference is inside an existing output-localized cohort. Broad input-only traits are not exclusion masks. The residual table therefore prioritizes new causes without hiding a mismatch merely because an unrelated structure coexists elsewhere in its sentence. + +| Rank | Expected → actual first cell | Residual cases | +|---:|---|---:| +| 1 | `U+281B ⠛ -> U+2823 ⠣` | 147 | +| 2 | `U+2810 ⠐ -> U+2832 ⠲` | 98 | +| 3 | `U+2811 ⠑ -> U+283B ⠻` | 97 | +| 4 | `U+2800 ⠀ -> U+2832 ⠲` | 79 | +| 5 | `U+280E ⠎ -> U+2829 ⠩` | 77 | +| 6 | `U+2811 ⠑ -> U+2822 ⠢` | 68 | +| 7 | `U+2826 ⠦ -> U+2810 ⠐` | 67 | +| 8 | `U+280A ⠊ -> U+2814 ⠔` | 60 | +| 9 | `U+2810 ⠐ -> U+2815 ⠕` | 59 | +| 10 | `U+283C ⠼ -> U+2800 ⠀` | 55 | +| 11 | `U+2824 ⠤ -> U+2800 ⠀` | 54 | +| 12 | `U+2809 ⠉ -> U+2812 ⠒` | 47 | +| 13 | `U+2820 ⠠ -> U+2834 ⠴` | 47 | +| 14 | `U+2810 ⠐ -> U+2811 ⠑` | 46 | +| 15 | `U+2815 ⠕ -> U+2837 ⠷` | 41 | +| 16 | `U+2820 ⠠ -> U+281E ⠞` | 39 | +| 17 | `U+2820 ⠠ -> U+2832 ⠲` | 39 | +| 18 | `U+2802 ⠂ -> U+2810 ⠐` | 38 | +| 19 | `U+2809 ⠉ -> U+2821 ⠡` | 35 | +| 20 | `U+2834 ⠴ -> U+2800 ⠀` | 32 | + +### Residual `U+281B ⠛ -> U+2823 ⠣` + +- `sentence_01.json` #1753: 지난해 말 경기주택도시공사(GH)에서 퇴직한 전씨는 ‘성남FC 불법 후원금 의혹’과 관련해 검찰 조사를 받은 바 있으며 ‘GH 합숙소 의혹’에도 연루된 것으로 알려졌다. + - expected: `⠈⠿⠇⠦⠄⠴⠠⠠⠛⠓⠠⠴⠝⠠⠎⠀⠓⠽⠨⠕⠁⠚⠒⠀` + - actual: `⠈⠿⠇⠦⠄⠴⠠⠠⠣⠠⠴⠝⠠⠎⠀⠓⠽⠨⠕⠁⠚⠒⠀⠨` + - first differing cell (zero-based): 31 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #24: 대전대학교(총장 남상호)는 창업보육센터 입주기업인 ㈜티알(대표 김병수)이 최근 트랜스글로벌헬스케어(TGH)와 세계최초 AI 기반 만성폐쇄성폐질환(COPD) 진단기인 ‘The Spirokit’(더스피로킷)에 대한 물품공급 계약을 체결했다고 지난 14일 밝혔다. + - expected: `⠝⠎⠦⠄⠴⠠⠠⠞⠛⠓⠠⠴⠧⠀⠠⠝⠈⠌⠰⠽⠰⠥⠀⠴` + - actual: `⠝⠎⠦⠄⠴⠠⠠⠞⠣⠠⠴⠧⠀⠠⠝⠈⠌⠰⠽⠰⠥⠀⠴⠠` + - first differing cell (zero-based): 113 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #292: GH 주거 분야 전문인력과 HUG(주택도시보증공사), 변호사, 법무사 등 부동산·금융 전문인력이 상주하며 전세사기 피해자에게 부동산 법률, 긴급 금융지원, 주거지원 등 종합적인 맞춤형 상담을 제공한다. + - expected: `⠴⠠⠠⠛⠓⠲⠀⠨⠍⠈⠎⠀⠘⠛⠜⠀⠨⠾⠑⠛⠟⠐⠱⠁` + - actual: `⠴⠠⠠⠣⠲⠀⠨⠍⠈⠎⠀⠘⠛⠜⠀⠨⠾⠑⠛⠟⠐⠱⠁⠈` + - first differing cell (zero-based): 3 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #1853: 경기주택도시공사(GH)가 층간 소음 등 아파트 품질 사각지대를 일소하고, 건설산업 근로자의 적정 임금을 보장하는 등 경기도 대표 공공기관으로써 사회적 책임 실천에 나선다. + - expected: `⠈⠿⠇⠦⠄⠴⠠⠠⠛⠓⠠⠴⠫⠀⠰⠪⠶⠫⠒⠀⠠⠥⠪⠢` + - actual: `⠈⠿⠇⠦⠄⠴⠠⠠⠣⠠⠴⠫⠀⠰⠪⠶⠫⠒⠀⠠⠥⠪⠢⠀` + - first differing cell (zero-based): 21 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### Residual `U+2810 ⠐ -> U+2832 ⠲` + +- `sentence_01.json` #1675: LS일렉트릭은 8일부터 오는 10일까지 서울 삼성동 코엑스에서 열리는 ‘스마트공장·자동화산업전(SF+AW) 2023’ 전시회에 국내 기업 중 최대 규모로 참가한다. + - expected: `⠾⠦⠄⠴⠠⠠⠎⠋⠐⠖⠠⠠⠁⠺⠠⠴⠀⠼⠃⠚⠃⠉⠴⠄` + - actual: `⠾⠦⠄⠴⠠⠠⠎⠋⠲⠢⠴⠠⠠⠁⠺⠠⠴⠀⠼⠃⠚⠃⠉⠴` + - first differing cell (zero-based): 99 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #885: 지난 13일 온라인 컨퍼런스로 진행된 밸류데이에서 KT&G는 중장기(2024년~2026년) 주주환원 계획을 공개하고, 3대 핵심사업인 NGP(Next Generation Products)·글로벌CC(궐련담배)·건기식(건강기능식품) 중심의 미래비전 이행 경과를 공유했다. + - expected: `⠕⠙⠥⠉⠞⠎⠐⠜⠐⠆⠈⠮⠐⠥⠘⠞⠴⠠⠠⠉⠉⠦⠄⠈` + - actual: `⠕⠙⠥⠉⠞⠎⠐⠜⠲⠐⠆⠈⠮⠐⠥⠘⠞⠴⠠⠠⠉⠉⠦⠄` + - first differing cell (zero-based): 167 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #2536: 13일 금융투자업계에 따르면 오는 16일과 20일 KB증권(AA+)과 한국금융지주(AA-)가 각 4600억원, 1300억원의 회사채 발행을 위한 수요예측을 진행할 예정인 것으로 전해졌다. 한국금융지주는 한국투자증권을 주력 자회사로 둔 금융지주사다. + - expected: `⠍⠦⠄⠴⠠⠠⠁⠁⠐⠤⠠⠴⠫⠀⠫⠁⠀⠼⠙⠋⠚⠚⠹⠏` + - actual: `⠍⠦⠄⠴⠠⠠⠁⠁⠲⠤⠠⠴⠫⠀⠫⠁⠀⠼⠙⠋⠚⠚⠹⠏` + - first differing cell (zero-based): 93 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #1069: HMM·삼성중공업·파나시아·한국선급 등 4개사는 지난 27일 선박 이산화탄소 포집·액화 저장 기술(OCCS) 통합 실증 연구를 위한 업무협약(MOU)을 체결했다고 28일 밝혔다. + - expected: `⠴⠠⠠⠓⠍⠍⠐⠆⠇⠢⠠⠻⠨⠍⠶⠈⠿⠎⠃⠐⠆⠙⠉⠠` + - actual: `⠴⠠⠠⠓⠍⠍⠲⠐⠆⠇⠢⠠⠻⠨⠍⠶⠈⠿⠎⠃⠐⠆⠙⠉` + - first differing cell (zero-based): 6 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### Residual `U+2811 ⠑ -> U+283B ⠻` + +- `sentence_01.json` #1903: 김현용 현대차증권 연구원은 “전사적자원관리(ERP), 생산관리프로그램(MES) 등 기존의 기업향 솔루션에 이번 인수로 구매공급망관리(SRM)이 더해지며 사업 포트폴리오가 한층 단단해진 것으로 판단된다”고 설명했다. + - expected: `⠒⠐⠕⠦⠄⠴⠠⠠⠑⠗⠏⠠⠴⠐⠀⠠⠗⠶⠇⠒⠈⠧⠒⠐` + - actual: `⠒⠐⠕⠦⠄⠴⠠⠠⠻⠏⠠⠴⠐⠀⠠⠗⠶⠇⠒⠈⠧⠒⠐⠕` + - first differing cell (zero-based): 48 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #672: 충남교육청은 사무행정(태안여고), ERP(천안여자상업고) 및 대회홍보크리에이터(논산여자상업고) 3개 종목에서 대상인 교육부장관상을 받았고 △금상 6개 △은상 16개 △동상 23개 등 총 48개 수상의 영예를 누렸다. + - expected: `⠥⠠⠴⠐⠀⠴⠠⠠⠑⠗⠏⠦⠄⠰⠾⠣⠒⠱⠨⠇⠶⠎⠃⠈` + - actual: `⠥⠠⠴⠐⠀⠴⠠⠠⠻⠏⠦⠄⠰⠾⠣⠒⠱⠨⠇⠶⠎⠃⠈⠥` + - first differing cell (zero-based): 37 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #751: 올 하반기 위하고 기반 차세대 ‘스마트A 10’으로 본격적인 고객 전환이 예정된 가운데 세무회계사무소 전용 전사적자원관리 시스템(ERP)인 ‘위하고T’와 개인용 앱 ‘나하고(NAHAGO)’도 업계에서 주목받고 있다. + - expected: `⠓⠝⠢⠦⠄⠴⠠⠠⠑⠗⠏⠠⠴⠟⠀⠠⠦⠍⠗⠚⠈⠥⠴⠠` + - actual: `⠓⠝⠢⠦⠄⠴⠠⠠⠻⠏⠠⠴⠟⠀⠠⠦⠍⠗⠚⠈⠥⠴⠠⠞` + - first differing cell (zero-based): 127 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #4038: 현대건설은 현지 건설사 이알버드(ERBUD), 유니베프(UNIBEP)와 신재생에너지 및 인프라 분야 협력에 관한 업무협약도 체결했다. 이를 통해 신공항, 도심 인프라, 스마트시티 분야 관련 사업을 함께 추진한다. + - expected: `⠎⠊⠪⠦⠄⠴⠠⠠⠑⠗⠃⠥⠙⠠⠴⠐⠀⠩⠉⠕⠘⠝⠙⠪` + - actual: `⠎⠊⠪⠦⠄⠴⠠⠠⠻⠃⠥⠙⠠⠴⠐⠀⠩⠉⠕⠘⠝⠙⠪⠦` + - first differing cell (zero-based): 33 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### Residual `U+2800 ⠀ -> U+2832 ⠲` + +- `sentence_01.json` #4520: 방탄소년단(BTS) 10주년 기념 불꽃쇼가 지난 17일 서울 영등포구 여의도 한강공원에서 열린 방탄소년단 데뷔 10주년 FESTA @여의도(BTS 10th Anniversary FESTA @Yeouido)에서 펼쳐지고 있다. + - expected: `⠽⠀⠠⠠⠋⠑⠌⠁⠀⠈⠁⠠⠽⠑⠳⠊⠙⠕⠠⠴⠝⠠⠎⠀` + - actual: `⠽⠀⠠⠠⠋⠑⠌⠁⠲⠀⠈⠁⠴⠠⠽⠑⠳⠊⠙⠕⠠⠴⠝⠠` + - first differing cell (zero-based): 162 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #1841: 한편 현대차그룹은 현대차 아이오닉6와 제네시스 GV70 전동화 모델이 미국 고속도로 안전보험협회(IIHS)가 발표한 충돌평가에서 최고 등급인 ‘톱 세이프티 픽 플러스(TSP +)’를 받았다고 전했다. + - expected: `⠦⠄⠴⠠⠠⠞⠎⠏⠀⠐⠖⠠⠴⠴⠄⠐⠮⠀⠘⠔⠣⠌⠊⠈` + - actual: `⠦⠄⠴⠠⠠⠞⠎⠏⠲⠀⠢⠠⠴⠴⠄⠐⠮⠀⠘⠔⠣⠌⠊⠈` + - first differing cell (zero-based): 174 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #19: 해당 펀드는 전기차와 2차전지 및 2차전지 연관산업인 ESS /VPP(가상발전소)(14%) 등 전세계 친환경 기술 기업에 투자하는 상품이다. + - expected: `⠟⠀⠴⠠⠠⠑⠎⠎⠀⠸⠌⠠⠠⠧⠏⠏⠦⠄⠫⠇⠶⠘⠂⠨` + - actual: `⠟⠀⠴⠠⠠⠑⠎⠎⠲⠀⠸⠌⠴⠠⠠⠧⠏⠏⠦⠄⠫⠇⠶⠘` + - first differing cell (zero-based): 58 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### Residual `U+280E ⠎ -> U+2829 ⠩` + +- `sentence_01.json` #556: 현재 구룡마을은 서울주택도시공사(SH) 주도로 재개발 사업이 추진되고 있다. 지난 2020년 6월 실시계획 인가 내용에 따르면 임대주택 1천107가구와 공공분양 991가구, 민간분양 740가구가 들어설 계획이다. + - expected: `⠈⠿⠇⠦⠄⠴⠠⠠⠎⠓⠠⠴⠀⠨⠍⠊⠥⠐⠥⠀⠨⠗⠈⠗` + - actual: `⠈⠿⠇⠦⠄⠴⠠⠠⠩⠠⠴⠀⠨⠍⠊⠥⠐⠥⠀⠨⠗⠈⠗⠘` + - first differing cell (zero-based): 35 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #431: 173cm, 68kg의 다소 왜소한 체격이지만 스크럼 하프(SH) 포지션을 맡아 포워드와 백스 사이에서 날카로운 볼 연결과 공격방향을 결정하는 타고난 판단력이 강점이다. + - expected: `⠚⠙⠪⠦⠄⠴⠠⠠⠎⠓⠠⠴⠀⠙⠥⠨⠕⠠⠡⠮⠀⠑⠦⠣` + - actual: `⠚⠙⠪⠦⠄⠴⠠⠠⠩⠠⠴⠀⠙⠥⠨⠕⠠⠡⠮⠀⠑⠦⠣⠀` + - first differing cell (zero-based): 55 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #1606: 행복주택 당첨자들의 개인정보가 유출됐다. 서울주택도시공사(SH)는 직원의 실수로 발생한 사고라며 사과했지만, 공공기관으로서 수많은 서울시민의 신상 명세를 보관하고 있는 만큼 주의가 필요하다는 지적이 나온다. + - expected: `⠈⠿⠇⠦⠄⠴⠠⠠⠎⠓⠠⠴⠉⠵⠀⠨⠕⠁⠏⠒⠺⠀⠠⠕` + - actual: `⠈⠿⠇⠦⠄⠴⠠⠠⠩⠠⠴⠉⠵⠀⠨⠕⠁⠏⠒⠺⠀⠠⠕⠂` + - first differing cell (zero-based): 60 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #141: 서울 강동구 고덕강일지구에 이어 강서구 마곡지구에도 ‘반값 아파트’가 공급된다. 서울주택도시공사(SH)는 내년까지 8000가구 이상의 토지임대부 분양주택을 공급할 방침이다. + - expected: `⠈⠿⠇⠦⠄⠴⠠⠠⠎⠓⠠⠴⠉⠵⠀⠉⠗⠉⠡⠠⠫⠨⠕⠀` + - actual: `⠈⠿⠇⠦⠄⠴⠠⠠⠩⠠⠴⠉⠵⠀⠉⠗⠉⠡⠠⠫⠨⠕⠀⠼` + - first differing cell (zero-based): 93 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### Residual `U+2811 ⠑ -> U+2822 ⠢` + +- `sentence_01.json` #1627: 또 “한국무역협회와 공동으로 기업 차원의 규제 대응 및 유럽경제협력네트워크(EEN) 프로그램 등을 활용한 산업 협력 방안을 논의하는 자리를 늘려가길 희망한다”며 양 기관 간 협력을 주문했다. + - expected: `⠋⠪⠦⠄⠴⠠⠠⠑⠑⠝⠠⠴⠀⠙⠪⠐⠥⠈⠪⠐⠗⠢⠀⠊` + - actual: `⠋⠪⠦⠄⠴⠠⠠⠑⠢⠠⠴⠀⠙⠪⠐⠥⠈⠪⠐⠗⠢⠀⠊⠪` + - first differing cell (zero-based): 81 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #2143: 황새는 세계자연보전연맹 적색자료 목록에 ‘위기(EN)종’으로 분류된 국제보호종으로 세계에 약 2천500개체가 살아있는 것으로 전해진다. 우리나라 황새는 1971년 충북 음성군에서 밀렵꾼에게 잡히면서 자취를 감췄다. + - expected: `⠗⠈⠕⠦⠄⠴⠠⠠⠑⠝⠠⠴⠨⠿⠴⠄⠪⠐⠥⠀⠘⠛⠐⠩` + - actual: `⠗⠈⠕⠦⠄⠴⠠⠠⠢⠠⠴⠨⠿⠴⠄⠪⠐⠥⠀⠘⠛⠐⠩⠊` + - first differing cell (zero-based): 50 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #1156: 콘텐츠 플랫폼 기업 리디는 CJ ENM과 지적재산권(IP) 사업 확장을 위한 전략적 제휴(MOU)를 맺었다고 15일 밝혔다. + - expected: `⠴⠠⠠⠉⠚⠀⠠⠠⠑⠝⠍⠲⠈⠧⠀⠨⠕⠨⠹⠨⠗⠇⠒⠈` + - actual: `⠴⠠⠠⠉⠚⠀⠠⠠⠢⠍⠲⠈⠧⠀⠨⠕⠨⠹⠨⠗⠇⠒⠈⠏` + - first differing cell (zero-based): 37 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #2282: 2019년 11월 서울대공원은 캐니언파크로 알락꼬리여우원숭이 7마리를 양도했고, 12월에는 대구의 한 체험동물원에 14마리를 넘겼다. 알락꼬리여우원숭이는 세계자연보전연맹(IUCN)의 멸종위기종 목록인 적색 목록상 위기를 의미하는 EN(Endangered) 범주에 포함돼 있다. + - expected: `⠕⠚⠉⠵⠀⠴⠠⠠⠑⠝⠐⠣⠠⠢⠙⠁⠝⠛⠻⠫⠐⠜⠲⠀` + - actual: `⠕⠚⠉⠵⠀⠴⠠⠠⠢⠐⠣⠠⠢⠙⠁⠝⠛⠻⠫⠐⠜⠲⠀⠘` + - first differing cell (zero-based): 222 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### Residual `U+2826 ⠦ -> U+2810 ⠐` + +- `sentence_01.json` #1318: 비제조업 업황 BSI(73)는 2p 상승했다. 비제조업 업황 BSI가 전월 대비 상승한 것은 지난 2022년 8월 이후 6개월 만에 처음이다. + - expected: `⠶⠀⠴⠠⠠⠃⠎⠊⠦⠄⠼⠛⠉⠠⠴⠉⠵⠀⠼⠃⠴⠏⠲⠀` + - actual: `⠶⠀⠴⠠⠠⠃⠎⠊⠐⠣⠼⠛⠉⠴⠐⠜⠲⠉⠵⠀⠼⠃⠴⠏` + - first differing cell (zero-based): 21 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #1010: 첫 외국인 방문객으로는 말레이시아에서 온 Tan Chen Loon(47)씨 가족이다. 부인과 아들과 함께 공원을 방문했다가 행운의 방문객이 됐다. + - expected: `⠡⠢⠀⠠⠇⠕⠕⠝⠦⠄⠼⠙⠛⠠⠴⠠⠠⠕⠀⠫⠨⠭⠕⠊` + - actual: `⠡⠢⠀⠠⠇⠕⠕⠝⠐⠣⠼⠙⠛⠴⠐⠜⠲⠠⠠⠕⠀⠫⠨⠭` + - first differing cell (zero-based): 52 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #869: 카메라를 총처럼 든 남자가 있다. 피사체는 2019년 칸영화제 심사위원상을 수상한 말리 출신 이민자 영화 감독 래드 리다. 사진작가 JR(40)의 예술 인생은 이 사진으로 바뀌었다. + - expected: `⠁⠫⠀⠴⠠⠠⠚⠗⠦⠄⠼⠙⠚⠠⠴⠺⠀⠌⠠⠯⠀⠟⠠⠗` + - actual: `⠁⠫⠀⠴⠠⠠⠚⠗⠐⠣⠼⠙⠚⠴⠐⠜⠲⠺⠀⠌⠠⠯⠀⠟` + - first differing cell (zero-based): 119 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #1027: 5월 업황에 대한 전망 BSI(74)는 한 달 새 1포인트 올랐다. 제조업(72)에서 3포인트, 비제조업(76)에서 1포인트 상승했다. BSI에 소비자동향지수(CSI)를 반영한 4월 경제심리지수(ESI)는 전월보다 2.3포인트 상승한 93.8을 기록했다. + - expected: `⠶⠀⠴⠠⠠⠃⠎⠊⠦⠄⠼⠛⠙⠠⠴⠉⠵⠀⠚⠒⠀⠊⠂⠀` + - actual: `⠶⠀⠴⠠⠠⠃⠎⠊⠐⠣⠼⠛⠙⠴⠐⠜⠲⠉⠵⠀⠚⠒⠀⠊` + - first differing cell (zero-based): 28 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### Residual `U+280A ⠊ -> U+2814 ⠔` + +- `sentence_01.json` #362: 현재는 임상시험승인계획(IND) 준비 단계다. 임상 1상 신청은 작년 말을 목표로 했으나 중국 파트너사인 통화동보제약의 임상용 인슐린 원료 공급 일정 지연 등을 이유로 올해 2월로 미뤄졌다. + - expected: `⠚⠽⠁⠦⠄⠴⠠⠠⠊⠝⠙⠠⠴⠀⠨⠛⠘⠕⠀⠊⠒⠈⠌⠊` + - actual: `⠚⠽⠁⠦⠄⠴⠠⠠⠔⠙⠠⠴⠀⠨⠛⠘⠕⠀⠊⠒⠈⠌⠊⠲` + - first differing cell (zero-based): 30 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #230: INC는 아이디어(I)-니즈(N)-역량(C)의 융합을 뜻하며, 파괴적 혁신과 기업가적 대학으로서 산학협력을 활성화하기 위한 방법론으로, 지속가능한 가치창출형 산학협력을 위한 브랜드이다. + - expected: `⠴⠠⠠⠊⠝⠉⠲⠉⠵⠀⠣⠕⠊⠕⠎⠦⠄⠴⠠⠊⠠⠴⠤⠉` + - actual: `⠴⠠⠠⠔⠉⠲⠉⠵⠀⠣⠕⠊⠕⠎⠦⠄⠴⠠⠊⠠⠴⠤⠉⠕` + - first differing cell (zero-based): 3 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #371: 인터넷 인프라 전문기업 케이아이엔엑스(KINX)는 ‘2023년 중소기업 클라우드 서비스 보급·확산 사업’의 공급기업으로 4년 연속 선정됐다고 26일 밝혔다. + - expected: `⠠⠪⠦⠄⠴⠠⠠⠅⠊⠝⠭⠠⠴⠉⠵⠀⠠⠦⠼⠃⠚⠃⠉⠀` + - actual: `⠠⠪⠦⠄⠴⠠⠠⠅⠔⠭⠠⠴⠉⠵⠀⠠⠦⠼⠃⠚⠃⠉⠀⠉` + - first differing cell (zero-based): 39 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #311: 멕시코 이민국(INM)이 운영하는 이 수용시설에는 화재 당시 중남미 출신 이민자 68명이 수용돼 있었던 것으로 추정된다. 시설에 있던 사람들 대부분은 미국으로 향하던 베네수엘라인들이었던 것으로 알려졌다. + - expected: `⠈⠍⠁⠦⠄⠴⠠⠠⠊⠝⠍⠠⠴⠕⠀⠛⠻⠚⠉⠵⠀⠕⠀⠠` + - actual: `⠈⠍⠁⠦⠄⠴⠠⠠⠔⠍⠠⠴⠕⠀⠛⠻⠚⠉⠵⠀⠕⠀⠠⠍` + - first differing cell (zero-based): 19 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### Residual `U+2810 ⠐ -> U+2815 ⠕` + +- `sentence_01.json` #3848: ‘천안시 승격 60주년 KBS 열린음악회’에는 가수 김연자, 소찬휘, 김범룡, 최성수, 우연이, 김영임, 고영열, 김필, 김기태, 원어스(ONEUS)가 출연해 전 세대가 즐길 수 있는 다양한 공연을 선사한다. + - expected: `⠎⠠⠪⠦⠄⠴⠠⠠⠐⠕⠥⠎⠠⠴⠫⠀⠰⠯⠡⠚⠗⠀⠨⠾` + - actual: `⠎⠠⠪⠦⠄⠴⠠⠠⠕⠝⠑⠥⠎⠠⠴⠫⠀⠰⠯⠡⠚⠗⠀⠨` + - first differing cell (zero-based): 133 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #771: 이번 연구 결과 기존 AhR 활성화합물인 rutaecarpine(한약재), hydrocortisone(항염제, 호르몬), alantolactone(항염제, 살선충제)과 이번에 새롭게 발견된 물질들은 전체 발현된 독성의 2.6%~49%를 설명하는 것으로 나타났다. + - expected: `⠁⠝⠞⠕⠇⠁⠉⠞⠐⠕⠦⠄⠚⠶⠱⠢⠨⠝⠐⠀⠇⠂⠠⠾` + - actual: `⠁⠝⠞⠕⠇⠁⠉⠞⠕⠝⠑⠦⠄⠚⠶⠱⠢⠨⠝⠐⠀⠇⠂⠠` + - first differing cell (zero-based): 107 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #2198: 이번 파트너십으로 LG AI연구원은 퓨리오사AI가 개발 중인 2세대 AI 반도체 레니게이드(Renegade)로 초거대 AI 엑사원(EXAONE) 기반의 ‘생성형 AI’ 상용 기술을 검증한다. + - expected: `⠦⠄⠴⠠⠠⠑⠭⠁⠐⠕⠠⠴⠀⠈⠕⠘⠒⠺⠀⠠⠦⠠⠗⠶` + - actual: `⠦⠄⠴⠠⠠⠑⠭⠁⠕⠝⠑⠠⠴⠀⠈⠕⠘⠒⠺⠀⠠⠦⠠⠗` + - first differing cell (zero-based): 131 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### Residual `U+283C ⠼ -> U+2800 ⠀` + +- `sentence_01.json` #1702: 조광페인트의 신사업은 방열소재(TIM) 분야다. 조광페인트는- 2차전지 소재인 CK이엠솔루션을 자회사로 두고 관련 사업에 전력을 다하고 있다. + - expected: `⠙⠝⠟⠓⠪⠉⠵⠤⠼⠃⠰⠣⠨⠾⠨⠕⠀⠠⠥⠨⠗⠟⠀⠴` + - actual: `⠙⠝⠟⠓⠪⠉⠵⠤⠀⠼⠃⠰⠣⠨⠾⠨⠕⠀⠠⠥⠨⠗⠟⠀` + - first differing cell (zero-based): 56 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #1610: 16일 한국석유공사 유가정보시스템 오피넷에 따르면 7월 둘째 주(9~ 12일) 울산 주유소 휘발유 평균 판매 가격은 전주(1천538.23원)보다 3.12원 상승한 L(리터)당 1천541.35원을 기록했다. + - expected: `⠨⠍⠦⠄⠼⠊⠈⠔⠼⠁⠃⠕⠂⠠⠴⠀⠯⠇⠒⠀⠨⠍⠩⠠` + - actual: `⠨⠍⠦⠄⠼⠊⠈⠔⠀⠼⠁⠃⠕⠂⠠⠴⠀⠯⠇⠒⠀⠨⠍⠩` + - first differing cell (zero-based): 66 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #341: 미래에셋자산운용은 미국 대표지수에 환헤지형으로 투자하는 ‘TIGER 미국S&P500TR(H) 상장지수펀드(ETF)’와 ‘TIGER 미국나스닥100TR(H) ETF’ 순자산 합계가 1000억원을 돌파했다고 26일 밝혔다. + - expected: `⠈⠍⠁⠉⠠⠪⠊⠁⠼⠁⠚⠚⠴⠠⠠⠞⠗⠐⠣⠠⠓⠐⠜⠀` + - actual: `⠈⠍⠁⠉⠠⠪⠊⠁⠀⠀⠼⠁⠚⠚⠠⠠⠞⠗⠦⠠⠓⠴⠀⠴` + - first differing cell (zero-based): 138 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #2974: 보고서는 팬데믹 이전의 추세를 웃도는 가계 저축액을 초과저축으로 정의했다. 팬데믹 기간 초과저축은 지난해 명목 국내총생산(GDP)의 4.7~ 6.0%, 명목 민간소비의 9.7~12.4% 수준이다. + - expected: `⠺⠀⠼⠙⠲⠛⠈⠔⠼⠋⠲⠚⠴⠏⠐⠀⠑⠻⠑⠭⠀⠑⠟⠫` + - actual: `⠺⠀⠼⠙⠲⠛⠈⠔⠀⠼⠋⠲⠚⠴⠏⠐⠀⠑⠻⠑⠭⠀⠑⠟` + - first differing cell (zero-based): 140 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +## Cross-cutting input-only structural cohorts + +These are cross-cutting input-only structural cohorts, not new primary classes and not engine routing rules. Candidate selection never changes a case's existing primary class by itself. Only cohort members already classified as `pending_rule_review` form a pending subcluster; exact and other-primary members are controls that retain their existing outcomes. A separate classifier may use independently justified, output-localized PDF evidence, as in the rule-34 three-cell contradiction below. The `uppercase_roman_headword_closed_multiword_parenthetical` gate requires a two-or-more character uppercase ASCII headword immediately followed by a closed parenthesis whose contents are two or more ASCII Roman words separated only by spaces. Because the contents admit only letters and spaces, visible operators, subscript/superscript notation, and nested parentheses are excluded deterministically. The `mixed_roman_korean_word_before_uppercase_headword_expansion` gate further requires a whitespace-delimited preceding word that contains both ASCII Roman and Korean and ends in Korean. It locates the following uppercase headword itself, including its immediately preceding emitted cell, so an earlier Roman entry in the same sentence cannot satisfy the output audit. The `single_capital_followed_by_parenthesized_digits` gate requires a standalone capital, a non-empty closed ASCII-digit parenthetical, and alphanumeric outer boundaries. It likewise includes the emitted entry-boundary cell in localization. The `uppercase_roman_run_followed_by_hyphen_digits` gate requires a maximal uppercase ASCII run, literal ASCII hyphen-minus, a non-empty digit run, and alphanumeric outer boundaries. Its localized range includes the current encoded run and its immediately preceding cell, keeping `F-35`-style routing separate from both parenthesized digits and headword expansions. The `uppercase_alphanumeric_roman_digit_sequence` gate covers the wider rule-35/math collision: a capital-led uppercase/digit sequence with optional internal ASCII hyphen/full stop and trailing prose comma/colon/semicolon. It localizes only the current entry cells and does not decide whether an `A-3`-shaped surface is an identifier or a mathematical expression. The `standalone_multi_character_uppercase_roman_word` gate finds maximal ASCII-letter runs of two or more capitals with non-alphanumeric boundaries; a run immediately followed by `(` is excluded so the HCA-style headword itself is not counted by both gates. The `korean_prefixed_closed_allcaps_parenthetical` gate requires an immediately preceding Korean character and a closed body of two or more uppercase ASCII letters. It intentionally contains both acronym annotations (`책임자(COO)`) and scientific formulae (`일산화탄소(CO)`) so their semantic collision remains measurable. The `korean_prefixed_closed_roman_annotation_rule_34_order` gate accepts the narrower rule-34 body grammar after an immediately preceding Korean character and localizes only the current engine's Korean opening-parenthesis cells. This distinguishes the PDF's parenthesis-before-Roman-indicator order from unrelated differences later in the same sentence. The `multi_character_allcaps_roman_runs_joined_by_middle_dot` gate requires two maximal ASCII-letter runs of at least two capitals joined directly by U+00B7, with non-alphanumeric outer boundaries. It records shapes such as `AI·SW` without assigning prose, mathematics, or science semantics. The `roman_run_immediately_before_attached_middle_dot_boundary` gate is output-localized: it requires a maximal ASCII-letter run immediately before U+00B7 and an attached Korean character or ASCII-letter run after it, then searches for that whole current-engine signature in the actual output. It therefore isolates the Roman terminator boundary without treating unrelated middle dots elsewhere in the sentence as causal. The `attached_ascii_roman_to_korean_script_boundary` gate requires an ASCII letter run immediately followed by Korean script and localizes only the current mode marker there: either rule 29's final `⠲` or rule 39's opening `⠸⠷`. It deliberately does not infer the dominant language of the sentence from that surface boundary. The `korean_majority_same_token_roman_sandwich_non_domain` scope audit mirrors the narrower rule-39 implementation gate: the nearest script characters on both sides of one Korean segment are ASCII Roman, the input is Korean-majority by first-script word counts, and the segment is not dot-delimited like the official domain example. It measures change scope but is not itself an output-localized causal classifier. The `korean_inline_parenthesized_single_arithmetic_operator` gate requires an immediate `Korean(` + one rule-45 arithmetic operator + `)Korean` span. Unlike broad coexistence traits, it also locates the current engine's emitted structure and counts a mismatch as signature-local only when the sentence's first differing cell falls inside that output range. The `attached_plus_followed_by_parenthesized_korean_gloss` gate requires literal `한글+(한글)` with a non-empty all-Korean gloss. It anchors the real prefix immediately before `+` and verifies the current neutral-Korean output signature, distinguishing rule-46 spacing at the sign from unrelated differences elsewhere without deciding whether a name is mathematical. The `allcaps_roman_run_containing_ou` gate finds maximal, alphanumeric-delimited uppercase ASCII runs containing adjacent `OU`. It locates the independently encoded run signature in the complete current output and counts only first differences inside that signature as localized. The `allcaps_roman_run_containing_st` gate applies the same output-position requirement to adjacent `ST`; it is kept separate because UEB 10.12 makes contraction use depend on how an abbreviation or acronym is pronounced. The `uppercase_ascii_segments_joined_by_ampersand_capitalization` gate narrows attached ampersand runs to uppercase-only segments and locates the complete current Korean-context output. The run must begin its whitespace-delimited token, matching the current token-level capitals-word pre-emission scope; Korean-attached and punctuation-prefixed occurrences remain controls. UEB 8.4.2 says a nonalphabetic symbol terminates capitals word mode, while the official `AT&T` and `B&B` examples restart capitalization after `&`. The broader `capitals_word_mode_previously_spanning_nonletter_scope` gate reproduces the former token predicate across ampersands, hyphens, digits, and other nonletters that separate uppercase runs. Trailing nonletters after the final run are excluded because their output is unchanged. It is a broad change-scope/regression audit and is deliberately not an output-cause localizer. The `attached_ascii_roman_segments_joined_by_ampersand` gate requires non-empty ASCII-letter segments joined directly by `&`, excludes spaced/Korean/alphanumeric continuations, and localizes only the current output cell immediately before the ampersand through an independently encoded real-input prefix. The `ampersand_before_attached_ascii_roman_segment` gate covers the distinct one-sided UEB `&c` boundary: `&` is followed by a complete ASCII-letter segment, while an ASCII alphanumeric or another ampersand immediately before it and a digit continuation after it are excluded. Its output range is anchored by independently encoding the real input prefix before each occurrence, then includes only the current Rule-71/29 entry boundary; a second pre-fix anchor through `&` retains the former exit location. The `ascii_apostrophe_between_ascii_letter_runs` gate requires a straight apostrophe with an ASCII letter immediately on both sides and expands only across those two letter runs. It excludes detached quotation marks and numeric measurement marks, then locates the complete current mixed-Korean output signature. UEB 8.4.2 directly supplies `O'Hara`, `DON'T`, and `THAT'S` as controls. The `consecutive_ascii_roman_words_whitespace_boundary` gate requires two adjacent ASCII-letter words separated only by whitespace. For each boundary it independently encodes the real input prefix ending after the first word, then localizes only the current rule-29 terminator at that position or the full output's replacing blank. This prevents an unrelated Roman occurrence elsewhere in the sentence from satisfying the audit. Slash and other punctuation-separated forms are deliberately excluded. The `decimal_point_between_ascii_digits` gate finds whitespace-delimited words containing `digit.digit` and reproduces each whole word in a neutral Korean context, so suffixes and punctuation remain part of the current-engine signature. The `compact_numeric_ascii_letter_suffix` gate finds a numeric prefix immediately followed by ASCII letters, includes the immediately preceding output cell as its entry boundary, and retains suffix-specific outcome counts. It intentionally includes both possible rule-69 units and ambiguous variable/identifier forms; membership alone does not assign unit semantics. The `rule69_ascii_unit_before_terminator_skipping_symbol` gate is narrower: it accepts only ASCII unit spellings already supported by rule 69, includes the immediately following rule-33/34 punctuation cell in the localized signature, and does not infer new units. The `tight_triangle_mark_immediately_before_korean` gate requires literal `△한글` with no input space and includes the first following Korean cell in its localized output range, so an observed missing-space difference is measured at the mark boundary. The `attached_korean_auxiliary_itda_spacing` gate is input-only after the rule-49 correction: it measures Korean tokens ending in attached `있다` without claiming a current output signature or deciding whether orthographic correction may override the printed input. + +| Cluster | Candidates | Exact | Mismatch | Conflicting-reference cases | +|---|---:|---:|---:|---:| +| `allcaps_roman_run_beginning_with_pure_letter_shortform` | 3242 | 2809 | 433 | 0 | +| `allcaps_roman_run_containing_ar` | 1022 | 480 | 542 | 0 | +| `allcaps_roman_run_containing_ed` | 816 | 362 | 454 | 0 | +| `allcaps_roman_run_containing_ou` | 1816 | 128 | 1688 | 0 | +| `allcaps_roman_run_containing_st` | 1479 | 815 | 664 | 0 | +| `ampersand_before_attached_ascii_roman_segment` | 30 | 14 | 16 | 0 | +| `ascii_apostrophe_between_ascii_letter_runs` | 147 | 106 | 41 | 0 | +| `ascii_roman_tail_comma_before_digit_korean_token` | 58 | 34 | 24 | 0 | +| `attached_ascii_roman_segments_joined_by_ampersand` | 802 | 687 | 115 | 0 | +| `attached_ascii_roman_to_korean_script_boundary` | 17693 | 15206 | 2487 | 0 | +| `attached_korean_auxiliary_itda_spacing` | 95 | 84 | 11 | 0 | +| `attached_korean_to_roman_hyphen_boundary` | 105 | 84 | 21 | 0 | +| `attached_plus_followed_by_parenthesized_korean_gloss` | 16 | 4 | 12 | 0 | +| `capitals_word_mode_previously_spanning_nonletter_scope` | 1733 | 1270 | 463 | 0 | +| `closed_roman_parenthetical_after_non_ascii_letter_boundary` | 63959 | 57175 | 6784 | 0 | +| `compact_numeric_ascii_letter_suffix` | 2975 | 2492 | 483 | 0 | +| `consecutive_ascii_roman_words_whitespace_boundary` | 4679 | 3314 | 1365 | 0 | +| `decimal_point_between_ascii_digits` | 4546 | 4106 | 440 | 0 | +| `korean_inline_parenthesized_single_arithmetic_operator` | 23 | 22 | 1 | 0 | +| `korean_majority_same_token_roman_sandwich_non_domain` | 947 | 736 | 211 | 0 | +| `korean_prefixed_closed_allcaps_parenthetical` | 54492 | 48892 | 5600 | 0 | +| `korean_prefixed_closed_roman_annotation_rule_34_order` | 64382 | 57748 | 6634 | 0 | +| `korean_prefixed_roman_parenthetical_followed_by_allcaps_hyphen_suffix` | 11 | 1 | 10 | 0 | +| `mixed_roman_korean_word_before_uppercase_headword_expansion` | 10 | 7 | 3 | 0 | +| `multi_character_allcaps_roman_runs_joined_by_middle_dot` | 97 | 0 | 97 | 0 | +| `percent_point_unit_list_comma` | 7 | 5 | 2 | 0 | +| `pure_allcaps_segment_before_hyphen_and_multi_allcaps_segment_after` | 448 | 359 | 89 | 0 | +| `roman_hyphenated_word_after_whitespace_following_korean_word` | 361 | 277 | 84 | 0 | +| `roman_parenthetical_headword_after_whitespace_following_korean_word` | 4695 | 3945 | 750 | 0 | +| `roman_run_after_whitespace_following_closed_roman_enclosure` | 1093 | 580 | 513 | 0 | +| `roman_run_immediately_before_attached_middle_dot_boundary` | 577 | 0 | 577 | 0 | +| `rule69_ascii_unit_before_terminator_skipping_symbol` | 440 | 385 | 55 | 0 | +| `single_capital_followed_by_parenthesized_digits` | 1361 | 1351 | 10 | 0 | +| `spaced_comma_between_ascii_digit_runs` | 217 | 195 | 22 | 0 | +| `standalone_multi_character_uppercase_roman_word` | 62411 | 55551 | 6860 | 0 | +| `tight_triangle_mark_immediately_before_korean` | 377 | 317 | 60 | 0 | +| `uppercase_alphanumeric_roman_digit_sequence` | 3429 | 2861 | 568 | 0 | +| `uppercase_ascii_run_immediately_after_digit_in_roman_sequence` | 1896 | 1569 | 327 | 0 | +| `uppercase_ascii_run_immediately_after_hyphen_in_roman_sequence` | 952 | 724 | 228 | 0 | +| `uppercase_ascii_segments_joined_by_ampersand_capitalization` | 439 | 361 | 78 | 0 | +| `uppercase_roman_headword_closed_multiword_parenthetical` | 175 | 113 | 62 | 0 | +| `uppercase_roman_run_followed_by_hyphen_digits` | 571 | 504 | 67 | 0 | +| `uppercase_word_after_whitespace_continuing_ascii_roman_text` | 1729 | 1115 | 614 | 0 | + +### `allcaps_roman_run_beginning_with_pure_letter_shortform` + +Of the 3242 candidates, 321 are the actual `pending_rule_review` subcluster. The other 2921 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 433 mismatches were evaluable and 52 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2820 ⠠ -> U+2830 ⠰`: 50 +- `U+2830 ⠰ -> U+2820 ⠠`: 2 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 112 +- `pending_rule_review`: 321 + +Representative `exact` samples: + +- `sentence_01.json` #8: SK증권은 2일 삼성SDI에 대해 증설 투자를 위한 자금 여력이 가장 우수한 기업이라고 설명하며 완성차와 추가적인 조인트벤처(JV) 설립이 기대된다고 분석했다. 투자의견 ‘매수’와 목표주가 91만원을 유지했다. + - expected: `⠴⠠⠠⠎⠅⠲⠨⠪⠶⠈⠏⠒⠵⠀⠼⠃⠕⠂⠀⠇⠢⠠⠻⠴` + - actual: `⠴⠠⠠⠎⠅⠲⠨⠪⠶⠈⠏⠒⠵⠀⠼⠃⠕⠂⠀⠇⠢⠠⠻⠴` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #356: 한편, 이번에 새롭게 추가한 ‘나이스플러스(NEIS+)’를 통해 수강신청, 학습기록 관리, 수업피드백 등을 관리하여 학생, 학부모, 교사가 서로 소통하는 완벽한 학교 수업지원 플랫폼으로 거듭나기를 기대하고 있다. + - expected: `⠚⠒⠙⠡⠐⠀⠕⠘⠾⠝⠀⠠⠗⠐⠥⠃⠈⠝⠀⠰⠍⠫⠚⠒` + - actual: `⠚⠒⠙⠡⠐⠀⠕⠘⠾⠝⠀⠠⠗⠐⠥⠃⠈⠝⠀⠰⠍⠫⠚⠒` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #30: “챗GPT보다 가벼운 여러 대규모언어모델(LLM)이 출현해 오픈AI 독점 구도에 균열을 내고 있다. 한국어 특화 LLM으로 글로벌 경쟁에 대응하면서 세부 서비스(튜닝) 모델에서 차별성을 확보하는 전략이 중요하다.” + - expected: `⠦⠰⠗⠄⠴⠠⠠⠛⠏⠞⠲⠘⠥⠊⠀⠫⠘⠱⠛⠀⠱⠐⠎⠀` + - actual: `⠦⠰⠗⠄⠴⠠⠠⠛⠏⠞⠲⠘⠥⠊⠀⠫⠘⠱⠛⠀⠱⠐⠎⠀` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #81: SGI는 “산업연관분석을 활용해 우리나라의 대일 수출 증가가 국내총생산(GDP)에 미치는 영향을 계산해 보면 경제성장률은 0.1%포인트 높아질 것”이라고 밝혔다. + - expected: `⠴⠠⠠⠎⠛⠊⠲⠉⠵⠀⠦⠇⠒⠎⠃⠡⠈⠧⠒⠘⠛⠠⠹⠮` + - actual: `⠴⠠⠠⠎⠛⠊⠲⠉⠵⠀⠦⠇⠒⠎⠃⠡⠈⠧⠒⠘⠛⠠⠹⠮` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_01.json` #3447: 점검 내용은 폐수 처리 업체 등록 기준, 준수 사항, 방류 수 배출 허용 기준 준수 여부 등이다. 수질원격감시체계(TMS) 설치 지원 사업 상담·기술 지원도 병행한다. + - expected: `⠕⠰⠝⠈⠌⠦⠄⠴⠠⠠⠞⠍⠎⠠⠴⠀⠠⠞⠰⠕⠀⠨⠕⠏` + - actual: `⠕⠰⠝⠈⠌⠦⠄⠴⠰⠠⠠⠞⠍⠎⠠⠴⠀⠠⠞⠰⠕⠀⠨⠕` + - first differing cell (zero-based): 110 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #1533: 지난 21일 개통된 4세대 교육행정 정보시스템 ‘나이스(NEIS)’에 대한 현직 교사들의 불만족도가 높은 가운데 울산에서도 시스템 불안정에 대한 문제가 제기됐다. + - expected: `⠉⠣⠕⠠⠪⠦⠄⠴⠠⠠⠝⠑⠊⠎⠠⠴⠴⠄⠝⠀⠊⠗⠚⠒` + - actual: `⠉⠣⠕⠠⠪⠦⠄⠴⠰⠠⠠⠝⠑⠊⠎⠠⠴⠴⠄⠝⠀⠊⠗⠚` + - first differing cell (zero-based): 58 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #5589: 플래티넘 모델에만 탑재된 사양으로는 헤드업디스플레이(HUD), 레인 센서, 파노라믹뷰 모니터, 디지털 리어뷰 미러, 2열 열선시트, 자동 전조등 시스템(AFS) 등이 있다. + - expected: `⠠⠪⠓⠝⠢⠦⠄⠴⠠⠠⠁⠋⠎⠠⠴⠀⠊⠪⠶⠕⠀⠕⠌⠊` + - actual: `⠠⠪⠓⠝⠢⠦⠄⠴⠰⠠⠠⠁⠋⠎⠠⠴⠀⠊⠪⠶⠕⠀⠕⠌` + - first differing cell (zero-based): 147 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #119: 21일 국제금융센터에 따르면 KB국민은행의 지난 17일 신용부도스와프(CDS) 프리미엄은 43bp(1bp는 0.01%포인트)로 일주일 전 대비 1bp 상승하는 데 그쳤다. + - expected: `⠠⠪⠧⠙⠪⠦⠄⠴⠠⠠⠉⠙⠎⠠⠴⠀⠙⠪⠐⠕⠑⠕⠎⠢` + - actual: `⠠⠪⠧⠙⠪⠦⠄⠴⠰⠠⠠⠉⠙⠎⠠⠴⠀⠙⠪⠐⠕⠑⠕⠎` + - first differing cell (zero-based): 74 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #191: 소프트웨어 측면에서는 스마트 TV 독자 운영체제 웹(web)OS의 진화를 앞세워 맞춤형 고객경험과 CDX(Cross Device eXperience) 경험을 강화한다. + - expected: `⠰⠝⠨⠝⠀⠏⠗⠃⠴⠐⠣⠺⠑⠃⠐⠜⠠⠠⠕⠎⠲⠺⠀⠨` + - actual: `⠰⠝⠨⠝⠀⠏⠗⠃⠦⠄⠴⠺⠑⠃⠠⠴⠴⠠⠠⠕⠎⠲⠺⠀` + - first differing cell (zero-based): 48 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #99: 황승주 씨는 “기능성 소화불량증, 특히 PDS(식후불편감증후군) 복부팽만감 유형으로 고통 받는 환자들에게 적용가능한 안전하고 높은 치료효율의 한약소재를 개발해 국민건강에 이바지하고자 한다”고 말했다. + - expected: `⠀⠓⠪⠁⠚⠕⠀⠴⠰⠠⠠⠏⠙⠎⠦⠄⠠⠕⠁⠚⠍⠘⠯⠙` + - actual: `⠀⠓⠪⠁⠚⠕⠀⠴⠠⠠⠏⠙⠎⠦⠄⠠⠕⠁⠚⠍⠘⠯⠙⠡` + - first differing cell (zero-based): 45 + - current primary/reason: `corpus_suspect` / `ueb_grade1_before_nonstanding_opening_parenthesis` +- `sentence_03.json` #6: 지난해 9월 HMM과 파나시아가 선박용 탄소포집 시스템 공동연구에 대한 업무협약(MOU)을 체결한 뒤 7개월 만에 내놓은 연구성과다. + - expected: `⠜⠁⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠒⠀⠊⠍⠗` + - actual: `⠜⠁⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠒⠀⠊⠍⠗⠀` + - first differing cell (zero-based): 83 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #19: 이날 행사에는 구 대표 외에 권봉석 (주)LG 최고운영책임자(COO), 박일평 LG사이언스파크대표(사장)를 비롯해 각 계열사 최고기술책임자(CTO), 최고데이터책임자(CDO), 최고인사책임자(CHO) 등이 참석해 국내 이공계 R&D 인재 400여명과 만났다. + - expected: `⠁⠕⠢⠨⠦⠄⠴⠠⠠⠉⠓⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢⠠` + - actual: `⠁⠕⠢⠨⠦⠄⠴⠠⠉⠠⠓⠠⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢` + - first differing cell (zero-based): 188 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `allcaps_roman_run_containing_ar` + +Of the 1022 candidates, 521 are the actual `pending_rule_review` subcluster. The other 501 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 542 mismatches were evaluable and 416 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2801 ⠁ -> U+281C ⠜`: 411 +- `U+2810 ⠐ -> U+2815 ⠕`: 3 +- `U+2800 ⠀ -> U+2820 ⠠`: 2 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 18 +- `pending_rule_review`: 521 +- `unsupported_character_review`: 3 + +Representative `exact` samples: + +- `sentence_01.json` #54: 에버소울은 출시 전부터 지스타(G-STAR) 2022, 에이지에프(AGF) 2022 등 굵직한 국내 게임 및 서브컬처 행사에 출품해 기대를 모았다. 12월 말에는 사전 예약자 150만 명을 모객했다. + - expected: `⠝⠘⠎⠠⠥⠯⠵⠀⠰⠯⠠⠕⠀⠨⠾⠘⠍⠓⠎⠀⠨⠕⠠⠪` + - actual: `⠝⠘⠎⠠⠥⠯⠵⠀⠰⠯⠠⠕⠀⠨⠾⠘⠍⠓⠎⠀⠨⠕⠠⠪` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #344: 한편 이장우 대전시장은 이틀간의 대만 방문을 마무리하고 싱가포르로 이동해 18일까지 Merck(머크) 사, 국립싱가포르대학 바이오연구단, A-STAR 및 바이오폴리스 등 바이오산업 관련 해외 기업과 연구기관을 방문할 예정이다. + - expected: `⠚⠒⠙⠡⠀⠕⠨⠶⠍⠀⠊⠗⠨⠾⠠⠕⠨⠶⠵⠀⠕⠓⠮⠫` + - actual: `⠚⠒⠙⠡⠀⠕⠨⠶⠍⠀⠊⠗⠨⠾⠠⠕⠨⠶⠵⠀⠕⠓⠮⠫` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #192: 국내 시총 상위 10개 기업에 투자하는 ETF 역시 코스피를 뛰어넘었다. 일례로 NH-아문디자산운용의 하나로(HANARO) 200 TOP10 ETF는 연초 이후 17.6% 올라 같은 기간 코스피 상승률(14.3%)을 넘어섰다. + - expected: `⠈⠍⠁⠉⠗⠀⠠⠕⠰⠿⠀⠇⠶⠍⠗⠀⠼⠁⠚⠈⠗⠀⠈⠕` + - actual: `⠈⠍⠁⠉⠗⠀⠠⠕⠰⠿⠀⠇⠶⠍⠗⠀⠼⠁⠚⠈⠗⠀⠈⠕` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #596: 롯데백화점은 바닷가와 도심 등에 이어 올 봄에는 서울 경희궁 공원 입구에 ‘리얼스 마켓(RE:EARTH)’을 연다고 10일 밝혔다. + - expected: `⠐⠥⠄⠊⠝⠘⠗⠁⠚⠧⠨⠎⠢⠵⠀⠘⠊⠄⠫⠧⠀⠊⠥⠠` + - actual: `⠐⠥⠄⠊⠝⠘⠗⠁⠚⠧⠨⠎⠢⠵⠀⠘⠊⠄⠫⠧⠀⠊⠥⠠` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_01.json` #142: ‘메타버스존’에서는 증강현실(AR)·가상현실(VR)에 적용되는 혁신기술이 소개된다. 글라스를 착용하고 최첨단 3D 센싱모듈이 구현하는 가상현실을 관람객이 직접 체험할 수 있도록 했다. + - expected: `⠠⠕⠂⠦⠄⠴⠠⠠⠁⠗⠠⠴⠐⠆⠫⠇⠶⠚⠡⠠⠕⠂⠦⠄` + - actual: `⠠⠕⠂⠦⠄⠴⠠⠠⠜⠠⠴⠐⠆⠫⠇⠶⠚⠡⠠⠕⠂⠦⠄⠴` + - first differing cell (zero-based): 34 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #28: 체납된 세금은 전국 어디서나 은행 현금자동인출기(ATM)를 이용해 고지서 없이 현금과 신용카드로 납부가 가능하며, 자동응답시스템(ARS) 지방세 납부서비스(044-300-7114)를 이용해 가상계좌를 확인해 계좌이체를 하거나 신용카드로 납부할 수도 있다. + - expected: `⠓⠝⠢⠦⠄⠴⠠⠠⠁⠗⠎⠠⠴⠀⠨⠕⠘⠶⠠⠝⠀⠉⠃⠘` + - actual: `⠓⠝⠢⠦⠄⠴⠠⠠⠜⠎⠠⠴⠀⠨⠕⠘⠶⠠⠝⠀⠉⠃⠘⠍` + - first differing cell (zero-based): 129 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #52: 종합식품기업 하림이 공룡 모양으로 생긴 동그랑땡 ‘용가리 땡’을 20일 출시했다. 용가리땡에는 특별 제작한 증강현실(AR) 공룡 카드까지 한 장씩 들어 있어 아이들의 많은 관심이 예상된다. + - expected: `⠠⠕⠂⠦⠄⠴⠠⠠⠁⠗⠠⠴⠀⠈⠿⠐⠬⠶⠀⠋⠊⠪⠠⠫` + - actual: `⠠⠕⠂⠦⠄⠴⠠⠠⠜⠠⠴⠀⠈⠿⠐⠬⠶⠀⠋⠊⠪⠠⠫⠨` + - first differing cell (zero-based): 126 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #187: 주정차단속 ARS 알림서비스는 기존 불법 주정차 문자 알림서비스를 업그레이드 해 문자와 함께 자동응답서비스(ARS)로도 주정차단속을 알려주는 시스템이다. + - expected: `⠊⠒⠠⠭⠀⠴⠠⠠⠁⠗⠎⠲⠀⠣⠂⠐⠕⠢⠠⠎⠘⠕⠠⠪` + - actual: `⠊⠒⠠⠭⠀⠴⠠⠠⠜⠎⠲⠀⠣⠂⠐⠕⠢⠠⠎⠘⠕⠠⠪⠉` + - first differing cell (zero-based): 14 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #142: ‘메타버스존’에서는 증강현실(AR)·가상현실(VR)에 적용되는 혁신기술이 소개된다. 글라스를 착용하고 최첨단 3D 센싱모듈이 구현하는 가상현실을 관람객이 직접 체험할 수 있도록 했다. + - expected: `⠠⠕⠂⠦⠄⠴⠠⠠⠁⠗⠠⠴⠐⠆⠫⠇⠶⠚⠡⠠⠕⠂⠦⠄` + - actual: `⠠⠕⠂⠦⠄⠴⠠⠠⠜⠠⠴⠐⠆⠫⠇⠶⠚⠡⠠⠕⠂⠦⠄⠴` + - first differing cell (zero-based): 34 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #28: 체납된 세금은 전국 어디서나 은행 현금자동인출기(ATM)를 이용해 고지서 없이 현금과 신용카드로 납부가 가능하며, 자동응답시스템(ARS) 지방세 납부서비스(044-300-7114)를 이용해 가상계좌를 확인해 계좌이체를 하거나 신용카드로 납부할 수도 있다. + - expected: `⠓⠝⠢⠦⠄⠴⠠⠠⠁⠗⠎⠠⠴⠀⠨⠕⠘⠶⠠⠝⠀⠉⠃⠘` + - actual: `⠓⠝⠢⠦⠄⠴⠠⠠⠜⠎⠠⠴⠀⠨⠕⠘⠶⠠⠝⠀⠉⠃⠘⠍` + - first differing cell (zero-based): 129 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #52: 종합식품기업 하림이 공룡 모양으로 생긴 동그랑땡 ‘용가리 땡’을 20일 출시했다. 용가리땡에는 특별 제작한 증강현실(AR) 공룡 카드까지 한 장씩 들어 있어 아이들의 많은 관심이 예상된다. + - expected: `⠠⠕⠂⠦⠄⠴⠠⠠⠁⠗⠠⠴⠀⠈⠿⠐⠬⠶⠀⠋⠊⠪⠠⠫` + - actual: `⠠⠕⠂⠦⠄⠴⠠⠠⠜⠠⠴⠀⠈⠿⠐⠬⠶⠀⠋⠊⠪⠠⠫⠨` + - first differing cell (zero-based): 126 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #187: 주정차단속 ARS 알림서비스는 기존 불법 주정차 문자 알림서비스를 업그레이드 해 문자와 함께 자동응답서비스(ARS)로도 주정차단속을 알려주는 시스템이다. + - expected: `⠊⠒⠠⠭⠀⠴⠠⠠⠁⠗⠎⠲⠀⠣⠂⠐⠕⠢⠠⠎⠘⠕⠠⠪` + - actual: `⠊⠒⠠⠭⠀⠴⠠⠠⠜⠎⠲⠀⠣⠂⠐⠕⠢⠠⠎⠘⠕⠠⠪⠉` + - first differing cell (zero-based): 14 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `allcaps_roman_run_containing_ed` + +Of the 816 candidates, 386 are the actual `pending_rule_review` subcluster. The other 430 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 454 mismatches were evaluable and 341 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2811 ⠑ -> U+282B ⠫`: 339 +- `U+2810 ⠐ -> U+280E ⠎`: 1 +- `U+2810 ⠐ -> U+281D ⠝`: 1 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 68 +- `pending_rule_review`: 386 + +Representative `exact` samples: + +- `sentence_01.json` #26: LG전자 전시관 입구에는 올레드(OLED) 플렉서블 사이니지 260장을 이어 붙인 초대형 조형물 ‘올레드 지평선’이 관람객들의 이목을 집중시킨다. + - expected: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠀⠨⠾⠠⠕⠈⠧⠒⠀⠕⠃⠈⠍⠝⠉` + - actual: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠀⠨⠾⠠⠕⠈⠧⠒⠀⠕⠃⠈⠍⠝⠉` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #44: 아울러 국가응급진료정보망(NEDIS)의 개인정보 수집·연계 법적 근거를 마련해 구급활동일지, 건강보험진료기록과의 연계를 통해 응급환자에 대해 이송부터 응급실 진료, 의료기관 퇴원까지 단절 없는(seamless) 응급의료데이터 체계를 구축한다. + - expected: `⠣⠯⠐⠎⠀⠈⠍⠁⠫⠪⠶⠈⠪⠃⠨⠟⠐⠬⠨⠻⠘⠥⠑⠶` + - actual: `⠣⠯⠐⠎⠀⠈⠍⠁⠫⠪⠶⠈⠪⠃⠨⠟⠐⠬⠨⠻⠘⠥⠑⠶` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #423: 아이씨에이치는 그동안 삼성전자에 필름형 박막안테나(MFA), 전자파 차폐용 가스켓, IT기기용 테이프 등 스마트폰 부품 소재를 공급해왔으며, 지난해부터는 디스플레이용 복합소재 등 OLED 부품 소재로도 영역을 확대하고 있다. + - expected: `⠣⠕⠠⠠⠕⠝⠕⠰⠕⠉⠵⠀⠈⠪⠊⠿⠣⠒⠀⠇⠢⠠⠻⠨` + - actual: `⠣⠕⠠⠠⠕⠝⠕⠰⠕⠉⠵⠀⠈⠪⠊⠿⠣⠒⠀⠇⠢⠠⠻⠨` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #484: 윤석열 대통령은 4일 첨단 디스플레이산업과 관련해 “민간이 적기에 투자할 수 있도록 인센티브를 확보하고 OLED(유기발광다이오드) 기술 고도화를 지원하면서 글로벌시장에서 압도적인 1위를 계속 견지하도록 만들겠다”고 말했다. + - expected: `⠩⠒⠠⠹⠳⠀⠊⠗⠓⠿⠐⠻⠵⠀⠼⠙⠕⠂⠀⠰⠎⠢⠊⠒` + - actual: `⠩⠒⠠⠹⠳⠀⠊⠗⠓⠿⠐⠻⠵⠀⠼⠙⠕⠂⠀⠰⠎⠢⠊⠒` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_01.json` #51: 와이캅 픽셀은 와이어·패키지·렌즈가 필요 없는 ‘와이캅’ 기술을 기반으로 한다. 적녹청(RGB) 3개의 마이크로 LED를 수직 방향으로 적층한 세계 최초의 풀 컬러 원칩 기술이다. + - expected: `⠪⠐⠥⠀⠴⠠⠠⠇⠑⠙⠲⠐⠮⠀⠠⠍⠨⠕⠁⠀⠘⠶⠚⠜` + - actual: `⠪⠐⠥⠀⠴⠠⠠⠇⠫⠲⠐⠮⠀⠠⠍⠨⠕⠁⠀⠘⠶⠚⠜⠶` + - first differing cell (zero-based): 107 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #33: 지원은 학생이 영재교육종합데이터베이스(GED)를 통해 원서 접수하고 교사의 추천을 받아 진행하고, 자세한 내용은 학교 및 영재교육 기관별 홈페이지에 탑재되어 있다. + - expected: `⠠⠪⠦⠄⠴⠠⠠⠛⠑⠙⠠⠴⠐⠮⠀⠓⠿⠚⠗⠀⠏⠒⠠⠎` + - actual: `⠠⠪⠦⠄⠴⠠⠠⠛⠫⠠⠴⠐⠮⠀⠓⠿⠚⠗⠀⠏⠒⠠⠎⠀` + - first differing cell (zero-based): 40 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #209: 포드의 글로벌 트럭 디자인 DNA를 토대로 강인하면서도 다양한 사용 목적에 부합하는 실용적인 내·외부를 구현했다. 전면부 시그니처 C클램프 헤드라이트, 매트릭스 LED 헤드라이트, 포드(FORD) 레터링이 특징이다. + - expected: `⠁⠠⠪⠀⠴⠠⠠⠇⠑⠙⠲⠀⠚⠝⠊⠪⠐⠣⠕⠓⠪⠐⠀⠙` + - actual: `⠁⠠⠪⠀⠴⠠⠠⠇⠫⠲⠀⠚⠝⠊⠪⠐⠣⠕⠓⠪⠐⠀⠙⠥` + - first differing cell (zero-based): 159 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #388: 김 신부는 그동안 제작한 스테인드글라스 작품은 물론 회화·LED(발광다이오드)조명작품·도자기 등 60여점의 작품을 전시한다. 그는 “형상을 떠난 자유로움과 원초적인 아름다움에 대한 깊이를 관람객들에게 전달하고 싶다”고 밝혔다. + - expected: `⠚⠧⠐⠆⠴⠠⠠⠇⠑⠙⠦⠄⠘⠂⠈⠧⠶⠊⠣⠕⠥⠊⠪⠠` + - actual: `⠚⠧⠐⠆⠴⠠⠠⠇⠫⠦⠄⠘⠂⠈⠧⠶⠊⠣⠕⠥⠊⠪⠠⠴` + - first differing cell (zero-based): 61 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #35: 삼성전자는 올해 CES에서 77인치 OLED TV를 처음으로 공개할 예정이다. 이는 지난 8월 ‘국제정보디스플레이학술대회(IMID) 2022’에서 삼성디스플레이가 공개한 77인치 퀀텀닷(QD)-OLED 패널을 탑재한 제품이다. + - expected: `⠋⠏⠒⠓⠎⠢⠊⠄⠴⠐⠣⠟⠙⠐⠜⠤⠠⠠⠕⠇⠫⠲⠀⠙` + - actual: `⠋⠏⠒⠓⠎⠢⠊⠄⠦⠄⠴⠠⠠⠟⠙⠠⠴⠤⠴⠠⠠⠕⠇⠫` + - first differing cell (zero-based): 172 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #33: 지원은 학생이 영재교육종합데이터베이스(GED)를 통해 원서 접수하고 교사의 추천을 받아 진행하고, 자세한 내용은 학교 및 영재교육 기관별 홈페이지에 탑재되어 있다. + - expected: `⠠⠪⠦⠄⠴⠠⠠⠛⠑⠙⠠⠴⠐⠮⠀⠓⠿⠚⠗⠀⠏⠒⠠⠎` + - actual: `⠠⠪⠦⠄⠴⠠⠠⠛⠫⠠⠴⠐⠮⠀⠓⠿⠚⠗⠀⠏⠒⠠⠎⠀` + - first differing cell (zero-based): 40 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #209: 포드의 글로벌 트럭 디자인 DNA를 토대로 강인하면서도 다양한 사용 목적에 부합하는 실용적인 내·외부를 구현했다. 전면부 시그니처 C클램프 헤드라이트, 매트릭스 LED 헤드라이트, 포드(FORD) 레터링이 특징이다. + - expected: `⠁⠠⠪⠀⠴⠠⠠⠇⠑⠙⠲⠀⠚⠝⠊⠪⠐⠣⠕⠓⠪⠐⠀⠙` + - actual: `⠁⠠⠪⠀⠴⠠⠠⠇⠫⠲⠀⠚⠝⠊⠪⠐⠣⠕⠓⠪⠐⠀⠙⠥` + - first differing cell (zero-based): 159 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #62: LG디스플레이가 자사의 유기발광다이오드(OLED) TV 패널이 글로벌 친환경 인증기관인 카본 트러스트(Carbon Trust)로 부터 탄소발자국 인증을 획득했다고 19일 밝혔다. + - expected: `⠧⠶⠊⠣⠕⠥⠊⠪⠴⠐⠣⠠⠠⠕⠇⠫⠐⠜⠀⠠⠠⠞⠧⠲` + - actual: `⠧⠶⠊⠣⠕⠥⠊⠪⠦⠄⠴⠠⠠⠕⠇⠫⠠⠴⠀⠴⠠⠠⠞⠧` + - first differing cell (zero-based): 35 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` + +### `allcaps_roman_run_containing_ou` + +Of the 1816 candidates, 1679 are the actual `pending_rule_review` subcluster. The other 137 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 1688 mismatches were evaluable and 1506 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2815 ⠕ -> U+2833 ⠳`: 1504 +- `U+2810 ⠐ -> U+283D ⠽`: 2 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 9 +- `pending_rule_review`: 1679 + +Representative `exact` samples: + +- `sentence_01.json` #850: 신규 보급 단말기는 앞서 국내 신용카드사들이 합작해 만든 근거리 무선 통신(NFC) 단말기 결제 규격인 ‘저스터치(JUSTOUCH)’와 호환성을 갖춰야 한다. + - expected: `⠠⠟⠈⠩⠀⠘⠥⠈⠪⠃⠀⠊⠒⠑⠂⠈⠕⠉⠵⠀⠣⠲⠠⠎` + - actual: `⠠⠟⠈⠩⠀⠘⠥⠈⠪⠃⠀⠊⠒⠑⠂⠈⠕⠉⠵⠀⠣⠲⠠⠎` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #727: 또 천안 살펴유(YOU) 모바일 앱 시행, 1인가구 밀키트 지원사업을 실시해 1인 가구 등의 고독사 취약계층 모니터링 사업을 추진하는 등 고독사 예방에도 총력을 기울이고 있다. + - expected: `⠠⠊⠥⠀⠰⠾⠣⠒⠀⠇⠂⠙⠱⠩⠦⠄⠴⠠⠠⠽⠳⠠⠴⠀` + - actual: `⠠⠊⠥⠀⠰⠾⠣⠒⠀⠇⠂⠙⠱⠩⠦⠄⠴⠠⠠⠽⠳⠠⠴⠀` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #606: 앞서 방탄소년단(BTS)의 2020년 정규 4집 ‘MAP OF THE SOUL : 7’의 337만여 장 초동 기록을 넘긴 것이다. + - expected: `⠣⠲⠠⠎⠀⠘⠶⠓⠒⠠⠥⠉⠡⠊⠒⠦⠄⠴⠠⠠⠃⠞⠎⠠` + - actual: `⠣⠲⠠⠎⠀⠘⠶⠓⠒⠠⠥⠉⠡⠊⠒⠦⠄⠴⠠⠠⠃⠞⠎⠠` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #6975: 이 밖에 포시즌스 호텔 바 ‘오울(OUL)’에서는 후 환유 라인의 대표 성분인 ‘삼’을 비롯해 구기자, 식초, 청귤 등을 활용해 만든 칵테일 3종도 판매한다. + - expected: `⠕⠀⠘⠁⠁⠝⠀⠙⠥⠠⠕⠨⠵⠠⠪⠀⠚⠥⠓⠝⠂⠀⠘⠀` + - actual: `⠕⠀⠘⠁⠁⠝⠀⠙⠥⠠⠕⠨⠵⠠⠪⠀⠚⠥⠓⠝⠂⠀⠘⠀` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_01.json` #47: 다날은 계열사 ‘제프’가 국내 대체불가토큰(NFT) 거래소를 운영하는 ‘팔라’와 메타버스·NFT 협력 관련 협약(MOU)을 맺고 메타버스 플랫폼 ‘제프월드’의 인프라 확대를 추진한다고 3일 밝혔다. + - expected: `⠜⠁⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠑⠗⠅⠈⠥⠀⠑⠝⠓⠘` + - actual: `⠜⠁⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠑⠗⠅⠈⠥⠀⠑⠝⠓⠘⠎` + - first differing cell (zero-based): 114 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #128: 이번 방문에서 대표단은 우수 외투기업과 투자협약(MOU)을 체결하고 투자 상담, 기업정보 교류 등 적극적인 외자 유치 활동을 펼칠 계획이다. + - expected: `⠜⠁⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠈⠥⠀⠓⠍` + - actual: `⠜⠁⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠈⠥⠀⠓⠍⠨` + - first differing cell (zero-based): 48 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #1: 김 지사는 18일(이하 현지시각) 미국 코네티컷주 댄버리 린데 본사에서 산지브 람바(Sanjiv Lamba) 린데 회장, 성백석 린데코리아 회장, 조일교 아산시 부시장과 투자양해각서(MOU)을 체결했다. + - expected: `⠠⠎⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - actual: `⠠⠎⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - first differing cell (zero-based): 175 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #191: SK온은 23일 서울 종로구 SK서린사옥에서 에코프로머티리얼즈, 중국의 GEM(거린메이)과 전구체 생산을 위한 3자 합작법인인 지이엠코리아뉴에너지머티리얼즈㈜(이하 지이엠코리아) 설립 양해각서(MOU)를 체결했다고 밝혔다. + - expected: `⠠⠎⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠐⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈` + - actual: `⠠⠎⠦⠄⠴⠠⠠⠍⠳⠠⠴⠐⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈⠥` + - first differing cell (zero-based): 191 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `allcaps_roman_run_containing_st` + +Of the 1479 candidates, 625 are the actual `pending_rule_review` subcluster. The other 854 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 664 mismatches were evaluable and 473 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+280E ⠎ -> U+280C ⠌`: 466 +- `U+2820 ⠠ -> U+280C ⠌`: 2 +- `U+280B ⠋ -> U+2820 ⠠`: 1 +- `U+280C ⠌ -> U+280E ⠎`: 1 +- `U+2810 ⠐ -> U+2811 ⠑`: 1 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 39 +- `pending_rule_review`: 625 + +Representative `exact` samples: + +- `sentence_01.json` #44: 지니타임티켓은 지니뮤직 실시간 라이브 공연 플랫폼 ‘스테이지(STAYG)’를 통해 오프라인 공연을 특별가에 구매·감상할 수 있는 서비스다. + - expected: `⠨⠕⠉⠕⠓⠣⠕⠢⠓⠕⠋⠝⠄⠵⠀⠨⠕⠉⠕⠑⠩⠨⠕⠁` + - actual: `⠨⠕⠉⠕⠓⠣⠕⠢⠓⠕⠋⠝⠄⠵⠀⠨⠕⠉⠕⠑⠩⠨⠕⠁` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #82: 대전 이전을 진행 중인 방위사업청 과장급 직원 110여 명이 한국과학기술원(KAIST) 을지연구소가 주관한 ‘국방연구개발 아카데미’에 참석하기 위해 대전을 찾았다. + - expected: `⠊⠗⠨⠾⠀⠕⠨⠾⠮⠀⠨⠟⠚⠗⠶⠀⠨⠍⠶⠟⠀⠘⠶⠍` + - actual: `⠊⠗⠨⠾⠀⠕⠨⠾⠮⠀⠨⠟⠚⠗⠶⠀⠨⠍⠶⠟⠀⠘⠶⠍` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #281: 대구경북과학기술원(DGIST) 소속 구성원들이 ‘4월 과학의 달’을 맞아 과학연구 성과 등에 기여한 공로로 표창 수상자를 대거 배출했다. + - expected: `⠊⠗⠈⠍⠈⠻⠘⠍⠁⠈⠧⠚⠁⠈⠕⠠⠯⠏⠒⠦⠄⠴⠠⠠` + - actual: `⠊⠗⠈⠍⠈⠻⠘⠍⠁⠈⠧⠚⠁⠈⠕⠠⠯⠏⠒⠦⠄⠴⠠⠠` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #615: 지난달 연세대와도 인력 육성 업무협약을 체결한 지 2주 만이다. 앞서 포스코퓨처엠은 지난해부터 포스텍(POSTECH), 울산과학기술원(UNIST), 한양대, 대구경북과학기술원(DGIST) 등과도 배터리소재 인재 양성을 위한 협약을 맺은 바 있다. + - expected: `⠨⠕⠉⠒⠊⠂⠀⠡⠠⠝⠊⠗⠧⠊⠥⠀⠟⠐⠱⠁⠀⠩⠁⠠` + - actual: `⠨⠕⠉⠒⠊⠂⠀⠡⠠⠝⠊⠗⠧⠊⠥⠀⠟⠐⠱⠁⠀⠩⠁⠠` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_01.json` #167: 올해 전 세계 반도체 시장 전망도 암울하다. 세계반도체시장통계기구(WSTS)에 따르면 올해 글로벌 반도체 매출은 5천565억 달러(약 706조5천880억원)로 지난해(5천801억달러) 대비 4.1% 감소할 것으로 조사됐다. + - expected: `⠈⠍⠦⠄⠴⠠⠠⠺⠎⠞⠎⠠⠴⠝⠀⠠⠊⠐⠪⠑⠡⠀⠥⠂` + - actual: `⠈⠍⠦⠄⠴⠠⠠⠺⠌⠎⠠⠴⠝⠀⠠⠊⠐⠪⠑⠡⠀⠥⠂⠚` + - first differing cell (zero-based): 67 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #244: 충남대학교 이진숙 총장을 비롯한 관계자들은 5월 19일, 베트남 최고의 국립대인 하노이과학기술대학(HUST), 베트남국립농업대학(VNUA)을 연이어 방문해 ‘오픈 캠퍼스(open campus)’ 설립을 골자로 한 협정을 체결했다. + - expected: `⠁⠦⠄⠴⠠⠠⠓⠥⠎⠞⠠⠴⠐⠀⠘⠝⠓⠪⠉⠢⠈⠍⠁⠐` + - actual: `⠁⠦⠄⠴⠠⠠⠓⠥⠌⠠⠴⠐⠀⠘⠝⠓⠪⠉⠢⠈⠍⠁⠐⠕` + - first differing cell (zero-based): 101 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #142: 수많은 작품 OST를 책임지며 실력파 프로듀서로 인정받고 있는 필승불패W, 리디아(Lydia), 장석원이 힘을 합쳐 만든 곡으로 웰메이드 OST 탄생을 예감케 한다. + - expected: `⠙⠍⠢⠀⠴⠠⠠⠕⠎⠞⠲⠐⠮⠀⠰⠗⠁⠕⠢⠨⠕⠑⠱⠀` + - actual: `⠙⠍⠢⠀⠴⠠⠠⠕⠌⠲⠐⠮⠀⠰⠗⠁⠕⠢⠨⠕⠑⠱⠀⠠` + - first differing cell (zero-based): 17 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #375: 윤석열 대통령은 30일 캐서린 타이 미국 무역대표부(USTR) 대표를 만나 미 인플레이션감축법(IRA)·반도체지원법(CHIPS Act)과 관련해 “미국에 진출하는 한국 기업들이 어려움을 겪지 않도록 우호적인 방향으로 배려해 달라”고 당부했다. + - expected: `⠘⠍⠦⠄⠴⠠⠠⠥⠎⠞⠗⠠⠴⠀⠊⠗⠙⠬⠐⠮⠀⠑⠒⠉` + - actual: `⠘⠍⠦⠄⠴⠠⠠⠥⠌⠗⠠⠴⠀⠊⠗⠙⠬⠐⠮⠀⠑⠒⠉⠀` + - first differing cell (zero-based): 53 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #167: 올해 전 세계 반도체 시장 전망도 암울하다. 세계반도체시장통계기구(WSTS)에 따르면 올해 글로벌 반도체 매출은 5천565억 달러(약 706조5천880억원)로 지난해(5천801억달러) 대비 4.1% 감소할 것으로 조사됐다. + - expected: `⠈⠍⠦⠄⠴⠠⠠⠺⠎⠞⠎⠠⠴⠝⠀⠠⠊⠐⠪⠑⠡⠀⠥⠂` + - actual: `⠈⠍⠦⠄⠴⠠⠠⠺⠌⠎⠠⠴⠝⠀⠠⠊⠐⠪⠑⠡⠀⠥⠂⠚` + - first differing cell (zero-based): 67 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #244: 충남대학교 이진숙 총장을 비롯한 관계자들은 5월 19일, 베트남 최고의 국립대인 하노이과학기술대학(HUST), 베트남국립농업대학(VNUA)을 연이어 방문해 ‘오픈 캠퍼스(open campus)’ 설립을 골자로 한 협정을 체결했다. + - expected: `⠁⠦⠄⠴⠠⠠⠓⠥⠎⠞⠠⠴⠐⠀⠘⠝⠓⠪⠉⠢⠈⠍⠁⠐` + - actual: `⠁⠦⠄⠴⠠⠠⠓⠥⠌⠠⠴⠐⠀⠘⠝⠓⠪⠉⠢⠈⠍⠁⠐⠕` + - first differing cell (zero-based): 101 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #142: 수많은 작품 OST를 책임지며 실력파 프로듀서로 인정받고 있는 필승불패W, 리디아(Lydia), 장석원이 힘을 합쳐 만든 곡으로 웰메이드 OST 탄생을 예감케 한다. + - expected: `⠙⠍⠢⠀⠴⠠⠠⠕⠎⠞⠲⠐⠮⠀⠰⠗⠁⠕⠢⠨⠕⠑⠱⠀` + - actual: `⠙⠍⠢⠀⠴⠠⠠⠕⠌⠲⠐⠮⠀⠰⠗⠁⠕⠢⠨⠕⠑⠱⠀⠠` + - first differing cell (zero-based): 17 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #375: 윤석열 대통령은 30일 캐서린 타이 미국 무역대표부(USTR) 대표를 만나 미 인플레이션감축법(IRA)·반도체지원법(CHIPS Act)과 관련해 “미국에 진출하는 한국 기업들이 어려움을 겪지 않도록 우호적인 방향으로 배려해 달라”고 당부했다. + - expected: `⠘⠍⠦⠄⠴⠠⠠⠥⠎⠞⠗⠠⠴⠀⠊⠗⠙⠬⠐⠮⠀⠑⠒⠉` + - actual: `⠘⠍⠦⠄⠴⠠⠠⠥⠌⠗⠠⠴⠀⠊⠗⠙⠬⠐⠮⠀⠑⠒⠉⠀` + - first differing cell (zero-based): 53 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `ampersand_before_attached_ascii_roman_segment` + +Of the 30 candidates, 12 are the actual `pending_rule_review` subcluster. The other 18 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 16 mismatches were evaluable and 4 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2808 ⠈ -> U+2834 ⠴`: 3 +- `U+2834 ⠴ -> U+2820 ⠠`: 1 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 4 +- `pending_rule_review`: 12 + +Representative `exact` samples: + +- `sentence_01.json` #202: 제이스코홀딩스는 필리핀 니켈 광산사업을 공동 추진중인 EVM(EV Mining &Development)이 광산지질국(MGB)에 4천700헥타르(약 1천400만평)에 대한 탐사허가(EP)를 신청했다고 9일 밝혔다. + - expected: `⠨⠝⠕⠠⠪⠋⠥⠚⠥⠂⠊⠕⠶⠠⠪⠉⠵⠀⠙⠕⠂⠐⠕⠙` + - actual: `⠨⠝⠕⠠⠪⠋⠥⠚⠥⠂⠊⠕⠶⠠⠪⠉⠵⠀⠙⠕⠂⠐⠕⠙` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #1319: 그는 이날 ‘2023 UNIST 과학&ICT 콘서트’ 행사에 발표자로 나서 연구중심대학을 표방하고 시작한 포스텍, 광주과학기술원(GIST) 등도 20년이 지나면서 고전을 면치 못했다고 설명했다. + - expected: `⠈⠪⠉⠵⠀⠕⠉⠂⠀⠠⠦⠼⠃⠚⠃⠉⠀⠴⠠⠠⠥⠝⠊⠌` + - actual: `⠈⠪⠉⠵⠀⠕⠉⠂⠀⠠⠦⠼⠃⠚⠃⠉⠀⠴⠠⠠⠥⠝⠊⠌` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #3453: 3일 한화자산운용은 자사 펀드 직판 애플리케이션 ‘파인(PINE)’에 청년소장펀드 2종을 탑재했다고 밝혔다. 해당 펀드는 ‘한화 MZ픽 한국&K리츠’ 및 ‘한화 MZ픽 그린테크’ 2종이다. + - expected: `⠼⠉⠕⠂⠀⠚⠒⠚⠧⠨⠇⠒⠛⠬⠶⠵⠀⠨⠇⠀⠙⠾⠊⠪` + - actual: `⠼⠉⠕⠂⠀⠚⠒⠚⠧⠨⠇⠒⠛⠬⠶⠵⠀⠨⠇⠀⠙⠾⠊⠪` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_02.json` #974: 드림&Dream멘토링은 시간과 봉사정신을 내어주는(드림) 대학생 멘토와 그로 인해 꿈(Dream)을 이루는 고등학생 멘티가 함께 만들어 가는 이야기라는 뜻으로 멘티의 학교적응력 향상을 목표로 하는 프로그램이다. + - expected: `⠊⠪⠐⠕⠢⠴⠈⠯⠴⠠⠙⠗⠂⠍⠲⠑⠝⠒⠓⠥⠐⠕⠶⠵` + - actual: `⠊⠪⠐⠕⠢⠴⠈⠯⠠⠙⠗⠂⠍⠲⠑⠝⠒⠓⠥⠐⠕⠶⠵⠀` + - first differing cell (zero-based): 8 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #19011: 앞서 하이브는 이달 투모로우바이투게더(TXT)와 세븐틴을 예고했고, 내달 방탄소년단(BTS) 정국과 엔하이픈 그리고 앤팀(&TEAM)의 앨범 발매 발표한 바 있다. + - expected: `⠀⠗⠒⠓⠕⠢⠦⠄⠈⠯⠠⠠⠞⠂⠍⠠⠴⠺⠀⠗⠂⠘⠎⠢` + - actual: `⠀⠗⠒⠓⠕⠢⠦⠄⠴⠈⠯⠠⠠⠞⠂⠍⠠⠴⠺⠀⠗⠂⠘⠎` + - first differing cell (zero-based): 115 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #4742: 시스템 반도체 설계 전문기업 코아시아가 ‘삼성 파운드리 포럼(SFF)&SAFE™ 포럼 2023’에 참가해 적극적인 글로벌 고객사 확보에 나설 계획이라고 27일 밝혔다. + - expected: `⠐⠕⠀⠙⠥⠐⠎⠢⠴⠐⠣⠠⠠⠎⠋⠋⠐⠜⠈⠯⠠⠠⠎⠁` + - actual: `⠐⠕⠀⠙⠥⠐⠎⠢⠦⠄⠴⠠⠠⠎⠋⠋⠠⠴⠴⠈⠯⠠⠠⠎` + - first differing cell (zero-based): 57 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #974: 드림&Dream멘토링은 시간과 봉사정신을 내어주는(드림) 대학생 멘토와 그로 인해 꿈(Dream)을 이루는 고등학생 멘티가 함께 만들어 가는 이야기라는 뜻으로 멘티의 학교적응력 향상을 목표로 하는 프로그램이다. + - expected: `⠊⠪⠐⠕⠢⠴⠈⠯⠴⠠⠙⠗⠂⠍⠲⠑⠝⠒⠓⠥⠐⠕⠶⠵` + - actual: `⠊⠪⠐⠕⠢⠴⠈⠯⠠⠙⠗⠂⠍⠲⠑⠝⠒⠓⠥⠐⠕⠶⠵⠀` + - first differing cell (zero-based): 8 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #11394: 이어 1일 공개된 3차 라인업에는 동방신기(TVXQ!), 샤이니(SHINee), NCT 127, NCT DREAM, fromis_9(프로미스나인), CRAVITY, NewJeans, xikers, NiziU, &TEAM이 이름을 올리며 총 25팀이 출연을 확정했다. + - expected: `⠊⠿⠘⠶⠠⠟⠈⠕⠦⠄⠴⠠⠠⠞⠧⠭⠟⠖⠠⠴⠐⠀⠠⠜` + - actual: `⠊⠿⠘⠶⠠⠟⠈⠕⠀⠀⠦⠠⠠⠞⠧⠭⠟⠖⠴⠐⠀⠠⠜⠕` + - first differing cell (zero-based): 38 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `ascii_apostrophe_between_ascii_letter_runs` + +Of the 147 candidates, 37 are the actual `pending_rule_review` subcluster. The other 110 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 41 mismatches were evaluable and 3 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+281D ⠝ -> U+280A ⠊`: 2 +- `U+2820 ⠠ -> U+2803 ⠃`: 1 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 4 +- `pending_rule_review`: 37 + +Representative `exact` samples: + +- `sentence_01.json` #67: LG전자는 오는 5일부터 8일까지 ‘CES 2023’이 열리는 미국 라스베이거스 컨벤션센터(LVCC)에 ‘라이프스굿(Life's Good)’을 소개하는 광고판을 설치한다고 3일 밝혔다. + - expected: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠉⠵⠀⠥⠉⠵⠀⠼⠑⠕⠂⠘⠍⠓⠎` + - actual: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠉⠵⠀⠥⠉⠵⠀⠼⠑⠕⠂⠘⠍⠓⠎` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #425: ‘제로(Zero)’는 ‘대전 0시 축제(Daejeon Zero O'clock Festival)’의 영문 이름에서 따왔으며, 축제장에서 3D홀로그램 미디어 콘텐츠 운영을 맡고 있는 애드테크 전문기업 코스윌에서 개발했다. + - expected: `⠠⠦⠨⠝⠐⠥⠦⠄⠴⠠⠵⠻⠕⠠⠴⠴⠄⠉⠵⠀⠠⠦⠊⠗` + - actual: `⠠⠦⠨⠝⠐⠥⠦⠄⠴⠠⠵⠻⠕⠠⠴⠴⠄⠉⠵⠀⠠⠦⠊⠗` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #13417: 트랙리스트에 따르면 이번 앨범에는 첫 번째 트랙 ‘I'm Okay(아이엠 오케이)’를 시작으로 ‘오로라(AURORA)’, ‘PALACE(팰러스)’, ‘PARADE(퍼레이드)’까지 총 네 곡이 수록됐다. + - expected: `⠓⠪⠐⠗⠁⠐⠕⠠⠪⠓⠪⠝⠀⠠⠊⠐⠪⠑⠡⠀⠕⠘⠾⠀` + - actual: `⠓⠪⠐⠗⠁⠐⠕⠠⠪⠓⠪⠝⠀⠠⠊⠐⠪⠑⠡⠀⠕⠘⠾⠀` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #446: 추경호 부총리 겸 기획재정부 장관은 3일 국제신용평가사 무디스(Moody's) 연례 협의단에 미국 인플레이션 감축법(IRA) 등에 따른 국내 기업의 불확실성이 상당 부분 해소됐다고 강조했다. + - expected: `⠰⠍⠈⠻⠚⠥⠀⠘⠍⠰⠿⠐⠕⠀⠈⠱⠢⠀⠈⠕⠚⠽⠁⠨` + - actual: `⠰⠍⠈⠻⠚⠥⠀⠘⠍⠰⠿⠐⠕⠀⠈⠱⠢⠀⠈⠕⠚⠽⠁⠨` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_01.json` #5043: 국제올림픽위원회(IOC)가 승인하는 세계 최대 규모의 청소년 종합 스포츠대회인 국제청소년스포츠축제(International Children's Games)가 ‘다함께 꿈꾸는 미래’를 주제로 지난 6일 대구에서 막을 올렸다. + - expected: `⠁⠰⠝⠁⠇⠀⠠⠡⠝⠄⠎⠀⠠⠛⠁⠍⠑⠎⠠⠴⠫⠀⠠⠦` + - actual: `⠁⠰⠝⠁⠇⠀⠠⠡⠊⠇⠙⠗⠢⠄⠎⠀⠠⠛⠁⠍⠑⠎⠠⠴` + - first differing cell (zero-based): 117 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #6452: 종합 홈 인테리어 전문기업 한샘(대표 김진태)이 온라인몰 자녀방 가구 ‘아임빅(I'M BIG)’과 아동전문 패션브랜드 ‘히로(HIRO)’의 공동 팝업스토어 ‘I'm B!G X HIRO’를 오픈한다고 25일 밝혔다. + - expected: `⠦⠴⠠⠊⠄⠍⠀⠠⠠⠠⠃⠰⠖⠛⠀⠰⠭⠀⠓⠊⠗⠕⠠⠄` + - actual: `⠦⠴⠠⠊⠄⠍⠀⠠⠃⠖⠠⠛⠀⠰⠠⠭⠀⠠⠠⠓⠊⠗⠕⠴` + - first differing cell (zero-based): 160 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #5043: 국제올림픽위원회(IOC)가 승인하는 세계 최대 규모의 청소년 종합 스포츠대회인 국제청소년스포츠축제(International Children's Games)가 ‘다함께 꿈꾸는 미래’를 주제로 지난 6일 대구에서 막을 올렸다. + - expected: `⠁⠰⠝⠁⠇⠀⠠⠡⠝⠄⠎⠀⠠⠛⠁⠍⠑⠎⠠⠴⠫⠀⠠⠦` + - actual: `⠁⠰⠝⠁⠇⠀⠠⠡⠊⠇⠙⠗⠢⠄⠎⠀⠠⠛⠁⠍⠑⠎⠠⠴` + - first differing cell (zero-based): 117 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #12390: 이후에도 이 남성이 또다시 말을 걸며 접근하자, 피해 여성은 “노(No), 돈 바더 미(Don't bother me·귀찮게 하지 마세요)”라고 말한다. + - expected: `⠀⠃⠕⠮⠗⠀⠍⠑⠐⠆⠈⠍⠗⠰⠣⠒⠴⠈⠝⠀⠚⠨⠕⠀` + - actual: `⠀⠃⠕⠮⠗⠀⠍⠑⠲⠐⠆⠈⠍⠗⠰⠣⠒⠴⠈⠝⠀⠚⠨⠕` + - first differing cell (zero-based): 89 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #13506: 바이브(VIBE) 류재현 프로듀싱팀 VIP(VIBE IN PLAY)와 프로듀서 Drei, 가수 한동근이 네이버 웹툰 ‘신부가 필요해’의 첫 번째 음원 ‘I'll Be(아이 윌 비)’를 발매하고 차별화된 감성을 선사했다. + - expected: `⠶⠓⠕⠢⠀⠴⠠⠠⠧⠊⠏⠐⠣⠠⠠⠠⠧⠊⠃⠑⠀⠔⠀⠏` + - actual: `⠶⠓⠕⠢⠀⠴⠠⠠⠠⠧⠊⠏⠐⠣⠧⠊⠃⠑⠀⠊⠝⠀⠏⠇` + - first differing cell (zero-based): 40 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #6452: 종합 홈 인테리어 전문기업 한샘(대표 김진태)이 온라인몰 자녀방 가구 ‘아임빅(I'M BIG)’과 아동전문 패션브랜드 ‘히로(HIRO)’의 공동 팝업스토어 ‘I'm B!G X HIRO’를 오픈한다고 25일 밝혔다. + - expected: `⠦⠴⠠⠊⠄⠍⠀⠠⠠⠠⠃⠰⠖⠛⠀⠰⠭⠀⠓⠊⠗⠕⠠⠄` + - actual: `⠦⠴⠠⠊⠄⠍⠀⠠⠃⠖⠠⠛⠀⠰⠠⠭⠀⠠⠠⠓⠊⠗⠕⠴` + - first differing cell (zero-based): 160 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `ascii_roman_tail_comma_before_digit_korean_token` + +Of the 58 candidates, 20 are the actual `pending_rule_review` subcluster. The other 38 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 24 mismatches were evaluable and 0 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Mismatch primary-class distribution: + +- `corpus_suspect`: 4 +- `pending_rule_review`: 20 + +Representative `exact` samples: + +- `sentence_01.json` #1754: 이 부사장은 인공지능(AI) 챗봇인 챗GPT가 1990년대에 등장한 PC, 2000년대의 인터넷, 2010년대에 출시된 스마트폰 못지 않게 반도체 시장에도 큰 파급력을 가질 것이라고 내다봤다. + - expected: `⠕⠀⠘⠍⠇⠨⠶⠵⠀⠟⠈⠿⠨⠕⠉⠪⠶⠦⠄⠴⠠⠠⠁⠊` + - actual: `⠕⠀⠘⠍⠇⠨⠶⠵⠀⠟⠈⠿⠨⠕⠉⠪⠶⠦⠄⠴⠠⠠⠁⠊` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #239: 세계 금연의 날(World No Tabacco Day, 2023년 5월 31일)은 세계보건기구(WHO)가 담배가 전 세계적으로 심각한 문제임을 인식시키고 ‘담배 연기 없는 사회’를 만들기 위하여 1987년 제정한 기념일이다. + - expected: `⠠⠝⠈⠌⠀⠈⠪⠢⠡⠺⠀⠉⠂⠦⠄⠴⠠⠸⠺⠀⠠⠝⠕⠀` + - actual: `⠠⠝⠈⠌⠀⠈⠪⠢⠡⠺⠀⠉⠂⠦⠄⠴⠠⠸⠺⠀⠠⠝⠕⠀` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #2147: 용량은 512GB, 1테라바이트(TB), 두 가지 종류로 출시됐다. USB 연결 케이블 2종(C-to-C, C-to-A)과 함께 전용 범퍼케이스가 제공된다. + - expected: `⠬⠶⠐⠜⠶⠵⠀⠼⠑⠁⠃⠴⠠⠠⠛⠃⠐⠀⠼⠁⠀⠓⠝⠐` + - actual: `⠬⠶⠐⠜⠶⠵⠀⠼⠑⠁⠃⠴⠠⠠⠛⠃⠐⠀⠼⠁⠀⠓⠝⠐` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #4209: 사티아 나델라 MS 최고경영자(CEO)는 이날 “나의 계정에서 코파일럿과 채팅하게 돼 기쁘다”며 “AI 비서와 일하는 것은 1980년대의 PC, 1990년대의 인터넷, 21세기 모바일의 부상만큼이나 주목할 만하다”라고 자평했다. + - expected: `⠇⠓⠕⠣⠀⠉⠊⠝⠂⠐⠣⠀⠴⠠⠠⠍⠎⠲⠀⠰⠽⠈⠥⠈` + - actual: `⠇⠓⠕⠣⠀⠉⠊⠝⠂⠐⠣⠀⠴⠠⠠⠍⠎⠲⠀⠰⠽⠈⠥⠈` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_01.json` #322: 삼성전자는 5나노미터(nm, 1nm는 10억 분의 1m) 기반 신규 컨트롤러를 탑재한 PC용 고성능 비휘발성 메모리 익스프레스(NVMe) SSD ‘PM9C1a’를 양산한다고 12일 밝혔다. + - expected: `⠠⠪⠙⠪⠐⠝⠠⠪⠴⠐⠣⠠⠠⠝⠧⠍⠠⠄⠑⠐⠜⠀⠠⠠` + - actual: `⠠⠪⠙⠪⠐⠝⠠⠪⠦⠄⠴⠠⠝⠠⠧⠠⠍⠑⠠⠴⠀⠴⠠⠠` + - first differing cell (zero-based): 124 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #431: 173cm, 68kg의 다소 왜소한 체격이지만 스크럼 하프(SH) 포지션을 맡아 포워드와 백스 사이에서 날카로운 볼 연결과 공격방향을 결정하는 타고난 판단력이 강점이다. + - expected: `⠚⠙⠪⠦⠄⠴⠠⠠⠎⠓⠠⠴⠀⠙⠥⠨⠕⠠⠡⠮⠀⠑⠦⠣` + - actual: `⠚⠙⠪⠦⠄⠴⠠⠠⠩⠠⠴⠀⠙⠥⠨⠕⠠⠡⠮⠀⠑⠦⠣⠀` + - first differing cell (zero-based): 55 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #8811: HD현대중공업(A)과 LS일렉트릭(AA-) 등도 언더발행에 성공했다. HD현대중공업은 회사채 수요예측에서 1년6개월물 -29bp, 2년물 -20bp로 물량을 채웠다. LS일렉트릭의 3년물은 -6bp에 낙찰됐다. + - expected: `⠁⠦⠄⠴⠠⠠⠁⠁⠐⠤⠠⠴⠀⠊⠪⠶⠊⠥⠀⠾⠊⠎⠘⠂` + - actual: `⠁⠦⠄⠴⠠⠠⠁⠁⠲⠤⠠⠴⠀⠊⠪⠶⠊⠥⠀⠾⠊⠎⠘⠂` + - first differing cell (zero-based): 50 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #7444: 일반부는 하프(Half), 10km, 5km로 나눠 열리며 양양 웰컴 센터를 출발해 남대천 일출로와 낙산대교를 거쳐 동호해변을 반환점으로 한다. + - expected: `⠘⠍⠉⠵⠀⠚⠙⠪⠴⠐⠣⠠⠓⠁⠇⠋⠐⠜⠂⠀⠼⠁⠚⠅` + - actual: `⠘⠍⠉⠵⠀⠚⠙⠪⠦⠄⠴⠠⠓⠁⠇⠋⠠⠴⠐⠀⠼⠁⠚⠴` + - first differing cell (zero-based): 12 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` + +### `attached_ascii_roman_segments_joined_by_ampersand` + +Of the 802 candidates, 103 are the actual `pending_rule_review` subcluster. The other 699 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 115 mismatches were evaluable and 0 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Mismatch primary-class distribution: + +- `corpus_suspect`: 12 +- `pending_rule_review`: 103 + +Representative `exact` samples: + +- `sentence_01.json` #207: 우리금융지주가 약 7조원의 자본 여력을 등에 업고 인수·합병(M&A) 시장에 출격했다. 벤처캐피탈(VC) 인수를 시작으로 증권사와 보험사까지 인수한다는 계획이다. + - expected: `⠍⠐⠕⠈⠪⠢⠩⠶⠨⠕⠨⠍⠫⠀⠜⠁⠀⠼⠛⠨⠥⠏⠒⠺` + - actual: `⠍⠐⠕⠈⠪⠢⠩⠶⠨⠕⠨⠍⠫⠀⠜⠁⠀⠼⠛⠨⠥⠏⠒⠺` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #146: KT&G(사장 백복인)가 산업정책연구원(IPS)이 주관하고 윤경 ESG 포럼이 주최하는 ‘윤경 ESG 포럼 CEO 서약식’에 참여해 윤리경영 의지를 다졌다. + - expected: `⠴⠠⠠⠅⠞⠈⠯⠠⠛⠦⠄⠇⠨⠶⠀⠘⠗⠁⠘⠭⠟⠠⠴⠫` + - actual: `⠴⠠⠠⠅⠞⠈⠯⠠⠛⠦⠄⠇⠨⠶⠀⠘⠗⠁⠘⠭⠟⠠⠴⠫` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #191: 한국투자신탁운용의 에이스(ACE) 글로벌브랜드TOP10블룸버그 ETF는 연초 이후 22.6% 상승했다. 명품 기업 루이비통모에헤네시(LVMH), 대표 가치주로 꼽히는 프록터앤드갬블(P&G) 편입 비중이 높은 것이 특징이다. + - expected: `⠚⠒⠈⠍⠁⠓⠍⠨⠠⠟⠓⠁⠛⠬⠶⠺⠀⠝⠕⠠⠪⠦⠄⠴` + - actual: `⠚⠒⠈⠍⠁⠓⠍⠨⠠⠟⠓⠁⠛⠬⠶⠺⠀⠝⠕⠠⠪⠦⠄⠴` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #159: 또 액화천연가스(LNG) 분야 협력을 확대하는 한편 수소환원제철 관련 공동 연구·개발(R&D)을 추진해 유럽연합(EU)의 탄소국경조정세(CBAM)와 같은 글로벌 규제와 자원 무기화에 공동 대응한다는 전략이다. + - expected: `⠠⠊⠥⠀⠗⠁⠚⠧⠰⠾⠡⠫⠠⠪⠦⠄⠴⠠⠠⠇⠝⠛⠠⠴` + - actual: `⠠⠊⠥⠀⠗⠁⠚⠧⠰⠾⠡⠫⠠⠪⠦⠄⠴⠠⠠⠇⠝⠛⠠⠴` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_01.json` #1082: AMAT는 경기도에 반도체 장비 연구개발(R&D) 센터를 신설하기 위해 지난해 7월 산업통상자원부, 경기도와 투자의향 양해각서(MOU)를 체결했다. + - expected: `⠠⠎⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠐⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - actual: `⠠⠎⠦⠄⠴⠠⠠⠍⠳⠠⠴⠐⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - first differing cell (zero-based): 130 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #455: 먼저, 세계 최고의 R&D인프라와 인력을 갖춘 장점을 활용하여 국가첨단반도체 기술센터(ASTC)를 유치하고 대전을 반도체 연구·교육·실증 거점으로 조성할 계획이다. + - expected: `⠓⠎⠦⠄⠴⠠⠠⠁⠎⠞⠉⠠⠴⠐⠮⠀⠩⠰⠕⠚⠈⠥⠀⠊` + - actual: `⠓⠎⠦⠄⠴⠠⠠⠁⠌⠉⠠⠴⠐⠮⠀⠩⠰⠕⠚⠈⠥⠀⠊⠗` + - first differing cell (zero-based): 90 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #370: 26일 GS칼텍스는 서울 강남구 GS타워에서 이승훈 GS칼텍스 S&T본부장, 박진기 HMM 총괄부사장 등이 참석한 가운데 친환경 바이오선박유 사업 업무협약(MOU)을 체결했다고 밝혔다. + - expected: `⠜⠁⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈⠥` + - actual: `⠜⠁⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈⠥⠀` + - first differing cell (zero-based): 164 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #19: 이날 행사에는 구 대표 외에 권봉석 (주)LG 최고운영책임자(COO), 박일평 LG사이언스파크대표(사장)를 비롯해 각 계열사 최고기술책임자(CTO), 최고데이터책임자(CDO), 최고인사책임자(CHO) 등이 참석해 국내 이공계 R&D 인재 400여명과 만났다. + - expected: `⠁⠕⠢⠨⠦⠄⠴⠠⠠⠉⠓⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢⠠` + - actual: `⠁⠕⠢⠨⠦⠄⠴⠠⠉⠠⠓⠠⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢` + - first differing cell (zero-based): 188 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `attached_ascii_roman_to_korean_script_boundary` + +Of the 17693 candidates, 1719 are the actual `pending_rule_review` subcluster. The other 15974 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 2487 mismatches were evaluable and 0 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Mismatch primary-class distribution: + +- `corpus_suspect`: 766 +- `pending_rule_review`: 1719 +- `unsupported_character_review`: 2 + +Representative `exact` samples: + +- `sentence_01.json` #1: LG전자는 ‘업(UP)가전’을 무기로 내세우고 있다. 이번 CES에서도 LG 씽큐 앱에서 터치만으로 제품 색상을 바꿀 수 있는 무드업 냉장고를 비롯한 다양한 업가전을 선보일 예정이다. + - expected: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠉⠵⠀⠠⠦⠎⠃⠦⠄⠴⠠⠠⠥⠏⠠` + - actual: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠉⠵⠀⠠⠦⠎⠃⠦⠄⠴⠠⠠⠥⠏⠠` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #4: KDI에 따르면, 비제조업 업황BSI(기업경기실사지수) 전망치는 2월 72에서 3월 74로 개선되고 있다. 중국 관광객 유입에 대한 기대감이 확산된 영향으로 분석된다. + - expected: `⠴⠠⠠⠅⠙⠊⠲⠝⠀⠠⠊⠐⠪⠑⠡⠐⠀⠘⠕⠨⠝⠨⠥⠎` + - actual: `⠴⠠⠠⠅⠙⠊⠲⠝⠀⠠⠊⠐⠪⠑⠡⠐⠀⠘⠕⠨⠝⠨⠥⠎` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #5: 또한 전 세계적인 흐름인 ‘ESG경영’이 지역기업이 도입할 수 있도록 지원하고, 코트라(KOTRA) 등 유관기관과 협력 지방공공기관 및 지역기업의 해외 진출도 지원한다. + - expected: `⠠⠊⠥⠚⠒⠀⠨⠾⠀⠠⠝⠈⠌⠨⠹⠟⠀⠚⠪⠐⠪⠢⠟⠀` + - actual: `⠠⠊⠥⠚⠒⠀⠨⠾⠀⠠⠝⠈⠌⠨⠹⠟⠀⠚⠪⠐⠪⠢⠟⠀` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #2: 주요 지수는 장중 2% 이상 하락했으나 장 막판 스위스중앙은행(SNB)이 나서 CS에 대한 지원 방침을 밝히면서 나스닥지수가 반등하는 등 한숨을 돌렸다. + - expected: `⠨⠍⠬⠀⠨⠕⠠⠍⠉⠵⠀⠨⠶⠨⠍⠶⠀⠼⠃⠴⠏⠀⠕⠇` + - actual: `⠨⠍⠬⠀⠨⠕⠠⠍⠉⠵⠀⠨⠶⠨⠍⠶⠀⠼⠃⠴⠏⠀⠕⠇` + - current primary/reason: `exact` / `exact` + +Representative `exact_rule29_terminator` samples: + +- `sentence_01.json` #1: LG전자는 ‘업(UP)가전’을 무기로 내세우고 있다. 이번 CES에서도 LG 씽큐 앱에서 터치만으로 제품 색상을 바꿀 수 있는 무드업 냉장고를 비롯한 다양한 업가전을 선보일 예정이다. + - expected: `⠾⠀⠴⠠⠠⠉⠑⠎⠲⠝⠠⠎⠊⠥⠀⠴⠠⠠⠇⠛⠲⠀⠠⠠` + - actual: `⠾⠀⠴⠠⠠⠉⠑⠎⠲⠝⠠⠎⠊⠥⠀⠴⠠⠠⠇⠛⠲⠀⠠⠠` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #41: 올해로 창립 10주년을 맞이한 IWPG는 유엔 경제사회이사회(UN ECOSOC)와 글로벌소통국(DGC)에 등록된 국제 NGO로서, 전쟁 반대와 실질적인 평화의 바람을 일으키고 있다. + - expected: `⠀⠴⠠⠠⠊⠺⠏⠛⠲⠉⠵⠀⠩⠝⠒⠀⠈⠻⠨⠝⠇⠚⠽⠕` + - actual: `⠀⠴⠠⠠⠊⠺⠏⠛⠲⠉⠵⠀⠩⠝⠒⠀⠈⠻⠨⠝⠇⠚⠽⠕` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #5: 또한 전 세계적인 흐름인 ‘ESG경영’이 지역기업이 도입할 수 있도록 지원하고, 코트라(KOTRA) 등 유관기관과 협력 지방공공기관 및 지역기업의 해외 진출도 지원한다. + - expected: `⠠⠦⠴⠠⠠⠑⠎⠛⠲⠈⠻⠻⠴⠄⠕⠀⠨⠕⠱⠁⠈⠕⠎⠃` + - actual: `⠠⠦⠴⠠⠠⠑⠎⠛⠲⠈⠻⠻⠴⠄⠕⠀⠨⠕⠱⠁⠈⠕⠎⠃` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #2: 주요 지수는 장중 2% 이상 하락했으나 장 막판 스위스중앙은행(SNB)이 나서 CS에 대한 지원 방침을 밝히면서 나스닥지수가 반등하는 등 한숨을 돌렸다. + - expected: `⠠⠎⠀⠴⠠⠠⠉⠎⠲⠝⠀⠊⠗⠚⠒⠀⠨⠕⠏⠒⠀⠘⠶⠰` + - actual: `⠠⠎⠀⠴⠠⠠⠉⠎⠲⠝⠀⠊⠗⠚⠒⠀⠨⠕⠏⠒⠀⠘⠶⠰` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_01.json` #14: 새롭게 선보인 DRX 브랜딩과 팀 컬러는 2023년 1월 1일부터 DRX의 온오프라인 콘텐츠, 엠디(MD), SNS, 유튜브 등 모든 제작물에 적용될 예정이다. + - expected: `⠰⠪⠐⠀⠝⠢⠊⠕⠴⠐⠣⠠⠠⠍⠙⠐⠜⠂⠀⠠⠠⠎⠝⠎` + - actual: `⠰⠪⠐⠀⠝⠢⠊⠕⠦⠄⠴⠠⠠⠍⠙⠠⠴⠐⠀⠴⠠⠠⠎⠝` + - first differing cell (zero-based): 97 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #10: 설동호 대전교육감은 “학교 교육과정 및 여건에 맞는 인공지능(AI)교육 선도학교 운영을 통해 AI·SW교육 기반 구축과 학생들과 교원들의 미래 디지털 역량이 함양될 것이다.”라고 말했다. + - expected: `⠚⠗⠀⠴⠠⠠⠁⠊⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕⠘` + - actual: `⠚⠗⠀⠴⠠⠠⠁⠊⠲⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕` + - first differing cell (zero-based): 93 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #6: 지난해 9월 HMM과 파나시아가 선박용 탄소포집 시스템 공동연구에 대한 업무협약(MOU)을 체결한 뒤 7개월 만에 내놓은 연구성과다. + - expected: `⠜⠁⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠒⠀⠊⠍⠗` + - actual: `⠜⠁⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠒⠀⠊⠍⠗⠀` + - first differing cell (zero-based): 83 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #19: 이날 행사에는 구 대표 외에 권봉석 (주)LG 최고운영책임자(COO), 박일평 LG사이언스파크대표(사장)를 비롯해 각 계열사 최고기술책임자(CTO), 최고데이터책임자(CDO), 최고인사책임자(CHO) 등이 참석해 국내 이공계 R&D 인재 400여명과 만났다. + - expected: `⠁⠕⠢⠨⠦⠄⠴⠠⠠⠉⠓⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢⠠` + - actual: `⠁⠕⠢⠨⠦⠄⠴⠠⠉⠠⠓⠠⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢` + - first differing cell (zero-based): 188 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch_rule29_terminator` samples: + +- `sentence_01.json` #14: 새롭게 선보인 DRX 브랜딩과 팀 컬러는 2023년 1월 1일부터 DRX의 온오프라인 콘텐츠, 엠디(MD), SNS, 유튜브 등 모든 제작물에 적용될 예정이다. + - expected: `⠎⠀⠴⠠⠠⠙⠗⠭⠲⠺⠀⠷⠥⠙⠪⠐⠣⠟⠀⠋⠷⠓⠝⠒` + - actual: `⠎⠀⠴⠠⠠⠙⠗⠭⠲⠺⠀⠷⠥⠙⠪⠐⠣⠟⠀⠋⠷⠓⠝⠒` + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #10: 설동호 대전교육감은 “학교 교육과정 및 여건에 맞는 인공지능(AI)교육 선도학교 운영을 통해 AI·SW교육 기반 구축과 학생들과 교원들의 미래 디지털 역량이 함양될 것이다.”라고 말했다. + - expected: `⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕⠘⠒⠀⠈⠍⠰⠍⠁⠈` + - actual: `⠲⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕⠘⠒⠀⠈⠍⠰⠍⠁` + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #6: 지난해 9월 HMM과 파나시아가 선박용 탄소포집 시스템 공동연구에 대한 업무협약(MOU)을 체결한 뒤 7개월 만에 내놓은 연구성과다. + - expected: `⠂⠀⠴⠠⠠⠓⠍⠍⠲⠈⠧⠀⠙⠉⠠⠕⠣⠫⠀⠠⠾⠘⠁⠬` + - actual: `⠂⠀⠴⠠⠠⠓⠍⠍⠲⠈⠧⠀⠙⠉⠠⠕⠣⠫⠀⠠⠾⠘⠁⠬` + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #19: 이날 행사에는 구 대표 외에 권봉석 (주)LG 최고운영책임자(COO), 박일평 LG사이언스파크대표(사장)를 비롯해 각 계열사 최고기술책임자(CTO), 최고데이터책임자(CDO), 최고인사책임자(CHO) 등이 참석해 국내 이공계 R&D 인재 400여명과 만났다. + - expected: `⠙⠻⠀⠴⠠⠠⠇⠛⠲⠇⠕⠾⠠⠪⠙⠋⠪⠊⠗⠙⠬⠦⠄⠇` + - actual: `⠙⠻⠀⠴⠠⠠⠇⠛⠲⠇⠕⠾⠠⠪⠙⠋⠪⠊⠗⠙⠬⠦⠄⠇` + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `attached_korean_auxiliary_itda_spacing` + +Of the 95 candidates, 8 are the actual `pending_rule_review` subcluster. The other 87 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +Mismatch primary-class distribution: + +- `corpus_suspect`: 3 +- `pending_rule_review`: 8 + +Representative `exact` samples: + +- `sentence_01.json` #1416: ‘키우GO’ 서비스는 투자목표와 투자기간, 투자금액, 투자성향 등을 종합적으로 분석하여 현재 금융시장 상황에 적합한 자산배분 포트폴리오를 제공하는 투자일임(Wrap)서비스로, 21년 5월 서비스 출시 후 꾸준한 성장을 하고있다. + - expected: `⠠⠦⠋⠕⠍⠴⠠⠠⠛⠕⠴⠄⠀⠠⠎⠘⠕⠠⠪⠉⠵⠀⠓⠍` + - actual: `⠠⠦⠋⠕⠍⠴⠠⠠⠛⠕⠴⠄⠀⠠⠎⠘⠕⠠⠪⠉⠵⠀⠓⠍` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #242: 예산군의 주차장 관리 업무를 담당하는 임종생 씨(70)는 매일 아침 군청사 출입문에서 밝은 미소로 출근하는 직원과 민원인에게 매일 아침마다 인사를 건네고 있어 호평을 받고있다. + - expected: `⠌⠇⠒⠈⠛⠺⠀⠨⠍⠰⠣⠨⠶⠀⠈⠧⠒⠐⠕⠀⠎⠃⠑⠍` + - actual: `⠌⠇⠒⠈⠛⠺⠀⠨⠍⠰⠣⠨⠶⠀⠈⠧⠒⠐⠕⠀⠎⠃⠑⠍` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #74: 아울러 제주도개발공사는 국내 생수업계에서는 처음으로 재활용 페트(CR-PET)를 적용한 화학적 재활용 페트 ‘제주삼다수 리본(RE:Born)’을 개발하는 등 소재혁신을 통한 친환경 라인업도 확대하고있다. + - expected: `⠣⠯⠐⠎⠀⠨⠝⠨⠍⠊⠥⠈⠗⠘⠂⠈⠿⠇⠉⠵⠀⠈⠍⠁` + - actual: `⠣⠯⠐⠎⠀⠨⠝⠨⠍⠊⠥⠈⠗⠘⠂⠈⠿⠇⠉⠵⠀⠈⠍⠁` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #204: 이번 법안엔 소형모듈원전(SMR)을 분산에너지로 인정하는 내용도 담겨있다. 여야가 인정 여부를 두고 이견을 보였지만 최근 합의했다. + - expected: `⠕⠘⠾⠀⠘⠎⠃⠣⠒⠝⠒⠀⠠⠥⠚⠻⠑⠥⠊⠩⠂⠏⠒⠨` + - actual: `⠕⠘⠾⠀⠘⠎⠃⠣⠒⠝⠒⠀⠠⠥⠚⠻⠑⠥⠊⠩⠂⠏⠒⠨` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_01.json` #13832: 키움은 최근 4연패 부진에 빠져있다. 팀 득점(19)은 최하위를 기록중이고 팀 타율(.228) 9위, 홈런(1) 9위, OPS(.581) 10위 등 각종 타격지표가 모두 하위권에 머무르고 있는 것이 고민이다. + - expected: `⠢⠀⠓⠣⠩⠂⠦⠄⠼⠲⠃⠃⠓⠠⠴⠀⠼⠊⠍⠗⠐⠀⠚⠥` + - actual: `⠢⠀⠓⠣⠩⠂⠦⠄⠲⠼⠃⠃⠓⠠⠴⠀⠼⠊⠍⠗⠐⠀⠚⠥` + - first differing cell (zero-based): 81 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #2144: 황새는 세계자연보전연맹 적색자료목록에서 위기(EN)종으로 분류된 국제적 보호종으로 전 세계에서 2천499개체 정도 생존하고 있는 것으로 알려져있다. + - expected: `⠗⠈⠕⠦⠄⠴⠠⠠⠑⠝⠠⠴⠨⠿⠪⠐⠥⠀⠘⠛⠐⠩⠊⠽` + - actual: `⠗⠈⠕⠦⠄⠴⠠⠠⠢⠠⠴⠨⠿⠪⠐⠥⠀⠘⠛⠐⠩⠊⠽⠒` + - first differing cell (zero-based): 49 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #15510: 타이틀곡 ‘세월아’는 수많은 드라마의 OST를 프로듀싱한 작곡가 필승불패W, 지민(JAK), 건치가 의기투합하여 만든 곡으로 인생에 대한 공감 가는 가사와 애절한 멜로디 그리고 세련되면서 신나는 사운드가 담겨있다. + - expected: `⠑⠣⠺⠀⠴⠠⠠⠕⠎⠞⠲⠐⠮⠀⠙⠪⠐⠥⠊⠩⠠⠕⠶⠚` + - actual: `⠑⠣⠺⠀⠴⠠⠠⠕⠌⠲⠐⠮⠀⠙⠪⠐⠥⠊⠩⠠⠕⠶⠚⠒` + - first differing cell (zero-based): 39 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `attached_korean_to_roman_hyphen_boundary` + +Of the 105 candidates, 17 are the actual `pending_rule_review` subcluster. The other 88 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +Mismatch primary-class distribution: + +- `corpus_suspect`: 4 +- `pending_rule_review`: 17 + +Representative `exact` samples: + +- `sentence_01.json` #491: 자원 선순환 달성을 위한 폐플라스틱 재활용도 추진한다. 폐폴리스티렌을 열분해한 재활용스티렌(RSM) 제조 사업과 RSM을 고기능성 합성고무 SSBR에 적용시킨 에코-SSBR을 오는 2025년까지 상용화할 계획이다. + - expected: `⠨⠣⠏⠒⠀⠠⠾⠠⠛⠚⠧⠒⠀⠊⠂⠠⠻⠮⠀⠍⠗⠚⠒⠀` + - actual: `⠨⠣⠏⠒⠀⠠⠾⠠⠛⠚⠧⠒⠀⠊⠂⠠⠻⠮⠀⠍⠗⠚⠒⠀` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #1147: 올해 상반기 해외사절단을 이끌고 있는 김두겸 울산시장은 13일(현지시각) 아랍에미리트(UAE) 아부다비 중심에 위치한 국영석유회사인 애드낙(ADNOC) 본사에서 ‘울산시-ADNOC, 수소·암모니아산업 공동협력회의’를 개최했다. + - expected: `⠥⠂⠚⠗⠀⠇⠶⠘⠒⠈⠕⠀⠚⠗⠽⠇⠨⠞⠊⠒⠮⠀⠕⠠` + - actual: `⠥⠂⠚⠗⠀⠇⠶⠘⠒⠈⠕⠀⠚⠗⠽⠇⠨⠞⠊⠒⠮⠀⠕⠠` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #477: 미국의 전략핵잠수함이 운용하는 트라이던트-Ⅱ잠수함발사탄도미사일(SLBM)은 사거리가 1만2000㎞인 전략 핵무기다. 사실상 태평양 어디에서도 북한에 핵 타격을 가할 수 있는 역량을 갖춘 셈이다. + - expected: `⠑⠕⠈⠍⠁⠺⠀⠨⠾⠐⠜⠁⠚⠗⠁⠨⠢⠠⠍⠚⠢⠕⠀⠛` + - actual: `⠑⠕⠈⠍⠁⠺⠀⠨⠾⠐⠜⠁⠚⠗⠁⠨⠢⠠⠍⠚⠢⠕⠀⠛` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #976: 하쿠토-R 미션1에는 달 표면을 굴러다닐 로봇이 실려 있다. 일본 우주항공연구개발기구(JAXA)와 장난감 기업 토미가 함께 만든 지름 8㎝짜리 공 모양의 초소형 로봇 ‘소라큐’다. + - expected: `⠚⠋⠍⠓⠥⠤⠴⠠⠗⠲⠀⠑⠕⠠⠡⠼⠁⠝⠉⠵⠀⠊⠂⠀` + - actual: `⠚⠋⠍⠓⠥⠤⠴⠠⠗⠲⠀⠑⠕⠠⠡⠼⠁⠝⠉⠵⠀⠊⠂⠀` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_01.json` #441: 한국무역협회는 16일 아부다비에서 아랍에미리트(UAE) 연방상공회의소와 ‘한-UAE 경제협력위원회’ 설립을 위한 업무협약(MOU)을 체결했다고 17일 밝혔다. 이날 서명에는 구자열 한국무역협회 회장과 압둘라 마즈로이 UAE연방상의 회장이 참석했다. + - expected: `⠜⠁⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈⠥` + - actual: `⠜⠁⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈⠥⠀` + - first differing cell (zero-based): 128 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #8005: 대한무역투자진흥공사(KOTRA·코트라)는 윤석열 대통령의 아랍에미리트(UAE) 국빈 방문을 계기로 16일(현지시간) UAE 수도 아부다비에서 ‘한-UAE 비즈니스 상담회’를 개최했다고 밝혔다. + - expected: `⠴⠠⠠⠅⠕⠞⠗⠁⠐⠆⠋⠥⠓⠪⠐⠣⠠⠴⠉⠵⠀⠩⠒⠠` + - actual: `⠴⠠⠠⠅⠕⠞⠗⠁⠲⠐⠆⠋⠥⠓⠪⠐⠣⠠⠴⠉⠵⠀⠩⠒` + - first differing cell (zero-based): 29 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #3173: 매도자는 중견 PEF 운용사 자비스자산운용과 ST리더스프라이빗에쿼티(PE)가 공동으로 설립한 ‘에스티엘자비스2018의1사모투자합자회사’다. 자비스운용-ST리더스PE 컨소시엄은 지난 2018년 소신여객을 인수했는데 5년 만에 MC파트너스에 회사를 넘기게 됐다. + - expected: `⠬⠶⠈⠧⠀⠴⠠⠠⠎⠞⠲⠐⠕⠊⠎⠠⠪⠙⠪⠐⠣⠕⠘⠕` + - actual: `⠬⠶⠈⠧⠀⠴⠠⠠⠌⠲⠐⠕⠊⠎⠠⠪⠙⠪⠐⠣⠕⠘⠕⠄` + - first differing cell (zero-based): 44 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #6423: 한국수력원자력(한수원, 사장 황주호)은 한-UAE 포괄적 전략적 에너지 파트너십에 관한 공동선언과 관련해 UAE원자력공사(ENEC)와 ‘넷제로 가속화 전략적 협력 MOU’를 체결했다. + - expected: `⠈⠿⠇⠦⠄⠴⠠⠠⠑⠝⠑⠉⠠⠴⠧⠀⠠⠦⠉⠝⠄⠨⠝⠐` + - actual: `⠈⠿⠇⠦⠄⠴⠠⠠⠢⠑⠉⠠⠴⠧⠀⠠⠦⠉⠝⠄⠨⠝⠐⠥` + - first differing cell (zero-based): 129 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `attached_plus_followed_by_parenthesized_korean_gloss` + +Of the 16 candidates, 12 are the actual `pending_rule_review` subcluster. The other 4 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 12 mismatches were evaluable and 12 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2826 ⠦ -> U+2800 ⠀`: 12 + +Mismatch primary-class distribution: + +- `pending_rule_review`: 12 + +Representative `exact` samples: + +- `sentence_01.json` #12551: 최근 화제가 되고 있는 넷플릭스 ‘더 글로리’에서 문동은(송혜교)에게 “넝담”이라며 시비를 거는 추 선생 역으로 열연을 펼쳤다. 디즈니+(플러스) 오리지널 시리즈 ‘카지노’에도 출연했다. + - expected: `⠰⠽⠈⠵⠀⠚⠧⠨⠝⠫⠀⠊⠽⠈⠥⠀⠕⠌⠉⠵⠀⠉⠝⠄` + - actual: `⠰⠽⠈⠵⠀⠚⠧⠨⠝⠫⠀⠊⠽⠈⠥⠀⠕⠌⠉⠵⠀⠉⠝⠄` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #4383: 운영과정은 ‘도전’ 프로그램(5주 과정)과 ‘도전+(플러스)’ 프로그램(5개월 과정)으로 진행되며 프로그램 이수 시 50만 원에서 최대 300만 원까지(월별 참여수당 지급) 참여수당과 인센티브를 지원한다. + - expected: `⠛⠻⠈⠧⠨⠻⠵⠀⠠⠦⠊⠥⠨⠾⠴⠄⠀⠙⠪⠐⠥⠈⠪⠐` + - actual: `⠛⠻⠈⠧⠨⠻⠵⠀⠠⠦⠊⠥⠨⠾⠴⠄⠀⠙⠪⠐⠥⠈⠪⠐` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #4388: 청룡시리즈어워즈는 2022년 국내 최초로 오리지널 스트리밍 시리즈를 대상으로 열린 시상식이다. 넷플릭스부터 디즈니+(플러스), 애플TV+(플러스), 왓챠, 웨이브, 카카오TV, 쿠팡플레이, 티빙이 제작하거나 투자한 국내 드라마와 예능·교양을 대상으로 한다. + - expected: `⠰⠻⠐⠬⠶⠠⠕⠐⠕⠨⠪⠎⠏⠨⠪⠉⠵⠀⠼⠃⠚⠃⠃⠀` + - actual: `⠰⠻⠐⠬⠶⠠⠕⠐⠕⠨⠪⠎⠏⠨⠪⠉⠵⠀⠼⠃⠚⠃⠃⠀` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_01.json` #845: 부산광역시는 3일 오전 부산광역시청에서 초록우산어린이재단 부산본부, 월드비전 부산사업본부, 굿네이버스 영남본부와 ‘자립+(더하기) 동행 프로젝트’ 업무협약을 체결했다고 밝혔다. + - expected: `⠠⠦⠨⠐⠕⠃⠀⠢⠦⠄⠊⠎⠚⠈⠕⠠⠴⠀⠊⠿⠚⠗⠶⠀` + - actual: `⠠⠦⠨⠐⠕⠃⠀⠢⠀⠦⠄⠊⠎⠚⠈⠕⠠⠴⠀⠊⠿⠚⠗⠶` + - first differing cell (zero-based): 116 + - current primary/reason: `pending_rule_review` / `number_rule_review` +- `sentence_02.json` #168: 5주 과정인 ‘도전’ 프로그램과 5개월 과정인 ‘도전+(플러스)’ 프로그램으로 나뉘어 진행되며, 프로그램 이수 시 1개월에 50만 원씩 최대 300만 원까지 참여 수당과 인센티브를 지원한다. + - expected: `⠠⠦⠊⠥⠨⠾⠀⠢⠦⠄⠙⠮⠐⠎⠠⠪⠠⠴⠴⠄⠀⠙⠪⠐` + - actual: `⠠⠦⠊⠥⠨⠾⠀⠢⠀⠦⠄⠙⠮⠐⠎⠠⠪⠠⠴⠴⠄⠀⠙⠪` + - first differing cell (zero-based): 53 + - current primary/reason: `pending_rule_review` / `number_rule_review` +- `sentence_03.json` #12962: 1·2열 온도·모드·풍량을 각각 독립적으로 제어할 수있는 3존+(플러스)공조, 디지털키 2, 실내 지문 인증 시스템, 콘솔 암레스트 수납함 자외선 살균 기능, 콘솔 암레스트 열선 등을 탑재했다. + - expected: `⠵⠀⠼⠉⠨⠷⠀⠢⠦⠄⠙⠮⠐⠎⠠⠪⠠⠴⠈⠿⠨⠥⠐⠀` + - actual: `⠵⠀⠼⠉⠨⠷⠀⠢⠀⠦⠄⠙⠮⠐⠎⠠⠪⠠⠴⠈⠿⠨⠥⠐` + - first differing cell (zero-based): 62 + - current primary/reason: `pending_rule_review` / `number_rule_review` +- `sentence_04.json` #6494: LG유플러스(대표 황현식)는 중소 알뜰폰(MVNO) 사업자의 요금제를 판매하는 오프라인 컨설팅 전문매장 ‘알뜰폰+(플러스)’를 전국으로 확대한다고 31일 밝혔다. + - expected: `⠂⠠⠊⠮⠙⠷⠀⠢⠦⠄⠙⠮⠐⠎⠠⠪⠠⠴⠴⠄⠐⠮⠀⠨` + - actual: `⠂⠠⠊⠮⠙⠷⠀⠢⠀⠦⠄⠙⠮⠐⠎⠠⠪⠠⠴⠴⠄⠐⠮⠀` + - first differing cell (zero-based): 117 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #845: 부산광역시는 3일 오전 부산광역시청에서 초록우산어린이재단 부산본부, 월드비전 부산사업본부, 굿네이버스 영남본부와 ‘자립+(더하기) 동행 프로젝트’ 업무협약을 체결했다고 밝혔다. + - expected: `⠠⠦⠨⠐⠕⠃⠀⠢⠦⠄⠊⠎⠚⠈⠕⠠⠴⠀⠊⠿⠚⠗⠶⠀` + - actual: `⠠⠦⠨⠐⠕⠃⠀⠢⠀⠦⠄⠊⠎⠚⠈⠕⠠⠴⠀⠊⠿⠚⠗⠶` + - first differing cell (zero-based): 116 + - current primary/reason: `pending_rule_review` / `number_rule_review` +- `sentence_02.json` #168: 5주 과정인 ‘도전’ 프로그램과 5개월 과정인 ‘도전+(플러스)’ 프로그램으로 나뉘어 진행되며, 프로그램 이수 시 1개월에 50만 원씩 최대 300만 원까지 참여 수당과 인센티브를 지원한다. + - expected: `⠠⠦⠊⠥⠨⠾⠀⠢⠦⠄⠙⠮⠐⠎⠠⠪⠠⠴⠴⠄⠀⠙⠪⠐` + - actual: `⠠⠦⠊⠥⠨⠾⠀⠢⠀⠦⠄⠙⠮⠐⠎⠠⠪⠠⠴⠴⠄⠀⠙⠪` + - first differing cell (zero-based): 53 + - current primary/reason: `pending_rule_review` / `number_rule_review` +- `sentence_03.json` #12962: 1·2열 온도·모드·풍량을 각각 독립적으로 제어할 수있는 3존+(플러스)공조, 디지털키 2, 실내 지문 인증 시스템, 콘솔 암레스트 수납함 자외선 살균 기능, 콘솔 암레스트 열선 등을 탑재했다. + - expected: `⠵⠀⠼⠉⠨⠷⠀⠢⠦⠄⠙⠮⠐⠎⠠⠪⠠⠴⠈⠿⠨⠥⠐⠀` + - actual: `⠵⠀⠼⠉⠨⠷⠀⠢⠀⠦⠄⠙⠮⠐⠎⠠⠪⠠⠴⠈⠿⠨⠥⠐` + - first differing cell (zero-based): 62 + - current primary/reason: `pending_rule_review` / `number_rule_review` +- `sentence_04.json` #6494: LG유플러스(대표 황현식)는 중소 알뜰폰(MVNO) 사업자의 요금제를 판매하는 오프라인 컨설팅 전문매장 ‘알뜰폰+(플러스)’를 전국으로 확대한다고 31일 밝혔다. + - expected: `⠂⠠⠊⠮⠙⠷⠀⠢⠦⠄⠙⠮⠐⠎⠠⠪⠠⠴⠴⠄⠐⠮⠀⠨` + - actual: `⠂⠠⠊⠮⠙⠷⠀⠢⠀⠦⠄⠙⠮⠐⠎⠠⠪⠠⠴⠴⠄⠐⠮⠀` + - first differing cell (zero-based): 117 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `capitals_word_mode_previously_spanning_nonletter_scope` + +Of the 1733 candidates, 416 are the actual `pending_rule_review` subcluster. The other 1317 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +Mismatch primary-class distribution: + +- `corpus_suspect`: 47 +- `pending_rule_review`: 416 + +Representative `exact` samples: + +- `sentence_01.json` #17: 지난달 30일(현지시각) 뉴욕증권거래소(NYSE)에서 다우존스30산업평균지수는 전 거래일보다 73.55포인트(0.22%) 하락한 3만3147.25로 마감했다. 대형주 중심의 S&P500지수는 9.78포인트(0.25%) 하락한 3839.50으로, 기술주 중심의 나스닥지수는 11.61포인트(0.11%) 하락한 1만0466.4 + - expected: `⠨⠕⠉⠒⠊⠂⠀⠼⠉⠚⠕⠂⠦⠄⠚⠡⠨⠕⠠⠕⠫⠁⠠⠴` + - actual: `⠨⠕⠉⠒⠊⠂⠀⠼⠉⠚⠕⠂⠦⠄⠚⠡⠨⠕⠠⠕⠫⠁⠠⠴` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #23: 군은 14일 군청 소회의실에서 태안군 박경찬 부군수를 비롯해 충청남도·보령시·당진시·서천군 관계자 등 10여 명이 참석한 가운데 ‘화력발전 지역자원시설세 탄력세율 추진 T/F(태스크포스) 회의’를 개최했다고 밝혔다. + - expected: `⠈⠛⠵⠀⠼⠁⠙⠕⠂⠀⠈⠛⠰⠻⠀⠠⠥⠚⠽⠺⠠⠕⠂⠝` + - actual: `⠈⠛⠵⠀⠼⠁⠙⠕⠂⠀⠈⠛⠰⠻⠀⠠⠥⠚⠽⠺⠠⠕⠂⠝` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #50: 당시 CC(폐쇄)TV에 찍힌 장면을 보면 A씨는 엘리베이터를 기다리는 B씨를 발견하자 몰래 뒤로 다가가 갑자기 피해 여성의 머리를 뒤에서 돌려차기로 가격하는 등 폭행했다. + - expected: `⠊⠶⠠⠕⠀⠴⠠⠠⠉⠉⠦⠄⠙⠌⠠⠧⠗⠠⠴⠴⠠⠠⠞⠧` + - actual: `⠊⠶⠠⠕⠀⠴⠠⠠⠉⠉⠦⠄⠙⠌⠠⠧⠗⠠⠴⠴⠠⠠⠞⠧` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #263: 김 부위원장은 투자은행(IB)의 기업 신용 공여, 합병 제도 등 기업의 M&A와 관련한 다른 제도의 불합리한 규제도 정비하고 기업구조혁신펀드도 추가로 조성하겠다고 밝혔다. + - expected: `⠈⠕⠢⠀⠘⠍⠍⠗⠏⠒⠨⠶⠵⠀⠓⠍⠨⠣⠵⠚⠗⠶⠦⠄` + - actual: `⠈⠕⠢⠀⠘⠍⠍⠗⠏⠒⠨⠶⠵⠀⠓⠍⠨⠣⠵⠚⠗⠶⠦⠄` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_01.json` #90: 조 사장은 “앞으로도 이처럼 ‘더 나은 삶(Better Life)’을 실현하기 위해 최고의(First), 차별화된(Unique), 세상에 없던(New) F·U·N 고객경험을 제공하겠다”고 약속했다. + - expected: `⠈⠥⠺⠦⠄⠴⠠⠋⠌⠠⠴⠐⠀⠰⠣⠘⠳⠚⠧⠊⠽⠒⠦⠄` + - actual: `⠈⠥⠺⠦⠄⠴⠠⠋⠊⠗⠌⠠⠴⠐⠀⠰⠣⠘⠳⠚⠧⠊⠽⠒` + - first differing cell (zero-based): 81 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #10: 설동호 대전교육감은 “학교 교육과정 및 여건에 맞는 인공지능(AI)교육 선도학교 운영을 통해 AI·SW교육 기반 구축과 학생들과 교원들의 미래 디지털 역량이 함양될 것이다.”라고 말했다. + - expected: `⠚⠗⠀⠴⠠⠠⠁⠊⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕⠘` + - actual: `⠚⠗⠀⠴⠠⠠⠁⠊⠲⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕` + - first differing cell (zero-based): 93 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #370: 26일 GS칼텍스는 서울 강남구 GS타워에서 이승훈 GS칼텍스 S&T본부장, 박진기 HMM 총괄부사장 등이 참석한 가운데 친환경 바이오선박유 사업 업무협약(MOU)을 체결했다고 밝혔다. + - expected: `⠜⠁⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈⠥` + - actual: `⠜⠁⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈⠥⠀` + - first differing cell (zero-based): 164 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #19: 이날 행사에는 구 대표 외에 권봉석 (주)LG 최고운영책임자(COO), 박일평 LG사이언스파크대표(사장)를 비롯해 각 계열사 최고기술책임자(CTO), 최고데이터책임자(CDO), 최고인사책임자(CHO) 등이 참석해 국내 이공계 R&D 인재 400여명과 만났다. + - expected: `⠁⠕⠢⠨⠦⠄⠴⠠⠠⠉⠓⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢⠠` + - actual: `⠁⠕⠢⠨⠦⠄⠴⠠⠉⠠⠓⠠⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢` + - first differing cell (zero-based): 188 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `closed_roman_parenthetical_after_non_ascii_letter_boundary` + +Of the 63959 candidates, 5599 are the actual `pending_rule_review` subcluster. The other 58360 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 6784 mismatches were evaluable and 418 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2826 ⠦ -> U+2834 ⠴`: 114 +- `U+2834 ⠴ -> U+2826 ⠦`: 36 +- `U+2826 ⠦ -> U+2800 ⠀`: 29 +- `U+2820 ⠠ -> U+280E ⠎`: 17 +- `U+2810 ⠐ -> U+2834 ⠴`: 14 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 1181 +- `pending_rule_review`: 5599 +- `unsupported_character_review`: 4 + +Representative `exact` samples: + +- `sentence_01.json` #1: LG전자는 ‘업(UP)가전’을 무기로 내세우고 있다. 이번 CES에서도 LG 씽큐 앱에서 터치만으로 제품 색상을 바꿀 수 있는 무드업 냉장고를 비롯한 다양한 업가전을 선보일 예정이다. + - expected: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠉⠵⠀⠠⠦⠎⠃⠦⠄⠴⠠⠠⠥⠏⠠` + - actual: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠉⠵⠀⠠⠦⠎⠃⠦⠄⠴⠠⠠⠥⠏⠠` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #1: 세종시 주최, 세종시탄소중립지원센터 주관으로 열린 이번 포럼의 주요의제는 기업의 이에스지(ESG) 경영과 사회적 책임, 국내 온실가스 배출권거래제 현황과 대응 방안을 마련하기 위해 개최됐다. + - expected: `⠠⠝⠨⠿⠠⠕⠀⠨⠍⠰⠽⠐⠀⠠⠝⠨⠿⠠⠕⠓⠒⠠⠥⠨` + - actual: `⠠⠝⠨⠿⠠⠕⠀⠨⠍⠰⠽⠐⠀⠠⠝⠨⠿⠠⠕⠓⠒⠠⠥⠨` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #2: 연구소는 전남대학교, 순천대학교, 전남도립대학교, 동신대학교, 한국에너지공과대학교, 충남대학교, 부산대학교 등 드론, 빅데이터, 인공지능(AI) 관련 대학 교수와 전남테크노파크, ㈜엘시스 등 전문가로 기획연구팀을 구성한다. + - expected: `⠡⠈⠍⠠⠥⠉⠵⠀⠨⠾⠉⠢⠊⠗⠚⠁⠈⠬⠐⠀⠠⠛⠰⠾` + - actual: `⠡⠈⠍⠠⠥⠉⠵⠀⠨⠾⠉⠢⠊⠗⠚⠁⠈⠬⠐⠀⠠⠛⠰⠾` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #1: 일본 정부는 이날 국가안전보장회의(NSC)를 열어 대응 방침을 논의했다. 이어 북한의 미사일 발사에 대해 “엄중히 항의하고 강하게 비난한다”고 밝혔다. + - expected: `⠕⠂⠘⠷⠀⠨⠻⠘⠍⠉⠵⠀⠕⠉⠂⠀⠈⠍⠁⠫⠣⠒⠨⠾` + - actual: `⠕⠂⠘⠷⠀⠨⠻⠘⠍⠉⠵⠀⠕⠉⠂⠀⠈⠍⠁⠫⠣⠒⠨⠾` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_01.json` #39: 소프트웨어정책연구소(SPRi)는 ‘2023년 SW산업 10대 이슈 전망’을 통해 올해 가장 주요한 이슈로 인공지능 기반 모델 고도화를 1위로 선정했다. + - expected: `⠈⠍⠠⠥⠦⠄⠴⠠⠠⠎⠏⠠⠗⠊⠠⠴⠉⠵⠀⠠⠦⠼⠃⠚` + - actual: `⠈⠍⠠⠥⠦⠄⠴⠠⠎⠠⠏⠠⠗⠊⠠⠴⠉⠵⠀⠠⠦⠼⠃⠚` + - first differing cell (zero-based): 23 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #15: 충남도 농정의 의사결정 과정에 민간의 주도적 참여를 이끌고 현장 의견을 반영하기 위한 민관 농정협의체 ‘충남 쎈(SSEn)농위원회’가 본격 출범했다. + - expected: `⠠⠠⠝⠒⠦⠄⠴⠠⠠⠎⠎⠠⠢⠠⠴⠉⠿⠍⠗⠏⠒⠚⠽⠴` + - actual: `⠠⠠⠝⠒⠦⠄⠴⠠⠎⠠⠎⠠⠑⠝⠠⠴⠉⠿⠍⠗⠏⠒⠚⠽` + - first differing cell (zero-based): 109 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #286: 캘리포니아 레드우드 시티의 재무고문인 로렌스 폰은 “텍사스 ETF의 이름을 주를 상징하는 ‘론스타(Lone Star) ETF’ 또는 ‘리멤버 알라모(Alamo) ETF’로 명명하는 것도 괜찮을 것”이라고 말했다. + - expected: `⠀⠠⠦⠐⠷⠠⠪⠓⠴⠐⠣⠠⠇⠐⠕⠀⠠⠌⠜⠐⠜⠀⠠⠠` + - actual: `⠀⠠⠦⠐⠷⠠⠪⠓⠦⠄⠴⠠⠇⠐⠕⠀⠠⠌⠜⠠⠴⠀⠴⠠` + - first differing cell (zero-based): 91 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #19: 이날 행사에는 구 대표 외에 권봉석 (주)LG 최고운영책임자(COO), 박일평 LG사이언스파크대표(사장)를 비롯해 각 계열사 최고기술책임자(CTO), 최고데이터책임자(CDO), 최고인사책임자(CHO) 등이 참석해 국내 이공계 R&D 인재 400여명과 만났다. + - expected: `⠁⠕⠢⠨⠦⠄⠴⠠⠠⠉⠓⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢⠠` + - actual: `⠁⠕⠢⠨⠦⠄⠴⠠⠉⠠⠓⠠⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢` + - first differing cell (zero-based): 188 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #14: 새롭게 선보인 DRX 브랜딩과 팀 컬러는 2023년 1월 1일부터 DRX의 온오프라인 콘텐츠, 엠디(MD), SNS, 유튜브 등 모든 제작물에 적용될 예정이다. + - expected: `⠰⠪⠐⠀⠝⠢⠊⠕⠴⠐⠣⠠⠠⠍⠙⠐⠜⠂⠀⠠⠠⠎⠝⠎` + - actual: `⠰⠪⠐⠀⠝⠢⠊⠕⠦⠄⠴⠠⠠⠍⠙⠠⠴⠐⠀⠴⠠⠠⠎⠝` + - first differing cell (zero-based): 97 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #10: 설동호 대전교육감은 “학교 교육과정 및 여건에 맞는 인공지능(AI)교육 선도학교 운영을 통해 AI·SW교육 기반 구축과 학생들과 교원들의 미래 디지털 역량이 함양될 것이다.”라고 말했다. + - expected: `⠚⠗⠀⠴⠠⠠⠁⠊⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕⠘` + - actual: `⠚⠗⠀⠴⠠⠠⠁⠊⠲⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕` + - first differing cell (zero-based): 93 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #1: 김 지사는 18일(이하 현지시각) 미국 코네티컷주 댄버리 린데 본사에서 산지브 람바(Sanjiv Lamba) 린데 회장, 성백석 린데코리아 회장, 조일교 아산시 부시장과 투자양해각서(MOU)을 체결했다. + - expected: `⠠⠎⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - actual: `⠠⠎⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - first differing cell (zero-based): 175 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #19: 이날 행사에는 구 대표 외에 권봉석 (주)LG 최고운영책임자(COO), 박일평 LG사이언스파크대표(사장)를 비롯해 각 계열사 최고기술책임자(CTO), 최고데이터책임자(CDO), 최고인사책임자(CHO) 등이 참석해 국내 이공계 R&D 인재 400여명과 만났다. + - expected: `⠁⠕⠢⠨⠦⠄⠴⠠⠠⠉⠓⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢⠠` + - actual: `⠁⠕⠢⠨⠦⠄⠴⠠⠉⠠⠓⠠⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢` + - first differing cell (zero-based): 188 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `compact_numeric_ascii_letter_suffix` + +Of the 2975 candidates, 423 are the actual `pending_rule_review` subcluster. The other 2552 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 483 mismatches were evaluable and 41 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2810 ⠐ -> U+2832 ⠲`: 20 +- `U+2820 ⠠ -> U+2834 ⠴`: 3 +- `U+2800 ⠀ -> U+2832 ⠲`: 2 +- `U+2808 ⠈ -> U+2832 ⠲`: 2 +- `U+2811 ⠑ -> U+283B ⠻`: 2 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 60 +- `pending_rule_review`: 423 + +Representative `exact` samples: + +- `sentence_01.json` #340: 한국 국채(외국환평형기금채 5년물 기준)의 신용부도스와프(CDS) 프리미엄은 12월 월평균 53bp로 나타났다. 지난 10월(61bp) 이후 하락 추세다. + - expected: `⠚⠒⠈⠍⠁⠀⠈⠍⠁⠰⠗⠦⠄⠽⠈⠍⠁⠚⠧⠒⠙⠻⠚⠻` + - actual: `⠚⠒⠈⠍⠁⠀⠈⠍⠁⠰⠗⠦⠄⠽⠈⠍⠁⠚⠧⠒⠙⠻⠚⠻` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #29: 어획량이 감소하면서 지난해 12월 기준 1상자(20kg)당 위판가가 24만 원까지 치솟으면서 자원 증강 필요성이 끊임없이 제기돼 왔다. + - expected: `⠎⠚⠽⠁⠐⠜⠶⠕⠀⠫⠢⠠⠥⠚⠑⠡⠠⠎⠀⠨⠕⠉⠒⠚` + - actual: `⠎⠚⠽⠁⠐⠜⠶⠕⠀⠫⠢⠠⠥⠚⠑⠡⠠⠎⠀⠨⠕⠉⠒⠚` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #59: 이들 중 A씨(31) 등 2명은 지난 3월 25일 김해공항으로 필로폰 968g 등을 팬티 속에 숨겨 입국한 혐의를 받고 있다. + - expected: `⠕⠊⠮⠀⠨⠍⠶⠀⠴⠠⠁⠲⠠⠠⠕⠦⠄⠼⠉⠁⠠⠴⠀⠊` + - actual: `⠕⠊⠮⠀⠨⠍⠶⠀⠴⠠⠁⠲⠠⠠⠕⠦⠄⠼⠉⠁⠠⠴⠀⠊` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #57: 미국지질조사국(USGS)에 따르면 이날 에콰도르 항구도시 과야킬에서 남쪽으로 약 80km 떨어진 지점에서 규모 6.8의 지진이 발생했다. 지진의 깊이는 66㎞다. + - expected: `⠑⠕⠈⠍⠁⠨⠕⠨⠕⠂⠨⠥⠇⠈⠍⠁⠦⠄⠴⠠⠠⠥⠎⠛` + - actual: `⠑⠕⠈⠍⠁⠨⠕⠨⠕⠂⠨⠥⠇⠈⠍⠁⠦⠄⠴⠠⠠⠥⠎⠛` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_01.json` #711: 램(RAM)과 저장용량은 전작과 동일할 것으로 예상된다. 기본 모델과 프로 모델은 8GB(기가바이트) 램에 256GB 내장메모리, 울트라는 12GB 램에 256GB·512GB·1TB 내장메모리가 유력하다. + - expected: `⠃⠑⠋⠴⠠⠠⠛⠃⠐⠆⠼⠑⠁⠃⠴⠠⠠⠛⠃⠐⠆⠼⠁⠴` + - actual: `⠃⠑⠋⠴⠠⠠⠛⠃⠲⠐⠆⠼⠑⠁⠃⠴⠠⠠⠛⠃⠲⠐⠆⠼` + - first differing cell (zero-based): 161 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #445: 박람회에서는 다양한 분야의 디지털 교육 프로그램을 한자리에서 체험할 수 있도록 인공지능(AI) 코스웨어·학습플랫폼, 인공지능(AI) 교과교육, 인공지능(AI) 학습지원, 3D·가상현실(VR)·메타버스 교육, 소프트웨어(SW)·코딩·로봇 교육 등 체험 공간을 운영할 예정이다. + - expected: `⠒⠐⠀⠼⠉⠴⠠⠙⠐⠆⠫⠇⠶⠚⠡⠠⠕⠂⠦⠄⠴⠠⠠⠧` + - actual: `⠒⠐⠀⠼⠉⠴⠠⠙⠲⠐⠆⠫⠇⠶⠚⠡⠠⠕⠂⠦⠄⠴⠠⠠` + - first differing cell (zero-based): 179 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #3023: 이마트24가 판매하는 ‘더빅삼각김밥’은 일반 삼각김밥(100g~110g)에 비해 중량을 약 50% 늘렸다. 밥 한 공기(200g)와 비슷한 양을 1500~2000원에 먹을 수 있다는 게 이마트24의 설명이다. + - expected: `⠦⠄⠼⠁⠚⠚⠴⠛⠈⠔⠼⠁⠁⠚⠰⠛⠠⠴⠝⠀⠘⠕⠚⠗` + - actual: `⠦⠄⠼⠁⠚⠚⠴⠛⠲⠈⠔⠼⠁⠁⠚⠴⠛⠠⠴⠝⠀⠘⠕⠚` + - first differing cell (zero-based): 59 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #2127: 또한, 새로운 표준 파워트레인인 B5 엔진은 가솔린 기반의 마일드 하이브리드 엔진이다. 최고출력 250마력(5700rpm), 최대토크 35.7kg·m(1800~4800rpm)의 성능을 갖췄다. + - expected: `⠼⠉⠑⠲⠛⠴⠅⠛⠐⠆⠴⠍⠐⠣⠼⠁⠓⠚⠚⠈⠔⠼⠙⠓` + - actual: `⠼⠉⠑⠲⠛⠴⠅⠛⠲⠐⠆⠴⠍⠐⠣⠼⠁⠓⠚⠚⠈⠔⠼⠙` + - first differing cell (zero-based): 129 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #89: “인공지능(AI), 6G 등 핵심 기술을 위한 투자도 늘리는 동시에 전기차 충전, 디지털 헬스, 웹OS 기반의 콘텐츠 서비스 등 많은 영역으로 사업 포트폴리오를 확장하고 있습니다.” + - expected: `⠟⠈⠿⠨⠕⠉⠪⠶⠴⠐⠣⠠⠠⠁⠊⠐⠜⠂⠀⠼⠋⠠⠛⠲` + - actual: `⠟⠈⠿⠨⠕⠉⠪⠶⠦⠄⠴⠠⠠⠁⠊⠠⠴⠐⠀⠼⠋⠴⠠⠛` + - first differing cell (zero-based): 9 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #61: 사업 대상은 주차단위구획 50개 이상의 충전기 의무설치 대상 시설인 공동주택과 공중이용시설로 완속충전기(7 ~ 11kw 미만) 약 83기, 콘센트형(3kw) 약 285기를 지원한다. + - expected: `⠨⠾⠈⠕⠦⠄⠼⠛⠈⠔⠼⠁⠁⠴⠅⠺⠲⠀⠑⠕⠑⠒⠠⠴` + - actual: `⠨⠾⠈⠕⠦⠄⠼⠛⠀⠈⠔⠀⠼⠁⠁⠴⠅⠺⠲⠀⠑⠕⠑⠒` + - first differing cell (zero-based): 104 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #341: 미래에셋자산운용은 미국 대표지수에 환헤지형으로 투자하는 ‘TIGER 미국S&P500TR(H) 상장지수펀드(ETF)’와 ‘TIGER 미국나스닥100TR(H) ETF’ 순자산 합계가 1000억원을 돌파했다고 26일 밝혔다. + - expected: `⠈⠍⠁⠉⠠⠪⠊⠁⠼⠁⠚⠚⠴⠠⠠⠞⠗⠐⠣⠠⠓⠐⠜⠀` + - actual: `⠈⠍⠁⠉⠠⠪⠊⠁⠀⠀⠼⠁⠚⠚⠠⠠⠞⠗⠦⠠⠓⠴⠀⠴` + - first differing cell (zero-based): 138 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #115: ‘Busan is Good(부산이라 좋다)’이라는 새로운 도시 표어의 조형물을 공개하고, 3차원(3D)으로 표현한 도시상징 표지(CI) 영상을 상영한다. + - expected: `⠁⠝⠀⠊⠎⠀⠠⠛⠙⠦⠄⠘⠍⠇⠒⠕⠐⠣⠀⠨⠥⠴⠊⠠` + - actual: `⠁⠝⠀⠊⠎⠀⠠⠛⠕⠕⠙⠦⠄⠘⠍⠇⠒⠕⠐⠣⠀⠨⠥⠴` + - first differing cell (zero-based): 15 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `consecutive_ascii_roman_words_whitespace_boundary` + +Of the 4679 candidates, 1251 are the actual `pending_rule_review` subcluster. The other 3428 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 1365 mismatches were evaluable and 5 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2800 ⠀ -> U+2832 ⠲`: 4 +- `U+2815 ⠕ -> U+2800 ⠀`: 1 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 114 +- `pending_rule_review`: 1251 + +Representative `exact` samples: + +- `sentence_01.json` #18: 2023년형 신제품은 매터(Matter)와 HCA(Home Connectivity Alliance) 표준을 지원하는 스마트싱스 허브(SmartThings Hub)를 기반으로 다양한 기기를 자동으로 연결하고, 제어·관리할 수 있는 것이 특징이다. + - expected: `⠼⠃⠚⠃⠉⠀⠉⠡⠚⠻⠀⠠⠟⠨⠝⠙⠍⠢⠵⠀⠑⠗⠓⠎` + - actual: `⠼⠃⠚⠃⠉⠀⠉⠡⠚⠻⠀⠠⠟⠨⠝⠙⠍⠢⠵⠀⠑⠗⠓⠎` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #26: 공주대학교 인재개발실에서 취득한 ISO 21001(교육기관경영시스템)은 국제표준화기구(ISO:International Organization for Standardization)에서 34개국 140여명의 전문가 그룹에 의해 개발돼 2018년에 제정되었다. + - expected: `⠈⠿⠨⠍⠊⠗⠚⠁⠈⠬⠀⠟⠨⠗⠈⠗⠘⠂⠠⠕⠂⠝⠠⠎` + - actual: `⠈⠿⠨⠍⠊⠗⠚⠁⠈⠬⠀⠟⠨⠗⠈⠗⠘⠂⠠⠕⠂⠝⠠⠎` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #66: 올해는 더욱 많은 인원을 수용할 수 있는 ‘Mint Breeze Stage(잔디마당)’의 5월 13일 헤드라이너로 발탁되며 다시 한 번 페스티벌 관객들을 들썩이게 할 전망이다. + - expected: `⠥⠂⠚⠗⠉⠵⠀⠊⠎⠍⠁⠀⠑⠒⠴⠵⠀⠟⠏⠒⠮⠀⠠⠍` + - actual: `⠥⠂⠚⠗⠉⠵⠀⠊⠎⠍⠁⠀⠑⠒⠴⠵⠀⠟⠏⠒⠮⠀⠠⠍` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #149: 이창용 한국은행 총재가 국제결제은행(BIS) 토론회에서 “한국 성인의 16%가 가상통화 거래를 위한 은행 계좌를 갖고 있다”며 “골칫거리 중 하나(one of headache to me)”라고 말했다. + - expected: `⠕⠰⠣⠶⠬⠶⠀⠚⠒⠈⠍⠁⠵⠚⠗⠶⠀⠰⠿⠨⠗⠫⠀⠈` + - actual: `⠕⠰⠣⠶⠬⠶⠀⠚⠒⠈⠍⠁⠵⠚⠗⠶⠀⠰⠿⠨⠗⠫⠀⠈` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_02.json` #13516: 특히 어린이들이 직접 체험할 수 있는 DIY(Do It Yourself) 제품을 선보일 예정이다. 핑크퐁뿐만 아니라 다양한 캐릭터 지식재산권(IP) 콜라보 제품으로 포트폴리오를 확장할 계획이다. + - expected: `⠠⠙⠊⠽⠐⠣⠠⠙⠕⠀⠠⠭⠀⠠⠽⠗⠋⠐⠜⠲⠀⠨⠝⠙` + - actual: `⠠⠙⠊⠽⠐⠣⠠⠙⠀⠠⠭⠀⠠⠽⠗⠋⠐⠜⠲⠀⠨⠝⠙⠍` + - first differing cell (zero-based): 47 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #13446: 방탄소년단, 엔하이픈, 아이즈원, 엑소 백현 등 인기 아이돌 앨범 작업에참여한 danke(당케, lalala studio), 케이지(KZ)뿐만 아니라, 알앤비 싱어송라이터비오(B.O.) 등 쟁쟁한 프로듀서진이 참여해 완성도를 높였다. + - expected: `⠀⠴⠇⠁⠇⠁⠇⠁⠀⠌⠥⠙⠊⠕⠠⠴⠐⠀⠋⠝⠕⠨⠕⠦` + - actual: `⠀⠴⠇⠁⠇⠁⠇⠁⠲⠀⠎⠞⠥⠙⠊⠕⠴⠐⠀⠋⠝⠕⠨⠕` + - first differing cell (zero-based): 93 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #35: 삼성전자는 올해 CES에서 77인치 OLED TV를 처음으로 공개할 예정이다. 이는 지난 8월 ‘국제정보디스플레이학술대회(IMID) 2022’에서 삼성디스플레이가 공개한 77인치 퀀텀닷(QD)-OLED 패널을 탑재한 제품이다. + - expected: `⠋⠏⠒⠓⠎⠢⠊⠄⠴⠐⠣⠟⠙⠐⠜⠤⠠⠠⠕⠇⠫⠲⠀⠙` + - actual: `⠋⠏⠒⠓⠎⠢⠊⠄⠦⠄⠴⠠⠠⠟⠙⠠⠴⠤⠴⠠⠠⠕⠇⠫` + - first differing cell (zero-based): 172 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #14: 한국타이어앤테크놀로지㈜(대표이사 이수일, 이하 한국타이어)가 국제자동차연맹(FIA)이 주관하는 ‘FIA 주니어 ERC(FIA Junior ERC, 이하 주니어 ERC)’ 대회에 레이싱 타이어를 독점 공급한다. + - expected: `⠍⠉⠕⠎⠀⠴⠠⠠⠑⠗⠉⠦⠄⠴⠠⠠⠋⠊⠁⠀⠠⠚⠥⠝` + - actual: `⠍⠉⠕⠎⠀⠴⠠⠠⠻⠉⠦⠄⠴⠠⠠⠋⠊⠁⠀⠠⠚⠥⠝⠊` + - first differing cell (zero-based): 114 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #1: 김 지사는 18일(이하 현지시각) 미국 코네티컷주 댄버리 린데 본사에서 산지브 람바(Sanjiv Lamba) 린데 회장, 성백석 린데코리아 회장, 조일교 아산시 부시장과 투자양해각서(MOU)을 체결했다. + - expected: `⠠⠎⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - actual: `⠠⠎⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - first differing cell (zero-based): 175 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #62: LG디스플레이가 자사의 유기발광다이오드(OLED) TV 패널이 글로벌 친환경 인증기관인 카본 트러스트(Carbon Trust)로 부터 탄소발자국 인증을 획득했다고 19일 밝혔다. + - expected: `⠧⠶⠊⠣⠕⠥⠊⠪⠴⠐⠣⠠⠠⠕⠇⠫⠐⠜⠀⠠⠠⠞⠧⠲` + - actual: `⠧⠶⠊⠣⠕⠥⠊⠪⠦⠄⠴⠠⠠⠕⠇⠫⠠⠴⠀⠴⠠⠠⠞⠧` + - first differing cell (zero-based): 35 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` + +### `decimal_point_between_ascii_digits` + +Of the 4546 candidates, 401 are the actual `pending_rule_review` subcluster. The other 4145 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 440 mismatches were evaluable and 127 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2814 ⠔ -> U+280A ⠊`: 19 +- `U+2800 ⠀ -> U+2812 ⠒`: 16 +- `U+2810 ⠐ -> U+2832 ⠲`: 13 +- `U+2800 ⠀ -> U+283C ⠼`: 10 +- `U+2820 ⠠ -> U+2834 ⠴`: 10 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 39 +- `pending_rule_review`: 401 + +Representative `exact` samples: + +- `sentence_01.json` #2: 이번 종합시행계획은 과학기술분야 연구개발 예산 5조2천418억, 정보통신방송기술(ICT) 분야 연구개발 예산 1조4천308억원을 대상으로 하며 지원 예산은 지난해(6조4천161억원) 보다 약 3.9% 증가한 규모이다. + - expected: `⠕⠘⠾⠀⠨⠿⠚⠃⠠⠕⠚⠗⠶⠈⠌⠚⠽⠁⠵⠀⠈⠧⠚⠁` + - actual: `⠕⠘⠾⠀⠨⠿⠚⠃⠠⠕⠚⠗⠶⠈⠌⠚⠽⠁⠵⠀⠈⠧⠚⠁` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #5: 현재의 경기상황을 보여주는 동행지수(1월 기준, 99.8→99.4)·앞으로의 흐름을 보여주는 선행지수(98.8→98.5)는 모두 기준(100)보다 낮은 수준에서 하락세를 지속하면서 경기 부진을 반영하는 모양새다. + - expected: `⠚⠡⠨⠗⠺⠀⠈⠻⠈⠕⠇⠶⠚⠧⠶⠮⠀⠘⠥⠱⠨⠍⠉⠵` + - actual: `⠚⠡⠨⠗⠺⠀⠈⠻⠈⠕⠇⠶⠚⠧⠶⠮⠀⠘⠥⠱⠨⠍⠉⠵` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #8: LG전자의 1분기 영업이익은 전년동기대비 22.9% 감소한 1조4974억원이다. 특히 2009년 국제회계기준(IFRS) 도입 이후 처음으로 삼성전자의 영업이익을 넘어섰다. + - expected: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠣⠺⠀⠼⠁⠘⠛⠈⠕⠀⠻⠎⠃⠕⠕` + - actual: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠣⠺⠀⠼⠁⠘⠛⠈⠕⠀⠻⠎⠃⠕⠕` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #43: 시카고상품거래소(CME) 페드워치에 따르면 연준이 이번 FOMC에서 금리를 동결할 가능성은 지난 15일 45.4%에서 17일 18.1%로 줄었다. 반면 0.25%포인트 인상할 가능성은 54.6%에서 81.9%로 상승했다. + - expected: `⠠⠕⠋⠈⠥⠇⠶⠙⠍⠢⠈⠎⠐⠗⠠⠥⠦⠄⠴⠠⠠⠉⠍⠑` + - actual: `⠠⠕⠋⠈⠥⠇⠶⠙⠍⠢⠈⠎⠐⠗⠠⠥⠦⠄⠴⠠⠠⠉⠍⠑` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_01.json` #414: 셀트리온은 지난 5일 유럽의약품청(EMA)에 유플라이마의 20㎎/0.2ml(이하 20㎎) 제형을 추가하는 품목 변경 허가 신청을 했다고 16일 밝혔다. + - expected: `⠺⠀⠼⠃⠚⠴⠍⠛⠸⠌⠼⠚⠲⠃⠍⠇⠦⠄⠕⠚⠀⠼⠃⠚` + - actual: `⠺⠀⠼⠃⠚⠴⠍⠛⠲⠸⠌⠼⠚⠲⠃⠴⠍⠇⠦⠄⠕⠚⠀⠼` + - first differing cell (zero-based): 60 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #420: 대전대학교(총장 남상호) LINC3.0사업단(단장 이영환)은 지난 6일 특화 분야 ICC(기업협업) 참여학과 중심의 기술개발 프로그램인 ‘2023 혜화 All-SET 기술사업화’ 운영을 위한 선정평가를 진행했다고 7일 밝혔다. + - expected: `⠥⠠⠴⠀⠴⠠⠠⠇⠔⠉⠼⠉⠲⠚⠇⠎⠃⠊⠒⠦⠄⠊⠒⠨` + - actual: `⠥⠠⠴⠀⠴⠠⠠⠇⠊⠝⠉⠼⠉⠲⠚⠇⠎⠃⠊⠒⠦⠄⠊⠒` + - first differing cell (zero-based): 30 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #413: 이날 ‘올해 1분기(1~3월) 분기 실적’을 발표한 주요 기업은 사회연결망(SNS) 기업인 메타(META↑0.89%)와 글로벌 호텔 체인 힐튼 월드와이드 홀딩스(HLT↓3.41%), 세계 최대 항공기 제조업체 보잉(BA↑0.42%) 입니다. + - expected: `⠄⠴⠠⠠⠍⠑⠞⠁⠀⠰⠒⠕⠀⠼⠚⠲⠓⠊⠴⠏⠠⠴⠧⠀` + - actual: `⠄⠴⠠⠠⠍⠑⠞⠁⠲⠀⠰⠒⠕⠀⠼⠚⠲⠓⠊⠴⠏⠠⠴⠧` + - first differing cell (zero-based): 98 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #957: 1분기 실질 국내총소득(GDI)은 0.8% 늘어 증가율이 실질 GDP(0.3%)를 웃돌았다. 원유·천연가스 등 주요 수입품 가격 하락폭이 반도체 등 주요 수출품 가격 하락폭보다 커 교역조건이 개선됐기 때문이다. + - expected: `⠠⠕⠂⠨⠕⠂⠀⠴⠰⠠⠠⠛⠙⠏⠦⠄⠼⠚⠲⠉⠴⠏⠠⠴` + - actual: `⠠⠕⠂⠨⠕⠂⠀⠴⠠⠠⠛⠙⠏⠐⠣⠼⠚⠲⠉⠴⠏⠴⠐⠜` + - first differing cell (zero-based): 65 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #167: 올해 전 세계 반도체 시장 전망도 암울하다. 세계반도체시장통계기구(WSTS)에 따르면 올해 글로벌 반도체 매출은 5천565억 달러(약 706조5천880억원)로 지난해(5천801억달러) 대비 4.1% 감소할 것으로 조사됐다. + - expected: `⠈⠍⠦⠄⠴⠠⠠⠺⠎⠞⠎⠠⠴⠝⠀⠠⠊⠐⠪⠑⠡⠀⠥⠂` + - actual: `⠈⠍⠦⠄⠴⠠⠠⠺⠌⠎⠠⠴⠝⠀⠠⠊⠐⠪⠑⠡⠀⠥⠂⠚` + - first differing cell (zero-based): 67 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #420: 대전대학교(총장 남상호) LINC3.0사업단(단장 이영환)은 지난 6일 특화 분야 ICC(기업협업) 참여학과 중심의 기술개발 프로그램인 ‘2023 혜화 All-SET 기술사업화’ 운영을 위한 선정평가를 진행했다고 7일 밝혔다. + - expected: `⠥⠠⠴⠀⠴⠠⠠⠇⠔⠉⠼⠉⠲⠚⠇⠎⠃⠊⠒⠦⠄⠊⠒⠨` + - actual: `⠥⠠⠴⠀⠴⠠⠠⠇⠊⠝⠉⠼⠉⠲⠚⠇⠎⠃⠊⠒⠦⠄⠊⠒` + - first differing cell (zero-based): 30 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #147: 20일(현지시간) 미국부동산중개인협회(NAR)는 지난 3월 기존 주택 매매 건수가 444만건으로 전월보다 2.4% 줄었다고 밝혔다. 전년 동기와 비교하면 22% 급감했다. + - expected: `⠚⠽⠦⠄⠴⠠⠠⠝⠁⠗⠠⠴⠉⠵⠀⠨⠕⠉⠒⠀⠼⠉⠏⠂` + - actual: `⠚⠽⠦⠄⠴⠠⠠⠝⠜⠠⠴⠉⠵⠀⠨⠕⠉⠒⠀⠼⠉⠏⠂⠀` + - first differing cell (zero-based): 46 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #49: 유럽중앙은행(ECB)이 크레디트스위스(CS)의 유동성 위기에도 ‘빅스텝’(기준금리를 한 번에 0.5%포인트 인상)을 단행하자, 미국 연방준비제도(Fed·연준) 역시 금리 인상을 멈추지 않을 것이라는 전망이 힘을 얻고 있다. + - expected: `⠊⠥⠦⠄⠴⠠⠋⠫⠐⠆⠡⠨⠛⠠⠴⠀⠱⠁⠠⠕⠀⠈⠪⠢` + - actual: `⠊⠥⠦⠄⠴⠠⠋⠫⠲⠐⠆⠡⠨⠛⠠⠴⠀⠱⠁⠠⠕⠀⠈⠪` + - first differing cell (zero-based): 149 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `korean_inline_parenthesized_single_arithmetic_operator` + +Of the 23 candidates, 1 are the actual `pending_rule_review` subcluster. The other 22 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 1 mismatches were evaluable and 0 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Mismatch primary-class distribution: + +- `pending_rule_review`: 1 + +Representative `exact` samples: + +- `sentence_01.json` #2921: 경제자유도가 높아지면 1인당 GDP도 개선되는 것으로 나타났다. OECD 회원국의 2021년 경제자유도와 1인당 GDP간 관계를 분석한 결과, 경제자유도와 1인당 GDP 간에는 정(+)의 상관관계(상관계수 +0.46)를 보였다. + - expected: `⠈⠻⠨⠝⠨⠣⠩⠊⠥⠫⠀⠉⠥⠲⠣⠨⠕⠑⠡⠀⠼⠁⠟⠊` + - actual: `⠈⠻⠨⠝⠨⠣⠩⠊⠥⠫⠀⠉⠥⠲⠣⠨⠕⠑⠡⠀⠼⠁⠟⠊` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #417: 시는 세종고용복지플러스(+)센터, 세종여성새로일하기센터, 세종과학기술인협회 등과 공동으로 기업지원사업안내는 물론, 청년들이 면접 부담을 덜 수 있도록 청년면접비지원사업도 동시 진행하고 있다. + - expected: `⠠⠕⠉⠵⠀⠠⠝⠨⠿⠈⠥⠬⠶⠘⠭⠨⠕⠙⠮⠐⠎⠠⠪⠦` + - actual: `⠠⠕⠉⠵⠀⠠⠝⠨⠿⠈⠥⠬⠶⠘⠭⠨⠕⠙⠮⠐⠎⠠⠪⠦` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #12396: “2027년까지 수출과 매출에서 우리 중소기업이 차지하는 비율이 50% 이상이 되는 K-중소기업 50플러스(+)를 달성하는 데 있어서 메인비즈 기업이 선봉장으로 나서주기를 바랍니다.” + - expected: `⠦⠼⠃⠚⠃⠛⠀⠉⠡⠠⠫⠨⠕⠀⠠⠍⠰⠯⠈⠧⠀⠑⠗⠰` + - actual: `⠦⠼⠃⠚⠃⠛⠀⠉⠡⠠⠫⠨⠕⠀⠠⠍⠰⠯⠈⠧⠀⠑⠗⠰` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #798: 여기에 주요 산유국 협의체인 OPEC 플러스(+)가 미국의 직간접적인 압박에도 추가 감산을 결정하면서 중동에 대한 미국의 영향력이 약화하고 있다는 분석도 힘을 얻고 있다. + - expected: `⠱⠈⠕⠝⠀⠨⠍⠬⠀⠇⠒⠩⠈⠍⠁⠀⠚⠱⠃⠺⠰⠝⠟⠀` + - actual: `⠱⠈⠕⠝⠀⠨⠍⠬⠀⠇⠒⠩⠈⠍⠁⠀⠚⠱⠃⠺⠰⠝⠟⠀` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_02.json` #948: 유니브시티(UNIV+CITY)는 대학교(UNIVERSITY)와 시(CITY)의 합성어로 더하기(+)는 시와 대학, 기업 등과의 상생을 의미한다. 파란과 빨간, 노란색으로 자유롭고 간편함을 표현한 손 글씨(캘리그라피)를 통해 대학도시 천안의 젊음을 상징한다. + - expected: `⠄⠴⠠⠠⠥⠝⠊⠧⠐⠖⠠⠠⠉⠰⠽⠠⠴⠉⠵⠀⠊⠗⠚⠁` + - actual: `⠄⠴⠠⠠⠥⠝⠊⠧⠲⠢⠴⠠⠠⠉⠰⠽⠠⠴⠉⠵⠀⠊⠗⠚` + - first differing cell (zero-based): 18 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `korean_majority_same_token_roman_sandwich_non_domain` + +Of the 947 candidates, 201 are the actual `pending_rule_review` subcluster. The other 746 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +Mismatch primary-class distribution: + +- `corpus_suspect`: 10 +- `pending_rule_review`: 201 + +Representative `exact` samples: + +- `sentence_01.json` #314: 이와 관련해서 FAA는 “FAA항공정보(NOTAMS) 업데이트에 관여하는 시스템이 작동하지 않고 있다”며 “복구작업을 벌이고 있지만 현재로서는 복구 시점을 예상하기 어렵다”고 말했다. + - expected: `⠕⠧⠀⠈⠧⠒⠐⠡⠚⠗⠠⠎⠀⠴⠠⠠⠋⠁⠁⠲⠉⠵⠀⠦` + - actual: `⠕⠧⠀⠈⠧⠒⠐⠡⠚⠗⠠⠎⠀⠴⠠⠠⠋⠁⠁⠲⠉⠵⠀⠦` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #122: 이번 연수는 자연과 상생하는 청정환경 도시를 조성하고 정보통신기술(IT)·생명공학기술(BT) 등 첨단산업단지를 만들어가는 제주도의 우수정책 사례를 벤치마킹함으로써, 심도 있는 정책의정 구현을 모색하기 위해 마련되었다. + - expected: `⠕⠘⠾⠀⠡⠠⠍⠉⠵⠀⠨⠣⠡⠈⠧⠀⠇⠶⠠⠗⠶⠚⠉⠵` + - actual: `⠕⠘⠾⠀⠡⠠⠍⠉⠵⠀⠨⠣⠡⠈⠧⠀⠇⠶⠠⠗⠶⠚⠉⠵` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #50: 당시 CC(폐쇄)TV에 찍힌 장면을 보면 A씨는 엘리베이터를 기다리는 B씨를 발견하자 몰래 뒤로 다가가 갑자기 피해 여성의 머리를 뒤에서 돌려차기로 가격하는 등 폭행했다. + - expected: `⠊⠶⠠⠕⠀⠴⠠⠠⠉⠉⠦⠄⠙⠌⠠⠧⠗⠠⠴⠴⠠⠠⠞⠧` + - actual: `⠊⠶⠠⠕⠀⠴⠠⠠⠉⠉⠦⠄⠙⠌⠠⠧⠗⠠⠴⠴⠠⠠⠞⠧` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #133: 우크라이나 전쟁 피해 지원 방안을 논의하는 이 회의에는 프랑스·독일·이탈리아·스페인 등 법무부 장관 20여명과 국제형사재판소(ICC)·유럽연합(EU) 관계자 등이 참석했다. 지난달 18일 출국한 한 장관은 22일 귀국한다. + - expected: `⠍⠋⠪⠐⠣⠕⠉⠀⠨⠾⠨⠗⠶⠀⠙⠕⠚⠗⠀⠨⠕⠏⠒⠀` + - actual: `⠍⠋⠪⠐⠣⠕⠉⠀⠨⠾⠨⠗⠶⠀⠙⠕⠚⠗⠀⠨⠕⠏⠒⠀` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_01.json` #142: ‘메타버스존’에서는 증강현실(AR)·가상현실(VR)에 적용되는 혁신기술이 소개된다. 글라스를 착용하고 최첨단 3D 센싱모듈이 구현하는 가상현실을 관람객이 직접 체험할 수 있도록 했다. + - expected: `⠠⠕⠂⠦⠄⠴⠠⠠⠁⠗⠠⠴⠐⠆⠫⠇⠶⠚⠡⠠⠕⠂⠦⠄` + - actual: `⠠⠕⠂⠦⠄⠴⠠⠠⠜⠠⠴⠐⠆⠫⠇⠶⠚⠡⠠⠕⠂⠦⠄⠴` + - first differing cell (zero-based): 34 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #230: INC는 아이디어(I)-니즈(N)-역량(C)의 융합을 뜻하며, 파괴적 혁신과 기업가적 대학으로서 산학협력을 활성화하기 위한 방법론으로, 지속가능한 가치창출형 산학협력을 위한 브랜드이다. + - expected: `⠴⠠⠠⠊⠝⠉⠲⠉⠵⠀⠣⠕⠊⠕⠎⠦⠄⠴⠠⠊⠠⠴⠤⠉` + - actual: `⠴⠠⠠⠔⠉⠲⠉⠵⠀⠣⠕⠊⠕⠎⠦⠄⠴⠠⠊⠠⠴⠤⠉⠕` + - first differing cell (zero-based): 3 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #1041: 사모투자펀드(PEF) 운용사 IMM크레딧앤솔루션(IMM CS·이하 ICS)이 KT클라우드에 6000억원을 수혈하며 주요 주주로 올라선다. 대규모 투자금을 유치한 KT클라우드는 데이터센터 확충 등 본격적인 신사업 확대에 나설 방침이다. + - expected: `⠊⠍⠍⠀⠠⠠⠉⠎⠐⠆⠕⠚⠀⠴⠠⠠⠊⠉⠎⠠⠴⠕⠀⠴` + - actual: `⠊⠍⠍⠀⠠⠠⠉⠎⠲⠐⠆⠕⠚⠀⠴⠠⠠⠊⠉⠎⠠⠴⠕⠀` + - first differing cell (zero-based): 64 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #338: 체험에 앞서 간단한 O·X 퀴즈가 진행됐다. 액화석유가스(LPG)·액화천연가스(LNG)·부탄가스의 차이점을 설명하고 누출사고 시 대응 방법 등을 O·X로 답하는 방식이다. + - expected: `⠊⠒⠚⠒⠀⠴⠠⠕⠐⠆⠴⠠⠭⠲⠀⠋⠍⠗⠨⠪⠫⠀⠨⠟` + - actual: `⠊⠒⠚⠒⠀⠴⠠⠕⠲⠐⠆⠴⠠⠭⠲⠀⠋⠍⠗⠨⠪⠫⠀⠨` + - first differing cell (zero-based): 22 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `korean_prefixed_closed_allcaps_parenthetical` + +Of the 54492 candidates, 4549 are the actual `pending_rule_review` subcluster. The other 49943 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +Mismatch primary-class distribution: + +- `corpus_suspect`: 1047 +- `pending_rule_review`: 4549 +- `unsupported_character_review`: 4 + +Representative `exact` samples: + +- `sentence_01.json` #1: LG전자는 ‘업(UP)가전’을 무기로 내세우고 있다. 이번 CES에서도 LG 씽큐 앱에서 터치만으로 제품 색상을 바꿀 수 있는 무드업 냉장고를 비롯한 다양한 업가전을 선보일 예정이다. + - expected: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠉⠵⠀⠠⠦⠎⠃⠦⠄⠴⠠⠠⠥⠏⠠` + - actual: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠉⠵⠀⠠⠦⠎⠃⠦⠄⠴⠠⠠⠥⠏⠠` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #1: 세종시 주최, 세종시탄소중립지원센터 주관으로 열린 이번 포럼의 주요의제는 기업의 이에스지(ESG) 경영과 사회적 책임, 국내 온실가스 배출권거래제 현황과 대응 방안을 마련하기 위해 개최됐다. + - expected: `⠠⠝⠨⠿⠠⠕⠀⠨⠍⠰⠽⠐⠀⠠⠝⠨⠿⠠⠕⠓⠒⠠⠥⠨` + - actual: `⠠⠝⠨⠿⠠⠕⠀⠨⠍⠰⠽⠐⠀⠠⠝⠨⠿⠠⠕⠓⠒⠠⠥⠨` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #2: 연구소는 전남대학교, 순천대학교, 전남도립대학교, 동신대학교, 한국에너지공과대학교, 충남대학교, 부산대학교 등 드론, 빅데이터, 인공지능(AI) 관련 대학 교수와 전남테크노파크, ㈜엘시스 등 전문가로 기획연구팀을 구성한다. + - expected: `⠡⠈⠍⠠⠥⠉⠵⠀⠨⠾⠉⠢⠊⠗⠚⠁⠈⠬⠐⠀⠠⠛⠰⠾` + - actual: `⠡⠈⠍⠠⠥⠉⠵⠀⠨⠾⠉⠢⠊⠗⠚⠁⠈⠬⠐⠀⠠⠛⠰⠾` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #1: 일본 정부는 이날 국가안전보장회의(NSC)를 열어 대응 방침을 논의했다. 이어 북한의 미사일 발사에 대해 “엄중히 항의하고 강하게 비난한다”고 밝혔다. + - expected: `⠕⠂⠘⠷⠀⠨⠻⠘⠍⠉⠵⠀⠕⠉⠂⠀⠈⠍⠁⠫⠣⠒⠨⠾` + - actual: `⠕⠂⠘⠷⠀⠨⠻⠘⠍⠉⠵⠀⠕⠉⠂⠀⠈⠍⠁⠫⠣⠒⠨⠾` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_01.json` #14: 새롭게 선보인 DRX 브랜딩과 팀 컬러는 2023년 1월 1일부터 DRX의 온오프라인 콘텐츠, 엠디(MD), SNS, 유튜브 등 모든 제작물에 적용될 예정이다. + - expected: `⠰⠪⠐⠀⠝⠢⠊⠕⠴⠐⠣⠠⠠⠍⠙⠐⠜⠂⠀⠠⠠⠎⠝⠎` + - actual: `⠰⠪⠐⠀⠝⠢⠊⠕⠦⠄⠴⠠⠠⠍⠙⠠⠴⠐⠀⠴⠠⠠⠎⠝` + - first differing cell (zero-based): 97 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #10: 설동호 대전교육감은 “학교 교육과정 및 여건에 맞는 인공지능(AI)교육 선도학교 운영을 통해 AI·SW교육 기반 구축과 학생들과 교원들의 미래 디지털 역량이 함양될 것이다.”라고 말했다. + - expected: `⠚⠗⠀⠴⠠⠠⠁⠊⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕⠘` + - actual: `⠚⠗⠀⠴⠠⠠⠁⠊⠲⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕` + - first differing cell (zero-based): 93 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #1: 김 지사는 18일(이하 현지시각) 미국 코네티컷주 댄버리 린데 본사에서 산지브 람바(Sanjiv Lamba) 린데 회장, 성백석 린데코리아 회장, 조일교 아산시 부시장과 투자양해각서(MOU)을 체결했다. + - expected: `⠠⠎⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - actual: `⠠⠎⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - first differing cell (zero-based): 175 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #19: 이날 행사에는 구 대표 외에 권봉석 (주)LG 최고운영책임자(COO), 박일평 LG사이언스파크대표(사장)를 비롯해 각 계열사 최고기술책임자(CTO), 최고데이터책임자(CDO), 최고인사책임자(CHO) 등이 참석해 국내 이공계 R&D 인재 400여명과 만났다. + - expected: `⠁⠕⠢⠨⠦⠄⠴⠠⠠⠉⠓⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢⠠` + - actual: `⠁⠕⠢⠨⠦⠄⠴⠠⠉⠠⠓⠠⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢` + - first differing cell (zero-based): 188 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `korean_prefixed_closed_roman_annotation_rule_34_order` + +Of the 64382 candidates, 5450 are the actual `pending_rule_review` subcluster. The other 58932 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 6634 mismatches were evaluable and 1166 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2834 ⠴ -> U+2826 ⠦`: 1163 +- `U+2810 ⠐ -> U+2826 ⠦`: 3 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 1180 +- `pending_rule_review`: 5450 +- `unsupported_character_review`: 4 + +Representative `exact` samples: + +- `sentence_01.json` #1: LG전자는 ‘업(UP)가전’을 무기로 내세우고 있다. 이번 CES에서도 LG 씽큐 앱에서 터치만으로 제품 색상을 바꿀 수 있는 무드업 냉장고를 비롯한 다양한 업가전을 선보일 예정이다. + - expected: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠉⠵⠀⠠⠦⠎⠃⠦⠄⠴⠠⠠⠥⠏⠠` + - actual: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠉⠵⠀⠠⠦⠎⠃⠦⠄⠴⠠⠠⠥⠏⠠` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #1: 세종시 주최, 세종시탄소중립지원센터 주관으로 열린 이번 포럼의 주요의제는 기업의 이에스지(ESG) 경영과 사회적 책임, 국내 온실가스 배출권거래제 현황과 대응 방안을 마련하기 위해 개최됐다. + - expected: `⠠⠝⠨⠿⠠⠕⠀⠨⠍⠰⠽⠐⠀⠠⠝⠨⠿⠠⠕⠓⠒⠠⠥⠨` + - actual: `⠠⠝⠨⠿⠠⠕⠀⠨⠍⠰⠽⠐⠀⠠⠝⠨⠿⠠⠕⠓⠒⠠⠥⠨` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #2: 연구소는 전남대학교, 순천대학교, 전남도립대학교, 동신대학교, 한국에너지공과대학교, 충남대학교, 부산대학교 등 드론, 빅데이터, 인공지능(AI) 관련 대학 교수와 전남테크노파크, ㈜엘시스 등 전문가로 기획연구팀을 구성한다. + - expected: `⠡⠈⠍⠠⠥⠉⠵⠀⠨⠾⠉⠢⠊⠗⠚⠁⠈⠬⠐⠀⠠⠛⠰⠾` + - actual: `⠡⠈⠍⠠⠥⠉⠵⠀⠨⠾⠉⠢⠊⠗⠚⠁⠈⠬⠐⠀⠠⠛⠰⠾` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #1: 일본 정부는 이날 국가안전보장회의(NSC)를 열어 대응 방침을 논의했다. 이어 북한의 미사일 발사에 대해 “엄중히 항의하고 강하게 비난한다”고 밝혔다. + - expected: `⠕⠂⠘⠷⠀⠨⠻⠘⠍⠉⠵⠀⠕⠉⠂⠀⠈⠍⠁⠫⠣⠒⠨⠾` + - actual: `⠕⠂⠘⠷⠀⠨⠻⠘⠍⠉⠵⠀⠕⠉⠂⠀⠈⠍⠁⠫⠣⠒⠨⠾` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_01.json` #14: 새롭게 선보인 DRX 브랜딩과 팀 컬러는 2023년 1월 1일부터 DRX의 온오프라인 콘텐츠, 엠디(MD), SNS, 유튜브 등 모든 제작물에 적용될 예정이다. + - expected: `⠰⠪⠐⠀⠝⠢⠊⠕⠴⠐⠣⠠⠠⠍⠙⠐⠜⠂⠀⠠⠠⠎⠝⠎` + - actual: `⠰⠪⠐⠀⠝⠢⠊⠕⠦⠄⠴⠠⠠⠍⠙⠠⠴⠐⠀⠴⠠⠠⠎⠝` + - first differing cell (zero-based): 97 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #78: 마지막으로 신규 회원사인 ㈜그레비티(대표 최항주)에 대한 소개와 서중석 교수의 발제로 ‘글쓰기에 필요한 다양한 생성형(Generative) AI Searcher’에 대한 토론이 진행됐다. + - expected: `⠀⠠⠗⠶⠠⠻⠚⠻⠴⠐⠣⠠⠛⠢⠻⠁⠞⠊⠧⠑⠐⠜⠀⠠` + - actual: `⠀⠠⠗⠶⠠⠻⠚⠻⠦⠄⠴⠠⠛⠢⠻⠁⠞⠊⠧⠑⠠⠴⠀⠴` + - first differing cell (zero-based): 116 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_03.json` #37: 경찰은 주변 폐쇄회로(CC)TV 등을 분석해 A씨 동선을 추적했고 14일 오후 경기도 양주의 한 주택에서 A씨를 긴급체포했다. + - expected: `⠌⠠⠧⠗⠚⠽⠐⠥⠴⠐⠣⠠⠠⠉⠉⠐⠜⠠⠠⠞⠧⠲⠀⠊` + - actual: `⠌⠠⠧⠗⠚⠽⠐⠥⠦⠄⠴⠠⠠⠉⠉⠠⠴⠴⠠⠠⠞⠧⠲⠀` + - first differing cell (zero-based): 21 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_04.json` #42: 경찰은 폐쇄회로(CC)TV 분석과 쇠구슬 판매 업체 탐문 수사, 국과수 발사지점 방향성 감정 등을 통해 발사 의심 세대를 특정해 이날 피의자 A씨를 자택에서 검거했다. + - expected: `⠌⠠⠧⠗⠚⠽⠐⠥⠴⠐⠣⠠⠠⠉⠉⠐⠜⠠⠠⠞⠧⠲⠀⠘` + - actual: `⠌⠠⠧⠗⠚⠽⠐⠥⠦⠄⠴⠠⠠⠉⠉⠠⠴⠴⠠⠠⠞⠧⠲⠀` + - first differing cell (zero-based): 16 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` + +Representative `mismatch` samples: + +- `sentence_01.json` #14: 새롭게 선보인 DRX 브랜딩과 팀 컬러는 2023년 1월 1일부터 DRX의 온오프라인 콘텐츠, 엠디(MD), SNS, 유튜브 등 모든 제작물에 적용될 예정이다. + - expected: `⠰⠪⠐⠀⠝⠢⠊⠕⠴⠐⠣⠠⠠⠍⠙⠐⠜⠂⠀⠠⠠⠎⠝⠎` + - actual: `⠰⠪⠐⠀⠝⠢⠊⠕⠦⠄⠴⠠⠠⠍⠙⠠⠴⠐⠀⠴⠠⠠⠎⠝` + - first differing cell (zero-based): 97 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #10: 설동호 대전교육감은 “학교 교육과정 및 여건에 맞는 인공지능(AI)교육 선도학교 운영을 통해 AI·SW교육 기반 구축과 학생들과 교원들의 미래 디지털 역량이 함양될 것이다.”라고 말했다. + - expected: `⠚⠗⠀⠴⠠⠠⠁⠊⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕⠘` + - actual: `⠚⠗⠀⠴⠠⠠⠁⠊⠲⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕` + - first differing cell (zero-based): 93 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #1: 김 지사는 18일(이하 현지시각) 미국 코네티컷주 댄버리 린데 본사에서 산지브 람바(Sanjiv Lamba) 린데 회장, 성백석 린데코리아 회장, 조일교 아산시 부시장과 투자양해각서(MOU)을 체결했다. + - expected: `⠠⠎⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - actual: `⠠⠎⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - first differing cell (zero-based): 175 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #19: 이날 행사에는 구 대표 외에 권봉석 (주)LG 최고운영책임자(COO), 박일평 LG사이언스파크대표(사장)를 비롯해 각 계열사 최고기술책임자(CTO), 최고데이터책임자(CDO), 최고인사책임자(CHO) 등이 참석해 국내 이공계 R&D 인재 400여명과 만났다. + - expected: `⠁⠕⠢⠨⠦⠄⠴⠠⠠⠉⠓⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢⠠` + - actual: `⠁⠕⠢⠨⠦⠄⠴⠠⠉⠠⠓⠠⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢` + - first differing cell (zero-based): 188 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `korean_prefixed_roman_parenthetical_followed_by_allcaps_hyphen_suffix` + +Of the 11 candidates, 0 are the actual `pending_rule_review` subcluster. The other 11 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 10 mismatches were evaluable and 0 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Mismatch primary-class distribution: + +- `corpus_suspect`: 10 + +Representative `exact` samples: + +- `sentence_02.json` #10613: 오스템임플란트 경영권 지분을 인수한 사모펀드(PEF) 연합군 유니슨캐피탈코리아(UCK)-MBK파트너스가 2차 공개매수를 거쳐 총 96.1% 지분을 확보했다. 오스템임플란트는 자발적상장폐지 요건을 넘겨 상장폐지 절차를 밟는다. + - expected: `⠥⠠⠪⠓⠝⠢⠕⠢⠙⠮⠐⠣⠒⠓⠪⠀⠈⠻⠻⠈⠏⠒⠀⠨` + - actual: `⠥⠠⠪⠓⠝⠢⠕⠢⠙⠮⠐⠣⠒⠓⠪⠀⠈⠻⠻⠈⠏⠒⠀⠨` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_01.json` #35: 삼성전자는 올해 CES에서 77인치 OLED TV를 처음으로 공개할 예정이다. 이는 지난 8월 ‘국제정보디스플레이학술대회(IMID) 2022’에서 삼성디스플레이가 공개한 77인치 퀀텀닷(QD)-OLED 패널을 탑재한 제품이다. + - expected: `⠋⠏⠒⠓⠎⠢⠊⠄⠴⠐⠣⠟⠙⠐⠜⠤⠠⠠⠕⠇⠫⠲⠀⠙` + - actual: `⠋⠏⠒⠓⠎⠢⠊⠄⠦⠄⠴⠠⠠⠟⠙⠠⠴⠤⠴⠠⠠⠕⠇⠫` + - first differing cell (zero-based): 172 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #7731: 카젬 전 사장은 지난 2017년 9월 한국GM 사장으로 취임했으며, 작년 6월 중국 상하이자동차(SAIC)-GM 총괄 부사장을 맡고 있다. + - expected: `⠚⠣⠕⠨⠊⠿⠰⠣⠴⠐⠣⠠⠠⠎⠁⠊⠉⠐⠜⠤⠠⠠⠛⠍` + - actual: `⠚⠣⠕⠨⠊⠿⠰⠣⠦⠄⠴⠠⠠⠎⠁⠊⠉⠠⠴⠤⠴⠠⠠⠛` + - first differing cell (zero-based): 91 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_03.json` #866: 제주삼다수 리본은 화학적 재활용 페트인 ‘스카이펫(SKYPET)-CR’을 사용한 제품으로, 제주개발공사가 SK케미칼과 손잡고 2021년 10월 업계 최초로 개발했다. + - expected: `⠠⠪⠋⠣⠕⠙⠝⠄⠴⠐⠣⠠⠠⠎⠅⠽⠏⠑⠞⠐⠜⠤⠠⠠` + - actual: `⠠⠪⠋⠣⠕⠙⠝⠄⠦⠄⠴⠠⠠⠎⠅⠽⠏⠑⠞⠠⠴⠤⠴⠠` + - first differing cell (zero-based): 47 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` + +### `mixed_roman_korean_word_before_uppercase_headword_expansion` + +Of the 10 candidates, 3 are the actual `pending_rule_review` subcluster. The other 7 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 3 mismatches were evaluable and 2 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2809 ⠉ -> U+2812 ⠒`: 1 +- `U+2811 ⠑ -> U+282B ⠫`: 1 + +Mismatch primary-class distribution: + +- `pending_rule_review`: 3 + +Representative `exact` samples: + +- `sentence_01.json` #18: 2023년형 신제품은 매터(Matter)와 HCA(Home Connectivity Alliance) 표준을 지원하는 스마트싱스 허브(SmartThings Hub)를 기반으로 다양한 기기를 자동으로 연결하고, 제어·관리할 수 있는 것이 특징이다. + - expected: `⠼⠃⠚⠃⠉⠀⠉⠡⠚⠻⠀⠠⠟⠨⠝⠙⠍⠢⠵⠀⠑⠗⠓⠎` + - actual: `⠼⠃⠚⠃⠉⠀⠉⠡⠚⠻⠀⠠⠟⠨⠝⠙⠍⠢⠵⠀⠑⠗⠓⠎` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #15523: KAI(한국항공우주산업)은 AH(Airbus Helicopters)와 향후 10년간 예측되는 소형무장헬기(LAH)와 수리온(KUH) 300대 규모 생산 물량에 대한 선제적 통합 발주 계약에 서명했다고 31일 밝혔다. + - expected: `⠴⠠⠠⠅⠁⠊⠦⠄⠚⠒⠈⠍⠁⠚⠶⠈⠿⠍⠨⠍⠇⠒⠎⠃` + - actual: `⠴⠠⠠⠅⠁⠊⠦⠄⠚⠒⠈⠍⠁⠚⠶⠈⠿⠍⠨⠍⠇⠒⠎⠃` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_01.json` #3656: 제너시스BBQ 그룹이 미국 현지시간 19일 뉴저지주 잉글우드(Englewood)에 BSK(BBQ Smart Kitchen) 1호점을 그랜드 오픈하고 배달·포장 전문 매장을 통한 기하급수 성장을 이어간다고 밝혔다. + - expected: `⠠⠅⠊⠞⠡⠢⠐⠜⠲⠀⠼⠁⠀⠚⠥⠨⠎⠢⠮⠀⠈⠪⠐⠗` + - actual: `⠠⠅⠊⠞⠡⠢⠐⠜⠀⠼⠁⠀⠚⠥⠨⠎⠢⠮⠀⠈⠪⠐⠗⠒` + - first differing cell (zero-based): 104 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #9391: 이지스자산운용이 사옥인 여의도 세우빌딩의 2층을 리모델링하고 미국 그린빌딩위원회(USGBC)의 LEED(Leadership in Energy and Environmental Design) 골드(Gold) 등급 인증을 획득했다고 6일 밝혔다. + - expected: `⠴⠺⠀⠴⠠⠠⠇⠑⠑⠙⠐⠣⠠⠇⠂⠙⠻⠩⠊⠏⠀⠔⠀⠠` + - actual: `⠴⠺⠀⠴⠠⠠⠇⠑⠫⠐⠣⠠⠇⠂⠙⠻⠩⠊⠏⠀⠔⠀⠠⠢` + - first differing cell (zero-based): 95 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `multi_character_allcaps_roman_runs_joined_by_middle_dot` + +Of the 97 candidates, 96 are the actual `pending_rule_review` subcluster. The other 1 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +Mismatch primary-class distribution: + +- `corpus_suspect`: 1 +- `pending_rule_review`: 96 + +Representative `mismatch` samples: + +- `sentence_01.json` #398: 방송통신위원회는 이동통신 3사(SK텔레콤·KT·LG유플러스), 한국정보통신진흥협회(KAIT)와 협력한다. 오는 16일부터 각 통신사 명의로 가입자에게 ‘스미싱 문자 주의 안내’ 문자 메시지를 순차 발송할 예정이다. + - expected: `⠢⠐⠆⠴⠠⠠⠅⠞⠐⠆⠴⠠⠠⠇⠛⠲⠩⠙⠮⠐⠎⠠⠪⠠` + - actual: `⠢⠐⠆⠴⠠⠠⠅⠞⠲⠐⠆⠴⠠⠠⠇⠛⠲⠩⠙⠮⠐⠎⠠⠪` + - first differing cell (zero-based): 51 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #10: 설동호 대전교육감은 “학교 교육과정 및 여건에 맞는 인공지능(AI)교육 선도학교 운영을 통해 AI·SW교육 기반 구축과 학생들과 교원들의 미래 디지털 역량이 함양될 것이다.”라고 말했다. + - expected: `⠚⠗⠀⠴⠠⠠⠁⠊⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕⠘` + - actual: `⠚⠗⠀⠴⠠⠠⠁⠊⠲⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕` + - first differing cell (zero-based): 93 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #1112: 12일(현지시간) 미국 ABC·NBC방송 등에 따르면 유튜버이자 스카이다이버인 트레버 제이컵(29)는 지난 2021년 12월 자신의 유튜브 채널에 12분 47초짜리 비행 영상을 올렸다. + - expected: `⠁⠀⠴⠠⠠⠁⠃⠉⠐⠆⠴⠠⠠⠝⠃⠉⠲⠘⠶⠠⠿⠀⠊⠪` + - actual: `⠁⠀⠴⠠⠠⠁⠃⠉⠲⠐⠆⠴⠠⠠⠝⠃⠉⠲⠘⠶⠠⠿⠀⠊` + - first differing cell (zero-based): 30 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #2998: 독일 빌트암존탁이 여론조사기관에 의뢰해 지난 17~21일 유권자 1266명을 대상으로 실시한 여론조사에서 지지율 22%를 기록해 1위인 CDU·CSU(기독사회당) 지지율(26%)과 불과 4%포인트 차이를 나타냈다. + - expected: `⠟⠀⠴⠠⠠⠉⠙⠥⠐⠆⠴⠠⠠⠉⠎⠥⠦⠄⠈⠕⠊⠭⠇⠚` + - actual: `⠟⠀⠴⠠⠠⠉⠙⠥⠲⠐⠆⠴⠠⠠⠉⠎⠥⠦⠄⠈⠕⠊⠭⠇` + - first differing cell (zero-based): 130 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `percent_point_unit_list_comma` + +Of the 7 candidates, 2 are the actual `pending_rule_review` subcluster. The other 5 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 2 mismatches were evaluable and 0 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Mismatch primary-class distribution: + +- `pending_rule_review`: 2 + +Representative `exact` samples: + +- `sentence_01.json` #1767: 부산은행도 서민금융 상품 ‘새희망홀씨’ 대출 금리를 1%p 내렸다. 주담대와 전세대출, 신용대출도 최대 0.8%포인트(p), 0.85%p, 0.6%p씩 금리를 내리기로 했다. + - expected: `⠘⠍⠇⠒⠵⠚⠗⠶⠊⠥⠀⠠⠎⠑⠟⠈⠪⠢⠩⠶⠀⠇⠶⠙` + - actual: `⠘⠍⠇⠒⠵⠚⠗⠶⠊⠥⠀⠠⠎⠑⠟⠈⠪⠢⠩⠶⠀⠇⠶⠙` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #12305: 1일 금융감독원에 따르면 3월말 국내은행의 BIS기준 보통주자본비율, 기본자본비율, 총자본비율은 각각 12.88%, 14.24%, 15.58%로 나타났다. 전 분기 대비 각각 0.28%포인트(p), 0.33%p, 0.29%p 올랐다. + - expected: `⠼⠁⠕⠂⠀⠈⠪⠢⠩⠶⠫⠢⠊⠭⠏⠒⠝⠀⠠⠊⠐⠪⠑⠡` + - actual: `⠼⠁⠕⠂⠀⠈⠪⠢⠩⠶⠫⠢⠊⠭⠏⠒⠝⠀⠠⠊⠐⠪⠑⠡` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_02.json` #15633: 4일 금융감독원에 따르면 6월 말 국내은행의 BIS기준 보통주자본비율, 기본자본비율, 총자본비율은 각각 12.98%, 14.27%, 15.62%로 나타났다. 전 분기 대비 각각 0.08%포인트(p), 0.01%p, 0.04%p 올랐다. + - expected: `⠐⠀⠼⠚⠲⠚⠙⠴⠴⠏⠏⠀⠥⠂⠐⠣⠌⠊⠲` + - actual: `⠐⠀⠼⠚⠲⠚⠙⠴⠏⠏⠀⠥⠂⠐⠣⠌⠊⠲` + - first differing cell (zero-based): 194 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `pure_allcaps_segment_before_hyphen_and_multi_allcaps_segment_after` + +Of the 448 candidates, 80 are the actual `pending_rule_review` subcluster. The other 368 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 89 mismatches were evaluable and 0 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Mismatch primary-class distribution: + +- `corpus_suspect`: 9 +- `pending_rule_review`: 80 + +Representative `exact` samples: + +- `sentence_01.json` #54: 에버소울은 출시 전부터 지스타(G-STAR) 2022, 에이지에프(AGF) 2022 등 굵직한 국내 게임 및 서브컬처 행사에 출품해 기대를 모았다. 12월 말에는 사전 예약자 150만 명을 모객했다. + - expected: `⠝⠘⠎⠠⠥⠯⠵⠀⠰⠯⠠⠕⠀⠨⠾⠘⠍⠓⠎⠀⠨⠕⠠⠪` + - actual: `⠝⠘⠎⠠⠥⠯⠵⠀⠰⠯⠠⠕⠀⠨⠾⠘⠍⠓⠎⠀⠨⠕⠠⠪` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #313: 순천향대(총장 김승우)는 지난 3일 한국국제협력단(KOICA)과 함께 우즈베키스탄의 수도 타슈겐트에서 스타트업 지원센터 ‘U-ENTER(Uzbekistan Entrepreneurship Innovation Center)’ 준공식을 개최했다고 밝혔다. + - expected: `⠠⠛⠰⠾⠚⠜⠶⠊⠗⠦⠄⠰⠿⠨⠶⠀⠈⠕⠢⠠⠪⠶⠍⠠` + - actual: `⠠⠛⠰⠾⠚⠜⠶⠊⠗⠦⠄⠰⠿⠨⠶⠀⠈⠕⠢⠠⠪⠶⠍⠠` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #74: 아울러 제주도개발공사는 국내 생수업계에서는 처음으로 재활용 페트(CR-PET)를 적용한 화학적 재활용 페트 ‘제주삼다수 리본(RE:Born)’을 개발하는 등 소재혁신을 통한 친환경 라인업도 확대하고있다. + - expected: `⠣⠯⠐⠎⠀⠨⠝⠨⠍⠊⠥⠈⠗⠘⠂⠈⠿⠇⠉⠵⠀⠈⠍⠁` + - actual: `⠣⠯⠐⠎⠀⠨⠝⠨⠍⠊⠥⠈⠗⠘⠂⠈⠿⠇⠉⠵⠀⠈⠍⠁` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #883: 그룹 방탄소년단(BTS) 슈가의 첫 공식 솔로 음반 ‘D-데이’(D-DAY)가 발매 당일 한터차트 기준 107만장이 넘게 팔려나가며 첫날 판매량으로는 K팝 솔로 가수 최고 기록을 경신했다. + - expected: `⠈⠪⠐⠍⠃⠀⠘⠶⠓⠒⠠⠥⠉⠡⠊⠒⠦⠄⠴⠠⠠⠃⠞⠎` + - actual: `⠈⠪⠐⠍⠃⠀⠘⠶⠓⠒⠠⠥⠉⠡⠊⠒⠦⠄⠴⠠⠠⠃⠞⠎` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_01.json` #149: 탠덤 OLED를 탄성있는 플라스틱 기판에 결합한 것이 LG디스플레이의 차량용 플라스틱(P)-OLED다. 차량용 P-OLED는 LCD 대비 소비전력을 60% 줄이고, 무게는 80%나 저감해 전기차 시대에도 적합한 디스플레이라는 평가다. + - expected: `⠮⠐⠣⠠⠪⠓⠕⠁⠴⠐⠣⠠⠏⠐⠜⠤⠠⠠⠕⠇⠫⠲⠊⠲` + - actual: `⠮⠐⠣⠠⠪⠓⠕⠁⠦⠄⠴⠠⠏⠠⠴⠤⠴⠠⠠⠕⠇⠫⠲⠊` + - first differing cell (zero-based): 87 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #8478: 논문 제목은 ‘Strong electron-phonon coupling driven charge density wave states in stoichiometric 1T-VS2 crystals’이다. 논문은 화학기상수송법으로 제작된 이황화바나듐(VS2)에서 관찰한 완벽한 결정성을 가진 양자상전이 현상을 다뤘다. + - expected: `⠑⠇⠑⠉⠞⠗⠕⠝⠤⠏⠓⠕⠝⠕⠝⠀⠉⠳⠏⠇⠬⠀⠙⠗` + - actual: `⠑⠇⠑⠉⠞⠗⠕⠝⠔⠏⠓⠕⠝⠕⠝⠀⠉⠳⠏⠇⠬⠀⠙⠗` + - first differing cell (zero-based): 28 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #947: BMW코리아 미래재단이 ‘2023 서울안전한마당’에 이동식 에너지 저장소(ESS)인 ‘넥스트 그린 투-고(NEXT GREEN TO-GO)’ 부스를 마련하고 체험형 교육 프로그램을 운영한다. + - expected: `⠤⠈⠥⠦⠄⠴⠠⠠⠠⠝⠑⠭⠞⠀⠛⠗⠑⠢⠀⠞⠕⠤⠛⠠` + - actual: `⠤⠈⠥⠦⠄⠴⠠⠠⠝⠑⠭⠞⠀⠠⠠⠛⠗⠑⠢⠀⠠⠠⠞⠕` + - first differing cell (zero-based): 103 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #6743: 변압기 절연유로 사용되는 발암물질인 폴리염화비페닐(PCBs)도 평균 0.003pg WHO-TEQ/㎥으로, 최근 2년 평균 0.004pg WHO-TEQ/㎥ 대비 감소세를 유지했다. + - expected: `⠓⠕⠤⠠⠠⠞⠑⠟⠸⠌⠍⠘⠼⠉⠪⠐⠥⠐⠀⠰⠽⠈⠵⠀` + - actual: `⠓⠕⠤⠠⠠⠞⠑⠟⠲⠸⠌⠍⠘⠼⠉⠪⠐⠥⠐⠀⠰⠽⠈⠵` + - first differing cell (zero-based): 92 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `roman_hyphenated_word_after_whitespace_following_korean_word` + +Of the 361 candidates, 73 are the actual `pending_rule_review` subcluster. The other 288 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 84 mismatches were evaluable and 6 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2834 ⠴ -> U+2800 ⠀`: 6 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 11 +- `pending_rule_review`: 73 + +Representative `exact` samples: + +- `sentence_01.json` #52: 과학기술정보통신부와 개인정보보호위원회의 공동 고시 기준에 따라 한국인터넷진흥원(KISA)이 인증하는 ISMS-P는 고객 정보보호 및 개인정보보호를 위한 일련의 조치와 활동이 국가공인 인증기준에 적합함을 증명하는 제도다. + - expected: `⠈⠧⠚⠁⠈⠕⠠⠯⠨⠻⠘⠥⠓⠿⠠⠟⠘⠍⠧⠀⠈⠗⠟⠨` + - actual: `⠈⠧⠚⠁⠈⠕⠠⠯⠨⠻⠘⠥⠓⠿⠠⠟⠘⠍⠧⠀⠈⠗⠟⠨` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #344: 한편 이장우 대전시장은 이틀간의 대만 방문을 마무리하고 싱가포르로 이동해 18일까지 Merck(머크) 사, 국립싱가포르대학 바이오연구단, A-STAR 및 바이오폴리스 등 바이오산업 관련 해외 기업과 연구기관을 방문할 예정이다. + - expected: `⠚⠒⠙⠡⠀⠕⠨⠶⠍⠀⠊⠗⠨⠾⠠⠕⠨⠶⠵⠀⠕⠓⠮⠫` + - actual: `⠚⠒⠙⠡⠀⠕⠨⠶⠍⠀⠊⠗⠨⠾⠠⠕⠨⠶⠵⠀⠕⠓⠮⠫` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #569: 새울원전 관계자는 “APR 1400의 유럽 수출 노형인 EU-APR은 유럽사업자요건(EUR) 인증을 취득했고, 미국 이외 국가로는 처음으로 미국 원자력규제위원회 설계인증을 받아 안전성과 기술력을 국제적으로 인증받고 있다”고 밝혔다. + - expected: `⠠⠗⠯⠏⠒⠨⠾⠀⠈⠧⠒⠈⠌⠨⠉⠵⠀⠦⠴⠠⠠⠁⠏⠗` + - actual: `⠠⠗⠯⠏⠒⠨⠾⠀⠈⠧⠒⠈⠌⠨⠉⠵⠀⠦⠴⠠⠠⠁⠏⠗` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #937: 혼다 CR-V는 예상보다 크고 강한 차다. 준중형 스포츠유틸리티차(SUV)로 규정되지만, 동급에선 공간도 넉넉하고 충분한 힘도 갖췄다. 6년 만에 6세대 완전변경 모델로 힘과 덩치를 모두 키워서 돌아왔다. + - expected: `⠚⠷⠊⠀⠴⠠⠠⠉⠗⠤⠰⠠⠧⠲⠉⠵⠀⠌⠇⠶⠘⠥⠊⠀` + - actual: `⠚⠷⠊⠀⠴⠠⠠⠉⠗⠤⠰⠠⠧⠲⠉⠵⠀⠌⠇⠶⠘⠥⠊⠀` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_02.json` #10082: SK텔레콤이 무선 네트워크 품질 관리 AI(인공지능) 솔루션인 A-STAR(Access-Infra Service for Targeting & Action Recommendation)를 개발해 자사 전국 기지국에 적용했다고 28일 밝혔다. + - expected: `⠥⠂⠐⠍⠠⠡⠟⠀⠴⠠⠁⠤⠠⠠⠌⠜⠐⠣⠠⠁⠒⠑⠎⠎` + - actual: `⠥⠂⠐⠍⠠⠡⠟⠀⠀⠠⠁⠔⠠⠠⠎⠞⠁⠗⠦⠠⠁⠉⠉⠑` + - first differing cell (zero-based): 69 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #5926: 랩지노믹스는 자체 개발한 암 진단검사 서비스 ‘캔서스캔(CancerSCAN)’을 비롯해 국내 최초로 NGS-NIPT(Non-Invasive Prenatal Test·비침습 산전선별검사) 서비스 ‘맘가드(MomGuard)’를 공급하고 있다. + - expected: `⠀⠰⠽⠰⠥⠐⠥⠀⠴⠠⠠⠝⠛⠎⠤⠠⠠⠝⠊⠏⠞⠦⠄⠴` + - actual: `⠀⠰⠽⠰⠥⠐⠥⠀⠀⠠⠠⠝⠛⠎⠔⠠⠠⠝⠊⠏⠞⠦⠠⠝` + - first differing cell (zero-based): 98 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #149: 탠덤 OLED를 탄성있는 플라스틱 기판에 결합한 것이 LG디스플레이의 차량용 플라스틱(P)-OLED다. 차량용 P-OLED는 LCD 대비 소비전력을 60% 줄이고, 무게는 80%나 저감해 전기차 시대에도 적합한 디스플레이라는 평가다. + - expected: `⠮⠐⠣⠠⠪⠓⠕⠁⠴⠐⠣⠠⠏⠐⠜⠤⠠⠠⠕⠇⠫⠲⠊⠲` + - actual: `⠮⠐⠣⠠⠪⠓⠕⠁⠦⠄⠴⠠⠏⠠⠴⠤⠴⠠⠠⠕⠇⠫⠲⠊` + - first differing cell (zero-based): 87 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #420: 대전대학교(총장 남상호) LINC3.0사업단(단장 이영환)은 지난 6일 특화 분야 ICC(기업협업) 참여학과 중심의 기술개발 프로그램인 ‘2023 혜화 All-SET 기술사업화’ 운영을 위한 선정평가를 진행했다고 7일 밝혔다. + - expected: `⠥⠠⠴⠀⠴⠠⠠⠇⠔⠉⠼⠉⠲⠚⠇⠎⠃⠊⠒⠦⠄⠊⠒⠨` + - actual: `⠥⠠⠴⠀⠴⠠⠠⠇⠊⠝⠉⠼⠉⠲⠚⠇⠎⠃⠊⠒⠦⠄⠊⠒` + - first differing cell (zero-based): 30 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #3036: 스마트 로봇기업 RP가 일본 오웰 사와 R-BOT(알봇)의 일본 사업 진출을 위한 업무협약(MOU)을 체결했다고 23일 밝혔다. + - expected: `⠜⠁⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈⠥` + - actual: `⠜⠁⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈⠥⠀` + - first differing cell (zero-based): 92 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #1528: 서부발전은 18일(현지시간) 오만에서 오만수전력조달공사(OPWP)가 주최한 ‘오만 마나 500㎿ 태양광발전 계약 서명식’에 파트너사인 프랑스 EDF-R과 함께 참석했다고 밝혔다. + - expected: `⠣⠶⠠⠪⠀⠴⠠⠠⠑⠙⠋⠤⠰⠠⠗⠲⠈⠧⠀⠚⠢⠠⠈⠝` + - actual: `⠣⠶⠠⠪⠀⠴⠠⠠⠫⠋⠤⠰⠠⠗⠲⠈⠧⠀⠚⠢⠠⠈⠝⠀` + - first differing cell (zero-based): 139 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `roman_parenthetical_headword_after_whitespace_following_korean_word` + +Of the 4695 candidates, 640 are the actual `pending_rule_review` subcluster. The other 4055 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 750 mismatches were evaluable and 4 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2820 ⠠ -> U+2834 ⠴`: 2 +- `U+2834 ⠴ -> U+2800 ⠀`: 2 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 109 +- `pending_rule_review`: 640 +- `unsupported_character_review`: 1 + +Representative `exact` samples: + +- `sentence_01.json` #4: LG전자가 AI(인공지능) 전문가인 김정희 전무를 인공지능연구소 수장으로 영입해 고객 상황에 최적화된 솔루션을 먼저 제안할 수 있는 AI 기술 고도화에 박차를 가한다. + - expected: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠫⠀⠴⠠⠠⠁⠊⠦⠄⠟⠈⠿⠨⠕⠉` + - actual: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠫⠀⠴⠠⠠⠁⠊⠦⠄⠟⠈⠿⠨⠕⠉` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #47: 정 의원은 지난 2월 통계청에서 발표한 2022년 합계출산율이 0.78명으로 OECD(경제협력개발기구)에 가입한 38개국 중 유일하게 출산율 1명대 이하를 기록하는 등 초저출생 현상이 가속화되고 있다고 지적했다. + - expected: `⠨⠻⠀⠺⠏⠒⠵⠀⠨⠕⠉⠒⠀⠼⠃⠏⠂⠀⠓⠿⠈⠌⠰⠻` + - actual: `⠨⠻⠀⠺⠏⠒⠵⠀⠨⠕⠉⠒⠀⠼⠃⠏⠂⠀⠓⠿⠈⠌⠰⠻` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #39: 앞서 지난 15일 수원역에서 무궁화 열차에 탑승하려던 장애인 승객 A씨가 자신의 휠체어 탑승이 거부당한 사연이 SNS(사회관계망서비스)를 통해 확산했다. + - expected: `⠣⠲⠠⠎⠀⠨⠕⠉⠒⠀⠼⠁⠑⠕⠂⠀⠠⠍⠏⠒⠱⠁⠝⠠` + - actual: `⠣⠲⠠⠎⠀⠨⠕⠉⠒⠀⠼⠁⠑⠕⠂⠀⠠⠍⠏⠒⠱⠁⠝⠠` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #146: 인천시가 지역 내 교량과 터널에 설치된 방음시설 중 화재에 취약한 PMMA(폴리메타크릴산메틸) 소재를 불연성 재질인 유리로 교체하기로 했다. + - expected: `⠟⠰⠾⠠⠕⠫⠀⠨⠕⠱⠁⠀⠉⠗⠀⠈⠬⠐⠜⠶⠈⠧⠀⠓` + - actual: `⠟⠰⠾⠠⠕⠫⠀⠨⠕⠱⠁⠀⠉⠗⠀⠈⠬⠐⠜⠶⠈⠧⠀⠓` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_01.json` #24409: 한편 경북은 배터리 규제자유특구(2019)를 시작으로 이차전지 재사용·재활용 산업을 선점했고, 이차전지 혁신거버넌스 출범(2022.11), 이차전지 산업생태계 구축 MOU(2023.2) 등 각종 국가정책사업을 다수 유치해 이차전지 산업생태계를 완성해가고 있다. + - expected: `⠌⠀⠈⠍⠰⠍⠁⠀⠴⠠⠠⠍⠕⠥⠦⠄⠼⠃⠚⠃⠉⠲⠃⠠` + - actual: `⠌⠀⠈⠍⠰⠍⠁⠀⠀⠠⠠⠍⠕⠥⠦⠼⠃⠚⠃⠉⠲⠃⠴⠀` + - first differing cell (zero-based): 159 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #24207: 3일 업계에 따르면 삼성전자는 올 하반기 출시할 갤럭시S23 FE(팬에디션)에 자체 개발한 AP인 엑시노스2200을 투입할 예정이다. + - expected: `⠕⠴⠠⠎⠼⠃⠉⠀⠠⠠⠋⠑⠦⠄⠙⠗⠒⠝⠊⠕⠠⠡⠠⠴` + - actual: `⠕⠴⠠⠎⠼⠃⠉⠀⠴⠠⠠⠋⠑⠦⠄⠙⠗⠒⠝⠊⠕⠠⠡⠠` + - first differing cell (zero-based): 58 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #5981: 통합번호 ‘109’는 긴급 구급·구조 번호인 ‘119’와 같이 자살이 ‘구조가 필요한 긴급한 상황’이라는 인식을 줄 수 있고, ‘한 명의 생명(1)도, 자살 zero(0), 구하자(9)’라는 의미를 포함하고 있다. + - expected: `⠊⠥⠐⠀⠨⠇⠂⠀⠴⠵⠻⠕⠦⠄⠼⠚⠠⠴⠐⠀⠈⠍⠚⠨` + - actual: `⠊⠥⠐⠀⠨⠇⠂⠀⠀⠵⠑⠗⠕⠦⠼⠚⠴⠐⠀⠈⠍⠚⠨⠦` + - first differing cell (zero-based): 145 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #191: 소프트웨어 측면에서는 스마트 TV 독자 운영체제 웹(web)OS의 진화를 앞세워 맞춤형 고객경험과 CDX(Cross Device eXperience) 경험을 강화한다. + - expected: `⠰⠝⠨⠝⠀⠏⠗⠃⠴⠐⠣⠺⠑⠃⠐⠜⠠⠠⠕⠎⠲⠺⠀⠨` + - actual: `⠰⠝⠨⠝⠀⠏⠗⠃⠦⠄⠴⠺⠑⠃⠠⠴⠴⠠⠠⠕⠎⠲⠺⠀` + - first differing cell (zero-based): 48 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #14: 한국타이어앤테크놀로지㈜(대표이사 이수일, 이하 한국타이어)가 국제자동차연맹(FIA)이 주관하는 ‘FIA 주니어 ERC(FIA Junior ERC, 이하 주니어 ERC)’ 대회에 레이싱 타이어를 독점 공급한다. + - expected: `⠍⠉⠕⠎⠀⠴⠠⠠⠑⠗⠉⠦⠄⠴⠠⠠⠋⠊⠁⠀⠠⠚⠥⠝` + - actual: `⠍⠉⠕⠎⠀⠴⠠⠠⠻⠉⠦⠄⠴⠠⠠⠋⠊⠁⠀⠠⠚⠥⠝⠊` + - first differing cell (zero-based): 114 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #292: GH 주거 분야 전문인력과 HUG(주택도시보증공사), 변호사, 법무사 등 부동산·금융 전문인력이 상주하며 전세사기 피해자에게 부동산 법률, 긴급 금융지원, 주거지원 등 종합적인 맞춤형 상담을 제공한다. + - expected: `⠴⠠⠠⠛⠓⠲⠀⠨⠍⠈⠎⠀⠘⠛⠜⠀⠨⠾⠑⠛⠟⠐⠱⠁` + - actual: `⠴⠠⠠⠣⠲⠀⠨⠍⠈⠎⠀⠘⠛⠜⠀⠨⠾⠑⠛⠟⠐⠱⠁⠈` + - first differing cell (zero-based): 3 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #191: SK온은 23일 서울 종로구 SK서린사옥에서 에코프로머티리얼즈, 중국의 GEM(거린메이)과 전구체 생산을 위한 3자 합작법인인 지이엠코리아뉴에너지머티리얼즈㈜(이하 지이엠코리아) 설립 양해각서(MOU)를 체결했다고 밝혔다. + - expected: `⠠⠎⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠐⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈` + - actual: `⠠⠎⠦⠄⠴⠠⠠⠍⠳⠠⠴⠐⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈⠥` + - first differing cell (zero-based): 191 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `roman_run_after_whitespace_following_closed_roman_enclosure` + +Of the 1093 candidates, 163 are the actual `pending_rule_review` subcluster. The other 930 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 513 mismatches were evaluable and 5 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2811 ⠑ -> U+282B ⠫`: 1 +- `U+2811 ⠑ -> U+283B ⠻`: 1 +- `U+2815 ⠕ -> U+2837 ⠷`: 1 +- `U+2820 ⠠ -> U+280E ⠎`: 1 +- `U+2825 ⠥ -> U+2820 ⠠`: 1 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 349 +- `pending_rule_review`: 163 +- `unsupported_character_review`: 1 + +Representative `exact` samples: + +- `sentence_01.json` #337: ‘아리송(ARISONG)’, ‘Boyfriend’에 이어 ‘오로라’에도 인기 안무가 리정이 안무 메이킹에 참여해 기대를 모은다. 리정과 7인 7색 매력의 시그니처가 만나 어떤 색다른 퍼포먼스를 선사할지 이목이 집중된다. + - expected: `⠠⠦⠣⠐⠕⠠⠿⠦⠄⠴⠠⠠⠜⠊⠎⠰⠛⠠⠴⠴⠄⠐⠀⠠` + - actual: `⠠⠦⠣⠐⠕⠠⠿⠦⠄⠴⠠⠠⠜⠊⠎⠰⠛⠠⠴⠴⠄⠐⠀⠠` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #461: 주요 연수 내용은 교육현장의 요구를 적극적으로 반영해 인공지능(AI), ChatGTP, 영어그림책 등을 활용한 다양한 교수학습 방법, 세계시민교육 실천사례, 영미권 원어민 강사와의 협력수업 방법 등으로 구성했다. + - expected: `⠨⠍⠬⠀⠡⠠⠍⠀⠉⠗⠬⠶⠵⠀⠈⠬⠩⠁⠚⠡⠨⠶⠺⠀` + - actual: `⠨⠍⠬⠀⠡⠠⠍⠀⠉⠗⠬⠶⠵⠀⠈⠬⠩⠁⠚⠡⠨⠶⠺⠀` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #46: 삼성자산운용은 인도 시장에 투자하는 ‘KODEX 인도 Nifty50’, ‘KODEX 인도 Nifty50 레버리지’ 상장지수펀드(ETF) 2종을 21일 상장한다고 밝혔다. + - expected: `⠇⠢⠠⠻⠨⠇⠒⠛⠬⠶⠵⠀⠟⠊⠥⠀⠠⠕⠨⠶⠝⠀⠓⠍` + - actual: `⠇⠢⠠⠻⠨⠇⠒⠛⠬⠶⠵⠀⠟⠊⠥⠀⠠⠕⠨⠶⠝⠀⠓⠍` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #298: HMM은 카타르 하마드에서 당초 수량보다 16개를 추가한 586개의 임시주택 컨테이너를 다목적선(MPV) ‘HMM 울산호’에 선적했다. 이 선박은 27일 출항, 다음달 10일경 튀르키예 이스켄데룬에 도착할 예정이다. + - expected: `⠴⠠⠠⠓⠍⠍⠲⠵⠀⠋⠓⠐⠪⠀⠚⠑⠊⠪⠝⠠⠎⠀⠊⠶` + - actual: `⠴⠠⠠⠓⠍⠍⠲⠵⠀⠋⠓⠐⠪⠀⠚⠑⠊⠪⠝⠠⠎⠀⠊⠶` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_01.json` #17229: 김태현은 연령별 대표팀을 거치며 성장한 엘리트다. 김태현은 U-17, U-20, U-23 대표팀을 거쳤으며 지난 2020년에는 김학범 감독의 지도 속 아시아축구연맹(AFC) U-23 아시안컵에도 나섰다. + - expected: `⠓⠗⠚⠡⠵⠀⠴⠠⠥⠤⠼⠁⠛⠂⠀⠰⠠⠥⠤⠼⠃⠚⠂⠀` + - actual: `⠓⠗⠚⠡⠵⠀⠴⠠⠠⠠⠥⠤⠼⠁⠛⠐⠀⠴⠥⠤⠼⠃⠚⠐` + - first differing cell (zero-based): 58 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #15163: 삼성전자는 HDR10+ GAMING 기술을 지난해 이후에 출시된 7시리즈 이상의 오디세이 게이밍 모니터와 120Hz 이상을 지원하는 QLED 70·80시리즈, OLED, 네오(Neo) QLED 등 TV에 적용했다. + - expected: `⠉⠵⠀⠴⠠⠠⠟⠇⠑⠙⠀⠼⠛⠚⠐⠆⠼⠓⠚⠠⠕⠐⠕⠨` + - actual: `⠉⠵⠀⠴⠠⠠⠟⠇⠫⠀⠼⠛⠚⠐⠆⠼⠓⠚⠠⠕⠐⠕⠨⠪` + - first differing cell (zero-based): 120 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #4022: 예탁원은 13일 “KOFR OIS(오버나이트 인덱스 스와프) 시장 형성에 필요한 KOFR OIS 추정 금리커브와 KOFR 현물상품 출시를 위해 필요한 텀(Term) KOFR 개발을 추진한다”고 밝혔다. + - expected: `⠕⠂⠀⠦⠴⠠⠠⠅⠕⠋⠗⠀⠠⠠⠕⠊⠎⠦⠄⠥⠘⠎⠉⠣` + - actual: `⠕⠂⠀⠦⠴⠠⠠⠅⠷⠗⠀⠠⠠⠕⠊⠎⠦⠄⠥⠘⠎⠉⠣⠕` + - first differing cell (zero-based): 18 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #5928: 이번 평가에서 현대건설은 ‘유럽연합(EU) 택소노미 기반 지속가능매출산정’ ‘SBTi 승인’ ‘생물다양성 위험성 평가 실시’ ‘임직원 복지제도 확대’ 등을 우수 성과로 인정받았다. + - expected: `⠻⠴⠄⠀⠠⠦⠴⠠⠠⠎⠃⠞⠠⠄⠊⠲⠀⠠⠪⠶⠟⠴⠄⠀` + - actual: `⠻⠴⠄⠀⠠⠦⠴⠠⠎⠠⠃⠠⠞⠊⠲⠀⠠⠪⠶⠟⠴⠄⠀⠠` + - first differing cell (zero-based): 78 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #14: 새롭게 선보인 DRX 브랜딩과 팀 컬러는 2023년 1월 1일부터 DRX의 온오프라인 콘텐츠, 엠디(MD), SNS, 유튜브 등 모든 제작물에 적용될 예정이다. + - expected: `⠰⠪⠐⠀⠝⠢⠊⠕⠴⠐⠣⠠⠠⠍⠙⠐⠜⠂⠀⠠⠠⠎⠝⠎` + - actual: `⠰⠪⠐⠀⠝⠢⠊⠕⠦⠄⠴⠠⠠⠍⠙⠠⠴⠐⠀⠴⠠⠠⠎⠝` + - first differing cell (zero-based): 97 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #78: 마지막으로 신규 회원사인 ㈜그레비티(대표 최항주)에 대한 소개와 서중석 교수의 발제로 ‘글쓰기에 필요한 다양한 생성형(Generative) AI Searcher’에 대한 토론이 진행됐다. + - expected: `⠀⠠⠗⠶⠠⠻⠚⠻⠴⠐⠣⠠⠛⠢⠻⠁⠞⠊⠧⠑⠐⠜⠀⠠` + - actual: `⠀⠠⠗⠶⠠⠻⠚⠻⠦⠄⠴⠠⠛⠢⠻⠁⠞⠊⠧⠑⠠⠴⠀⠴` + - first differing cell (zero-based): 116 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_03.json` #286: 캘리포니아 레드우드 시티의 재무고문인 로렌스 폰은 “텍사스 ETF의 이름을 주를 상징하는 ‘론스타(Lone Star) ETF’ 또는 ‘리멤버 알라모(Alamo) ETF’로 명명하는 것도 괜찮을 것”이라고 말했다. + - expected: `⠀⠠⠦⠐⠷⠠⠪⠓⠴⠐⠣⠠⠇⠐⠕⠀⠠⠌⠜⠐⠜⠀⠠⠠` + - actual: `⠀⠠⠦⠐⠷⠠⠪⠓⠦⠄⠴⠠⠇⠐⠕⠀⠠⠌⠜⠠⠴⠀⠴⠠` + - first differing cell (zero-based): 91 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #62: LG디스플레이가 자사의 유기발광다이오드(OLED) TV 패널이 글로벌 친환경 인증기관인 카본 트러스트(Carbon Trust)로 부터 탄소발자국 인증을 획득했다고 19일 밝혔다. + - expected: `⠧⠶⠊⠣⠕⠥⠊⠪⠴⠐⠣⠠⠠⠕⠇⠫⠐⠜⠀⠠⠠⠞⠧⠲` + - actual: `⠧⠶⠊⠣⠕⠥⠊⠪⠦⠄⠴⠠⠠⠕⠇⠫⠠⠴⠀⠴⠠⠠⠞⠧` + - first differing cell (zero-based): 35 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` + +### `roman_run_immediately_before_attached_middle_dot_boundary` + +Of the 577 candidates, 570 are the actual `pending_rule_review` subcluster. The other 7 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 577 mismatches were evaluable and 516 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2810 ⠐ -> U+2832 ⠲`: 516 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 7 +- `pending_rule_review`: 570 + +Representative `localized_mismatch` samples: + +- `sentence_01.json` #57: 아키에이지 워는 원작 아키에이지 개발사 엑스엘게임즈(각자대표 송재경, 최관호)가 개발 중인 신작 PC·모바일 크로스플랫폼 대규모다중접속역할수행게임(MMORPG)이다. 원작의 향수를 자극하는 스토리와 캐릭터, 언리얼 엔진4를 활용한 고퀄리티 그래픽이 특징이다. + - expected: `⠨⠁⠀⠴⠠⠠⠏⠉⠐⠆⠑⠥⠘⠣⠕⠂⠀⠋⠪⠐⠥⠠⠪⠙` + - actual: `⠨⠁⠀⠴⠠⠠⠏⠉⠲⠐⠆⠑⠥⠘⠣⠕⠂⠀⠋⠪⠐⠥⠠⠪` + - first differing cell (zero-based): 92 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #10: 설동호 대전교육감은 “학교 교육과정 및 여건에 맞는 인공지능(AI)교육 선도학교 운영을 통해 AI·SW교육 기반 구축과 학생들과 교원들의 미래 디지털 역량이 함양될 것이다.”라고 말했다. + - expected: `⠚⠗⠀⠴⠠⠠⠁⠊⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕⠘` + - actual: `⠚⠗⠀⠴⠠⠠⠁⠊⠲⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕` + - first differing cell (zero-based): 93 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #91: 지난 13일 브라질에서 우수제조관리기준(BGMP) 인증을 받으며 중남미시장 진출에도 첫발을 내디뎠다. BGMP 인증으로 HA·Ca필러가 브라질을 비롯한 남미 국가들에 진입하는 데 속도가 붙을 것이란 전망이다. + - expected: `⠐⠥⠀⠴⠠⠠⠓⠁⠐⠆⠴⠠⠉⠁⠲⠙⠕⠂⠐⠎⠫⠀⠘⠪` + - actual: `⠐⠥⠀⠴⠠⠠⠓⠁⠲⠐⠆⠴⠠⠉⠁⠲⠙⠕⠂⠐⠎⠫⠀⠘` + - first differing cell (zero-based): 121 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #49: 유럽중앙은행(ECB)이 크레디트스위스(CS)의 유동성 위기에도 ‘빅스텝’(기준금리를 한 번에 0.5%포인트 인상)을 단행하자, 미국 연방준비제도(Fed·연준) 역시 금리 인상을 멈추지 않을 것이라는 전망이 힘을 얻고 있다. + - expected: `⠊⠥⠦⠄⠴⠠⠋⠫⠐⠆⠡⠨⠛⠠⠴⠀⠱⠁⠠⠕⠀⠈⠪⠢` + - actual: `⠊⠥⠦⠄⠴⠠⠋⠫⠲⠐⠆⠡⠨⠛⠠⠴⠀⠱⠁⠠⠕⠀⠈⠪` + - first differing cell (zero-based): 149 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #57: 아키에이지 워는 원작 아키에이지 개발사 엑스엘게임즈(각자대표 송재경, 최관호)가 개발 중인 신작 PC·모바일 크로스플랫폼 대규모다중접속역할수행게임(MMORPG)이다. 원작의 향수를 자극하는 스토리와 캐릭터, 언리얼 엔진4를 활용한 고퀄리티 그래픽이 특징이다. + - expected: `⠨⠁⠀⠴⠠⠠⠏⠉⠐⠆⠑⠥⠘⠣⠕⠂⠀⠋⠪⠐⠥⠠⠪⠙` + - actual: `⠨⠁⠀⠴⠠⠠⠏⠉⠲⠐⠆⠑⠥⠘⠣⠕⠂⠀⠋⠪⠐⠥⠠⠪` + - first differing cell (zero-based): 92 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #10: 설동호 대전교육감은 “학교 교육과정 및 여건에 맞는 인공지능(AI)교육 선도학교 운영을 통해 AI·SW교육 기반 구축과 학생들과 교원들의 미래 디지털 역량이 함양될 것이다.”라고 말했다. + - expected: `⠚⠗⠀⠴⠠⠠⠁⠊⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕⠘` + - actual: `⠚⠗⠀⠴⠠⠠⠁⠊⠲⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕` + - first differing cell (zero-based): 93 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #91: 지난 13일 브라질에서 우수제조관리기준(BGMP) 인증을 받으며 중남미시장 진출에도 첫발을 내디뎠다. BGMP 인증으로 HA·Ca필러가 브라질을 비롯한 남미 국가들에 진입하는 데 속도가 붙을 것이란 전망이다. + - expected: `⠐⠥⠀⠴⠠⠠⠓⠁⠐⠆⠴⠠⠉⠁⠲⠙⠕⠂⠐⠎⠫⠀⠘⠪` + - actual: `⠐⠥⠀⠴⠠⠠⠓⠁⠲⠐⠆⠴⠠⠉⠁⠲⠙⠕⠂⠐⠎⠫⠀⠘` + - first differing cell (zero-based): 121 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #49: 유럽중앙은행(ECB)이 크레디트스위스(CS)의 유동성 위기에도 ‘빅스텝’(기준금리를 한 번에 0.5%포인트 인상)을 단행하자, 미국 연방준비제도(Fed·연준) 역시 금리 인상을 멈추지 않을 것이라는 전망이 힘을 얻고 있다. + - expected: `⠊⠥⠦⠄⠴⠠⠋⠫⠐⠆⠡⠨⠛⠠⠴⠀⠱⠁⠠⠕⠀⠈⠪⠢` + - actual: `⠊⠥⠦⠄⠴⠠⠋⠫⠲⠐⠆⠡⠨⠛⠠⠴⠀⠱⠁⠠⠕⠀⠈⠪` + - first differing cell (zero-based): 149 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `rule69_ascii_unit_before_terminator_skipping_symbol` + +Of the 440 candidates, 53 are the actual `pending_rule_review` subcluster. The other 387 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 55 mismatches were evaluable and 3 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2810 ⠐ -> U+2802 ⠂`: 1 +- `U+2824 ⠤ -> U+2814 ⠔`: 1 +- `U+283C ⠼ -> U+2800 ⠀`: 1 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 2 +- `pending_rule_review`: 53 + +Representative `exact` samples: + +- `sentence_01.json` #1168: 16일 리튬플러스는 충남 금산군 추부공장에서 생산된 배터리급 초고순도 수산화리튬 1.4톤(ton)을 판매했다고 밝혔다. 지난달 31일 600kg(킬로그램)에 이은 두번째 출하다. + - expected: `⠼⠁⠋⠕⠂⠀⠐⠕⠓⠩⠢⠙⠮⠐⠎⠠⠪⠉⠵⠀⠰⠍⠶⠉` + - actual: `⠼⠁⠋⠕⠂⠀⠐⠕⠓⠩⠢⠙⠮⠐⠎⠠⠪⠉⠵⠀⠰⠍⠶⠉` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #29: 어획량이 감소하면서 지난해 12월 기준 1상자(20kg)당 위판가가 24만 원까지 치솟으면서 자원 증강 필요성이 끊임없이 제기돼 왔다. + - expected: `⠎⠚⠽⠁⠐⠜⠶⠕⠀⠫⠢⠠⠥⠚⠑⠡⠠⠎⠀⠨⠕⠉⠒⠚` + - actual: `⠎⠚⠽⠁⠐⠜⠶⠕⠀⠫⠢⠠⠥⠚⠑⠡⠠⠎⠀⠨⠕⠉⠒⠚` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #81: 이번에 개발한 신제품의 용량은 기존 제품(16GB)보다 50% 높아졌다. D램 용량이 높아지면 시스템이 데이터를 더 빠르고 효율적으로 처리할 수 있게 된다. + - expected: `⠕⠘⠾⠝⠀⠈⠗⠘⠂⠚⠒⠀⠠⠟⠨⠝⠙⠍⠢⠺⠀⠬⠶⠐` + - actual: `⠕⠘⠾⠝⠀⠈⠗⠘⠂⠚⠒⠀⠠⠟⠨⠝⠙⠍⠢⠺⠀⠬⠶⠐` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #996: 기존에 월 데이터 이용량이 60GB인 20대 고객의 경우 기존에는 6만9000원(110GB)짜리 요금제만 선택이 가능했다. 하지만 이제 6만1000원(60GB)짜리 요금제 이용이 가능해 매달 8000원을 절약할 수 있다. + - expected: `⠈⠕⠨⠷⠝⠀⠏⠂⠀⠊⠝⠕⠓⠎⠀⠕⠬⠶⠐⠜⠶⠕⠀⠼` + - actual: `⠈⠕⠨⠷⠝⠀⠏⠂⠀⠊⠝⠕⠓⠎⠀⠕⠬⠶⠐⠜⠶⠕⠀⠼` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_02.json` #13659: 위례공원 맨발 황톳길(1200m)은 7월 말 개장하며, 중앙공원 맨발 황톳길( 1200m)은 8월 초 일부 구간(500m)을 우선 개장한 뒤 9월 중 모두 개통한다. + - expected: `⠓⠥⠄⠈⠕⠂⠦⠄⠼⠁⠃⠚⠚⠴⠍⠠⠴⠵⠀⠼⠓⠏⠂⠀` + - actual: `⠓⠥⠄⠈⠕⠂⠦⠄⠀⠼⠁⠃⠚⠚⠴⠍⠠⠴⠵⠀⠼⠓⠏⠂` + - first differing cell (zero-based): 81 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #22445: 또한 라이트급(-70kg) 토너먼트에서는 ‘한국 귀화 파이터’ 난딘에르덴(남양주 팀피니쉬)과 ‘슈토 환태평양 챔피언’ 데바나 슈타로(COBRA KAI), 아르투르 솔로비예프(MFP)와 맥스 더 바디(BRAVE GYM)가 4강전에 나선다. + - expected: `⠕⠓⠪⠈⠪⠃⠦⠄⠤⠼⠛⠚⠴⠅⠛⠠⠴⠀⠓⠥⠉⠎⠑⠾` + - actual: `⠕⠓⠪⠈⠪⠃⠦⠄⠔⠼⠛⠚⠴⠅⠛⠠⠴⠀⠓⠥⠉⠎⠑⠾` + - first differing cell (zero-based): 16 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #322: 삼성전자는 5나노미터(nm, 1nm는 10억 분의 1m) 기반 신규 컨트롤러를 탑재한 PC용 고성능 비휘발성 메모리 익스프레스(NVMe) SSD ‘PM9C1a’를 양산한다고 12일 밝혔다. + - expected: `⠠⠪⠙⠪⠐⠝⠠⠪⠴⠐⠣⠠⠠⠝⠧⠍⠠⠄⠑⠐⠜⠀⠠⠠` + - actual: `⠠⠪⠙⠪⠐⠝⠠⠪⠦⠄⠴⠠⠝⠠⠧⠠⠍⠑⠠⠴⠀⠴⠠⠠` + - first differing cell (zero-based): 124 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #431: 173cm, 68kg의 다소 왜소한 체격이지만 스크럼 하프(SH) 포지션을 맡아 포워드와 백스 사이에서 날카로운 볼 연결과 공격방향을 결정하는 타고난 판단력이 강점이다. + - expected: `⠚⠙⠪⠦⠄⠴⠠⠠⠎⠓⠠⠴⠀⠙⠥⠨⠕⠠⠡⠮⠀⠑⠦⠣` + - actual: `⠚⠙⠪⠦⠄⠴⠠⠠⠩⠠⠴⠀⠙⠥⠨⠕⠠⠡⠮⠀⠑⠦⠣⠀` + - first differing cell (zero-based): 55 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #4925: 1일 삼성전자에 따르면 2023년 1월부터 7월까지 판매된 삼성 Neo QLED·QLED TV 3대 중 1대는 85형 또는 98형(247cm)으로 집계됐다. + - expected: `⠝⠑⠕⠀⠠⠠⠟⠇⠑⠙⠐⠆⠴⠠⠠⠟⠇⠑⠙⠀⠠⠠⠞⠧` + - actual: `⠝⠑⠕⠀⠠⠠⠟⠇⠫⠲⠐⠆⠴⠠⠠⠟⠇⠫⠀⠠⠠⠞⠧⠀` + - first differing cell (zero-based): 72 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #2527: 삼성전자가 선폭 2·3나노(㎚·10억분의 1m) 수준의 반도체 설계에 필요한 ‘공정설계키트(PDK)’를 국내 팹리스(반도체 설계 업체)에 제공하는 등 파운드리(위탁 생산) 생태계 키우기에 나선다. + - expected: `⠉⠉⠥⠦⠄⠴⠝⠍⠐⠆⠼⠁⠚⠹⠘⠛⠺⠀⠼⠁⠴⠍⠠⠴` + - actual: `⠉⠉⠥⠦⠄⠴⠝⠍⠲⠐⠆⠼⠁⠚⠹⠘⠛⠺⠀⠼⠁⠴⠍⠠` + - first differing cell (zero-based): 29 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `single_capital_followed_by_parenthesized_digits` + +Of the 1361 candidates, 6 are the actual `pending_rule_review` subcluster. The other 1355 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 10 mismatches were evaluable and 4 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2834 ⠴ -> U+2800 ⠀`: 4 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 4 +- `pending_rule_review`: 6 + +Representative `exact` samples: + +- `sentence_01.json` #80: 4일 경기 안성경찰서 등에 따르면 남성 A(54)씨는 지난 2일 오후 9시53분께 경기 안성의 주차장 인근에서 전처인 B(53)씨 흉기로 찔러 살해했다. + - expected: `⠼⠙⠕⠂⠀⠈⠻⠈⠕⠀⠣⠒⠠⠻⠈⠻⠰⠣⠂⠠⠎⠀⠊⠪` + - actual: `⠼⠙⠕⠂⠀⠈⠻⠈⠕⠀⠣⠒⠠⠻⠈⠻⠰⠣⠂⠠⠎⠀⠊⠪` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #186: 3일 사회관계망서비스(SNS)를 통해 유포된 동영상에는 중학생인 A(14)양이 지난달 30일 태안의 한 지하 주차장에서 B(15)양으로부터 일방적으로 폭행을 당하고 주변에 있던 학생들은 이를 웃으며 방관하는 장면이 담겼다. + - expected: `⠼⠉⠕⠂⠀⠇⠚⠽⠈⠧⠒⠈⠌⠑⠶⠠⠎⠘⠕⠠⠪⠦⠄⠴` + - actual: `⠼⠉⠕⠂⠀⠇⠚⠽⠈⠧⠒⠈⠌⠑⠶⠠⠎⠘⠕⠠⠪⠦⠄⠴` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #51: 20일 한국장기조직기증원에 따르면 A(11)군은 지난 3일 학교에 가기 위해 횡단보도를 건너다가 시내버스에 치여 병원으로 이송돼 치료받았지만 회복하지 못하고 뇌사 상태에 빠졌다. + - expected: `⠼⠃⠚⠕⠂⠀⠚⠒⠈⠍⠁⠨⠶⠈⠕⠨⠥⠨⠕⠁⠈⠕⠨⠪` + - actual: `⠼⠃⠚⠕⠂⠀⠚⠒⠈⠍⠁⠨⠶⠈⠕⠨⠥⠨⠕⠁⠈⠕⠨⠪` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #1629: 23일 법조계에 따르면 전주지법 제12형사부(김도형 부장판사)는 살인, 공갈, 성매매 알선 행위 등 처벌에 관한 법률 위반 혐의로 기소된 A(28)씨에게 징역 17년을 선고했다. + - expected: `⠼⠃⠉⠕⠂⠀⠘⠎⠃⠨⠥⠈⠌⠝⠀⠠⠊⠐⠪⠑⠡⠀⠨⠾` + - actual: `⠼⠃⠉⠕⠂⠀⠘⠎⠃⠨⠥⠈⠌⠝⠀⠠⠊⠐⠪⠑⠡⠀⠨⠾` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_01.json` #9855: 부부인 A·B씨는 지난해 5월 14일부터 17일 사이 전남 여수에서 모텔을 운영하면서 딸 C(37)씨가 지적장애를 앓는 이모 D(60)씨를 폭행해 사망에 이를 때까지 방치한 혐의로 기소됐다. + - expected: `⠍⠘⠍⠟⠀⠴⠠⠁⠐⠆⠴⠠⠃⠲⠠⠠⠕⠉⠵⠀⠨⠕⠉⠒` + - actual: `⠍⠘⠍⠟⠀⠴⠠⠁⠲⠐⠆⠴⠠⠃⠲⠠⠠⠕⠉⠵⠀⠨⠕⠉` + - first differing cell (zero-based): 9 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #16790: 서울 수서경찰서는 폐쇄회로(CC)TV 등을 토대로 용의자 동선을 추적, 이날 오후 2시쯤 서울 강동구 성내동 주거지에서 A(42)씨를 체포했다. + - expected: `⠌⠠⠧⠗⠚⠽⠐⠥⠴⠐⠣⠠⠠⠉⠉⠐⠜⠠⠠⠞⠧⠲⠀⠊` + - actual: `⠌⠠⠧⠗⠚⠽⠐⠥⠦⠄⠴⠠⠠⠉⠉⠠⠴⠴⠠⠠⠞⠧⠲⠀` + - first differing cell (zero-based): 27 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_03.json` #639: 2일 법조계 등에 따르면 서울중앙지법 형사 18단독 이준구 판사는 지난달 26일 업무상 과실치상 혐의로 재판에 넘겨진 업주 A(64)에 대해 무죄를 선고했다. + - expected: `⠨⠟⠀⠎⠃⠨⠍⠀⠴⠠⠁⠦⠄⠼⠋⠙⠠⠴⠝⠀⠊⠗⠚⠗` + - actual: `⠨⠟⠀⠎⠃⠨⠍⠀⠀⠁⠦⠼⠋⠙⠴⠀⠀⠝⠀⠊⠗⠚⠗⠀` + - first differing cell (zero-based): 120 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #3330: 부산지검 마약범죄특별수사팀(팀장 박성민 강력범죄수사부장)은 시가 216억 상당의 마약류를 태국에서 밀반입한 혐의로 총책 A씨(31)와 운반책 B(31), C(30)씨를 구속기소했다고 10일 밝혔다. + - expected: `⠀⠛⠘⠒⠰⠗⠁⠀⠴⠠⠃⠦⠄⠼⠉⠁⠠⠴⠐⠀⠴⠠⠉⠦` + - actual: `⠀⠛⠘⠒⠰⠗⠁⠀⠀⠃⠦⠼⠉⠁⠴⠐⠀⠴⠠⠉⠦⠄⠼⠉` + - first differing cell (zero-based): 146 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `spaced_comma_between_ascii_digit_runs` + +Of the 217 candidates, 22 are the actual `pending_rule_review` subcluster. The other 195 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 22 mismatches were evaluable and 1 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2802 ⠂ -> U+2810 ⠐`: 1 + +Mismatch primary-class distribution: + +- `pending_rule_review`: 22 + +Representative `exact` samples: + +- `sentence_01.json` #498: 신제품은 15.6인치(15Z90RT) 울트라슬림과 14인치(14Z90RS)·16인치(16Z90RS) 그램 스타일 등으로 구성된다. 아울러 그램 17, 16, 15, 14 등도 선보일 예정이다. + - expected: `⠠⠟⠨⠝⠙⠍⠢⠵⠀⠼⠁⠑⠲⠋⠟⠰⠕⠦⠄⠼⠁⠑⠴⠠` + - actual: `⠠⠟⠨⠝⠙⠍⠢⠵⠀⠼⠁⠑⠲⠋⠟⠰⠕⠦⠄⠼⠁⠑⠴⠠` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #442: 시는 총사업비 59억 원을 투입해 1, 2단계 사업을 완료했으며, 부지면적 6만 3천㎡에 산악 지형용 자전거(MTB) 연습을 위한 펌프트랙 5개의 코스와 조명시설을 갖춘 축구장 2면을 조성했다. + - expected: `⠠⠕⠉⠵⠀⠰⠿⠇⠎⠃⠘⠕⠀⠼⠑⠊⠹⠀⠏⠒⠮⠀⠓⠍` + - actual: `⠠⠕⠉⠵⠀⠰⠿⠇⠎⠃⠘⠕⠀⠼⠑⠊⠹⠀⠏⠒⠮⠀⠓⠍` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #3215: 현재경기판단(69, 5포인트), 향후경기전망(78, 4포인트)의 상승 폭이 상대적으로 컸다. 또 소비자지출전망(113)도 2포인트 올랐다. 생활형편전망(93)과 가계수입전망(98), 현재생활형편(89)은 각각 1포인트 상승했다. + - expected: `⠚⠡⠨⠗⠈⠻⠈⠕⠙⠒⠊⠒⠦⠄⠼⠋⠊⠐⠀⠼⠑⠀⠙⠥` + - actual: `⠚⠡⠨⠗⠈⠻⠈⠕⠙⠒⠊⠒⠦⠄⠼⠋⠊⠐⠀⠼⠑⠀⠙⠥` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #299: 국책연구기관인 한국환경연구원(KEI)이 한국수력원자력(한수원)이 낸 ‘신한울 원전 3, 4호기 환경영향평가 재협의 초안’에 부정적인 의견을 밝혔다. KEI는 한수원 조사에서 해산어류(바닷물고기류) 영향이 ‘매우 형식적’으로 이뤄졌다고 지적했다. + - expected: `⠈⠍⠁⠰⠗⠁⠡⠈⠍⠈⠕⠈⠧⠒⠟⠀⠚⠒⠈⠍⠁⠚⠧⠒` + - actual: `⠈⠍⠁⠰⠗⠁⠡⠈⠍⠈⠕⠈⠧⠒⠟⠀⠚⠒⠈⠍⠁⠚⠧⠒` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_03.json` #19569: 배우 겸 2PM 멤버 황찬성이 AK-69, 2AM 이창민과 함께 부른 ‘인 투 더 파이어(Into the fire)’가 일본 TV 애니메이션 ‘리:몬스터(Re:Monster)’의 오프닝 주제가로 선정됐다. + - expected: `⠠⠠⠁⠅⠤⠼⠋⠊⠂⠀⠼⠃⠠⠠⠁⠍⠲⠀⠕⠰⠣⠶⠑⠟` + - actual: `⠠⠠⠁⠅⠤⠼⠋⠊⠐⠀⠼⠃⠴⠠⠠⠁⠍⠲⠀⠕⠰⠣⠶⠑` + - first differing cell (zero-based): 42 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #11298: 4세대 대표 아이돌 그룹 에스파(aespa)의 첫 단독 리얼리티 ‘에스파의 싱크로드(연출 진선미 제작 SM C&C STUDIO)’ 5, 6회가 오는 11일 오전 11시 웨이브(Wavve)에서 독점 공개된다. + - expected: `⠉⠀⠌⠥⠙⠊⠕⠠⠄⠠⠴⠴⠄⠀⠼⠑⠐⠀⠼⠋⠀⠚⠽⠫` + - actual: `⠉⠀⠌⠥⠙⠊⠕⠠⠴⠠⠄⠴⠄⠀⠼⠑⠐⠀⠼⠋⠀⠚⠽⠫` + - first differing cell (zero-based): 111 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #537: 11일 Macker와 ZAYDA, 12에는 Killa Ton과 Bagagee Viphex13, 13일에는 EDM씬의 트렌드를 주도하는 쥬디(JOODY)와 아쉬코(ASHIKO)가 출연했다. + - expected: `⠂⠝⠉⠵⠀⠴⠠⠠⠑⠙⠍⠲⠠⠠⠟⠺⠀⠓⠪⠐⠝⠒⠊⠪` + - actual: `⠂⠝⠉⠵⠀⠴⠠⠠⠫⠍⠲⠠⠠⠟⠺⠀⠓⠪⠐⠝⠒⠊⠪⠐` + - first differing cell (zero-based): 83 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #13832: 이어 르세라핌은 오늘 정오 글로벌 팬 커뮤니티 플랫폼 위버스(Weverse)와 공식 SNS 채널을 통해 오는 3월 18, 19일 양일간 개최되는 팬미팅에 대한 자세한 정보를 공개했다. + - expected: `⠎⠠⠪⠦⠄⠴⠠⠺⠐⠑⠎⠑⠠⠴⠧⠀⠈⠿⠠⠕⠁⠀⠴⠠` + - actual: `⠎⠠⠪⠦⠄⠴⠠⠺⠑⠧⠻⠎⠑⠠⠴⠧⠀⠈⠿⠠⠕⠁⠀⠴` + - first differing cell (zero-based): 62 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #7083: 경기도교육청(교육감 임태희)이 코로나19 장기화로 학습, 신체 건강, 사회성, 심리·정서 등 결손이 발생한 초등 3, 4학년의 개별 맞춤형 성장을 지원하는 ‘더(T?H?E) 자람 프로젝트’의 현장 안착을 지원한다. + - expected: `⠊⠎⠦⠄⠴⠠⠞⠦⠠⠓⠦⠠⠑⠠⠴⠀⠨⠐⠣⠢⠀⠙⠪⠐` + - actual: `⠊⠎⠦⠄⠴⠠⠞⠦⠰⠠⠓⠦⠰⠠⠑⠠⠴⠀⠨⠐⠣⠢⠀⠙` + - first differing cell (zero-based): 160 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `standalone_multi_character_uppercase_roman_word` + +Of the 62411 candidates, 5659 are the actual `pending_rule_review` subcluster. The other 56752 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +Mismatch primary-class distribution: + +- `corpus_suspect`: 1197 +- `pending_rule_review`: 5659 +- `unsupported_character_review`: 4 + +Representative `exact` samples: + +- `sentence_01.json` #1: LG전자는 ‘업(UP)가전’을 무기로 내세우고 있다. 이번 CES에서도 LG 씽큐 앱에서 터치만으로 제품 색상을 바꿀 수 있는 무드업 냉장고를 비롯한 다양한 업가전을 선보일 예정이다. + - expected: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠉⠵⠀⠠⠦⠎⠃⠦⠄⠴⠠⠠⠥⠏⠠` + - actual: `⠴⠠⠠⠇⠛⠲⠨⠾⠨⠉⠵⠀⠠⠦⠎⠃⠦⠄⠴⠠⠠⠥⠏⠠` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #1: 세종시 주최, 세종시탄소중립지원센터 주관으로 열린 이번 포럼의 주요의제는 기업의 이에스지(ESG) 경영과 사회적 책임, 국내 온실가스 배출권거래제 현황과 대응 방안을 마련하기 위해 개최됐다. + - expected: `⠠⠝⠨⠿⠠⠕⠀⠨⠍⠰⠽⠐⠀⠠⠝⠨⠿⠠⠕⠓⠒⠠⠥⠨` + - actual: `⠠⠝⠨⠿⠠⠕⠀⠨⠍⠰⠽⠐⠀⠠⠝⠨⠿⠠⠕⠓⠒⠠⠥⠨` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #2: 연구소는 전남대학교, 순천대학교, 전남도립대학교, 동신대학교, 한국에너지공과대학교, 충남대학교, 부산대학교 등 드론, 빅데이터, 인공지능(AI) 관련 대학 교수와 전남테크노파크, ㈜엘시스 등 전문가로 기획연구팀을 구성한다. + - expected: `⠡⠈⠍⠠⠥⠉⠵⠀⠨⠾⠉⠢⠊⠗⠚⠁⠈⠬⠐⠀⠠⠛⠰⠾` + - actual: `⠡⠈⠍⠠⠥⠉⠵⠀⠨⠾⠉⠢⠊⠗⠚⠁⠈⠬⠐⠀⠠⠛⠰⠾` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #1: 일본 정부는 이날 국가안전보장회의(NSC)를 열어 대응 방침을 논의했다. 이어 북한의 미사일 발사에 대해 “엄중히 항의하고 강하게 비난한다”고 밝혔다. + - expected: `⠕⠂⠘⠷⠀⠨⠻⠘⠍⠉⠵⠀⠕⠉⠂⠀⠈⠍⠁⠫⠣⠒⠨⠾` + - actual: `⠕⠂⠘⠷⠀⠨⠻⠘⠍⠉⠵⠀⠕⠉⠂⠀⠈⠍⠁⠫⠣⠒⠨⠾` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_01.json` #14: 새롭게 선보인 DRX 브랜딩과 팀 컬러는 2023년 1월 1일부터 DRX의 온오프라인 콘텐츠, 엠디(MD), SNS, 유튜브 등 모든 제작물에 적용될 예정이다. + - expected: `⠰⠪⠐⠀⠝⠢⠊⠕⠴⠐⠣⠠⠠⠍⠙⠐⠜⠂⠀⠠⠠⠎⠝⠎` + - actual: `⠰⠪⠐⠀⠝⠢⠊⠕⠦⠄⠴⠠⠠⠍⠙⠠⠴⠐⠀⠴⠠⠠⠎⠝` + - first differing cell (zero-based): 97 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #10: 설동호 대전교육감은 “학교 교육과정 및 여건에 맞는 인공지능(AI)교육 선도학교 운영을 통해 AI·SW교육 기반 구축과 학생들과 교원들의 미래 디지털 역량이 함양될 것이다.”라고 말했다. + - expected: `⠚⠗⠀⠴⠠⠠⠁⠊⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕⠘` + - actual: `⠚⠗⠀⠴⠠⠠⠁⠊⠲⠐⠆⠴⠠⠠⠎⠺⠲⠈⠬⠩⠁⠀⠈⠕` + - first differing cell (zero-based): 93 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #1: 김 지사는 18일(이하 현지시각) 미국 코네티컷주 댄버리 린데 본사에서 산지브 람바(Sanjiv Lamba) 린데 회장, 성백석 린데코리아 회장, 조일교 아산시 부시장과 투자양해각서(MOU)을 체결했다. + - expected: `⠠⠎⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - actual: `⠠⠎⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - first differing cell (zero-based): 175 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #19: 이날 행사에는 구 대표 외에 권봉석 (주)LG 최고운영책임자(COO), 박일평 LG사이언스파크대표(사장)를 비롯해 각 계열사 최고기술책임자(CTO), 최고데이터책임자(CDO), 최고인사책임자(CHO) 등이 참석해 국내 이공계 R&D 인재 400여명과 만났다. + - expected: `⠁⠕⠢⠨⠦⠄⠴⠠⠠⠉⠓⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢⠠` + - actual: `⠁⠕⠢⠨⠦⠄⠴⠠⠉⠠⠓⠠⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢` + - first differing cell (zero-based): 188 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `tight_triangle_mark_immediately_before_korean` + +Of the 377 candidates, 54 are the actual `pending_rule_review` subcluster. The other 323 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 60 mismatches were evaluable and 3 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+280A ⠊ -> U+2800 ⠀`: 1 +- `U+2818 ⠘ -> U+2800 ⠀`: 1 +- `U+2829 ⠩ -> U+2800 ⠀`: 1 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 6 +- `pending_rule_review`: 54 + +Representative `exact` samples: + +- `sentence_01.json` #3907: 웹케시그룹은 △청구·결제·수납 솔루션 ‘위빌(WeBILL)’ △글로벌 중견·대기업 자금관리 솔루션 ‘위엠비에이(WeMBA)’ △전자세금계산서 발행 솔루션 ‘위택스(WeTAX)’ △글로벌 통합 자금관리 시스템 ‘위지엠비에이(WeGMBA)’ 등의 글로벌 전략 상품도 순차적으로 출시 예정이다. + - expected: `⠏⠗⠃⠋⠝⠠⠕⠈⠪⠐⠍⠃⠵⠀⠸⠬⠀⠰⠻⠈⠍⠐⠆⠈` + - actual: `⠏⠗⠃⠋⠝⠠⠕⠈⠪⠐⠍⠃⠵⠀⠸⠬⠀⠰⠻⠈⠍⠐⠆⠈` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #17: 시범사업은 서해안권·백제문화권·서부내륙권을 대표하는 4개 시군의 관광자원 특성을 반영한 △문화치유 △해양치유 △마을맞춤 △엠지(MZ)맞춤 등 유형별 코스를 집중 발굴해 워케이션 상품을 기획했다. + - expected: `⠠⠕⠘⠎⠢⠇⠎⠃⠵⠀⠠⠎⠚⠗⠣⠒⠈⠏⠒⠐⠆⠘⠗⠁` + - actual: `⠠⠕⠘⠎⠢⠇⠎⠃⠵⠀⠠⠎⠚⠗⠣⠒⠈⠏⠒⠐⠆⠘⠗⠁` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #244: 올해 모집 분야는 △에듀테크&콘텐츠 △라이프스타일 △정보통신기술(ICT)&디지털 기반 혁신기술 등이다. 시리즈A 단계까지 법인 등록 스타트업이면 지원할 수 있다. + - expected: `⠥⠂⠚⠗⠀⠑⠥⠨⠕⠃⠀⠘⠛⠜⠉⠵⠀⠸⠬⠀⠝⠊⠩⠓` + - actual: `⠥⠂⠚⠗⠀⠑⠥⠨⠕⠃⠀⠘⠛⠜⠉⠵⠀⠸⠬⠀⠝⠊⠩⠓` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #7059: 현재 남양주시의 놀이체험시설은 △놀자람(화도) △까꿍놀이터(진접) △도르르(호평) △북(Book)놀이터(별내) △아이꿈놀이터(와부) 총 5개소로, 기존에 무료로 운영 중인 아이꿈놀이터를 제외한 유료 시설 4개소에 대해 무료 서비스가 제공된다. + - expected: `⠚⠡⠨⠗⠀⠉⠢⠜⠶⠨⠍⠠⠕⠺⠀⠉⠥⠂⠕⠰⠝⠚⠎⠢` + - actual: `⠚⠡⠨⠗⠀⠉⠢⠜⠶⠨⠍⠠⠕⠺⠀⠉⠥⠂⠕⠰⠝⠚⠎⠢` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_01.json` #6407: 정부도 이에 대응해 △Upper-mid 대역(7~24GHz) 기술 △커버리지 확대 기술 △소프트웨어(SW) 중심 네트워크 △에너지 절감 △공급망 안보 강화 등 5대 분야에 대해 기술개발을 추진한다. + - expected: `⠊⠙⠲⠀⠊⠗⠱⠁⠦⠄⠼⠛⠈⠔⠼⠃⠙⠴⠠⠛⠠⠓⠵⠠` + - actual: `⠊⠙⠲⠀⠊⠗⠱⠁⠀⠀⠦⠼⠛⠈⠔⠼⠃⠙⠠⠠⠛⠓⠵⠴` + - first differing cell (zero-based): 36 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #672: 충남교육청은 사무행정(태안여고), ERP(천안여자상업고) 및 대회홍보크리에이터(논산여자상업고) 3개 종목에서 대상인 교육부장관상을 받았고 △금상 6개 △은상 16개 △동상 23개 등 총 48개 수상의 영예를 누렸다. + - expected: `⠥⠠⠴⠐⠀⠴⠠⠠⠑⠗⠏⠦⠄⠰⠾⠣⠒⠱⠨⠇⠶⠎⠃⠈` + - actual: `⠥⠠⠴⠐⠀⠴⠠⠠⠻⠏⠦⠄⠰⠾⠣⠒⠱⠨⠇⠶⠎⠃⠈⠥` + - first differing cell (zero-based): 37 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #2221: 개선사업에는 △교량 상징조형물 및 배면·교각 이미지 연출 발광다이오드(LED)조명 △상징조형물 내부 은하수 조명 △기상전광판 4개 △상부 레이저빔 등이 설치됐다. + - expected: `⠊⠪⠦⠄⠴⠠⠠⠇⠑⠙⠠⠴⠨⠥⠑⠻⠀⠸⠬⠀⠇⠶⠨⠕` + - actual: `⠊⠪⠦⠄⠴⠠⠠⠇⠫⠠⠴⠨⠥⠑⠻⠀⠸⠬⠀⠇⠶⠨⠕⠶` + - first differing cell (zero-based): 74 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #7425: 모집 분야는 △창업 성장 및 일자리 지원 사업 △청년 메이커(maker) 지원 사업 △온라인플랫폼 활성화 지원 사업 △청년 창업 공간지원(s/w, h/w) △워케이션 지원 사업이다. + - expected: `⠄⠴⠎⠸⠌⠺⠂⠀⠓⠸⠌⠺⠠⠴⠀⠸⠬⠀⠏⠋⠝⠕⠠⠡` + - actual: `⠄⠴⠎⠸⠌⠺⠂⠀⠴⠓⠴⠸⠌⠺⠠⠴⠀⠸⠬⠀⠏⠋⠝⠕` + - first differing cell (zero-based): 141 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `uppercase_alphanumeric_roman_digit_sequence` + +Of the 3429 candidates, 462 are the actual `pending_rule_review` subcluster. The other 2967 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 568 mismatches were evaluable and 9 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2834 ⠴ -> U+2800 ⠀`: 9 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 106 +- `pending_rule_review`: 462 + +Representative `exact` samples: + +- `sentence_01.json` #17: 지난달 30일(현지시각) 뉴욕증권거래소(NYSE)에서 다우존스30산업평균지수는 전 거래일보다 73.55포인트(0.22%) 하락한 3만3147.25로 마감했다. 대형주 중심의 S&P500지수는 9.78포인트(0.25%) 하락한 3839.50으로, 기술주 중심의 나스닥지수는 11.61포인트(0.11%) 하락한 1만0466.4 + - expected: `⠨⠕⠉⠒⠊⠂⠀⠼⠉⠚⠕⠂⠦⠄⠚⠡⠨⠕⠠⠕⠫⠁⠠⠴` + - actual: `⠨⠕⠉⠒⠊⠂⠀⠼⠉⠚⠕⠂⠦⠄⠚⠡⠨⠕⠠⠕⠫⠁⠠⠴` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #167: 2023년 제1회 상담사례 워크숍은 정신건강 임상심리사인 김한우 수퍼바이저(월덴3 아카데미 대표)가 ‘기질 및 성격검사(TCI), 미네소타 다면적 인성 검사(MMPI-2), 문장완성검사(SCT) 활용을 위한 심리평가 슈퍼비전’이라는 주제로 진행하였다. + - expected: `⠼⠃⠚⠃⠉⠀⠉⠡⠀⠨⠝⠼⠁⠀⠚⠽⠀⠇⠶⠊⠢⠇⠐⠌` + - actual: `⠼⠃⠚⠃⠉⠀⠉⠡⠀⠨⠝⠼⠁⠀⠚⠽⠀⠇⠶⠊⠢⠇⠐⠌` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #69: 그룹 아스트로 멤버 문빈(25)이 갑작스럽게 세상을 떠난 데 대한 연예계 추모 물결이 이어지고 있는 가운데, 그가 생전 활약했던 주요 무대인 KBS2 ‘뮤직뱅크’ 측도 애도에 동참한다. + - expected: `⠈⠪⠐⠍⠃⠀⠣⠠⠪⠓⠪⠐⠥⠀⠑⠝⠢⠘⠎⠀⠑⠛⠘⠟` + - actual: `⠈⠪⠐⠍⠃⠀⠣⠠⠪⠓⠪⠐⠥⠀⠑⠝⠢⠘⠎⠀⠑⠛⠘⠟` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #41: 마쓰노 장관은 5월 히로시마 주요 7개국(G7) 정상회의에 윤 대통령을 초청할 것인지에 대해서는 “초청국에 대해서는 현재 검토 중이며 아무것도 결정되지 않았다”고 밝혔다. + - expected: `⠑⠠⠠⠪⠉⠥⠀⠨⠶⠈⠧⠒⠵⠀⠼⠑⠏⠂⠀⠚⠕⠐⠥⠠` + - actual: `⠑⠠⠠⠪⠉⠥⠀⠨⠶⠈⠧⠒⠵⠀⠼⠑⠏⠂⠀⠚⠕⠐⠥⠠` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_01.json` #7679: 또한 최근 지아이노베이션은 GI-301이 임상 1a상 파트A에서 단회 투여에도 우수한 IgE 감소 효과 확인했다고 밝혔다. 더불어 유럽종양학회(ESMO)에서 GI-101(CD80-1gC4 Fc-IL2V)의 임상 1/2상 결과도 발표할 예정이다. + - expected: `⠍⠕⠠⠴⠝⠠⠎⠀⠴⠠⠠⠛⠊⠤⠼⠁⠚⠁⠐⠣⠠⠠⠉⠙` + - actual: `⠍⠕⠠⠴⠝⠠⠎⠀⠀⠠⠠⠛⠊⠔⠼⠁⠚⠁⠦⠠⠠⠉⠙⠼` + - first differing cell (zero-based): 149 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #2507: 엔비디아의 H100(67 테라플롭스)은 A100(19.5 테라플롭스)에 비해 3배 이상 높은 연산량을 제공하는 컴퓨팅 자원이다. 1 테라플롭스(TF)는 1초에 1조개의 계산을 할 수 있는 속도다. + - expected: `⠥⠃⠠⠪⠠⠴⠵⠀⠴⠠⠁⠼⠁⠚⠚⠦⠄⠼⠁⠊⠲⠑⠀⠓` + - actual: `⠥⠃⠠⠪⠠⠴⠵⠀⠀⠠⠁⠼⠁⠚⠚⠦⠼⠁⠊⠲⠑⠀⠓⠝` + - first differing cell (zero-based): 37 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #10938: 인천검단 AA13-1·2 블록 입주 예정자들이 지하주차장 붕괴 사고로 인한 입주 지연과 관련해 한국토지주택공사(LH)와 GS건설이 제시한 보상안을 24일 수용하기로 했다. + - expected: `⠰⠾⠈⠎⠢⠊⠒⠀⠴⠠⠠⠁⠁⠼⠁⠉⠤⠼⠁⠐⠆⠼⠃⠀` + - actual: `⠰⠾⠈⠎⠢⠊⠒⠀⠀⠠⠠⠁⠁⠼⠁⠉⠔⠼⠁⠐⠼⠃⠀⠀` + - first differing cell (zero-based): 9 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #1766: 삼성전자는 부품·수리 도구·설명서·동영상 등으로 구성된 ‘자가수리 프로그램’을 국내에 도입한다고 30일 밝혔다. 우선 갤럭시 S20·S21·S22 시리즈와 노트북(갤럭시북 프로 15.6인치)·고선명(HD) TV 일부 제품이 대상이다. + - expected: `⠈⠗⠂⠐⠹⠠⠕⠀⠴⠠⠎⠼⠃⠚⠐⠆⠴⠠⠎⠼⠃⠁⠐⠆` + - actual: `⠈⠗⠂⠐⠹⠠⠕⠀⠀⠠⠎⠼⠃⠚⠐⠠⠎⠼⠃⠁⠐⠠⠎⠼` + - first differing cell (zero-based): 123 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #87: 4일 진원생명과학은 자체 개발 흡인작용 피내 접종기 진덤(GeneDerm)을 이용한 코로나19 DNA 백신(GLS-5310)의 안전성·면역원성에 대한 임상1상 결과를 국제감염병학회 (ISID) 학술지인 국제감염질환저널에 발표했다고 4일 밝혔다. + - expected: `⠢⠘⠻⠚⠁⠚⠽⠀⠦⠄⠴⠠⠠⠊⠎⠊⠙⠠⠴⠀⠚⠁⠠⠯` + - actual: `⠢⠘⠻⠚⠁⠚⠽⠀⠴⠐⠣⠠⠠⠊⠎⠊⠙⠐⠜⠲⠀⠚⠁⠠` + - first differing cell (zero-based): 172 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #324: 남녀단식 TT1 ~ TT10(지체), T11(지적), DF(청각) 등 12개 세부 종목에서 랭킹 포인트 60점을 걸고 승부를 펼친 결과 총 24명의 우승자가 탄생했다. + - expected: `⠀⠴⠠⠠⠞⠞⠼⠁⠈⠔⠠⠠⠞⠞⠼⠁⠚⠦⠄⠨⠕⠰⠝⠠` + - actual: `⠀⠴⠠⠠⠞⠞⠼⠁⠀⠈⠔⠀⠴⠠⠠⠞⠞⠼⠁⠚⠦⠄⠨⠕` + - first differing cell (zero-based): 17 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #64: SK에코플랜트는 20일 종로구 수송동 본사에서 현대엔지니어링, USNC와 ‘수소 마이크로 허브(H2 Micro Hub)’ 구축을 위한 3자간 업무협약(MOU)을 체결했다고 밝혔다. + - expected: `⠜⠁⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈⠥` + - actual: `⠜⠁⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈⠥⠀` + - first differing cell (zero-based): 148 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #850: SK하이닉스는 세계 최초로 D램 단품 칩 12개를 수직으로 적층해 최고 용량인 24기가바이트(GB) HBM3 신제품을 개발하는 데 성공했다고 20일 밝혔다. + - expected: `⠈⠕⠫⠘⠣⠕⠓⠪⠴⠐⠣⠠⠠⠛⠃⠐⠜⠀⠠⠠⠓⠃⠍⠼` + - actual: `⠈⠕⠫⠘⠣⠕⠓⠪⠦⠄⠴⠠⠠⠛⠃⠠⠴⠀⠴⠠⠠⠓⠃⠍` + - first differing cell (zero-based): 95 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` + +### `uppercase_ascii_run_immediately_after_digit_in_roman_sequence` + +Of the 1896 candidates, 272 are the actual `pending_rule_review` subcluster. The other 1624 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 327 mismatches were evaluable and 3 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2830 ⠰ -> U+2820 ⠠`: 3 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 55 +- `pending_rule_review`: 272 + +Representative `exact` samples: + +- `sentence_01.json` #25: 세븐일레븐은 올해 말 정식 버전 오픈을 목표로 하며, 이를 통해 주력상품을 구매하고 배송까지 받을 수 있는 차세대 오프라인을 위한 온라인(O4O) 시스템을 구축해 나갈 계획이다. + - expected: `⠠⠝⠘⠵⠕⠂⠐⠝⠘⠵⠵⠀⠥⠂⠚⠗⠀⠑⠂⠀⠨⠻⠠⠕` + - actual: `⠠⠝⠘⠵⠕⠂⠐⠝⠘⠵⠵⠀⠥⠂⠚⠗⠀⠑⠂⠀⠨⠻⠠⠕` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #51: 먼저, 차세대 비디오 압축표준(VVC) 분야 64건과 5G 이동통신(NR) 분야 48건 등 시장 수요가 큰 상용표준특허가 다수 포함, 향후 상당한 특허 기술료가 전망된다. + - expected: `⠑⠾⠨⠎⠐⠀⠰⠣⠠⠝⠊⠗⠀⠘⠕⠊⠕⠥⠀⠣⠃⠰⠍⠁` + - actual: `⠑⠾⠨⠎⠐⠀⠰⠣⠠⠝⠊⠗⠀⠘⠕⠊⠕⠥⠀⠣⠃⠰⠍⠁` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #47: 이수화학이 미국 과학 기술 엔지니어링 회사 KBR과 전고체 배터리 소재 황화리튬(Li2S)의 상업공정 공동 개발에 나선다. 양사는 이를 위해 공동개발 계약을 체결했다고 20일 밝혔다. + - expected: `⠕⠠⠍⠚⠧⠚⠁⠕⠀⠑⠕⠈⠍⠁⠀⠈⠧⠚⠁⠀⠈⠕⠠⠯` + - actual: `⠕⠠⠍⠚⠧⠚⠁⠕⠀⠑⠕⠈⠍⠁⠀⠈⠧⠚⠁⠀⠈⠕⠠⠯` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #78: 차체 중량은 55㎏인데, 체중 100㎏의 사람을 태울 수 있다. 배터리 출력은 1000W(와트)로, 도심 길거리에서 쓰는 킥보드와 비슷한 출력이다. + - expected: `⠰⠣⠰⠝⠀⠨⠍⠶⠐⠜⠶⠵⠀⠼⠑⠑⠴⠅⠛⠲⠟⠊⠝⠐` + - actual: `⠰⠣⠰⠝⠀⠨⠍⠶⠐⠜⠶⠵⠀⠼⠑⠑⠴⠅⠛⠲⠟⠊⠝⠐` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_03.json` #8207: 대한건축사협회가 주최하는 ‘한국건축산업대전(KAFF)’은 2006년부터 시작된 국내 최대 B2B·B2G(기업-정부간거래) 중심 건축자재·설비·기술 전문 전시회로, 올해는 관련 기업 100여개사가 참가해 코로나19 이후 최대 규모로 개최됐다. + - expected: `⠊⠗⠀⠴⠠⠃⠼⠃⠰⠠⠃⠐⠆⠴⠠⠃⠼⠃⠰⠠⠛⠦⠄⠈` + - actual: `⠊⠗⠀⠴⠠⠃⠼⠃⠠⠃⠲⠐⠆⠴⠠⠃⠼⠃⠠⠛⠦⠄⠈⠕` + - first differing cell (zero-based): 97 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #89: “인공지능(AI), 6G 등 핵심 기술을 위한 투자도 늘리는 동시에 전기차 충전, 디지털 헬스, 웹OS 기반의 콘텐츠 서비스 등 많은 영역으로 사업 포트폴리오를 확장하고 있습니다.” + - expected: `⠟⠈⠿⠨⠕⠉⠪⠶⠴⠐⠣⠠⠠⠁⠊⠐⠜⠂⠀⠼⠋⠠⠛⠲` + - actual: `⠟⠈⠿⠨⠕⠉⠪⠶⠦⠄⠴⠠⠠⠁⠊⠠⠴⠐⠀⠼⠋⠴⠠⠛` + - first differing cell (zero-based): 9 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #445: 박람회에서는 다양한 분야의 디지털 교육 프로그램을 한자리에서 체험할 수 있도록 인공지능(AI) 코스웨어·학습플랫폼, 인공지능(AI) 교과교육, 인공지능(AI) 학습지원, 3D·가상현실(VR)·메타버스 교육, 소프트웨어(SW)·코딩·로봇 교육 등 체험 공간을 운영할 예정이다. + - expected: `⠒⠐⠀⠼⠉⠴⠠⠙⠐⠆⠫⠇⠶⠚⠡⠠⠕⠂⠦⠄⠴⠠⠠⠧` + - actual: `⠒⠐⠀⠼⠉⠴⠠⠙⠲⠐⠆⠫⠇⠶⠚⠡⠠⠕⠂⠦⠄⠴⠠⠠` + - first differing cell (zero-based): 179 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #123: 이날 이네오스는 자동차 위탁 생산 업체인 마그나슈타이어와 함께 새로운 4X4 전기차(EV)를 개발한다고 발표했다. 이네오스는 신차 양산 시점을 2026년으로 목표하고 있다. + - expected: `⠠⠗⠐⠥⠛⠀⠼⠙⠡⠼⠙⠀⠨⠾⠈⠕⠰⠣⠦⠄⠴⠠⠠⠑` + - actual: `⠠⠗⠐⠥⠛⠀⠼⠙⠴⠠⠭⠼⠙⠀⠨⠾⠈⠕⠰⠣⠦⠄⠴⠠` + - first differing cell (zero-based): 62 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #115: ‘Busan is Good(부산이라 좋다)’이라는 새로운 도시 표어의 조형물을 공개하고, 3차원(3D)으로 표현한 도시상징 표지(CI) 영상을 상영한다. + - expected: `⠁⠝⠀⠊⠎⠀⠠⠛⠙⠦⠄⠘⠍⠇⠒⠕⠐⠣⠀⠨⠥⠴⠊⠠` + - actual: `⠁⠝⠀⠊⠎⠀⠠⠛⠕⠕⠙⠦⠄⠘⠍⠇⠒⠕⠐⠣⠀⠨⠥⠴` + - first differing cell (zero-based): 15 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `uppercase_ascii_run_immediately_after_hyphen_in_roman_sequence` + +Of the 952 candidates, 209 are the actual `pending_rule_review` subcluster. The other 743 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 228 mismatches were evaluable and 7 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2820 ⠠ -> U+2830 ⠰`: 5 +- `U+2830 ⠰ -> U+2820 ⠠`: 2 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 19 +- `pending_rule_review`: 209 + +Representative `exact` samples: + +- `sentence_01.json` #52: 과학기술정보통신부와 개인정보보호위원회의 공동 고시 기준에 따라 한국인터넷진흥원(KISA)이 인증하는 ISMS-P는 고객 정보보호 및 개인정보보호를 위한 일련의 조치와 활동이 국가공인 인증기준에 적합함을 증명하는 제도다. + - expected: `⠈⠧⠚⠁⠈⠕⠠⠯⠨⠻⠘⠥⠓⠿⠠⠟⠘⠍⠧⠀⠈⠗⠟⠨` + - actual: `⠈⠧⠚⠁⠈⠕⠠⠯⠨⠻⠘⠥⠓⠿⠠⠟⠘⠍⠧⠀⠈⠗⠟⠨` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #313: 순천향대(총장 김승우)는 지난 3일 한국국제협력단(KOICA)과 함께 우즈베키스탄의 수도 타슈겐트에서 스타트업 지원센터 ‘U-ENTER(Uzbekistan Entrepreneurship Innovation Center)’ 준공식을 개최했다고 밝혔다. + - expected: `⠠⠛⠰⠾⠚⠜⠶⠊⠗⠦⠄⠰⠿⠨⠶⠀⠈⠕⠢⠠⠪⠶⠍⠠` + - actual: `⠠⠛⠰⠾⠚⠜⠶⠊⠗⠦⠄⠰⠿⠨⠶⠀⠈⠕⠢⠠⠪⠶⠍⠠` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #74: 아울러 제주도개발공사는 국내 생수업계에서는 처음으로 재활용 페트(CR-PET)를 적용한 화학적 재활용 페트 ‘제주삼다수 리본(RE:Born)’을 개발하는 등 소재혁신을 통한 친환경 라인업도 확대하고있다. + - expected: `⠣⠯⠐⠎⠀⠨⠝⠨⠍⠊⠥⠈⠗⠘⠂⠈⠿⠇⠉⠵⠀⠈⠍⠁` + - actual: `⠣⠯⠐⠎⠀⠨⠝⠨⠍⠊⠥⠈⠗⠘⠂⠈⠿⠇⠉⠵⠀⠈⠍⠁` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #883: 그룹 방탄소년단(BTS) 슈가의 첫 공식 솔로 음반 ‘D-데이’(D-DAY)가 발매 당일 한터차트 기준 107만장이 넘게 팔려나가며 첫날 판매량으로는 K팝 솔로 가수 최고 기록을 경신했다. + - expected: `⠈⠪⠐⠍⠃⠀⠘⠶⠓⠒⠠⠥⠉⠡⠊⠒⠦⠄⠴⠠⠠⠃⠞⠎` + - actual: `⠈⠪⠐⠍⠃⠀⠘⠶⠓⠒⠠⠥⠉⠡⠊⠒⠦⠄⠴⠠⠠⠃⠞⠎` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_01.json` #7967: 네비웍스가 개발한 가상훈련 플랫폼 VTB-X(Virtual Training Block)을 바탕으로 만들어진 XR 훈련은 시뮬레이션, AI(인공지능) 기반 시나리오 자동 생성, 자동 평가 시스템 등이 구축됐다. + - expected: `⠀⠴⠠⠠⠧⠞⠃⠤⠠⠭⠐⠣⠠⠧⠊⠗⠞⠥⠁⠇⠀⠠⠞⠗` + - actual: `⠀⠴⠠⠠⠧⠞⠃⠤⠰⠠⠭⠐⠣⠠⠧⠊⠗⠞⠥⠁⠇⠀⠠⠞` + - first differing cell (zero-based): 41 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #5190: 이어진 2부에서는 “청년들의 시작(START)과 성장(UP)! 대전청이 응원합니다.”를 모토로 ‘종이비행기 날리기’ 국가대표의 강의와 공연이 결합된 ‘Lecture Concert’를 진행해 참관한 ‘UNI-C.O.N.그룹’ 구성원들의 뜨거운 반응을 이끌어 냈다. + - expected: `⠦⠴⠠⠠⠥⠝⠊⠤⠠⠉⠲⠠⠕⠲⠠⠝⠲⠈⠪⠐⠍⠃⠴⠄` + - actual: `⠦⠴⠠⠠⠥⠝⠊⠤⠰⠠⠉⠲⠠⠕⠲⠠⠝⠲⠈⠪⠐⠍⠃⠴` + - first differing cell (zero-based): 183 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #7591: 알파드는 다양한 주행 상황에서도 최상의 승차감을 보여주겠다고 작정한 차다. 도요타 TNGA-K(Toyota New Global Architecture-K) 플랫폼을 기반으로 차체 강성을 높였고 소음·진동(NVH) 저감 설계가 반영된 결과다. + - expected: `⠴⠠⠠⠞⠝⠛⠁⠤⠠⠅⠐⠣⠠⠞⠕⠽⠕⠞⠁⠀⠠⠝⠑⠺` + - actual: `⠴⠠⠠⠞⠝⠛⠁⠤⠰⠠⠅⠐⠣⠠⠞⠕⠽⠕⠞⠁⠀⠠⠝⠑` + - first differing cell (zero-based): 84 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #149: 탠덤 OLED를 탄성있는 플라스틱 기판에 결합한 것이 LG디스플레이의 차량용 플라스틱(P)-OLED다. 차량용 P-OLED는 LCD 대비 소비전력을 60% 줄이고, 무게는 80%나 저감해 전기차 시대에도 적합한 디스플레이라는 평가다. + - expected: `⠮⠐⠣⠠⠪⠓⠕⠁⠴⠐⠣⠠⠏⠐⠜⠤⠠⠠⠕⠇⠫⠲⠊⠲` + - actual: `⠮⠐⠣⠠⠪⠓⠕⠁⠦⠄⠴⠠⠏⠠⠴⠤⠴⠠⠠⠕⠇⠫⠲⠊` + - first differing cell (zero-based): 87 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #420: 대전대학교(총장 남상호) LINC3.0사업단(단장 이영환)은 지난 6일 특화 분야 ICC(기업협업) 참여학과 중심의 기술개발 프로그램인 ‘2023 혜화 All-SET 기술사업화’ 운영을 위한 선정평가를 진행했다고 7일 밝혔다. + - expected: `⠥⠠⠴⠀⠴⠠⠠⠇⠔⠉⠼⠉⠲⠚⠇⠎⠃⠊⠒⠦⠄⠊⠒⠨` + - actual: `⠥⠠⠴⠀⠴⠠⠠⠇⠊⠝⠉⠼⠉⠲⠚⠇⠎⠃⠊⠒⠦⠄⠊⠒` + - first differing cell (zero-based): 30 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #947: BMW코리아 미래재단이 ‘2023 서울안전한마당’에 이동식 에너지 저장소(ESS)인 ‘넥스트 그린 투-고(NEXT GREEN TO-GO)’ 부스를 마련하고 체험형 교육 프로그램을 운영한다. + - expected: `⠤⠈⠥⠦⠄⠴⠠⠠⠠⠝⠑⠭⠞⠀⠛⠗⠑⠢⠀⠞⠕⠤⠛⠠` + - actual: `⠤⠈⠥⠦⠄⠴⠠⠠⠝⠑⠭⠞⠀⠠⠠⠛⠗⠑⠢⠀⠠⠠⠞⠕` + - first differing cell (zero-based): 103 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #1528: 서부발전은 18일(현지시간) 오만에서 오만수전력조달공사(OPWP)가 주최한 ‘오만 마나 500㎿ 태양광발전 계약 서명식’에 파트너사인 프랑스 EDF-R과 함께 참석했다고 밝혔다. + - expected: `⠣⠶⠠⠪⠀⠴⠠⠠⠑⠙⠋⠤⠰⠠⠗⠲⠈⠧⠀⠚⠢⠠⠈⠝` + - actual: `⠣⠶⠠⠪⠀⠴⠠⠠⠫⠋⠤⠰⠠⠗⠲⠈⠧⠀⠚⠢⠠⠈⠝⠀` + - first differing cell (zero-based): 139 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `uppercase_ascii_segments_joined_by_ampersand_capitalization` + +Of the 439 candidates, 68 are the actual `pending_rule_review` subcluster. The other 371 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 78 mismatches were evaluable and 1 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+280E ⠎ -> U+2829 ⠩`: 1 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 10 +- `pending_rule_review`: 68 + +Representative `exact` samples: + +- `sentence_01.json` #858: 이에 따라 SK하이닉스가 설비투자(CAPEX) 규모를 올해 50%이상 감축하지만, S&P는 SK하이닉스가 투자 축소만으로 한계가 있다고 지적했다. + - expected: `⠕⠝⠀⠠⠊⠐⠣⠀⠴⠠⠠⠎⠅⠲⠚⠣⠕⠉⠕⠁⠠⠪⠫⠀` + - actual: `⠕⠝⠀⠠⠊⠐⠣⠀⠴⠠⠠⠎⠅⠲⠚⠣⠕⠉⠕⠁⠠⠪⠫⠀` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #146: KT&G(사장 백복인)가 산업정책연구원(IPS)이 주관하고 윤경 ESG 포럼이 주최하는 ‘윤경 ESG 포럼 CEO 서약식’에 참여해 윤리경영 의지를 다졌다. + - expected: `⠴⠠⠠⠅⠞⠈⠯⠠⠛⠦⠄⠇⠨⠶⠀⠘⠗⠁⠘⠭⠟⠠⠴⠫` + - actual: `⠴⠠⠠⠅⠞⠈⠯⠠⠛⠦⠄⠇⠨⠶⠀⠘⠗⠁⠘⠭⠟⠠⠴⠫` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #647: 온유는 지난 3월 6일 첫 정규 앨범 ‘써클(Circle)’로 컴백한다. 이번 앨범은 몽환적인 R&B 장르의 타이틀곡 ‘O(Circle)’(써클)을 비롯한 다채로운 분위기의 10곡으로 구성됐다. + - expected: `⠷⠩⠉⠵⠀⠨⠕⠉⠒⠀⠼⠉⠏⠂⠀⠼⠋⠕⠂⠀⠰⠎⠄⠀` + - actual: `⠷⠩⠉⠵⠀⠨⠕⠉⠒⠀⠼⠉⠏⠂⠀⠼⠋⠕⠂⠀⠰⠎⠄⠀` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #263: 김 부위원장은 투자은행(IB)의 기업 신용 공여, 합병 제도 등 기업의 M&A와 관련한 다른 제도의 불합리한 규제도 정비하고 기업구조혁신펀드도 추가로 조성하겠다고 밝혔다. + - expected: `⠈⠕⠢⠀⠘⠍⠍⠗⠏⠒⠨⠶⠵⠀⠓⠍⠨⠣⠵⠚⠗⠶⠦⠄` + - actual: `⠈⠕⠢⠀⠘⠍⠍⠗⠏⠒⠨⠶⠵⠀⠓⠍⠨⠣⠵⠚⠗⠶⠦⠄` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_02.json` #1061: SH&E 아카데미에서는 온라인과 오프라인 형태로 16개의 교육 과정이 진행된다. 화학류와 가스류, 소방안전을 비롯해 국제표준화기구(ISO) 인증 안전교육도 포함된다. + - expected: `⠴⠠⠠⠎⠓⠈⠯⠠⠑⠲⠀⠣⠋⠊⠝⠑⠕⠝⠠⠎⠉⠵⠀⠷` + - actual: `⠴⠠⠠⠩⠈⠯⠠⠑⠲⠀⠣⠋⠊⠝⠑⠕⠝⠠⠎⠉⠵⠀⠷⠐` + - first differing cell (zero-based): 3 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #1164: 위메이드의 위믹스가 국내 가상자산 거래소 코인원에 재상장되는 가운데, 자회사 블루포션게임즈가 위메이드와 P&E(Play and Earn) 사업을 위한 업무협약(MOU)을 체결한 것이 부각되고 있다. + - expected: `⠜⠁⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠒⠀⠸⠎⠕` + - actual: `⠜⠁⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠒⠀⠸⠎⠕⠀` + - first differing cell (zero-based): 142 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #455: 먼저, 세계 최고의 R&D인프라와 인력을 갖춘 장점을 활용하여 국가첨단반도체 기술센터(ASTC)를 유치하고 대전을 반도체 연구·교육·실증 거점으로 조성할 계획이다. + - expected: `⠓⠎⠦⠄⠴⠠⠠⠁⠎⠞⠉⠠⠴⠐⠮⠀⠩⠰⠕⠚⠈⠥⠀⠊` + - actual: `⠓⠎⠦⠄⠴⠠⠠⠁⠌⠉⠠⠴⠐⠮⠀⠩⠰⠕⠚⠈⠥⠀⠊⠗` + - first differing cell (zero-based): 90 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #370: 26일 GS칼텍스는 서울 강남구 GS타워에서 이승훈 GS칼텍스 S&T본부장, 박진기 HMM 총괄부사장 등이 참석한 가운데 친환경 바이오선박유 사업 업무협약(MOU)을 체결했다고 밝혔다. + - expected: `⠜⠁⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈⠥` + - actual: `⠜⠁⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠈⠥⠀` + - first differing cell (zero-based): 164 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #19: 이날 행사에는 구 대표 외에 권봉석 (주)LG 최고운영책임자(COO), 박일평 LG사이언스파크대표(사장)를 비롯해 각 계열사 최고기술책임자(CTO), 최고데이터책임자(CDO), 최고인사책임자(CHO) 등이 참석해 국내 이공계 R&D 인재 400여명과 만났다. + - expected: `⠁⠕⠢⠨⠦⠄⠴⠠⠠⠉⠓⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢⠠` + - actual: `⠁⠕⠢⠨⠦⠄⠴⠠⠉⠠⠓⠠⠕⠠⠴⠀⠊⠪⠶⠕⠀⠰⠣⠢` + - first differing cell (zero-based): 188 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `uppercase_roman_headword_closed_multiword_parenthetical` + +Of the 175 candidates, 53 are the actual `pending_rule_review` subcluster. The other 122 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +Mismatch primary-class distribution: + +- `corpus_suspect`: 9 +- `pending_rule_review`: 53 + +Representative `exact` samples: + +- `sentence_01.json` #18: 2023년형 신제품은 매터(Matter)와 HCA(Home Connectivity Alliance) 표준을 지원하는 스마트싱스 허브(SmartThings Hub)를 기반으로 다양한 기기를 자동으로 연결하고, 제어·관리할 수 있는 것이 특징이다. + - expected: `⠼⠃⠚⠃⠉⠀⠉⠡⠚⠻⠀⠠⠟⠨⠝⠙⠍⠢⠵⠀⠑⠗⠓⠎` + - actual: `⠼⠃⠚⠃⠉⠀⠉⠡⠚⠻⠀⠠⠟⠨⠝⠙⠍⠢⠵⠀⠑⠗⠓⠎` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #313: 순천향대(총장 김승우)는 지난 3일 한국국제협력단(KOICA)과 함께 우즈베키스탄의 수도 타슈겐트에서 스타트업 지원센터 ‘U-ENTER(Uzbekistan Entrepreneurship Innovation Center)’ 준공식을 개최했다고 밝혔다. + - expected: `⠠⠛⠰⠾⠚⠜⠶⠊⠗⠦⠄⠰⠿⠨⠶⠀⠈⠕⠢⠠⠪⠶⠍⠠` + - actual: `⠠⠛⠰⠾⠚⠜⠶⠊⠗⠦⠄⠰⠿⠨⠶⠀⠈⠕⠢⠠⠪⠶⠍⠠` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #6968: 홈플러스의 자체 브랜드 PB(Private Brand) 상품 200여종도 몽골 시장에 진출했다. K-푸드(Korean-Food) 열풍 전진기지로 몽골 현지 ‘서클(CIRCLE)’ 그룹이 운영하는 할인점을 택했다. + - expected: `⠚⠥⠢⠙⠮⠐⠎⠠⠪⠺⠀⠨⠰⠝⠀⠘⠪⠐⠗⠒⠊⠪⠀⠴` + - actual: `⠚⠥⠢⠙⠮⠐⠎⠠⠪⠺⠀⠨⠰⠝⠀⠘⠪⠐⠗⠒⠊⠪⠀⠴` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #2384: 마술사들의 등용문인 국제마술대회는 국내 최초로 세계마술연맹(FISM)의 인증을 받은 ‘FISM QC(Qualified Contest) BIMF’라는 이름으로 더욱 특별하게 진행된다. + - expected: `⠑⠠⠯⠇⠊⠮⠺⠀⠊⠪⠶⠬⠶⠑⠛⠟⠀⠈⠍⠁⠨⠝⠑⠠` + - actual: `⠑⠠⠯⠇⠊⠮⠺⠀⠊⠪⠶⠬⠶⠑⠛⠟⠀⠈⠍⠁⠨⠝⠑⠠` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_01.json` #168: 제주항공은 국제항공운송협회(IATA)가 주관하는 국제 항공운송 표준 감사 제도 ‘IOSA(IATA Operation Safety Audit) ISM 14th Edition’ 인증을 마쳐 세계 기준의 안전 관리시스템을 입증받았다고 6일 밝혔다. + - expected: `⠠⠊⠎⠍⠀⠼⠁⠙⠞⠓⠀⠠⠫⠊⠰⠝⠴⠄⠀⠟⠨⠪⠶⠮` + - actual: `⠠⠊⠎⠍⠀⠼⠁⠙⠐⠞⠓⠀⠠⠫⠊⠰⠝⠴⠄⠀⠟⠨⠪⠶` + - first differing cell (zero-based): 129 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #733: 아산시는 6일, 자매도시인 말레이시아 현지 최대 신선 과실류 수입업체인 CTG(Chop Tong Guan)와 농특산물 수출 확대를 위한 업무협약(MOU)을 체결했다. + - expected: `⠜⠁⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - actual: `⠜⠁⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠰⠝⠈⠳⠚⠗⠌⠊⠲` + - first differing cell (zero-based): 136 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #3330: 앞서 양사는 지난 3월 그린수소·암모니아의 원활한 생산·공급·활용을 위한 특수목적법인(SPC) 알 파탄 엘텍유브이씨 그린에너지 LLC(AL FATTAN LTechUVC Green Energy LLC)를 설립한 바 있다. + - expected: `⠟⠝⠉⠎⠨⠕⠀⠴⠰⠠⠠⠇⠇⠉⠐⠣⠰⠠⠠⠁⠇⠀⠠⠠` + - actual: `⠟⠝⠉⠎⠨⠕⠀⠴⠠⠠⠇⠇⠉⠐⠣⠰⠠⠠⠁⠇⠀⠠⠠⠋` + - first differing cell (zero-based): 129 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #2928: 이밖에 미국에서 많은 구독자와 공신력 있는 외식 전문지인 ‘QSR(Quick Service Restaurant)’ 매거진과 ‘매쉬드(Mashed)’를 통해 K-치킨의 대표 브랜드로 소개된 바 있다. + - expected: `⠠⠟⠎⠗⠐⠣⠠⠟⠥⠊⠉⠅⠀⠠⠎⠻⠧⠊⠉⠑⠀⠠⠗⠑` + - actual: `⠠⠟⠎⠗⠐⠣⠠⠟⠅⠀⠠⠎⠻⠧⠊⠉⠑⠀⠠⠗⠑⠌⠁⠥` + - first differing cell (zero-based): 66 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `uppercase_roman_run_followed_by_hyphen_digits` + +Of the 571 candidates, 50 are the actual `pending_rule_review` subcluster. The other 521 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 67 mismatches were evaluable and 0 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Mismatch primary-class distribution: + +- `corpus_suspect`: 17 +- `pending_rule_review`: 50 + +Representative `exact` samples: + +- `sentence_01.json` #88: 진원생명과학은 현재 미국에서 코로나19에 관한 mRNA 또는 아데노바이러스(Ad26) 벡터 백신 접종자들을 대상으로 GLS-5310의 부스터 샷 임상연구를 진행 중이다. + - expected: `⠨⠟⠏⠒⠠⠗⠶⠑⠻⠈⠧⠚⠁⠵⠀⠚⠡⠨⠗⠀⠑⠕⠈⠍` + - actual: `⠨⠟⠏⠒⠠⠗⠶⠑⠻⠈⠧⠚⠁⠵⠀⠚⠡⠨⠗⠀⠑⠕⠈⠍` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #167: 2023년 제1회 상담사례 워크숍은 정신건강 임상심리사인 김한우 수퍼바이저(월덴3 아카데미 대표)가 ‘기질 및 성격검사(TCI), 미네소타 다면적 인성 검사(MMPI-2), 문장완성검사(SCT) 활용을 위한 심리평가 슈퍼비전’이라는 주제로 진행하였다. + - expected: `⠼⠃⠚⠃⠉⠀⠉⠡⠀⠨⠝⠼⠁⠀⠚⠽⠀⠇⠶⠊⠢⠇⠐⠌` + - actual: `⠼⠃⠚⠃⠉⠀⠉⠡⠀⠨⠝⠼⠁⠀⠚⠽⠀⠇⠶⠊⠢⠇⠐⠌` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #217: LG이노텍은 인공지능(AI)을 활용한 시뮬레이션을 통해 일반 자성소재 대비 에너지 손실은 최대 40% 줄이고, 파워 밀도는 3배 높아진 ‘고효율 페라이트’ 자성소재(X-2)를 독자적으로 개발해 넥슬림에 적용했다고 설명했다. + - expected: `⠴⠠⠠⠇⠛⠲⠕⠉⠥⠓⠝⠁⠵⠀⠟⠈⠿⠨⠕⠉⠪⠶⠦⠄` + - actual: `⠴⠠⠠⠇⠛⠲⠕⠉⠥⠓⠝⠁⠵⠀⠟⠈⠿⠨⠕⠉⠪⠶⠦⠄` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #48: 올트먼은 GPT-4가 기존 ‘GPT-3.5’보다 정확도가 40% 이상 높아졌지만, 이를 정보의 주요 출처로 사용해서는 안된다고도 지적했다. 거짓을 사실처럼 보이게 하는 이른바 ‘환각(hallucination) 현상’에 주의해야 한다는 입장이다. + - expected: `⠥⠂⠓⠪⠑⠾⠵⠀⠴⠠⠠⠛⠏⠞⠤⠼⠙⠫⠀⠈⠕⠨⠷⠀` + - actual: `⠥⠂⠓⠪⠑⠾⠵⠀⠴⠠⠠⠛⠏⠞⠤⠼⠙⠫⠀⠈⠕⠨⠷⠀` + - current primary/reason: `exact` / `exact` + +Representative `mismatch` samples: + +- `sentence_01.json` #87: 4일 진원생명과학은 자체 개발 흡인작용 피내 접종기 진덤(GeneDerm)을 이용한 코로나19 DNA 백신(GLS-5310)의 안전성·면역원성에 대한 임상1상 결과를 국제감염병학회 (ISID) 학술지인 국제감염질환저널에 발표했다고 4일 밝혔다. + - expected: `⠢⠘⠻⠚⠁⠚⠽⠀⠦⠄⠴⠠⠠⠊⠎⠊⠙⠠⠴⠀⠚⠁⠠⠯` + - actual: `⠢⠘⠻⠚⠁⠚⠽⠀⠴⠐⠣⠠⠠⠊⠎⠊⠙⠐⠜⠲⠀⠚⠁⠠` + - first differing cell (zero-based): 172 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #1355: 최신 한국형 화물창 기술(KC-2)을 적용한 국내 최초 LNG(액화천연가스) 벙커링 전용 선박인 ‘블루 웨일호’ (Blue Whale)가 10일 운항을 시작했다. + - expected: `⠗⠕⠂⠚⠥⠴⠄⠀⠦⠄⠴⠠⠃⠇⠥⠑⠀⠠⠱⠁⠇⠑⠠⠴` + - actual: `⠗⠕⠂⠚⠥⠴⠄⠀⠴⠐⠣⠠⠃⠇⠥⠑⠀⠠⠱⠁⠇⠑⠐⠜` + - first differing cell (zero-based): 114 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #1607: 오픈AI의 챗GPT(GPT-4)가 소믈리에로 데뷔했다. 와인 플랫폼 서비스 ‘칠링’이 챗GPT를 기반으로 한 인공지능(AI) 소믈리에 기능을 추가한 것이다. + - expected: `⠠⠠⠛⠏⠞⠤⠼⠙⠐⠜⠲⠫⠀⠠⠥⠑⠮⠐⠕⠝⠐⠥⠀⠊` + - actual: `⠠⠠⠛⠏⠞⠤⠼⠙⠴⠐⠜⠲⠫⠀⠠⠥⠑⠮⠐⠕⠝⠐⠥⠀` + - first differing cell (zero-based): 30 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #6606: 훈련전대는 해군·해병대 장병 420여 명(각 170여 명·250여 명)과 일출봉함(LST-Ⅱ, 4900t급), 상륙돌격장갑차(KAAV) 6대, K-808 차륜형장갑차 2대, K-55 자주포 2문, K-77 사격지휘장갑차 1대로 구성돼 있다. + - expected: `⠚⠢⠦⠄⠴⠠⠠⠇⠎⠞⠤⠠⠠⠊⠊⠂⠀⠼⠙⠊⠚⠚⠞⠲` + - actual: `⠚⠢⠦⠄⠴⠠⠠⠇⠌⠤⠠⠠⠊⠊⠂⠀⠼⠙⠊⠚⠚⠞⠲⠈` + - first differing cell (zero-based): 79 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +### `uppercase_word_after_whitespace_continuing_ascii_roman_text` + +Of the 1729 candidates, 544 are the actual `pending_rule_review` subcluster. The other 1185 candidates are exact or existing non-pending-primary controls; this membership alone does not reclassify them. + +For this output-signature audit, 614 mismatches were evaluable and 4 have their first differing cell inside the detected structure's current-engine output range. The remaining mismatches are controls against causal over-attribution. + +Localized first-difference transitions: + +- `U+2800 ⠀ -> U+2832 ⠲`: 1 +- `U+2801 ⠁ -> U+281C ⠜`: 1 +- `U+280E ⠎ -> U+280C ⠌`: 1 +- `U+2811 ⠑ -> U+283B ⠻`: 1 + +Mismatch primary-class distribution: + +- `corpus_suspect`: 70 +- `pending_rule_review`: 544 + +Representative `exact` samples: + +- `sentence_01.json` #32: 노스홀 메인 부스에서는 신기술인 ‘메타(META) 테크놀로지’를 적용해 화질을 혁신한 3세대 OLED TV 패널을 발표할 예정이다. + - expected: `⠉⠥⠠⠪⠚⠥⠂⠀⠑⠝⠟⠀⠘⠍⠠⠪⠝⠠⠎⠉⠵⠀⠠⠟` + - actual: `⠉⠥⠠⠪⠚⠥⠂⠀⠑⠝⠟⠀⠘⠍⠠⠪⠝⠠⠎⠉⠵⠀⠠⠟` + - current primary/reason: `exact` / `exact` +- `sentence_02.json` #41: 올해로 창립 10주년을 맞이한 IWPG는 유엔 경제사회이사회(UN ECOSOC)와 글로벌소통국(DGC)에 등록된 국제 NGO로서, 전쟁 반대와 실질적인 평화의 바람을 일으키고 있다. + - expected: `⠥⠂⠚⠗⠐⠥⠀⠰⠣⠶⠐⠕⠃⠀⠼⠁⠚⠨⠍⠉⠡⠮⠀⠑` + - actual: `⠥⠂⠚⠗⠐⠥⠀⠰⠣⠶⠐⠕⠃⠀⠼⠁⠚⠨⠍⠉⠡⠮⠀⠑` + - current primary/reason: `exact` / `exact` +- `sentence_03.json` #302: 건국대 경영전문대학원(KU MBA)은 2015년부터 국제경영대학발전협의회(AACSB) 인증을 보유하고 있다. 고등교육법 시행령에 따라 교육부가 인가한 13개 경영전문대학원 중 하나다. 경영전문대학원인 만큼 엄격하게 교육 품질을 관리한다. + - expected: `⠈⠾⠈⠍⠁⠊⠗⠀⠈⠻⠻⠨⠾⠑⠛⠊⠗⠚⠁⠏⠒⠦⠄⠴` + - actual: `⠈⠾⠈⠍⠁⠊⠗⠀⠈⠻⠻⠨⠾⠑⠛⠊⠗⠚⠁⠏⠒⠦⠄⠴` + - current primary/reason: `exact` / `exact` +- `sentence_04.json` #363: 켈리(KELLY)는 ‘KEEP NATUALLY’의 줄임말로 인위적인 것을 최소화하고 자연주의적인 원료, 공법, 맛을 추구한다는 의미를 담고 있다. + - expected: `⠋⠝⠂⠐⠕⠦⠄⠴⠠⠠⠅⠑⠇⠇⠽⠠⠴⠉⠵⠀⠠⠦⠴⠠` + - actual: `⠋⠝⠂⠐⠕⠦⠄⠴⠠⠠⠅⠑⠇⠇⠽⠠⠴⠉⠵⠀⠠⠦⠴⠠` + - current primary/reason: `exact` / `exact` + +Representative `localized_mismatch` samples: + +- `sentence_01.json` #23829: 증강현실(AR)과 인공지능(AI) 기술을 접목한 3D AR 아바타 제작 애플리케이션인 ‘제페토’를 활용해 비대면 동아리 소모임과 심뇌혈관질환 예방사업 운영 및 홍보 등을 한다. + - expected: `⠠⠕⠂⠦⠄⠴⠠⠠⠁⠗⠠⠴⠈⠧⠀⠟⠈⠿⠨⠕⠉⠪⠶⠦` + - actual: `⠠⠕⠂⠦⠄⠴⠠⠠⠜⠠⠴⠈⠧⠀⠟⠈⠿⠨⠕⠉⠪⠶⠦⠄` + - first differing cell (zero-based): 15 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_02.json` #14: 한국타이어앤테크놀로지㈜(대표이사 이수일, 이하 한국타이어)가 국제자동차연맹(FIA)이 주관하는 ‘FIA 주니어 ERC(FIA Junior ERC, 이하 주니어 ERC)’ 대회에 레이싱 타이어를 독점 공급한다. + - expected: `⠍⠉⠕⠎⠀⠴⠠⠠⠑⠗⠉⠦⠄⠴⠠⠠⠋⠊⠁⠀⠠⠚⠥⠝` + - actual: `⠍⠉⠕⠎⠀⠴⠠⠠⠻⠉⠦⠄⠴⠠⠠⠋⠊⠁⠀⠠⠚⠥⠝⠊` + - first differing cell (zero-based): 114 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #17921: 한국타이어앤테크놀로지㈜(대표이사 이수일, 이하 한국타이어)가 오는 8월 21일, 컴포트 타이어 브랜드 ‘키너지(Kinergy)’의 신제품인 사계절용 밸런스 타이어 ‘키너지 ST AS(Kinergy ST AS)’를 국내에 새롭게 출시한다. + - expected: `⠉⠎⠨⠕⠀⠴⠠⠠⠎⠞⠀⠠⠠⠁⠎⠐⠣⠠⠅⠔⠻⠛⠽⠀` + - actual: `⠉⠎⠨⠕⠀⠴⠠⠠⠌⠀⠠⠠⠁⠎⠐⠣⠠⠅⠔⠻⠛⠽⠀⠠` + - first differing cell (zero-based): 162 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_04.json` #3324: 삼성전자는 이날 생성형 인공지능(AI) 서버에 적용되는 서버용 SSD ‘PM1743’과 쿼드러플 레벨 셀(QLC) 낸드 기반 256TB SSD도 선보였다. + - expected: `⠶⠀⠴⠠⠠⠎⠎⠙⠀⠠⠦⠠⠠⠏⠍⠼⠁⠛⠙⠉⠠⠴⠲⠈` + - actual: `⠶⠀⠴⠠⠠⠎⠎⠙⠲⠀⠠⠦⠴⠠⠠⠏⠍⠼⠁⠛⠙⠉⠴⠄` + - first differing cell (zero-based): 68 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +Representative `mismatch` samples: + +- `sentence_01.json` #35: 삼성전자는 올해 CES에서 77인치 OLED TV를 처음으로 공개할 예정이다. 이는 지난 8월 ‘국제정보디스플레이학술대회(IMID) 2022’에서 삼성디스플레이가 공개한 77인치 퀀텀닷(QD)-OLED 패널을 탑재한 제품이다. + - expected: `⠋⠏⠒⠓⠎⠢⠊⠄⠴⠐⠣⠟⠙⠐⠜⠤⠠⠠⠕⠇⠫⠲⠀⠙` + - actual: `⠋⠏⠒⠓⠎⠢⠊⠄⠦⠄⠴⠠⠠⠟⠙⠠⠴⠤⠴⠠⠠⠕⠇⠫` + - first differing cell (zero-based): 172 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_02.json` #14: 한국타이어앤테크놀로지㈜(대표이사 이수일, 이하 한국타이어)가 국제자동차연맹(FIA)이 주관하는 ‘FIA 주니어 ERC(FIA Junior ERC, 이하 주니어 ERC)’ 대회에 레이싱 타이어를 독점 공급한다. + - expected: `⠍⠉⠕⠎⠀⠴⠠⠠⠑⠗⠉⠦⠄⠴⠠⠠⠋⠊⠁⠀⠠⠚⠥⠝` + - actual: `⠍⠉⠕⠎⠀⠴⠠⠠⠻⠉⠦⠄⠴⠠⠠⠋⠊⠁⠀⠠⠚⠥⠝⠊` + - first differing cell (zero-based): 114 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` +- `sentence_03.json` #48: WIS 전시장 내 KT DS 전시관에서는 실제 현장에 적용되는 통합 대시보드 화면과 위험 구역을 감지·경고하는 지능형 폐쇄회로(CC)TV를 통해 에스패스를 체험할 수 있다. + - expected: `⠌⠠⠧⠗⠚⠽⠐⠥⠴⠐⠣⠠⠠⠉⠉⠐⠜⠠⠠⠞⠧⠲⠐⠮` + - actual: `⠌⠠⠧⠗⠚⠽⠐⠥⠦⠄⠴⠠⠠⠉⠉⠠⠴⠴⠠⠠⠞⠧⠲⠐` + - first differing cell (zero-based): 128 + - current primary/reason: `corpus_suspect` / `rule34_roman_indicator_before_opening_parenthesis` +- `sentence_04.json` #2340: 제네시스 GV70 전동화 모델(전기차 버전)이 독일 자동차 전문지의 비교 평가에서 1위를 차지했다. 비교 대상인 아우디 Q8 e-트론, 벤츠 EQE SUV(스포츠유틸리티차)를 제쳤다. + - expected: `⠕⠀⠴⠠⠟⠼⠓⠀⠰⠑⠲⠤⠓⠪⠐⠷⠐⠀⠘⠝⠒⠰⠪⠀` + - actual: `⠕⠀⠴⠠⠟⠼⠓⠀⠑⠲⠤⠓⠪⠐⠷⠐⠀⠘⠝⠒⠰⠪⠀⠴` + - first differing cell (zero-based): 117 + - current primary/reason: `pending_rule_review` / `foreign_text_rule_review` + +## UEB grade-1 first-difference cohorts + +These cohorts are defined by both an input boundary and the sentence's actual first-difference transition. They therefore do not claim every mismatch merely coexisting with an ASCII run. Candidate, exact, mismatch, and primary-class counts remain cross-cutting controls; only the reported target transition is the localized residual under review. The reverse transition is retained separately rather than folded into the target. + +| Cohort | Candidates | Exact controls | Mismatch | Target localized | Reverse | +|---|---:|---:|---:|---:|---:| +| `allcaps_roman_run_beginning_with_pure_letter_shortform` | 3242 | 2809 | 433 | 2 | 50 | +| `uppercase_ascii_run_immediately_after_digit_in_roman_sequence` | 1896 | 1569 | 327 | 0 | 3 | +| `uppercase_ascii_run_immediately_after_hyphen_in_roman_sequence` | 952 | 724 | 228 | 5 | 2 | +| `pure_allcaps_segment_before_hyphen_and_multi_allcaps_segment_after` | 448 | 359 | 89 | 0 | 0 | + +### All-caps shortform prefix at an attached Roman entry + +UEB 2024 rule 5.7.2 requires grade-1 mode when a letters-sequence could be read as a shortform or as containing one. Rule 10.9.7 covers a standing-alone shortform-shaped sequence, rule 10.9.8 covers a sequence at the beginning of a longer word (the PDF example is `LLC`), and rule 5.8.1 places grade 1 before capitalization. The current standalone ASCII-token route already supplies that guard for complete shortform-shaped controls such as `AC`, `CD`, `IMM`, and `AG`; the attached Korean-word/parenthetical route enters directly at the capital marker and accounts for part of the localized `⠰ -> ⠠` signature. This is a routing distinction supported independently by the PDF, not an expected-output lookup. The implemented boundary is only rule 10.9.7's complete pure-letter shortform. Longer runs such as `GDP`, `LLM`, and the PDF's rule-10.9.8 `LLC` example remain in the broad diagnostic cohort but are not generalized in Korean routing: that broader experiment regressed its exact controls. A shortform appearing later would require the still-distinct grade-1 word rule 10.9.9. + +Implementation-boundary experiment (all numbers are full-corpus exact matches, with the committed analyzer-only checkpoint as baseline): + +| Boundary | Exact / 83,528 | Change | Decision | +|---|---:|---:|---| +| Analyzer-only baseline | 66,546 | — | control | +| Prefix guard extended through the uppercase token route | 65,264 | -1,282 | rejected | +| Same-token token route narrowed, rule-28 prefix retained | 65,355 | -1,191 | rejected | +| Uppercase token route restored, rule-28 prefix retained | 65,474 | -1,072 | rejected | +| Rule-28 complete shortform only | 66,683 | +137 | retained | + +At the retained boundary the broad cohort moves from 2,239 exact / 1,881 mismatch / 962 target-localized / 1 reverse to 2,376 exact / 1,744 mismatch / 778 target-localized / 42 reverse. These figures do not turn the remaining longer-prefix members into an engine rule; they preserve the failed broader trials as evidence that input shape alone is unsafe. + +Same-surface controls demonstrate why primary classes must not be changed by cohort membership: + +| Surface | Candidates | Exact | Mismatch | Target-localized | +|---|---:|---:|---:|---:| +| `AC` | 159 | 156 | 3 | 0 | +| `LLM` | 176 | 125 | 51 | 0 | +| `CD` | 65 | 51 | 14 | 0 | +| `IMM` | 27 | 23 | 4 | 0 | +| `AG` | 18 | 15 | 3 | 0 | +| `GDP` | 359 | 335 | 24 | 0 | +| `WD` | 10 | 10 | 0 | 0 | + +### Uppercase immediately after a digit + +UEB rule 5.6.1 says the numeric indicator establishes grade-1 mode, and rule 5.6.2 ends that mode at a space, hyphen, dash, or grade-1 terminator; a capitalization indicator is not a terminator. Korean rule 35 likewise keeps Roman letters and an adjacent number in one Roman section. The PDF's printed `3b`, `3B`, and `3m` examples distinguish the three following-letter classes: lowercase `a`-`j` retains `⠰` because its cells are numeric, a capital uses its capitalization indicator, and lowercase `k`-`z` needs no extra indicator. `Braille4All`, `M4G`, and `W1N` independently confirm the capital boundary inside longer alphanumeric strings. Before the engine change this cohort contained 330 localized `⠠ -> ⠰` cases. A blanket digit-to-letter removal reached 67,000/83,528 (+317) but was rejected: retaining `⠰` only for lowercase `a`-`j` recovers 10 exact cases and raises the result to 67,010. The wrapper control also exposes a separate routing boundary: a numeric run already preceded by an ASCII letter is part of the Roman identifier, not a fresh rule-69 compact unit. Preserving the rule-69 path for genuinely numeric-leading units while excluding that identifier boundary adds 2 more exact cases, for a final 67,012 (+329). The uppercase cohort moves from 756 exact / 1,140 mismatch / 330 target-localized / 1 reverse to 1,078 exact / 818 mismatch / 0 target-localized / 1 reverse. The remaining non-exact members are not attributed to the removed uppercase transition: their sentence-level first difference may lie in another structure and remains under its existing primary class. This numeric state change remains separate from both the complete-shortform guard and the hyphen continuation boundary below. + +### Uppercase immediately after a hyphen + +UEB rule 5.7.2 prints `CD-ROM` with one grade-1 indicator before `CD` and no second grade-1 indicator after the hyphen. Korean rule 29 similarly uses one Roman span for consecutive Roman text. Before the engine change, the broad diagnostic contained 952 candidates / 32 exact / 920 mismatch, with 312 localized `⠠ -> ⠰` and 2 reverse transitions. A blanket uppercase-suffix removal reached 67,222 (+210) but made the broad cohort's single-capital controls such as `Around-U`, `DALL-E`, `ISMS-P`, and `USB-C` non-exact; it was rejected. Requiring only a two-letter uppercase suffix reached 67,162 (+150) but regressed the mixed-prefix exact control `Ko-LLM`; it was also rejected. The retained boundary matches the complete PDF shape: the immediately adjacent prefix is a pure-uppercase letter segment and the immediately adjacent suffix is a pure-uppercase segment of at least two letters. It reaches 67,138 (+126) while preserving all 32 baseline exact controls. The broad diagnostic now contains 952 candidates / 158 exact / 794 mismatch, with 157 localized `⠠ -> ⠰` and 3 reverse transitions. The dedicated `pure_allcaps_segment_before_hyphen_and_multi_allcaps_segment_after` row reports only the implemented subset; broad mixed-case and single-capital members remain controls or pending review. `K-ALM` is the one new reverse surface but was already a mismatch before this change, not an exact regression. Digit-hyphen forms such as `F-35` remain excluded, and the complete-shortform guard still legitimately precedes `CD` in `CD-ROM`. + +### Attached Korean-to-Roman hyphen boundary + +Korean rule 29 opens a Roman section for Roman text in a Korean sentence, rule 33 keeps the hyphen as punctuation at the Korean/Roman boundary, and rules 35-36 own adjacent alphanumerics and Roman numerals. The retained production gate therefore routes an immediately attached capital-led or multi-letter Roman identifier as prose (`하쿠토-R`, `기장-KBO`, `온다-life`), but leaves a single lowercase variable and any token with an explicit mathematical operator on the mathematics route (`값-x`, `값-X+1`). The analyzer applies the encoder's selective U+2160-U+217F compatibility expansion before testing the word grammar, so `천궁-Ⅱ` is audited as the equivalent `천궁-II` boundary. + +Before this gate, the deterministic cohort contained 105 candidates / 62 exact / 43 mismatch. It now contains 105 candidates / 84 exact / 21 mismatch. The complete corpus exact-ID audit moved from 75,704 to 75,785 (+22), and every new exact ID belongs to this cohort; no formerly exact ID was lost. Cohort membership is input-only and never changes a primary class, so the remaining non-exact members retain their independent review causes. + + +## Roman-entry residual cohorts after grade-1 localization + +These three cohorts split the former dominant `⠴ -> blank` residual by the input structure at the actual first-difference location. Their entry boundary is anchored by independently encoding the input prefix and requiring it to equal the full current-engine output prefix; repeated Roman text elsewhere cannot satisfy the locator. A localized count is recorded only when no earlier output-localized cohort already claims that first difference. Candidate membership remains cross-cutting and never changes a primary class. + +Korean rule 29 requires a Roman indicator before Roman text in a Korean sentence. Rules 33-35 define the relevant hyphen, enclosure, and number boundaries. Independently, math rules 2, 6, 11, 12, and 45 permit subtraction, parentheses, Roman variables, and function notation with overlapping ASCII surface forms. The surface gates below therefore cannot by themselves exclude a mathematical reading. + +| Cohort | Candidates | Exact controls | Mismatch | Pending | Corpus suspect | Localized `⠴ -> blank` | Reverse `blank -> ⠴` | +|---|---:|---:|---:|---:|---:|---:|---:| +| `roman_hyphenated_word_after_whitespace_following_korean_word` | 361 | 277 | 84 | 73 | 11 | 6 | 0 | +| `roman_parenthetical_headword_after_whitespace_following_korean_word` | 4695 | 3945 | 750 | 640 | 109 | 2 | 0 | +| `korean_prefixed_roman_parenthetical_followed_by_allcaps_hyphen_suffix` | 11 | 1 | 10 | 0 | 10 | 0 | 0 | + +The whitespace parenthetical-headword cohort also retains 2 localized `⠠ -> ⠴` and 0 localized `⠴ -> ⠰` cases as separate transitions; they are not folded into the target. The exact controls demonstrate that the broad structure is already correct in many sentences, while the attached parenthetical-hyphen cohort has no exact control and includes cases already classified by the stricter rule-34 reference-order contradiction. Consequently none of these measurements authorizes an engine change; they are deterministic pending/corpus-review diagnostics only. + +### Consecutive Roman uppercase-word re-entry + +Korean rule 29 explicitly says that when two or more Roman items occur consecutively, the Roman indicator is placed only before the first and the terminator only after the last. Its printed `Los Angeles` and `Table of Contents` examples exercise multiword Roman sections; the rule-28 appendix independently supplies capitalization indicators inside that section. The current token phase can nevertheless insert an explicit Roman-entry event before a later uppercase word when the preceding Roman run began inside a mixed Korean/punctuation word. The character emitter is still in Roman mode at that point, so this is a candidate duplicate-event boundary rather than permission to rewrite arbitrary multiword ASCII text. + +Analyzer checkpoint `266e70c` fixed the pre-change baseline at 1,729 candidates, 386 exact controls, 1,343 mismatches, 1,272 pending members, and 385/1,343 evaluable mismatches localized to the current re-entry signature. Those localized transitions were 365 `⠠ -> ⠴`, 16 `⠰ -> ⠴`, no reverse `⠴ -> ⠠`, and four other transitions. Exact controls include contexts where the first Roman word already opened token-level mode, while parenthesized/mixed-token examples expose the duplicate event. + +The generalized fix makes an explicit `EnterEnglish` event idempotent when final emit state is already inside a Roman section; it neither names an input nor changes a new section's entry. The current measurement is 1729 candidates, 1115 exact controls, 614 mismatches, 544 pending members, and 4/614 localized mismatches. Current target counts are 0 `⠠ -> ⠴`, 0 `⠰ -> ⠴`, and 0 reverse `⠴ -> ⠠`. Cohort exact controls increase by 275, while corpus-wide exact matches rise from 67,138 to 67,442 (+304); corrected prefixes that still differ later remain mismatches, and applications outside this strict input gate account for the remaining net gain. The raw/residual reverse `⠴ -> ⠠` totals remain 68/65 before and after the change. The complete custom standard summary is 5,141 total, 5,141 success, 0 failure, and 0 skipped. + +### Closed Roman parenthetical after a non-ASCII-letter boundary + +Korean rule 34 (2024 Korean-rules PDF p.29) prints the opening enclosure before Roman entry in `링컨(Lincoln)`: the opening sequence is `⠦⠄⠴`. This output-position cohort finds a closed, non-nested parenthetical whose body begins with an ASCII letter and whose opening does not immediately follow another ASCII letter, then locates its complete current-engine signature without consulting the reference. Its localized boundary includes the two current output cells immediately before the opening plus the first three entry cells, so rule-11 math spacing can be separated from a difference later inside the parenthetical. Direct function-call shapes such as `f(x)` are excluded. The PDF's math rule 6 (p.59), rule 12 (pp.63-64), and rule 45 (p.75) nevertheless leave standalone `(x)` and other Roman-letter parenthetical mathematics as counterexamples, so the surface gate is not an engine-routing predicate. + +The cross-cutting input cohort contains 63959 candidates: 57175 exact controls and 6784 mismatches. Mismatch primary classes remain unchanged: 5599 `pending_rule_review`, 1181 `corpus_suspect`, 0 `comparison_method`, and 4 `unsupported_character_review`. Of 6784 evaluable mismatches, 418 have the first difference at the detected leading-spacing/entry boundary; these include 114 `U+2826 ⠦ -> U+2834 ⠴`, 36 `U+2834 ⠴ -> U+2826 ⠦`, and 29 `U+2826 ⠦ -> U+2800 ⠀` transitions; the exact localized reverse `U+2800 ⠀ -> U+2826 ⠦` occurs 5 times. After all earlier localized cohorts and this cohort are excluded, the raw-to-residual target count is 122 -> 5, the Roman-indicator reverse count is 41 -> 2, and the spacing target/reverse counts are 63 -> 21 and 5 -> 0. The short full-encoder form `웹3(Web3)` emits the PDF opening order as a routing control, while whitespace and quote boundaries reproduce the current Roman-first opening. Representative localized samples, with shard and index, are retained in the generated cluster sample table. Because exact controls are abundant and the PDF does not make this input shape semantically sufficient to exclude mathematics, no engine change or primary reclassification is inferred. The one raw/residual spacing reverse is a Korean-body `△(교육)` layout boundary, not a Roman-parenthetical member, and remains a separate rule-72/layout review. + + +The HCA-style headword-expansion gate now supplies one narrow prose-routing premise. Korean rules 29 and 34 require a fresh Roman section and continuous Roman transcription for a complete all-capitals headword followed by a closed, multiword Roman expansion. The implementation requires a headword of at least two ASCII capitals and at least two ASCII-letter words inside the parenthesis; digits, operators, scripts, nested brackets, and alphanumeric text after the closing parenthesis remain math-owned controls. Rule 34's `링컨(Lincoln)은` additionally proves that attached Korean text after the closing parenthesis stays on the prose route; the same boundary now covers the multiword form without admitting ASCII letters or digits in the trailer. Together these boundaries change 71 corpus cases from mismatch to exact and raise this cohort's exact controls from 17 to 87. The residual members still measure contraction, capitalization, earlier sentence differences, unsupported characters, and reference-order conflicts rather than authorizing a wider surface-form rule. + +The standalone-uppercase cohort is similarly ambiguous. Hangeul rule 28's appendix defines the capital-word indicator for two or more consecutive capitals, and rule 29 defines Roman indicators around Roman text in a Korean sentence. But math rule 12 also uses uppercase Roman variables, while science rule 7 uses uppercase runs in chemical formulas. The input gate cannot determine which semantic regime applies, so its output differences are observations to review, not permission to infer an engine rule from the corpus reference. + +The Korean-prefixed all-caps parenthetical cohort isolates that ambiguity more narrowly. Hangeul rules 28/29 require Roman and capitalization indicators for prose acronyms, while science rule 7 requires element-by-element capitals for chemical formulae. Both meanings can have the same input surface form. The observed `COO`/`NSC`/`MOU` output differences therefore do not justify disabling either algorithm without independent semantic evidence. + +Two narrower cohorts separate causes hidden by the frequent `U+2834 -> U+2800` cell transition. `single_capital_followed_by_parenthesized_digits` reproduces the current math-token routing of forms such as `A(14)`: Hangeul rules 29 and 34 govern a Roman section and a parenthesized Roman form, while math rule 6 independently defines parenthesized function notation such as `f(x)`. A capital and numeric argument do not remove that mathematical counterexample, so this localized routing difference remains pending rather than authorizing an input-shape exception. `mixed_roman_korean_word_before_uppercase_headword_expansion` separately targets the next Roman headword after a mixed Roman+Korean word (for example, a Korean particle attached to the previous Roman name). Its range is anchored to that later headword, not to the earlier Roman entry. The narrow rules-29/34 prose gate described above is now implemented, and this cohort no longer has a missing-entry localized transition. Its remaining localized differences are later Roman-letter/contraction differences. The two causes and their controls remain separately measurable instead of widening the headword grammar. + +The uppercase-Roman hyphen-digits cohort is a third independent cause. Hangeul rule 35 explicitly shows `D-100` as a Roman-and-number continuation (2024 Korean-rules PDF p.29), while math rule 2 defines subtraction and the math chapters allow uppercase Roman variables. The surface form alone therefore does not prove whether `F-35` is an identifier or a subtraction expression. This cohort records the current operator-routing signature and exact controls without merging it into either `A(14)` or HCA-style diagnostics. No engine change is made without both a safe semantic boundary and exact controls. + +The all-caps `OU` cohort isolates a frequent output transition without treating the reference as a rule. Hangeul rules 28, 29, and 32 delegate Roman-letter content to UEB (2024 Korean-rules PDF p.25 and following rules). UEB 10.12.1 says not to use a contraction when it is known or can be determined that an abbreviation or acronym's letters are pronounced separately, but says to use the contraction when that pronunciation is in doubt; UEB 10.12.2 otherwise uses contractions in abbreviations and acronyms (UEB 2024 PDF pp.191-192; Korean UEB translation PDF pp.182-183). Thus an expected `o` + `u` versus the current `ou` groupsign can be localized to an uppercase run, yet the surface run alone cannot distinguish a letter-by-letter initialism from a pronounceable word or acronym. That distinction needs lexical or semantic evidence absent from this input gate. Exact members are controls, identical-input conflicting references remain `corpus_suspect`, and no engine change is inferred. + +The all-caps Roman middle-dot cohort is also semantically underdetermined. Hangeul rule 29 defines Roman indicators around Roman text in a Korean sentence, and Hangeul rule 50 requires U+00B7 to be attached on both sides, but neither rule says that the punctuation joins the adjacent Roman runs into one Roman span. Math rule 2 separately defines the same printed dot as multiplication, and science rule 4 uses it inside chemical formulae. An input-only `AI·SW` gate therefore cannot prove which mode transition is required. Exact cases remain controls, mismatches retain their existing primary class, and no engine rule is inferred from their references. Representative samples are sentence-level evidence: when the reported first difference precedes the detected middle-dot span, the cohort must not be treated as the cause of that mismatch. + +The narrower Roman-before-middle-dot boundary cohort separates that semantic question from a checkable indicator boundary. Hangeul rule 29 requires a Roman terminator after Roman text. Rule 33 enumerates the punctuation that suppresses or moves that terminator, but does not include U+00B7; rule 50 requires the middle dot to be attached on both sides and does not state a Roman-terminator exception. Thus a localized reference that omits the terminator conflicts with the current rule-29/33 path on the available PDF text. This is conservative corpus/PDF-reference review evidence, not permission to remove the terminator or to reclassify non-localized cases. + +The inline parenthesized-operator cohort has an independently checkable spacing boundary. Hangeul rule 46 inserts spaces only when an operation or comparison sign is between Korean text, while the literal parentheses intervene in this gate. Hangeul rule 49 says punctuation spacing follows the print, and science rule 21 prints and brailles `(-)` and `(+)` with no spaces inside the parentheses. The output-signature count is therefore the implementation-candidate subset; mere sentence-level coexistence is retained only as a control. ASCII hyphen-minus is independently supported through the `Symbol`/rule-49 punctuation path, while `+`, `×`, `÷`, and `=` reach the `MathSymbol` spacing rule; end-to-end tests preserve tight parentheses on both paths. Analyzer checkpoint `30a9b10` recorded the pre-fix baseline as 23 candidates, 0 exact, 23 mismatches, and 19 mismatches whose first difference was signature-local. The generalized rule-46/49 fix is evaluated below against that immutable baseline rather than inferred from a reference string. + +The tight-triangle cohort is not an implementation premise. Hangeul rule 49 assigns `△` the omission-mark role and requires print spacing to be followed, while rule 72 also assigns the same glyph a bullet role but shows a print space after every bullet. A tight corpus input does not identify which role was intended, and adding a space absent from the input would contradict rule 49 unless independent layout evidence establishes a bullet. The localizer searches the complete actual output for a neutral-Korean, current-engine signature covering the mark and its first following Korean character; it neither encodes a context-sensitive sentence prefix in isolation nor reads the reference output. Tight marks followed by ASCII letters or digits remain outside this gate. Localized mismatches are therefore corpus/layout review evidence only. + +Corpus contradictions remain a separate gate: identical inputs with conflicting references are classified as `corpus_suspect` before these cohorts are recorded and would appear explicitly in each mismatch primary-class distribution. Their absence does not prove a reference correct; it only means that this deterministic contradiction test did not fire. + +### Uppercase Roman runs containing `AR` + +UEB 10.12.1 (2024 UEB PDF pp.191-192, printed pp.163-164) says not to use a contraction when letters within an abbreviation or acronym are known to be pronounced separately, but to use the contraction in case of doubt. Its official `DAR` example writes separate `a` and `r` cells, while UEB 10.12.2's official `START` example uses the `ar` groupsign. Both full-encoder controls pass. Thus an uppercase surface containing `AR` does not itself supply the pronunciation or lexical meaning needed to select either form. + +The output-localized cohort contains 1022 candidates, 480 exact controls, and 542 mismatches. Existing mismatch primary classes are preserved: 521 `pending_rule_review`, 18 `corpus_suspect`, 0 `comparison_method`, and 3 `unsupported_character_review`. Of 542 evaluable mismatches, 416 have their first difference inside the detected current-engine run: 411 `U+2801 ⠁ -> U+281C ⠜` and 0 `U+281C ⠜ -> U+2801 ⠁`. Across raw pending transitions and the final residual after localized cohorts, the target count is 413 -> 1; the reverse is 2 -> 2. Inputs denoting separately pronounced initialisms such as corpus `AR`/`ARS` coexist with exact or officially contracted controls. This is deterministic evidence for a pronunciation-dependent pending cohort, not an engine rule or primary reclassification. Representative shard/index samples are retained above. + +### Roman run after a closed Roman enclosure + +Korean rule 29 places one Roman indicator/terminator pair around consecutive Roman items, with `Los Angeles` and `Table of Contents` as its printed multiword examples. Rule 34 separately omits the Roman terminator when Roman text is enclosed by quotation marks or parentheses. Neither printed rule states whether a later Roman run after the enclosure, intervening punctuation, and whitespace is a continuation of that section or a fresh section. The input gate therefore detects only the structural boundary; it does not label the later run as semantically new. + +The cohort contains 1093 candidates, 580 exact controls, and 513 mismatches. Existing mismatch primary classes are preserved: 163 `pending_rule_review`, 349 `corpus_suspect`, and 1 `unsupported_character_review`. Of 513 evaluable mismatches, 5 are output-localized to the current later-run signature plus its one leading boundary cell: 0 `U+2834 ⠴ -> U+2830 ⠰` and 0 `U+2830 ⠰ -> U+2834 ⠴`. The target counted 2 raw and 333 residual cases before this cohort; it is now 1 final residual. Across raw and final residual maps, `U+2830 ⠰ -> U+2834 ⠴` is 1 -> 0. Exact controls coexist with both directions, and 321 mismatches are already independently identified corpus contradictions. Without a printed fresh-entry example or semantic enclosure model, changing continuation state would be reference-fitting; this remains a deterministic pending/corpus-review cohort only. Representative shard/index samples are retained above. + +### Uppercase Roman runs containing `ED` + +UEB 10.12.1 (2024 UEB PDF pp.191-192, printed pp.163-164) requires separate letters when an abbreviation's letters are known to be pronounced separately, and its official `OED` example writes `e` and `d` separately. Rule 10.12.2 otherwise uses contractions; its official `BEd` example uses the `ed` groupsign. Both full-encoder controls pass. Consequently an uppercase `ED` surface cannot establish pronunciation or abbreviation semantics by spelling alone. + +The cohort contains 816 candidates, 362 exact controls, and 454 mismatches. Existing mismatch primary classes remain 386 `pending_rule_review`, 68 `corpus_suspect`, 0 `comparison_method`, and 0 `unsupported_character_review`. Of 454 evaluable mismatches, 341 are localized to the detected current-engine run: 339 `U+2811 ⠑ -> U+282B ⠫` and 0 `U+282B ⠫ -> U+2811 ⠑`. Across raw pending transitions and the final residual after localized cohorts, the target is 344 -> 3 and the reverse is 0 -> 0. Corpus initialisms such as `LED` and `GED` require external pronunciation knowledge, while exact and official controls preserve contraction-bearing outcomes. No engine change or primary reclassification is made; representative shard/index samples are retained above. + +### Uppercase Roman runs containing `ST` + +UEB 10.12.1 (2024 UEB PDF pp.191-192, printed pp.163-164) says not to use a contraction when letters within an abbreviation or acronym are known to be pronounced separately, but to use the contraction in case of doubt. Rule 10.12.2 requires contractions in other abbreviations and acronyms. Consequently an uppercase `ST` surface alone cannot determine whether `s` + `t` or the `st` groupsign is required. The current cohort contains 1479 candidates, 815 exact controls, and 664 mismatches; primary classes remain 625 `pending_rule_review`, 39 `corpus_suspect`, 0 `comparison_method`, and 0 `unsupported_character_review`. Of 664 evaluable mismatches, 473 are localized to the detected current-engine run: 466 `U+280E ⠎ -> U+280C ⠌` and 1 `U+280C ⠌ -> U+280E ⠎`. The target's raw-to-residual count is 473 -> 6, and the reverse is 2 -> 1. Exact controls such as `STAYG`, `KAIST`, `DGIST`, and `POSTECH` coexist with localized mismatches such as `WSTS`, `HUST`, `OST`, and `USTR`; representative shard/index samples are preserved in the cluster table. This lexical/pronunciation distinction cannot be inferred from the input-only spelling, so no engine change or primary reclassification is made. + +### Uppercase segments joined by ampersand: capitalization extent + +UEB 8.4.2 (2024 UEB PDF p.118, printed p.90) terminates capitals word mode at a nonalphabetic symbol. UEB 3.1.1 and the capitalization examples (PDF pp.51 and 120, printed pp.23 and 92) consequently print `AT&T` as `⠠⠠⠁⠞⠈⠯⠠⠞` and `B&B` as `⠠⠃⠈⠯⠠⠃`: Roman mode remains continuous, but capitalization restarts for each ASCII-letter segment. The detector accepts only complete uppercase ASCII segments joined directly by `&`, with the same non-alphanumeric outer boundaries as the existing Roman-ampersand rule, and requires the run to begin its whitespace-delimited token. Korean-attached and punctuation-prefixed occurrences stay outside the change scope. + +At the analyzer-only checkpoint this cohort contained 439 candidates, 1 exact control, 438 mismatches, and 220 first differences localized inside the independently reproduced Korean-context signature. After the general capitalization correction it contains 439 candidates, 361 exact controls, and 78 mismatches. Existing remaining mismatch primaries are 68 `pending_rule_review`, 10 `corpus_suspect`, 0 `comparison_method`, and 0 `unsupported_character_review`. Of 78 evaluable mismatches, 1 have their first difference inside that signature. The sole pre-change exact member contained lowercase Roman text later in the same whitespace token and was outside the production predicate's actual change scope; its primary outcome was preserved. The cohort table above retains the transition distribution and shard/index samples. Capitalization extent is fixed by the official symbol examples and requires neither pronunciation nor corpus semantics; the diagnostic never changes a primary class. + +### Capitals-word nonletter change-scope audit + +This input-only scope exactly mirrors the former token predicate where it could incorrectly carry capitals-word handling across a nonletter. UEB 8.4.2 terminates that mode at the nonletter. Trailing nonletters after the final uppercase run are excluded because their output is unchanged. Before the correction all 1,733 candidates were mismatches and none was exact. The current run has 1733 candidates, 1270 exact controls, and 463 mismatches. A complete exact-ID set audit found 852 newly exact cases and zero cases lost from the 68,439-exact baseline, yielding 69,291/83,528 (82.96%). Two deliberately rejected wider predicates exposed why the boundary matters: requiring an entirely uppercase-only token lost 87 former exact cases, while treating every initial uppercase run as token-level capitals mode lost 33 mixed-case controls such as the UEB `TVOntario` class by bypassing its required capitals terminator. The retained predicate pre-emits only when the initial run has at least two capitals and every ASCII letter in the token is uppercase; Rule 28 independently restarts capitalization after the nonletter. This cohort remains a regression audit only: membership does not assign a primary class or attribute a first difference. + +### Attached Roman segments joined by ampersand + +UEB 3.1.1 (2024 UEB PDF p.51, printed p.23) directly prints `AT&T` and `B&B` with the attached ampersand `⠈⠯` and no mode boundary around it. Korean rule 71 (2024 Korean-rules PDF p.51, printed p.45) assigns the same `⠈⠯` cells, while rule 29 places Roman entry before a Roman section and termination after its last item. Before the engine change, the Korean-context path instead exited before `&`, wrapped the information symbol as a separate Roman section, and re-entered for the following letters. + +The baseline was 802 candidates / 0 exact / 802 mismatch, preserving 776 `pending_rule_review`, 10 `corpus_suspect`, and 16 `unsupported_character_review` primary classifications. Its real-prefix localizer assigned only the output cell immediately before `&`: 356/802 mismatches localized, all 356 were `U+2808 ⠈ -> U+2832 ⠲`, the raw-to-residual target count was 411 -> 53, and `U+2832 ⠲ -> U+2808 ⠈` was 0 raw / 0 residual. + +The implemented gate shares the analyzer's complete-run predicate: one or more non-empty ASCII-letter segments joined directly by `&`, with non-alphanumeric outer boundaries. It keeps the existing Roman mode open and suppresses only rule 71's redundant wrapper around that ampersand. Spaced `Marks & Spencer`, Korean `가&나`, empty segments, and outer digit continuations remain outside the gate. Official full-encoder controls `AT&T` and `B&B` pass, and the Korean rule-71 spaced example `종이접기 & 클레이아트` retains its independent `⠴⠈⠯⠲` section. + +After the change the same cohort contains 802 candidates, 687 exact and 115 mismatch. Current mismatch primary classes remain evaluator-owned: 103 `pending_rule_review`, 12 `corpus_suspect`, 0 `unsupported_character_review`, and 0 `comparison_method`. The localizer evaluates all 115 remaining mismatches and finds 0 target-localized cases (0 `U+2808 ⠈ -> U+2832 ⠲`); current raw-to-residual target count is 24 -> 21, while `U+2832 ⠲ -> U+2808 ⠈` remains 0 raw / 0 residual. Cohort exact increases by 273 and corpus-wide exact increases by the same 273, from 67,442 to 67,715. Because every changed input must satisfy this shared predicate and the baseline had no exact member, this boundary has no exact regression. The remaining 529 candidates differ elsewhere or retain an existing comparison, corpus-suspect, unsupported, or pending cause. Representative shard/index samples are retained above. + +### Ampersand before an attached ASCII Roman segment + +This is the residual boundary not covered by the implemented `A&B` gate. Official UEB 3.1.1 (2024 UEB PDF p.51, printed p.23) prints `&c (etc)` as `⠈⠯⠉ ⠐⠣⠑⠞⠉⠐⠜`, with no mode break between the ampersand and `c`; its `AT&T` and `B&B` examples give the same attached behavior between Roman segments. Korean rule 71 (2024 Korean-rules PDF pp.51-52, printed pp.45-46) wraps an ampersand in Roman indicators when needed to distinguish it from Hangul, while rules 29 and 32 require one Roman section for consecutive Roman material and UEB transcription inside that section. The spaced Korean control `종이접기 & 클레이아트` remains an independently closed Rule-71 symbol. + +The diagnostic checkpoint baseline was 30 candidates / 0 corpus exact / 30 mismatch, preserving 26 `pending_rule_review` and 4 `corpus_suspect` primary classes. Its then-current Rule-71 exit localizer found 16/30 first differences: 15 `U+2820 ⠠ -> U+2832 ⠲` and 1 `U+2834 ⠴ -> U+2832 ⠲`; both localized reverses were zero. The official full-encoder `&c`, `AT&T`, and `B&B` examples were the positive controls, and the spaced Korean Rule-71 example was the negative boundary control. + +The implemented rule is limited to an ampersand followed by a complete attached ASCII-letter segment, with no left ASCII alphanumeric or adjacent ampersand and no trailing digit continuation. Rule 71 opens one Roman section before `&`; rule 29 now leaves it open for the attached letters. It does not name a corpus input or inspect a reference. After the change, the cohort has 30 candidates, 14 exact and 16 mismatch; the corpus-wide exact total rises by the same 12 cases, from 68,175 to 68,187, so no exact regression occurs inside or outside this gate. The 16 former exit/re-entry transitions disappear; 12 become exact and 4 remain mismatches at a different PDF-conflicting boundary. Existing mismatch primary classes remain 12 `pending_rule_review`, 4 `corpus_suspect`, 0 `comparison_method`, and 0 `unsupported_character_review`. + +Of 16 current evaluable mismatches, 4 are localized to the occurrence-specific entry signature: 3 `U+2808 ⠈ -> U+2834 ⠴` where a parenthesized `&TEAM` reference omits Rule 71's required Roman indicator, and 1 `U+2834 ⠴ -> U+2820 ⠠` where a `드림&Dream` reference inserts another Roman indicator inside the same continuous section. Their raw-to-residual counts are 22 -> 19 and 26 -> 25; the corresponding reverse maps are 0 -> 0 and 61 -> 47. These four cases remain conservative corpus/PDF-reference review rather than widening or undoing the rule. Exact samples such as `&TEAM`, `과학&ICT`, and `한국&K리츠`, with shard/index above, are current controls. Primary classifications are never changed by this cohort. + +### ASCII apostrophe between Roman letter runs + +This output-localized cohort requires a straight ASCII apostrophe with an ASCII letter immediately on both sides. UEB 8.4.2 (2024 UEB PDF pp.118-119, printed pp.90-91) directly prints `O'Hara`, `DON'T`, and `THAT'S` with the apostrophe cell inside the same Roman word; capitals-word mode may end at the apostrophe, but the Roman section itself does not. Detached quotation marks, Korean single quotation marks, digit-adjacent measurement signs, and an apostrophe at the end of one whitespace-delimited token before another Roman word are excluded. + +The diagnostic checkpoint had 147 candidates / 0 exact / 147 mismatch. Its occurrence-specific localizer put 112 first differences on the apostrophe boundary: 30 expected apostrophe cell `U+2804 ⠄` versus actual capital indicator `U+2820 ⠠`, and 82 expected `U+2804 ⠄` versus actual Roman indicator `U+2834 ⠴`; neither target had a localized reverse. The implementation keeps only a same-token apostrophe with ASCII letters immediately on both sides in the current Roman section, delegates its cell to the existing UEB section-7 punctuation encoder, and restarts capitals mode for an uppercase run after the nonalphabetic apostrophe. Korean Rule 37 still suppresses whole-word contractions at a Roman entry. A rejected broader route made the detached `Guitar' Listening` control exact, so the final gate explicitly does not look through whitespace. + +After the correction the cohort has 147 candidates, 106 exact controls, and 41 mismatches. Of 41 evaluable residual mismatches, 3 place their first difference inside the independently encoded current signature, but those residual transitions are other letter/spacing differences rather than either former apostrophe transition. The complete corpus exact-ID audit found 68 newly exact cases and zero formerly exact cases lost, raising the corpus total from 69,291 to 69,359. Cohort membership itself never rewrites a primary class; the engine result may make a member exact or expose an independently classified residual. + +### Spaced comma between ASCII digit runs + +This output-localized cohort requires a comma immediately after an ASCII digit, one or more following whitespace characters, and another ASCII digit. Korean rule 41 (2024 Korean-rules PDF p.33, printed p.27) assigns `⠂` only when the comma is *attached* between digits, as in the exact control `9,375명`; rule 49 (PDF pp.37-38, printed pp.31-32) assigns the ordinary Korean comma `⠐`, illustrated by `근면, 검소, 협동은 ...`. Rule 33 (PDF p.28, printed p.22) independently keeps Korean punctuation at Roman-to-Korean boundaries. By contrast, official UEB 7 (2024 UEB PDF p.103, printed p.75) defines its prose comma as `⠂`, and UEB 6.2.1 (PDF p.94, printed p.66) retains `⠂` inside attached numeric forms such as `3,500`. Those two surfaces are negative controls and are excluded by this gate. + +The diagnostic baseline was 217 candidates / 7 exact / 210 mismatch, with 177 occurrence-specific `U+2810 ⠐ -> U+2802 ⠂` first differences and no localized reverse. Rule 41 had looked through `remaining_words`, incorrectly treating whitespace as if the following digit were attached. The implementation now inspects only the next character in the same token. It neither names a corpus input nor consults expected output; attached numbers and UEB punctuation remain owned by their existing routes. + +After the correction, the cohort has 217 candidates / 195 exact / 22 mismatch. Existing mismatch primaries remain 22 `pending_rule_review`, 0 `corpus_suspect`, 0 `comparison_method`, and 0 `unsupported_character_review`. Of 22 evaluable current mismatches, 1 localize to the comma cell: 0 `U+2810 ⠐ -> U+2802 ⠂` and 1 `U+2802 ⠂ -> U+2810 ⠐`. The cohort gains 174 exact cases. Across all pending cases, the current raw-to-residual counts are 6 -> 6 for the target and 39 -> 38 for the reverse. The standard controls `9,375명`, `창세기 12,1-9`, and `근면, 검소, 협동은 ...` retain their PDF cells. No primary class is changed by the cohort. + +### ASCII/Roman-tail comma before a digit-led Korean token + +This companion cohort is disjoint from the preceding digit-comma gate: the comma is immediately preceded by an ASCII letter, followed after whitespace by a token that starts with a digit and contains Korean script. Korean rule 33 (2024 Korean-rules PDF p.28, printed p.22) says that punctuation with different UEB and Korean cells, including comma, is written as Korean punctuation at a Roman-to-Korean boundary and suppresses the Roman terminator. Rule 49 supplies `⠐`; UEB 7 supplies `⠂` only while the comma remains inside English text. Requiring Korean script in the right token is therefore the negative control against reclassifying an English date or number sequence from surface punctuation alone. + +The diagnostic baseline was 58 candidates / 0 exact / 58 mismatch. Of those, 23 had `U+2810 ⠐ -> U+2802 ⠂` at the comma inside the independently encoded complete boundary signature and none had the reverse. The same rule-41 correction removes the cross-token ASCII-letter lookup; rule 33 and the existing English-symbol route then choose the punctuation from the actual surrounding scripts. + +After the correction, this cohort has 58 candidates / 34 exact / 24 mismatch. Existing mismatch primaries remain 20 `pending_rule_review`, 4 `corpus_suspect`, 0 `comparison_method`, and 0 `unsupported_character_review`. Of 24 evaluable current mismatches, 0 localize to the comma-cell signature: 0 `U+2810 ⠐ -> U+2802 ⠂` and 0 `U+2802 ⠂ -> U+2810 ⠐`. Three cases become exact. The official rule-33 `KTX, 새마을호` boundary and UEB prose comma remain independent standard controls. The detector and localizer do not read expected output to choose a route, and membership does not change a primary class. + +### Percent-point unit list comma + +Korean rule 69 attachment 2 (2024 Korean-rules PDF p.50, printed p.44) explicitly defines `%p` as the percent-point unit. This cohort requires two complete numeric `%p` tokens separated by comma plus whitespace, so rule 49's ordinary Korean comma is the punctuation boundary; it does not infer arbitrary ASCII suffixes as units. The diagnostic baseline was 7 candidates / 0 exact / 7 mismatch, with one localized `U+2810 ⠐ -> U+2802 ⠂` and no reverse. + +After the correction, this cohort has 7 candidates / 5 exact / 2 mismatch, preserving 2 `pending_rule_review`, 0 `corpus_suspect`, 0 `comparison_method`, and 0 `unsupported_character_review` mismatch primaries. Of 2 evaluable current mismatches, 0 localize to the independently encoded complete unit pair: 0 `U+2810 ⠐ -> U+2802 ⠂` and 0 `U+2802 ⠂ -> U+2810 ⠐`. Five cases become exact; the other two retain independent earlier differences. One of those five is also in the Roman-tail cohort, leaving four disjoint `%p` gains. Thus +174 in the numeric-list cohort, +3 in the Roman-tail cohort, and +4 disjoint here account for all +181 corpus exact gains, with zero exact-set regressions. The engine contains only the general rule-41 same-token boundary, not a `%p` special case. + +### Attached Korean `있다` spacing normalization + +Korean rule 49 (2024 Korean-rules PDF p.37, printed p.31) says that spacing follows the print input. The PDF consistently retains an explicit space in `그리고 있다` (physical p.18), `살고 있다` (p.26), and `수강하고 있다` (p.30), but it gives no example authorizing a transcriber to insert a missing print space. The current pre-fix token normalizer nevertheless split any Korean token ending in attached `있다`. + +The diagnostic baseline had 95 candidates / 0 exact / 95 mismatch. All 95 were in the exact former implementation scope; 74 first differences were at the inserted blank: 73 `U+2815 ⠕ -> U+2800 ⠀`, one `U+2823 ⠣ -> U+2800 ⠀`, and no localized reverse. The absence of a baseline exact member is the in-scope regression control. + +After removing that input-correcting transformation, the cohort has 95 candidates / 84 exact / 11 mismatch, preserving 8 `pending_rule_review`, 3 `corpus_suspect`, 0 `comparison_method`, and 0 `unsupported_character_review` mismatch primaries. Of 0 evaluable current mismatches, 0 still localize to an inserted blank: 0 `U+2815 ⠕ -> U+2800 ⠀`, 0 `U+2823 ⠣ -> U+2800 ⠀`, and 0 `U+2800 ⠀ -> U+2815 ⠕`. Seventy-one cases become exact; the other 24 retain independent earlier differences. Corpus exact increases by the same 71 with zero exact-set regressions. The three explicitly spaced PDF forms remain full-encoder negative controls, so removing correction of missing input whitespace does not remove a printed space. The local rule-47 standard case had accidentally transcribed PDF physical p.36 `덮여 있다` as attached `덮여있다` while retaining the PDF's spaced braille; correcting that input transcription restores the complete 5,141/5,141 standard summary without an engine exception. + +Current uppercase-Roman hyphen-digits measurement: 571 candidates, 504 exact controls, 67 mismatches, 50 members in the actual `pending_rule_review` subcluster, and 0/67 evaluable mismatches whose first difference is inside the target run plus its entry boundary. It remains distinct from parenthesized digits and headword expansions; no engine change is inferred. + +Current uppercase alphanumeric Roman-digit sequence measurement: 3429 candidates, 2861 exact controls, 568 mismatches, 462 members in the actual `pending_rule_review` subcluster, and 9/568 evaluable mismatches whose first difference is localized to the current entry cells. The target `⠴ -> blank` occurs 9 times and the reverse occurs 0 times. Korean rule 35 (2024 Korean-rules PDF pp.29-30, printed pp.23-24) explicitly treats `MP3`, `A4`, `KF94`, and `D-100` as Roman-number sequences. Math rules 11/12 (PDF pp.63-65, printed pp.57-59) separately route mathematical Roman notation without the prose Roman indicator and with two-cell spacing. Official UEB 6.5.1-6.5.2 (2024 UEB PDF p.96, printed p.68) determines numeric/grade-1 continuation only after the surrounding mode has been chosen. Thus surfaces such as `A-3` cannot be assigned identifier or variable semantics from this structure alone. Exact controls and both directions are retained; no engine change is inferred. + +Current consecutive ASCII-Roman word-boundary measurement: 4679 candidates, 3314 exact controls, 1365 mismatches, 1251 members in the actual `pending_rule_review` subcluster, and 5/1365 evaluable mismatches whose first difference is localized to the current boundary cell. The target reference blank versus current terminator (`⠀ -> ⠲`) occurs 4 times; the exact reverse (`⠲ -> ⠀`) occurs 0 times. Korean rule 29 (2024 Korean-rules PDF p.26, printed p.20) explicitly treats consecutive Roman items as one Roman section, with one Roman indicator before the first item and one terminator after the last; its printed `Los Angeles` and `Table of Contents` examples are exact controls. Rule 34 (PDF p.29, printed p.23) separately omits the terminator before a closing enclosure, while official UEB 9.7.1 (2024 UEB PDF p.137, printed p.109) prints the multiword prose enclosure `plays (such as Romeo and Juliet)` with ordinary internal spaces and nested closing punctuation. Together they support preserving a complete letter-and-space Roman parenthetical as prose, but not a fragment containing a function-call opener, digit, operator, or nested bracket. The input-only gate does not decide whether punctuation-separated text such as `ESS /VPP` continues the same section, so that variant remains outside this cohort. Primary classes are preserved. Before the narrow token-routing change this cohort was 4,679 candidates / 2,058 exact / 2,621 mismatch, with 128 localized `⠀ -> ⠲` and zero reverse. The cause was the final `Letters)` token of a whitespace-split Roman parenthetical being rerouted as math, which made the preceding Roman word terminate early and introduced an extra blank. After excluding only a backwards-verified complete Roman parenthetical tail from math routing, the cohort is 2,132 exact / 2,547 mismatch, with 23 target-localized and zero reverse; corpus-wide exact likewise rises by 74, from 68,101 to 68,175. The 105 removed localized boundaries include 74 newly exact cases and 31 cases that still differ elsewhere. Remaining targets such as `SYNO PEM-1`, `ACE Fair(2020)`, and `Mnet K-POP` do not satisfy the complete-parenthetical gate and remain separate residuals rather than widening this rule. + +Current single-capital parenthesized-digits measurement: 1361 candidates, 1351 exact controls, 10 mismatches, 6 members in the actual `pending_rule_review` subcluster, and 4/10 evaluable mismatches whose first difference is inside the target run plus its entry boundary. No engine change is inferred from the ambiguous prose/function surface form. + +Current mixed Roman+Korean boundary before uppercase headword-expansion measurement: 10 candidates, 7 exact controls, 3 mismatches, 3 members in the actual `pending_rule_review` subcluster, and 2/3 evaluable mismatches whose first difference is localized to the later headword's entry boundary/output. The detector cannot be satisfied by the earlier Roman entry. The narrow rules-29/34 headword-expansion route is active; these residuals therefore identify a separate state or Roman-letter difference. + +Current compact numeric+ASCII-suffix measurement: 2975 candidates, 2492 exact controls, 483 mismatches, 423 members in the actual `pending_rule_review` subcluster, and 41/483 evaluable mismatches whose first difference is inside the complete current-engine output signature or its immediate entry boundary. Rule 40 requires the numeric indicator and rule 69 requires Roman indicators around a Roman-written unit, but the input shape alone cannot prove that every ASCII suffix is a unit. + +| ASCII suffix | Candidates | Exact | Mismatch | Localized first diff | +|---|---:|---:|---:|---:| +| `m` | 364 | 312 | 52 | 4 | +| `km` | 308 | 286 | 22 | 2 | +| `G` | 237 | 212 | 25 | 1 | +| `kg` | 209 | 189 | 20 | 3 | +| `D` | 167 | 133 | 34 | 4 | +| `p` | 140 | 111 | 29 | 1 | +| `g` | 138 | 116 | 22 | 3 | +| `t` | 128 | 115 | 13 | 1 | +| `M` | 104 | 96 | 8 | 0 | +| `cm` | 69 | 58 | 11 | 0 | +| `B` | 62 | 60 | 2 | 0 | +| `GB` | 61 | 48 | 13 | 3 | +| `GWh` | 57 | 53 | 4 | 0 | +| `ha` | 53 | 51 | 2 | 0 | +| `TV` | 52 | 45 | 7 | 1 | +| `S` | 45 | 44 | 1 | 0 | +| `GW` | 44 | 42 | 2 | 0 | +| `X` | 39 | 25 | 14 | 0 | +| `bp` | 39 | 24 | 15 | 0 | +| `K` | 38 | 21 | 17 | 0 | +| `MW` | 37 | 36 | 1 | 0 | +| `TURN` | 33 | 21 | 12 | 0 | +| `L` | 31 | 31 | 0 | 0 | +| `mm` | 29 | 24 | 5 | 1 | +| `egin` | 28 | 28 | 0 | 0 | + +Entry-boundary pre-fix baseline: 2,975 candidates, 1,649 exact controls, 1,326 mismatches, 1,259 pending members, and 356 localized first differences; the dominant reference number-sign versus current space transition accounted for 296 cases. Rules 68 and 69 already accept semantic Unicode compatibility-unit forms. The implementation derives compact ASCII spellings only from their all-letter NFKC decompositions, reuses the owning rule's PDF-defined cells, chooses the longest complete spelling, and rejects partial suffix matches. It does not extend recognition to separated English words or arbitrary corpus suffixes. The same cohort now has 1,759 exact controls, 1,216 mismatches, 1,148 pending members, and 261 localized first differences; corpus-wide exact matches moved from 66,436 to 66,546 (+110). Rule 69's printed `160㎎/㎗` example directly controls `160mg`; the additional full-encoder `240mg`/`240㎎` pair proves that recognition is invariant to the numeric value and also passes. Rule 68's printed `10,000㎡는 1㏊이다` example controls the `ha` spelling through U+33CA's NFKC decomposition; `15.2ha`/`15.2㏊` now passes without a spelling-specific output branch. The 53-case `ha` suffix control moved from 0 exact to 18 exact; its remaining 35 mismatches have independent later or surrounding differences. A full U+3300..U+33FF owner audit found 73 Rules 68/69 glyphs with pure-ASCII NFKC decompositions, 73 distinct spellings, and therefore no current duplicate-spelling owner collision. Production nevertheless groups every owner before resolution and excludes a spelling if any owner cells differ; a synthetic collision test proves that this is not first-wins behavior. An exhaustive test also compares every one of the 73 derived spellings with every owner glyph at both the unit-cell and full-encoder boundaries in a neutral Korean measurement context. That audit exposed the pre-existing explicit `cal` mapping's missing ordinary terminator; rule 69 requires the terminator, while the existing slash-boundary function removes it for the printed `cal/㎠/min` context. Restoring the ordinary terminator and extending that slash-continuation boundary made the audit pass without changing the corpus-wide 66,546 exact total. Ambiguous pure-English inputs remain on UEB: the standard controls `3m` and `4.m` are retained and pass, rather than being globally forced into a Korean unit route. + +Current rule-69 ASCII-unit punctuation-boundary measurement: 440 candidates, 385 exact controls, 55 mismatches, 53 members in the actual `pending_rule_review` subcluster, and 3/55 evaluable mismatches whose first difference is localized to the unit-plus-punctuation output signature. PDF rule 69 requires a Roman terminator after a Roman-written unit in the ordinary case, while rules 33/34 omit it at the listed punctuation or enclosing-mark boundary; the rule 46 PDF example `체중(kg)` is the minimal parenthesized-unit control. Membership is restricted to rule-69 spellings already supported by the engine and does not infer unit semantics for arbitrary ASCII suffixes. + Analyzer pre-fix baseline: 440 candidates, 0 exact controls, 440 mismatches, 435 pending members, and 9 signature-local first differences. After the generalized rule-33/34 boundary override and the matching non-math routing guard, the same cohort has 325 exact controls and 115 mismatches. Corpus-wide exact matches moved from 66,039 to 66,436 (+397); the additional gains are applications of the same boundary rule outside this strict ASCII detector, including compatibility-unit forms. The complete standard suite remains 5,141/5,141. + +Current decimal-point measurement: 4546 candidates, 4106 exact controls, 440 mismatches, 401 members in the actual `pending_rule_review` subcluster, and 127/440 evaluable mismatches whose first difference is inside the complete decimal-containing word's current-engine signature. Hangeul rules 43 and 48 keep an ASCII point between digits in the numeric sequence and encode it as the decimal point; rules 35 and 69 supply controls for adjacent Roman-number chains and Roman units. This is an implementation-candidate audit, not permission to specialize on a corpus reference. +Analyzer checkpoint `ed20f99` recorded the pre-fix baseline as 4,546 candidates, 2,584 exact controls, 1,962 mismatches, 1,905 pending members, and 875 localized first differences. Its dominant localized transition was reference decimal point `U+2832 ⠲` versus current Roman indicator `U+2834 ⠴` in 647 cases. After the generalized rule-43/48 guard, that transition is absent: 525 cases become exact and the other corrected prefixes expose later independent mismatches. + +Current all-caps `OU` measurement: 1816 candidates, 128 exact controls, 1688 mismatches, 1679 members in the actual `pending_rule_review` subcluster, and 1506/1688 evaluable mismatches whose first difference is inside the current-engine output signature for the detected run. This is a pronunciation-sensitive UEB review cohort, not an engine routing rule. + +Current standalone-uppercase measurement: 62411 candidates, 55551 exact controls, 6860 mismatches, and 5659 members in the actual `pending_rule_review` subcluster. Its high frequency does not make it causal: the same input shape is exact in many cases, and a sentence containing the shape may first differ at another Roman, numeric, or punctuation structure. No engine change is inferred from this cohort. + +Current Korean-prefixed all-caps parenthetical measurement: 54492 candidates, 48892 exact controls, 5600 mismatches, and 4549 members in the actual `pending_rule_review` subcluster. This is a semantic-collision audit, not an engine routing rule; no implementation change is inferred from its reference outputs. + +Current rule-34 opening-order measurement: 64382 structural candidates, 57748 exact controls, 6634 mismatches, and 1166/6634 evaluable mismatches whose first difference is inside the current engine's Korean opening-parenthesis cells. Only the 1163 localized first-cell transitions have the reference/current order `⠴` versus `⠦`. After requiring the complete reference `⠴⠐⠣` versus current/PDF `⠦⠄⠴` three-cell signature and preserving higher-priority comparison classifications, 1160 are classified with the dedicated rule-34 contradiction reason; mere coexistence with a Korean-prefixed Roman annotation does not change a primary class. + +NIKL Q&A #325 clarifies that the six lower wordsigns named by Korean Rule 37 remain expanded when Roman words are discussed in Korean context, while a recognizable English title or phrase follows UEB 10.5 and uses a lower wordsign only when it stands alone and satisfies the lower-sign adjacency restriction. The encoder distinguishes those contexts from input structure, capitalization, and enclosure boundaries; analyzer references and competitor fields do not affect routing. + +Current UEB non-standing parenthesis/grade-1 contradiction measurement: 70 cases contain an all-capitals letters-sequence that resembles a shortform but is followed immediately by an opening round, square, or curly parenthesis. UEB 2.6.2 permits those opening symbols before a standing-alone sequence, while 2.6.3 does not permit them after one. The classifier requires complete-sentence equality after removing only reference-side grade-1 cells immediately before the localized capitals indicators; all other differences remain pending review. + +Current UEB capitalized-passage contradiction measurement: 3 cases contain at least three consecutive capitalized symbols-sequences and differ from the current UEB 8.5.2-8.5.3 path only by replacing the one passage indicator/terminator pair with separate one- or two-cell capitalization indicators. The classifier requires equality of the complete sentence after deleting exactly those structurally counted separate indicators; unrelated Roman, punctuation, contraction, or spacing differences remain pending review. + +Current all-caps Roman middle-dot measurement: 97 candidates, 0 exact controls, 97 mismatches, and 96 members in the actual `pending_rule_review` subcluster. This cross-cutting cohort preserves every primary class and is not an engine routing rule. + +Current Roman-before-middle-dot boundary measurement: 577 candidates, 0 exact controls, 577 mismatches, 570 members in the actual `pending_rule_review` subcluster, and 516/577 evaluable mismatches whose first difference is localized to the one current terminator immediately before the middle dot. The locator encodes each real input prefix ending at the dot, so identifier state such as `K-ICS·...` is retained without consulting expected. This raises localized target coverage from 342 to 417 and reduces the then-leading residual `⠐ -> ⠲` transition from 297 to 223. Rules 29, 33, and 50 support the current terminator path but do not support the localized reference omission; no engine change or primary-class rewrite is inferred. + +Current attached Roman-to-Korean boundary measurement: 17693 candidates, 15206 exact controls, 2487 mismatches, 1719 members in the actual `pending_rule_review` subcluster, and 0/2487 evaluable mismatches whose first difference is localized to the current boundary marker. The localized target `⠲ -> ⠸` occurs 0 times and the reverse `⠸ -> ⠲` occurs 0 times. Korean rule 29 (2024 Korean-rules PDF p.26, printed p.20) requires a Roman terminator around Roman text in a Korean sentence, whereas rule 39 (PDF pp.31-32, printed pp.25-26) applies the Korean opening/closing markers only when Roman text is the sentence's main language. Among exact candidates, 13004 expose the current rule-29 terminator at such a boundary and 0 expose the current rule-39 opening. Its printed controls include both an English sentence (`What is 김치 in English?`) and the address `www.대통령.kr` inside a Korean sentence. Official UEB 2.4.7 (2024 UEB PDF p.41, printed p.13) confirms that a UEB mode does not extend through a switch to another braille code, but does not choose whether this Korean boundary belongs to a Korean-main or Roman-main context. Therefore the attached script boundary alone cannot distinguish ordinary Korean prose from an embedded Roman-domain context. Primary classes are preserved and no engine change is inferred without a narrower input-derived dominance gate. At diagnostic checkpoint `57c4608`, this same cohort was 17,693 candidates / 12,724 exact / 4,969 mismatch, with 268 localized `⠲ -> ⠸`, zero reverse, 10,780 exact rule-29 markers, and zero exact rule-39 markers. After narrowing the engine by dominance and the PDF domain shape, it is 12,924 exact / 4,769 mismatch with no localized target or reverse; corpus-wide exact rises from 67,715 to 68,101 (+386). The remaining 0 mismatch with a current rule-39 opening is a list-heavy Korean sentence whose ASCII-leading brand tokens satisfy the mechanical word-count majority; its first difference is not localized to this boundary, so it remains a semantic pending control rather than grounds for another engine branch. + +Current rule-39 narrowed-scope audit: 947 candidates, 736 exact controls, 211 mismatches, and 201 members in the actual `pending_rule_review` subcluster. This input-derived scope is recorded separately from the direct-boundary output-localizer; primary classes are unchanged. At the implementation checkpoint, this conservative first-script-word approximation contains 385 exact cases, accounting for all but one of the corpus-wide +386 net gain; no localized reverse transition is observed in the direct-boundary cohort. + +Current inline parenthesized-operator measurement: 23 candidates, 22 exact controls, 1 mismatches, 1 members in the actual `pending_rule_review` subcluster, and 0/1 evaluable mismatches whose first differing cell is inside the emitted structure. Primary classes are preserved. + At that implementation checkpoint, the strict cohort moved from 0 to 17 exact cases; the corpus-wide total moved from 65,491 to 65,514 (+23 exact) because the same PDF-backed spacing rule also applied outside the stricter Korean-boundary audit gate. These are immutable checkpoint counts rather than the report's later cumulative total. The complete standard suite remained 5,141/5,141. + +Current attached plus + parenthesized Korean-gloss measurement: 16 candidates, 4 exact controls, 12 mismatches, 12 members in the actual `pending_rule_review` subcluster, and 12/12 evaluable mismatches whose first differing cell is inside the current emitted structure. Hangeul rule 46 supplies the operation-sign spacing control, but the surface form alone does not establish whether a brand or program name uses `+` mathematically. Exact and localized mismatch references coexist for `도전+(플러스)`, so no engine change or primary reclassification is inferred. + +Current tight-triangle measurement: 377 candidates, 317 exact controls, 60 mismatches, 54 members in the actual `pending_rule_review` subcluster, and 3/60 evaluable mismatches whose first difference is inside the `△` plus first-Korean output range. No engine change is inferred. + +## Encoding-error diagnostics + +The audit starts from all raw encoding errors, then separates cases already resolved by a comparison method or corpus contradiction. The message, family, and singleton tables below count only unresolved encoding-error review cases. These diagnostics are not additional primary classes. A singleton unsupported character is a character that also fails when encoded by itself. Such a failure remains a review candidate until the PDF independently establishes support. + +| Encoding-error audit | Cases | +|---|---:| +| Raw encoding errors | 4 | +| Resolved by comparison method | 0 | +| Excluded as corpus suspect | 0 | +| Unresolved encoding-error review cases | 4 | +| Explained by singleton unsupported character(s) | 4 | +| Multiple singleton unsupported characters | 0 | +| Unclassified without a singleton explanation | 0 | + +| Error message | Cases | +|---|---:| +| `Invalid symbol character` | 4 | + +| Error family | Cases | +|---|---:| +| `punctuation_or_layout_symbol` | 4 | + +Families are diagnostics, not automatic normalization permissions. Rules 68/69 compatibility-unit support removed that error family from the current run; `enclosed_organization_mark` and layout symbols still have no confirmed rule. + +| Singleton error character | Cases containing it | NFKC decomposition | Family | +|---|---:|---|---| +| `U+260F ☏` | 3 | `☏` | `punctuation_or_layout_symbol` | +| `U+2665 ♥` | 1 | `♥` | `punctuation_or_layout_symbol` | + +## Shards + +| Shard | Exact | Total | Accuracy | +|---|---:|---:|---:| +| `sentence_01.json` | 22505 | 25000 | 90.02% | +| `sentence_02.json` | 22630 | 25000 | 90.52% | +| `sentence_03.json` | 22855 | 25000 | 91.42% | +| `sentence_04.json` | 7795 | 8528 | 91.40% | + +## Overlapping mismatch traits + +| Trait | Count | +|---|---:| +| `contains_ascii_digits` | 5759 | +| `contains_ascii_letters` | 7694 | +| `contains_delimiter_or_quote` | 7743 | +| `input_not_nfkc` | 243 | + +## Samples + +### `foreign_text_rule_review` + +- `sentence_01.json` #39: 소프트웨어정책연구소(SPRi)는 ‘2023년 SW산업 10대 이슈 전망’을 통해 올해 가장 주요한 이슈로 인공지능 기반 모델 고도화를 1위로 선정했다. + - expected: `⠈⠍⠠⠥⠦⠄⠴⠠⠠⠎⠏⠠⠗⠊⠠⠴⠉⠵⠀⠠⠦⠼⠃⠚` + - actual: `⠈⠍⠠⠥⠦⠄⠴⠠⠎⠠⠏⠠⠗⠊⠠⠴⠉⠵⠀⠠⠦⠼⠃⠚` +- `sentence_01.json` #47: 다날은 계열사 ‘제프’가 국내 대체불가토큰(NFT) 거래소를 운영하는 ‘팔라’와 메타버스·NFT 협력 관련 협약(MOU)을 맺고 메타버스 플랫폼 ‘제프월드’의 인프라 확대를 추진한다고 3일 밝혔다. + - expected: `⠜⠁⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠮⠀⠑⠗⠅⠈⠥⠀⠑⠝⠓⠘` + - actual: `⠜⠁⠦⠄⠴⠠⠠⠍⠳⠠⠴⠮⠀⠑⠗⠅⠈⠥⠀⠑⠝⠓⠘⠎` +- `sentence_01.json` #50: 부산광역시는 3일 오후 부산광역시청 영상회의실에서 종합화학소재기업 ㈜금양과 이차전지 생산기지 건립을 위한 8천억원 규모의 투자 양해각서(MOU)를 체결한다고 밝혔다. + - expected: `⠠⠎⠦⠄⠴⠠⠠⠍⠕⠥⠠⠴⠐⠮⠀⠰⠝⠈⠳⠚⠒⠊⠈⠥` + - actual: `⠠⠎⠦⠄⠴⠠⠠⠍⠳⠠⠴⠐⠮⠀⠰⠝⠈⠳⠚⠒⠊⠈⠥⠀` +- `sentence_01.json` #51: 와이캅 픽셀은 와이어·패키지·렌즈가 필요 없는 ‘와이캅’ 기술을 기반으로 한다. 적녹청(RGB) 3개의 마이크로 LED를 수직 방향으로 적층한 세계 최초의 풀 컬러 원칩 기술이다. + - expected: `⠪⠐⠥⠀⠴⠠⠠⠇⠑⠙⠲⠐⠮⠀⠠⠍⠨⠕⠁⠀⠘⠶⠚⠜` + - actual: `⠪⠐⠥⠀⠴⠠⠠⠇⠫⠲⠐⠮⠀⠠⠍⠨⠕⠁⠀⠘⠶⠚⠜⠶` +- `sentence_01.json` #57: 아키에이지 워는 원작 아키에이지 개발사 엑스엘게임즈(각자대표 송재경, 최관호)가 개발 중인 신작 PC·모바일 크로스플랫폼 대규모다중접속역할수행게임(MMORPG)이다. 원작의 향수를 자극하는 스토리와 캐릭터, 언리얼 엔진4를 활용한 고퀄리티 그래픽이 특징이다. + - expected: `⠨⠁⠀⠴⠠⠠⠏⠉⠐⠆⠑⠥⠘⠣⠕⠂⠀⠋⠪⠐⠥⠠⠪⠙` + - actual: `⠨⠁⠀⠴⠠⠠⠏⠉⠲⠐⠆⠑⠥⠘⠣⠕⠂⠀⠋⠪⠐⠥⠠⠪` + +### `number_rule_review` + +- `sentence_01.json` #845: 부산광역시는 3일 오전 부산광역시청에서 초록우산어린이재단 부산본부, 월드비전 부산사업본부, 굿네이버스 영남본부와 ‘자립+(더하기) 동행 프로젝트’ 업무협약을 체결했다고 밝혔다. + - expected: `⠠⠦⠨⠐⠕⠃⠀⠢⠦⠄⠊⠎⠚⠈⠕⠠⠴⠀⠊⠿⠚⠗⠶⠀` + - actual: `⠠⠦⠨⠐⠕⠃⠀⠢⠀⠦⠄⠊⠎⠚⠈⠕⠠⠴⠀⠊⠿⠚⠗⠶` +- `sentence_01.json` #7774: 한편 이번 조사는 대구시가 ㈜리얼미터에 의뢰해 전화면접·온라인(7:3)을 통해 실시했고 응답률은 15.6%, 표본오차는 95% 신뢰수준에서 ±3.1%포인트다. + - expected: `⠨⠛⠝⠠⠎⠀⠢⠔⠀⠼⠉⠲⠁⠴⠏⠀⠙⠥⠟⠓⠪⠊⠲` + - actual: `⠨⠛⠝⠠⠎⠀⠢⠔⠼⠉⠲⠁⠴⠏⠀⠙⠥⠟⠓⠪⠊⠲` +- `sentence_01.json` #10976: 자살예방 캠페인 캐릭터인 더더(+), 배로(×), 빼요(–), 누미(÷) ‘생명지키미들’의 카카오톡 이모티톤을 18일 오후 2시부터 무료로 배포한다. + - expected: `⠐⠀⠠⠘⠗⠬⠦⠄⠔⠠⠴⠐⠀⠉⠍⠑⠕⠦⠄⠌⠌⠠⠴⠀` + - actual: `⠐⠀⠠⠘⠗⠬⠦⠄⠠⠤⠠⠴⠐⠀⠉⠍⠑⠕⠦⠄⠌⠌⠠⠴` +- `sentence_01.json` #10977: 자살예방 캠페인 캐릭터 ‘생명지키미들’은 총 4종으로, 사랑과 희망을 더해주는 ‘더더(+)’, 행복을 마구마구 불려주는 ‘배로(×)’, 슬픔을 잊게 해주는 ‘빼요(–)’, 걱정과 고민을 듣고 나눠주는 ‘누미(÷)’로 구성돼 있다. + - expected: `⠠⠦⠠⠘⠗⠬⠦⠄⠔⠠⠴⠴⠄⠐⠀⠈⠹⠨⠻⠈⠧⠀⠈⠥` + - actual: `⠠⠦⠠⠘⠗⠬⠦⠄⠠⠤⠠⠴⠴⠄⠐⠀⠈⠹⠨⠻⠈⠧⠀⠈` +- `sentence_01.json` #13879: 탬파베이는 83득점 20실점이다. 득실 마진이 +63점이나 된다. 메이저리그 역사에서 개막 10경기 득실자 마진에서 역대 3위 기록이다. 그런데 1~2위는 앞서 언급된 1884년 마룬스(+106)와 마룬스(+73) 기록이다. + - expected: `⠂⠀⠑⠨⠟⠕⠀⠢⠀⠼⠋⠉⠨⠎⠢⠕⠉⠀⠊⠽⠒⠊⠲⠀` + - actual: `⠂⠀⠑⠨⠟⠕⠀⠢⠼⠋⠉⠨⠎⠢⠕⠉⠀⠊⠽⠒⠊⠲⠀⠑` + +### `punctuation_rule_review` + +- `sentence_01.json` #18647: 배우 박성웅, 오대환, 오달수, 주석태 주연 정통 하드보일드 액션 영화 ‘더와일드:야수들의 전쟁’(감독:김봉한/제작:(주)아센디오, (주)제이앤씨미디어그룹/이하 더와일드)의 개봉 소식과 함께 보도스틸이 공개돼 눈길을 끈다. + - expected: `⠎⠧⠕⠂⠊⠪⠐⠂⠀⠜⠠⠍⠊⠮⠺⠀⠨⠾⠨⠗⠶⠴⠄⠦` + - actual: `⠎⠧⠕⠂⠊⠪⠐⠂⠜⠠⠍⠊⠮⠺⠀⠨⠾⠨⠗⠶⠴⠄⠦⠄` +- `sentence_03.json` #18547: 남태우는 디즈니+(플러스)의 오리지널 시리즈 ‘한강’(연출/극본 김상철)에서 국제범죄 수사대 고형민 경사 역할을 맡아 연기 변신을 선보인다. + - expected: `⠊⠕⠨⠪⠉⠕⠀⠢⠦⠄⠙⠮⠐⠎⠠⠪⠠⠴⠺⠀⠥⠐⠕⠨` + - actual: `⠊⠕⠨⠪⠉⠕⠀⠢⠀⠦⠄⠙⠮⠐⠎⠠⠪⠠⠴⠺⠀⠥⠐⠕` +- `sentence_03.json` #19763: 디즈니+ 드라마 ‘무빙’을 언급하던 한효주 곁에 다가온 조인성은 “굉장히 무서운 와이프였어요~”라고 극 중 아내 자랑(?)을 하며 남편 손님과 눈빛 교환을 하는 모습으로 폭소를 안긴다. + - expected: `⠊⠕⠨⠪⠉⠕⠀⠢⠀⠊⠪⠐⠣⠑⠀⠠⠦⠑⠍⠘⠕⠶⠴⠄` + - actual: `⠊⠕⠨⠪⠉⠕⠀⠀⠢⠀⠊⠪⠐⠣⠑⠀⠠⠦⠑⠍⠘⠕⠶⠴` +- `sentence_03.json` #21665: 그러자 김숙은 “그러면 우재야. ‘홍김동전’ 잠깐 쉬어라”며 급 하차 권유(?)를 하고, 홍진경은 “김치 없냐”면서 느닷없이 김치를 찾는 등 총체적 난국이 펼쳐졌다. + - expected: `⠕⠢⠠⠍⠁⠵⠀⠦⠁⠒⠀⠍⠨⠗⠜⠲⠀⠠⠦⠚⠿⠈⠕⠢` + - actual: `⠕⠢⠠⠍⠁⠵⠀⠦⠈⠪⠐⠎⠑⠡⠀⠍⠨⠗⠜⠲⠀⠠⠦⠚` + +### `roman_ellipsis_uses_korean_cells_in_roman_enclosure` + +- `sentence_01.json` #13772: 이후 아이브는 다양한 노래들로 간식 퀴즈를 진행했고, 최신곡 스테이씨(STAYC)의 ‘파피’(Poppy)부터 터보의 ‘러브 이즈’(Love Is…)까지 맞히며 맛있는 간식들을 획득했다. + - expected: `⠇⠕⠧⠑⠀⠠⠊⠎⠠⠠⠠⠠⠴⠠⠫⠨⠕⠀⠑⠅⠚⠕⠑⠱` + - actual: `⠇⠕⠧⠑⠀⠠⠊⠎⠲⠲⠲⠠⠴⠠⠫⠨⠕⠀⠑⠅⠚⠕⠑⠱` + +### `rule34_roman_indicator_before_opening_parenthesis` + +- `sentence_01.json` #14: 새롭게 선보인 DRX 브랜딩과 팀 컬러는 2023년 1월 1일부터 DRX의 온오프라인 콘텐츠, 엠디(MD), SNS, 유튜브 등 모든 제작물에 적용될 예정이다. + - expected: `⠰⠪⠐⠀⠝⠢⠊⠕⠴⠐⠣⠠⠠⠍⠙⠐⠜⠂⠀⠠⠠⠎⠝⠎` + - actual: `⠰⠪⠐⠀⠝⠢⠊⠕⠦⠄⠴⠠⠠⠍⠙⠠⠴⠐⠀⠴⠠⠠⠎⠝` +- `sentence_01.json` #35: 삼성전자는 올해 CES에서 77인치 OLED TV를 처음으로 공개할 예정이다. 이는 지난 8월 ‘국제정보디스플레이학술대회(IMID) 2022’에서 삼성디스플레이가 공개한 77인치 퀀텀닷(QD)-OLED 패널을 탑재한 제품이다. + - expected: `⠋⠏⠒⠓⠎⠢⠊⠄⠴⠐⠣⠟⠙⠐⠜⠤⠠⠠⠕⠇⠫⠲⠀⠙` + - actual: `⠋⠏⠒⠓⠎⠢⠊⠄⠦⠄⠴⠠⠠⠟⠙⠠⠴⠤⠴⠠⠠⠕⠇⠫` +- `sentence_01.json` #77: 삼성전자가 미국 라스베이거스에서 열리는 세계 최대 전자 전시회 ‘CES 2023’ 개막을 앞두고 77인치 유기발광다이오드(OLED) TV 등 2023년형 TV 신제품을 대거 공개했다. + - expected: `⠧⠶⠊⠣⠕⠥⠊⠪⠴⠐⠣⠠⠠⠕⠇⠫⠐⠜⠀⠠⠠⠞⠧⠲` + - actual: `⠧⠶⠊⠣⠕⠥⠊⠪⠦⠄⠴⠠⠠⠕⠇⠫⠠⠴⠀⠴⠠⠠⠞⠧` +- `sentence_01.json` #79: B씨의 신고로 수사에 착수한 경찰은 인근 폐쇄회로(CC)TV와 탐문수색 등을 바탕으로 A씨를 특정해 지난달 29일 검거해 31일 구속했다. + - expected: `⠌⠠⠧⠗⠚⠽⠐⠥⠴⠐⠣⠠⠠⠉⠉⠐⠜⠠⠠⠞⠧⠲⠧⠀` + - actual: `⠌⠠⠧⠗⠚⠽⠐⠥⠦⠄⠴⠠⠠⠉⠉⠠⠴⠴⠠⠠⠞⠧⠲⠧` +- `sentence_01.json` #83: 삼성전자는 이번 행사에서 77형 유기발광다이오드(OLED) TV를 첫 공개하기도 했다. 지난해 처음 출시한 삼성 OLED TV는 55형, 65형과 함께 초대형 77형 모델이 추가된 셈이다. + - expected: `⠧⠶⠊⠣⠕⠥⠊⠪⠴⠐⠣⠠⠠⠕⠇⠫⠐⠜⠀⠠⠠⠞⠧⠲` + - actual: `⠧⠶⠊⠣⠕⠥⠊⠪⠦⠄⠴⠠⠠⠕⠇⠫⠠⠴⠀⠴⠠⠠⠞⠧` + +### `ueb_capitalized_passage_written_as_separate_capital_words` + +- `sentence_01.json` #12980: 민희가 속한 크래비티는 오늘(14일) 오후 6시에 SBS M, SBS FiL ‘더쇼’에서 미니 5집 ‘마스터 : 피스’의 타이틀곡 ‘그루비(Groovy)’로 컴백 무대를 펼친다. + - expected: `⠋⠠⠕⠝⠀⠴⠠⠠⠎⠃⠎⠀⠰⠠⠍⠂⠀⠠⠠⠎⠃⠎⠀⠠` + - actual: `⠋⠠⠕⠝⠀⠴⠠⠠⠠⠎⠃⠎⠀⠰⠍⠂⠀⠎⠃⠎⠠⠄⠀⠠` +- `sentence_03.json` #16532: IPX(구 라인프렌즈)의 글로벌 인기 캐릭터 IP BT21이 방탄소년단(BTS) 데뷔 10주년 ‘FESTA(2023 BTS FESTA)’에 특별 참여한다. + - expected: `⠉⠡⠀⠠⠦⠴⠠⠠⠋⠑⠌⠁⠐⠣⠼⠃⠚⠃⠉⠀⠠⠠⠃⠞` + - actual: `⠉⠡⠀⠠⠦⠴⠠⠠⠠⠋⠑⠌⠁⠐⠣⠼⠃⠚⠃⠉⠀⠃⠞⠎` +- `sentence_03.json` #20410: 에이핑크는 올해 4월 미니 10집 ‘셀프(SELF)’를 내고 타이틀곡 ‘D N D’로 활동했다. 최근에는 크리스마스 음원도 발매했다. 멤버 개인 활동도 병행했다. + - expected: `⠮⠈⠭⠀⠠⠦⠴⠠⠙⠀⠰⠠⠝⠀⠰⠠⠙⠴⠄⠐⠥⠀⠚⠧` + - actual: `⠮⠈⠭⠀⠠⠦⠴⠠⠠⠠⠙⠀⠰⠝⠀⠰⠙⠠⠄⠴⠄⠐⠥⠀` + +### `ueb_grade1_before_nonstanding_opening_parenthesis` + +- `sentence_01.json` #1573: 넥슨(대표 이정헌)이 글로벌 게임개발자 콘퍼런스 ‘GDC(Game Developers Conference) 2023’에 참가해 대규모다중접속역할수행게임(MMORPG)과 블록체인 기술간 결합을 주제로 강연을 진행한다고 3일 발표했다. + - expected: `⠐⠾⠠⠪⠀⠠⠦⠴⠰⠠⠠⠛⠙⠉⠐⠣⠠⠛⠁⠍⠑⠀⠠⠙` + - actual: `⠐⠾⠠⠪⠀⠠⠦⠴⠠⠠⠛⠙⠉⠐⠣⠠⠛⠁⠍⠑⠀⠠⠙⠑` +- `sentence_01.json` #5005: “정부 전용 초거대AI 개발에 민간 기업의 LLM(초거대언어모델)을 활용하는데 정부 내부 문서나 데이터는 별도로 분리된 공간에서 안전하게 학습시킬 것이다.” + - expected: `⠀⠈⠕⠎⠃⠺⠀⠴⠰⠠⠠⠇⠇⠍⠦⠄⠰⠥⠈⠎⠊⠗⠾⠎` + - actual: `⠀⠈⠕⠎⠃⠺⠀⠴⠠⠠⠇⠇⠍⠦⠄⠰⠥⠈⠎⠊⠗⠾⠎⠑` +- `sentence_01.json` #8272: 배우 이선균(48)과 그룹 빅뱅 출신 가수 GD(35·권지용)가 마약 투약 혐의로 입건된 가운데 이들의 마약 공급책은 의사인 것으로 드러났다. + - expected: `⠠⠟⠀⠫⠠⠍⠀⠴⠰⠠⠠⠛⠙⠦⠄⠼⠉⠑⠐⠆⠈⠏⠒⠨` + - actual: `⠠⠟⠀⠫⠠⠍⠀⠴⠠⠠⠛⠙⠦⠄⠼⠉⠑⠐⠆⠈⠏⠒⠨⠕` +- `sentence_01.json` #8459: 롯데정보통신은 자사가 개발한 대화형 인공지능(AI) 모델이 한국지능정보사회진흥원(NIA)와 업스테이지가 공동으로 주최하는 한국어 언어모델 리더보드인 ‘오픈 코-LLM(Open Ko-LLM)’에서 1위를 달성했다고 1일 밝혔다. + - expected: `⠥⠙⠵⠀⠋⠥⠤⠴⠰⠠⠠⠇⠇⠍⠐⠣⠠⠕⠏⠢⠀⠠⠅⠕` + - actual: `⠥⠙⠵⠀⠋⠥⠤⠴⠠⠠⠇⠇⠍⠐⠣⠠⠕⠏⠢⠀⠠⠅⠕⠤` +- `sentence_01.json` #8525: 이어 “재정지출을 늘려 성장률 상승이 물가 상승을 따라잡을 수 있으면 실질적 GDP(국내총생산)은 늘어난다”며 “이런 문제를 단선적으로 접근하는 것이 정부의 근본적 문제”라고 지적했다. + - expected: `⠂⠨⠕⠂⠨⠹⠀⠴⠰⠠⠠⠛⠙⠏⠦⠄⠈⠍⠁⠉⠗⠰⠿⠠` + - actual: `⠂⠨⠕⠂⠨⠹⠀⠴⠠⠠⠛⠙⠏⠦⠄⠈⠍⠁⠉⠗⠰⠿⠠⠗` + +### `unsupported_character_review` + +- `sentence_02.json` #6204: 지방세와 세외수입 체납액은 전국 어디서나 은행 자동인출기(ATM)을 이용해 고지서 없이도 납부할 수 있고 가상계좌 혹은 ARS자동응답시스템(☏043-850-7400)을 통해 신용카드로 납부할 수 있다. + - expected: `⠨⠕⠘⠶⠠⠝⠧⠀⠠⠝⠽⠠⠍⠕⠃⠀⠰⠝⠉⠃⠗⠁⠵⠀` + - actual: `` + - error: `Invalid symbol character` +- `sentence_02.json` #6827: 납부기간은 다음달 4일까지로 고지서를 이용해 금융기관에 방문, 납부하거나 고지서에 기재된 납부 전용계좌(가상계좌)로 이체 또는 은행 현금입출금기(ATM), 금융결제원 인터넷지로(www.giro.or.kr), 위택스(www.wetax.go.kr), ARS(☏043-850-7400) 등으로 납부할 수 있다. + - expected: `⠉⠃⠘⠍⠈⠕⠫⠒⠵⠀⠊⠣⠪⠢⠊⠂⠀⠼⠙⠕⠂⠠⠫⠨` + - actual: `` + - error: `Invalid symbol character` +- `sentence_02.json` #6990: 한편, 지방세, 세외수입 체납액은 전국 어디서나 은행 자동인출기(ATM)를 이용해 고지서 없이도 납부 할 수 있고 가상계좌 혹은 ARS자동응답시스템(☏043-850-7400)을 통해 신용카드로도 납부 가능하다. + - expected: `⠚⠒⠙⠡⠐⠀⠨⠕⠘⠶⠠⠝⠐⠀⠠⠝⠽⠠⠍⠕⠃⠀⠰⠝` + - actual: `` + - error: `Invalid symbol character` +- `sentence_03.json` #16738: 김동현은 23일 자신의 사회관계망서비스(SNS)에 “드디어 산후조리원으로 토봉이 처음 안아보는 날♥ 우리 막내 딸 건강하게 태어나줘서 넘 고마웡”이라는 글과 함께 사진을 게재했다. + - expected: `⠈⠕⠢⠊⠿⠚⠡⠵⠀⠼⠃⠉⠕⠂⠀⠨⠠⠟⠺⠀⠇⠚⠽⠈` + - actual: `` + - error: `Invalid symbol character` + +## PDF-derived state gates + +The rule 37 example `그는 Can you help me?라고 도움을 요청했다.` distinguishes the first word after the roman indicator (`Can`, whose whole-word sign is suppressed) from the interior word `you` in the uninterrupted ASCII phrase (whose UEB wordsign is retained). The `prev_is_ascii_word && next_is_ascii_word` gate expresses that phrase-interior position rather than matching an input string. + +The rule 39 example `What is 김치 in English?` resumes the surrounding English passage after the Korean span. The `english_dominant_wrap_active` gate therefore retains the UEB wordsign for the resumed `in`, instead of treating it as a fresh rule 37 entry word. + +## Rule 69 compatibility-unit scope + +The engine accepts 96 scientific/measurement glyphs from Unicode CJK Compatibility, derives their Roman spelling with NFKC, and applies rules 68/69 rather than whole-word UEB. The accepted glyph set and panic-free encoding property are fixed by inline tests. The official Unicode names distinguish `U+337A ㍺` SQUARE IU (accepted) from `U+33D1 ㏑` SQUARE LN, `U+33D2 ㏒` SQUARE LOG, and `U+33DA ㏚` SQUARE PR (not units, rejected). See the [Unicode CJK Compatibility names list](https://www.unicode.org/charts/nameslist/n_3300.html). + +## Rule 36 Roman-numeral presentation forms + +Rule 36 says that a Roman numeral is written with the corresponding Roman letters. The encoder therefore applies compatibility decomposition only to Unicode Roman Numerals U+2160–U+217F and sends the ASCII spelling through the existing rule-36 algorithm. Encoder regressions compare Unicode presentations with ASCII equivalents in the PDF sentence and in attached-Korean, particle-adjacent, and lower-case contexts. U+2180 `ↀ` and unrelated NFKC characters such as `㈜` are explicit non-targets. + +The transition audit reconstructs the immediately preceding engine behavior: direct and NFC encoding rejected U+2160–U+217F, while the analyzer's existing NFKC comparison path already used the same ASCII Roman spelling. This avoids a saved-output lookup and keeps the transition reproducible from the current corpus. + +Presentation-form cases audited: 37. + +| Previous observation → current observation | Cases | +|---|---:| +| `encoding_error -> encoded_mismatch_pending_rule_review` | 15 | +| `nfkc_input_equivalent -> exact` | 22 | + +Remaining complex encoding errors: 0. These cases still contain another character that fails independently, so disappearance of the `roman_numeral_presentation` family does not imply that every former error case now encodes successfully. + + +## Rules 34/54 Korean-prefixed Roman annotations + +Rule 34 says that when Roman text is enclosed by quotation marks or brackets, the Roman terminator is omitted; its PDF example is `링컨(Lincoln)은 미국의 제16대 대통령이다.` The example's cells put the printed Korean opening parenthesis (`⠦⠄`) before the Roman indicator (`⠴`). Rule 54 says that text immediately after an opening bracket and immediately before a closing bracket is attached. Together these establish the Korean-prefix + closed-Roman-annotation context independently of corpus expected values. A following comma or period is outside the already closed annotation and must not cause its Roman contents to be rerouted as mathematics. + +The implementation gate exists only inside `split_mixed_math_word`, after the prefix has been proved entirely Korean. It accepts a fully closed parenthesized Roman word (including ASCII digits such as `O4O`) plus ordinary trailing prose punctuation. The corpus audit treats the opposite localized reference prefix (`⠴⠐⠣`, Roman indicator plus UEB opening parenthesis) as a data-reference contradiction only when all three cells and the real input position agree. Broad sentence-level coexistence is retained as an exact or existing-primary control. This audit does not alter engine routing. The global math detector is byte-for-byte unchanged; regression tests preserve its existing standalone results for `(x)`, `(A)`, and `(abc)`, while explicit forms such as `(x+1)`, `(a/b)`, and `(x₁)` remain math candidates. + +Against the immediately preceding 63,399-exact run, exact matches increased by 2,092. The observable primary totals changed as follows: `comparison_method` 290→303, `pending_rule_review` 19,636→17,543, and `unsupported_character_review` 203→191. Raw encoding errors stayed at 450; errors resolved by a comparison method changed 247→259 and unresolved review errors changed 203→191. + +## Rule evidence and change log + +| Stage | Standard cases | Corpus exact | Corpus accuracy | Evidence | +|---|---:|---:|---:|---| +| Parent commit `3cfeae0` | 5,141/5,141 | 57,732/83,528 | 69.12% | Reproduced with release tests | +| Rules 28/29 indicator ordering | 5,141/5,141 | 61,652/83,528 | 73.81% | Roman indicator now precedes UEB grade-1/capital indicators; rule 35 roman-number continuity retained | +| Rules 37/39 shared UEB groupsign algorithm | 5,141/5,141 | 63,239/83,528 | 75.71% | Korean Roman sections reuse UEB preference/morphology rules; entry wordsigns and English-dominant resume are state-gated | +| Rules 68/69 compatibility unit algorithm | 5,141/5,141 | 63,388/83,528 | 75.89% | 96 Unicode unit presentation forms use one decomposition/letter-run/attachment algorithm; encoding errors fell from 434 to 226 | +| Rule 36 Unicode Roman-numeral presentation normalization | 5,141/5,141 | 63,399/83,528 | 75.90% | U+2160–U+217F use the corresponding Roman-letter spelling; 11 NFKC-equivalent observations became exact, 23 errors became encoded mismatches pending review, and 3 remain blocked by `㈜` | +| Rules 34/54 Korean-prefixed closed Roman annotation routing | 5,141/5,141 | 65,491/83,528 | 78.41% | A fully closed Roman annotation after an all-Korean prefix stays on the prose encoder path, including attached comma/period and alphanumeric forms such as `O4O`; exact matches increased by 2,092 while the global math detector remained unchanged | +| Rules 43/48 decimal-point ownership | 5,141/5,141 | 66,039/83,528 | 79.06% | A period directly between ASCII digits remains on the numeric punctuation path even when its word or sentence also contains Roman text; 647 localized Roman-entry differences were removed and 525 cases became exact | +| Rules 33/34/69 Roman-unit punctuation boundary | 5,141/5,141 | 66,436/83,528 | 79.54% | Rule-69 units retain their ordinary terminator at end/Korean/slash boundaries but omit it before rule-33/34 punctuation or enclosing marks; compact unit tokens with that boundary stay off the math path; 397 cases became exact | +| Rules 68/69 compact compatibility-derived ASCII units | 5,141/5,141 | 66,546/83,528 | 79.67% | Compact ASCII unit spellings are derived from the engine's already accepted Unicode compatibility-unit forms and reuse their owning-rule cells, with longest-complete matching and no expansion to separated English words; `160mg`, numeric-invariance control `240mg`, and Rule-68 `ha` controls are retained; 110 cases became exact | +| UEB numeric-mode letter classes in Roman identifiers | 5,141/5,141 | 67,012/83,528 | 80.23% | Lowercase `a`-`j` retains grade 1 after digits, capitals use capitalization, and lowercase `k`-`z` needs no extra indicator; numeric-leading Rule-69 units remain separate | +| UEB complete all-caps segments across hyphen | 5,141/5,141 | 67,138/83,528 | 80.38% | The grade-1 restart is omitted only between a complete uppercase prefix and an uppercase suffix of at least two letters; mixed/single-capital and digit-hyphen controls remain unchanged | +| Rule-29 consecutive Roman entry idempotence | 5,141/5,141 | 67,442/83,528 | 80.74% | An explicit Roman-entry event is ignored only when final emit state is already inside the same Roman section | +| Rules 29/71 complete attached Roman ampersand run | 5,141/5,141 | 67,715/83,528 | 81.07% | Complete ASCII-letter segments joined by `&` retain one Roman section; official `AT&T`, `B&B`, and spaced Korean Rule-71 controls delimit the safe boundary; 273 cases became exact | +| Rules 29/39 Roman-to-Korean mode ownership | 5,141/5,141 | 68,101/83,528 | 81.53% | Same-token Korean is wrapped only in a Roman-majority document or the official dot-delimited `www.대통령.kr` domain shape; all three rule-39 PDF controls remain exact and 386 corpus cases became exact | +| Rules 29/34 multiword Roman parenthetical continuity | 5,141/5,141 | 68,175/83,528 | 81.62% | A backwards-verified, closed letter-and-space Roman parenthetical tail stays on the prose route; function calls, digits, operators, nested brackets, and punctuation-separated forms remain outside; 74 cases became exact | +| Rules 29/32/71 ampersand before attached Roman segment | 5,141/5,141 | 68,187/83,528 | 81.63% | Official UEB `&c`, `AT&T`, and `B&B` keep one Roman section across attached `&`; the spaced Korean Rule-71 example, left ASCII alphanumerics, repeated ampersands, and trailing digits delimit the gate; 12 cases became exact | +| UEB 8.4.2 capitals-word nonletter boundary | 5,141/5,141 | 69,291/83,528 | 82.96% | Starting from the accepted 68,439-exact checkpoint, capitals-word handling ends at each nonletter and later uppercase runs restart independently; official `AT&T`, `B&B`, `MP3`, and `TVOntario` delimit the token/span boundary; 852 cases became exact and the complete exact-ID audit found zero former exact cases lost | +| UEB 8.4.2 same-token internal Roman apostrophe | 5,141/5,141 | 69,359/83,528 | 83.04% | A straight apostrophe stays in the Roman section only with immediate same-token ASCII letters on both sides, while capitals mode restarts for an uppercase suffix; official `O'Hara`, `DON'T`, `THAT'S`, and `SHE'LL` plus detached quote and measurement controls delimit the gate; 68 cases became exact and the complete exact-ID audit found zero former exact cases lost | +| Rules 29/34 all-caps headword with closed multiword Roman expansion | 5,141/5,141 | 69,389/83,528 | 83.07% | A complete two-or-more-capital headword followed by a closed expansion of at least two ASCII-letter words stays on the prose route; digits, operators, nesting, scripts, and alphanumeric trailers remain math controls; the cohort's exact count rose 17→47 with 30 corpus-wide gains | +| Rule 34 Korean trailer after closed multiword Roman expansion | 5,141/5,141 | 69,430/83,528 | 83.12% | Rule 34's `링컨(Lincoln)은` establishes that attached Korean text after `)` remains prose; applying the same boundary to closed multiword expansions adds 41 corpus-wide exact matches and raises the headword cohort 47→87, while ASCII-letter and digit trailers remain excluded | + +The latest full `cargo test -p braillify test_by_testcase --release -- --nocapture` run was accepted from its custom testcase summary, not the trailing filtered harness: `총 테스트 케이스: 5141`, `성공: 5141`, `실패: 0`, and `Skip (limitation): 0`. + +Engine changes must add a row only after both the 5,141-case standard suite and this full analysis have been rerun. Suspect-reference clusters stay in this report; they are not engine targets without independent PDF evidence. diff --git a/docs/corpus-analysis/NIKL_2025_V1_inquiry.md b/docs/corpus-analysis/NIKL_2025_V1_inquiry.md new file mode 100644 index 00000000..8b500d05 --- /dev/null +++ b/docs/corpus-analysis/NIKL_2025_V1_inquiry.md @@ -0,0 +1,482 @@ +# 2025년 국립국어원 말뭉치 점자 참조값 및 2024 개정 한국 점자 규정 해석 질의서 + +## 1. 문의 취지 + +안녕하세요. + +2025년 국립국어원 한국어–점자 병렬 말뭉치의 묵자 입력과 점자 참조값을 2024년 개정 「한국 점자 규정」에 따라 대조하는 과정에서, 규정만으로는 일관된 판단이 어렵거나 말뭉치 참조값의 확인이 필요한 유형을 발견하여 문의드립니다. + +이 질의서는 말뭉치의 오류를 미리 단정하려는 것이 아닙니다. 다음 세 경우를 구분하여 자동 점역기가 임의의 예외 규칙을 만들지 않도록 하는 것이 목적입니다. + +1. 현행 규정으로 점형을 결정할 수 있고 말뭉치 참조값의 정정 여부만 확인하면 되는 경우 +2. 발음·문서 장르·수식 여부 같은 외부 의미 정보가 있어야 점형을 결정할 수 있는 경우 +3. 현행 규정에 명시되지 않은 문자 또는 경계여서 별도의 해석이 필요한 경우 + +가능하시다면 각 질문에 대해 다음 사항을 함께 알려 주시기를 부탁드립니다. + +- 올바른 점자 셀 배열 +- 적용되는 규정의 항·붙임·예 번호 +- 해당 말뭉치 참조값의 정정 필요 여부 +- 묵자 표면형만으로 결정 가능한지, 별도의 발음·문서 모드·의미 주석이 필요한지 +- 정정이 필요할 때 공식 정오표 또는 차기 말뭉치 버전에 반영되는지 + +## 2. 대조 방법과 재현 범위 + +- 분석 대상: 83,528문장 +- 현재 완전 일치: 75,726문장(90.66%) +- 불일치 또는 인코딩 불가: 7,802문장 +- 분류 결과: 규정 검토 대기 6,505건, 규정과 참조값의 모순 후보 1,293건, 미지원 문자 검토 4건 +- 모순 후보의 구성: 제34항 괄호/로마자표 순서 1,160건, UEB 비독립 대문자 글자열 앞의 불필요한 1급 점자표 70건, 제37항 여섯 단어의 구간 내부 단어 약자 59건, UEB 대문자 구절을 개별 대문자표로 적은 참조 3건, 로마자 U+2026 말줄임표의 상충 참조 1건 +- 비교 전용 정규화 분류: 0건. 과거 NFKC로만 같았던 잔여 8건은 관련 규정에 근거한 문자별 처리로 모두 해소했으며, 전체 입력에 대한 일괄 NFKC는 적용하지 않음 +- 비교 필드: 묵자 `input`과 말뭉치 점자 참조값 `unicode`만 사용 +- 완전 일치의 정의: 공백을 포함한 유니코드 점자 셀 배열이 처음부터 끝까지 동일한 경우 +- 위치 표기: `sentence_XX.json #N`에서 `N`은 사람이 확인하기 쉬운 1부터 시작하는 배열 순번 +- 계측 방법: 문장에 특정 표면형이 존재하는지만 세지 않고, 기대 출력과 현재 규칙 기반 출력의 **최초 차이 셀**이 그 구조의 실제 출력 범위 안에 있는지도 별도로 확인 +- 정확 대조군: 같은 구조를 포함하면서 전체 출력이 완전히 일치하는 문장 +- 역전 대조: 기대와 현재 출력의 셀 전이가 반대 방향으로 나타나는 경우를 별도 계수 + +경쟁 제품의 출력 필드는 읽거나 정답으로 비교하지 않았습니다. 또한 아래 건수는 서로 겹칠 수 있는 진단 코호트이므로 합산하면 전체 불일치 건수가 되지 않습니다. 코호트에 포함됐다는 이유만으로 기존의 1차 분류를 바꾸지도 않았습니다. + +규정 근거는 [2024 개정 한국 점자 규정 PDF](../2024%20개정%20한국%20점자%20규정.pdf)와 같은 저장소에 보존된 2024 UEB 규정을 확인했습니다. 상세 계측과 대표 출력은 [전체 재현 보고서](NIKL_2025_V1.md)에 있습니다. + +## 3. 우선 답변을 요청드리는 사항 + +### 질문 1. 제34항의 괄호와 로마자표 순서 + +2024 개정 한국 점자 규정 제34항의 `링컨(Lincoln)은 …` 예에서는 다음 순서로 읽힙니다. + +1. 한글 여는 소괄호 `⠦⠄` +2. 로마자표 `⠴` +3. 괄호 안 로마자 +4. 한글 닫는 소괄호 + +현재 점역기도 이 순서를 따릅니다. 그런데 말뭉치에는 한글 바로 뒤의 닫힌 로마자 주석에서 기대값이 `⠴⠐⠣`(로마자표 + UEB 여는 괄호)로 시작하고, 현재 규정 경로가 `⠦⠄⠴`(한글 여는 괄호 + 로마자표)로 시작하는 사례가 1,160건 있습니다. 단순히 같은 문장에 괄호가 있다는 이유가 아니라, 실제 입력 위치에서 이 세 셀의 순서가 모두 반대로 확인되는 경우만 셌습니다. + +대표 사례: + +- `sentence_01.json #35`: `… 국제정보디스플레이학술대회(IMID) 2022 …` +- `sentence_03.json #48`: `… 폐쇄회로(CC)TV …` +- `sentence_01.json #77`: `… OLED …`가 포함된 동일 유형 문장 + +문의: + +1. 한글 문장 안에서 `한글(로마자)`를 점역할 때 여는 괄호와 로마자표의 올바른 순서는 `⠦⠄⠴`입니까? +2. 그렇다면 `⠴⠐⠣`로 시작하는 위 1,160건은 말뭉치 참조값 정정 대상입니까? +3. 괄호 안이 약어, 일반 영단어, 로마자와 숫자의 결합이어도 같은 순서를 적용합니까? + +### 질문 2. 로마자 바로 뒤에 붙은 가운뎃점(U+00B7) 앞의 로마자 종료표 + +제29항은 한글 문장 안의 로마자 앞뒤에 로마자표와 로마자 종료표를 쓰도록 합니다. 제33항은 종료표를 생략하거나 문장 부호 앞으로 옮기는 경계를 열거하지만, 가운뎃점은 그 목록에 없습니다. 제50항은 가운뎃점을 앞뒤 말에 붙여 쓰도록 규정합니다. + +`로마자·한글/로마자` 경계 후보 577건은 완전 일치 대조군이 0건이고 모두 불일치입니다. 그중 516건은 최초 차이가 정확히 현재 출력의 로마자 종료표 위치에 있으며, 말뭉치는 가운뎃점 `⠐`을 기대하지만 규정 기반 출력은 먼저 로마자 종료표 `⠲`을 냅니다. 더 좁은 `AI·SW`형 순수 대문자 코호트도 97건 모두 불일치입니다. + +대표 사례: + +- `sentence_01.json #57`: `PC·모바일` +- `sentence_02.json #10`: `AI·SW교육` +- `sentence_03.json #91`: `HA·Ca필러` +- `sentence_04.json #49`: `Fed·연준` + +문의: + +1. `PC·모바일`에서 `PC` 뒤의 로마자 종료표를 써야 합니까, 생략해야 합니까? +2. 가운뎃점이 로마자 구간을 이어 주는 기호라면 그 근거 항과 적용 범위는 무엇입니까? +3. `AI·SW`, 화학식, 곱셈 기호처럼 같은 표면형이 서로 다른 의미를 가질 때, 말뭉치에 의미/모드 주석이 필요합니까? +4. 현행 규정상 종료표를 생략할 근거가 없다면 위 참조값들은 정정 대상입니까? + +### 질문 3. `△한글`에서 삼각형 뒤 공백의 소유권 + +제49항은 삼각형 문장 부호의 점형을 정하고 인쇄물의 띄어쓰기를 따르게 합니다. 제72항은 `△`를 항목 표지로도 사용하지만 PDF 예는 줄 배치 또는 공백이 있는 목록입니다. 말뭉치에는 입력 자체가 `△한글`처럼 붙어 있는데 기대 점자에는 삼각형 뒤 공백이 들어간 사례가 반복됩니다. + +- 후보 377건 / 완전 일치 315건 / 불일치 62건 +- 현재 출력의 삼각형과 바로 뒤 첫 한글 셀 범위에 최초 차이가 있는 사례 3건 +- 대표: `sentence_01.json #3907` `△청구…`, `sentence_02.json #17` `△문화치유`, `sentence_03.json #244` `△에듀테크&콘텐츠` + +문의: + +1. 묵자에 공백이 없는 `△문화치유`도 항목 표지라는 의미만으로 점자에서는 `△` 뒤를 띄어야 합니까? +2. 제49항의 “묵자의 띄어쓰기를 따른다”와 제72항의 항목 표지 관례 중 어느 규정이 우선합니까? +3. 문장 부호인 삼각형과 목록 표지인 삼각형을 구분하려면 구조 또는 의미 주석이 필요합니까? +4. 입력에 없는 공백을 기대값에 넣는 것이 맞다면 말뭉치의 묵자 입력도 함께 정규화해야 합니까? + +### 질문 4. 약어·두문자어의 발음에 따른 UEB 약자 사용 + +제32항에 따라 로마자 구간 내부는 UEB를 따르는 것으로 이해했습니다. UEB 10.12.1은 약어의 글자를 따로 발음하는 것이 알려져 있으면 해당 글자 결합 약자를 쓰지 않고, 발음을 모르면 약자를 쓰도록 합니다. 같은 대문자 표면형이 브랜드명·단어처럼 발음되기도 하고 이니셜로 하나씩 발음되기도 하므로 묵자 철자만으로는 결과를 결정할 수 없습니다. + +현재 측정: + +| 철자 구조 | 후보 | 완전 일치 | 불일치 | 목표 최초 차이 | 역방향 | +|---|---:|---:|---:|---:|---:| +| `OU` 포함 | 1,816 | 128 | 1,688 | `o ⠕ → ou 약자 ⠳` 1,502 | 0 | +| `ST` 포함 | 1,479 | 812 | 667 | `s ⠎ → st 약자 ⠌` 466 | 별도 확정 안 함 | +| `AR` 포함 | 1,022 | 477 | 545 | `a ⠁ → ar 약자 ⠜` 411 | 별도 확정 안 함 | +| `ED` 포함 | 816 | 362 | 454 | `e ⠑ → ed 약자 ⠫` 339 | 0 | + +완전 일치 대조에는 `JUSTOUCH`(`sentence_01.json #850`), `YOU`(`sentence_02.json #727`), `STAYG`(`sentence_01.json #44`), `KAIST`(`sentence_02.json #82`), `DGIST`(`sentence_03.json #281`), `OLED`(`sentence_01.json #26`) 등이 있습니다. 반면 `MOU`, `AR/ARS`, `LED/GED`, `GH/GHP`, `ERP/ERBUD`, `SH` 계열에서는 글자별 발음 여부에 따라 참조값이 달라지는 것으로 보입니다. + +문의: + +1. 말뭉치 구축 시 약어의 실제 한국어 발음(글자별 발음/단어 발음)을 조사하여 약자 사용 여부를 결정했습니까? +2. 발음 정보가 없는 자동 점역기는 UEB의 “의심스러우면 약자를 사용”하는 기본값을 그대로 적용해야 합니까? +3. `MOU`, `OLED`, `LED`, `GH`, `ERP`처럼 관용 발음이 여러 개인 항목에는 발음 사전 또는 발음 주석을 제공할 수 있습니까? +4. 동일 철자에 서로 다른 참조값이 허용된다면 그 선택을 재현할 수 있는 최소 메타데이터는 무엇입니까? + +### 질문 5. `A(14)`형 표면을 인명 표지로 볼지 수식으로 볼지 + +`A(14)`처럼 단일 대문자 뒤에 숫자 괄호가 붙는 후보는 1,361건이며 완전 일치가 1,351건, 불일치가 10건입니다. 그중 4건은 최초 차이가 해당 진입 경계 안에 있습니다. 대다수 기사 문맥은 일반문 경로로 재현됐지만, 같은 표면은 함수·수학 변수·문항 번호일 수도 있으므로 남은 사례를 표면형만으로 일반화하지 않았습니다. + +대표 사례: + +- `sentence_01.json #80`: `A(54)씨` +- `sentence_02.json #186`: `A(14)양` +- `sentence_03.json #51`: `A(11)군` + +문의: + +1. 일반 기사 문장 속 `A(14)양`은 로마자+숫자 괄호로 점역해야 합니까, 수학식으로 점역해야 합니까? +2. 조사 `씨/양/군` 같은 문맥만으로 일반 규칙을 정해도 됩니까? +3. 표면형만으로 수식 반례를 완전히 배제할 수 없다면 말뭉치에 “일반문/수식” 모드 주석이 필요합니까? +4. 현재 참조값을 만든 점역 과정에서 이 형태를 수식으로 분류한 기준이 있다면 공개가 가능합니까? + +### 질문 6. `HCA(Home Connectivity Alliance)`형 약어 풀이의 로마자 구간 재진입 + +대문자 표제어 뒤에 닫힌 괄호가 있고, 괄호 안에 공백으로 구분된 로마자 단어가 둘 이상인 구조는 175건입니다. 제29항과 제34항을 보수적으로 적용하여, 표제어가 ASCII 대문자 2자 이상이고 괄호 안이 ASCII 글자 단어 2개 이상이며 숫자·연산자·중첩 괄호·다른 문자가 없는 경우만 일반 로마자 풀이로 처리했습니다. 이어 제34항 공식 예시 `링컨(Lincoln)은`에 따라 닫는 괄호 뒤에 붙은 한글과 문장 부호도 일반문 경계로 처리하되, 영문·숫자 꼬리는 계속 제외했습니다. 현재 완전 일치는 113건, 불일치는 62건입니다. 이 구조는 **입력 형태를 모은 교차 진단군일 뿐 1차 분류를 바꾸지 않습니다**. + +정확 대조: + +- `sentence_01.json #3647`: `TB(Top View Battle)` +- `sentence_02.json #313`: `U-ENTER(Uzbekistan Entrepreneurship Innovation Center)` +- `sentence_03.json #20370`: `KINGDOM(Moonlight Tears)` +- `sentence_04.json #2928`: `QSR(Quick Service Restaurant)` + +불일치 대표: + +- `sentence_01.json #18`: `매터(Matter)와 HCA(Home Connectivity Alliance) 표준…` + +이 사례는 위의 보수적 제29·34항 적용으로 완전 일치가 되었습니다. 닫는 괄호 뒤에 조사가 붙는 `HCA(...)를` 형태도 제34항의 `링컨(Lincoln)은`과 같은 경계로 처리한 뒤에는 이 진단군에서 새 로마자표가 생략되는 국소 차이가 남지 않았습니다. 남은 국소 차이는 로마자 글자·약자 점역 차이이며, 표면 구조만으로 수식 반례를 완전히 배제할 수 있는지에 관한 의미 모드 질문은 여전히 남습니다. + +문의: + +1. `Matter` 뒤 한글 조사 `와`가 나오면 첫 로마자 구간은 종료되고, 뒤의 `HCA`에서 로마자표를 다시 써야 합니까? +2. `HCA(Home Connectivity Alliance)` 전체는 일반 로마자 풀이로 보아야 합니까? +3. 같은 형태의 수식 가능성을 배제하려면 어떤 의미/문서 모드 정보가 필요합니까? +4. 제29항의 “둘 이상의 로마자가 이어 나올 때”에서 한글 조사나 한글 단어가 사이에 있으면 연속성이 명백히 끝나는 것으로 보아도 됩니까? + +### 질문 7. `F-35`형 식별자와 수학식의 구분 + +대문자 로마자 run + 하이픈 + 숫자 구조는 571건이며 완전 일치 503건, 불일치 68건입니다. 현재 잔여 중 최초 차이가 해당 구조의 진입/출력 범위에 직접 국소화된 사례는 0건입니다. 제35항은 `D-100`을 로마자와 숫자가 이어지는 예로 제시하지만, 수학 규정은 대문자 변수와 마이너스를 별도 경로로 처리하므로 표면형의 의미 충돌 여부는 별도로 문의합니다. + +정확 대조에는 `GLS-5310`(`sentence_01.json #88`), `MMPI-2`(`sentence_02.json #167`), `X-2`(`sentence_03.json #217`), `GPT-4`(`sentence_04.json #48`)가 있습니다. 불일치 대표는 `sentence_04.json #323`의 `F-35`입니다. + +문의: + +1. 일반 기사 속 `F-35`, `AH-64`, `K-9` 같은 기종·모델명은 제35항의 로마자-숫자 연속 규칙으로 처리합니까? +2. 동일한 `A-3`형 표면이 수학 변수와 수의 뺄셈일 때는 어떤 입력 정보로 구분해야 합니까? +3. 하이픈-minus U+002D 하나만 제공된 말뭉치에서 식별자와 마이너스를 자동 판별하도록 요구합니까, 아니면 의미 모드 주석이 필요합니까? + +### 질문 8. 공백·괄호를 사이에 둔 로마자 구간의 연속 범위 + +제29항의 `Los Angeles`, `Table of Contents`는 공백으로 나뉜 여러 로마자 단어를 하나의 로마자 구간으로 처리하는 근거가 됩니다. 다만 실제 문장에서는 괄호, 쉼표, 슬래시, 숫자, 한글 조사 등을 사이에 둔 경우가 많아 구간이 언제 끝나는지 불명확합니다. + +| 진단 구조 | 후보 | 완전 일치 | 불일치 | 구조 안 최초 차이 | +|---|---:|---:|---:|---:| +| 연속 ASCII 로마자 단어의 공백 경계 | 4,679 | 3,229 | 1,450 | 7 | +| 닫힌 로마자 괄호 뒤 공백 후 새 로마자 | 1,093 | 571 | 522 | 6 | +| 앞 한글 단어 뒤 공백 후 로마자 괄호 표제어 | 4,695 | 3,933 | 762 | 4 | +| 기존 로마자 뒤 공백 후 대문자 단어 | 1,729 | 1,086 | 643 | 4 | + +정확 대조로는 PDF 예와 같은 다단어 로마자, `Global X`(`sentence_01.json #2785`), `BYD), BMW`(`sentence_02.json #1282`), `KODEX 인도 Nifty50`(`sentence_03.json #46`) 등이 공존합니다. 남은 대표 경계에는 `SYNO PEM-1`, `ACE Fair(2020)`, `Mnet K-POP`, `ESS /VPP`가 있습니다. + +문의: + +1. 공백만 있는 연속 로마자 단어는 언제나 하나의 로마자 구간입니까? +2. 닫는 괄호 뒤 공백, 쉼표, 슬래시는 구간을 끝내는 경계입니까, 아니면 열거된 로마자 전체가 하나의 구간입니까? +3. 괄호로 닫힌 로마자 설명 뒤에 이어지는 새 로마자 고유명은 로마자표를 새로 써야 합니까? +4. 규정의 “이어 나올 때”를 자동 점역기가 판단할 수 있도록 경계별 예를 추가해 주실 수 있습니까? + +### 질문 9. 앰퍼샌드와 뒤따르는 로마자·숫자의 구간 + +UEB에는 `AT&T`, `B&B`, `&c`처럼 앰퍼샌드와 로마자가 붙어 있는 예가 있고, 한국 점자 규정 제71항에는 한글 사이의 독립적인 앰퍼샌드 예가 있습니다. 이를 근거로 완전한 `A&B`형 로마자 run은 한 구간으로 처리할 수 있었지만, 숫자가 이어지거나 한쪽에만 로마자가 붙는 경우는 남아 있습니다. + +- 붙은 로마자 `A&B`형: 802건 / 완전 일치 684건 / 불일치 118건 +- `&c`처럼 오른쪽 로마자만 붙은 좁은 구조: 30건 / 완전 일치 14건 / 불일치 16건 +- 남은 불일치는 다른 로마자 약자·숫자 연속·문장 경계 차이와 겹칠 수 있으므로 앰퍼샌드 자체의 문제로 일괄 분류하지 않음 + +문의: + +1. `S&P500`, `R&D`, `AT&T5G`에서 앰퍼샌드는 하나의 로마자 구간 안에 있습니까? +2. 앰퍼샌드 뒤 숫자가 이어질 때 제35항의 로마자-숫자 연속 규칙이 그대로 적용됩니까? +3. 한글과 붙은 `과학&ICT`, 독립 기호인 `종이접기 & 클레이아트`, 괄호 안 `&TEAM`은 각각 어느 항을 우선 적용합니까? +4. 공백 유무만으로 독립 기호와 로마자 구간 내부 기호를 구분해도 됩니까? + +### 질문 10. 숫자+ASCII 접미부의 단위·식별자·분수 해석 + +숫자 바로 뒤에 ASCII 문자가 붙은 구조는 2,975건이며 완전 일치 2,483건, 불일치 492건입니다. 42건은 최초 차이가 그 전체 토큰 또는 진입 경계 안에 있습니다. 제69항의 공식 단위와 Unicode 호환 단위에서 규정상 도출되는 철자는 일반화했지만, 같은 표면형이 단위·변수·모델명일 수 있어 모든 ASCII 접미부를 단위로 보지는 않았습니다. + +정확 대조에는 `118.0GW`(`sentence_01.json #343`), `20kg`(`sentence_02.json #29`), `692g`(`sentence_03.json #60`) 등이 있습니다. 남은 대표 사례는 다음과 같습니다. + +- `sentence_01.json #633`: `3.5~8.5m` +- `sentence_02.json #893`: `50~800m` +- `sentence_03.json #140`: `1/2` +- `sentence_04.json #871`: `98M` + +문의: + +1. `m`, `M`, `G`, `p`, `bp`, `GB`처럼 대소문자와 분야에 따라 뜻이 달라지는 접미부는 말뭉치에서 어떤 기준으로 단위로 판정합니까? +2. `3.5~8.5m`에서 범위 기호와 단위의 로마자표/종료표 범위는 어떻게 됩니까? +3. 일반문에 있는 ASCII `1/2`는 단순 슬래시 표기입니까, 수학 분수입니까? 의미가 분수여도 입력이 LaTeX가 아니라면 어떤 규칙을 적용합니까? +4. `98M`이 수량 단위, 모델명, 변수 중 무엇인지 표면형만으로 판정할 수 없을 때 필요한 메타데이터는 무엇입니까? +5. 단위의 대소문자는 묵자 그대로 보존하여 서로 다른 점역 결과를 내야 합니까? + +### 질문 11. ASCII 문장 부호와 한국어 문장 부호의 정규화 + +말뭉치에는 책·작품 제목을 ASCII `<...>`로 표시한 사례가 있으나 제49항은 별도의 한국어 문장 부호 `〈...〉`(U+3008/U+3009)를 규정합니다. 이 경계의 상위 잔여 최초 차이는 98건입니다. 또한 ASCII 하이픈-minus, en dash, 줄표가 기사 편집 과정에서 혼용된 사례가 있고, 관련 잔여 전이는 약 80건입니다. + +대표 사례: + +- ASCII `<제목>`이 겹낫표/홑화살괄호 의미로 쓰인 기사 제목 +- `sentence_01.json #189`: `… 써봐 - 슈퍼 캐리` +- `sentence_02.json #2352`: `이이남 -각 사람에게…` +- `sentence_03.json #4223`: en dash 사용 +- `sentence_04.json #976`: `하쿠토-R` + +문의: + +1. ASCII `<`와 `>`가 제목 표지로 쓰였을 때 자동으로 U+3008/U+3009 의미의 한국어 문장 부호로 정규화해야 합니까? +2. 아니면 코드 포인트가 다르면 수학의 부등호 또는 ASCII 기호로 점역해야 합니까? +3. U+002D `-`, U+2013 `–`, 한국어 줄표가 혼용된 입력은 원문 코드 포인트를 보존해야 합니까, 의미에 맞게 정규화해야 합니까? +4. 제49항의 “묵자의 띄어쓰기를 따른다”는 입력에 있는 공백을 그대로 보존하라는 뜻입니까? + +### 질문 12. 호환 문자의 선택적 정규화 정책 + +현재 `comparison_method` 잔여는 0건입니다. 전체 입력을 NFKC로 바꾼 것이 아니라, 제36항의 Unicode 로마 숫자와 제68·69항의 Unicode 호환 단위처럼 현행 규정으로 대응 철자를 도출할 수 있는 문자만 선택적으로 처리했습니다. 직전 잔여 8건도 U+2026 말줄임표 1건과 호환 단위의 로마자/숫자 구간 경계 7건으로 나누어 규정에 따라 처리했습니다. 따라서 `㈜` 같은 다른 호환 문자를 자동 분해하는 일반 NFKC 규칙은 두지 않았습니다. + +문의: + +1. 말뭉치 참조값은 원 `input` 코드 포인트를 기준으로 작성됐습니까, NFKC 등 사전 정규화를 거친 문자열을 기준으로 작성됐습니까? +2. Unicode 호환 단위는 그 호환 문자의 표준 분해 철자를 사용해 제68·69항을 적용하는 것이 맞습니까? +3. 공식 권장 정규화 형식(NFC/NFKC)과, 정규화하면 안 되는 예외 문자 목록이 있습니까? +4. 문자별 규정 근거가 없는 경우에는 원 코드 포인트를 보존하고 미지원으로 보고하는 것이 맞습니까? + +### 질문 13. 현행 규정에서 점형을 찾지 못한 문자 4건 + +현재 단독 인코딩도 실패하며 2024 개정 한국 점자 규정에서 독립적인 점형 근거를 찾지 못한 사례는 4건뿐입니다. + +| 문자 | 포함 문장 수 | NFKC 분해 | 말뭉치 참조에서 관찰된 처리 | 현재 진단 | +|---|---:|---|---|---| +| `☏` U+260F | 3 | 그대로 | 기호 자리에 로마자 `TEL`에 해당하는 점형이 들어간 것으로 관찰됨 | 공식 근거 미확인 | +| `♥` U+2665 | 1 | 그대로 | 기호 위치에 추가 공백이 들어간 것으로 관찰됨 | 공식 근거 미확인 | + +위 관찰은 참조값을 역산해 구현하기 위한 근거로 사용하지 않았습니다. UEB 11.7.2의 전사자 정의 도형은 독자에게 정의를 제공해야 하므로, 범용 자동 점역기의 고정 점형으로 채택하지 않았습니다. + +문의: + +1. `☏`와 `♥`에 공식 권장 점형이 있습니까? +2. `☏`를 `TEL`로 풀어 쓰거나 `♥`를 공백으로 대체하는 것이 말뭉치 구축 지침에 따른 의도적 처리입니까? +3. 전사자 정의 기호라면 문서마다 점형과 설명을 함께 제공해야 합니까? +4. 공식 규정의 지원 범위 밖이라면 자동 점역기는 오류를 반환해야 합니까, 원문을 보존해야 합니까, 또는 별도의 대체 텍스트 입력을 요구해야 합니까? + +### 질문 14. 묵자 입력에 없는 띄어쓰기나 교정의 허용 범위 + +제49항은 문장 부호의 띄어쓰기를 묵자에 따르도록 합니다. 분석 과정에서도 입력에 붙은 `있다`를 임의로 띄어 쓰던 전처리를 제거했을 때 71건이 정확해졌고, PDF에 실제 공백이 있는 표준 예는 공백을 그대로 유지했습니다. 반대로 `△한글` 등 일부 말뭉치 참조값은 입력에 없는 공백을 요구하는 것으로 보입니다. + +문의: + +1. 점역기는 말뭉치의 `input`을 문자 단위로 충실히 점역해야 합니까, 맞춤법·띄어쓰기 오류를 먼저 교정해야 합니까? +2. 교정을 허용한다면 공식 교정 규칙과 원문/교정문 대응 정보를 제공할 수 있습니까? +3. 참조값이 교정된 문장을 기준으로 만들어졌다면 교정된 묵자 필드도 함께 제공할 수 있습니까? +4. 입력에 없는 공백을 점자 참조값에만 넣는 것이 평가상 의도된 동작입니까? + +### 질문 15. 로마자·단위·숫자 뒤에 붙은 괄호를 어느 점자 체계로 적는지 + +앞에서 설명한 제34항의 1,160건은 **한글 뒤 괄호 안에 로마자가 있는 경우**입니다. 이와 반대로 `BSI(73)`, `Merit(4위)`, `M(41)`, `43bp(1bp…)`처럼 로마자·단위 뒤의 괄호 안에 숫자 또는 한글이 있는 경우도 별도로 남습니다. 기존 output-localized 코호트를 제외한 뒤에도 기대 한글 여는 소괄호의 첫 셀 `⠦`와 현재 UEB 여는 괄호의 첫 셀 `⠐`이 충돌하는 잔여가 66건입니다. + +대표 사례: + +- `sentence_01.json #1318`: `BSI(73)` +- `sentence_02.json #286`: `Merit(4위)` +- `sentence_03.json #551`: `조너선M(41)` +- `sentence_04.json #119`: `43bp(1bp는 0.01%포인트)` + +제33항의 괄호 속 `(, : ; ―)`는 괄호 자체를 열거한 것이 아니라 쉼표·쌍점·쌍반점·줄표의 목록으로 읽히므로, 위 소괄호의 점형을 직접 결정하지는 않는 것으로 이해했습니다. 제34항은 “로마자가 괄호 등으로 묶일 때”를 규정하지만, 위 사례에서 괄호로 묶인 내용은 로마자가 아니라 숫자 또는 한글입니다. + +문의: + +1. `BSI(73)`의 괄호는 로마자 구간 내부의 UEB 괄호입니까, 한글 점자의 소괄호입니까? +2. 괄호 안이 숫자, 한글, 로마자일 때 각각 로마자 구간의 종료 위치가 달라집니까? +3. `43bp(1bp는 …)`처럼 괄호 안이 단위 설명일 때는 바깥 단위의 로마자 구간을 먼저 닫아야 합니까? +4. 괄호의 점형을 결정하는 기준이 “괄호 앞 문자”, “괄호 안 주언어”, “전체 문장의 주언어” 중 무엇인지 예와 함께 알려 주실 수 있습니까? + +### 질문 16. U+002D 마이너스와 이름에 붙인 `+(풀이)`의 공백 + +제46항의 공식 예는 연산 기호 양옆을 띄우며, 뺄셈 기호에는 U+2212 `−`를 사용합니다. 반면 말뭉치 입력의 `음(-)극`, `마이너스(-)`는 U+002D HYPHEN-MINUS를 사용하면서 참조값에서는 제46항의 뺄셈 기호 셀 `⠔`을 요구합니다. 현재 점역기는 코드 포인트를 보존하여 U+2212는 뺄셈 기호, U+002D는 제49항 문장 부호 경로의 붙임표로 처리합니다. + +대표 사례: + +- `sentence_01.json #21108`: `양(+)극과 음(-)극` — `-`는 U+002D이나 참조값은 `⠔` +- `sentence_02.json #23164`: `마이너스(-)였으나` — 같은 전이가 최초 차이로 재현됨 + +또한 상표·프로그램 이름에 붙은 `한글+(한글 풀이)` 구조는 16건입니다. 현재 제46항 경로와 전체 참조값이 일치하는 문장은 4건, 불일치는 12건이며, 불일치 12건 모두 최초 차이가 현재 출력의 해당 구조 안에 있습니다. 특히 동일한 `도전+(플러스)` 표면형에서 공백을 요구하는 참조와 생략하는 참조가 모두 존재합니다. + +- 일치: `sentence_02.json #4383`의 `도전+(플러스)` +- 불일치: `sentence_02.json #168`, `#6210`의 `도전+(플러스)` +- 불일치: `sentence_01.json #845`의 `자립+(더하기)` +- 불일치: `sentence_03.json #18547`, `#18734`의 `디즈니+(플러스)` + +문의: + +1. 묵자 입력이 U+002D이어도 주변 의미가 음극·마이너스이면 U+2212와 같은 뺄셈 기호로 정규화해야 합니까? +2. 코드 포인트만으로 구분해야 한다면 위 U+002D 사례의 참조값은 붙임표 점형으로 정정해야 합니까? +3. `도전+(플러스)`처럼 이름 뒤에 기호의 한글 풀이를 괄호로 붙인 경우에도 제46항에 따라 `+` 양옆을 띄어야 합니까? +4. 동일 표면형의 참조값에서 공백이 서로 다른 사례는 어느 쪽으로 통일해야 합니까? + +### 질문 17. 로마자 구간 안 U+2026 말줄임표의 상충 참조 + +제32항에 따라 한글 문장 안의 로마자 구간은 UEB를 적용하는 것으로 이해했습니다. UEB 2024 제7.3절은 U+2026 `…` 말줄임표를 `⠲⠲⠲`로 제시합니다. 반면 한국 점자 규정 제53항의 한글 문장 부호 말줄임표는 `⠠⠠⠠`입니다. + +말뭉치에는 같은 구조인 `ASCII 로마자 + U+2026 + 닫는 괄호`에 서로 다른 참조가 있습니다. + +- `sentence_01.json #2530`: `…(I AM…)…` — 참조값은 UEB `⠲⠲⠲`이며 현재 규정 기반 출력과 완전 일치 +- `sentence_01.json #13772`: `…(Love Is…)…` — 참조값은 한국어 말줄임표 `⠠⠠⠠`이며 현재 UEB 출력 `⠲⠲⠲`과 불일치 + +두 사례 모두 U+2026이 로마자 바로 뒤, 닫는 소괄호 바로 앞에 있습니다. 따라서 두 번째 사례 1건은 표면 구조나 코드 포인트 차이로 설명할 수 없는 상충 참조로 분류했습니다. + +문의: + +1. 한글 문장 안 괄호에 든 로마자 구간의 U+2026은 UEB 제7.3절에 따라 `⠲⠲⠲`로 적는 것이 맞습니까? +2. 그렇다면 `Love Is…`의 `⠠⠠⠠` 참조는 `⠲⠲⠲`로 정정해야 합니까? +3. 로마자 구간 안에서도 한국어 문장 부호 점형을 적용하는 예외가 있다면, 그 조건과 근거 항은 무엇입니까? + +### 질문 18. 제37항의 여섯 단어를 로마자 구간 내부에서 단어 약자로 적은 참조 59건 + +제37항 붙임의 공식 예 `be, his, was, were의 약자를 바르게 쓰시오.`에서는 다음 세 위치가 한 문장에 함께 제시됩니다. + +- `be`: 로마자표 바로 뒤 +- `his`, `was`: 로마자 구간 내부 +- `were`: 로마자 종료표 바로 앞 + +공식 점자는 네 단어를 모두 단어 약자로 적지 않고 알파벳과 적용 가능한 묶음 약자로 풀어 씁니다. 따라서 현행 점역기도 같은 로마자 구간에 있는 `be`, `enough`, `his`, `in`, `was`, `were`를 풀어 씁니다. + +그런데 말뭉치에는 이 여섯 단어 중 하나 이상을 구간 내부에서 UEB 하위 단어 약자로 적은 참조가 59건 있습니다. 현재 표준 출력에서 해당 단어의 풀어 쓴 셀만 단어 약자 한 셀로 바꾸면 문장 전체가 완전 일치하며, 그 외 셀 차이는 없습니다. 이 동작은 위 제37항 공식 예의 구간 내부 `his`, `was`와 직접 충돌하므로 정확도를 위해 표준을 어기는 변경은 유지하지 않았습니다. + +대표 사례: + +- `sentence_01.json #507`: `Frontiers in Drug Delivery` +- `sentence_01.json #602`: `Trends in Biotechnology` +- `sentence_01.json #13800`: `I'll Be There` +- `sentence_02.json #4388`: `Boys, Be Different` +- `sentence_03.json #3516`: `Boys will be Boys` + +문의: + +1. 제37항의 여섯 단어는 공식 예와 같이 로마자 구간의 처음·중간·끝 어디에서나 단어 약자를 쓰지 않는 것이 맞습니까? +2. 그렇다면 위 59건에서 사용된 `be ⠆`, `in ⠔` 등의 하위 단어 약자 참조는 풀어 쓴 점형으로 정정해야 합니까? +3. 제37항 붙임의 “로마자 종료표 앞에서도”가 구간 중간에서는 단어 약자를 허용한다는 뜻이라면, 공식 예의 내부 단어 `his`, `was`를 풀어 쓴 이유는 무엇입니까? +4. 대문자로 시작하는 `In`, `Be`에도 같은 원칙을 적용하되 대문자표만 앞세우면 됩니까? + +### 질문 19. UEB 제8.5.2·8.5.3의 대문자 구절 대신 개별 대문자표를 쓴 참조 3건 + +제32항에 따라 한글 문장 안의 로마자 구간에는 UEB를 적용하는 것으로 이해했습니다. UEB 2024 제8.5.2는 구절을 세 개 이상의 기호열(symbols-sequence)로 정의하고 비알파벳 기호를 포함할 수 있다고 규정합니다. 제8.5.3은 대문자 구절의 마지막 적용 기호열 바로 뒤에 대문자 종료표를 두도록 합니다. 공식 예 `CAUTION: WET PAINT!`, `THE BBC AFRICA NEWS`, `A SELF-MADE MAN`, `A.A. (ALAN ALEXANDER) MILNE`도 공백·문장 부호·하이픈·괄호를 포함한 세 개 이상의 기호열에 대문자 구절표 `⠠⠠⠠`와 종료표 `⠠⠄`를 사용합니다. + +이 규칙을 특정 입력 목록이 아니라 연속된 대문자 기호열의 구조로 일반화하자, 직전 엔진과의 전체 말뭉치 대조에서 새로 완전 일치한 문장은 161건, 기존 완전 일치에서 벗어난 문장은 아래 3건으로 순증 158건이었습니다. 세 손실은 모두 현재 출력의 대문자 구절표 한 쌍을 말뭉치 참조의 기호열별 한 칸 또는 두 칸 대문자표로 바꾸면 문장 전체가 정확히 일치합니다. 로마자 철자, 약자, 문장 부호, 공백 등 다른 차이가 하나라도 있는 사례는 이 3건에 포함하지 않았습니다. + +| 위치 | 입력의 관련 구간 | 말뭉치 참조의 대문자 처리 | UEB 제8.5.2·8.5.3 처리 | +|---|---|---|---| +| `sentence_01.json #12980` | `SBS M, SBS FiL` | `⠠⠠SBS … ⠰⠠M … ⠠⠠SBS` | `⠠⠠⠠SBS … ⠰M … SBS⠠⠄` | +| `sentence_03.json #16532` | `FESTA(2023 BTS FESTA)` | `FESTA`, `BTS`, `FESTA`마다 `⠠⠠` | 첫 `FESTA` 앞 `⠠⠠⠠`, 마지막 적용 기호열 뒤 `⠠⠄` | +| `sentence_03.json #20410` | `D N D` | `⠠D … ⠰⠠N … ⠰⠠D` | `⠠⠠⠠D … ⠰N … ⠰D⠠⠄` | + +표의 로마자 철자는 대문자표 위치를 보이기 위한 축약 표기이며, `⠰`은 해당 한 글자 기호열에 필요한 1급 점자표로 그대로 유지됩니다. 첫 사례의 뒤쪽 `FiL`은 혼합 대소문자이므로 현재 구현은 그 앞에서 대문자 구절을 종료합니다. + +문의: + +1. 위 세 관련 구간은 각각 UEB 제8.5.2의 세 개 이상 대문자 기호열로 보아 대문자 구절표와 종료표를 쓰는 것이 맞습니까? +2. 그렇다면 기호열마다 개별 대문자표를 사용한 위 3건의 말뭉치 참조값은 대문자 구절 표기로 정정해야 합니까? +3. 방송 채널명, 행사명, 노래 제목, 한 글자 표제어처럼 고유명사의 종류에 따라 제8.5.2 적용을 배제하는 예외가 있습니까? +4. 쉼표, 숫자, 소괄호, 하이픈 및 뒤따르는 혼합 대소문자는 기호열 수와 대문자 구절의 시작·종료 범위에 각각 어떤 영향을 줍니까? + +### 질문 20. 여는 괄호가 바로 뒤따르는 대문자 글자열 앞의 1급 점자표 70건 + +UEB 2024 제2.6은 글자·글자열의 앞뒤에 공백이나 하이픈·대시가 있거나, 제2.6.2와 제2.6.3에 열거된 기호만 끼어 있을 때 이를 “독립되어 있다(standing alone)”고 봅니다. 제2.6.2는 **글자열 앞에 올 수 있는** 기호에 여는 소괄호·대괄호·중괄호를 포함하지만, 제2.6.3의 **글자열 뒤에 올 수 있는** 기호에는 닫는 괄호만 포함하고 여는 괄호는 포함하지 않습니다. 따라서 `GDC(`, `LLM(`처럼 여는 괄호가 바로 뒤따르는 대문자 글자열은 독립된 글자열이 아닌 것으로 해석했습니다. + +UEB 제5.7.2와 제10.9.7은 글자열이 독립되어 있고 하위 단어 약자(shortform)로 잘못 읽힐 수 있을 때 1급 점자표를 사용하도록 합니다. 제10.9.8은 하위 단어 약자로 시작하는 더 긴 알파벳 단어에 별도 규칙을 둡니다. 위와 같이 여는 괄호가 뒤따르는 비독립 글자열에는 이 조건이 성립하지 않으므로, 현재 점역기는 대문자 단어표 `⠠⠠`는 쓰되 그 앞의 1급 점자표 `⠰`은 쓰지 않습니다. + +말뭉치에는 이 위치에 `⠰`을 추가한 참조가 70건 있습니다. 아래 70건은 입력 철자나 괄호 안 내용으로 선별하지 않았습니다. 먼저 비독립 구조와 하위 단어 약자 충돌 가능성을 입력만으로 검출한 뒤, 참조값에서 현재 표준 출력의 해당 `⠠⠠` 바로 앞에 있는 `⠰` 한 칸만 제거하면 문장 전체가 완전 일치하는 경우로 한정했습니다. 철자·약자·괄호·공백 등 다른 셀 차이가 하나라도 있는 사례는 포함하지 않았습니다. + +| 위치 | 입력의 관련 구간 | 말뭉치 참조 | UEB 해석에 따른 현재 출력 | +|---|---|---|---| +| `sentence_01.json #1573` | `GDC(Game Developers Conference)` | `…⠴⠰⠠⠠⠛⠙⠉⠐⠣…` | `…⠴⠠⠠⠛⠙⠉⠐⠣…` | +| `sentence_01.json #5005` | `LLM(초거대언어모델)` | `…⠴⠰⠠⠠⠇⠇⠍⠦⠄…` | `…⠴⠠⠠⠇⠇⠍⠦⠄…` | +| `sentence_01.json #8525` | `GDP(국내총생산)` | `…⠴⠰⠠⠠⠛⠙⠏⠦⠄…` | `…⠴⠠⠠⠛⠙⠏⠦⠄…` | + +문의: + +1. 여는 소괄호·대괄호·중괄호가 바로 뒤따르는 대문자 글자열은 UEB 제2.6.3에 따라 독립된 글자열이 아닌 것이 맞습니까? +2. 그렇다면 위 70건의 참조값에서 대문자 단어표 앞의 1급 점자표 `⠰`을 삭제해야 합니까? +3. 괄호 안 내용이 앞 약어의 영문 풀이·한글 풀이·숫자 등의 부가 설명이라는 의미 관계가 제2.6의 독립 여부를 바꾸는 예외가 있습니까? +4. 같은 원칙을 여는 소괄호·대괄호·중괄호 모두에, 그리고 괄호 안이 로마자·한글·숫자인 경우 모두에 동일하게 적용합니까? + +### 질문 21. 로마자 구간 내부 `*`·`+`와 식별자·수식의 의미 경계 + +제32항에 따라 한글 문장 안의 로마자 구간에는 UEB를 적용하는 것으로 이해했습니다. UEB 2024 제3.3.1은 별표를 의미와 관계없이 묵자의 위치와 띄어쓰기에 따라 적도록 하며, 공식 예 `M*A*S*H`를 하나의 연속된 UEB 표기로 제시합니다. UEB 제3.17은 더하기표를 `⠐⠖`로 제시합니다. 반면 한글 점자 제45·46항과 수학 점자 규정에도 연산 기호가 있으므로, 같은 ASCII 표면형이 상품명·등급·합성어·수식·화학식을 모두 나타낼 수 있습니다. + +현재 점역기는 다음과 같이 표면 구조만으로 확정 가능한 범위만 일반화했습니다. + +- `M*A*S*H`처럼 비어 있지 않은 로마자 구간 사이의 `*`는 UEB 제3.3.1에 따라 로마자 구간을 끝내지 않고 `⠐⠔`로 적습니다. 숫자만의 `2*3`, 한글과 붙은 별표, 독립 별표는 이 규칙에 포함하지 않습니다. +- 오른쪽 피연산자가 없이 끝나는 `A+`, `TV+`, `24K+`, `C++`와 대소문자가 식별자임을 드러내는 `Dog+Yoga`형은 일반문에서 UEB `⠐⠖`를 사용합니다. 명시적 수식 모드와 완결된 `A+B`, `AB+C`, `sin+cos`는 수식 경로에 남깁니다. +- 한글 접두부·조사·닫힌 설명 괄호는 로마자 종결 식별자의 바깥 문맥으로만 사용하며, 특정 상표나 말뭉치 입력 문자열을 직접 매핑하지 않습니다. + +이 일반화 직전·직후의 83,528문장 전체 대조에서 별표 변경은 완전 일치 2건 순증, 더하기표 변경은 63건 순증이었고 기존 완전 일치 손실은 모두 0건이었습니다. 별표가 포함된 다른 1건도 해당 `PDC*line` 구간 자체는 참조와 같아졌으나 문장 안의 별도 차이 때문에 전체 일치 수에는 포함되지 않았습니다. + +그러나 다음 표면형은 입력만으로 의미를 확정할 수 없어 강제 변환하지 않았습니다. + +- 식별자·합성어로 보이는 사례: `SF+AW`, `TECH+TALK`, `YOUTH+TEEN`, `X+U`, `UNIV+CITY` +- 수식·기간 표기로도 읽히는 사례: `A+B`, `N+1`, `T+1` +- 화학식으로 보이는 사례: `134Cs+137Cs` +- 소문자 합성어와 함수 합의 충돌: `new+retro`, `she+recovery` 대 `sin+cos` + +문의: + +1. 한글 문장 안의 로마자 구간에서 `M*A*S*H`형 별표는 UEB 제3.3.1에 따라 하나의 로마자 구간과 묵자 띄어쓰기를 유지하는 것이 맞습니까? 이때 한글 점자 제60항의 별표 앞뒤 띄어쓰기는 로마자 구간 밖의 한글 문맥에만 적용합니까? +2. 오른쪽 피연산자가 없는 `A+`, `TV+`, `C++`형은 로마자 식별자·등급의 일부로 보아 UEB `⠐⠖`를 쓰는 것이 맞습니까? +3. `TECH+TALK`와 `AB+C`처럼 대문자 글자열 내부의 `+`가 식별자 결합인지 수학 덧셈인지 표면형만으로 구분하는 공식 기준이 있습니까? 글자열 길이·대소문자·괄호 속 한글 풀이를 판정 근거로 사용할 수 있습니까? +4. `N+1`, `T+1`, `134Cs+137Cs`는 각각 일반 로마자 구간, 수학 점자, 화학식용 UEB 중 어느 규정을 우선 적용해야 합니까? +5. 표면형만으로 판정할 수 없다면 말뭉치에 일반문/식별자/수식/화학식 의미 모드를 제공해야 합니까? + +## 4. 말뭉치 구축·정정 절차에 관한 공통 질문 + +1. 각 참조값을 작성할 때 적용한 한국 점자 규정 및 UEB 판본은 무엇입니까? +2. 일반문/수식/화학식/프로그래밍 코드/식별자/단위/고유명사/약어 발음을 구분하는 내부 주석이 있습니까? +3. 내부 주석이 있다면 공개 데이터에도 포함하거나 판정 기준을 문서화할 수 있습니까? +4. 재현 가능한 참조값 오류를 신고할 공식 창구와 필요한 최소 자료는 무엇입니까? +5. `sentence_XX.json #N`과 같은 shard/index는 버전 간에 안정적인 식별자로 사용할 수 있습니까? 아니라면 문장 ID를 제공할 수 있습니까? +6. 정정 사항은 정오표, 패치 버전, 다음 연도 말뭉치 중 어디에 반영됩니까? +7. 동일 입력의 참조값이 판본에 따라 바뀔 때 변경 이력과 적용 규정 항을 제공할 수 있습니까? + +## 5. 요청드리는 답변 양식 + +아래 형식으로 답변해 주시면 규칙 구현, 데이터 정정, 의미 주석 필요 항목을 서로 섞지 않고 반영할 수 있습니다. + +| 질문/사례 ID | 올바른 점자 셀 배열 | 근거 규정 | 말뭉치 정정 필요 | 표면형만으로 판정 가능 | 필요한 의미/발음 주석 | 비고/반영 예정 버전 | +|---|---|---|---|---|---|---| +| 예: 질문 1 / `sentence_01.json #35` | | | 예/아니요 | 예/아니요 | | | + +셀 배열은 가능하다면 유니코드 점자와 점 번호 표기를 함께 부탁드립니다. 하나의 문장 안에 여러 문제가 있으면, 문의한 표면형 주변의 최소 구간만 답변해 주셔도 됩니다. + +## 6. 우선순위 요약 + +답변을 한 번에 모두 제공하기 어렵다면 다음 순서로 우선 확인을 부탁드립니다. + +1. **제34항 괄호/로마자표 순서 1,160건**: 규정 예와 반대인 3셀 참조 서명으로 재현됨 +2. **UEB 비독립 대문자 글자열 앞의 1급 점자표 70건**: 여는 괄호가 뒤따르는 글자열은 제2.6.3의 독립 조건을 충족하지 않으며, 참조의 `⠰` 한 칸만 제거하면 문장 전체가 일치함 +3. **제37항 여섯 단어의 단어 약자 59건**: 공식 예의 구간 내부 `his`, `was`와 반대이며, 약자 치환만으로 전체 참조가 일치함 +4. **UEB 대문자 구절 대신 개별 대문자표를 쓴 참조 3건**: 제8.5.2·8.5.3 적용 결과와 대문자 지시표만 다르고 나머지 문장 전체가 일치함 +5. **로마자+가운뎃점 경계 577건**: 완전 일치 대조 0건, 516건이 현재 종료표 위치에 직접 국소화됨 +6. **약어 발음과 UEB 약자**: 정확/불일치가 동일 철자 구조에 공존하여 발음 정보 없이는 결정 불가 +7. **로마자 U+2026 상충 참조 1건**: 동일 구조의 `I AM…`은 UEB 점형, `Love Is…`는 한국어 점형을 요구함 +8. **`A(14)`, `HCA(...)`, `F-35` 및 `△한글` 잔여**: 대다수는 표준 경로로 해소됐으나 일반문·식별자·수식·항목 표지의 의미 모드 확인이 필요함 +9. **로마자 구간 내부 `*`·`+`의 의미 경계**: 종결 식별자와 공식 `M*A*S*H`는 일반화했으나, 대문자 합성어·수식·화학식의 동일 표면형에는 의미 모드가 필요함 +10. **U+002D 마이너스와 `+(풀이)` 공백**: 코드 포인트와 의미 정규화, 동일 표면형의 상충 참조 확인 +11. **미지원 문자 4건(`☏` 3건, `♥` 1건)**: 공식 점형 또는 대체 텍스트 정책 확인이 필요함. 정규화 비교 잔여는 0건임 + +위 항목들의 공식 해석을 받기 전에는 말뭉치 참조값에 맞추기 위한 개별 입력 예외나 기대값 역산 규칙을 추가하지 않고, 재현 가능한 진단으로만 보존할 예정입니다. diff --git a/libs/braillify/examples/nikl_corpus_analyze.rs b/libs/braillify/examples/nikl_corpus_analyze.rs new file mode 100644 index 00000000..db1becb8 --- /dev/null +++ b/libs/braillify/examples/nikl_corpus_analyze.rs @@ -0,0 +1,9344 @@ +//! Reproducible NIKL Korean–Korean Braille Parallel Corpus 2025 v1.0 analysis. +//! +//! Run from the workspace root: +//! `cargo run --release -p braillify --example nikl_corpus_analyze` +//! +//! This is an offline evaluation tool. It deliberately deserializes only `input` and +//! `unicode`; the read-only competitor fields `world` and `jeomsarang` are neither loaded nor +//! compared. + +use std::collections::{BTreeMap, BTreeSet}; +use std::fs::{self, File}; +use std::io::BufReader; +use std::path::{Path, PathBuf}; +use std::thread; +use std::time::Instant; + +use serde::{Deserialize, Serialize}; +use unicode_normalization::UnicodeNormalization; + +#[derive(Clone, Deserialize)] +struct CorpusCase { + input: String, + unicode: String, +} + +#[derive(Clone)] +struct LocatedCase { + shard: String, + index: usize, + case: CorpusCase, +} + +#[derive(Clone)] +struct EncodedCase { + located: LocatedCase, + actual: Result, + nfc_actual: Option>, + nfkc_actual: Option>, + singleton_unsupported_characters: Vec, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Serialize)] +#[serde(rename_all = "snake_case")] +enum PrimaryClass { + Exact, + ImplementationDefect, + UnsupportedCharacterReview, + UnclassifiedEncodingErrorReview, + CorpusSuspect, + ComparisonMethod, + PendingRuleReview, +} + +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Serialize)] +#[serde(rename_all = "snake_case")] +enum Reason { + Exact, + ConflictingDuplicateReference, + Rule34RomanIndicatorBeforeOpeningParenthesis, + RomanEllipsisUsesKoreanCellsInRomanEnclosure, + UebGrade1BeforeNonstandingOpeningParenthesis, + UebCapitalizedPassageWrittenAsSeparateCapitalWords, + BrailleWhitespaceEquivalent, + NfcInputEquivalent, + NfkcInputEquivalent, + RomanIndicatorAfterCapitalIndicator, + UnsupportedCharacterReview, + UnclassifiedEncodingErrorReview, + ForeignTextRuleReview, + NumberRuleReview, + PunctuationRuleReview, + KoreanRuleReview, +} + +#[derive(Debug, Serialize)] +struct Sample { + shard: String, + index: usize, + input: String, + expected_excerpt: String, + actual_excerpt: String, + error: Option, +} + +#[derive(Debug, Default, Serialize)] +struct ShardStats { + total: usize, + exact: usize, +} + +#[derive(Debug, Serialize)] +struct ErrorCharacterStats { + cases: usize, + nfkc: String, + family: &'static str, +} + +#[derive(Debug, Serialize)] +struct Rule36ComplexErrorSample { + shard: String, + index: usize, + input: String, + other_unsupported_characters: Vec, +} + +#[derive(Debug, Default, Serialize)] +struct Rule36TransitionAudit { + presentation_cases: usize, + observed_transitions: BTreeMap, + remaining_complex_errors: usize, + remaining_complex_error_samples: Vec, +} + +#[derive(Debug, Serialize)] +struct UnclassifiedEncodingErrorSample { + shard: String, + index: usize, + input: String, + error: String, +} + +#[derive(Debug, Serialize)] +struct MultipleSingletonErrorSample { + shard: String, + index: usize, + input: String, + unsupported_characters: Vec, +} + +#[derive(Debug, Default, Serialize)] +struct EncodingErrorAudit { + raw_total: usize, + resolved_by_comparison_method: usize, + excluded_as_corpus_suspect: usize, + unresolved_review_total: usize, + explained_by_singleton_unsupported: usize, + multiple_singleton_unsupported: usize, + multiple_singleton_samples: Vec, + unclassified_without_singleton: usize, + unclassified_samples: Vec, +} + +#[derive(Clone, Debug, Serialize)] +struct PendingRuleReviewClusterSample { + shard: String, + index: usize, + input: String, + expected_excerpt: String, + actual_excerpt: String, + first_difference_cell: Option, + error: Option, + primary_class: String, + reason: String, +} + +#[derive(Debug, Default, Serialize)] +struct PendingRuleReviewClusterStats { + candidates: usize, + exact: usize, + mismatch: usize, + conflicting_reference_cases: usize, + output_signature_mismatches_evaluated: usize, + first_difference_in_output_signature: usize, + first_difference_in_output_signature_transitions: BTreeMap, + actual_output_signature_outcomes: BTreeMap, + mismatch_primary_classes: BTreeMap, + samples: BTreeMap>, +} + +#[derive(Debug, Default, Serialize)] +struct FirstDifferenceTransitionStats { + cases: usize, + samples: Vec, +} + +#[derive(Debug, Serialize)] +struct AnalysisReport { + corpus: &'static str, + total: usize, + exact: usize, + mismatch: usize, + exact_percent: f64, + duplicate_inputs: usize, + conflicting_duplicate_inputs: usize, + primary_classes: BTreeMap, + reasons: BTreeMap, + encoding_error_messages: BTreeMap, + encoding_error_families: BTreeMap, + singleton_error_characters: BTreeMap, + encoding_error_audit: EncodingErrorAudit, + rule_36_transition_audit: Rule36TransitionAudit, + // Cross-cutting input cohorts; only members whose existing primary class + // is PendingRuleReview are pending-rule-review subclusters. + pending_rule_review_clusters: BTreeMap, + pending_first_difference_cell_transitions: BTreeMap, + pending_first_difference_transitions_after_localized_cohorts: + BTreeMap, + compact_numeric_ascii_suffixes: BTreeMap, + grade1_shortform_prefix_surfaces: BTreeMap, + grade1_numeric_continuation_surfaces: BTreeMap, + grade1_hyphen_continuation_surfaces: BTreeMap, + overlapping_traits: BTreeMap, + shards: BTreeMap, + samples: BTreeMap>, +} + +#[derive(Debug)] +struct Config { + report_path: PathBuf, + json_path: PathBuf, + sample_limit: usize, + threads: usize, +} + +impl Config { + fn parse() -> Result { + let workspace = Path::new(env!("CARGO_MANIFEST_DIR")).join("../.."); + let mut config = Self { + report_path: workspace.join("docs/corpus-analysis/NIKL_2025_V1.md"), + json_path: workspace.join("target/nikl-corpus-analysis.json"), + sample_limit: 5, + threads: thread::available_parallelism() + .map_or(1, usize::from) + .min(8), + }; + + let mut args = std::env::args().skip(1); + while let Some(arg) = args.next() { + match arg.as_str() { + "--report" => { + config.report_path = + PathBuf::from(args.next().ok_or("--report requires a path")?); + } + "--json" => { + config.json_path = PathBuf::from(args.next().ok_or("--json requires a path")?); + } + "--sample-limit" => { + config.sample_limit = args + .next() + .ok_or("--sample-limit requires a number")? + .parse() + .map_err(|_| "--sample-limit must be a positive integer")?; + } + "--threads" => { + config.threads = args + .next() + .ok_or("--threads requires a number")? + .parse() + .map_err(|_| "--threads must be a positive integer")?; + if config.threads == 0 { + return Err("--threads must be at least 1".to_string()); + } + } + "--help" | "-h" => { + println!( + "nikl_corpus_analyze [--report PATH] [--json PATH] \ + [--sample-limit N] [--threads N]" + ); + std::process::exit(0); + } + _ => return Err(format!("unknown argument: {arg}")), + } + } + Ok(config) + } +} + +fn load_cases() -> Result, String> { + let corpus_dir = Path::new(env!("CARGO_MANIFEST_DIR")).join("../../test_cases/corpus"); + let mut paths = fs::read_dir(&corpus_dir) + .map_err(|error| format!("cannot read {}: {error}", corpus_dir.display()))? + .map(|entry| entry.map(|entry| entry.path())) + .collect::, _>>() + .map_err(|error| format!("cannot enumerate corpus shards: {error}"))?; + paths.retain(|path| { + path.file_name() + .and_then(|name| name.to_str()) + .is_some_and(braillify::corpus_analysis::is_sentence_corpus_shard_name) + }); + paths.sort(); + let shard_count = paths.len(); + + let mut located = Vec::new(); + for path in paths { + let shard = path + .file_name() + .and_then(|name| name.to_str()) + .ok_or_else(|| format!("non-Unicode shard path: {}", path.display()))? + .to_string(); + let cases: Vec = serde_json::from_reader(BufReader::new( + File::open(&path) + .map_err(|error| format!("cannot open {}: {error}", path.display()))?, + )) + .map_err(|error| format!("invalid JSON in {}: {error}", path.display()))?; + located.extend( + cases + .into_iter() + .enumerate() + .map(|(index, case)| LocatedCase { + shard: shard.clone(), + index: index + 1, + case, + }), + ); + } + validate_corpus_shape(shard_count, located.len())?; + Ok(located) +} + +fn validate_corpus_shape(shard_count: usize, case_count: usize) -> Result<(), String> { + if shard_count == 0 { + return Err("no NIKL corpus shards matched sentence_*.json".to_string()); + } + if case_count == 0 { + return Err("NIKL corpus shards contained zero cases".to_string()); + } + Ok(()) +} + +fn singleton_unsupported_set_with( + cases: &[LocatedCase], + fails_alone: &mut dyn FnMut(char) -> bool, +) -> BTreeSet { + cases + .iter() + .flat_map(|located| located.case.input.chars()) + .collect::>() + .into_iter() + .filter(|ch| fails_alone(*ch)) + .collect() +} + +fn singleton_unsupported_set(cases: &[LocatedCase]) -> BTreeSet { + singleton_unsupported_set_with(cases, &mut |ch| { + braillify::encode_to_unicode(&ch.to_string()).is_err() + }) +} + +fn encode_cases(cases: &[LocatedCase], thread_count: usize) -> Vec { + // Compute singleton support exactly once per distinct corpus character. + // `BTreeSet` keeps both probing and report assignment deterministic; workers + // only perform membership lookups instead of re-encoding common marks such + // as `㈜` hundreds of times across error sentences. + let singleton_unsupported = singleton_unsupported_set(cases); + let chunk_size = cases.len().div_ceil(thread_count); + let mut chunks = thread::scope(|scope| { + let singleton_unsupported = &singleton_unsupported; + cases + .chunks(chunk_size.max(1)) + .map(|chunk| { + scope.spawn(move || { + chunk + .iter() + .cloned() + .map(|located| { + let actual = braillify::encode_to_unicode(&located.case.input); + let singleton_unsupported_characters = if actual.is_err() { + located + .case + .input + .chars() + .collect::>() + .into_iter() + .filter(|ch| singleton_unsupported.contains(ch)) + .collect() + } else { + Vec::new() + }; + let nfc: String = located.case.input.nfc().collect(); + let nfc_actual = (nfc != located.case.input) + .then(|| braillify::encode_to_unicode(&nfc)); + let nfkc: String = located.case.input.nfkc().collect(); + let nfkc_actual = (nfkc != located.case.input) + .then(|| braillify::encode_to_unicode(&nfkc)); + EncodedCase { + located, + actual, + nfc_actual, + nfkc_actual, + singleton_unsupported_characters, + } + }) + .collect::>() + }) + }) + .collect::>() + .into_iter() + .map(|handle| handle.join().expect("analysis worker panicked")) + .collect::>() + }); + chunks.drain(..).flatten().collect() +} + +fn normalized_braille_whitespace(text: &str) -> String { + text.chars() + .map(|ch| match ch { + ' ' | '\t' | '\r' | '\n' | '\u{00a0}' | '\u{3000}' => '\u{2800}', + _ => ch, + }) + .collect() +} + +/// Correct only the ordering defect justified by Korean rules 28 appendix and 29: +/// the Korean roman indicator must precede UEB grade-1/capital indicators. +fn roman_before_capital_order(text: &str) -> String { + text.replace("⠠⠠⠠⠴", "⠴⠠⠠⠠") + .replace("⠰⠠⠠⠴", "⠴⠰⠠⠠") + .replace("⠠⠠⠴", "⠴⠠⠠") +} + +fn conflicting_inputs(cases: &[LocatedCase]) -> (usize, BTreeSet) { + let mut references = BTreeMap::>::new(); + for located in cases { + references + .entry(located.case.input.clone()) + .or_default() + .insert(located.case.unicode.clone()); + } + let duplicate_count = cases.len().saturating_sub(references.len()); + let conflicting = references + .into_iter() + .filter_map(|(input, values)| (values.len() > 1).then_some(input)) + .collect(); + (duplicate_count, conflicting) +} + +fn classify(encoded: &EncodedCase, conflicting: &BTreeSet) -> (PrimaryClass, Reason) { + let expected = &encoded.located.case.unicode; + match &encoded.actual { + Ok(actual) if actual == expected => (PrimaryClass::Exact, Reason::Exact), + _ if conflicting.contains(&encoded.located.case.input) => ( + PrimaryClass::CorpusSuspect, + Reason::ConflictingDuplicateReference, + ), + Ok(actual) + if normalized_braille_whitespace(actual) == normalized_braille_whitespace(expected) => + { + ( + PrimaryClass::ComparisonMethod, + Reason::BrailleWhitespaceEquivalent, + ) + } + _ if encoded + .nfc_actual + .as_ref() + .is_some_and(|result| result.as_ref().is_ok_and(|actual| actual == expected)) => + { + (PrimaryClass::ComparisonMethod, Reason::NfcInputEquivalent) + } + _ if encoded + .nfkc_actual + .as_ref() + .is_some_and(|result| result.as_ref().is_ok_and(|actual| actual == expected)) => + { + (PrimaryClass::ComparisonMethod, Reason::NfkcInputEquivalent) + } + _ if is_rule_34_reference_order_contradiction(encoded) => ( + PrimaryClass::CorpusSuspect, + Reason::Rule34RomanIndicatorBeforeOpeningParenthesis, + ), + _ if is_roman_ellipsis_reference_contradiction(encoded) => ( + PrimaryClass::CorpusSuspect, + Reason::RomanEllipsisUsesKoreanCellsInRomanEnclosure, + ), + _ if is_ueb_grade1_before_nonstanding_opening_parenthesis_reference_contradiction( + encoded, + ) => + { + ( + PrimaryClass::CorpusSuspect, + Reason::UebGrade1BeforeNonstandingOpeningParenthesis, + ) + } + _ if is_ueb_capitalized_passage_reference_contradiction(encoded) => ( + PrimaryClass::CorpusSuspect, + Reason::UebCapitalizedPassageWrittenAsSeparateCapitalWords, + ), + Ok(actual) if roman_before_capital_order(actual) == *expected => ( + PrimaryClass::ImplementationDefect, + Reason::RomanIndicatorAfterCapitalIndicator, + ), + Err(_) if !encoded.singleton_unsupported_characters.is_empty() => ( + PrimaryClass::UnsupportedCharacterReview, + Reason::UnsupportedCharacterReview, + ), + Err(_) => ( + PrimaryClass::UnclassifiedEncodingErrorReview, + Reason::UnclassifiedEncodingErrorReview, + ), + Ok(_) + if encoded + .located + .case + .input + .chars() + .any(|ch| ch.is_ascii_alphabetic()) => + { + ( + PrimaryClass::PendingRuleReview, + Reason::ForeignTextRuleReview, + ) + } + Ok(_) + if encoded + .located + .case + .input + .chars() + .any(|ch| ch.is_ascii_digit()) => + { + (PrimaryClass::PendingRuleReview, Reason::NumberRuleReview) + } + Ok(_) + if encoded + .located + .case + .input + .chars() + .any(is_delimiter_or_quote) => + { + ( + PrimaryClass::PendingRuleReview, + Reason::PunctuationRuleReview, + ) + } + Ok(_) => (PrimaryClass::PendingRuleReview, Reason::KoreanRuleReview), + } +} + +fn is_roman_numeral_presentation(ch: char) -> bool { + (0x2160..=0x217f).contains(&(ch as u32)) +} + +/// Reconstruct an observable before/after transition for the rule-36 cohort +/// without assigning today's primary-class policy to the historical run. +/// Before targeted normalization, direct/NFC encoding failed; an exact NFKC +/// path was observable separately. The current side reports only whether the +/// case is exact, an encoded mismatch awaiting rule review, or still blocked by +/// another independently unsupported singleton character. +fn rule_36_observed_transition(encoded: &EncodedCase) -> Option<&'static str> { + if !encoded + .located + .case + .input + .chars() + .any(is_roman_numeral_presentation) + { + return None; + } + + let expected = &encoded.located.case.unicode; + let before = if encoded + .nfkc_actual + .as_ref() + .is_some_and(|result| result.as_ref().is_ok_and(|actual| actual == expected)) + { + "nfkc_input_equivalent" + } else { + "encoding_error" + }; + let after = match &encoded.actual { + Ok(actual) if actual == expected => "exact", + Ok(_) => "encoded_mismatch_pending_rule_review", + Err(_) if !encoded.singleton_unsupported_characters.is_empty() => { + "unsupported_character_review" + } + Err(_) => "unclassified_encoding_error_review", + }; + Some(match (before, after) { + ("nfkc_input_equivalent", "exact") => "nfkc_input_equivalent -> exact", + ("encoding_error", "encoded_mismatch_pending_rule_review") => { + "encoding_error -> encoded_mismatch_pending_rule_review" + } + ("encoding_error", "unsupported_character_review") => { + "encoding_error -> unsupported_character_review" + } + ("encoding_error", "unclassified_encoding_error_review") => { + "encoding_error -> unclassified_encoding_error_review" + } + _ => "other_observed_transition", + }) +} + +fn is_delimiter_or_quote(ch: char) -> bool { + matches!( + ch, + '(' | ')' | '[' | ']' | '{' | '}' | '“' | '”' | '‘' | '’' | '"' | '\'' + ) +} + +const UPPERCASE_ROMAN_HEADWORD_EXPANSION: &str = + "uppercase_roman_headword_closed_multiword_parenthetical"; +const STANDALONE_UPPERCASE_ROMAN_WORD: &str = "standalone_multi_character_uppercase_roman_word"; +const KOREAN_PREFIXED_ALLCAPS_PARENTHETICAL: &str = "korean_prefixed_closed_allcaps_parenthetical"; +const KOREAN_PREFIXED_CLOSED_ROMAN_ANNOTATION: &str = + "korean_prefixed_closed_roman_annotation_rule_34_order"; +const ALLCAPS_ROMAN_MIDDLE_DOT_RUNS: &str = + "multi_character_allcaps_roman_runs_joined_by_middle_dot"; +const ROMAN_RUN_BEFORE_MIDDLE_DOT_BOUNDARY: &str = + "roman_run_immediately_before_attached_middle_dot_boundary"; +const ATTACHED_ASCII_ROMAN_TO_KOREAN_BOUNDARY: &str = + "attached_ascii_roman_to_korean_script_boundary"; +const KOREAN_MAJORITY_ROMAN_SANDWICH_NON_DOMAIN: &str = + "korean_majority_same_token_roman_sandwich_non_domain"; +const KOREAN_INLINE_PARENTHESIZED_OPERATOR: &str = + "korean_inline_parenthesized_single_arithmetic_operator"; +const ATTACHED_PLUS_PARENTHESIZED_KOREAN_GLOSS: &str = + "attached_plus_followed_by_parenthesized_korean_gloss"; +const TIGHT_TRIANGLE_BEFORE_KOREAN: &str = "tight_triangle_mark_immediately_before_korean"; +const ATTACHED_KOREAN_AUXILIARY_ITDA_SPACING: &str = "attached_korean_auxiliary_itda_spacing"; +const ALLCAPS_ROMAN_RUN_CONTAINING_OU: &str = "allcaps_roman_run_containing_ou"; +const ALLCAPS_ROMAN_RUN_CONTAINING_ST: &str = "allcaps_roman_run_containing_st"; +const ALLCAPS_ROMAN_RUN_CONTAINING_AR: &str = "allcaps_roman_run_containing_ar"; +const ALLCAPS_ROMAN_RUN_CONTAINING_ED: &str = "allcaps_roman_run_containing_ed"; +const ROMAN_RUN_AFTER_CLOSED_ROMAN_ENCLOSURE: &str = + "roman_run_after_whitespace_following_closed_roman_enclosure"; +const ATTACHED_ASCII_ROMAN_SEGMENTS_JOINED_BY_AMPERSAND: &str = + "attached_ascii_roman_segments_joined_by_ampersand"; +const UPPERCASE_ASCII_SEGMENTS_JOINED_BY_AMPERSAND: &str = + "uppercase_ascii_segments_joined_by_ampersand_capitalization"; +const CAPITALS_WORD_NONLETTER_CHANGE_SCOPE: &str = + "capitals_word_mode_previously_spanning_nonletter_scope"; +const AMPERSAND_BEFORE_ATTACHED_ASCII_ROMAN_SEGMENT: &str = + "ampersand_before_attached_ascii_roman_segment"; +const ASCII_APOSTROPHE_BETWEEN_ASCII_LETTERS: &str = "ascii_apostrophe_between_ascii_letter_runs"; +const SPACED_COMMA_BETWEEN_ASCII_DIGIT_RUNS: &str = "spaced_comma_between_ascii_digit_runs"; +const ASCII_ROMAN_COMMA_BEFORE_DIGIT_KOREAN_TOKEN: &str = + "ascii_roman_tail_comma_before_digit_korean_token"; +const PERCENT_POINT_UNIT_LIST_COMMA: &str = "percent_point_unit_list_comma"; +const SINGLE_CAPITAL_PARENTHESIZED_DIGITS: &str = "single_capital_followed_by_parenthesized_digits"; +const MIXED_ROMAN_KOREAN_BEFORE_HEADWORD_EXPANSION: &str = + "mixed_roman_korean_word_before_uppercase_headword_expansion"; +const UPPERCASE_ROMAN_HYPHEN_DIGITS: &str = "uppercase_roman_run_followed_by_hyphen_digits"; +const UPPERCASE_ALPHANUMERIC_ROMAN_DIGIT_SEQUENCE: &str = + "uppercase_alphanumeric_roman_digit_sequence"; +const DECIMAL_POINT_BETWEEN_DIGITS: &str = "decimal_point_between_ascii_digits"; +const COMPACT_NUMERIC_ASCII_SUFFIX: &str = "compact_numeric_ascii_letter_suffix"; +const RULE69_ASCII_UNIT_BEFORE_TERMINATOR_SKIPPING_SYMBOL: &str = + "rule69_ascii_unit_before_terminator_skipping_symbol"; +const ALLCAPS_SHORTFORM_PREFIX_COLLISION: &str = + "allcaps_roman_run_beginning_with_pure_letter_shortform"; +const ROMAN_UPPERCASE_AFTER_DIGIT: &str = + "uppercase_ascii_run_immediately_after_digit_in_roman_sequence"; +const ROMAN_UPPERCASE_AFTER_HYPHEN: &str = + "uppercase_ascii_run_immediately_after_hyphen_in_roman_sequence"; +const PURE_ALLCAPS_HYPHEN_MULTI_ALLCAPS: &str = + "pure_allcaps_segment_before_hyphen_and_multi_allcaps_segment_after"; +const KOREAN_TO_ROMAN_HYPHEN_BOUNDARY: &str = "attached_korean_to_roman_hyphen_boundary"; +const ROMAN_HYPHENATED_WORD_AFTER_KOREAN_WORD: &str = + "roman_hyphenated_word_after_whitespace_following_korean_word"; +const ROMAN_PARENTHETICAL_HEADWORD_AFTER_KOREAN_WORD: &str = + "roman_parenthetical_headword_after_whitespace_following_korean_word"; +const KOREAN_PREFIXED_ROMAN_PARENTHETICAL_HYPHEN_SUFFIX: &str = + "korean_prefixed_roman_parenthetical_followed_by_allcaps_hyphen_suffix"; +const CONSECUTIVE_ROMAN_UPPERCASE_WORD_REENTRY: &str = + "uppercase_word_after_whitespace_continuing_ascii_roman_text"; +const CONSECUTIVE_ASCII_ROMAN_WORD_BOUNDARY: &str = + "consecutive_ascii_roman_words_whitespace_boundary"; +const ROMAN_PARENTHETICAL_AFTER_NONLETTER_BOUNDARY: &str = + "closed_roman_parenthetical_after_non_ascii_letter_boundary"; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +struct InputSpan { + start_byte: usize, + end_byte: usize, +} + +/// Finds whitespace-delimited words containing an ASCII decimal point between +/// digits. Rule 43 explicitly keeps punctuation between digits in the same +/// numeric sequence, and rule 48 assigns the decimal-point cell. The whole +/// word is retained so the output locator reproduces suffix contexts such as +/// `%`, Roman units, Korean text, and closing punctuation. +fn decimal_word_spans(input: &str) -> Vec { + let mut spans = BTreeSet::new(); + for (dot_byte, _) in input.match_indices('.') { + let previous = input[..dot_byte].chars().next_back(); + let next = input[dot_byte + 1..].chars().next(); + if !previous.is_some_and(|ch| ch.is_ascii_digit()) + || !next.is_some_and(|ch| ch.is_ascii_digit()) + { + continue; + } + + let start_byte = input[..dot_byte] + .char_indices() + .rev() + .find_map(|(byte, ch)| ch.is_whitespace().then_some(byte + ch.len_utf8())) + .unwrap_or(0); + let end_byte = input[dot_byte + 1..] + .char_indices() + .find_map(|(offset, ch)| ch.is_whitespace().then_some(dot_byte + 1 + offset)) + .unwrap_or(input.len()); + spans.insert((start_byte, end_byte)); + } + spans + .into_iter() + .map(|(start_byte, end_byte)| InputSpan { + start_byte, + end_byte, + }) + .collect() +} + +/// Finds a compact numeric prefix followed immediately by one or more ASCII +/// letters, with alphanumeric outer boundaries. The shape includes rule-69 +/// units but deliberately does not declare every suffix a unit: mathematical +/// variables and identifiers can share the same surface form. +fn compact_numeric_ascii_suffix_spans(input: &str) -> Vec { + let bytes = input.as_bytes(); + let mut spans = Vec::new(); + let mut cursor = 0usize; + while cursor < bytes.len() { + if !bytes[cursor].is_ascii_digit() + || input[..cursor] + .chars() + .next_back() + .is_some_and(|ch| ch.is_ascii_alphanumeric()) + { + cursor += input[cursor..] + .chars() + .next() + .expect("cursor must remain on a character boundary") + .len_utf8(); + continue; + } + + let start_byte = cursor; + while bytes + .get(cursor) + .is_some_and(|byte| byte.is_ascii_digit() || matches!(*byte, b',' | b'.')) + { + cursor += 1; + } + let suffix_start = cursor; + while bytes.get(cursor).is_some_and(u8::is_ascii_alphabetic) { + cursor += 1; + } + if cursor > suffix_start + && input[cursor..] + .chars() + .next() + .is_none_or(|ch| !ch.is_ascii_alphanumeric()) + { + spans.push(InputSpan { + start_byte, + end_byte: cursor, + }); + } + } + spans +} + +fn compact_numeric_ascii_suffix(span: InputSpan, input: &str) -> &str { + input[span.start_byte..span.end_byte] + .trim_start_matches(|ch: char| ch.is_ascii_digit() || matches!(ch, ',' | '.')) +} + +/// Refines the broad compact-suffix cohort to spellings already recognized by +/// rule 69, immediately followed by punctuation for which rules 33/34 omit the +/// Roman terminator. The span includes that punctuation so the output locator +/// measures the exact unit-to-punctuation boundary rather than mere coexistence. +fn rule69_ascii_unit_before_terminator_skipping_symbol_spans(input: &str) -> Vec { + const RULE69_ASCII_UNITS: &[&str] = &["min", "cal", "cm", "kg", "in", "mm", "GB", "m", "h"]; + + compact_numeric_ascii_suffix_spans(input) + .into_iter() + .filter_map(|span| { + let suffix = compact_numeric_ascii_suffix(span, input); + let symbol = input[span.end_byte..].chars().next()?; + (RULE69_ASCII_UNITS.contains(&suffix) + && matches!( + symbol, + '.' | '?' + | '!' + | '…' + | '⋯' + | '"' + | '\'' + | '”' + | '’' + | '」' + | '』' + | '〉' + | '》' + | '(' + | ')' + | ']' + | '}' + | ',' + | ':' + | ';' + | '―' + )) + .then_some(InputSpan { + start_byte: span.start_byte, + end_byte: span.end_byte + symbol.len_utf8(), + }) + }) + .collect() +} + +fn first_difference_at_rule69_ascii_unit_terminator_boundary(item: &EncodedCase) -> bool { + first_difference_in_compact_numeric_ascii_suffix_spans( + item, + &rule69_ascii_unit_before_terminator_skipping_symbol_spans(&item.located.case.input), + ) +} + +fn first_difference_in_compact_numeric_ascii_suffix(item: &EncodedCase) -> bool { + first_difference_in_compact_numeric_ascii_suffix_spans( + item, + &compact_numeric_ascii_suffix_spans(&item.located.case.input), + ) +} + +fn first_difference_in_compact_numeric_ascii_suffix_spans( + item: &EncodedCase, + spans: &[InputSpan], +) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + korean_context_signature_ranges(&item.located.case.input, actual, spans, 1) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +fn first_difference_in_decimal_word(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + korean_context_signature_ranges( + &item.located.case.input, + actual, + &decimal_word_spans(&item.located.case.input), + 0, + ) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +/// Finds rule-34-shaped annotations whose opening parenthesis immediately +/// follows Korean script and whose closed body contains only ordinary Roman +/// letters, digits, apostrophes, periods, or hyphens. This is an input gate; +/// the separate output locator decides whether a first difference is at the +/// opening-parenthesis order established by the PDF example. +fn korean_prefixed_closed_roman_annotation_spans(input: &str) -> Vec { + let mut spans = Vec::new(); + for (open_byte, _) in input.match_indices('(') { + if !input[..open_byte] + .chars() + .next_back() + .is_some_and(is_korean_script) + { + continue; + } + + let body_start = open_byte + 1; + let Some(close_offset) = input[body_start..].find(')') else { + continue; + }; + let close_byte = body_start + close_offset; + let body = &input[body_start..close_byte]; + if !body.is_empty() + && body.chars().any(|ch| ch.is_ascii_alphabetic()) + && body + .chars() + .all(|ch| ch.is_ascii_alphanumeric() || matches!(ch, '-' | '\'' | '.')) + { + spans.push(InputSpan { + start_byte: open_byte, + end_byte: close_byte + 1, + }); + } + } + spans +} + +/// Locate only the current engine's Korean opening-parenthesis cells. The +/// signature is derived from a neutral Korean probe and verified at the cell +/// offset obtained by encoding the real prefix, never from corpus expected. +fn korean_prefixed_annotation_opening_ranges( + input: &str, + actual: &str, +) -> Vec> { + let actual_cells = actual.chars().collect::>(); + let korean = braillify::encode_to_unicode("가").expect("neutral Korean probe must encode"); + let korean_with_open = + braillify::encode_to_unicode("가(").expect("Korean opening-parenthesis probe must encode"); + let korean_cells = korean.chars().count(); + let opening = korean_with_open + .chars() + .skip(korean_cells) + .collect::>(); + + korean_prefixed_closed_roman_annotation_spans(input) + .into_iter() + .filter_map(|span| { + let prefix = braillify::encode_to_unicode(&input[..span.start_byte]).ok()?; + let start = prefix.chars().count(); + let end = start.checked_add(opening.len())?; + (actual_cells.get(start..end) == Some(opening.as_slice())).then_some(start..end) + }) + .collect() +} + +fn first_difference_in_korean_prefixed_annotation_opening(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + korean_prefixed_annotation_opening_ranges(&item.located.case.input, actual) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +/// The PDF's rule-34 example emits the printed Korean opening parenthesis +/// before entering Roman mode: `⠦⠄⠴`. A corpus reference that instead starts +/// this same localized input structure with Roman mode plus the UEB opening +/// parenthesis (`⠴⠐⠣`) contradicts that explicit order. Requiring both +/// three-cell signatures avoids reclassifying unrelated mismatches in the +/// broad input cohort. +fn is_rule_34_reference_order_contradiction(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + let expected = &item.located.case.unicode; + if actual == expected { + return false; + } + + let difference = first_difference_cell(expected, actual); + let expected_cells = expected.chars().collect::>(); + let actual_cells = actual.chars().collect::>(); + expected_cells.get(difference..difference + 3) == Some(&['⠴', '⠐', '⠣']) + && actual_cells.get(difference..difference + 3) == Some(&['⠦', '⠄', '⠴']) + && korean_prefixed_annotation_opening_ranges(&item.located.case.input, actual) + .into_iter() + .any(|range| range.start == difference) +} + +/// UEB 2024 section 7.3 assigns U+2026 the same three full-stop cells as the +/// print spelling `...`. When the ellipsis is attached to Roman letters and +/// immediately closes their enclosure, a reference using Korean rule-53 +/// middle-dot cells is an independently reproducible standard conflict. +fn is_roman_ellipsis_reference_contradiction(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + let expected = &item.located.case.unicode; + if actual == expected { + return false; + } + + let has_roman_enclosed_ellipsis = item + .located + .case + .input + .chars() + .collect::>() + .windows(3) + .any(|window| { + window[0].is_ascii_alphabetic() + && window[1] == '…' + && matches!(window[2], ')' | ']' | '}' | '”' | '’' | '」' | '』') + }); + if !has_roman_enclosed_ellipsis { + return false; + } + + let difference = first_difference_cell(expected, actual); + let expected_cells = expected.chars().collect::>(); + let actual_cells = actual.chars().collect::>(); + expected_cells.get(difference..difference + 3) == Some(&['⠠', '⠠', '⠠']) + && actual_cells.get(difference..difference + 3) == Some(&['⠲', '⠲', '⠲']) +} + +#[derive(Clone, Copy)] +struct AnalyzerCapitalizedGroup { + start: usize, + end: usize, + capital_count: usize, +} + +fn is_capitals_opening_punctuation(ch: char) -> bool { + matches!( + ch, + '\'' | '"' | '\u{2018}' | '\u{201c}' | '(' | '[' | '{' | '〈' | '《' | '「' | '『' + ) +} + +fn is_capitals_closing_quote(ch: char) -> bool { + matches!( + ch, + '\'' | '"' | '\u{2019}' | '\u{201d}' | '〉' | '》' | '」' | '』' + ) +} + +fn analyzer_capitalized_group(word: &str) -> Option { + let chars = word.chars().collect::>(); + if chars.iter().any(char::is_ascii_lowercase) { + return None; + } + + let start = chars.iter().position(char::is_ascii_uppercase)?; + if !chars[..start] + .iter() + .copied() + .all(is_capitals_opening_punctuation) + { + return None; + } + let last_capital = chars.iter().rposition(char::is_ascii_uppercase)?; + if chars[start..=last_capital] + .iter() + .any(|ch| is_korean_script(*ch)) + { + return None; + } + + let mut end = chars.len(); + for index in last_capital + 1..chars.len() { + let ch = chars[index]; + let opens_attached_korean_gloss = matches!(ch, '(' | '[' | '{') + && chars[index + 1..] + .iter() + .any(|next| is_korean_script(*next)); + if is_korean_script(ch) || is_capitals_closing_quote(ch) || opens_attached_korean_gloss { + end = index; + break; + } + } + + Some(AnalyzerCapitalizedGroup { + start, + end, + capital_count: chars.iter().filter(|ch| ch.is_ascii_uppercase()).count(), + }) +} + +/// Return the number of per-sequence capitalization cells used by the +/// non-passage spelling for each UEB 8.5.2 passage candidate in `input`. +/// A single capital uses one cell; a multi-letter capitals word uses two. +fn separate_capital_indicator_counts_for_passages(input: &str) -> Vec { + let words = input.split_whitespace().collect::>(); + let groups = words + .iter() + .map(|word| analyzer_capitalized_group(word)) + .collect::>(); + let word_lengths = words + .iter() + .map(|word| word.chars().count()) + .collect::>(); + let mut counts = Vec::new(); + let mut index = 0usize; + + while index + 2 < words.len() { + let Some(current) = groups[index] else { + index += 1; + continue; + }; + let Some(first) = groups[index + 1] else { + index += 1; + continue; + }; + let Some(second) = groups[index + 2] else { + index += 1; + continue; + }; + if current.end != word_lengths[index] + || first.start != 0 + || first.end != word_lengths[index + 1] + || second.start != 0 + { + index += 1; + continue; + } + + let mut end = index + 3; + while end < words.len() && groups[end].is_some_and(|group| group.start == 0) { + end += 1; + } + let separate_cells = groups[index..end] + .iter() + .flatten() + .map(|group| if group.capital_count == 1 { 1 } else { 2 }) + .sum(); + counts.push(separate_cells); + index = end; + } + + counts +} + +fn matches_after_removing_capital_cells( + expected: &[char], + actual: &[char], + expected_index: usize, + actual_index: usize, + removed: usize, + required_removed: usize, + failed: &mut BTreeSet<(usize, usize, usize)>, +) -> bool { + let state = (expected_index, actual_index, removed); + if failed.contains(&state) { + return false; + } + if expected_index == expected.len() && actual_index == actual.len() { + return removed == required_removed; + } + + if expected.get(expected_index) == actual.get(actual_index) + && matches_after_removing_capital_cells( + expected, + actual, + expected_index + 1, + actual_index + 1, + removed, + required_removed, + failed, + ) + { + return true; + } + if removed < required_removed + && expected.get(expected_index) == Some(&'⠠') + && matches_after_removing_capital_cells( + expected, + actual, + expected_index + 1, + actual_index, + removed + 1, + required_removed, + failed, + ) + { + return true; + } + + failed.insert(state); + false +} + +/// UEB 2024 8.5.2 requires one capitals-passage indicator for three or more +/// capitalized symbols-sequences, and 8.5.3 places its terminator immediately +/// after the final affected sequence. A reference is classified only when the +/// complete sentence becomes identical by replacing that exact five-cell +/// passage pair with the structurally required one-/two-cell indicators for +/// each sequence. Unrelated differences therefore remain pending review. +fn is_ueb_capitalized_passage_reference_contradiction(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + let expected = &item.located.case.unicode; + if actual == expected { + return false; + } + + let indicator_counts = separate_capital_indicator_counts_for_passages(&item.located.case.input); + if indicator_counts.is_empty() { + return false; + } + let actual = actual.chars().collect::>(); + let expected = expected.chars().collect::>(); + + for start in 0..actual.len().saturating_sub(2) { + if actual.get(start..start + 3) != Some(&['⠠', '⠠', '⠠']) + || expected.get(..start) != actual.get(..start) + { + continue; + } + for terminator in start + 3..actual.len().saturating_sub(1) { + if actual.get(terminator..terminator + 2) != Some(&['⠠', '⠄']) { + continue; + } + let actual_suffix_start = terminator + 2; + let suffix_len = actual.len() - actual_suffix_start; + let Some(expected_segment_end) = expected.len().checked_sub(suffix_len) else { + continue; + }; + if expected_segment_end < start + || expected.get(expected_segment_end..) != actual.get(actual_suffix_start..) + { + continue; + } + + let actual_content = &actual[start + 3..terminator]; + let expected_segment = &expected[start..expected_segment_end]; + for required_removed in &indicator_counts { + if expected_segment.len() != actual_content.len() + required_removed { + continue; + } + if matches_after_removing_capital_cells( + expected_segment, + actual_content, + 0, + 0, + 0, + *required_removed, + &mut BTreeSet::new(), + ) { + return true; + } + } + } + } + false +} + +/// Finds maximal all-caps ASCII runs containing the adjacent letters `OU`. +/// +/// This is an input gate for a pronunciation-sensitive UEB diagnostic, not a +/// claim that the run is an initialism. Alphanumeric outer boundaries exclude +/// fragments of identifiers while retaining parenthesized and standalone runs. +fn allcaps_roman_runs_containing_pair(input: &str, pair: &[u8; 2]) -> Vec { + let bytes = input.as_bytes(); + let mut runs = Vec::new(); + let mut cursor = 0; + while cursor < bytes.len() { + if !bytes[cursor].is_ascii_alphabetic() { + cursor += input[cursor..] + .chars() + .next() + .expect("cursor must remain on a character boundary") + .len_utf8(); + continue; + } + + let start_byte = cursor; + while cursor < bytes.len() && bytes[cursor].is_ascii_alphabetic() { + cursor += 1; + } + let end_byte = cursor; + let run = &input[start_byte..end_byte]; + let previous = input[..start_byte].chars().next_back(); + let next = input[end_byte..].chars().next(); + if run.len() >= 2 + && run.bytes().all(|byte| byte.is_ascii_uppercase()) + && run.as_bytes().windows(2).any(|window| window == pair) + && previous.is_none_or(|ch| !ch.is_ascii_alphanumeric()) + && next.is_none_or(|ch| !ch.is_ascii_alphanumeric()) + { + runs.push(InputSpan { + start_byte, + end_byte, + }); + } + } + runs +} + +fn allcaps_roman_runs_containing_ou(input: &str) -> Vec { + allcaps_roman_runs_containing_pair(input, b"OU") +} + +fn allcaps_roman_runs_containing_st(input: &str) -> Vec { + allcaps_roman_runs_containing_pair(input, b"ST") +} + +fn allcaps_roman_runs_containing_ar(input: &str) -> Vec { + allcaps_roman_runs_containing_pair(input, b"AR") +} + +fn allcaps_roman_runs_containing_ed(input: &str) -> Vec { + allcaps_roman_runs_containing_pair(input, b"ED") +} + +fn enclosure_contains_ascii_roman(input: &str, closer_byte: usize, closer: char) -> bool { + let (opener, search_end) = match closer { + ')' => ('(', closer_byte), + '’' => ('‘', closer_byte), + '”' => ('“', closer_byte), + _ => return false, + }; + let Some(open_byte) = input[..search_end].rfind(opener) else { + return false; + }; + input[open_byte + opener.len_utf8()..closer_byte] + .chars() + .any(|ch| ch.is_ascii_alphabetic()) +} + +/// Finds an ASCII-letter run after whitespace that follows a closed +/// Roman-containing parenthesis or curly-quoted span. The +/// optional comma/colon/semicolon and opening quote model corpus punctuation; +/// no output or reference cells participate in this input gate. +fn roman_run_after_closed_roman_enclosure_spans(input: &str) -> Vec { + let bytes = input.as_bytes(); + let mut spans = Vec::new(); + let mut cursor = 0usize; + while cursor < bytes.len() { + if !bytes[cursor].is_ascii_alphabetic() { + cursor += input[cursor..] + .chars() + .next() + .expect("cursor must remain on a character boundary") + .len_utf8(); + continue; + } + + let start_byte = cursor; + while cursor < bytes.len() && bytes[cursor].is_ascii_alphabetic() { + cursor += 1; + } + let end_byte = cursor; + if input[start_byte..end_byte].is_empty() + || input[..start_byte] + .chars() + .next_back() + .is_some_and(|ch| ch.is_ascii_alphanumeric()) + || input[end_byte..] + .chars() + .next() + .is_some_and(|ch| ch.is_ascii_alphanumeric()) + { + continue; + } + + let mut boundary = start_byte; + while let Some((offset, ch)) = input[..boundary].char_indices().next_back() { + if matches!(ch, '‘' | '“') { + boundary = offset; + } else { + break; + } + } + + let mut saw_whitespace = false; + while let Some((offset, ch)) = input[..boundary].char_indices().next_back() { + if ch.is_whitespace() { + saw_whitespace = true; + boundary = offset; + } else { + break; + } + } + if !saw_whitespace { + continue; + } + + while let Some((offset, ch)) = input[..boundary].char_indices().next_back() { + if matches!(ch, ',' | ':' | ';') { + boundary = offset; + } else { + break; + } + } + let Some((closer_byte, closer)) = input[..boundary].char_indices().next_back() else { + continue; + }; + if enclosure_contains_ascii_roman(input, closer_byte, closer) { + spans.push(InputSpan { + start_byte, + end_byte, + }); + } + } + spans +} + +/// Finds complete ASCII-letter sequences joined directly by one or more +/// ampersands, such as the UEB §3.1.1 examples `AT&T` and `B&B`. Whitespace, +/// Korean text, empty segments, and alphanumeric outer continuations are +/// excluded so the cohort is an attached Roman-symbol boundary, not broad +/// sentence-level ampersand coexistence. +fn attached_ascii_roman_ampersand_spans(input: &str) -> Vec { + let bytes = input.as_bytes(); + let mut spans = Vec::new(); + let mut cursor = 0usize; + while cursor < bytes.len() { + if !bytes[cursor].is_ascii_alphabetic() { + cursor += input[cursor..] + .chars() + .next() + .expect("cursor must remain on a character boundary") + .len_utf8(); + continue; + } + + let start_byte = cursor; + while bytes + .get(cursor) + .is_some_and(|byte| byte.is_ascii_alphabetic() || *byte == b'&') + { + cursor += 1; + } + let end_byte = cursor; + let run = &input[start_byte..end_byte]; + let previous = input[..start_byte].chars().next_back(); + let next = input[end_byte..].chars().next(); + if run.contains('&') + && run.split('&').all(|segment| { + !segment.is_empty() && segment.bytes().all(|byte| byte.is_ascii_alphabetic()) + }) + && previous.is_none_or(|ch| !ch.is_ascii_alphanumeric()) + && next.is_none_or(|ch| !ch.is_ascii_alphanumeric()) + { + spans.push(InputSpan { + start_byte, + end_byte, + }); + } + } + spans +} + +/// Narrows the attached-Roman ampersand cohort to all-capital ASCII segments +/// that begin their whitespace-delimited token. The current token-level bug +/// can pre-emit one capitals-word indicator only in that position; a run +/// attached after Korean or opening punctuation already takes the correct +/// per-segment character path and is an out-of-scope control. +/// UEB 8.4.2 terminates capitals word mode at the nonalphabetic ampersand, and +/// the official UEB 3.1.1/8.4 examples `AT&T` and `B&B` therefore restart +/// capitalization for the following segment. This detector uses only the +/// printed input shape and does not inspect a corpus reference. +fn uppercase_ascii_ampersand_spans(input: &str) -> Vec { + attached_ascii_roman_ampersand_spans(input) + .into_iter() + .filter(|span| { + input[..span.start_byte] + .chars() + .next_back() + .is_none_or(char::is_whitespace) + && input[span.start_byte..span.end_byte] + .split('&') + .all(|segment| segment.bytes().all(|byte| byte.is_ascii_uppercase())) + }) + .collect() +} + +/// Finds a straight ASCII apostrophe joining non-empty ASCII-letter runs. +/// UEB 8.4.2 prints this structure inside `O'Hara`, `DON'T`, and `THAT'S`; +/// detached quotation marks, measurement marks, and Korean quote punctuation +/// are excluded by the immediate-letter requirement. +fn ascii_internal_apostrophe_spans(input: &str) -> Vec { + let indexed = input.char_indices().collect::>(); + let mut spans = BTreeSet::new(); + for index in 1..indexed.len().saturating_sub(1) { + if indexed[index].1 != '\'' + || !indexed[index - 1].1.is_ascii_alphabetic() + || !indexed[index + 1].1.is_ascii_alphabetic() + { + continue; + } + let mut start = index - 1; + while start > 0 && indexed[start - 1].1.is_ascii_alphabetic() { + start -= 1; + } + let mut end = index + 2; + while end < indexed.len() && indexed[end].1.is_ascii_alphabetic() { + end += 1; + } + spans.insert(( + indexed[start].0, + indexed.get(end).map_or(input.len(), |(byte, _)| *byte), + )); + } + spans + .into_iter() + .map(|(start_byte, end_byte)| InputSpan { + start_byte, + end_byte, + }) + .collect() +} + +/// Reproduces the complete input scope of the former token-level capitals-word +/// predicate where it can change output: a whitespace-delimited token starts +/// with ASCII, has no lowercase ASCII letters, and an uppercase letter occurs +/// again after an intervening nonletter. UEB 8.4.2 says that nonletter +/// terminates capitals word mode. A trailing digit or Korean suffix after the +/// final uppercase run is excluded because both old and new paths emit the +/// same capitals indicator before that run. This broad scope audit measures +/// regressions; unlike the ampersand cohort it does not claim that the first +/// difference belongs to one particular symbol. +fn capitals_word_nonletter_change_scope_spans(input: &str) -> Vec { + let mut spans = Vec::new(); + let mut start_byte = 0usize; + for (end_byte, ch) in input + .char_indices() + .chain(std::iter::once((input.len(), ' '))) + { + if ch != ' ' { + continue; + } + if start_byte < end_byte { + let token = &input[start_byte..end_byte]; + let token_chars = token.chars().collect::>(); + let ascii_letters = token_chars + .iter() + .copied() + .filter(|candidate| candidate.is_ascii_alphabetic()) + .collect::>(); + let uppercase_after_nonletter = token_chars.iter().enumerate().any(|(index, ch)| { + ch.is_ascii_uppercase() + && token_chars[..index] + .iter() + .any(|previous| !previous.is_ascii_alphabetic()) + }); + if token + .chars() + .next() + .is_some_and(|first| first.is_ascii_alphabetic()) + && ascii_letters.len() >= 2 + && ascii_letters + .iter() + .all(|letter| letter.is_ascii_uppercase()) + && uppercase_after_nonletter + { + spans.push(InputSpan { + start_byte, + end_byte, + }); + } + } + start_byte = end_byte + 1; + } + spans +} + +/// Finds an ampersand immediately followed by a complete ASCII-letter segment +/// when no ASCII alphanumeric precedes it. This is the one-sided shape of the +/// UEB §3.1.1 `&c` example, kept separate from the already implemented `A&B` +/// cohort and from digit/identifier continuations. +fn ampersand_before_attached_ascii_roman_spans(input: &str) -> Vec { + let bytes = input.as_bytes(); + let mut spans = Vec::new(); + for (ampersand_byte, ch) in input.char_indices() { + if ch != '&' + || input[..ampersand_byte] + .chars() + .next_back() + .is_some_and(|previous| previous.is_ascii_alphanumeric() || previous == '&') + || !bytes + .get(ampersand_byte + 1) + .is_some_and(u8::is_ascii_alphabetic) + { + continue; + } + + let mut end_byte = ampersand_byte + 1; + while bytes.get(end_byte).is_some_and(u8::is_ascii_alphabetic) { + end_byte += 1; + } + if bytes.get(end_byte).is_some_and(u8::is_ascii_alphanumeric) { + continue; + } + spans.push(InputSpan { + start_byte: ampersand_byte, + end_byte, + }); + } + spans +} + +/// Finds a comma immediately after an ASCII digit and followed, after one or +/// more whitespace characters, by another ASCII digit. Korean rule 41 is +/// explicitly limited to a comma *attached* between digits, so this cohort +/// isolates spaced numeric-list punctuation without claiming Roman prose +/// commas or attached digit grouping. +fn spaced_comma_between_ascii_digit_run_spans(input: &str) -> Vec { + input + .match_indices(',') + .filter_map(|(comma_byte, comma)| { + input[..comma_byte] + .chars() + .next_back() + .is_some_and(|ch| ch.is_ascii_digit()) + .then_some(())?; + + let after_comma = comma_byte + comma.len(); + let mut following = input[after_comma..].char_indices(); + let (first_offset, first) = following.next()?; + if first_offset != 0 || !first.is_whitespace() { + return None; + } + + let (digit_offset, digit) = following.find(|(_, ch)| !ch.is_whitespace())?; + if !digit.is_ascii_digit() { + return None; + } + + Some(InputSpan { + start_byte: comma_byte, + end_byte: after_comma + digit_offset + digit.len_utf8(), + }) + }) + .collect() +} + +/// Finds a whitespace-delimited ASCII/Roman tail ending in a comma, followed +/// by a token that begins with a digit and contains Korean script. This is the +/// rule-33 boundary represented by corpus surfaces such as an English title or +/// an ASCII unit before a year/count carrying a Korean suffix. Pure English +/// prose and a following digit-only token are excluded. +fn ascii_roman_comma_before_digit_korean_token_spans(input: &str) -> Vec { + input + .match_indices(',') + .filter_map(|(comma_byte, comma)| { + if !input[..comma_byte] + .chars() + .next_back() + .is_some_and(|ch| ch.is_ascii_alphabetic()) + { + return None; + } + + let after_comma = comma_byte + comma.len(); + let first_nonspace = input[after_comma..] + .char_indices() + .find(|(_, ch)| !ch.is_whitespace())?; + if first_nonspace.0 == 0 || !first_nonspace.1.is_ascii_digit() { + return None; + } + let right_start = after_comma + first_nonspace.0; + let right_end = input[right_start..] + .char_indices() + .find_map(|(offset, ch)| ch.is_whitespace().then_some(right_start + offset)) + .unwrap_or(input.len()); + if !input[right_start..right_end].chars().any(is_korean_script) { + return None; + } + + let left_start = input[..comma_byte] + .char_indices() + .rev() + .find_map(|(byte, ch)| ch.is_whitespace().then_some(byte + ch.len_utf8())) + .unwrap_or(0); + Some(InputSpan { + start_byte: left_start, + end_byte: right_end, + }) + }) + .collect() +} + +/// Finds a comma between two whitespace-separated numeric `%p` unit tokens. +/// Korean rule 69 attachment 2 defines `%p` as the percent-point unit; rule 49 +/// therefore owns the comma separating complete measurements. This remains a +/// diagnostic scope cohort rather than an input-specific engine branch. +fn percent_point_unit_list_comma_spans(input: &str) -> Vec { + input + .match_indices("%p,") + .filter_map(|(unit_byte, matched)| { + if !input[..unit_byte] + .chars() + .next_back() + .is_some_and(|ch| ch.is_ascii_digit()) + { + return None; + } + let comma_byte = unit_byte + matched.len() - 1; + let after_comma = comma_byte + 1; + let (right_offset, right_first) = input[after_comma..] + .char_indices() + .find(|(_, ch)| !ch.is_whitespace())?; + if right_offset == 0 || !right_first.is_ascii_digit() { + return None; + } + let right_start = after_comma + right_offset; + let right_end = input[right_start..] + .char_indices() + .find_map(|(offset, ch)| ch.is_whitespace().then_some(right_start + offset)) + .unwrap_or(input.len()); + let right_token = input[right_start..right_end] + .trim_end_matches(|ch: char| !ch.is_ascii_alphanumeric()); + if !right_token.ends_with("%p") { + return None; + } + let left_start = input[..unit_byte] + .char_indices() + .rev() + .find_map(|(byte, ch)| ch.is_whitespace().then_some(byte + ch.len_utf8())) + .unwrap_or(0); + Some(InputSpan { + start_byte: left_start, + end_byte: right_end, + }) + }) + .collect() +} + +/// Finds maximal pure-uppercase ASCII letter runs whose beginning is itself a +/// pure-letter UEB shortform abbreviation. UEB 5.7.2 and 10.9.7-10.9.8 require +/// grade 1 both for a complete shortform-shaped letters-sequence (`WD`) and for +/// a longer word beginning with one (`PDS`, `LLM`, `GDP`). +fn allcaps_shortform_prefix_spans(input: &str) -> Vec { + let bytes = input.as_bytes(); + let mut spans = Vec::new(); + let mut cursor = 0usize; + while cursor < bytes.len() { + if !bytes[cursor].is_ascii_alphabetic() { + cursor += input[cursor..] + .chars() + .next() + .expect("cursor must remain on a character boundary") + .len_utf8(); + continue; + } + + let start_byte = cursor; + while bytes.get(cursor).is_some_and(u8::is_ascii_alphabetic) { + cursor += 1; + } + let end_byte = cursor; + let run = &input[start_byte..end_byte]; + // The isolated prefix is current-engine evidence only: no corpus + // expected/reference value participates in this candidate gate. For a + // pure all-caps letters-sequence, a leading ⠰ is the engine's existing + // UEB 5.7.2/10.9.7 shortform-collision decision. + let has_shortform_prefix = (2..=run.len()).any(|end| { + braillify::encode_to_unicode(&run[..end]).is_ok_and(|encoded| encoded.starts_with('⠰')) + }); + if run.len() >= 2 + && run.bytes().all(|byte| byte.is_ascii_uppercase()) + && has_shortform_prefix + && input[..start_byte] + .chars() + .next_back() + .is_none_or(|previous| !previous.is_ascii_alphanumeric()) + && input[end_byte..] + .chars() + .next() + .is_none_or(|next| !next.is_ascii_alphanumeric()) + { + spans.push(InputSpan { + start_byte, + end_byte, + }); + } + } + spans +} + +/// Narrow the shortform-collision audit to a letters-sequence followed +/// immediately by an opening round, square, or curly parenthesis. UEB 2.6.2 +/// permits these symbols before a standing-alone sequence, but the exhaustive +/// following-symbol list in 2.6.3 does not permit them after one. Consequently +/// 5.7.2/10.9.7 cannot introduce grade 1 merely because the capital sequence, +/// considered in isolation, resembles a shortform. +fn allcaps_shortform_before_nonstanding_opening_group_spans(input: &str) -> Vec { + allcaps_shortform_prefix_spans(input) + .into_iter() + .filter(|span| { + input[span.end_byte..] + .chars() + .next() + .is_some_and(|ch| matches!(ch, '(' | '[' | '{')) + }) + .collect() +} + +fn nonstanding_shortform_capitals_ranges( + input: &str, + actual: &str, + spans: &[InputSpan], +) -> Vec> { + let mut ranges = BTreeSet::new(); + for span in spans { + let mut letter_cells = String::new(); + let mut valid = true; + for letter in input[span.start_byte..span.end_byte].chars() { + let Ok(encoded) = + braillify::encode_to_unicode(&letter.to_ascii_lowercase().to_string()) + else { + valid = false; + break; + }; + if encoded.chars().count() != 1 { + valid = false; + break; + } + letter_cells.push_str(&encoded); + } + if !valid { + continue; + } + let signature = format!("⠠⠠{letter_cells}"); + for (start_byte, _) in actual.match_indices(&signature) { + let start = actual[..start_byte].chars().count(); + ranges.insert((start, start + signature.chars().count())); + } + } + ranges.into_iter().map(|(start, end)| start..end).collect() +} + +/// UEB 2.6.1-2.6.3 makes a letters-sequence followed immediately by an +/// opening grouping sign not standing alone. This classifier accepts a corpus +/// contradiction only when every difference in the complete sentence is an +/// extra reference-side grade-1 cell immediately before the current capitals +/// indicator of one of those structurally detected sequences. +fn is_ueb_grade1_before_nonstanding_opening_parenthesis_reference_contradiction( + item: &EncodedCase, +) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + let expected = &item.located.case.unicode; + if actual == expected { + return false; + } + + let spans = allcaps_shortform_before_nonstanding_opening_group_spans(&item.located.case.input); + if spans.is_empty() { + return false; + } + + let expected_cells = expected.chars().collect::>(); + let actual_cells = actual.chars().collect::>(); + let mut expected_index = 0usize; + let mut actual_index = 0usize; + let mut removed_at_actual = Vec::new(); + while expected_index < expected_cells.len() && actual_index < actual_cells.len() { + if expected_cells[expected_index] == actual_cells[actual_index] { + expected_index += 1; + actual_index += 1; + continue; + } + if expected_cells[expected_index] == '⠰' && actual_cells[actual_index] == '⠠' { + removed_at_actual.push(actual_index); + expected_index += 1; + continue; + } + return false; + } + if expected_index != expected_cells.len() + || actual_index != actual_cells.len() + || removed_at_actual.is_empty() + { + return false; + } + + let ranges = nonstanding_shortform_capitals_ranges(&item.located.case.input, actual, &spans); + removed_at_actual + .iter() + .all(|position| ranges.iter().any(|range| range.contains(position))) +} + +/// Finds maximal ASCII alphanumeric identifiers containing an immediate +/// digit-to-uppercase transition (`O4O`, `Li2S`, `V2X`). The numeric indicator +/// itself sets grade-1 mode under UEB 5.6.1, and 5.6.2 does not terminate that +/// mode at a capital indicator; the cohort measures whether an extra `⠰` is +/// nevertheless emitted at this exact boundary. +fn roman_uppercase_after_digit_spans(input: &str) -> Vec { + let bytes = input.as_bytes(); + let mut spans = Vec::new(); + let mut cursor = 0usize; + while cursor < bytes.len() { + if !bytes[cursor].is_ascii_alphanumeric() { + cursor += input[cursor..] + .chars() + .next() + .expect("cursor must remain on a character boundary") + .len_utf8(); + continue; + } + let start_byte = cursor; + while bytes.get(cursor).is_some_and(u8::is_ascii_alphanumeric) { + cursor += 1; + } + let end_byte = cursor; + let run = &bytes[start_byte..end_byte]; + if run.iter().any(u8::is_ascii_alphabetic) + && run + .windows(2) + .any(|pair| pair[0].is_ascii_digit() && pair[1].is_ascii_uppercase()) + { + spans.push(InputSpan { + start_byte, + end_byte, + }); + } + } + spans +} + +/// Finds maximal ASCII Roman identifiers containing a hyphen immediately +/// followed by an uppercase run (`U-ENTER`, `CD-ROM`). This is independent of +/// the digit transition above: Korean rule 29 keeps one Roman section around +/// consecutive Roman text, while UEB 5.6.2 gives a hyphen separate significance +/// only when terminating numeric grade-1 mode. +fn roman_uppercase_after_hyphen_spans(input: &str) -> Vec { + let bytes = input.as_bytes(); + let mut spans = Vec::new(); + let mut cursor = 0usize; + while cursor < bytes.len() { + if !bytes[cursor].is_ascii_alphanumeric() { + cursor += input[cursor..] + .chars() + .next() + .expect("cursor must remain on a character boundary") + .len_utf8(); + continue; + } + let start_byte = cursor; + while bytes + .get(cursor) + .is_some_and(|byte| byte.is_ascii_alphanumeric() || *byte == b'-') + { + cursor += 1; + } + let end_byte = cursor; + let run = &bytes[start_byte..end_byte]; + if run.iter().any(u8::is_ascii_alphabetic) + && run + .windows(2) + .any(|pair| pair[0] == b'-' && pair[1].is_ascii_uppercase()) + { + spans.push(InputSpan { + start_byte, + end_byte, + }); + } + } + spans +} + +/// Narrows the broad hyphen continuation diagnostic to the UEB 5.7.2 +/// `CD-ROM` boundary implemented by the engine: the immediately adjacent +/// letter segment before the hyphen is pure uppercase, and the immediately +/// adjacent segment after it is pure uppercase with at least two letters. +fn pure_allcaps_hyphen_multi_allcaps_spans(input: &str) -> Vec { + roman_uppercase_after_hyphen_spans(input) + .into_iter() + .filter(|span| { + let run = &input.as_bytes()[span.start_byte..span.end_byte]; + run.iter().enumerate().any(|(hyphen, byte)| { + if *byte != b'-' { + return false; + } + let prefix_start = run[..hyphen] + .iter() + .rposition(|byte| !byte.is_ascii_alphabetic()) + .map_or(0, |index| index + 1); + let prefix = &run[prefix_start..hyphen]; + let suffix = &run[hyphen + 1..]; + let suffix_len = suffix + .iter() + .take_while(|byte| byte.is_ascii_alphabetic()) + .count(); + let suffix_letters = &suffix[..suffix_len]; + + !prefix.is_empty() + && prefix.iter().all(u8::is_ascii_uppercase) + && suffix_letters.len() >= 2 + && suffix_letters.iter().all(u8::is_ascii_uppercase) + }) + }) + .collect() +} + +/// Mirrors the production grammar for an attached Korean-to-Roman hyphen +/// boundary (`하쿠토-R`, `기장-KBO`) without consulting corpus braille. +/// +/// The encoder selectively expands U+2160-U+217F Roman-numeral presentation +/// characters before token routing, so this audit applies the same expansion +/// (`천궁-Ⅱ` -> `천궁-II`). A capital initial or a multi-letter identifier +/// distinguishes prose labels from a single lowercase algebra variable, while +/// an explicit operator keeps the token in the mathematics cohort. +fn is_korean_to_roman_hyphen_boundary_word(word: &str) -> bool { + let mut normalized = String::with_capacity(word.len()); + for ch in word.chars() { + if (0x2160..=0x217f).contains(&(ch as u32)) { + normalized.extend(std::iter::once(ch).nfkc()); + } else { + normalized.push(ch); + } + } + let chars = normalized.chars().collect::>(); + + chars.windows(3).enumerate().any(|(index, window)| { + if !is_korean_script(window[0]) || window[1] != '-' || !window[2].is_ascii_alphabetic() { + return false; + } + + let roman_tail = &chars[index + 2..]; + let identifier_len = roman_tail + .iter() + .take_while(|ch| ch.is_ascii_alphanumeric()) + .count(); + let letter_count = roman_tail[..identifier_len] + .iter() + .filter(|ch| ch.is_ascii_alphabetic()) + .count(); + let identifier_is_unambiguous = window[2].is_ascii_uppercase() || letter_count >= 2; + let has_explicit_math_operator = roman_tail.iter().any(|ch| { + matches!( + *ch, + '+' | '−' + | '×' + | '÷' + | '=' + | '<' + | '>' + | '≤' + | '≥' + | '≠' + | '≈' + | '^' + | '_' + | '/' + | '*' + | '|' + | '∈' + | '∉' + | '⊂' + | '⊃' + | '∧' + | '∨' + ) + }); + + identifier_is_unambiguous && !has_explicit_math_operator + }) +} + +fn korean_to_roman_hyphen_boundary_spans(input: &str) -> Vec { + let mut spans = Vec::new(); + let mut token_start = None; + for (index, ch) in input.char_indices() { + if ch.is_whitespace() { + if let Some(start_byte) = token_start.take() + && is_korean_to_roman_hyphen_boundary_word(&input[start_byte..index]) + { + spans.push(InputSpan { + start_byte, + end_byte: index, + }); + } + } else if token_start.is_none() { + token_start = Some(index); + } + } + if let Some(start_byte) = token_start + && is_korean_to_roman_hyphen_boundary_word(&input[start_byte..]) + { + spans.push(InputSpan { + start_byte, + end_byte: input.len(), + }); + } + spans +} + +fn preceding_whitespace_word_contains_korean(input: &str, start_byte: usize) -> bool { + let before = &input[..start_byte]; + if !before.chars().next_back().is_some_and(char::is_whitespace) { + return false; + } + before + .trim_end_matches(char::is_whitespace) + .rsplit(char::is_whitespace) + .next() + .is_some_and(|word| word.chars().any(is_korean_script)) +} + +/// Narrows the broad hyphen-continuation trait to a new Roman word after a +/// whitespace-delimited Korean-containing word. This separates entry-mode +/// routing (`A-STAR`) from the already measured grade-1 boundary inside a +/// Roman run (`CD-ROM`). It remains diagnostic because math rule 2 gives the +/// same hyphen-minus a subtraction reading. +fn roman_hyphenated_word_after_korean_word_spans(input: &str) -> Vec { + roman_uppercase_after_hyphen_spans(input) + .into_iter() + .filter(|span| { + input.as_bytes()[span.start_byte].is_ascii_alphabetic() + && preceding_whitespace_word_contains_korean(input, span.start_byte) + }) + .collect() +} + +/// Finds a two-or-more-letter Roman headword immediately after a whitespace- +/// delimited Korean-containing word and immediately before a non-empty closed +/// parenthetical. The parenthetical may contain prose or notation; no semantic +/// choice between Korean rule 29 and math rules 6/12/45 is inferred. +fn roman_parenthetical_headword_after_korean_word_spans(input: &str) -> Vec { + let bytes = input.as_bytes(); + let mut spans = Vec::new(); + for (open, _) in input.match_indices('(') { + let mut start_byte = open; + while start_byte > 0 && bytes[start_byte - 1].is_ascii_alphabetic() { + start_byte -= 1; + } + if open - start_byte < 2 || !preceding_whitespace_word_contains_korean(input, start_byte) { + continue; + } + let body_start = open + 1; + let Some((close_offset, close)) = input[body_start..] + .char_indices() + .find(|(_, ch)| matches!(ch, '(' | ')')) + else { + continue; + }; + if close == ')' && close_offset > 0 { + spans.push(InputSpan { + start_byte, + end_byte: open, + }); + } + } + spans +} + +/// Separates an attached rule-34-shaped Roman parenthetical followed by a +/// hyphenated all-caps suffix from both whitespace Roman entry and ordinary +/// all-caps hyphen continuation. The strict body/suffix gate is structural +/// evidence only; rule 34 and math rules 6/12 still permit competing modes. +fn korean_prefixed_roman_parenthetical_hyphen_suffix_spans(input: &str) -> Vec { + let bytes = input.as_bytes(); + let mut spans = Vec::new(); + for (open, _) in input.match_indices('(') { + if !input[..open] + .chars() + .next_back() + .is_some_and(is_korean_script) + { + continue; + } + let body_start = open + 1; + let Some(close_offset) = input[body_start..].find(')') else { + continue; + }; + let close = body_start + close_offset; + let body = &input[body_start..close]; + if body.len() < 2 || !body.bytes().all(|byte| byte.is_ascii_uppercase()) { + continue; + } + if bytes.get(close + 1) != Some(&b'-') { + continue; + } + let suffix_start = close + 2; + let mut end_byte = suffix_start; + while bytes.get(end_byte).is_some_and(u8::is_ascii_uppercase) { + end_byte += 1; + } + if end_byte - suffix_start < 2 + || input[end_byte..] + .chars() + .next() + .is_some_and(|next| next.is_ascii_alphanumeric()) + { + continue; + } + spans.push(InputSpan { + start_byte: open, + end_byte, + }); + } + spans +} + +/// Finds an uppercase word-start run after whitespace when the preceding +/// whitespace-delimited word ends in an ASCII letter. Korean rule 29 treats +/// consecutive Roman text as one section, but the uppercase token phase can +/// independently request another entry before this second word. Punctuation +/// and non-letter endings deliberately break this structural gate. +fn consecutive_roman_uppercase_word_spans(input: &str) -> Vec { + let bytes = input.as_bytes(); + let mut spans = Vec::new(); + for (start_byte, ch) in input.char_indices() { + if !ch.is_ascii_uppercase() + || !input[..start_byte] + .chars() + .next_back() + .is_some_and(char::is_whitespace) + { + continue; + } + let before = input[..start_byte].trim_end_matches(char::is_whitespace); + if !before + .chars() + .next_back() + .is_some_and(|ch| ch.is_ascii_alphabetic()) + { + continue; + } + let mut end_byte = start_byte; + while bytes.get(end_byte).is_some_and(u8::is_ascii_uppercase) { + end_byte += 1; + } + if end_byte - start_byte >= 2 + && !input[end_byte..] + .chars() + .next() + .is_some_and(|next| next.is_ascii_alphabetic()) + { + spans.push(InputSpan { + start_byte, + end_byte, + }); + } + } + spans +} + +/// Finds adjacent ASCII-letter words separated only by one or more whitespace +/// characters. The gate is deliberately indifferent to capitalization and +/// lexical meaning; rule 29's official `Los Angeles` and `Table of Contents` +/// are exact controls for this structural boundary. +fn consecutive_ascii_roman_word_boundary_spans(input: &str) -> Vec { + let bytes = input.as_bytes(); + let mut spans = Vec::new(); + let mut cursor = 0usize; + while cursor < bytes.len() { + if !bytes[cursor].is_ascii_alphabetic() + || input[..cursor] + .chars() + .next_back() + .is_some_and(|previous| previous.is_ascii_alphanumeric()) + { + cursor += input[cursor..] + .chars() + .next() + .expect("cursor must remain on a character boundary") + .len_utf8(); + continue; + } + let start_byte = cursor; + while bytes.get(cursor).is_some_and(u8::is_ascii_alphabetic) { + cursor += 1; + } + let whitespace_start = cursor; + while input[cursor..] + .chars() + .next() + .is_some_and(char::is_whitespace) + { + cursor += input[cursor..] + .chars() + .next() + .expect("whitespace cursor must remain valid") + .len_utf8(); + } + if cursor == whitespace_start || !bytes.get(cursor).is_some_and(u8::is_ascii_alphabetic) { + cursor = whitespace_start.max(start_byte + 1); + continue; + } + while bytes.get(cursor).is_some_and(u8::is_ascii_alphabetic) { + cursor += 1; + } + spans.push(InputSpan { + start_byte, + end_byte: cursor, + }); + cursor = whitespace_start; + } + spans +} + +/// Locates only the current cell at a consecutive-Roman whitespace boundary. +/// A real input prefix ending after the first word normally ends in rule 29's +/// terminator. The complete output either retains that cell (the suspected +/// premature exit) or replaces its position with the printed whitespace. +fn consecutive_ascii_roman_boundary_actual_ranges( + input: &str, + actual: &str, +) -> Vec> { + let actual_cells = actual.chars().collect::>(); + consecutive_ascii_roman_word_boundary_spans(input) + .into_iter() + .filter_map(|span| { + let boundary_offset = input[span.start_byte..span.end_byte] + .char_indices() + .find_map(|(offset, ch)| ch.is_whitespace().then_some(offset))?; + let boundary_byte = span.start_byte + boundary_offset; + let prefix = braillify::encode_to_unicode(&input[..boundary_byte]).ok()?; + let prefix_cells = prefix.chars().collect::>(); + let boundary = prefix_cells.len().checked_sub(1)?; + if prefix_cells.get(boundary) != Some(&'⠲') { + return None; + } + if actual_cells.get(..prefix_cells.len()) == Some(prefix_cells.as_slice()) { + return Some((boundary, boundary + 1)); + } + if actual_cells.get(..boundary) == Some(&prefix_cells[..boundary]) + && actual_cells.get(boundary) == Some(&'⠀') + { + return Some((boundary, boundary + 1)); + } + None + }) + .collect::>() + .into_iter() + .map(|(start, end)| start..end) + .collect() +} + +fn first_difference_at_consecutive_ascii_roman_word_boundary(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + consecutive_ascii_roman_boundary_actual_ranges(&item.located.case.input, actual) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +/// Finds a closed, non-nested parenthetical whose body starts with a Roman +/// letter and whose opening does not immediately follow another ASCII letter. +/// This includes rule-34 enclosure contexts after Korean, digits, whitespace, +/// or quotes while excluding direct function-call shapes such as `f(x)`. +/// Standalone `(x)` remains an intentional math-rule-6 control, so this is an +/// analyzer cohort rather than an engine routing predicate. +fn roman_parenthetical_after_nonletter_boundary_spans(input: &str) -> Vec { + let mut spans = Vec::new(); + for (open, _) in input.match_indices('(') { + if input[..open] + .chars() + .next_back() + .is_some_and(|previous| previous.is_ascii_alphabetic()) + { + continue; + } + let body_start = open + 1; + if !input[body_start..] + .chars() + .next() + .is_some_and(|first| first.is_ascii_alphabetic()) + { + continue; + } + let Some((close_offset, close)) = input[body_start..] + .char_indices() + .find(|(_, ch)| matches!(ch, '(' | ')')) + else { + continue; + }; + if close == ')' { + spans.push(InputSpan { + start_byte: open, + end_byte: body_start + close_offset + 1, + }); + } + } + spans +} + +/// Locates each detected run in the full current-engine output by searching +/// for that run's independently encoded signature. This uses neither the +/// corpus reference nor a hard-coded braille value. +fn current_engine_signature_ranges( + input: &str, + actual: &str, + spans: &[InputSpan], + leading_boundary_cells: usize, +) -> Vec> { + let mut ranges = BTreeSet::new(); + for candidate in spans { + let run = &input[candidate.start_byte..candidate.end_byte]; + let Ok(signature) = braillify::encode_to_unicode(run) else { + continue; + }; + let signature_cells = signature.chars().count(); + for (start_byte, _) in actual.match_indices(&signature) { + let signature_start = actual[..start_byte].chars().count(); + let start = signature_start.saturating_sub(leading_boundary_cells); + ranges.insert((start, signature_start + signature_cells)); + } + } + ranges.into_iter().map(|(start, end)| start..end).collect() +} + +/// Produces the current mixed-Korean routing signature for a candidate. This +/// differs from encoding the candidate in isolation when the pure-English UEB +/// preflight owns the isolated text but the mixed document routes it as math. +fn korean_context_signature(run: &str) -> Option { + let left = braillify::encode_to_unicode("가").ok()?; + let right = braillify::encode_to_unicode("나").ok()?; + let probe = braillify::encode_to_unicode(&format!("가 {run} 나")).ok()?; + let probe_cells = probe.chars().collect::>(); + let start = left.chars().count(); + let end = probe_cells.len().checked_sub(right.chars().count())?; + let middle = probe_cells.get(start..end)?; + let first_content = middle.iter().position(|cell| *cell != '⠀')?; + let last_content = middle.iter().rposition(|cell| *cell != '⠀')?; + Some(middle[first_content..=last_content].iter().collect()) +} + +fn korean_context_signature_ranges( + input: &str, + actual: &str, + spans: &[InputSpan], + leading_boundary_cells: usize, +) -> Vec> { + let mut ranges = BTreeSet::new(); + for candidate in spans { + let run = &input[candidate.start_byte..candidate.end_byte]; + let Some(signature) = korean_context_signature(run) else { + continue; + }; + let signature_cells = signature.chars().count(); + for (start_byte, _) in actual.match_indices(&signature) { + let signature_start = actual[..start_byte].chars().count(); + ranges.insert(( + signature_start.saturating_sub(leading_boundary_cells), + signature_start + signature_cells, + )); + } + } + ranges.into_iter().map(|(start, end)| start..end).collect() +} + +/// Produces the current signature of a Roman candidate embedded inside one +/// mixed Korean word. This intentionally complements the space-delimited probe +/// above: token-level capitalization can add grade 1 to standalone `WD`, while +/// the residual under review occurs in attached forms such as `한글(WD)`. +fn mixed_korean_word_signature(run: &str) -> Option { + let left = braillify::encode_to_unicode("가").ok()?; + let right = braillify::encode_to_unicode("나").ok()?; + let probe = braillify::encode_to_unicode(&format!("가{run}나")).ok()?; + let probe_cells = probe.chars().collect::>(); + let start = left.chars().count(); + let end = probe_cells.len().checked_sub(right.chars().count())?; + Some(probe_cells.get(start..end)?.iter().collect()) +} + +fn roman_entry_signature_ranges( + input: &str, + actual: &str, + spans: &[InputSpan], + leading_boundary_cells: usize, +) -> Vec> { + let mut ranges = BTreeSet::new(); + for candidate in spans { + let run = &input[candidate.start_byte..candidate.end_byte]; + for signature in [ + korean_context_signature(run), + mixed_korean_word_signature(run), + ] + .into_iter() + .flatten() + { + // The neutral Korean probe appends its own current-engine exit cell. + // A corpus candidate followed by a closing parenthesis can suppress + // that exit under Korean rule 34, so search both the complete probe + // and the same current-engine signature without only that generated + // trailing boundary. Candidate letters and indicators are untouched. + let without_probe_exit = signature + .char_indices() + .next_back() + .map(|(last, _)| signature[..last].to_string()); + for searchable in [Some(signature), without_probe_exit] + .into_iter() + .flatten() + .filter(|candidate| !candidate.is_empty()) + { + let signature_cells = searchable.chars().count(); + for (start_byte, _) in actual.match_indices(&searchable) { + let signature_start = actual[..start_byte].chars().count(); + ranges.insert(( + signature_start.saturating_sub(leading_boundary_cells), + signature_start + signature_cells, + )); + } + } + } + } + ranges.into_iter().map(|(start, end)| start..end).collect() +} + +fn first_difference_in_grade1_cohort_spans(item: &EncodedCase, spans: &[InputSpan]) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + let expected_cell = item.located.case.unicode.chars().nth(first_difference); + let actual_cell = actual.chars().nth(first_difference); + if !matches!( + (expected_cell, actual_cell), + (Some('\u{2830}'), Some('\u{2820}')) | (Some('\u{2820}'), Some('\u{2830}')) + ) { + return false; + } + roman_entry_signature_ranges(&item.located.case.input, actual, spans, 1) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +/// Locates the actual output boundary corresponding to an input span start. +/// The prefix is encoded independently and must be byte-for-byte equal to the +/// full output prefix, so a repeated Roman surface elsewhere cannot satisfy +/// the locator. The short range covers only mode-entry cells, not the run. +fn current_engine_input_entry_ranges( + input: &str, + actual: &str, + spans: &[InputSpan], + entry_cells: usize, +) -> Vec> { + let actual_cells = actual.chars().collect::>(); + let mut ranges = BTreeSet::new(); + for span in spans { + let Ok(prefix) = braillify::encode_to_unicode(&input[..span.start_byte]) else { + continue; + }; + let prefix_cells = prefix.chars().collect::>(); + if actual_cells.starts_with(&prefix_cells) { + let start = prefix_cells.len(); + let end = start.saturating_add(entry_cells).min(actual_cells.len()); + if start < end { + ranges.insert((start, end)); + } + } + } + ranges.into_iter().map(|(start, end)| start..end).collect() +} + +fn first_difference_at_input_span_entry( + item: &EncodedCase, + spans: &[InputSpan], + entry_cells: usize, +) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + current_engine_input_entry_ranges(&item.located.case.input, actual, spans, entry_cells) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +/// Locates the opening of a complete parenthetical through its independently +/// encoded current-engine signature. This complements prefix offsets: a +/// trailing digit can change how an isolated prefix exits Roman/number mode, +/// and whitespace belongs before rather than inside the parenthetical entry. +#[cfg(test)] +fn current_engine_parenthetical_entry_ranges( + input: &str, + actual: &str, + spans: &[InputSpan], + entry_cells: usize, +) -> Vec> { + roman_entry_signature_ranges(input, actual, spans, 0) + .into_iter() + .filter_map(|signature| { + let end = signature + .start + .saturating_add(entry_cells) + .min(signature.end); + (signature.start < end).then_some((signature.start, end)) + }) + .collect::>() + .into_iter() + .map(|(start, end)| start..end) + .collect() +} + +/// Extends the occurrence-specific parenthetical entry range by the two cells +/// immediately before the current opening. Math rule 11 emits a two-blank +/// boundary there, whereas Korean rules 34/54 attach an enclosure to adjacent +/// prose. This is an output localizer only: it does not decide which semantic +/// route owns the parenthetical. +fn current_engine_parenthetical_leading_boundary_ranges( + input: &str, + actual: &str, + spans: &[InputSpan], +) -> Vec> { + let mut ranges = current_engine_input_entry_ranges(input, actual, spans, 5) + .into_iter() + .map(|range| (range.start, range.end)) + .collect::>(); + ranges.extend( + roman_entry_signature_ranges(input, actual, spans, 2) + .into_iter() + .filter_map(|signature| { + let end = signature.start.saturating_add(5).min(signature.end); + (signature.start < end).then_some((signature.start, end)) + }), + ); + ranges.into_iter().map(|(start, end)| start..end).collect() +} + +fn first_difference_at_parenthetical_boundary(item: &EncodedCase, spans: &[InputSpan]) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + current_engine_parenthetical_leading_boundary_ranges(&item.located.case.input, actual, spans) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +fn allcaps_ou_actual_ranges(input: &str, actual: &str) -> Vec> { + current_engine_signature_ranges(input, actual, &allcaps_roman_runs_containing_ou(input), 0) +} + +fn first_difference_in_allcaps_ou_run(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + allcaps_ou_actual_ranges(&item.located.case.input, actual) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +fn allcaps_st_actual_ranges(input: &str, actual: &str) -> Vec> { + current_engine_signature_ranges(input, actual, &allcaps_roman_runs_containing_st(input), 0) +} + +fn first_difference_in_allcaps_st_run(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + allcaps_st_actual_ranges(&item.located.case.input, actual) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +fn allcaps_ar_actual_ranges(input: &str, actual: &str) -> Vec> { + current_engine_signature_ranges(input, actual, &allcaps_roman_runs_containing_ar(input), 0) +} + +fn first_difference_in_allcaps_ar_run(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + allcaps_ar_actual_ranges(&item.located.case.input, actual) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +fn allcaps_ed_actual_ranges(input: &str, actual: &str) -> Vec> { + current_engine_signature_ranges(input, actual, &allcaps_roman_runs_containing_ed(input), 0) +} + +fn first_difference_in_allcaps_ed_run(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + allcaps_ed_actual_ranges(&item.located.case.input, actual) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +fn roman_after_closed_enclosure_actual_ranges( + input: &str, + actual: &str, +) -> Vec> { + current_engine_signature_ranges( + input, + actual, + &roman_run_after_closed_roman_enclosure_spans(input), + 1, + ) +} + +fn first_difference_in_roman_after_closed_enclosure(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + roman_after_closed_enclosure_actual_ranges(&item.located.case.input, actual) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +fn attached_roman_ampersand_boundary_ranges( + input: &str, + actual: &str, +) -> Vec> { + let actual_cells = actual.chars().collect::>(); + let mut ranges = BTreeSet::new(); + for span in attached_ascii_roman_ampersand_spans(input) { + for (offset, _) in input[span.start_byte..span.end_byte].match_indices('&') { + let ampersand_byte = span.start_byte + offset; + let Ok(prefix) = braillify::encode_to_unicode(&input[..ampersand_byte]) else { + continue; + }; + let prefix_cells = prefix.chars().collect::>(); + if actual_cells.starts_with(&prefix_cells) && !prefix_cells.is_empty() { + let boundary = prefix_cells.len() - 1; + ranges.insert((boundary, boundary + 1)); + } + } + } + ranges.into_iter().map(|(start, end)| start..end).collect() +} + +fn first_difference_in_attached_roman_ampersand(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + attached_roman_ampersand_boundary_ranges(&item.located.case.input, actual) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +fn first_difference_in_uppercase_ascii_ampersand(item: &EncodedCase) -> bool { + first_difference_in_korean_context_signature_spans( + item, + &uppercase_ascii_ampersand_spans(&item.located.case.input), + 0, + ) +} + +/// Locates the Rule-71/29 boundary around an ampersand whose right-hand ASCII +/// Roman segment is attached. The real input prefix before each occurrence +/// anchors the current complete entry signature; the independently encoded +/// prefix through `&` additionally retains the pre-fix terminator location. +/// Another ampersand in the sentence therefore cannot satisfy this audit. +fn ampersand_before_ascii_roman_boundary_ranges( + input: &str, + actual: &str, +) -> Vec> { + let actual_cells = actual.chars().collect::>(); + let spans = ampersand_before_attached_ascii_roman_spans(input); + let mut ranges = current_engine_input_entry_ranges(input, actual, &spans, 7) + .into_iter() + .map(|range| (range.start, range.end)) + .collect::>(); + for span in spans { + let ampersand_end = span.start_byte + 1; + let Ok(prefix) = braillify::encode_to_unicode(&input[..ampersand_end]) else { + continue; + }; + let prefix_cells = prefix.chars().collect::>(); + if actual_cells.starts_with(&prefix_cells) && !prefix_cells.is_empty() { + let boundary = prefix_cells.len() - 1; + ranges.insert((boundary, boundary.saturating_add(3).min(actual_cells.len()))); + } + } + ranges.into_iter().map(|(start, end)| start..end).collect() +} + +fn first_difference_after_ampersand_before_ascii_roman(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + ampersand_before_ascii_roman_boundary_ranges(&item.located.case.input, actual) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +/// Locates the emitted comma cell by encoding the real prefix immediately +/// before each occurrence. The prefix must match the complete current output, +/// so another comma elsewhere in the sentence cannot satisfy the audit. +fn spaced_numeric_list_comma_actual_ranges( + input: &str, + actual: &str, +) -> Vec> { + let actual_cells = actual.chars().collect::>(); + spaced_comma_between_ascii_digit_run_spans(input) + .into_iter() + .filter_map(|span| { + let prefix = braillify::encode_to_unicode(&input[..span.start_byte]).ok()?; + let prefix_cells = prefix.chars().collect::>(); + if !actual_cells.starts_with(&prefix_cells) || prefix_cells.len() >= actual_cells.len() + { + return None; + } + Some((prefix_cells.len(), prefix_cells.len() + 1)) + }) + .collect::>() + .into_iter() + .map(|(start, end)| start..end) + .collect() +} + +fn first_difference_at_spaced_numeric_list_comma(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + spaced_numeric_list_comma_actual_ranges(&item.located.case.input, actual) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +fn first_difference_at_ascii_roman_comma_before_digit_korean(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + if !matches!( + ( + item.located.case.unicode.chars().nth(first_difference), + actual.chars().nth(first_difference) + ), + (Some('⠐'), Some('⠂')) | (Some('⠂'), Some('⠐')) + ) { + return false; + } + current_engine_signature_ranges( + &item.located.case.input, + actual, + &ascii_roman_comma_before_digit_korean_token_spans(&item.located.case.input), + 0, + ) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +fn first_difference_at_percent_point_unit_list_comma(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + if !matches!( + ( + item.located.case.unicode.chars().nth(first_difference), + actual.chars().nth(first_difference) + ), + (Some('⠐'), Some('⠂')) | (Some('⠂'), Some('⠐')) + ) { + return false; + } + current_engine_signature_ranges( + &item.located.case.input, + actual, + &percent_point_unit_list_comma_spans(&item.located.case.input), + 0, + ) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +/// Finds a standalone single capital immediately followed by a non-empty, +/// closed ASCII-digit parenthetical, such as `A(14)`. The span is deliberately +/// semantic-neutral: prose labels and mathematical function notation can share +/// this surface form. +fn single_capital_parenthesized_digit_spans(input: &str) -> Vec { + let bytes = input.as_bytes(); + let mut spans = Vec::new(); + for (start_byte, ch) in input.char_indices() { + if !ch.is_ascii_uppercase() + || braillify::corpus_analysis::has_ascii_alphanumeric_before(input, start_byte) + { + continue; + } + let open = start_byte + 1; + if bytes.get(open) != Some(&b'(') { + continue; + } + let mut cursor = open + 1; + let digit_start = cursor; + while bytes.get(cursor).is_some_and(u8::is_ascii_digit) { + cursor += 1; + } + if cursor == digit_start || bytes.get(cursor) != Some(&b')') { + continue; + } + let end_byte = cursor + 1; + if input[end_byte..] + .chars() + .next() + .is_some_and(|next| next.is_ascii_alphanumeric()) + { + continue; + } + spans.push(InputSpan { + start_byte, + end_byte, + }); + } + spans +} + +/// Finds a maximal uppercase ASCII run followed by ASCII hyphen-minus and a +/// non-empty digit run, such as `D-100`, `F-35`, or `AH-64`. Identifier and +/// subtraction readings deliberately remain separate semantic possibilities. +fn uppercase_roman_hyphen_digit_spans(input: &str) -> Vec { + let bytes = input.as_bytes(); + let mut spans = Vec::new(); + let mut cursor = 0; + while cursor < bytes.len() { + if !bytes[cursor].is_ascii_uppercase() { + cursor += input[cursor..] + .chars() + .next() + .expect("cursor must remain on a character boundary") + .len_utf8(); + continue; + } + let start_byte = cursor; + while bytes.get(cursor).is_some_and(u8::is_ascii_uppercase) { + cursor += 1; + } + if input[..start_byte] + .chars() + .next_back() + .is_some_and(|previous| previous.is_ascii_alphanumeric()) + || bytes.get(cursor) != Some(&b'-') + { + continue; + } + cursor += 1; + let digit_start = cursor; + while bytes.get(cursor).is_some_and(u8::is_ascii_digit) { + cursor += 1; + } + if cursor == digit_start + || input[cursor..] + .chars() + .next() + .is_some_and(|next| next.is_ascii_alphanumeric()) + { + continue; + } + spans.push(InputSpan { + start_byte, + end_byte: cursor, + }); + } + spans +} + +/// Finds a maximal uppercase/digit Roman sequence beginning with a capital, +/// optionally joined by ASCII hyphen-minus or full stop. A trailing prose +/// comma/colon/semicolon is included in the observed boundary. Rule 35 covers +/// the Roman-number reading, while math rules 11/12 leave the same surface +/// potentially ambiguous as a variable expression; this detector assigns +/// neither meaning. +fn uppercase_alphanumeric_roman_digit_spans(input: &str) -> Vec { + let bytes = input.as_bytes(); + let mut spans = Vec::new(); + let mut cursor = 0usize; + while cursor < bytes.len() { + if !bytes[cursor].is_ascii_uppercase() + || input[..cursor] + .chars() + .next_back() + .is_some_and(|previous| previous.is_ascii_alphanumeric()) + { + cursor += input[cursor..] + .chars() + .next() + .expect("cursor must remain on a character boundary") + .len_utf8(); + continue; + } + + let start_byte = cursor; + let mut has_digit = false; + while bytes + .get(cursor) + .is_some_and(|byte| byte.is_ascii_uppercase() || byte.is_ascii_digit()) + { + has_digit |= bytes[cursor].is_ascii_digit(); + cursor += 1; + } + while bytes + .get(cursor) + .is_some_and(|byte| matches!(byte, b'-' | b'.')) + && bytes + .get(cursor + 1) + .is_some_and(|byte| byte.is_ascii_uppercase() || byte.is_ascii_digit()) + { + cursor += 1; + while bytes + .get(cursor) + .is_some_and(|byte| byte.is_ascii_uppercase() || byte.is_ascii_digit()) + { + has_digit |= bytes[cursor].is_ascii_digit(); + cursor += 1; + } + } + let sequence_end = cursor; + if bytes + .get(cursor) + .is_some_and(|byte| matches!(byte, b',' | b':' | b';')) + { + cursor += 1; + } + if has_digit + && input[cursor..] + .chars() + .next() + .is_none_or(|next| !next.is_ascii_alphanumeric()) + { + spans.push(InputSpan { + start_byte, + end_byte: cursor, + }); + } else if cursor == start_byte { + cursor += 1; + } else if cursor > sequence_end { + cursor = sequence_end; + } + } + spans +} + +fn first_difference_in_signature_spans( + item: &EncodedCase, + spans: &[InputSpan], + leading_boundary_cells: usize, +) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + current_engine_signature_ranges( + &item.located.case.input, + actual, + spans, + leading_boundary_cells, + ) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +fn first_difference_in_korean_context_signature_spans( + item: &EncodedCase, + spans: &[InputSpan], + leading_boundary_cells: usize, +) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + korean_context_signature_ranges( + &item.located.case.input, + actual, + spans, + leading_boundary_cells, + ) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +fn first_difference_at_ascii_internal_apostrophe(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + let actual_len = actual.chars().count(); + ascii_internal_apostrophe_spans(&item.located.case.input) + .into_iter() + .filter_map(|span| { + let relative = item.located.case.input[span.start_byte..span.end_byte].find('\'')?; + let apostrophe_byte = span.start_byte + relative; + let prefix = + braillify::encode_to_unicode(&item.located.case.input[..apostrophe_byte]).ok()?; + let prefix_len = prefix.chars().count(); + let common = prefix + .chars() + .zip(actual.chars()) + .take_while(|(left, right)| left == right) + .count(); + // Encoding a real prefix in isolation may append only its final + // Roman boundary. Reject anchors that diverge earlier, because an + // unrelated prior mismatch must not be attributed here. + (prefix_len.saturating_sub(common) <= 2) + .then(|| common..std::cmp::min(common + 5, actual_len)) + }) + .any(|range| range.contains(&first_difference)) +} + +/// Only output-localized cohorts may claim a first difference. Broad input-only +/// coexistence traits are intentionally absent: excluding them would hide +/// unrelated causes merely because a sentence also contains Roman text. +fn first_difference_claimed_by_prior_localized_cohort(item: &EncodedCase) -> bool { + first_difference_in_allcaps_ou_run(item) + || first_difference_in_compact_numeric_ascii_suffix(item) + || first_difference_in_decimal_word(item) + || first_difference_in_korean_prefixed_annotation_opening(item) + || first_difference_in_inline_parenthesized_operator(item) + || first_difference_in_attached_plus_parenthesized_korean_gloss(item) + || first_difference_in_tight_triangle(item) + || first_difference_at_roman_middle_dot_boundary(item) + || first_difference_in_uppercase_ascii_ampersand(item) + || first_difference_at_ascii_internal_apostrophe(item) + || first_difference_in_signature_spans( + item, + &single_capital_parenthesized_digit_spans(&item.located.case.input), + 1, + ) + || first_difference_in_signature_spans( + item, + &mixed_roman_korean_before_headword_expansion_spans(&item.located.case.input), + 1, + ) + || first_difference_in_korean_context_signature_spans( + item, + &uppercase_roman_hyphen_digit_spans(&item.located.case.input), + 1, + ) +} + +fn first_difference_claimed_before_roman_entry_residual(item: &EncodedCase) -> bool { + first_difference_claimed_by_prior_localized_cohort(item) + || first_difference_in_grade1_cohort_spans( + item, + &allcaps_shortform_prefix_spans(&item.located.case.input), + ) + || first_difference_in_grade1_cohort_spans( + item, + &roman_uppercase_after_digit_spans(&item.located.case.input), + ) + || first_difference_in_grade1_cohort_spans( + item, + &roman_uppercase_after_hyphen_spans(&item.located.case.input), + ) +} + +fn first_difference_in_current_roman_entry_signature( + item: &EncodedCase, + spans: &[InputSpan], +) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + roman_entry_signature_ranges(&item.located.case.input, actual, spans, 0) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +fn first_difference_claimed_before_consecutive_roman_reentry(item: &EncodedCase) -> bool { + first_difference_claimed_before_roman_entry_residual(item) + || first_difference_at_input_span_entry( + item, + &roman_hyphenated_word_after_korean_word_spans(&item.located.case.input), + 2, + ) + || first_difference_at_input_span_entry( + item, + &roman_parenthetical_headword_after_korean_word_spans(&item.located.case.input), + 2, + ) + || first_difference_at_input_span_entry( + item, + &korean_prefixed_roman_parenthetical_hyphen_suffix_spans(&item.located.case.input), + 2, + ) +} + +fn first_difference_claimed_before_nonletter_parenthetical(item: &EncodedCase) -> bool { + first_difference_claimed_before_consecutive_roman_reentry(item) + || first_difference_in_current_roman_entry_signature( + item, + &consecutive_roman_uppercase_word_spans(&item.located.case.input), + ) +} + +fn first_difference_claimed_before_allcaps_st(item: &EncodedCase) -> bool { + first_difference_claimed_before_nonletter_parenthetical(item) + || first_difference_at_parenthetical_boundary( + item, + &roman_parenthetical_after_nonletter_boundary_spans(&item.located.case.input), + ) +} + +fn first_difference_claimed_before_attached_roman_ampersand(item: &EncodedCase) -> bool { + first_difference_claimed_before_allcaps_st(item) || first_difference_in_allcaps_st_run(item) +} + +fn first_difference_claimed_before_allcaps_ar(item: &EncodedCase) -> bool { + first_difference_claimed_before_attached_roman_ampersand(item) + || first_difference_in_attached_roman_ampersand(item) +} + +fn first_difference_claimed_before_ampersand_right_roman(item: &EncodedCase) -> bool { + first_difference_claimed_before_allcaps_ar(item) + || first_difference_after_ampersand_before_ascii_roman(item) +} + +fn first_difference_claimed_before_roman_after_closed_enclosure(item: &EncodedCase) -> bool { + first_difference_claimed_before_ampersand_right_roman(item) + || first_difference_in_allcaps_ar_run(item) +} + +fn first_difference_claimed_before_allcaps_ed(item: &EncodedCase) -> bool { + first_difference_claimed_before_roman_after_closed_enclosure(item) + || first_difference_in_roman_after_closed_enclosure(item) +} + +fn first_difference_claimed_before_attached_ascii_roman_to_korean(item: &EncodedCase) -> bool { + first_difference_claimed_before_allcaps_ed(item) || first_difference_in_allcaps_ed_run(item) +} + +fn first_difference_claimed_before_uppercase_alphanumeric_roman_digit(item: &EncodedCase) -> bool { + first_difference_claimed_before_attached_ascii_roman_to_korean(item) + || first_difference_at_attached_ascii_roman_to_korean_boundary(item) +} + +fn first_difference_claimed_before_consecutive_ascii_roman_boundary(item: &EncodedCase) -> bool { + first_difference_claimed_before_uppercase_alphanumeric_roman_digit(item) + || first_difference_at_input_span_entry( + item, + &uppercase_alphanumeric_roman_digit_spans(&item.located.case.input), + 2, + ) +} + +fn first_difference_claimed_by_localized_cohort(item: &EncodedCase) -> bool { + first_difference_claimed_before_consecutive_ascii_roman_boundary(item) + || first_difference_at_consecutive_ascii_roman_word_boundary(item) + || first_difference_at_spaced_numeric_list_comma(item) + || first_difference_at_ascii_roman_comma_before_digit_korean(item) + || first_difference_at_percent_point_unit_list_comma(item) +} + +/// Input-only candidate gate for acronym expansions such as +/// `HCA(Home Connectivity Alliance)`. +/// +/// This is deliberately an analyzer diagnostic, not an engine rule. Requiring +/// only ASCII letters and spaces inside the closed parenthesis also excludes +/// visible operators, subscript/superscript notation, and nested parentheses. +fn uppercase_roman_headword_expansion_spans(input: &str) -> Vec { + let bytes = input.as_bytes(); + let mut spans = Vec::new(); + for (open, _) in input.match_indices('(') { + let mut headword_start = open; + while headword_start > 0 && bytes[headword_start - 1].is_ascii_alphabetic() { + headword_start -= 1; + } + let headword = &input[headword_start..open]; + if headword.len() < 2 || !headword.bytes().all(|byte| byte.is_ascii_uppercase()) { + continue; + } + + let parenthetical_tail = &input[open + 1..]; + let Some(close) = parenthetical_tail.find(')') else { + continue; + }; + let contents = &parenthetical_tail[..close]; + if contents.is_empty() + || contents.trim_matches(' ') != contents + || !contents + .bytes() + .all(|byte| byte.is_ascii_alphabetic() || byte == b' ') + { + continue; + } + + if contents.split_ascii_whitespace().count() >= 2 { + spans.push(InputSpan { + start_byte: headword_start, + end_byte: open, + }); + } + } + spans +} + +fn has_uppercase_roman_headword_expansion(input: &str) -> bool { + !uppercase_roman_headword_expansion_spans(input).is_empty() +} + +/// Narrows the HCA-style diagnostic to a position-sensitive mode boundary: +/// a preceding whitespace-delimited word contains Roman letters and ends in +/// Korean, followed by an uppercase headword expansion. This identifies a +/// mixed Roman+Korean particle boundary without naming a particular particle. +fn mixed_roman_korean_before_headword_expansion_spans(input: &str) -> Vec { + uppercase_roman_headword_expansion_spans(input) + .into_iter() + .filter(|span| { + let before = &input[..span.start_byte]; + if !before.chars().next_back().is_some_and(char::is_whitespace) { + return false; + } + let previous_word = before + .trim_end_matches(char::is_whitespace) + .rsplit(char::is_whitespace) + .next() + .unwrap_or(""); + previous_word.chars().any(|ch| ch.is_ascii_alphabetic()) + && previous_word.chars().any(is_korean_script) + && previous_word + .chars() + .next_back() + .is_some_and(is_korean_script) + }) + .collect() +} + +/// Finds a maximal, alphanumeric-delimited ASCII letter run of two or more +/// capitals, excluding the headword of an immediately following parenthetical +/// expansion already covered by `UPPERCASE_ROMAN_HEADWORD_EXPANSION`. +fn has_standalone_uppercase_roman_word(input: &str) -> bool { + let bytes = input.as_bytes(); + let mut cursor = 0; + while cursor < bytes.len() { + if !bytes[cursor].is_ascii_alphabetic() { + cursor += input[cursor..] + .chars() + .next() + .expect("cursor must remain on a character boundary") + .len_utf8(); + continue; + } + + let start = cursor; + while cursor < bytes.len() && bytes[cursor].is_ascii_alphabetic() { + cursor += 1; + } + let run = &input[start..cursor]; + let previous = input[..start].chars().next_back(); + let next = input[cursor..].chars().next(); + let is_alphanumeric_delimited = previous.is_none_or(|ch| !ch.is_ascii_alphanumeric()) + && next.is_none_or(|ch| !ch.is_ascii_alphanumeric()); + if run.len() >= 2 + && run.bytes().all(|byte| byte.is_ascii_uppercase()) + && is_alphanumeric_delimited + && next != Some('(') + { + return true; + } + } + false +} + +fn is_korean_script(ch: char) -> bool { + matches!(ch as u32, 0x3131..=0x318e | 0xac00..=0xd7a3) +} + +/// Cross-cutting shape shared by prose acronyms and scientific formulae: +/// an immediate Korean prefix followed by a closed, two-or-more-letter +/// all-caps ASCII parenthetical such as `책임자(COO)` or `일산화탄소(CO)`. +fn has_korean_prefixed_allcaps_parenthetical(input: &str) -> bool { + for (open, _) in input.match_indices('(') { + if !input[..open] + .chars() + .next_back() + .is_some_and(is_korean_script) + { + continue; + } + let tail = &input[open + 1..]; + let Some(close) = tail.find(')') else { + continue; + }; + let body = &tail[..close]; + if body.len() >= 2 && body.bytes().all(|byte| byte.is_ascii_uppercase()) { + return true; + } + } + false +} + +/// Cross-cutting input-only shape such as `AI·SW`: two maximal ASCII-letter +/// runs of at least two capitals joined directly by U+00B7 MIDDLE DOT. +/// +/// This deliberately does not assign prose, mathematics, or science +/// semantics. The 2024 rules use the same character as Korean punctuation +/// and as a multiplication mark, so the shape remains an analyzer cohort. +fn has_allcaps_roman_middle_dot_runs(input: &str) -> bool { + let bytes = input.as_bytes(); + for (middle_dot, _) in input.match_indices('·') { + let mut left_start = middle_dot; + while left_start > 0 && bytes[left_start - 1].is_ascii_alphabetic() { + left_start -= 1; + } + + let right_start = middle_dot + '·'.len_utf8(); + let mut right_end = right_start; + while right_end < bytes.len() && bytes[right_end].is_ascii_alphabetic() { + right_end += 1; + } + + let left = &input[left_start..middle_dot]; + let right = &input[right_start..right_end]; + let previous = input[..left_start].chars().next_back(); + let next = input[right_end..].chars().next(); + let has_alphanumeric_boundaries = previous.is_none_or(|ch| !ch.is_ascii_alphanumeric()) + && next.is_none_or(|ch| !ch.is_ascii_alphanumeric()); + if left.len() >= 2 + && right.len() >= 2 + && left.bytes().all(|byte| byte.is_ascii_uppercase()) + && right.bytes().all(|byte| byte.is_ascii_uppercase()) + && has_alphanumeric_boundaries + { + return true; + } + } + false +} + +/// Finds an ASCII-letter run followed immediately by U+00B7 and either the +/// first following Korean character or the complete following ASCII-letter +/// run. The span is syntactic only: it does not infer punctuation, product-name, +/// or mathematical semantics from the middle dot. +fn roman_run_before_middle_dot_boundary_spans(input: &str) -> Vec { + let bytes = input.as_bytes(); + let mut spans = Vec::new(); + for (middle_dot, mark) in input.match_indices('·') { + let mut left_start = middle_dot; + while left_start > 0 && bytes[left_start - 1].is_ascii_alphabetic() { + left_start -= 1; + } + if left_start == middle_dot + || input[..left_start] + .chars() + .next_back() + .is_some_and(|ch| ch.is_ascii_alphanumeric()) + { + continue; + } + + let right_start = middle_dot + mark.len(); + let Some(first_right) = input[right_start..].chars().next() else { + continue; + }; + let right_end = if first_right.is_ascii_alphabetic() { + let mut end = right_start; + while end < bytes.len() && bytes[end].is_ascii_alphabetic() { + end += 1; + } + end + } else if is_korean_script(first_right) { + right_start + first_right.len_utf8() + } else { + continue; + }; + spans.push(InputSpan { + start_byte: left_start, + end_byte: right_end, + }); + } + spans +} + +/// Locates only the current terminator immediately before an attached middle +/// dot by encoding the real input prefix ending at that boundary. This keeps +/// hyphen/identifier state such as `K-ICS·...` without consulting expected. +fn roman_middle_dot_boundary_actual_ranges( + input: &str, + actual: &str, +) -> Vec> { + let actual_cells = actual.chars().collect::>(); + roman_run_before_middle_dot_boundary_spans(input) + .into_iter() + .filter_map(|span| { + let middle_dot_offset = input[span.start_byte..span.end_byte].find('·')?; + let middle_dot_byte = span.start_byte + middle_dot_offset; + let prefix = braillify::encode_to_unicode(&input[..middle_dot_byte]).ok()?; + if !actual.starts_with(&prefix) { + return None; + } + let end = prefix.chars().count(); + let start = end.checked_sub(1)?; + (actual_cells.get(start) == Some(&'⠲')).then_some((start, end)) + }) + .collect::>() + .into_iter() + .map(|(start, end)| start..end) + .collect() +} + +fn first_difference_at_roman_middle_dot_boundary(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + roman_middle_dot_boundary_actual_ranges(&item.located.case.input, actual) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +/// Finds a maximal ASCII-letter run followed immediately by a Korean script +/// character. This is deliberately only a script boundary: it does not infer +/// whether the surrounding sentence is Korean- or Roman-dominant. +fn attached_ascii_roman_to_korean_spans(input: &str) -> Vec { + let bytes = input.as_bytes(); + input + .char_indices() + .filter_map(|(korean_byte, korean)| { + if !is_korean_script(korean) || korean_byte == 0 { + return None; + } + let mut roman_start = korean_byte; + while roman_start > 0 && bytes[roman_start - 1].is_ascii_alphabetic() { + roman_start -= 1; + } + (roman_start < korean_byte).then_some(InputSpan { + start_byte: roman_start, + end_byte: korean_byte + korean.len_utf8(), + }) + }) + .collect() +} + +/// Locates only the current mode marker at an attached Roman-to-Korean +/// boundary. Encoding the real prefix establishes the cells before the +/// boundary without consulting expected. The current full output can either +/// retain rule 29's final Roman terminator or replace it with rule 39's +/// two-cell Korean opening marker. +fn attached_ascii_roman_to_korean_actual_ranges( + input: &str, + actual: &str, +) -> Vec> { + let actual_cells = actual.chars().collect::>(); + attached_ascii_roman_to_korean_spans(input) + .into_iter() + .filter_map(|span| { + let korean_byte = input[span.start_byte..span.end_byte] + .char_indices() + .find_map(|(offset, ch)| { + is_korean_script(ch).then_some(span.start_byte + offset) + })?; + let prefix = braillify::encode_to_unicode(&input[..korean_byte]).ok()?; + let prefix_cells = prefix.chars().collect::>(); + let terminator = prefix_cells.len().checked_sub(1)?; + if prefix_cells.get(terminator) != Some(&'⠲') { + return None; + } + if actual_cells.get(..prefix_cells.len()) == Some(prefix_cells.as_slice()) { + return Some((terminator, prefix_cells.len())); + } + if actual_cells.get(..terminator) == Some(&prefix_cells[..terminator]) + && actual_cells.get(terminator..terminator + 2) == Some(&['⠸', '⠷']) + { + return Some((terminator, terminator + 2)); + } + None + }) + .collect::>() + .into_iter() + .map(|(start, end)| start..end) + .collect() +} + +fn first_difference_at_attached_ascii_roman_to_korean_boundary(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + attached_ascii_roman_to_korean_actual_ranges(&item.located.case.input, actual) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +/// Mirrors the document-level word-count gate used by rule 39 closely enough +/// for a corpus scope audit: each whitespace-delimited word contributes its +/// first ASCII letter or Korean script character. +fn input_is_english_majority(input: &str) -> bool { + let mut english_words = 0usize; + let mut korean_words = 0usize; + for word in input.split_whitespace() { + match word + .chars() + .find(|ch| ch.is_ascii_alphabetic() || is_korean_script(*ch)) + { + Some(ch) if ch.is_ascii_alphabetic() => english_words += 1, + Some(_) => korean_words += 1, + None => {} + } + } + english_words >= korean_words.max(1) +} + +/// Scope audit for the engine branch changed after the boundary diagnosis: +/// Korean-majority input, one whitespace-delimited token, Korean script with +/// the nearest script character on both sides being ASCII Roman, excluding +/// the PDF's dot-delimited `www.대통령.kr` structure. +fn korean_majority_roman_sandwich_non_domain_spans(input: &str) -> Vec { + if input_is_english_majority(input) { + return Vec::new(); + } + let mut spans = Vec::new(); + let mut word_start = 0usize; + for word in input.split_inclusive(char::is_whitespace) { + let word_body = word.trim_end_matches(char::is_whitespace); + let chars = word_body.char_indices().collect::>(); + let mut cursor = 0usize; + while cursor < chars.len() { + if !is_korean_script(chars[cursor].1) { + cursor += 1; + continue; + } + let segment_start = cursor; + while cursor < chars.len() && is_korean_script(chars[cursor].1) { + cursor += 1; + } + let segment_end = cursor; + let left_is_roman = chars[..segment_start].iter().rev().find_map(|(_, ch)| { + (ch.is_ascii_alphabetic() || is_korean_script(*ch)) + .then_some(ch.is_ascii_alphabetic()) + }) == Some(true); + let right_is_roman = chars[segment_end..].iter().find_map(|(_, ch)| { + (ch.is_ascii_alphabetic() || is_korean_script(*ch)) + .then_some(ch.is_ascii_alphabetic()) + }) == Some(true); + if !left_is_roman || !right_is_roman { + continue; + } + let korean_start = chars[segment_start].0; + let korean_end = chars + .get(segment_end) + .map_or(word_body.len(), |(byte, _)| *byte); + let dot_delimited = word_body[..korean_start].ends_with('.') + && word_body[korean_end..].starts_with('.'); + if !dot_delimited { + spans.push(InputSpan { + start_byte: word_start + korean_start, + end_byte: word_start + korean_end, + }); + } + } + word_start += word.len(); + } + spans +} + +fn record_attached_ascii_roman_to_korean_marker_outcomes( + stats: &mut PendingRuleReviewClusterStats, + item: &EncodedCase, + actual: &str, + primary_key: &str, + reason_key: &str, + sample_limit: usize, +) { + let ranges = attached_ascii_roman_to_korean_actual_ranges(&item.located.case.input, actual); + let actual_cells = actual.chars().collect::>(); + let outcome = if primary_key == "exact" { + "exact" + } else { + "mismatch" + }; + let mut seen = BTreeSet::new(); + for range in ranges { + let Some(first) = actual_cells.get(range.start) else { + continue; + }; + let marker = match first { + '⠲' => "rule29_terminator", + '⠸' => "rule39_hangul_opening", + _ => continue, + }; + if !seen.insert(marker) { + continue; + } + *stats + .actual_output_signature_outcomes + .entry(format!("{outcome}:{marker}")) + .or_insert(0) += 1; + + let bucket = stats + .samples + .entry(format!("{outcome}_{marker}")) + .or_default(); + if bucket.len() >= sample_limit + || bucket + .iter() + .any(|existing| existing.shard == item.located.shard) + { + continue; + } + let start = range.start.saturating_sub(8); + let expected_excerpt = item + .located + .case + .unicode + .chars() + .skip(start) + .take(24) + .collect(); + let actual_excerpt = actual.chars().skip(start).take(24).collect(); + bucket.push(PendingRuleReviewClusterSample { + shard: item.located.shard.clone(), + index: item.located.index, + input: item.located.case.input.clone(), + expected_excerpt, + actual_excerpt, + first_difference_cell: None, + error: None, + primary_class: primary_key.to_string(), + reason: reason_key.to_string(), + }); + } +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +struct InlineParenthesizedOperator { + open_byte: usize, + operator: char, +} + +/// Finds `한글()한글` spans without assigning a meaning from the +/// corpus reference. The operator set is exactly the arithmetic set named by +/// Hangeul rules 45/46; parentheses remain visible input boundaries. +fn inline_parenthesized_operators(input: &str) -> Vec { + let mut matches = Vec::new(); + for (open_byte, _) in input.match_indices('(') { + if !input[..open_byte] + .chars() + .next_back() + .is_some_and(is_korean_script) + { + continue; + } + + let tail = &input[open_byte + 1..]; + let Some(operator) = tail.chars().next() else { + continue; + }; + if !matches!( + operator, + '+' | '-' | '\u{2212}' | '\u{00d7}' | '\u{00f7}' | '=' + ) { + continue; + } + let after_operator = &tail[operator.len_utf8()..]; + let Some(after_close) = after_operator.strip_prefix(')') else { + continue; + }; + if after_close.chars().next().is_some_and(is_korean_script) { + matches.push(InlineParenthesizedOperator { + open_byte, + operator, + }); + } + } + matches +} + +/// Returns the current engine's actual cell ranges for the detected input +/// structures. This is deliberately derived from encoding a neutral Korean +/// probe rather than from the corpus reference. A range is retained only when +/// the full sentence has the same current-engine signature at the cell offset +/// produced by the prefix ending immediately before `(`. +fn inline_parenthesized_operator_actual_ranges( + input: &str, + actual: &str, +) -> Vec> { + let actual_cells = actual.chars().collect::>(); + let left_probe_cells = braillify::encode_to_unicode("가") + .expect("neutral Korean probe must encode") + .chars() + .count(); + let right_probe_cells = braillify::encode_to_unicode("나") + .expect("neutral Korean probe must encode") + .chars() + .count(); + + inline_parenthesized_operators(input) + .into_iter() + .filter_map(|candidate| { + let prefix = braillify::encode_to_unicode(&input[..candidate.open_byte]).ok()?; + let start = prefix.chars().count(); + let probe = + braillify::encode_to_unicode(&format!("가({})나", candidate.operator)).ok()?; + let probe_cells = probe.chars().collect::>(); + let end = probe_cells.len().checked_sub(right_probe_cells)?; + let signature = probe_cells.get(left_probe_cells..end)?; + let actual_end = start.checked_add(signature.len())?; + (actual_cells.get(start..actual_end) == Some(signature)).then_some(start..actual_end) + }) + .collect() +} + +fn first_difference_in_inline_parenthesized_operator(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + inline_parenthesized_operator_actual_ranges(&item.located.case.input, actual) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +/// Finds `한글+(한글 풀이)` without inferring whether the plus sign is an +/// arithmetic operator or part of a brand name. The strict body grammar keeps +/// this diagnostic separate from Roman/math parentheticals. +fn attached_plus_parenthesized_korean_gloss_spans(input: &str) -> Vec { + let mut spans = Vec::new(); + for (plus_byte, marker) in input.match_indices("+(") { + if !input[..plus_byte] + .chars() + .next_back() + .is_some_and(is_korean_script) + { + continue; + } + + let body_start = plus_byte + marker.len(); + let Some(close_offset) = input[body_start..].find(')') else { + continue; + }; + let close_byte = body_start + close_offset; + let body = &input[body_start..close_byte]; + if !body.is_empty() && body.chars().all(is_korean_script) { + spans.push(InputSpan { + start_byte: plus_byte, + end_byte: close_byte + 1, + }); + } + } + spans +} + +/// Anchors each gloss at the output of the real input prefix immediately +/// before `+`, then verifies a current-engine signature made from the actual +/// input span in neutral Korean context. No corpus reference cells are used. +fn attached_plus_parenthesized_korean_gloss_actual_ranges( + input: &str, + actual: &str, +) -> Vec> { + let actual_cells = actual.chars().collect::>(); + let left_probe_cells = braillify::encode_to_unicode("가") + .expect("neutral Korean probe must encode") + .chars() + .count(); + + attached_plus_parenthesized_korean_gloss_spans(input) + .into_iter() + .filter_map(|span| { + let prefix = braillify::encode_to_unicode(&input[..span.start_byte]).ok()?; + let start = prefix.chars().count(); + let run = &input[span.start_byte..span.end_byte]; + let probe = braillify::encode_to_unicode(&format!("가{run}")).ok()?; + let signature = probe.chars().skip(left_probe_cells).collect::>(); + let end = start.checked_add(signature.len())?; + (actual_cells.get(start..end) == Some(signature.as_slice())).then_some(start..end) + }) + .collect() +} + +fn first_difference_in_attached_plus_parenthesized_korean_gloss(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + attached_plus_parenthesized_korean_gloss_actual_ranges(&item.located.case.input, actual) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +fn tight_triangle_spans(input: &str) -> Vec { + input + .match_indices('△') + .filter_map(|(byte, mark)| { + let next = input[byte + mark.len()..].chars().next()?; + is_korean_script(next).then_some(InputSpan { + start_byte: byte, + end_byte: byte + mark.len() + next.len_utf8(), + }) + }) + .collect() +} + +fn tight_triangle_positions(input: &str) -> Vec { + tight_triangle_spans(input) + .into_iter() + .map(|span| span.start_byte) + .collect() +} + +/// Current-engine ranges for `△한글`, including the first Korean cell after +/// the mark. A missing reference space therefore differs inside this range, +/// while unrelated earlier sentence differences do not count as causal. +fn tight_triangle_actual_ranges(input: &str, actual: &str) -> Vec> { + korean_context_signature_ranges(input, actual, &tight_triangle_spans(input), 0) +} + +fn first_difference_in_tight_triangle(item: &EncodedCase) -> bool { + let Ok(actual) = &item.actual else { + return false; + }; + if actual == &item.located.case.unicode { + return false; + } + let first_difference = first_difference_cell(&item.located.case.unicode, actual); + tight_triangle_actual_ranges(&item.located.case.input, actual) + .into_iter() + .any(|range| range.contains(&first_difference)) +} + +/// Mirrors the removed token-normalization rule's narrow input shape: a +/// whitespace-delimited Korean token ends in `있다` (optionally followed by a +/// full stop) but contains no printed space before that suffix. Membership is +/// descriptive; it does not decide whether orthography may override rule 49's +/// instruction to follow print spacing. +fn has_attached_korean_auxiliary_itda(input: &str) -> bool { + for word in input.split_inclusive(char::is_whitespace) { + let body = word.trim_end_matches(char::is_whitespace); + let suffix = ["있다.", "있다"] + .into_iter() + .find(|suffix| body.ends_with(suffix)); + if let Some(suffix) = suffix { + let suffix_start = body.len() - suffix.len(); + let prefix = &body[..suffix_start]; + if !prefix.is_empty() && prefix.chars().any(is_korean_script) { + return true; + } + } + } + false +} + +macro_rules! enum_key { + ($value:expr) => {{ + serde_json::to_value($value) + .expect("enum serialization must succeed") + .as_str() + .expect("enum must serialize as a string") + .to_string() + }}; +} + +fn excerpt_pair(expected: &str, actual: &str) -> (String, String) { + let expected_chars = expected.chars().collect::>(); + let actual_chars = actual.chars().collect::>(); + let first_diff = first_difference_cell(expected, actual); + let start = first_diff.saturating_sub(8); + let expected_excerpt = expected_chars.iter().skip(start).take(24).collect(); + let actual_excerpt = actual_chars.iter().skip(start).take(24).collect(); + (expected_excerpt, actual_excerpt) +} + +fn first_difference_cell(expected: &str, actual: &str) -> usize { + expected + .chars() + .zip(actual.chars()) + .position(|(left, right)| left != right) + .unwrap_or_else(|| expected.chars().count().min(actual.chars().count())) +} + +fn cell_transition_key(expected: &str, actual: &str, index: usize) -> String { + let label = |text: &str| { + text.chars().nth(index).map_or_else( + || "".to_string(), + |cell| format!("U+{:04X} {cell}", cell as u32), + ) + }; + format!("{} -> {}", label(expected), label(actual)) +} + +fn record_pending_first_difference_transition( + transitions: &mut BTreeMap, + item: &EncodedCase, + primary_key: &str, + reason_key: &str, + sample_limit: usize, +) { + let Ok(actual) = &item.actual else { + return; + }; + let expected = &item.located.case.unicode; + if actual == expected { + return; + } + let first_difference = first_difference_cell(expected, actual); + let key = cell_transition_key(expected, actual, first_difference); + let stats = transitions.entry(key).or_default(); + stats.cases += 1; + if stats.samples.len() >= sample_limit + || stats + .samples + .iter() + .any(|sample| sample.shard == item.located.shard) + { + return; + } + let (expected_excerpt, actual_excerpt) = excerpt_pair(expected, actual); + stats.samples.push(PendingRuleReviewClusterSample { + shard: item.located.shard.clone(), + index: item.located.index, + input: item.located.case.input.clone(), + expected_excerpt, + actual_excerpt, + first_difference_cell: Some(first_difference), + error: None, + primary_class: primary_key.to_string(), + reason: reason_key.to_string(), + }); +} + +fn is_compatibility_unit_decomposition(ch: char, nfkc: &str) -> bool { + matches!( + ch as u32, + 0x3371..=0x337a + | 0x3380..=0x33c6 + | 0x33c8..=0x33cc + | 0x33ce..=0x33d0 + | 0x33d3..=0x33d9 + | 0x33db..=0x33df + | 0x33ff + ) && nfkc.chars().any(|part| part.is_ascii_alphabetic()) + && nfkc.chars().all(|part| { + part.is_ascii_alphabetic() + || matches!(part, '2' | '3' | '/' | '\u{2044}' | '\u{2215}' | 'μ') + }) +} + +fn encoding_error_family(ch: char) -> &'static str { + let nfkc = ch.to_string().nfkc().collect::(); + if is_compatibility_unit_decomposition(ch, &nfkc) { + "compatibility_unit_symbol" + } else if is_roman_numeral_presentation(ch) + && nfkc.chars().all(|part| { + matches!( + part.to_ascii_uppercase(), + 'I' | 'V' | 'X' | 'L' | 'C' | 'D' | 'M' + ) + }) + { + "roman_numeral_presentation" + } else if matches!(ch, '\u{3214}' | '\u{321c}') { + "enclosed_organization_mark" + } else if ch == '\u{2113}' { + "letterlike_unit_symbol" + } else if matches!( + ch, + '\u{02d1}' + | '\u{2025}' + | '\u{2502}' + | '\u{25b2}' + | '\u{25b4}' + | '\u{260f}' + | '\u{2665}' + | '\u{2e31}' + | '\u{302e}' + ) { + "punctuation_or_layout_symbol" + } else { + "other_unsupported_symbol" + } +} + +fn record_structural_cohort_case( + stats: &mut PendingRuleReviewClusterStats, + item: &EncodedCase, + primary_key: &str, + reason_key: &str, + sample_limit: usize, + first_difference_in_output_signature: Option, + include_localized_sample_bucket: bool, +) { + stats.candidates += 1; + let outcome = if primary_key == "exact" { + stats.exact += 1; + "exact" + } else { + stats.mismatch += 1; + if let Some(is_in_signature) = first_difference_in_output_signature { + stats.output_signature_mismatches_evaluated += 1; + stats.first_difference_in_output_signature += usize::from(is_in_signature); + if is_in_signature && let Ok(actual) = &item.actual { + let expected = &item.located.case.unicode; + let first_difference = first_difference_cell(expected, actual); + *stats + .first_difference_in_output_signature_transitions + .entry(cell_transition_key(expected, actual, first_difference)) + .or_insert(0) += 1; + } + } + if reason_key == "conflicting_duplicate_reference" { + stats.conflicting_reference_cases += 1; + } + *stats + .mismatch_primary_classes + .entry(primary_key.to_string()) + .or_insert(0) += 1; + "mismatch" + }; + let expected = &item.located.case.unicode; + let (actual, error) = match &item.actual { + Ok(actual) => (actual.as_str(), None), + Err(error) => ("", Some(error.clone())), + }; + let (expected_excerpt, actual_excerpt) = if actual == expected { + ( + expected.chars().take(24).collect(), + actual.chars().take(24).collect(), + ) + } else { + excerpt_pair(expected, actual) + }; + let first_difference_cell = + (actual != expected).then(|| first_difference_cell(expected, actual)); + let sample = PendingRuleReviewClusterSample { + shard: item.located.shard.clone(), + index: item.located.index, + input: item.located.case.input.clone(), + expected_excerpt, + actual_excerpt, + first_difference_cell, + error, + primary_class: primary_key.to_string(), + reason: reason_key.to_string(), + }; + let mut bucket_names = vec![outcome]; + if include_localized_sample_bucket && first_difference_in_output_signature == Some(true) { + bucket_names.push("localized_mismatch"); + } + for bucket_name in bucket_names { + let bucket = stats.samples.entry(bucket_name.to_string()).or_default(); + if bucket.len() < sample_limit + && !bucket + .iter() + .any(|existing| existing.shard == item.located.shard) + { + bucket.push(sample.clone()); + } + } +} + +fn analyze( + cases: Vec, + encoded: Vec, + sample_limit: usize, +) -> AnalysisReport { + let (duplicate_inputs, conflicting) = conflicting_inputs(&cases); + let mut primary_classes = BTreeMap::new(); + let mut reasons = BTreeMap::new(); + let mut encoding_error_messages = BTreeMap::new(); + let mut encoding_error_families = BTreeMap::new(); + let mut singleton_error_characters = BTreeMap::::new(); + let mut encoding_error_audit = EncodingErrorAudit::default(); + let mut traits = BTreeMap::new(); + let mut shards = BTreeMap::::new(); + let mut samples = BTreeMap::>::new(); + let mut rule_36_transition_audit = Rule36TransitionAudit::default(); + let mut pending_rule_review_clusters = BTreeMap::from([ + ( + RULE69_ASCII_UNIT_BEFORE_TERMINATOR_SKIPPING_SYMBOL.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + COMPACT_NUMERIC_ASCII_SUFFIX.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + DECIMAL_POINT_BETWEEN_DIGITS.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ALLCAPS_ROMAN_RUN_CONTAINING_OU.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ALLCAPS_ROMAN_RUN_CONTAINING_ST.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ALLCAPS_ROMAN_RUN_CONTAINING_AR.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ALLCAPS_ROMAN_RUN_CONTAINING_ED.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ROMAN_RUN_AFTER_CLOSED_ROMAN_ENCLOSURE.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + UPPERCASE_ASCII_SEGMENTS_JOINED_BY_AMPERSAND.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + CAPITALS_WORD_NONLETTER_CHANGE_SCOPE.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ATTACHED_ASCII_ROMAN_SEGMENTS_JOINED_BY_AMPERSAND.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + AMPERSAND_BEFORE_ATTACHED_ASCII_ROMAN_SEGMENT.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ASCII_APOSTROPHE_BETWEEN_ASCII_LETTERS.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + SPACED_COMMA_BETWEEN_ASCII_DIGIT_RUNS.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ASCII_ROMAN_COMMA_BEFORE_DIGIT_KOREAN_TOKEN.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + PERCENT_POINT_UNIT_LIST_COMMA.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ALLCAPS_ROMAN_MIDDLE_DOT_RUNS.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ROMAN_RUN_BEFORE_MIDDLE_DOT_BOUNDARY.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ATTACHED_ASCII_ROMAN_TO_KOREAN_BOUNDARY.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + KOREAN_MAJORITY_ROMAN_SANDWICH_NON_DOMAIN.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + KOREAN_PREFIXED_ALLCAPS_PARENTHETICAL.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + KOREAN_PREFIXED_CLOSED_ROMAN_ANNOTATION.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + KOREAN_INLINE_PARENTHESIZED_OPERATOR.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ATTACHED_PLUS_PARENTHESIZED_KOREAN_GLOSS.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + MIXED_ROMAN_KOREAN_BEFORE_HEADWORD_EXPANSION.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + SINGLE_CAPITAL_PARENTHESIZED_DIGITS.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + STANDALONE_UPPERCASE_ROMAN_WORD.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + TIGHT_TRIANGLE_BEFORE_KOREAN.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ATTACHED_KOREAN_AUXILIARY_ITDA_SPACING.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + UPPERCASE_ROMAN_HEADWORD_EXPANSION.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + UPPERCASE_ROMAN_HYPHEN_DIGITS.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + UPPERCASE_ALPHANUMERIC_ROMAN_DIGIT_SEQUENCE.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ALLCAPS_SHORTFORM_PREFIX_COLLISION.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ROMAN_UPPERCASE_AFTER_DIGIT.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ROMAN_UPPERCASE_AFTER_HYPHEN.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + PURE_ALLCAPS_HYPHEN_MULTI_ALLCAPS.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + KOREAN_TO_ROMAN_HYPHEN_BOUNDARY.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ROMAN_HYPHENATED_WORD_AFTER_KOREAN_WORD.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ROMAN_PARENTHETICAL_HEADWORD_AFTER_KOREAN_WORD.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + KOREAN_PREFIXED_ROMAN_PARENTHETICAL_HYPHEN_SUFFIX.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + CONSECUTIVE_ROMAN_UPPERCASE_WORD_REENTRY.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + CONSECUTIVE_ASCII_ROMAN_WORD_BOUNDARY.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ( + ROMAN_PARENTHETICAL_AFTER_NONLETTER_BOUNDARY.to_string(), + PendingRuleReviewClusterStats::default(), + ), + ]); + let mut pending_first_difference_cell_transitions = BTreeMap::new(); + let mut pending_first_difference_transitions_after_localized_cohorts = BTreeMap::new(); + let mut compact_numeric_ascii_suffixes = BTreeMap::new(); + let mut grade1_shortform_prefix_surfaces = BTreeMap::new(); + let mut grade1_numeric_continuation_surfaces = BTreeMap::new(); + let mut grade1_hyphen_continuation_surfaces = BTreeMap::new(); + let mut exact = 0usize; + + for item in &encoded { + let (primary, reason) = classify(item, &conflicting); + if let Some(transition) = rule_36_observed_transition(item) { + rule_36_transition_audit.presentation_cases += 1; + *rule_36_transition_audit + .observed_transitions + .entry(transition.to_string()) + .or_insert(0) += 1; + + if item.actual.is_err() { + let other_unsupported_characters = item + .singleton_unsupported_characters + .iter() + .copied() + .filter(|ch| !is_roman_numeral_presentation(*ch)) + .map(|ch| format!("U+{:04X} {ch}", ch as u32)) + .collect::>(); + rule_36_transition_audit.remaining_complex_errors += 1; + rule_36_transition_audit + .remaining_complex_error_samples + .push(Rule36ComplexErrorSample { + shard: item.located.shard.clone(), + index: item.located.index, + input: item.located.case.input.clone(), + other_unsupported_characters, + }); + } + } + let primary_key = enum_key!(&primary); + let reason_key = enum_key!(&reason); + *primary_classes.entry(primary_key.clone()).or_insert(0) += 1; + *reasons.entry(reason_key.clone()).or_insert(0) += 1; + + if primary == PrimaryClass::PendingRuleReview { + record_pending_first_difference_transition( + &mut pending_first_difference_cell_transitions, + item, + &primary_key, + &reason_key, + sample_limit, + ); + if !first_difference_claimed_by_localized_cohort(item) { + record_pending_first_difference_transition( + &mut pending_first_difference_transitions_after_localized_cohorts, + item, + &primary_key, + &reason_key, + sample_limit, + ); + } + } + + for (cluster, present, localized_first_difference, localized_samples) in [ + ( + RULE69_ASCII_UNIT_BEFORE_TERMINATOR_SKIPPING_SYMBOL, + !rule69_ascii_unit_before_terminator_skipping_symbol_spans( + &item.located.case.input, + ) + .is_empty(), + Some(first_difference_at_rule69_ascii_unit_terminator_boundary( + item, + )), + true, + ), + ( + COMPACT_NUMERIC_ASCII_SUFFIX, + !compact_numeric_ascii_suffix_spans(&item.located.case.input).is_empty(), + Some(first_difference_in_compact_numeric_ascii_suffix(item)), + true, + ), + ( + DECIMAL_POINT_BETWEEN_DIGITS, + !decimal_word_spans(&item.located.case.input).is_empty(), + Some(first_difference_in_decimal_word(item)), + true, + ), + ( + ALLCAPS_ROMAN_RUN_CONTAINING_OU, + !allcaps_roman_runs_containing_ou(&item.located.case.input).is_empty(), + Some(first_difference_in_allcaps_ou_run(item)), + false, + ), + ( + ALLCAPS_ROMAN_RUN_CONTAINING_ST, + !allcaps_roman_runs_containing_st(&item.located.case.input).is_empty(), + Some( + !first_difference_claimed_before_allcaps_st(item) + && first_difference_in_allcaps_st_run(item), + ), + true, + ), + ( + ALLCAPS_ROMAN_RUN_CONTAINING_AR, + !allcaps_roman_runs_containing_ar(&item.located.case.input).is_empty(), + Some( + !first_difference_claimed_before_allcaps_ar(item) + && first_difference_in_allcaps_ar_run(item), + ), + true, + ), + ( + ROMAN_RUN_AFTER_CLOSED_ROMAN_ENCLOSURE, + !roman_run_after_closed_roman_enclosure_spans(&item.located.case.input).is_empty(), + Some( + !first_difference_claimed_before_roman_after_closed_enclosure(item) + && first_difference_in_roman_after_closed_enclosure(item), + ), + true, + ), + ( + ALLCAPS_ROMAN_RUN_CONTAINING_ED, + !allcaps_roman_runs_containing_ed(&item.located.case.input).is_empty(), + Some( + !first_difference_claimed_before_allcaps_ed(item) + && first_difference_in_allcaps_ed_run(item), + ), + true, + ), + ( + UPPERCASE_ASCII_SEGMENTS_JOINED_BY_AMPERSAND, + !uppercase_ascii_ampersand_spans(&item.located.case.input).is_empty(), + Some(first_difference_in_uppercase_ascii_ampersand(item)), + true, + ), + ( + CAPITALS_WORD_NONLETTER_CHANGE_SCOPE, + !capitals_word_nonletter_change_scope_spans(&item.located.case.input).is_empty(), + None, + false, + ), + ( + ATTACHED_ASCII_ROMAN_SEGMENTS_JOINED_BY_AMPERSAND, + !attached_ascii_roman_ampersand_spans(&item.located.case.input).is_empty(), + Some( + !first_difference_claimed_before_attached_roman_ampersand(item) + && first_difference_in_attached_roman_ampersand(item), + ), + true, + ), + ( + AMPERSAND_BEFORE_ATTACHED_ASCII_ROMAN_SEGMENT, + !ampersand_before_attached_ascii_roman_spans(&item.located.case.input).is_empty(), + Some( + !first_difference_claimed_before_allcaps_ar(item) + && first_difference_after_ampersand_before_ascii_roman(item), + ), + true, + ), + ( + ASCII_APOSTROPHE_BETWEEN_ASCII_LETTERS, + !ascii_internal_apostrophe_spans(&item.located.case.input).is_empty(), + Some(first_difference_at_ascii_internal_apostrophe(item)), + true, + ), + ( + SPACED_COMMA_BETWEEN_ASCII_DIGIT_RUNS, + !spaced_comma_between_ascii_digit_run_spans(&item.located.case.input).is_empty(), + Some(first_difference_at_spaced_numeric_list_comma(item)), + true, + ), + ( + ASCII_ROMAN_COMMA_BEFORE_DIGIT_KOREAN_TOKEN, + !ascii_roman_comma_before_digit_korean_token_spans(&item.located.case.input) + .is_empty(), + Some(first_difference_at_ascii_roman_comma_before_digit_korean( + item, + )), + true, + ), + ( + PERCENT_POINT_UNIT_LIST_COMMA, + !percent_point_unit_list_comma_spans(&item.located.case.input).is_empty(), + Some(first_difference_at_percent_point_unit_list_comma(item)), + true, + ), + ( + ALLCAPS_ROMAN_MIDDLE_DOT_RUNS, + has_allcaps_roman_middle_dot_runs(&item.located.case.input), + None, + false, + ), + ( + ROMAN_RUN_BEFORE_MIDDLE_DOT_BOUNDARY, + !roman_run_before_middle_dot_boundary_spans(&item.located.case.input).is_empty(), + Some(first_difference_at_roman_middle_dot_boundary(item)), + true, + ), + ( + ATTACHED_ASCII_ROMAN_TO_KOREAN_BOUNDARY, + !attached_ascii_roman_to_korean_spans(&item.located.case.input).is_empty(), + Some( + !first_difference_claimed_before_attached_ascii_roman_to_korean(item) + && first_difference_at_attached_ascii_roman_to_korean_boundary(item), + ), + true, + ), + ( + KOREAN_MAJORITY_ROMAN_SANDWICH_NON_DOMAIN, + !korean_majority_roman_sandwich_non_domain_spans(&item.located.case.input) + .is_empty(), + None, + false, + ), + ( + KOREAN_PREFIXED_ALLCAPS_PARENTHETICAL, + has_korean_prefixed_allcaps_parenthetical(&item.located.case.input), + None, + false, + ), + ( + KOREAN_PREFIXED_CLOSED_ROMAN_ANNOTATION, + !korean_prefixed_closed_roman_annotation_spans(&item.located.case.input).is_empty(), + Some(first_difference_in_korean_prefixed_annotation_opening(item)), + true, + ), + ( + KOREAN_INLINE_PARENTHESIZED_OPERATOR, + !inline_parenthesized_operators(&item.located.case.input).is_empty(), + Some(first_difference_in_inline_parenthesized_operator(item)), + false, + ), + ( + ATTACHED_PLUS_PARENTHESIZED_KOREAN_GLOSS, + !attached_plus_parenthesized_korean_gloss_spans(&item.located.case.input) + .is_empty(), + Some(first_difference_in_attached_plus_parenthesized_korean_gloss(item)), + true, + ), + ( + MIXED_ROMAN_KOREAN_BEFORE_HEADWORD_EXPANSION, + !mixed_roman_korean_before_headword_expansion_spans(&item.located.case.input) + .is_empty(), + Some(first_difference_in_signature_spans( + item, + &mixed_roman_korean_before_headword_expansion_spans(&item.located.case.input), + 1, + )), + false, + ), + ( + SINGLE_CAPITAL_PARENTHESIZED_DIGITS, + !single_capital_parenthesized_digit_spans(&item.located.case.input).is_empty(), + Some(first_difference_in_signature_spans( + item, + &single_capital_parenthesized_digit_spans(&item.located.case.input), + 1, + )), + false, + ), + ( + STANDALONE_UPPERCASE_ROMAN_WORD, + has_standalone_uppercase_roman_word(&item.located.case.input), + None, + false, + ), + ( + TIGHT_TRIANGLE_BEFORE_KOREAN, + !tight_triangle_positions(&item.located.case.input).is_empty(), + Some(first_difference_in_tight_triangle(item)), + false, + ), + ( + ATTACHED_KOREAN_AUXILIARY_ITDA_SPACING, + has_attached_korean_auxiliary_itda(&item.located.case.input), + None, + true, + ), + ( + UPPERCASE_ROMAN_HEADWORD_EXPANSION, + has_uppercase_roman_headword_expansion(&item.located.case.input), + None, + false, + ), + ( + UPPERCASE_ROMAN_HYPHEN_DIGITS, + !uppercase_roman_hyphen_digit_spans(&item.located.case.input).is_empty(), + Some(first_difference_in_korean_context_signature_spans( + item, + &uppercase_roman_hyphen_digit_spans(&item.located.case.input), + 1, + )), + false, + ), + ( + UPPERCASE_ALPHANUMERIC_ROMAN_DIGIT_SEQUENCE, + !uppercase_alphanumeric_roman_digit_spans(&item.located.case.input).is_empty(), + Some( + !first_difference_claimed_before_uppercase_alphanumeric_roman_digit(item) + && first_difference_at_input_span_entry( + item, + &uppercase_alphanumeric_roman_digit_spans(&item.located.case.input), + 2, + ), + ), + true, + ), + ( + ALLCAPS_SHORTFORM_PREFIX_COLLISION, + !allcaps_shortform_prefix_spans(&item.located.case.input).is_empty(), + Some( + !first_difference_claimed_by_prior_localized_cohort(item) + && first_difference_in_grade1_cohort_spans( + item, + &allcaps_shortform_prefix_spans(&item.located.case.input), + ), + ), + true, + ), + ( + ROMAN_UPPERCASE_AFTER_DIGIT, + !roman_uppercase_after_digit_spans(&item.located.case.input).is_empty(), + Some( + !first_difference_claimed_by_prior_localized_cohort(item) + && first_difference_in_grade1_cohort_spans( + item, + &roman_uppercase_after_digit_spans(&item.located.case.input), + ), + ), + true, + ), + ( + ROMAN_UPPERCASE_AFTER_HYPHEN, + !roman_uppercase_after_hyphen_spans(&item.located.case.input).is_empty(), + Some( + !first_difference_claimed_by_prior_localized_cohort(item) + && first_difference_in_grade1_cohort_spans( + item, + &roman_uppercase_after_hyphen_spans(&item.located.case.input), + ), + ), + true, + ), + ( + PURE_ALLCAPS_HYPHEN_MULTI_ALLCAPS, + !pure_allcaps_hyphen_multi_allcaps_spans(&item.located.case.input).is_empty(), + Some( + !first_difference_claimed_by_prior_localized_cohort(item) + && first_difference_in_grade1_cohort_spans( + item, + &pure_allcaps_hyphen_multi_allcaps_spans(&item.located.case.input), + ), + ), + true, + ), + ( + KOREAN_TO_ROMAN_HYPHEN_BOUNDARY, + !korean_to_roman_hyphen_boundary_spans(&item.located.case.input).is_empty(), + None, + false, + ), + ( + ROMAN_HYPHENATED_WORD_AFTER_KOREAN_WORD, + !roman_hyphenated_word_after_korean_word_spans(&item.located.case.input).is_empty(), + Some( + !first_difference_claimed_before_roman_entry_residual(item) + && first_difference_at_input_span_entry( + item, + &roman_hyphenated_word_after_korean_word_spans( + &item.located.case.input, + ), + 2, + ), + ), + true, + ), + ( + ROMAN_PARENTHETICAL_HEADWORD_AFTER_KOREAN_WORD, + !roman_parenthetical_headword_after_korean_word_spans(&item.located.case.input) + .is_empty(), + Some( + !first_difference_claimed_before_roman_entry_residual(item) + && first_difference_at_input_span_entry( + item, + &roman_parenthetical_headword_after_korean_word_spans( + &item.located.case.input, + ), + 2, + ), + ), + true, + ), + ( + KOREAN_PREFIXED_ROMAN_PARENTHETICAL_HYPHEN_SUFFIX, + !korean_prefixed_roman_parenthetical_hyphen_suffix_spans(&item.located.case.input) + .is_empty(), + Some( + !first_difference_claimed_before_roman_entry_residual(item) + && first_difference_at_input_span_entry( + item, + &korean_prefixed_roman_parenthetical_hyphen_suffix_spans( + &item.located.case.input, + ), + 2, + ), + ), + true, + ), + ( + CONSECUTIVE_ROMAN_UPPERCASE_WORD_REENTRY, + !consecutive_roman_uppercase_word_spans(&item.located.case.input).is_empty(), + Some( + !first_difference_claimed_before_consecutive_roman_reentry(item) + && first_difference_in_current_roman_entry_signature( + item, + &consecutive_roman_uppercase_word_spans(&item.located.case.input), + ), + ), + true, + ), + ( + CONSECUTIVE_ASCII_ROMAN_WORD_BOUNDARY, + !consecutive_ascii_roman_word_boundary_spans(&item.located.case.input).is_empty(), + Some( + !first_difference_claimed_before_consecutive_ascii_roman_boundary(item) + && first_difference_at_consecutive_ascii_roman_word_boundary(item), + ), + true, + ), + ( + ROMAN_PARENTHETICAL_AFTER_NONLETTER_BOUNDARY, + !roman_parenthetical_after_nonletter_boundary_spans(&item.located.case.input) + .is_empty(), + Some( + !first_difference_claimed_before_nonletter_parenthetical(item) + && first_difference_at_parenthetical_boundary( + item, + &roman_parenthetical_after_nonletter_boundary_spans( + &item.located.case.input, + ), + ), + ), + true, + ), + ] { + if !present { + continue; + } + let stats = pending_rule_review_clusters + .get_mut(cluster) + .expect("registered pending-rule-review cluster must exist"); + record_structural_cohort_case( + stats, + item, + &primary_key, + &reason_key, + sample_limit, + localized_first_difference, + localized_samples, + ); + if cluster == ATTACHED_ASCII_ROMAN_TO_KOREAN_BOUNDARY + && let Ok(actual) = &item.actual + { + record_attached_ascii_roman_to_korean_marker_outcomes( + stats, + item, + actual, + &primary_key, + &reason_key, + sample_limit, + ); + } + } + + let mut suffix_spans = BTreeMap::>::new(); + for span in compact_numeric_ascii_suffix_spans(&item.located.case.input) { + suffix_spans + .entry(compact_numeric_ascii_suffix(span, &item.located.case.input).to_string()) + .or_default() + .push(span); + } + for (suffix, spans) in suffix_spans { + let localized = first_difference_in_compact_numeric_ascii_suffix_spans(item, &spans); + record_structural_cohort_case( + compact_numeric_ascii_suffixes.entry(suffix).or_default(), + item, + &primary_key, + &reason_key, + sample_limit, + Some(localized), + false, + ); + } + + for (target, spans) in [ + ( + &mut grade1_shortform_prefix_surfaces, + allcaps_shortform_prefix_spans(&item.located.case.input), + ), + ( + &mut grade1_numeric_continuation_surfaces, + roman_uppercase_after_digit_spans(&item.located.case.input), + ), + ( + &mut grade1_hyphen_continuation_surfaces, + roman_uppercase_after_hyphen_spans(&item.located.case.input), + ), + ] { + let mut surface_spans = BTreeMap::>::new(); + for span in spans { + surface_spans + .entry(item.located.case.input[span.start_byte..span.end_byte].to_string()) + .or_default() + .push(span); + } + for (surface, matching_spans) in surface_spans { + let localized = !first_difference_claimed_by_prior_localized_cohort(item) + && first_difference_in_grade1_cohort_spans(item, &matching_spans); + record_structural_cohort_case( + target.entry(surface).or_default(), + item, + &primary_key, + &reason_key, + sample_limit, + Some(localized), + true, + ); + } + } + + let shard = shards.entry(item.located.shard.clone()).or_default(); + shard.total += 1; + if primary == PrimaryClass::Exact { + exact += 1; + shard.exact += 1; + continue; + } + + let input = &item.located.case.input; + if let Err(error) = &item.actual { + encoding_error_audit.raw_total += 1; + if primary == PrimaryClass::ComparisonMethod { + encoding_error_audit.resolved_by_comparison_method += 1; + } else if primary == PrimaryClass::CorpusSuspect { + encoding_error_audit.excluded_as_corpus_suspect += 1; + } else { + encoding_error_audit.unresolved_review_total += 1; + if item.singleton_unsupported_characters.is_empty() { + encoding_error_audit.unclassified_without_singleton += 1; + if encoding_error_audit.unclassified_samples.len() < sample_limit { + encoding_error_audit.unclassified_samples.push( + UnclassifiedEncodingErrorSample { + shard: item.located.shard.clone(), + index: item.located.index, + input: item.located.case.input.clone(), + error: error.clone(), + }, + ); + } + } else { + encoding_error_audit.explained_by_singleton_unsupported += 1; + if item.singleton_unsupported_characters.len() > 1 { + encoding_error_audit.multiple_singleton_unsupported += 1; + encoding_error_audit.multiple_singleton_samples.push( + MultipleSingletonErrorSample { + shard: item.located.shard.clone(), + index: item.located.index, + input: item.located.case.input.clone(), + unsupported_characters: item + .singleton_unsupported_characters + .iter() + .map(|ch| format!("U+{:04X} {ch}", *ch as u32)) + .collect(), + }, + ); + } + } + *encoding_error_messages.entry(error.clone()).or_insert(0) += 1; + let mut case_families = BTreeSet::new(); + for ch in &item.singleton_unsupported_characters { + let key = format!("U+{:04X} {ch}", *ch as u32); + let family = encoding_error_family(*ch); + case_families.insert(family); + singleton_error_characters + .entry(key) + .and_modify(|stats| stats.cases += 1) + .or_insert_with(|| ErrorCharacterStats { + cases: 1, + nfkc: ch.to_string().nfkc().collect(), + family, + }); + } + for family in case_families { + *encoding_error_families + .entry(family.to_string()) + .or_insert(0) += 1; + } + } + } + for (name, present) in [ + ( + "contains_ascii_letters", + input.chars().any(|ch| ch.is_ascii_alphabetic()), + ), + ( + "contains_ascii_digits", + input.chars().any(|ch| ch.is_ascii_digit()), + ), + ( + "contains_delimiter_or_quote", + input.chars().any(is_delimiter_or_quote), + ), + ( + "contains_non_ascii_whitespace", + input.chars().any(|ch| ch.is_whitespace() && ch != ' '), + ), + ("input_not_nfc", input.nfc().ne(input.chars())), + ("input_not_nfkc", input.nfkc().ne(input.chars())), + ] { + if present { + *traits.entry(name.to_string()).or_insert(0) += 1; + } + } + + let bucket = samples.entry(reason_key).or_default(); + if bucket.len() < sample_limit { + let (actual_excerpt, error) = match &item.actual { + Ok(actual) => (actual.as_str(), None), + Err(error) => ("", Some(error.clone())), + }; + let (expected_excerpt, actual_excerpt) = + excerpt_pair(&item.located.case.unicode, actual_excerpt); + bucket.push(Sample { + shard: item.located.shard.clone(), + index: item.located.index, + input: input.clone(), + expected_excerpt, + actual_excerpt, + error, + }); + } + } + + let total = encoded.len(); + AnalysisReport { + corpus: "NIKL Korean-Korean Braille Parallel Corpus 2025 v1.0", + total, + exact, + mismatch: total - exact, + exact_percent: exact as f64 / total as f64 * 100.0, + duplicate_inputs, + conflicting_duplicate_inputs: conflicting.len(), + primary_classes, + reasons, + encoding_error_messages, + encoding_error_families, + singleton_error_characters, + encoding_error_audit, + rule_36_transition_audit, + pending_rule_review_clusters, + pending_first_difference_cell_transitions, + pending_first_difference_transitions_after_localized_cohorts, + compact_numeric_ascii_suffixes, + grade1_shortform_prefix_surfaces, + grade1_numeric_continuation_surfaces, + grade1_hyphen_continuation_surfaces, + overlapping_traits: traits, + shards, + samples, + } +} + +fn markdown(report: &AnalysisReport) -> String { + let mut text = String::new(); + text.push_str("# NIKL 2025 v1.0 corpus analysis\n\n"); + text.push_str( + "> Generated by `cargo run --release -p braillify --example nikl_corpus_analyze`. \ + The tool reads only `input` and `unicode`; it never loads or compares the read-only \ + `world` or `jeomsarang` fields.\n\n", + ); + text.push_str("## Current measurement\n\n"); + text.push_str("| Metric | Count |\n|---|---:|\n"); + text.push_str(&format!("| Total | {} |\n", report.total)); + text.push_str(&format!("| Exact | {} |\n", report.exact)); + text.push_str(&format!("| Mismatch | {} |\n", report.mismatch)); + text.push_str(&format!( + "| Exact accuracy | {:.2}% |\n", + report.exact_percent + )); + text.push_str(&format!( + "| Duplicate records | {} |\n", + report.duplicate_inputs + )); + text.push_str(&format!( + "| Inputs with conflicting references | {} |\n\n", + report.conflicting_duplicate_inputs + )); + + text.push_str("## Classification policy\n\n"); + text.push_str( + "Primary classes are evidence gates, not permissions to change the engine. \ + `implementation_defect` is restricted to defects independently confirmed from the PDF \ + (currently the rules 28/29 roman-indicator ordering signature). \ + `unsupported_character_review` contains encoding failures fully explained by one or more \ + singleton characters whose support obligation has not been confirmed from the PDF. \ + `unclassified_encoding_error_review` contains other encoding failures until a PDF-backed \ + implementation obligation or a reproducible comparison/corpus issue is established. \ + `pending_rule_review` contains foreign-text, number, punctuation, and Korean candidates \ + that have not yet been resolved against the PDF. `corpus_suspect` is reserved for \ + independently detectable contradictions: conflicting duplicate references or a localized \ + reference-cell signature that contradicts an explicit PDF/UEB rule. \ + `comparison_method` requires equality after a named normalization.\n\n", + ); + text.push_str("| Primary class | Count |\n|---|---:|\n"); + for (name, count) in &report.primary_classes { + text.push_str(&format!("| `{name}` | {count} |\n")); + } + text.push_str("\n| Reproducible reason | Count |\n|---|---:|\n"); + for (name, count) in &report.reasons { + text.push_str(&format!("| `{name}` | {count} |\n")); + } + + text.push_str("\n## Pending first-difference cell transitions\n\n"); + text.push_str( + "This ranking is a diagnostic selector, not an implementation rule. It counts only \ + current `pending_rule_review` cases whose encoder call succeeded, keyed by the expected \ + and actual cell at the sentence's first differing position. Candidate implementation \ + work must still bind a transition to a localized input structure, exact controls, and \ + independent PDF evidence.\n\n", + ); + let mut ranked_transitions = report + .pending_first_difference_cell_transitions + .iter() + .collect::>(); + ranked_transitions.sort_by(|(left_key, left), (right_key, right)| { + right + .cases + .cmp(&left.cases) + .then_with(|| left_key.cmp(right_key)) + }); + text.push_str("| Rank | Expected → actual first cell | Cases |\n|---:|---|---:|\n"); + for (rank, (transition, stats)) in ranked_transitions.iter().take(20).enumerate() { + text.push_str(&format!( + "| {} | `{transition}` | {} |\n", + rank + 1, + stats.cases + )); + } + for (transition, stats) in ranked_transitions.iter().take(10) { + text.push_str(&format!("\n### `{transition}`\n\n")); + for sample in &stats.samples { + text.push_str(&format!( + "- `{}` #{}: {}\n - expected: `{}`\n - actual: `{}`\n - first differing cell (zero-based): {}\n - current primary/reason: `{}` / `{}`\n", + sample.shard, + sample.index, + sample + .input + .chars() + .take(180) + .collect::() + .trim_end() + .replace('`', "\\`"), + sample.expected_excerpt, + sample.actual_excerpt, + sample.first_difference_cell.unwrap_or(0), + sample.primary_class, + sample.reason + )); + } + } + + text.push_str("\n## Residual first-difference transitions after localized cohorts\n\n"); + text.push_str( + "This ranking removes only cases whose first difference is inside an existing \ + output-localized cohort. Broad input-only traits are not exclusion masks. The residual \ + table therefore prioritizes new causes without hiding a mismatch merely because an \ + unrelated structure coexists elsewhere in its sentence.\n\n", + ); + let mut residual_transitions = report + .pending_first_difference_transitions_after_localized_cohorts + .iter() + .collect::>(); + residual_transitions.sort_by(|(left_key, left), (right_key, right)| { + right + .cases + .cmp(&left.cases) + .then_with(|| left_key.cmp(right_key)) + }); + text.push_str("| Rank | Expected → actual first cell | Residual cases |\n|---:|---|---:|\n"); + for (rank, (transition, stats)) in residual_transitions.iter().take(20).enumerate() { + text.push_str(&format!( + "| {} | `{transition}` | {} |\n", + rank + 1, + stats.cases + )); + } + for (transition, stats) in residual_transitions.iter().take(10) { + text.push_str(&format!("\n### Residual `{transition}`\n\n")); + for sample in &stats.samples { + text.push_str(&format!( + "- `{}` #{}: {}\n - expected: `{}`\n - actual: `{}`\n - first differing cell (zero-based): {}\n - current primary/reason: `{}` / `{}`\n", + sample.shard, + sample.index, + sample + .input + .chars() + .take(180) + .collect::() + .trim_end() + .replace('`', "\\`"), + sample.expected_excerpt, + sample.actual_excerpt, + sample.first_difference_cell.unwrap_or(0), + sample.primary_class, + sample.reason + )); + } + } + + text.push_str("\n## Cross-cutting input-only structural cohorts\n\n"); + text.push_str( + "These are cross-cutting input-only structural cohorts, not new primary classes and not \ + engine routing rules. Candidate selection never changes a case's existing primary \ + class by itself. Only cohort members already classified as `pending_rule_review` form a pending \ + subcluster; exact and other-primary members are controls that retain their existing \ + outcomes. A separate classifier may use independently justified, output-localized PDF \ + evidence, as in the rule-34 three-cell contradiction below. The \ + `uppercase_roman_headword_closed_multiword_parenthetical` gate requires a two-or-more \ + character uppercase ASCII headword immediately followed by a closed parenthesis whose \ + contents are two or more ASCII Roman words separated only by spaces. Because the \ + contents admit only letters and spaces, visible operators, subscript/superscript \ + notation, and nested parentheses are excluded deterministically. The \ + `mixed_roman_korean_word_before_uppercase_headword_expansion` gate further requires a \ + whitespace-delimited preceding word that contains both ASCII Roman and Korean and ends \ + in Korean. It locates the following uppercase headword itself, including its immediately \ + preceding emitted cell, so an earlier Roman entry in the same sentence cannot satisfy \ + the output audit. The `single_capital_followed_by_parenthesized_digits` gate requires a \ + standalone capital, a non-empty closed ASCII-digit parenthetical, and alphanumeric outer \ + boundaries. It likewise includes the emitted entry-boundary cell in localization. The \ + `uppercase_roman_run_followed_by_hyphen_digits` gate requires a maximal uppercase ASCII \ + run, literal ASCII hyphen-minus, a non-empty digit run, and alphanumeric outer \ + boundaries. Its localized range includes the current encoded run and its immediately \ + preceding cell, keeping `F-35`-style routing separate from both parenthesized digits and \ + headword expansions. The `uppercase_alphanumeric_roman_digit_sequence` gate covers the \ + wider rule-35/math collision: a capital-led uppercase/digit sequence with optional \ + internal ASCII hyphen/full stop and trailing prose comma/colon/semicolon. It localizes \ + only the current entry cells and does not decide whether an `A-3`-shaped surface is an \ + identifier or a mathematical expression. The \ + `standalone_multi_character_uppercase_roman_word` gate finds maximal ASCII-letter runs \ + of two or more capitals with non-alphanumeric boundaries; a run immediately followed \ + by `(` is excluded so the HCA-style headword itself is not counted by both gates. The \ + `korean_prefixed_closed_allcaps_parenthetical` gate requires an immediately preceding \ + Korean character and a closed body of two or more uppercase ASCII letters. It \ + intentionally contains both acronym annotations (`책임자(COO)`) and scientific \ + formulae (`일산화탄소(CO)`) so their semantic collision remains measurable. The \ + `korean_prefixed_closed_roman_annotation_rule_34_order` gate accepts the narrower \ + rule-34 body grammar after an immediately preceding Korean character and localizes only \ + the current engine's Korean opening-parenthesis cells. This distinguishes the PDF's \ + parenthesis-before-Roman-indicator order from unrelated differences later in the same \ + sentence. The \ + `multi_character_allcaps_roman_runs_joined_by_middle_dot` gate requires two maximal \ + ASCII-letter runs of at least two capitals joined directly by U+00B7, with \ + non-alphanumeric outer boundaries. It records shapes such as `AI·SW` without assigning \ + prose, mathematics, or science semantics. The \ + `roman_run_immediately_before_attached_middle_dot_boundary` gate is output-localized: \ + it requires a maximal ASCII-letter run immediately before U+00B7 and an attached \ + Korean character or ASCII-letter run after it, then searches for that whole \ + current-engine signature in the actual output. It therefore isolates the Roman \ + terminator boundary without treating unrelated middle dots elsewhere in the sentence \ + as causal. The `attached_ascii_roman_to_korean_script_boundary` gate requires an ASCII \ + letter run immediately followed by Korean script and localizes only the current mode \ + marker there: either rule 29's final `⠲` or rule 39's opening `⠸⠷`. It deliberately \ + does not infer the dominant language of the sentence from that surface boundary. The \ + `korean_majority_same_token_roman_sandwich_non_domain` scope audit mirrors the narrower \ + rule-39 implementation gate: the nearest script characters on both sides of one Korean \ + segment are ASCII Roman, the input is Korean-majority by first-script word counts, and \ + the segment is not dot-delimited like the official domain example. It measures change \ + scope but is not itself an output-localized causal classifier. The \ + `korean_inline_parenthesized_single_arithmetic_operator` gate requires an immediate \ + `Korean(` + one rule-45 arithmetic operator + `)Korean` span. Unlike broad coexistence \ + traits, it also locates the current engine's emitted structure and counts a mismatch as \ + signature-local only when the sentence's first differing cell falls inside that output \ + range. The `attached_plus_followed_by_parenthesized_korean_gloss` gate requires literal \ + `한글+(한글)` with a non-empty all-Korean gloss. It anchors the real prefix immediately \ + before `+` and verifies the current neutral-Korean output signature, distinguishing \ + rule-46 spacing at the sign from unrelated differences elsewhere without deciding \ + whether a name is mathematical. The `allcaps_roman_run_containing_ou` gate finds \ + maximal, alphanumeric-delimited \ + uppercase ASCII runs containing adjacent `OU`. It locates the independently encoded run \ + signature in the complete current output and counts only first differences inside that \ + signature as localized. The `allcaps_roman_run_containing_st` gate applies the same \ + output-position requirement to adjacent `ST`; it is kept separate because UEB 10.12 \ + makes contraction use depend on how an abbreviation or acronym is pronounced. The \ + `uppercase_ascii_segments_joined_by_ampersand_capitalization` gate narrows attached \ + ampersand runs to uppercase-only segments and locates the complete current Korean-context \ + output. The run must begin its whitespace-delimited token, matching the current \ + token-level capitals-word pre-emission scope; Korean-attached and punctuation-prefixed \ + occurrences remain controls. UEB 8.4.2 says a nonalphabetic symbol terminates capitals word mode, while the \ + official `AT&T` and `B&B` examples restart capitalization after `&`. The broader \ + `capitals_word_mode_previously_spanning_nonletter_scope` gate reproduces the former \ + token predicate across ampersands, hyphens, digits, and other nonletters that separate \ + uppercase runs. Trailing nonletters after the final run are excluded because their \ + output is unchanged. It is a \ + broad change-scope/regression audit and is deliberately not an output-cause localizer. The \ + `attached_ascii_roman_segments_joined_by_ampersand` gate requires non-empty ASCII-letter \ + segments joined directly by `&`, excludes spaced/Korean/alphanumeric continuations, and \ + localizes only the current output cell immediately before the ampersand through an \ + independently encoded real-input prefix. The \ + `ampersand_before_attached_ascii_roman_segment` gate covers the distinct one-sided \ + UEB `&c` boundary: `&` is followed by a complete ASCII-letter segment, while an ASCII \ + alphanumeric or another ampersand immediately before it and a digit continuation after \ + it are excluded. Its output range is anchored by independently encoding the real input \ + prefix before each occurrence, then includes only the current Rule-71/29 entry \ + boundary; a second pre-fix anchor through `&` retains the former exit location. The \ + `ascii_apostrophe_between_ascii_letter_runs` gate requires a straight apostrophe with \ + an ASCII letter immediately on both sides and expands only across those two letter \ + runs. It excludes detached quotation marks and numeric measurement marks, then locates \ + the complete current mixed-Korean output signature. UEB 8.4.2 directly supplies \ + `O'Hara`, `DON'T`, and `THAT'S` as controls. The \ + `consecutive_ascii_roman_words_whitespace_boundary` gate requires two adjacent \ + ASCII-letter words separated only by whitespace. For each boundary it independently \ + encodes the real input prefix ending after the first word, then localizes only the \ + current rule-29 terminator at that position or the full output's replacing blank. This \ + prevents an unrelated Roman occurrence elsewhere in the sentence from satisfying the \ + audit. Slash and other punctuation-separated forms are deliberately excluded. The \ + `decimal_point_between_ascii_digits` gate finds \ + whitespace-delimited words containing `digit.digit` and reproduces each whole word in a \ + neutral Korean context, so suffixes and punctuation remain part of the current-engine \ + signature. The `compact_numeric_ascii_letter_suffix` gate finds a numeric prefix \ + immediately followed by ASCII letters, includes the immediately preceding output cell \ + as its entry boundary, and retains suffix-specific outcome counts. It \ + intentionally includes both possible rule-69 units and ambiguous variable/identifier \ + forms; membership alone does not assign unit semantics. The \ + `rule69_ascii_unit_before_terminator_skipping_symbol` gate is narrower: it accepts only \ + ASCII unit spellings already supported by rule 69, includes the immediately following \ + rule-33/34 punctuation cell in the localized signature, and does not infer new units. The \ + `tight_triangle_mark_immediately_before_korean` gate requires literal \ + `△한글` with no input space and includes the first following Korean cell in its localized \ + output range, so an observed missing-space difference is measured at the mark boundary. \ + The `attached_korean_auxiliary_itda_spacing` gate is input-only after the rule-49 \ + correction: it measures Korean tokens ending in attached `있다` without claiming a \ + current output signature or deciding whether orthographic correction may override the \ + printed input.\n\n", + ); + text.push_str( + "| Cluster | Candidates | Exact | Mismatch | Conflicting-reference cases |\n\ + |---|---:|---:|---:|---:|\n", + ); + for (name, stats) in &report.pending_rule_review_clusters { + text.push_str(&format!( + "| `{name}` | {} | {} | {} | {} |\n", + stats.candidates, stats.exact, stats.mismatch, stats.conflicting_reference_cases + )); + } + for (name, stats) in &report.pending_rule_review_clusters { + text.push_str(&format!("\n### `{name}`\n\n")); + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "Of the {} candidates, {pending} are the actual `pending_rule_review` subcluster. \ + The other {} candidates are exact or existing non-pending-primary controls; this \ + membership alone does not reclassify them.\n\n", + stats.candidates, + stats.candidates - pending + )); + if stats.output_signature_mismatches_evaluated > 0 { + text.push_str(&format!( + "For this output-signature audit, {} mismatches were evaluable and {} have their \ + first differing cell inside the detected structure's current-engine output \ + range. The remaining mismatches are controls against causal over-attribution.\n\n", + stats.output_signature_mismatches_evaluated, + stats.first_difference_in_output_signature + )); + if !stats + .first_difference_in_output_signature_transitions + .is_empty() + { + let mut transitions = stats + .first_difference_in_output_signature_transitions + .iter() + .collect::>(); + transitions.sort_by(|(left_key, left), (right_key, right)| { + right.cmp(left).then_with(|| left_key.cmp(right_key)) + }); + text.push_str("Localized first-difference transitions:\n\n"); + for (transition, count) in transitions.into_iter().take(5) { + text.push_str(&format!("- `{transition}`: {count}\n")); + } + text.push('\n'); + } + } + text.push_str("Mismatch primary-class distribution:\n\n"); + for (primary, count) in &stats.mismatch_primary_classes { + text.push_str(&format!("- `{primary}`: {count}\n")); + } + for (outcome, samples) in &stats.samples { + text.push_str(&format!("\nRepresentative `{outcome}` samples:\n\n")); + for sample in samples { + text.push_str(&format!( + "- `{}` #{}: {}\n - expected: `{}`\n - actual: `{}`{}{}\n - current primary/reason: `{}` / `{}`\n", + sample.shard, + sample.index, + sample + .input + .chars() + .take(180) + .collect::() + .trim_end() + .replace('`', "\\`"), + sample.expected_excerpt, + sample.actual_excerpt, + sample + .error + .as_ref() + .map_or_else(String::new, |error| format!("\n - error: `{error}`")), + sample.first_difference_cell.map_or_else(String::new, |cell| { + format!("\n - first differing cell (zero-based): {cell}") + }), + sample.primary_class, + sample.reason + )); + } + } + } + text.push_str("\n## UEB grade-1 first-difference cohorts\n\n"); + text.push_str( + "These cohorts are defined by both an input boundary and the sentence's actual \ + first-difference transition. They therefore do not claim every mismatch merely \ + coexisting with an ASCII run. Candidate, exact, mismatch, and primary-class counts \ + remain cross-cutting controls; only the reported target transition is the localized \ + residual under review. The reverse transition is retained separately rather than \ + folded into the target.\n\n", + ); + text.push_str( + "| Cohort | Candidates | Exact controls | Mismatch | Target localized | Reverse |\n\ + |---|---:|---:|---:|---:|---:|\n", + ); + for (name, target, reverse) in [ + ( + ALLCAPS_SHORTFORM_PREFIX_COLLISION, + "U+2830 ⠰ -> U+2820 ⠠", + "U+2820 ⠠ -> U+2830 ⠰", + ), + ( + ROMAN_UPPERCASE_AFTER_DIGIT, + "U+2820 ⠠ -> U+2830 ⠰", + "U+2830 ⠰ -> U+2820 ⠠", + ), + ( + ROMAN_UPPERCASE_AFTER_HYPHEN, + "U+2820 ⠠ -> U+2830 ⠰", + "U+2830 ⠰ -> U+2820 ⠠", + ), + ( + PURE_ALLCAPS_HYPHEN_MULTI_ALLCAPS, + "U+2820 ⠠ -> U+2830 ⠰", + "U+2830 ⠰ -> U+2820 ⠠", + ), + ] { + let stats = report + .pending_rule_review_clusters + .get(name) + .expect("registered grade-1 cohort must exist"); + let target_count = stats + .first_difference_in_output_signature_transitions + .get(target) + .copied() + .unwrap_or(0); + let reverse_count = stats + .first_difference_in_output_signature_transitions + .get(reverse) + .copied() + .unwrap_or(0); + text.push_str(&format!( + "| `{name}` | {} | {} | {} | {target_count} | {reverse_count} |\n", + stats.candidates, stats.exact, stats.mismatch + )); + } + text.push_str( + "\n### All-caps shortform prefix at an attached Roman entry\n\n\ + UEB 2024 rule 5.7.2 requires grade-1 mode when a letters-sequence could \ + be read as a shortform or as containing one. Rule 10.9.7 covers a \ + standing-alone shortform-shaped sequence, rule 10.9.8 covers a sequence at \ + the beginning of a longer word (the PDF example is `LLC`), and rule 5.8.1 \ + places grade 1 before capitalization. The current standalone ASCII-token \ + route already supplies that guard for complete shortform-shaped controls \ + such as `AC`, `CD`, `IMM`, and `AG`; the attached Korean-word/parenthetical \ + route enters directly at the capital marker and accounts for part of the \ + localized `⠰ -> ⠠` signature. This is a routing distinction supported \ + independently by the PDF, not an expected-output lookup. The implemented \ + boundary is only rule 10.9.7's complete pure-letter shortform. Longer runs \ + such as `GDP`, `LLM`, and the PDF's rule-10.9.8 `LLC` example remain in the \ + broad diagnostic cohort but are not generalized in Korean routing: that \ + broader experiment regressed its exact controls. A shortform appearing \ + later would require the still-distinct grade-1 word rule 10.9.9.\n\n\ + Implementation-boundary experiment (all numbers are full-corpus exact \ + matches, with the committed analyzer-only checkpoint as baseline):\n\n\ + | Boundary | Exact / 83,528 | Change | Decision |\n\ + |---|---:|---:|---|\n\ + | Analyzer-only baseline | 66,546 | — | control |\n\ + | Prefix guard extended through the uppercase token route | 65,264 | -1,282 | rejected |\n\ + | Same-token token route narrowed, rule-28 prefix retained | 65,355 | -1,191 | rejected |\n\ + | Uppercase token route restored, rule-28 prefix retained | 65,474 | -1,072 | rejected |\n\ + | Rule-28 complete shortform only | 66,683 | +137 | retained |\n\n\ + At the retained boundary the broad cohort moves from 2,239 exact / 1,881 \ + mismatch / 962 target-localized / 1 reverse to 2,376 exact / 1,744 \ + mismatch / 778 target-localized / 42 reverse. These figures do not turn \ + the remaining longer-prefix members into an engine rule; they preserve \ + the failed broader trials as evidence that input shape alone is unsafe.\n\n", + ); + text.push_str( + "Same-surface controls demonstrate why primary classes must not be changed \ + by cohort membership:\n\n\ + | Surface | Candidates | Exact | Mismatch | Target-localized |\n\ + |---|---:|---:|---:|---:|\n", + ); + for surface in ["AC", "LLM", "CD", "IMM", "AG", "GDP", "WD"] { + if let Some(stats) = report.grade1_shortform_prefix_surfaces.get(surface) { + let localized = stats + .first_difference_in_output_signature_transitions + .get("U+2830 ⠰ -> U+2820 ⠠") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "| `{surface}` | {} | {} | {} | {localized} |\n", + stats.candidates, stats.exact, stats.mismatch + )); + } + } + for surface in ["AC", "LLM", "CD"] { + let Some(stats) = report.grade1_shortform_prefix_surfaces.get(surface) else { + continue; + }; + let exact = stats + .samples + .get("exact") + .and_then(|samples| samples.first()); + let localized = stats + .samples + .get("localized_mismatch") + .and_then(|samples| samples.first()); + if let (Some(exact), Some(localized)) = (exact, localized) { + text.push_str(&format!( + "\n- `{surface}` exact control: `{}` #{} — {}\n\ + - `{surface}` localized mismatch: `{}` #{} — {}\n", + exact.shard, + exact.index, + exact.input.chars().take(180).collect::(), + localized.shard, + localized.index, + localized.input.chars().take(180).collect::() + )); + } + } + text.push_str( + "\n### Uppercase immediately after a digit\n\n\ + UEB rule 5.6.1 says the numeric indicator establishes grade-1 mode, and \ + rule 5.6.2 ends that mode at a space, hyphen, dash, or grade-1 terminator; \ + a capitalization indicator is not a terminator. Korean rule 35 likewise \ + keeps Roman letters and an adjacent number in one Roman section. The PDF's \ + printed `3b`, `3B`, and `3m` examples distinguish the three following-letter \ + classes: lowercase `a`-`j` retains `⠰` because its cells are numeric, a \ + capital uses its capitalization indicator, and lowercase `k`-`z` needs no \ + extra indicator. `Braille4All`, `M4G`, and `W1N` independently confirm the \ + capital boundary inside longer alphanumeric strings. Before the engine \ + change this cohort contained 330 localized `⠠ -> ⠰` cases. A blanket \ + digit-to-letter removal reached 67,000/83,528 (+317) but was rejected: \ + retaining `⠰` only for lowercase `a`-`j` recovers 10 exact cases and raises \ + the result to 67,010. The wrapper control also exposes a separate routing \ + boundary: a numeric run already preceded by an ASCII letter is part of the \ + Roman identifier, not a fresh rule-69 compact unit. Preserving the rule-69 \ + path for genuinely numeric-leading units while excluding that identifier \ + boundary adds 2 more exact cases, for a final 67,012 (+329). The uppercase \ + cohort moves from 756 exact / 1,140 mismatch / 330 target-localized / 1 \ + reverse to 1,078 exact / 818 mismatch / 0 target-localized / 1 reverse. The remaining \ + non-exact members are not attributed to the removed uppercase transition: \ + their sentence-level first difference may lie in another structure and \ + remains under its existing primary class. This numeric state change remains \ + separate from both the complete-shortform guard and the hyphen continuation \ + boundary below.\n\n\ + ### Uppercase immediately after a hyphen\n\n\ + UEB rule 5.7.2 prints `CD-ROM` with one grade-1 indicator before `CD` and \ + no second grade-1 indicator after the hyphen. Korean rule 29 similarly \ + uses one Roman span for consecutive Roman text. Before the engine change, \ + the broad diagnostic contained 952 candidates / 32 exact / 920 mismatch, \ + with 312 localized `⠠ -> ⠰` and 2 reverse transitions. A blanket \ + uppercase-suffix removal reached 67,222 (+210) but made the broad cohort's \ + single-capital controls such as `Around-U`, `DALL-E`, `ISMS-P`, and `USB-C` \ + non-exact; it was rejected. Requiring only a two-letter uppercase suffix \ + reached 67,162 (+150) but regressed the mixed-prefix exact control `Ko-LLM`; \ + it was also rejected. The retained boundary matches the complete PDF shape: \ + the immediately adjacent prefix is a pure-uppercase letter segment and the \ + immediately adjacent suffix is a pure-uppercase segment of at least two \ + letters. It reaches 67,138 (+126) while preserving all 32 baseline exact \ + controls. The broad diagnostic now contains 952 candidates / 158 exact / \ + 794 mismatch, with 157 localized `⠠ -> ⠰` and 3 reverse transitions. The \ + dedicated `pure_allcaps_segment_before_hyphen_and_multi_allcaps_segment_after` \ + row reports only the implemented subset; broad mixed-case and single-capital \ + members remain controls or pending review. `K-ALM` is the one new reverse \ + surface but was already a mismatch before this change, not an exact \ + regression. Digit-hyphen forms such as `F-35` remain excluded, and the \ + complete-shortform guard still legitimately precedes `CD` in `CD-ROM`.\n\n", + ); + let korean_to_roman_hyphen = report + .pending_rule_review_clusters + .get(KOREAN_TO_ROMAN_HYPHEN_BOUNDARY) + .expect("registered Korean-to-Roman hyphen cohort must exist"); + let exact_with_separator = format!("{},{:03}", report.exact / 1_000, report.exact % 1_000); + text.push_str("### Attached Korean-to-Roman hyphen boundary\n\n"); + text.push_str(&format!( + "Korean rule 29 opens a Roman section for Roman text in a Korean sentence, rule 33 \ + keeps the hyphen as punctuation at the Korean/Roman boundary, and rules 35-36 own \ + adjacent alphanumerics and Roman numerals. The retained production gate therefore \ + routes an immediately attached capital-led or multi-letter Roman identifier as prose \ + (`하쿠토-R`, `기장-KBO`, `온다-life`), but leaves a single lowercase variable and any \ + token with an explicit mathematical operator on the mathematics route (`값-x`, \ + `값-X+1`). The analyzer applies the encoder's selective U+2160-U+217F compatibility \ + expansion before testing the word grammar, so `천궁-Ⅱ` is audited as the equivalent \ + `천궁-II` boundary.\n\n\ + Before this gate, the deterministic cohort contained 105 candidates / 62 exact / 43 \ + mismatch. It now contains {} candidates / {} exact / {} mismatch. The complete \ + corpus exact-ID audit moved from 75,704 to {} (+22), and every new exact ID belongs \ + to this cohort; no formerly exact ID was lost. Cohort membership is input-only and \ + never changes a primary class, so the remaining non-exact members retain their \ + independent review causes.\n\n", + korean_to_roman_hyphen.candidates, + korean_to_roman_hyphen.exact, + korean_to_roman_hyphen.mismatch, + exact_with_separator + )); + text.push_str("\n## Roman-entry residual cohorts after grade-1 localization\n\n"); + text.push_str( + "These three cohorts split the former dominant `⠴ -> blank` residual by the input \ + structure at the actual first-difference location. Their entry boundary is anchored by \ + independently encoding the input prefix and requiring it to equal the full current-engine \ + output prefix; repeated Roman text elsewhere cannot satisfy the locator. A localized count \ + is recorded only when no earlier output-localized cohort already claims that first \ + difference. Candidate membership remains cross-cutting and never changes a primary class.\n\n\ + Korean rule 29 requires a Roman indicator before Roman text in a Korean sentence. Rules \ + 33-35 define the relevant hyphen, enclosure, and number boundaries. Independently, math \ + rules 2, 6, 11, 12, and 45 permit subtraction, parentheses, Roman variables, and function \ + notation with overlapping ASCII surface forms. The surface gates below therefore cannot \ + by themselves exclude a mathematical reading.\n\n", + ); + text.push_str( + "| Cohort | Candidates | Exact controls | Mismatch | Pending | Corpus suspect | \ + Localized `⠴ -> blank` | Reverse `blank -> ⠴` |\n\ + |---|---:|---:|---:|---:|---:|---:|---:|\n", + ); + for name in [ + ROMAN_HYPHENATED_WORD_AFTER_KOREAN_WORD, + ROMAN_PARENTHETICAL_HEADWORD_AFTER_KOREAN_WORD, + KOREAN_PREFIXED_ROMAN_PARENTHETICAL_HYPHEN_SUFFIX, + ] { + let stats = report + .pending_rule_review_clusters + .get(name) + .expect("registered Roman-entry cohort must exist"); + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + let corpus_suspect = stats + .mismatch_primary_classes + .get("corpus_suspect") + .copied() + .unwrap_or(0); + let target = stats + .first_difference_in_output_signature_transitions + .get("U+2834 ⠴ -> U+2800 ⠀") + .copied() + .unwrap_or(0); + let reverse = stats + .first_difference_in_output_signature_transitions + .get("U+2800 ⠀ -> U+2834 ⠴") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "| `{name}` | {} | {} | {} | {pending} | {corpus_suspect} | {target} | \ + {reverse} |\n", + stats.candidates, stats.exact, stats.mismatch + )); + } + let parenthetical = report + .pending_rule_review_clusters + .get(ROMAN_PARENTHETICAL_HEADWORD_AFTER_KOREAN_WORD) + .expect("registered parenthetical-headword cohort must exist"); + let capital_to_roman = parenthetical + .first_difference_in_output_signature_transitions + .get("U+2820 ⠠ -> U+2834 ⠴") + .copied() + .unwrap_or(0); + let roman_to_grade1 = parenthetical + .first_difference_in_output_signature_transitions + .get("U+2834 ⠴ -> U+2830 ⠰") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nThe whitespace parenthetical-headword cohort also retains {capital_to_roman} localized \ + `⠠ -> ⠴` and {roman_to_grade1} localized `⠴ -> ⠰` cases as separate transitions; they \ + are not folded into the target. The exact controls demonstrate that the broad structure \ + is already correct in many sentences, while the attached parenthetical-hyphen cohort \ + has no exact control and includes cases already classified by the stricter rule-34 \ + reference-order contradiction. Consequently none of these measurements authorizes an \ + engine change; they are deterministic pending/corpus-review diagnostics only.\n\n" + )); + if let Some(stats) = report + .pending_rule_review_clusters + .get(CONSECUTIVE_ROMAN_UPPERCASE_WORD_REENTRY) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + let target = stats + .first_difference_in_output_signature_transitions + .get("U+2820 ⠠ -> U+2834 ⠴") + .copied() + .unwrap_or(0); + let grade1_to_roman = stats + .first_difference_in_output_signature_transitions + .get("U+2830 ⠰ -> U+2834 ⠴") + .copied() + .unwrap_or(0); + let reverse = stats + .first_difference_in_output_signature_transitions + .get("U+2834 ⠴ -> U+2820 ⠠") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "### Consecutive Roman uppercase-word re-entry\n\n\ + Korean rule 29 explicitly says that when two or more Roman items occur \ + consecutively, the Roman indicator is placed only before the first and the \ + terminator only after the last. Its printed `Los Angeles` and `Table of Contents` \ + examples exercise multiword Roman sections; the rule-28 appendix independently \ + supplies capitalization indicators inside that section. The current token phase can \ + nevertheless insert an explicit Roman-entry event before a later uppercase word when \ + the preceding Roman run began inside a mixed Korean/punctuation word. The character \ + emitter is still in Roman mode at that point, so this is a candidate duplicate-event \ + boundary rather than permission to rewrite arbitrary multiword ASCII text.\n\n\ + Analyzer checkpoint `266e70c` fixed the pre-change baseline at 1,729 candidates, 386 \ + exact controls, 1,343 mismatches, 1,272 pending members, and 385/1,343 evaluable \ + mismatches localized to the current re-entry signature. Those localized transitions \ + were 365 `⠠ -> ⠴`, 16 `⠰ -> ⠴`, no reverse `⠴ -> ⠠`, and four other transitions. \ + Exact controls include contexts where the first Roman word already opened token-level \ + mode, while parenthesized/mixed-token examples expose the duplicate event.\n\n\ + The generalized fix makes an explicit `EnterEnglish` event idempotent when final emit \ + state is already inside a Roman section; it neither names an input nor changes a new \ + section's entry. The current measurement is {} candidates, {} exact controls, {} \ + mismatches, {pending} pending members, and {}/{} localized mismatches. Current target \ + counts are {target} `⠠ -> ⠴`, {grade1_to_roman} `⠰ -> ⠴`, and {reverse} reverse \ + `⠴ -> ⠠`. Cohort exact controls increase by 275, while corpus-wide exact matches rise \ + from 67,138 to 67,442 (+304); corrected prefixes that still differ later remain \ + mismatches, and applications outside this strict input gate account for the remaining \ + net gain. The raw/residual reverse `⠴ -> ⠠` totals remain 68/65 before and after the \ + change. The complete custom standard summary is 5,141 total, 5,141 success, 0 failure, \ + and 0 skipped.\n\n", + stats.candidates, + stats.exact, + stats.mismatch, + stats.first_difference_in_output_signature, + stats.output_signature_mismatches_evaluated + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(ROMAN_PARENTHETICAL_AFTER_NONLETTER_BOUNDARY) + { + let primary_count = |name: &str| { + stats + .mismatch_primary_classes + .get(name) + .copied() + .unwrap_or(0) + }; + let localized_transition = |name: &str| { + stats + .first_difference_in_output_signature_transitions + .get(name) + .copied() + .unwrap_or(0) + }; + let residual_transition = |name: &str| { + report + .pending_first_difference_transitions_after_localized_cohorts + .get(name) + .map(|transition| transition.cases) + .unwrap_or(0) + }; + let raw_transition = |name: &str| { + report + .pending_first_difference_cell_transitions + .get(name) + .map(|transition| transition.cases) + .unwrap_or(0) + }; + let target = "U+2826 ⠦ -> U+2834 ⠴"; + let reverse = "U+2834 ⠴ -> U+2826 ⠦"; + let open_to_space = "U+2826 ⠦ -> U+2800 ⠀"; + let space_to_open = "U+2800 ⠀ -> U+2826 ⠦"; + text.push_str(&format!( + "### Closed Roman parenthetical after a non-ASCII-letter boundary\n\n\ + Korean rule 34 (2024 Korean-rules PDF p.29) prints the opening enclosure before \ + Roman entry in `링컨(Lincoln)`: the opening sequence is `⠦⠄⠴`. This output-position \ + cohort finds a closed, non-nested parenthetical whose body begins with an ASCII \ + letter and whose opening does not immediately follow another ASCII letter, then \ + locates its complete current-engine signature without consulting the reference. \ + Its localized boundary includes the two current output cells immediately before the \ + opening plus the first three entry cells, so rule-11 math spacing can be separated \ + from a difference later inside the parenthetical. Direct function-call shapes such \ + as `f(x)` are excluded. The PDF's math rule 6 \ + (p.59), rule 12 (pp.63-64), and rule 45 (p.75) nevertheless leave standalone `(x)` \ + and other Roman-letter parenthetical mathematics as counterexamples, so the \ + surface gate is not an engine-routing predicate.\n\n\ + The cross-cutting input cohort contains {} candidates: {} exact controls and {} \ + mismatches. Mismatch primary classes remain unchanged: {} `pending_rule_review`, \ + {} `corpus_suspect`, {} `comparison_method`, and {} \ + `unsupported_character_review`. Of {} evaluable mismatches, {} have the first \ + difference at the detected leading-spacing/entry boundary; these include {} \ + `{target}`, {} `{reverse}`, \ + and {} `{open_to_space}` transitions; the exact localized reverse `{space_to_open}` \ + occurs {} times. After all earlier localized cohorts and this cohort are excluded, \ + the raw-to-residual target count is {} -> {}, the Roman-indicator reverse count is \ + {} -> {}, and the spacing target/reverse counts are {} -> {} and {} -> {}. The \ + short full-encoder form `웹3(Web3)` emits the PDF opening \ + order as a routing control, while whitespace and quote boundaries reproduce the current Roman-first \ + opening. Representative localized samples, with shard and index, are retained in \ + the generated cluster sample table. Because exact controls are abundant and the PDF \ + does not make this input shape semantically sufficient to exclude mathematics, no \ + engine change or primary reclassification is inferred. The one raw/residual spacing \ + reverse is a Korean-body `△(교육)` layout boundary, not a Roman-parenthetical member, \ + and remains a separate rule-72/layout review.\n\n", + stats.candidates, + stats.exact, + stats.mismatch, + primary_count("pending_rule_review"), + primary_count("corpus_suspect"), + primary_count("comparison_method"), + primary_count("unsupported_character_review"), + stats.output_signature_mismatches_evaluated, + stats.first_difference_in_output_signature, + localized_transition(target), + localized_transition(reverse), + localized_transition(open_to_space), + localized_transition(space_to_open), + raw_transition(target), + residual_transition(target), + raw_transition(reverse), + residual_transition(reverse), + raw_transition(open_to_space), + residual_transition(open_to_space), + raw_transition(space_to_open), + residual_transition(space_to_open), + )); + } + text.push_str( + "\nThe HCA-style headword-expansion gate now supplies one narrow prose-routing premise. \ + Korean rules 29 and 34 require a fresh Roman section and continuous Roman transcription \ + for a complete all-capitals headword followed by a closed, multiword Roman expansion. \ + The implementation requires a headword of at least two ASCII capitals and at least two \ + ASCII-letter words inside the parenthesis; digits, operators, scripts, nested brackets, \ + and alphanumeric text after the closing parenthesis remain math-owned controls. Rule 34's \ + `링컨(Lincoln)은` additionally proves that attached Korean text after the closing \ + parenthesis stays on the prose route; the same boundary now covers the multiword form \ + without admitting ASCII letters or digits in the trailer. Together these boundaries \ + change 71 corpus cases from mismatch to exact and raise this cohort's exact controls from \ + 17 to 87. The residual members still measure contraction, capitalization, earlier sentence \ + differences, unsupported characters, and reference-order conflicts rather than \ + authorizing a wider surface-form rule.\n\n\ + The standalone-uppercase cohort is similarly ambiguous. Hangeul rule 28's appendix \ + defines the capital-word indicator for two or more consecutive capitals, and rule 29 \ + defines Roman indicators around Roman text in a Korean sentence. But math rule 12 also \ + uses uppercase Roman variables, while science rule 7 uses uppercase runs in chemical \ + formulas. The input gate cannot determine which semantic regime applies, so its output \ + differences are observations to review, not permission to infer an engine rule from the \ + corpus reference.\n\n\ + The Korean-prefixed all-caps parenthetical cohort isolates that ambiguity more narrowly. \ + Hangeul rules 28/29 require Roman and capitalization indicators for prose acronyms, \ + while science rule 7 requires element-by-element capitals for chemical formulae. Both \ + meanings can have the same input surface form. The observed `COO`/`NSC`/`MOU` output \ + differences therefore do not justify disabling either algorithm without independent \ + semantic evidence.\n\n\ + Two narrower cohorts separate causes hidden by the frequent `U+2834 -> U+2800` cell \ + transition. `single_capital_followed_by_parenthesized_digits` reproduces the current \ + math-token routing of forms such as `A(14)`: Hangeul rules 29 and 34 govern a Roman \ + section and a parenthesized Roman form, while math rule 6 independently defines \ + parenthesized function notation such as `f(x)`. A capital and numeric argument do not \ + remove that mathematical counterexample, so this localized routing difference remains \ + pending rather than authorizing an input-shape exception. \ + `mixed_roman_korean_word_before_uppercase_headword_expansion` separately targets the \ + next Roman headword after a mixed Roman+Korean word (for example, a Korean particle \ + attached to the previous Roman name). Its range is anchored to that later headword, not \ + to the earlier Roman entry. The narrow rules-29/34 prose gate described above is now \ + implemented, and this cohort no longer has a missing-entry localized transition. Its \ + remaining localized differences are later Roman-letter/contraction differences. The two \ + causes and their controls remain \ + separately measurable instead of widening the headword grammar.\n\n\ + The uppercase-Roman hyphen-digits cohort is a third independent cause. Hangeul rule 35 \ + explicitly shows `D-100` as a Roman-and-number continuation (2024 Korean-rules PDF \ + p.29), while math rule 2 defines subtraction and the math chapters allow uppercase Roman \ + variables. The surface form alone therefore does not prove whether `F-35` is an \ + identifier or a subtraction expression. This cohort records the current operator-routing \ + signature and exact controls without merging it into either `A(14)` or HCA-style \ + diagnostics. No engine change is made without both a safe semantic boundary and exact \ + controls.\n\n\ + The all-caps `OU` cohort isolates a frequent output transition without treating the \ + reference as a rule. Hangeul rules 28, 29, and 32 delegate Roman-letter content to UEB \ + (2024 Korean-rules PDF p.25 and following rules). \ + UEB 10.12.1 says not to use a contraction when it is known or can be determined that an \ + abbreviation or acronym's letters are pronounced separately, but says to use the \ + contraction when that pronunciation is in doubt; UEB 10.12.2 otherwise uses \ + contractions in abbreviations and acronyms (UEB 2024 PDF pp.191-192; Korean UEB \ + translation PDF pp.182-183). Thus an expected `o` + `u` versus the \ + current `ou` groupsign can be localized to an uppercase run, yet the surface run alone \ + cannot distinguish a letter-by-letter initialism from a pronounceable word or acronym. \ + That distinction needs lexical or semantic evidence absent from this input gate. Exact \ + members are controls, identical-input conflicting references remain `corpus_suspect`, \ + and no engine change is inferred.\n\n\ + The all-caps Roman middle-dot cohort is also semantically underdetermined. Hangeul rule \ + 29 defines Roman indicators around Roman text in a Korean sentence, and Hangeul rule 50 \ + requires U+00B7 to be attached on both sides, but neither rule says that the punctuation \ + joins the adjacent Roman runs into one Roman span. Math rule 2 separately defines the \ + same printed dot as multiplication, and science rule 4 uses it inside chemical \ + formulae. An input-only `AI·SW` gate therefore cannot prove which mode transition is \ + required. Exact cases remain controls, mismatches retain their existing primary class, \ + and no engine rule is inferred from their references. Representative samples are \ + sentence-level evidence: when the reported first difference precedes the detected \ + middle-dot span, the cohort must not be treated as the cause of that mismatch.\n\n\ + The narrower Roman-before-middle-dot boundary cohort separates that semantic question \ + from a checkable indicator boundary. Hangeul rule 29 requires a Roman terminator after \ + Roman text. Rule 33 enumerates the punctuation that suppresses or moves that terminator, \ + but does not include U+00B7; rule 50 requires the middle dot to be attached on both sides \ + and does not state a Roman-terminator exception. Thus a localized reference that omits \ + the terminator conflicts with the current rule-29/33 path on the available PDF text. \ + This is conservative corpus/PDF-reference review evidence, not permission to remove the \ + terminator or to reclassify non-localized cases.\n\n\ + The inline parenthesized-operator cohort has an independently checkable spacing boundary. \ + Hangeul rule 46 inserts spaces only when an operation or comparison sign is between \ + Korean text, while the literal parentheses intervene in this gate. Hangeul rule 49 says \ + punctuation spacing follows the print, and science rule 21 prints and brailles `(-)` and \ + `(+)` with no spaces inside the parentheses. The output-signature count is therefore the \ + implementation-candidate subset; mere sentence-level coexistence is retained only as a \ + control. ASCII hyphen-minus is independently supported through the `Symbol`/rule-49 \ + punctuation path, while `+`, `×`, `÷`, and `=` reach the `MathSymbol` spacing rule; \ + end-to-end tests preserve tight parentheses on both paths. Analyzer checkpoint `30a9b10` \ + recorded the pre-fix baseline as 23 candidates, 0 exact, 23 mismatches, and 19 \ + mismatches whose first difference was signature-local. The generalized rule-46/49 fix \ + is evaluated below against that immutable baseline rather than inferred from a reference \ + string.\n\n\ + The tight-triangle cohort is not an implementation premise. Hangeul rule 49 assigns `△` \ + the omission-mark role and requires print spacing to be followed, while rule 72 also \ + assigns the same glyph a bullet role but shows a print space after every bullet. A tight \ + corpus input does not identify which role was intended, and adding a space absent from \ + the input would contradict rule 49 unless independent layout evidence establishes a \ + bullet. The localizer searches the complete actual output for a neutral-Korean, \ + current-engine signature covering the mark and its first following Korean character; \ + it neither encodes a context-sensitive sentence prefix in isolation nor reads the \ + reference output. Tight marks followed by ASCII letters or digits remain outside this \ + gate. Localized mismatches are therefore corpus/layout review evidence only.\n\n\ + Corpus contradictions remain a separate gate: identical inputs with conflicting \ + references are classified as `corpus_suspect` before these cohorts are recorded and \ + would appear explicitly in each mismatch primary-class distribution. Their absence does \ + not prove a reference correct; it only means that this deterministic contradiction test \ + did not fire.\n", + ); + if let Some(stats) = report + .pending_rule_review_clusters + .get(ALLCAPS_ROMAN_RUN_CONTAINING_AR) + { + let primary_count = |name: &str| { + stats + .mismatch_primary_classes + .get(name) + .copied() + .unwrap_or(0) + }; + let localized_transition = |name: &str| { + stats + .first_difference_in_output_signature_transitions + .get(name) + .copied() + .unwrap_or(0) + }; + let raw_transition = |name: &str| { + report + .pending_first_difference_cell_transitions + .get(name) + .map(|transition| transition.cases) + .unwrap_or(0) + }; + let residual_transition = |name: &str| { + report + .pending_first_difference_transitions_after_localized_cohorts + .get(name) + .map(|transition| transition.cases) + .unwrap_or(0) + }; + let target = "U+2801 ⠁ -> U+281C ⠜"; + let reverse = "U+281C ⠜ -> U+2801 ⠁"; + text.push_str(&format!( + "\n### Uppercase Roman runs containing `AR`\n\n\ + UEB 10.12.1 (2024 UEB PDF pp.191-192, printed pp.163-164) says not to \ + use a contraction when letters within an abbreviation or acronym are known to be \ + pronounced separately, but to use the contraction in case of doubt. Its official \ + `DAR` example writes separate `a` and `r` cells, while UEB 10.12.2's official \ + `START` example uses the `ar` groupsign. Both full-encoder controls pass. Thus an \ + uppercase surface containing `AR` does not itself supply the pronunciation or \ + lexical meaning needed to select either form.\n\n\ + The output-localized cohort contains {} candidates, {} exact controls, and {} \ + mismatches. Existing mismatch primary classes are preserved: {} \ + `pending_rule_review`, {} `corpus_suspect`, {} `comparison_method`, and {} \ + `unsupported_character_review`. Of {} evaluable mismatches, {} have their first \ + difference inside the detected current-engine run: {} `{target}` and {} \ + `{reverse}`. Across raw pending transitions and the final residual after localized \ + cohorts, the target count is {} -> {}; the reverse is {} -> {}. Inputs denoting \ + separately pronounced initialisms such as corpus `AR`/`ARS` coexist with exact or \ + officially contracted controls. This is deterministic evidence for a \ + pronunciation-dependent pending cohort, not an engine rule or primary \ + reclassification. Representative shard/index samples are retained above.\n", + stats.candidates, + stats.exact, + stats.mismatch, + primary_count("pending_rule_review"), + primary_count("corpus_suspect"), + primary_count("comparison_method"), + primary_count("unsupported_character_review"), + stats.output_signature_mismatches_evaluated, + stats.first_difference_in_output_signature, + localized_transition(target), + localized_transition(reverse), + raw_transition(target), + residual_transition(target), + raw_transition(reverse), + residual_transition(reverse), + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(ROMAN_RUN_AFTER_CLOSED_ROMAN_ENCLOSURE) + { + let primary_count = |name: &str| { + stats + .mismatch_primary_classes + .get(name) + .copied() + .unwrap_or(0) + }; + let localized_transition = |name: &str| { + stats + .first_difference_in_output_signature_transitions + .get(name) + .copied() + .unwrap_or(0) + }; + let raw_transition = |name: &str| { + report + .pending_first_difference_cell_transitions + .get(name) + .map(|transition| transition.cases) + .unwrap_or(0) + }; + let residual_transition = |name: &str| { + report + .pending_first_difference_transitions_after_localized_cohorts + .get(name) + .map(|transition| transition.cases) + .unwrap_or(0) + }; + let target = "U+2834 ⠴ -> U+2830 ⠰"; + let reverse = "U+2830 ⠰ -> U+2834 ⠴"; + text.push_str(&format!( + "\n### Roman run after a closed Roman enclosure\n\n\ + Korean rule 29 places one Roman indicator/terminator pair around consecutive Roman \ + items, with `Los Angeles` and `Table of Contents` as its printed multiword examples. \ + Rule 34 separately omits the Roman terminator when Roman text is enclosed by \ + quotation marks or parentheses. Neither printed rule states whether a later Roman \ + run after the enclosure, intervening punctuation, and whitespace is a continuation \ + of that section or a fresh section. The input gate therefore detects only the \ + structural boundary; it does not label the later run as semantically new.\n\n\ + The cohort contains {} candidates, {} exact controls, and {} mismatches. Existing \ + mismatch primary classes are preserved: {} `pending_rule_review`, {} \ + `corpus_suspect`, and {} `unsupported_character_review`. Of {} evaluable \ + mismatches, {} are output-localized to the current later-run signature plus its one \ + leading boundary cell: {} `{target}` and {} `{reverse}`. The target counted {} raw \ + and 333 residual cases before this cohort; it is now {} final residual. Across raw \ + and final residual maps, `{reverse}` is {} -> {}. Exact controls coexist with both \ + directions, and 321 mismatches are already independently identified corpus \ + contradictions. Without a printed fresh-entry example or semantic enclosure model, \ + changing continuation state would be reference-fitting; this remains a deterministic \ + pending/corpus-review cohort only. Representative shard/index samples are retained \ + above.\n", + stats.candidates, + stats.exact, + stats.mismatch, + primary_count("pending_rule_review"), + primary_count("corpus_suspect"), + primary_count("unsupported_character_review"), + stats.output_signature_mismatches_evaluated, + stats.first_difference_in_output_signature, + localized_transition(target), + localized_transition(reverse), + raw_transition(target), + residual_transition(target), + raw_transition(reverse), + residual_transition(reverse), + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(ALLCAPS_ROMAN_RUN_CONTAINING_ED) + { + let primary_count = |name: &str| { + stats + .mismatch_primary_classes + .get(name) + .copied() + .unwrap_or(0) + }; + let localized_transition = |name: &str| { + stats + .first_difference_in_output_signature_transitions + .get(name) + .copied() + .unwrap_or(0) + }; + let raw_transition = |name: &str| { + report + .pending_first_difference_cell_transitions + .get(name) + .map(|transition| transition.cases) + .unwrap_or(0) + }; + let residual_transition = |name: &str| { + report + .pending_first_difference_transitions_after_localized_cohorts + .get(name) + .map(|transition| transition.cases) + .unwrap_or(0) + }; + let target = "U+2811 ⠑ -> U+282B ⠫"; + let reverse = "U+282B ⠫ -> U+2811 ⠑"; + text.push_str(&format!( + "\n### Uppercase Roman runs containing `ED`\n\n\ + UEB 10.12.1 (2024 UEB PDF pp.191-192, printed pp.163-164) requires \ + separate letters when an abbreviation's letters are known to be pronounced \ + separately, and its official `OED` example writes `e` and `d` separately. Rule \ + 10.12.2 otherwise uses contractions; its official `BEd` example uses the `ed` \ + groupsign. Both full-encoder controls pass. Consequently an uppercase `ED` surface \ + cannot establish pronunciation or abbreviation semantics by spelling alone.\n\n\ + The cohort contains {} candidates, {} exact controls, and {} mismatches. Existing \ + mismatch primary classes remain {} `pending_rule_review`, {} `corpus_suspect`, {} \ + `comparison_method`, and {} `unsupported_character_review`. Of {} evaluable \ + mismatches, {} are localized to the detected current-engine run: {} `{target}` and \ + {} `{reverse}`. Across raw pending transitions and the final residual after localized \ + cohorts, the target is {} -> {} and the reverse is {} -> {}. Corpus initialisms such \ + as `LED` and `GED` require external pronunciation knowledge, while exact and official \ + controls preserve contraction-bearing outcomes. No engine change or primary \ + reclassification is made; representative shard/index samples are retained above.\n", + stats.candidates, + stats.exact, + stats.mismatch, + primary_count("pending_rule_review"), + primary_count("corpus_suspect"), + primary_count("comparison_method"), + primary_count("unsupported_character_review"), + stats.output_signature_mismatches_evaluated, + stats.first_difference_in_output_signature, + localized_transition(target), + localized_transition(reverse), + raw_transition(target), + residual_transition(target), + raw_transition(reverse), + residual_transition(reverse), + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(ALLCAPS_ROMAN_RUN_CONTAINING_ST) + { + let primary_count = |name: &str| { + stats + .mismatch_primary_classes + .get(name) + .copied() + .unwrap_or(0) + }; + let localized_transition = |name: &str| { + stats + .first_difference_in_output_signature_transitions + .get(name) + .copied() + .unwrap_or(0) + }; + let raw_transition = |name: &str| { + report + .pending_first_difference_cell_transitions + .get(name) + .map(|transition| transition.cases) + .unwrap_or(0) + }; + let residual_transition = |name: &str| { + report + .pending_first_difference_transitions_after_localized_cohorts + .get(name) + .map(|transition| transition.cases) + .unwrap_or(0) + }; + let target = "U+280E ⠎ -> U+280C ⠌"; + let reverse = "U+280C ⠌ -> U+280E ⠎"; + text.push_str(&format!( + "\n### Uppercase Roman runs containing `ST`\n\n\ + UEB 10.12.1 (2024 UEB PDF pp.191-192, printed pp.163-164) says not to \ + use a contraction when letters within an abbreviation or acronym are known to be \ + pronounced separately, but to use the contraction in case of doubt. Rule 10.12.2 \ + requires contractions in other abbreviations and acronyms. Consequently an \ + uppercase `ST` surface alone cannot determine whether `s` + `t` or the `st` \ + groupsign is required. The current cohort contains {} candidates, {} exact controls, \ + and {} mismatches; primary classes remain {} `pending_rule_review`, {} \ + `corpus_suspect`, {} `comparison_method`, and {} \ + `unsupported_character_review`. Of {} evaluable mismatches, {} are localized to the \ + detected current-engine run: {} `{target}` and {} `{reverse}`. The target's \ + raw-to-residual count is {} -> {}, and the reverse is {} -> {}. Exact controls such \ + as `STAYG`, `KAIST`, `DGIST`, and `POSTECH` coexist with localized mismatches such as \ + `WSTS`, `HUST`, `OST`, and `USTR`; representative shard/index samples are preserved \ + in the cluster table. This lexical/pronunciation distinction cannot be inferred from \ + the input-only spelling, so no engine change or primary reclassification is made.\n", + stats.candidates, + stats.exact, + stats.mismatch, + primary_count("pending_rule_review"), + primary_count("corpus_suspect"), + primary_count("comparison_method"), + primary_count("unsupported_character_review"), + stats.output_signature_mismatches_evaluated, + stats.first_difference_in_output_signature, + localized_transition(target), + localized_transition(reverse), + raw_transition(target), + residual_transition(target), + raw_transition(reverse), + residual_transition(reverse), + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(UPPERCASE_ASCII_SEGMENTS_JOINED_BY_AMPERSAND) + { + let primary_count = |name: &str| { + stats + .mismatch_primary_classes + .get(name) + .copied() + .unwrap_or(0) + }; + text.push_str(&format!( + "\n### Uppercase segments joined by ampersand: capitalization extent\n\n\ + UEB 8.4.2 (2024 UEB PDF p.118, printed p.90) terminates capitals word \ + mode at a nonalphabetic symbol. UEB 3.1.1 and the capitalization examples \ + (PDF pp.51 and 120, printed pp.23 and 92) consequently print `AT&T` as \ + `⠠⠠⠁⠞⠈⠯⠠⠞` and `B&B` as `⠠⠃⠈⠯⠠⠃`: Roman mode remains \ + continuous, but capitalization restarts for each ASCII-letter segment. The detector \ + accepts only complete uppercase ASCII segments joined directly by `&`, with the same \ + non-alphanumeric outer boundaries as the existing Roman-ampersand rule, and requires \ + the run to begin its whitespace-delimited token. Korean-attached and \ + punctuation-prefixed occurrences stay outside the change scope.\n\n\ + At the analyzer-only checkpoint this cohort contained 439 candidates, 1 exact \ + control, 438 mismatches, and 220 first differences localized inside the independently \ + reproduced Korean-context signature. After the general capitalization correction it \ + contains {} candidates, {} exact controls, and {} mismatches. Existing remaining \ + mismatch primaries are {} `pending_rule_review`, {} `corpus_suspect`, {} \ + `comparison_method`, and {} `unsupported_character_review`. Of {} evaluable \ + mismatches, {} have their first difference inside that signature. The sole pre-change \ + exact member contained lowercase Roman text later in the same whitespace token and \ + was outside the production predicate's actual change scope; its primary outcome was \ + preserved. The cohort table above retains the transition distribution and shard/index \ + samples. Capitalization extent is fixed by the official symbol examples and requires \ + neither pronunciation nor corpus semantics; the diagnostic never changes a primary \ + class.\n", + stats.candidates, + stats.exact, + stats.mismatch, + primary_count("pending_rule_review"), + primary_count("corpus_suspect"), + primary_count("comparison_method"), + primary_count("unsupported_character_review"), + stats.output_signature_mismatches_evaluated, + stats.first_difference_in_output_signature, + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(CAPITALS_WORD_NONLETTER_CHANGE_SCOPE) + { + text.push_str(&format!( + "\n### Capitals-word nonletter change-scope audit\n\n\ + This input-only scope exactly mirrors the former token predicate where it could \ + incorrectly carry capitals-word handling across a nonletter. UEB 8.4.2 terminates \ + that mode at the nonletter. Trailing nonletters after the final uppercase run are \ + excluded because their output is unchanged. Before the correction all 1,733 \ + candidates were mismatches and none was exact. The current run has {} candidates, \ + {} exact controls, and {} mismatches. A complete exact-ID set audit found 852 newly \ + exact cases and zero cases lost from the 68,439-exact baseline, yielding \ + 69,291/83,528 (82.96%). Two deliberately rejected wider predicates exposed why the \ + boundary matters: requiring an entirely uppercase-only token lost 87 former exact \ + cases, while treating every initial uppercase run as token-level capitals mode lost \ + 33 mixed-case controls such as the UEB `TVOntario` class by bypassing its required \ + capitals terminator. The retained predicate pre-emits only when the initial run has \ + at least two capitals and every ASCII letter in the token is uppercase; Rule 28 \ + independently restarts capitalization after the nonletter. This cohort remains a \ + regression audit only: membership does not assign a primary class or attribute a \ + first difference.\n", + stats.candidates, stats.exact, stats.mismatch, + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(ATTACHED_ASCII_ROMAN_SEGMENTS_JOINED_BY_AMPERSAND) + { + let primary_count = |name: &str| { + stats + .mismatch_primary_classes + .get(name) + .copied() + .unwrap_or(0) + }; + let transition_count = |name: &str| { + stats + .first_difference_in_output_signature_transitions + .get(name) + .copied() + .unwrap_or(0) + }; + let raw_count = |name: &str| { + report + .pending_first_difference_cell_transitions + .get(name) + .map(|transition| transition.cases) + .unwrap_or(0) + }; + let residual_count = |name: &str| { + report + .pending_first_difference_transitions_after_localized_cohorts + .get(name) + .map(|transition| transition.cases) + .unwrap_or(0) + }; + let target = "U+2808 ⠈ -> U+2832 ⠲"; + let reverse = "U+2832 ⠲ -> U+2808 ⠈"; + text.push_str(&format!( + "\n### Attached Roman segments joined by ampersand\n\n\ + UEB 3.1.1 (2024 UEB PDF p.51, printed p.23) directly prints `AT&T` and \ + `B&B` with the attached ampersand `⠈⠯` and no mode boundary around it. Korean \ + rule 71 (2024 Korean-rules PDF p.51, printed p.45) assigns the same `⠈⠯` \ + cells, while rule 29 places Roman entry before a Roman section and termination after \ + its last item. Before the engine change, the Korean-context path instead exited before \ + `&`, wrapped the information symbol as a separate Roman section, and re-entered for \ + the following letters.\n\n\ + The baseline was 802 candidates / 0 exact / 802 mismatch, preserving 776 \ + `pending_rule_review`, 10 `corpus_suspect`, and 16 \ + `unsupported_character_review` primary classifications. Its real-prefix localizer \ + assigned only the output cell immediately before `&`: 356/802 mismatches localized, \ + all 356 were `{target}`, the raw-to-residual target count was 411 -> 53, and \ + `{reverse}` was 0 raw / 0 residual.\n\n\ + The implemented gate shares the analyzer's complete-run predicate: one or more \ + non-empty ASCII-letter segments joined directly by `&`, with non-alphanumeric outer \ + boundaries. It keeps the existing Roman mode open and suppresses only rule 71's \ + redundant wrapper around that ampersand. Spaced `Marks & Spencer`, Korean `가&나`, \ + empty segments, and outer digit continuations remain outside the gate. Official \ + full-encoder controls `AT&T` and `B&B` pass, and the Korean rule-71 spaced example \ + `종이접기 & 클레이아트` retains its independent `⠴⠈⠯⠲` section.\n\n\ + After the change the same cohort contains {} candidates, {} exact and {} mismatch. \ + Current mismatch primary classes remain evaluator-owned: {} `pending_rule_review`, {} \ + `corpus_suspect`, {} `unsupported_character_review`, and {} `comparison_method`. The \ + localizer evaluates all {} remaining mismatches and finds {} target-localized cases \ + ({} `{target}`); current raw-to-residual target count is {} -> {}, while `{reverse}` \ + remains {} raw / {} residual. Cohort exact increases by 273 and corpus-wide exact \ + increases by the same 273, from 67,442 to 67,715. Because every changed input must \ + satisfy this shared predicate and the baseline had no exact member, this boundary has \ + no exact regression. The remaining 529 candidates differ elsewhere or retain an \ + existing comparison, corpus-suspect, unsupported, or pending cause. Representative \ + shard/index samples are retained above.\n", + stats.candidates, + stats.exact, + stats.mismatch, + primary_count("pending_rule_review"), + primary_count("corpus_suspect"), + primary_count("unsupported_character_review"), + primary_count("comparison_method"), + stats.output_signature_mismatches_evaluated, + stats.first_difference_in_output_signature, + transition_count(target), + raw_count(target), + residual_count(target), + raw_count(reverse), + residual_count(reverse), + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(AMPERSAND_BEFORE_ATTACHED_ASCII_ROMAN_SEGMENT) + { + let primary_count = |name: &str| { + stats + .mismatch_primary_classes + .get(name) + .copied() + .unwrap_or(0) + }; + let localized_count = |name: &str| { + stats + .first_difference_in_output_signature_transitions + .get(name) + .copied() + .unwrap_or(0) + }; + let raw_count = |name: &str| { + report + .pending_first_difference_cell_transitions + .get(name) + .map(|transition| transition.cases) + .unwrap_or(0) + }; + let residual_count = |name: &str| { + report + .pending_first_difference_transitions_after_localized_cohorts + .get(name) + .map(|transition| transition.cases) + .unwrap_or(0) + }; + let omitted_indicator = "U+2808 ⠈ -> U+2834 ⠴"; + let omitted_indicator_reverse = "U+2834 ⠴ -> U+2808 ⠈"; + let duplicate_entry = "U+2834 ⠴ -> U+2820 ⠠"; + let duplicate_entry_reverse = "U+2820 ⠠ -> U+2834 ⠴"; + text.push_str(&format!( + "\n### Ampersand before an attached ASCII Roman segment\n\n\ + This is the residual boundary not covered by the implemented `A&B` gate. Official \ + UEB 3.1.1 (2024 UEB PDF p.51, printed p.23) prints `&c (etc)` as \ + `⠈⠯⠉ ⠐⠣⠑⠞⠉⠐⠜`, with no mode break between the ampersand and `c`; its \ + `AT&T` and `B&B` examples give the same attached behavior between Roman segments. \ + Korean rule 71 (2024 Korean-rules PDF pp.51-52, printed pp.45-46) wraps an \ + ampersand in Roman indicators when needed to distinguish it from Hangul, while \ + rules 29 and 32 require one Roman section for consecutive Roman material and UEB \ + transcription inside that section. The spaced Korean control `종이접기 & \ + 클레이아트` remains an independently closed Rule-71 symbol.\n\n\ + The diagnostic checkpoint baseline was 30 candidates / 0 corpus exact / 30 \ + mismatch, preserving 26 `pending_rule_review` and 4 `corpus_suspect` primary \ + classes. Its then-current Rule-71 exit localizer found 16/30 first differences: 15 \ + `U+2820 ⠠ -> U+2832 ⠲` and 1 `U+2834 ⠴ -> U+2832 ⠲`; both localized reverses \ + were zero. The official full-encoder `&c`, `AT&T`, and `B&B` examples were the \ + positive controls, and the spaced Korean Rule-71 example was the negative boundary \ + control.\n\n\ + The implemented rule is limited to an ampersand followed by a complete attached \ + ASCII-letter segment, with no left ASCII alphanumeric or adjacent ampersand and no \ + trailing digit continuation. Rule 71 opens one Roman section before `&`; rule 29 \ + now leaves it open for the attached letters. It does not name a corpus input or \ + inspect a reference. After the change, the cohort has {} candidates, {} exact and \ + {} mismatch; the corpus-wide exact total rises by the same 12 cases, from 68,175 to \ + 68,187, so no exact regression occurs inside or outside this gate. The 16 former \ + exit/re-entry transitions disappear; 12 become exact and 4 remain mismatches at a \ + different PDF-conflicting boundary. Existing mismatch primary classes remain {} \ + `pending_rule_review`, {} `corpus_suspect`, {} `comparison_method`, and {} \ + `unsupported_character_review`.\n\n\ + Of {} current evaluable mismatches, {} are localized to the occurrence-specific \ + entry signature: {} `{omitted_indicator}` where a parenthesized `&TEAM` reference \ + omits Rule 71's required Roman indicator, and {} `{duplicate_entry}` where a \ + `드림&Dream` reference inserts another Roman indicator inside the same continuous \ + section. Their raw-to-residual counts are {} -> {} and {} -> {}; the corresponding \ + reverse maps are {} -> {} and {} -> {}. These four cases remain conservative \ + corpus/PDF-reference review rather than widening or undoing the rule. Exact samples \ + such as `&TEAM`, `과학&ICT`, and `한국&K리츠`, with shard/index above, are current \ + controls. Primary classifications are never changed by this cohort.\n", + stats.candidates, + stats.exact, + stats.mismatch, + primary_count("pending_rule_review"), + primary_count("corpus_suspect"), + primary_count("comparison_method"), + primary_count("unsupported_character_review"), + stats.output_signature_mismatches_evaluated, + stats.first_difference_in_output_signature, + localized_count(omitted_indicator), + localized_count(duplicate_entry), + raw_count(omitted_indicator), + residual_count(omitted_indicator), + raw_count(duplicate_entry), + residual_count(duplicate_entry), + raw_count(omitted_indicator_reverse), + residual_count(omitted_indicator_reverse), + raw_count(duplicate_entry_reverse), + residual_count(duplicate_entry_reverse), + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(ASCII_APOSTROPHE_BETWEEN_ASCII_LETTERS) + { + text.push_str(&format!( + "\n### ASCII apostrophe between Roman letter runs\n\n\ + This output-localized cohort requires a straight ASCII apostrophe with an ASCII \ + letter immediately on both sides. UEB 8.4.2 (2024 UEB PDF pp.118-119, printed \ + pp.90-91) directly prints `O'Hara`, `DON'T`, and `THAT'S` with the apostrophe cell \ + inside the same Roman word; capitals-word mode may end at the apostrophe, but the \ + Roman section itself does not. Detached quotation marks, Korean single quotation \ + marks, digit-adjacent measurement signs, and an apostrophe at the end of one \ + whitespace-delimited token before another Roman word are excluded.\n\n\ + The diagnostic checkpoint had 147 candidates / 0 exact / 147 mismatch. Its \ + occurrence-specific localizer put 112 first differences on the apostrophe boundary: \ + 30 expected apostrophe cell `U+2804 ⠄` versus actual capital indicator `U+2820 ⠠`, \ + and 82 expected `U+2804 ⠄` versus actual Roman indicator `U+2834 ⠴`; neither target \ + had a localized reverse. The implementation keeps only a same-token apostrophe with \ + ASCII letters immediately on both sides in the current Roman section, delegates its \ + cell to the existing UEB section-7 punctuation encoder, and restarts capitals mode \ + for an uppercase run after the nonalphabetic apostrophe. Korean Rule 37 still \ + suppresses whole-word contractions at a Roman entry. A rejected broader route made \ + the detached `Guitar' Listening` control exact, so the final gate explicitly does not \ + look through whitespace.\n\n\ + After the correction the cohort has {} candidates, {} exact controls, and {} \ + mismatches. Of {} evaluable residual mismatches, {} place their first difference \ + inside the independently encoded current signature, but those residual transitions \ + are other letter/spacing differences rather than either former apostrophe transition. \ + The complete corpus exact-ID audit found 68 newly exact cases and zero formerly exact \ + cases lost, raising the corpus total from 69,291 to 69,359. Cohort membership itself \ + never rewrites a primary class; the engine result may make a member exact or expose \ + an independently classified residual.\n", + stats.candidates, + stats.exact, + stats.mismatch, + stats.output_signature_mismatches_evaluated, + stats.first_difference_in_output_signature, + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(SPACED_COMMA_BETWEEN_ASCII_DIGIT_RUNS) + { + let primary_count = |name: &str| { + stats + .mismatch_primary_classes + .get(name) + .copied() + .unwrap_or(0) + }; + let localized_count = |name: &str| { + stats + .first_difference_in_output_signature_transitions + .get(name) + .copied() + .unwrap_or(0) + }; + let raw_count = |name: &str| { + report + .pending_first_difference_cell_transitions + .get(name) + .map(|transition| transition.cases) + .unwrap_or(0) + }; + let residual_count = |name: &str| { + report + .pending_first_difference_transitions_after_localized_cohorts + .get(name) + .map(|transition| transition.cases) + .unwrap_or(0) + }; + let korean_to_ueb = "U+2810 ⠐ -> U+2802 ⠂"; + let ueb_to_korean = "U+2802 ⠂ -> U+2810 ⠐"; + text.push_str(&format!( + "\n### Spaced comma between ASCII digit runs\n\n\ + This output-localized cohort requires a comma immediately after an ASCII digit, \ + one or more following whitespace characters, and another ASCII digit. Korean rule \ + 41 (2024 Korean-rules PDF p.33, printed p.27) assigns `⠂` only when the comma is \ + *attached* between digits, as in the exact control `9,375명`; rule 49 (PDF \ + pp.37-38, printed pp.31-32) assigns the ordinary Korean comma `⠐`, illustrated by \ + `근면, 검소, 협동은 ...`. Rule 33 (PDF p.28, printed p.22) independently keeps \ + Korean punctuation at Roman-to-Korean boundaries. By contrast, official UEB 7 \ + (2024 UEB PDF p.103, printed p.75) defines its prose comma as `⠂`, and UEB 6.2.1 \ + (PDF p.94, printed p.66) retains `⠂` inside attached numeric forms such as \ + `3,500`. Those two surfaces are negative controls and are excluded by this gate.\n\n\ + The diagnostic baseline was 217 candidates / 7 exact / 210 mismatch, with 177 \ + occurrence-specific `{korean_to_ueb}` first differences and no localized reverse. \ + Rule 41 had looked through `remaining_words`, incorrectly treating whitespace as if \ + the following digit were attached. The implementation now inspects only the next \ + character in the same token. It neither names a corpus input nor consults expected \ + output; attached numbers and UEB punctuation remain owned by their existing routes.\n\n\ + After the correction, the cohort has {} candidates / {} exact / {} mismatch. \ + Existing mismatch primaries remain {} `pending_rule_review`, {} `corpus_suspect`, \ + {} `comparison_method`, and {} `unsupported_character_review`. Of {} evaluable \ + current mismatches, {} localize to the comma cell: {} `{korean_to_ueb}` and {} \ + `{ueb_to_korean}`. The cohort gains 174 exact cases. Across all pending cases, the \ + current raw-to-residual counts are {} -> {} for the target and {} -> {} for the \ + reverse. The standard controls `9,375명`, `창세기 12,1-9`, and `근면, 검소, \ + 협동은 ...` retain their PDF cells. No primary class is changed by the cohort.\n", + stats.candidates, + stats.exact, + stats.mismatch, + primary_count("pending_rule_review"), + primary_count("corpus_suspect"), + primary_count("comparison_method"), + primary_count("unsupported_character_review"), + stats.output_signature_mismatches_evaluated, + stats.first_difference_in_output_signature, + localized_count(korean_to_ueb), + localized_count(ueb_to_korean), + raw_count(korean_to_ueb), + residual_count(korean_to_ueb), + raw_count(ueb_to_korean), + residual_count(ueb_to_korean), + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(ASCII_ROMAN_COMMA_BEFORE_DIGIT_KOREAN_TOKEN) + { + let primary_count = |name: &str| { + stats + .mismatch_primary_classes + .get(name) + .copied() + .unwrap_or(0) + }; + let localized_count = |name: &str| { + stats + .first_difference_in_output_signature_transitions + .get(name) + .copied() + .unwrap_or(0) + }; + let korean_to_ueb = "U+2810 ⠐ -> U+2802 ⠂"; + let ueb_to_korean = "U+2802 ⠂ -> U+2810 ⠐"; + text.push_str(&format!( + "\n### ASCII/Roman-tail comma before a digit-led Korean token\n\n\ + This companion cohort is disjoint from the preceding digit-comma gate: the comma \ + is immediately preceded by an ASCII letter, followed after whitespace by a token \ + that starts with a digit and contains Korean script. Korean rule 33 (2024 \ + Korean-rules PDF p.28, printed p.22) says that punctuation with different UEB and \ + Korean cells, including comma, is written as Korean punctuation at a \ + Roman-to-Korean boundary and suppresses the Roman terminator. Rule 49 supplies \ + `⠐`; UEB 7 supplies `⠂` only while the comma remains inside English text. Requiring \ + Korean script in the right token is therefore the negative control against \ + reclassifying an English date or number sequence from surface punctuation alone.\n\n\ + The diagnostic baseline was 58 candidates / 0 exact / 58 mismatch. Of those, 23 \ + had `{korean_to_ueb}` at the comma inside the independently encoded complete \ + boundary signature and none had the reverse. The same rule-41 correction removes \ + the cross-token ASCII-letter lookup; rule 33 and the existing English-symbol route \ + then choose the punctuation from the actual surrounding scripts.\n\n\ + After the correction, this cohort has {} candidates / {} exact / {} mismatch. \ + Existing mismatch primaries remain {} `pending_rule_review`, {} `corpus_suspect`, \ + {} `comparison_method`, and {} `unsupported_character_review`. Of {} evaluable \ + current mismatches, {} localize to the comma-cell signature: {} \ + `{korean_to_ueb}` and {} `{ueb_to_korean}`. Three cases become exact. The official \ + rule-33 `KTX, 새마을호` boundary and UEB prose comma remain independent standard \ + controls. The detector and localizer do not read expected output to choose a route, \ + and membership does not change a primary class.\n", + stats.candidates, + stats.exact, + stats.mismatch, + primary_count("pending_rule_review"), + primary_count("corpus_suspect"), + primary_count("comparison_method"), + primary_count("unsupported_character_review"), + stats.output_signature_mismatches_evaluated, + stats.first_difference_in_output_signature, + localized_count(korean_to_ueb), + localized_count(ueb_to_korean), + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(PERCENT_POINT_UNIT_LIST_COMMA) + { + let primary_count = |name: &str| { + stats + .mismatch_primary_classes + .get(name) + .copied() + .unwrap_or(0) + }; + let localized_count = |name: &str| { + stats + .first_difference_in_output_signature_transitions + .get(name) + .copied() + .unwrap_or(0) + }; + let korean_to_ueb = "U+2810 ⠐ -> U+2802 ⠂"; + let ueb_to_korean = "U+2802 ⠂ -> U+2810 ⠐"; + text.push_str(&format!( + "\n### Percent-point unit list comma\n\n\ + Korean rule 69 attachment 2 (2024 Korean-rules PDF p.50, printed p.44) \ + explicitly defines `%p` as the percent-point unit. This cohort requires two \ + complete numeric `%p` tokens separated by comma plus whitespace, so rule 49's \ + ordinary Korean comma is the punctuation boundary; it does not infer arbitrary \ + ASCII suffixes as units. The diagnostic baseline was 7 candidates / 0 exact / 7 \ + mismatch, with one localized `{korean_to_ueb}` and no reverse.\n\n\ + After the correction, this cohort has {} candidates / {} exact / {} mismatch, \ + preserving {} `pending_rule_review`, {} `corpus_suspect`, {} `comparison_method`, \ + and {} `unsupported_character_review` mismatch primaries. Of {} evaluable current \ + mismatches, {} localize to the independently encoded complete unit pair: {} \ + `{korean_to_ueb}` and {} `{ueb_to_korean}`. Five cases become exact; the other two \ + retain independent earlier differences. One of those five is also in the \ + Roman-tail cohort, leaving four disjoint `%p` gains. Thus +174 in the numeric-list \ + cohort, +3 in the Roman-tail cohort, and +4 disjoint here account for all +181 \ + corpus exact gains, with zero exact-set regressions. The engine contains only the general \ + rule-41 same-token boundary, not a `%p` special case.\n", + stats.candidates, + stats.exact, + stats.mismatch, + primary_count("pending_rule_review"), + primary_count("corpus_suspect"), + primary_count("comparison_method"), + primary_count("unsupported_character_review"), + stats.output_signature_mismatches_evaluated, + stats.first_difference_in_output_signature, + localized_count(korean_to_ueb), + localized_count(ueb_to_korean), + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(ATTACHED_KOREAN_AUXILIARY_ITDA_SPACING) + { + let primary_count = |name: &str| { + stats + .mismatch_primary_classes + .get(name) + .copied() + .unwrap_or(0) + }; + let localized_count = |name: &str| { + stats + .first_difference_in_output_signature_transitions + .get(name) + .copied() + .unwrap_or(0) + }; + let expected_attached = "U+2815 ⠕ -> U+2800 ⠀"; + let expected_spaced = "U+2800 ⠀ -> U+2815 ⠕"; + let other_attached = "U+2823 ⠣ -> U+2800 ⠀"; + text.push_str(&format!( + "\n### Attached Korean `있다` spacing normalization\n\n\ + Korean rule 49 (2024 Korean-rules PDF p.37, printed p.31) says that spacing follows \ + the print input. The PDF consistently retains an explicit space in `그리고 있다` \ + (physical p.18), `살고 있다` (p.26), and `수강하고 있다` (p.30), but it gives no \ + example authorizing a transcriber to insert a missing print space. The current \ + pre-fix token normalizer nevertheless split any Korean token ending in attached \ + `있다`.\n\n\ + The diagnostic baseline had 95 candidates / 0 exact / 95 mismatch. All 95 were in \ + the exact former implementation scope; 74 first differences were at the inserted \ + blank: 73 `{expected_attached}`, one `{other_attached}`, and no localized reverse. \ + The absence of a baseline exact member is the in-scope regression control.\n\n\ + After removing that input-correcting transformation, the cohort has {} candidates / \ + {} exact / {} mismatch, preserving {} \ + `pending_rule_review`, {} `corpus_suspect`, {} `comparison_method`, and {} \ + `unsupported_character_review` mismatch primaries. Of {} evaluable current \ + mismatches, {} still localize to an inserted blank: {} `{expected_attached}`, {} \ + `{other_attached}`, and {} `{expected_spaced}`. Seventy-one cases become exact; the \ + other 24 retain independent earlier differences. Corpus exact increases by the same \ + 71 with zero exact-set regressions. The three explicitly spaced PDF forms remain \ + full-encoder negative controls, so removing correction of missing input whitespace \ + does not remove a printed space. The local rule-47 standard case had accidentally \ + transcribed PDF physical p.36 `덮여 있다` as attached `덮여있다` while retaining the \ + PDF's spaced braille; correcting that input transcription restores the complete \ + 5,141/5,141 standard summary without an engine exception.\n", + stats.candidates, + stats.exact, + stats.mismatch, + primary_count("pending_rule_review"), + primary_count("corpus_suspect"), + primary_count("comparison_method"), + primary_count("unsupported_character_review"), + stats.output_signature_mismatches_evaluated, + stats.first_difference_in_output_signature, + localized_count(expected_attached), + localized_count(other_attached), + localized_count(expected_spaced), + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(UPPERCASE_ROMAN_HYPHEN_DIGITS) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent uppercase-Roman hyphen-digits measurement: {} candidates, {} exact \ + controls, {} mismatches, {pending} members in the actual `pending_rule_review` \ + subcluster, and {}/{} evaluable mismatches whose first difference is inside the \ + target run plus its entry boundary. It remains distinct from parenthesized digits \ + and headword expansions; no engine change is inferred.\n", + stats.candidates, + stats.exact, + stats.mismatch, + stats.first_difference_in_output_signature, + stats.output_signature_mismatches_evaluated + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(UPPERCASE_ALPHANUMERIC_ROMAN_DIGIT_SEQUENCE) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + let target = stats + .first_difference_in_output_signature_transitions + .get("U+2834 ⠴ -> U+2800 ⠀") + .copied() + .unwrap_or(0); + let reverse = stats + .first_difference_in_output_signature_transitions + .get("U+2800 ⠀ -> U+2834 ⠴") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent uppercase alphanumeric Roman-digit sequence measurement: {} candidates, \ + {} exact controls, {} mismatches, {pending} members in the actual \ + `pending_rule_review` subcluster, and {}/{} evaluable mismatches whose first \ + difference is localized to the current entry cells. The target `⠴ -> blank` occurs \ + {target} times and the reverse occurs {reverse} times. Korean rule 35 (2024 \ + Korean-rules PDF pp.29-30, printed pp.23-24) explicitly treats `MP3`, `A4`, `KF94`, \ + and `D-100` as Roman-number sequences. Math rules 11/12 (PDF pp.63-65, printed \ + pp.57-59) separately route mathematical Roman notation without the prose Roman \ + indicator and with two-cell spacing. Official UEB 6.5.1-6.5.2 (2024 UEB PDF p.96, \ + printed p.68) determines numeric/grade-1 continuation only after the surrounding \ + mode has been chosen. Thus surfaces such as `A-3` cannot be assigned identifier or \ + variable semantics from this structure alone. Exact controls and both directions \ + are retained; no engine change is inferred.\n", + stats.candidates, + stats.exact, + stats.mismatch, + stats.first_difference_in_output_signature, + stats.output_signature_mismatches_evaluated + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(CONSECUTIVE_ASCII_ROMAN_WORD_BOUNDARY) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + let target = stats + .first_difference_in_output_signature_transitions + .get("U+2800 ⠀ -> U+2832 ⠲") + .copied() + .unwrap_or(0); + let reverse = stats + .first_difference_in_output_signature_transitions + .get("U+2832 ⠲ -> U+2800 ⠀") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent consecutive ASCII-Roman word-boundary measurement: {} candidates, {} \ + exact controls, {} mismatches, {pending} members in the actual \ + `pending_rule_review` subcluster, and {}/{} evaluable mismatches whose first \ + difference is localized to the current boundary cell. The target reference blank \ + versus current terminator (`⠀ -> ⠲`) occurs {target} times; the exact reverse \ + (`⠲ -> ⠀`) occurs {reverse} times. Korean rule 29 (2024 Korean-rules PDF p.26, \ + printed p.20) explicitly treats consecutive Roman items as one Roman section, with \ + one Roman indicator before the first item and one terminator after the last; its \ + printed `Los Angeles` and `Table of Contents` examples are exact controls. Rule 34 \ + (PDF p.29, printed p.23) separately omits the terminator before a closing enclosure, \ + while official UEB 9.7.1 (2024 UEB PDF p.137, printed p.109) prints the multiword \ + prose enclosure `plays (such as Romeo and Juliet)` with ordinary internal spaces \ + and nested closing punctuation. Together they support preserving a complete \ + letter-and-space Roman parenthetical as prose, but not a fragment containing a \ + function-call opener, digit, operator, or nested bracket. The input-only gate does not decide \ + whether punctuation-separated text such as `ESS /VPP` continues the same section, \ + so that variant remains outside this cohort. Primary classes are preserved. Before \ + the narrow token-routing change this cohort was 4,679 candidates / 2,058 exact / \ + 2,621 mismatch, with 128 localized `⠀ -> ⠲` and zero reverse. The cause was the \ + final `Letters)` token of a whitespace-split Roman parenthetical being rerouted as \ + math, which made the preceding Roman word terminate early and introduced an extra \ + blank. After excluding only a backwards-verified complete Roman parenthetical tail \ + from math routing, the cohort is 2,132 exact / 2,547 mismatch, with 23 target-localized \ + and zero reverse; corpus-wide exact likewise rises by 74, from 68,101 to 68,175. \ + The 105 removed localized boundaries include 74 newly exact cases and 31 cases that \ + still differ elsewhere. Remaining targets such as `SYNO PEM-1`, `ACE Fair(2020)`, \ + and `Mnet K-POP` do not satisfy the complete-parenthetical gate and remain separate \ + residuals rather than widening this rule.\n", + stats.candidates, + stats.exact, + stats.mismatch, + stats.first_difference_in_output_signature, + stats.output_signature_mismatches_evaluated + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(SINGLE_CAPITAL_PARENTHESIZED_DIGITS) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent single-capital parenthesized-digits measurement: {} candidates, {} \ + exact controls, {} mismatches, {pending} members in the actual \ + `pending_rule_review` subcluster, and {}/{} evaluable mismatches whose first \ + difference is inside the target run plus its entry boundary. No engine change is \ + inferred from the ambiguous prose/function surface form.\n", + stats.candidates, + stats.exact, + stats.mismatch, + stats.first_difference_in_output_signature, + stats.output_signature_mismatches_evaluated + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(MIXED_ROMAN_KOREAN_BEFORE_HEADWORD_EXPANSION) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent mixed Roman+Korean boundary before uppercase headword-expansion \ + measurement: {} candidates, {} exact controls, {} mismatches, {pending} members in \ + the actual `pending_rule_review` subcluster, and {}/{} evaluable mismatches whose \ + first difference is localized to the later headword's entry boundary/output. The \ + detector cannot be satisfied by the earlier Roman entry. The narrow rules-29/34 \ + headword-expansion route is active; these residuals therefore identify a separate \ + state or Roman-letter difference.\n", + stats.candidates, + stats.exact, + stats.mismatch, + stats.first_difference_in_output_signature, + stats.output_signature_mismatches_evaluated + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(COMPACT_NUMERIC_ASCII_SUFFIX) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent compact numeric+ASCII-suffix measurement: {} candidates, {} exact \ + controls, {} mismatches, {pending} members in the actual `pending_rule_review` \ + subcluster, and {}/{} evaluable mismatches whose first difference is inside the \ + complete current-engine output signature or its immediate entry boundary. Rule 40 \ + requires the numeric indicator and rule 69 requires Roman indicators around a \ + Roman-written unit, but the input \ + shape alone cannot prove that every ASCII suffix is a unit.\n\n", + stats.candidates, + stats.exact, + stats.mismatch, + stats.first_difference_in_output_signature, + stats.output_signature_mismatches_evaluated + )); + + let mut suffixes = report + .compact_numeric_ascii_suffixes + .iter() + .collect::>(); + suffixes.sort_by(|(left_key, left), (right_key, right)| { + right + .candidates + .cmp(&left.candidates) + .then_with(|| left_key.cmp(right_key)) + }); + text.push_str( + "| ASCII suffix | Candidates | Exact | Mismatch | Localized first diff |\n\ + |---|---:|---:|---:|---:|\n", + ); + for (suffix, suffix_stats) in suffixes.into_iter().take(25) { + text.push_str(&format!( + "| `{suffix}` | {} | {} | {} | {} |\n", + suffix_stats.candidates, + suffix_stats.exact, + suffix_stats.mismatch, + suffix_stats.first_difference_in_output_signature + )); + } + text.push_str( + "\nEntry-boundary pre-fix baseline: 2,975 candidates, 1,649 exact controls, 1,326 \ + mismatches, 1,259 pending members, and 356 localized first differences; the \ + dominant reference number-sign versus current space transition accounted for 296 \ + cases. Rules 68 and 69 already accept semantic Unicode compatibility-unit forms. \ + The implementation derives compact ASCII spellings only from their all-letter NFKC \ + decompositions, reuses the owning rule's PDF-defined cells, chooses the longest \ + complete spelling, and rejects partial suffix matches. It does not extend \ + recognition to separated English words or arbitrary corpus suffixes. The same \ + cohort now has 1,759 exact controls, 1,216 mismatches, 1,148 pending members, and \ + 261 localized first differences; corpus-wide exact matches moved from 66,436 to \ + 66,546 (+110). Rule 69's printed `160㎎/㎗` example directly controls `160mg`; the \ + additional full-encoder `240mg`/`240㎎` pair proves that recognition is invariant \ + to the numeric value and also passes. Rule 68's printed `10,000㎡는 1㏊이다` example \ + controls the `ha` spelling through U+33CA's NFKC decomposition; `15.2ha`/`15.2㏊` \ + now passes without a spelling-specific output branch. The 53-case `ha` suffix \ + control moved from 0 exact to 18 exact; its remaining 35 mismatches have independent \ + later or surrounding differences. A full U+3300..U+33FF owner audit found 73 \ + Rules 68/69 glyphs with pure-ASCII NFKC decompositions, 73 distinct spellings, and \ + therefore no current duplicate-spelling owner collision. Production nevertheless \ + groups every owner before resolution and excludes a spelling if any owner cells \ + differ; a synthetic collision test proves that this is not first-wins behavior. An \ + exhaustive test also compares every one of the 73 derived spellings with every \ + owner glyph at both the unit-cell and full-encoder boundaries in a neutral Korean \ + measurement context. That audit exposed \ + the pre-existing explicit `cal` mapping's missing ordinary terminator; rule 69 \ + requires the terminator, while the existing slash-boundary function removes it for \ + the printed `cal/㎠/min` context. Restoring the ordinary terminator and extending \ + that slash-continuation boundary made the audit pass without changing the \ + corpus-wide 66,546 exact total. Ambiguous pure-English inputs remain on UEB: the \ + standard controls `3m` and `4.m` are retained and pass, rather than being globally \ + forced into a Korean unit route.\n", + ); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(RULE69_ASCII_UNIT_BEFORE_TERMINATOR_SKIPPING_SYMBOL) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent rule-69 ASCII-unit punctuation-boundary measurement: {} candidates, {} \ + exact controls, {} mismatches, {pending} members in the actual \ + `pending_rule_review` subcluster, and {}/{} evaluable mismatches whose first \ + difference is localized to the unit-plus-punctuation output signature. PDF rule 69 \ + requires a Roman terminator after a Roman-written unit in the ordinary case, while \ + rules 33/34 omit it at the listed punctuation or enclosing-mark boundary; the rule \ + 46 PDF example `체중(kg)` is the minimal parenthesized-unit control. Membership is \ + restricted to rule-69 spellings already supported by the engine and does not infer \ + unit semantics for arbitrary ASCII suffixes.\n", + stats.candidates, + stats.exact, + stats.mismatch, + stats.first_difference_in_output_signature, + stats.output_signature_mismatches_evaluated + )); + text.push_str( + " Analyzer pre-fix baseline: 440 candidates, 0 exact controls, 440 mismatches, 435 \ + pending members, and 9 signature-local first differences. After the generalized \ + rule-33/34 boundary override and the matching non-math routing guard, the same cohort \ + has 325 exact controls and 115 mismatches. Corpus-wide exact matches moved from \ + 66,039 to 66,436 (+397); the additional gains are applications of the same boundary \ + rule outside this strict ASCII detector, including compatibility-unit forms. The \ + complete standard suite remains 5,141/5,141.\n", + ); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(DECIMAL_POINT_BETWEEN_DIGITS) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent decimal-point measurement: {} candidates, {} exact controls, {} \ + mismatches, {pending} members in the actual `pending_rule_review` subcluster, and \ + {}/{} evaluable mismatches whose first difference is inside the complete \ + decimal-containing word's current-engine signature. Hangeul rules 43 and 48 keep \ + an ASCII point between digits in the numeric sequence and encode it as the decimal \ + point; rules 35 and 69 supply controls for adjacent Roman-number chains and Roman \ + units. This is an implementation-candidate audit, not permission to specialize on \ + a corpus reference.\n", + stats.candidates, + stats.exact, + stats.mismatch, + stats.first_difference_in_output_signature, + stats.output_signature_mismatches_evaluated + )); + text.push_str( + "Analyzer checkpoint `ed20f99` recorded the pre-fix baseline as 4,546 candidates, \ + 2,584 exact controls, 1,962 mismatches, 1,905 pending members, and 875 localized \ + first differences. Its dominant localized transition was reference decimal point \ + `U+2832 ⠲` versus current Roman indicator `U+2834 ⠴` in 647 cases. After the \ + generalized rule-43/48 guard, that transition is absent: 525 cases become exact \ + and the other corrected prefixes expose later independent mismatches.\n", + ); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(ALLCAPS_ROMAN_RUN_CONTAINING_OU) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent all-caps `OU` measurement: {} candidates, {} exact controls, {} \ + mismatches, {pending} members in the actual `pending_rule_review` subcluster, and \ + {}/{} evaluable mismatches whose first difference is inside the current-engine \ + output signature for the detected run. This is a pronunciation-sensitive UEB \ + review cohort, not an engine routing rule.\n", + stats.candidates, + stats.exact, + stats.mismatch, + stats.first_difference_in_output_signature, + stats.output_signature_mismatches_evaluated + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(STANDALONE_UPPERCASE_ROMAN_WORD) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent standalone-uppercase measurement: {} candidates, {} exact controls, {} \ + mismatches, and {pending} members in the actual `pending_rule_review` subcluster. \ + Its high frequency does not make it causal: the same input shape is exact in many \ + cases, and a sentence containing the shape may first differ at another Roman, \ + numeric, or punctuation structure. No engine change is inferred from this cohort.\n", + stats.candidates, stats.exact, stats.mismatch + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(KOREAN_PREFIXED_ALLCAPS_PARENTHETICAL) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent Korean-prefixed all-caps parenthetical measurement: {} candidates, {} \ + exact controls, {} mismatches, and {pending} members in the actual \ + `pending_rule_review` subcluster. This is a semantic-collision audit, not an engine \ + routing rule; no implementation change is inferred from its reference outputs.\n", + stats.candidates, stats.exact, stats.mismatch + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(KOREAN_PREFIXED_CLOSED_ROMAN_ANNOTATION) + { + let rule_34_reference_conflicts = report + .reasons + .get("rule34_roman_indicator_before_opening_parenthesis") + .copied() + .unwrap_or(0); + let opposite_order = stats + .first_difference_in_output_signature_transitions + .get("U+2834 ⠴ -> U+2826 ⠦") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent rule-34 opening-order measurement: {} structural candidates, {} exact \ + controls, {} mismatches, and {}/{} evaluable mismatches whose first difference is \ + inside the current engine's Korean opening-parenthesis cells. Only the \ + {opposite_order} localized first-cell transitions have the reference/current order \ + `⠴` versus `⠦`. After requiring the complete reference `⠴⠐⠣` versus current/PDF \ + `⠦⠄⠴` three-cell signature and preserving higher-priority comparison \ + classifications, {rule_34_reference_conflicts} are classified with the dedicated \ + rule-34 contradiction reason; mere \ + coexistence with a Korean-prefixed Roman annotation does not change a primary \ + class.\n", + stats.candidates, + stats.exact, + stats.mismatch, + stats.first_difference_in_output_signature, + stats.output_signature_mismatches_evaluated + )); + } + text.push_str( + "\nNIKL Q&A #325 clarifies that the six lower wordsigns named by Korean Rule 37 \ + remain expanded when Roman words are discussed in Korean context, while a recognizable \ + English title or phrase follows UEB 10.5 and uses a lower wordsign only when it stands \ + alone and satisfies the lower-sign adjacency restriction. The encoder distinguishes \ + those contexts from input structure, capitalization, and enclosure boundaries; analyzer \ + references and competitor fields do not affect routing.\n", + ); + let nonstanding_parenthesis_grade1_reference_conflicts = report + .reasons + .get("ueb_grade1_before_nonstanding_opening_parenthesis") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent UEB non-standing parenthesis/grade-1 contradiction measurement: \ + {nonstanding_parenthesis_grade1_reference_conflicts} cases contain an all-capitals \ + letters-sequence that resembles a shortform but is followed immediately by an opening \ + round, square, or curly parenthesis. UEB 2.6.2 permits those opening symbols before a \ + standing-alone sequence, while 2.6.3 does not permit them after one. The classifier \ + requires complete-sentence equality after removing only reference-side grade-1 cells \ + immediately before the localized capitals indicators; all other differences remain \ + pending review.\n" + )); + let capitalized_passage_reference_conflicts = report + .reasons + .get("ueb_capitalized_passage_written_as_separate_capital_words") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent UEB capitalized-passage contradiction measurement: \ + {capitalized_passage_reference_conflicts} cases contain at least three consecutive \ + capitalized symbols-sequences and differ from the current UEB 8.5.2-8.5.3 path only \ + by replacing the one passage indicator/terminator pair with separate one- or two-cell \ + capitalization indicators. The classifier requires equality of the complete sentence \ + after deleting exactly those structurally counted separate indicators; unrelated \ + Roman, punctuation, contraction, or spacing differences remain pending review.\n" + )); + if let Some(stats) = report + .pending_rule_review_clusters + .get(ALLCAPS_ROMAN_MIDDLE_DOT_RUNS) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent all-caps Roman middle-dot measurement: {} candidates, {} exact controls, \ + {} mismatches, and {pending} members in the actual `pending_rule_review` subcluster. \ + This cross-cutting cohort preserves every primary class and is not an engine routing \ + rule.\n", + stats.candidates, stats.exact, stats.mismatch + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(ROMAN_RUN_BEFORE_MIDDLE_DOT_BOUNDARY) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent Roman-before-middle-dot boundary measurement: {} candidates, {} exact \ + controls, {} mismatches, {pending} members in the actual `pending_rule_review` \ + subcluster, and {}/{} evaluable mismatches whose first difference is localized to \ + the one current terminator immediately before the middle dot. The locator encodes \ + each real input prefix ending at the dot, so identifier state such as `K-ICS·...` \ + is retained without consulting expected. This raises localized target coverage from \ + 342 to 417 and reduces the then-leading residual `⠐ -> ⠲` transition from 297 to \ + 223. Rules 29, 33, and 50 support the current terminator path but do not support the \ + localized reference omission; no engine change or primary-class rewrite is \ + inferred.\n", + stats.candidates, + stats.exact, + stats.mismatch, + stats.first_difference_in_output_signature, + stats.output_signature_mismatches_evaluated + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(ATTACHED_ASCII_ROMAN_TO_KOREAN_BOUNDARY) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + let target = stats + .first_difference_in_output_signature_transitions + .get("U+2832 ⠲ -> U+2838 ⠸") + .copied() + .unwrap_or(0); + let reverse = stats + .first_difference_in_output_signature_transitions + .get("U+2838 ⠸ -> U+2832 ⠲") + .copied() + .unwrap_or(0); + let exact_terminator = stats + .actual_output_signature_outcomes + .get("exact:rule29_terminator") + .copied() + .unwrap_or(0); + let exact_hangul_opening = stats + .actual_output_signature_outcomes + .get("exact:rule39_hangul_opening") + .copied() + .unwrap_or(0); + let mismatch_hangul_opening = stats + .actual_output_signature_outcomes + .get("mismatch:rule39_hangul_opening") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent attached Roman-to-Korean boundary measurement: {} candidates, {} exact \ + controls, {} mismatches, {pending} members in the actual `pending_rule_review` \ + subcluster, and {}/{} evaluable mismatches whose first difference is localized to \ + the current boundary marker. The localized target `⠲ -> ⠸` occurs {target} times \ + and the reverse `⠸ -> ⠲` occurs {reverse} times. Korean rule 29 (2024 Korean-rules \ + PDF p.26, printed p.20) requires a Roman terminator around Roman text in a Korean \ + sentence, whereas rule 39 (PDF pp.31-32, printed pp.25-26) applies the Korean \ + opening/closing markers only when Roman text is the sentence's main \ + language. Among exact candidates, {exact_terminator} expose the current rule-29 \ + terminator at such a boundary and {exact_hangul_opening} expose the current rule-39 \ + opening. Its printed controls include both an English sentence (`What is 김치 in \ + English?`) and the address `www.대통령.kr` inside a Korean sentence. Official UEB \ + 2.4.7 (2024 UEB PDF p.41, printed p.13) confirms that a UEB mode does not extend \ + through a switch to another braille code, but does not choose whether this Korean \ + boundary belongs to a Korean-main or Roman-main context. Therefore the \ + attached script boundary alone cannot distinguish ordinary Korean prose from an \ + embedded Roman-domain context. Primary classes are preserved and no engine change \ + is inferred without a narrower input-derived dominance gate. At diagnostic \ + checkpoint `57c4608`, this same cohort was 17,693 candidates / 12,724 exact / 4,969 \ + mismatch, with 268 localized `⠲ -> ⠸`, zero reverse, 10,780 exact rule-29 markers, \ + and zero exact rule-39 markers. After narrowing the engine by dominance and the PDF \ + domain shape, it is 12,924 exact / 4,769 mismatch with no localized target or \ + reverse; corpus-wide exact rises from 67,715 to 68,101 (+386). The remaining \ + {mismatch_hangul_opening} mismatch with a current rule-39 opening is a list-heavy \ + Korean sentence whose ASCII-leading brand tokens satisfy the mechanical word-count \ + majority; its first difference is not localized to this boundary, so it remains a \ + semantic pending control rather than grounds for another engine branch.\n", + stats.candidates, + stats.exact, + stats.mismatch, + stats.first_difference_in_output_signature, + stats.output_signature_mismatches_evaluated + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(KOREAN_MAJORITY_ROMAN_SANDWICH_NON_DOMAIN) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent rule-39 narrowed-scope audit: {} candidates, {} exact controls, {} \ + mismatches, and {pending} members in the actual `pending_rule_review` subcluster. \ + This input-derived scope is recorded separately from the direct-boundary \ + output-localizer; primary classes are unchanged. At the implementation checkpoint, \ + this conservative first-script-word approximation contains 385 exact cases, \ + accounting for all but one of the corpus-wide +386 net gain; no localized reverse \ + transition is observed in the direct-boundary cohort.\n", + stats.candidates, stats.exact, stats.mismatch + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(KOREAN_INLINE_PARENTHESIZED_OPERATOR) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent inline parenthesized-operator measurement: {} candidates, {} exact \ + controls, {} mismatches, {pending} members in the actual `pending_rule_review` \ + subcluster, and {}/{} evaluable mismatches whose first differing cell is inside the \ + emitted structure. Primary classes are preserved.\n", + stats.candidates, + stats.exact, + stats.mismatch, + stats.first_difference_in_output_signature, + stats.output_signature_mismatches_evaluated + )); + text.push_str( + " At that implementation checkpoint, the strict cohort moved from 0 to 17 exact \ + cases; the corpus-wide total moved from 65,491 to 65,514 (+23 exact) because the same \ + PDF-backed spacing rule also applied outside the stricter Korean-boundary audit \ + gate. These are immutable checkpoint counts rather than the report's later cumulative \ + total. The complete standard suite remained 5,141/5,141.\n", + ); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(ATTACHED_PLUS_PARENTHESIZED_KOREAN_GLOSS) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent attached plus + parenthesized Korean-gloss measurement: {} candidates, \ + {} exact controls, {} mismatches, {pending} members in the actual \ + `pending_rule_review` subcluster, and {}/{} evaluable mismatches whose first \ + differing cell is inside the current emitted structure. Hangeul rule 46 supplies \ + the operation-sign spacing control, but the surface form alone does not establish \ + whether a brand or program name uses `+` mathematically. Exact and localized \ + mismatch references coexist for `도전+(플러스)`, so no engine change or primary \ + reclassification is inferred.\n", + stats.candidates, + stats.exact, + stats.mismatch, + stats.first_difference_in_output_signature, + stats.output_signature_mismatches_evaluated + )); + } + if let Some(stats) = report + .pending_rule_review_clusters + .get(TIGHT_TRIANGLE_BEFORE_KOREAN) + { + let pending = stats + .mismatch_primary_classes + .get("pending_rule_review") + .copied() + .unwrap_or(0); + text.push_str(&format!( + "\nCurrent tight-triangle measurement: {} candidates, {} exact controls, {} \ + mismatches, {pending} members in the actual `pending_rule_review` subcluster, and \ + {}/{} evaluable mismatches whose first difference is inside the `△` plus first-Korean \ + output range. No engine change is inferred.\n", + stats.candidates, + stats.exact, + stats.mismatch, + stats.first_difference_in_output_signature, + stats.output_signature_mismatches_evaluated + )); + } + + text.push_str("\n## Encoding-error diagnostics\n\n"); + text.push_str( + "The audit starts from all raw encoding errors, then separates cases already resolved by \ + a comparison method or corpus contradiction. The message, family, and singleton tables \ + below count only unresolved encoding-error review cases. These diagnostics are not \ + additional primary classes. \ + A singleton unsupported character is a character that also fails when encoded by itself. \ + Such a failure remains a review candidate until the PDF independently establishes support.\n\n", + ); + text.push_str("| Encoding-error audit | Cases |\n|---|---:|\n"); + text.push_str(&format!( + "| Raw encoding errors | {} |\n", + report.encoding_error_audit.raw_total + )); + text.push_str(&format!( + "| Resolved by comparison method | {} |\n", + report.encoding_error_audit.resolved_by_comparison_method + )); + text.push_str(&format!( + "| Excluded as corpus suspect | {} |\n", + report.encoding_error_audit.excluded_as_corpus_suspect + )); + text.push_str(&format!( + "| Unresolved encoding-error review cases | {} |\n", + report.encoding_error_audit.unresolved_review_total + )); + text.push_str(&format!( + "| Explained by singleton unsupported character(s) | {} |\n", + report + .encoding_error_audit + .explained_by_singleton_unsupported + )); + text.push_str(&format!( + "| Multiple singleton unsupported characters | {} |\n", + report.encoding_error_audit.multiple_singleton_unsupported + )); + text.push_str(&format!( + "| Unclassified without a singleton explanation | {} |\n\n", + report.encoding_error_audit.unclassified_without_singleton + )); + for sample in &report.encoding_error_audit.multiple_singleton_samples { + text.push_str(&format!( + "- compound `{}` #{}: {}\n - singleton unsupported: `{}`\n", + sample.shard, + sample.index, + sample.input.chars().take(180).collect::(), + sample.unsupported_characters.join(", ") + )); + } + for sample in &report.encoding_error_audit.unclassified_samples { + text.push_str(&format!( + "- unclassified `{}` #{}: {} (`{}`)\n", + sample.shard, + sample.index, + sample.input.chars().take(180).collect::(), + sample.error + )); + } + if !report + .encoding_error_audit + .multiple_singleton_samples + .is_empty() + || !report.encoding_error_audit.unclassified_samples.is_empty() + { + text.push('\n'); + } + text.push_str("| Error message | Cases |\n|---|---:|\n"); + for (name, count) in &report.encoding_error_messages { + text.push_str(&format!("| `{name}` | {count} |\n")); + } + text.push_str("\n| Error family | Cases |\n|---|---:|\n"); + for (name, count) in &report.encoding_error_families { + text.push_str(&format!("| `{name}` | {count} |\n")); + } + text.push_str( + "\nFamilies are diagnostics, not automatic normalization permissions. \ + Rules 68/69 compatibility-unit support removed that error family from the current run; \ + `enclosed_organization_mark` and layout symbols still have no confirmed rule.\n\n", + ); + text.push_str( + "| Singleton error character | Cases containing it | NFKC decomposition | Family |\n\ + |---|---:|---|---|\n", + ); + for (name, stats) in &report.singleton_error_characters { + text.push_str(&format!( + "| `{name}` | {} | `{}` | `{}` |\n", + stats.cases, + stats.nfkc.replace('`', "\\`"), + stats.family + )); + } + + text.push_str("\n## Shards\n\n| Shard | Exact | Total | Accuracy |\n|---|---:|---:|---:|\n"); + for (name, stats) in &report.shards { + text.push_str(&format!( + "| `{name}` | {} | {} | {:.2}% |\n", + stats.exact, + stats.total, + stats.exact as f64 / stats.total as f64 * 100.0 + )); + } + + text.push_str("\n## Overlapping mismatch traits\n\n| Trait | Count |\n|---|---:|\n"); + for (name, count) in &report.overlapping_traits { + text.push_str(&format!("| `{name}` | {count} |\n")); + } + + text.push_str("\n## Samples\n\n"); + for (reason, samples) in &report.samples { + text.push_str(&format!("### `{reason}`\n\n")); + for sample in samples { + let input = sample.input.chars().take(180).collect::(); + text.push_str(&format!( + "- `{}` #{}: {}\n - expected: `{}`\n - actual: `{}`{}\n", + sample.shard, + sample.index, + input.replace('`', "\\`"), + sample.expected_excerpt, + sample.actual_excerpt, + sample + .error + .as_ref() + .map_or_else(String::new, |error| format!("\n - error: `{error}`")) + )); + } + text.push('\n'); + } + + text.push_str("## PDF-derived state gates\n\n"); + text.push_str( + "The rule 37 example `그는 Can you help me?라고 도움을 요청했다.` distinguishes the first \ + word after the roman indicator (`Can`, whose whole-word sign is suppressed) from the \ + interior word `you` in the uninterrupted ASCII phrase (whose UEB wordsign is retained). \ + The `prev_is_ascii_word && next_is_ascii_word` gate expresses that phrase-interior \ + position rather than matching an input string.\n\n\ + The rule 39 example `What is 김치 in English?` resumes the surrounding English passage \ + after the Korean span. The `english_dominant_wrap_active` gate therefore retains the UEB \ + wordsign for the resumed `in`, instead of treating it as a fresh rule 37 entry word.\n\n", + ); + + text.push_str("## Rule 69 compatibility-unit scope\n\n"); + text.push_str( + "The engine accepts 96 scientific/measurement glyphs from Unicode CJK Compatibility, \ + derives their Roman spelling with NFKC, and applies rules 68/69 rather than whole-word \ + UEB. The accepted glyph set and panic-free encoding property are fixed by inline tests. \ + The official Unicode names distinguish `U+337A ㍺` SQUARE IU (accepted) from \ + `U+33D1 ㏑` SQUARE LN, `U+33D2 ㏒` SQUARE LOG, and `U+33DA ㏚` SQUARE PR \ + (not units, rejected). See the \ + [Unicode CJK Compatibility names list](https://www.unicode.org/charts/nameslist/n_3300.html).\n\n", + ); + + text.push_str("## Rule 36 Roman-numeral presentation forms\n\n"); + text.push_str( + "Rule 36 says that a Roman numeral is written with the corresponding Roman letters. \ + The encoder therefore applies compatibility decomposition only to Unicode Roman \ + Numerals U+2160–U+217F and sends the ASCII spelling through the existing rule-36 \ + algorithm. Encoder regressions compare Unicode presentations with ASCII equivalents \ + in the PDF sentence and in attached-Korean, particle-adjacent, and lower-case contexts. \ + U+2180 `ↀ` and unrelated NFKC characters such as `㈜` are explicit non-targets.\n\n", + ); + text.push_str( + "The transition audit reconstructs the immediately preceding engine behavior: direct \ + and NFC encoding rejected U+2160–U+217F, while the analyzer's existing NFKC comparison \ + path already used the same ASCII Roman spelling. This avoids a saved-output lookup and \ + keeps the transition reproducible from the current corpus.\n\n", + ); + text.push_str(&format!( + "Presentation-form cases audited: {}.\n\n", + report.rule_36_transition_audit.presentation_cases + )); + text.push_str("| Previous observation → current observation | Cases |\n|---|---:|\n"); + for (transition, count) in &report.rule_36_transition_audit.observed_transitions { + text.push_str(&format!("| `{transition}` | {count} |\n")); + } + text.push_str(&format!( + "\nRemaining complex encoding errors: {}. These cases still contain another character \ + that fails independently, so disappearance of the `roman_numeral_presentation` family \ + does not imply that every former error case now encodes successfully.\n\n", + report.rule_36_transition_audit.remaining_complex_errors + )); + for sample in &report + .rule_36_transition_audit + .remaining_complex_error_samples + { + let input = sample.input.chars().take(180).collect::(); + let unsupported = if sample.other_unsupported_characters.is_empty() { + "none detected".to_string() + } else { + sample.other_unsupported_characters.join(", ") + }; + text.push_str(&format!( + "- `{}` #{}: {}\n - other independently unsupported: `{}`\n", + sample.shard, + sample.index, + input.replace('`', "\\`"), + unsupported + )); + } + text.push('\n'); + + text.push_str("## Rules 34/54 Korean-prefixed Roman annotations\n\n"); + text.push_str( + "Rule 34 says that when Roman text is enclosed by quotation marks or brackets, the \ + Roman terminator is omitted; its PDF example is `링컨(Lincoln)은 미국의 제16대 \ + 대통령이다.` The example's cells put the printed Korean opening parenthesis \ + (`⠦⠄`) before the Roman indicator (`⠴`). Rule 54 says that text immediately after an opening bracket and \ + immediately before a closing bracket is attached. Together these establish the \ + Korean-prefix + closed-Roman-annotation context independently of corpus expected \ + values. A following comma or period is outside the already closed annotation and \ + must not cause its Roman contents to be rerouted as mathematics.\n\n\ + The implementation gate exists only inside `split_mixed_math_word`, after the prefix \ + has been proved entirely Korean. It accepts a fully closed parenthesized Roman word \ + (including ASCII digits such as `O4O`) plus ordinary trailing prose punctuation. \ + The corpus audit treats the opposite localized reference prefix (`⠴⠐⠣`, Roman \ + indicator plus UEB opening parenthesis) as a data-reference contradiction only when \ + all three cells and the real input position agree. Broad sentence-level coexistence is \ + retained as an exact or existing-primary control. This audit does not alter engine \ + routing. The global math detector is byte-for-byte unchanged; regression tests preserve its \ + existing standalone results for `(x)`, `(A)`, and `(abc)`, while explicit forms such \ + as `(x+1)`, `(a/b)`, and `(x₁)` remain math candidates.\n\n\ + Against the immediately preceding 63,399-exact run, exact matches increased by 2,092. \ + The observable primary totals changed as follows: `comparison_method` 290→303, \ + `pending_rule_review` 19,636→17,543, and `unsupported_character_review` 203→191. \ + Raw encoding errors stayed at 450; errors resolved by a comparison method changed \ + 247→259 and unresolved review errors changed 203→191.\n\n", + ); + + text.push_str("## Rule evidence and change log\n\n"); + text.push_str( + "| Stage | Standard cases | Corpus exact | Corpus accuracy | Evidence |\n\ + |---|---:|---:|---:|---|\n\ + | Parent commit `3cfeae0` | 5,141/5,141 | 57,732/83,528 | 69.12% | Reproduced with release tests |\n\ + | Rules 28/29 indicator ordering | 5,141/5,141 | 61,652/83,528 | 73.81% | Roman indicator now precedes UEB grade-1/capital indicators; rule 35 roman-number continuity retained |\n\ + | Rules 37/39 shared UEB groupsign algorithm | 5,141/5,141 | 63,239/83,528 | 75.71% | Korean Roman sections reuse UEB preference/morphology rules; entry wordsigns and English-dominant resume are state-gated |\n", + ); + text.push_str( + "| Rules 68/69 compatibility unit algorithm | 5,141/5,141 | 63,388/83,528 | 75.89% | 96 Unicode unit presentation forms use one decomposition/letter-run/attachment algorithm; encoding errors fell from 434 to 226 |\n", + ); + text.push_str( + "| Rule 36 Unicode Roman-numeral presentation normalization | 5,141/5,141 | 63,399/83,528 | 75.90% | U+2160–U+217F use the corresponding Roman-letter spelling; 11 NFKC-equivalent observations became exact, 23 errors became encoded mismatches pending review, and 3 remain blocked by `㈜` |\n", + ); + text.push_str( + "| Rules 34/54 Korean-prefixed closed Roman annotation routing | 5,141/5,141 | 65,491/83,528 | 78.41% | A fully closed Roman annotation after an all-Korean prefix stays on the prose encoder path, including attached comma/period and alphanumeric forms such as `O4O`; exact matches increased by 2,092 while the global math detector remained unchanged |\n", + ); + text.push_str( + "| Rules 43/48 decimal-point ownership | 5,141/5,141 | 66,039/83,528 | 79.06% | A period directly between ASCII digits remains on the numeric punctuation path even when its word or sentence also contains Roman text; 647 localized Roman-entry differences were removed and 525 cases became exact |\n", + ); + text.push_str( + "| Rules 33/34/69 Roman-unit punctuation boundary | 5,141/5,141 | 66,436/83,528 | 79.54% | Rule-69 units retain their ordinary terminator at end/Korean/slash boundaries but omit it before rule-33/34 punctuation or enclosing marks; compact unit tokens with that boundary stay off the math path; 397 cases became exact |\n", + ); + text.push_str( + "| Rules 68/69 compact compatibility-derived ASCII units | 5,141/5,141 | 66,546/83,528 | 79.67% | Compact ASCII unit spellings are derived from the engine's already accepted Unicode compatibility-unit forms and reuse their owning-rule cells, with longest-complete matching and no expansion to separated English words; `160mg`, numeric-invariance control `240mg`, and Rule-68 `ha` controls are retained; 110 cases became exact |\n", + ); + text.push_str( + "| UEB numeric-mode letter classes in Roman identifiers | 5,141/5,141 | 67,012/83,528 | 80.23% | Lowercase `a`-`j` retains grade 1 after digits, capitals use capitalization, and lowercase `k`-`z` needs no extra indicator; numeric-leading Rule-69 units remain separate |\n\ + | UEB complete all-caps segments across hyphen | 5,141/5,141 | 67,138/83,528 | 80.38% | The grade-1 restart is omitted only between a complete uppercase prefix and an uppercase suffix of at least two letters; mixed/single-capital and digit-hyphen controls remain unchanged |\n\ + | Rule-29 consecutive Roman entry idempotence | 5,141/5,141 | 67,442/83,528 | 80.74% | An explicit Roman-entry event is ignored only when final emit state is already inside the same Roman section |\n\ + | Rules 29/71 complete attached Roman ampersand run | 5,141/5,141 | 67,715/83,528 | 81.07% | Complete ASCII-letter segments joined by `&` retain one Roman section; official `AT&T`, `B&B`, and spaced Korean Rule-71 controls delimit the safe boundary; 273 cases became exact |\n\ + | Rules 29/39 Roman-to-Korean mode ownership | 5,141/5,141 | 68,101/83,528 | 81.53% | Same-token Korean is wrapped only in a Roman-majority document or the official dot-delimited `www.대통령.kr` domain shape; all three rule-39 PDF controls remain exact and 386 corpus cases became exact |\n\ + | Rules 29/34 multiword Roman parenthetical continuity | 5,141/5,141 | 68,175/83,528 | 81.62% | A backwards-verified, closed letter-and-space Roman parenthetical tail stays on the prose route; function calls, digits, operators, nested brackets, and punctuation-separated forms remain outside; 74 cases became exact |\n\ + | Rules 29/32/71 ampersand before attached Roman segment | 5,141/5,141 | 68,187/83,528 | 81.63% | Official UEB `&c`, `AT&T`, and `B&B` keep one Roman section across attached `&`; the spaced Korean Rule-71 example, left ASCII alphanumerics, repeated ampersands, and trailing digits delimit the gate; 12 cases became exact |\n\ + | UEB 8.4.2 capitals-word nonletter boundary | 5,141/5,141 | 69,291/83,528 | 82.96% | Starting from the accepted 68,439-exact checkpoint, capitals-word handling ends at each nonletter and later uppercase runs restart independently; official `AT&T`, `B&B`, `MP3`, and `TVOntario` delimit the token/span boundary; 852 cases became exact and the complete exact-ID audit found zero former exact cases lost |\n\ + | UEB 8.4.2 same-token internal Roman apostrophe | 5,141/5,141 | 69,359/83,528 | 83.04% | A straight apostrophe stays in the Roman section only with immediate same-token ASCII letters on both sides, while capitals mode restarts for an uppercase suffix; official `O'Hara`, `DON'T`, `THAT'S`, and `SHE'LL` plus detached quote and measurement controls delimit the gate; 68 cases became exact and the complete exact-ID audit found zero former exact cases lost |\n\ + | Rules 29/34 all-caps headword with closed multiword Roman expansion | 5,141/5,141 | 69,389/83,528 | 83.07% | A complete two-or-more-capital headword followed by a closed expansion of at least two ASCII-letter words stays on the prose route; digits, operators, nesting, scripts, and alphanumeric trailers remain math controls; the cohort's exact count rose 17→47 with 30 corpus-wide gains |\n\ + | Rule 34 Korean trailer after closed multiword Roman expansion | 5,141/5,141 | 69,430/83,528 | 83.12% | Rule 34's `링컨(Lincoln)은` establishes that attached Korean text after `)` remains prose; applying the same boundary to closed multiword expansions adds 41 corpus-wide exact matches and raises the headword cohort 47→87, while ASCII-letter and digit trailers remain excluded |\n", + ); + text.push_str( + "\nThe latest full `cargo test -p braillify test_by_testcase --release -- --nocapture` \ + run was accepted from its custom testcase summary, not the trailing filtered harness: \ + `총 테스트 케이스: 5141`, `성공: 5141`, `실패: 0`, and \ + `Skip (limitation): 0`.\n\n\ + Engine changes must add a row only after both the 5,141-case standard suite and \ + this full analysis have been rerun. Suspect-reference clusters stay in this report; \ + they are not engine targets without independent PDF evidence.\n", + ); + text +} + +fn write_file(path: &Path, contents: &str) -> Result<(), String> { + if let Some(parent) = path.parent() { + fs::create_dir_all(parent) + .map_err(|error| format!("cannot create {}: {error}", parent.display()))?; + } + fs::write(path, contents).map_err(|error| format!("cannot write {}: {error}", path.display())) +} + +fn run() -> Result<(), String> { + let started = Instant::now(); + let config = Config::parse()?; + let cases = load_cases()?; + let encoded = encode_cases(&cases, config.threads); + let report = analyze(cases, encoded, config.sample_limit); + let json = serde_json::to_string_pretty(&report) + .map_err(|error| format!("cannot serialize analysis JSON: {error}"))?; + write_file(&config.json_path, &json)?; + write_file(&config.report_path, &markdown(&report))?; + println!( + "NIKL corpus: {}/{} exact ({:.2}%), wall={:.3}s, report={}, json={}", + report.exact, + report.total, + report.exact_percent, + started.elapsed().as_secs_f64(), + config.report_path.display(), + config.json_path.display() + ); + Ok(()) +} + +fn main() { + if let Err(error) = run() { + eprintln!("nikl_corpus_analyze: {error}"); + std::process::exit(1); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn loads_all_sentence_corpus_shards_for_analysis() { + let cases = load_cases().expect("NIKL sentence shards must load"); + + assert_eq!(cases.len(), 83_528); + assert!( + cases + .iter() + .all(|located| located.shard.starts_with("sentence_") + && located.shard.ends_with(".json")) + ); + } + + #[rstest::rstest] + #[case::zero_shards(0, 0, Some("no NIKL corpus shards"))] + #[case::zero_cases(1, 0, Some("zero cases"))] + #[case::nonempty_corpus(4, 83_528, None)] + fn corpus_shape_must_be_nonempty( + #[case] shard_count: usize, + #[case] case_count: usize, + #[case] expected_error: Option<&str>, + ) { + let result = validate_corpus_shape(shard_count, case_count); + match expected_error { + Some(expected) => assert!(result.unwrap_err().contains(expected)), + None => assert_eq!(result, Ok(())), + } + } + + #[test] + fn singleton_cache_probes_each_distinct_corpus_character_once() { + let cases = ["㈜Aℓ", "㈜Bℓ"] + .into_iter() + .enumerate() + .map(|(index, input)| LocatedCase { + shard: "synthetic.json".to_string(), + index, + case: CorpusCase { + input: input.to_string(), + unicode: String::new(), + }, + }) + .collect::>(); + let mut calls = BTreeMap::::new(); + + let unsupported = singleton_unsupported_set_with(&cases, &mut |ch| { + *calls.entry(ch).or_insert(0) += 1; + matches!(ch, '㈜' | 'ℓ') + }); + + assert_eq!(unsupported, BTreeSet::from(['ℓ', '㈜'])); + assert_eq!(calls.len(), 4); + assert!(calls.values().all(|count| *count == 1)); + } + + #[test] + fn serializes_report_enum_keys_as_snake_case_strings() { + assert_eq!(enum_key!(&PrimaryClass::Exact), "exact"); + assert_eq!( + enum_key!(&Reason::UnsupportedCharacterReview), + "unsupported_character_review" + ); + } + + #[rstest::rstest] + #[case::compatibility_unit('㎏', "compatibility_unit_symbol")] + #[case::roman_numeral('Ⅱ', "roman_numeral_presentation")] + #[case::company_mark('㈜', "enclosed_organization_mark")] + #[case::layout_symbol('▲', "punctuation_or_layout_symbol")] + #[case::non_unit_square_log('㏒', "other_unsupported_symbol")] + fn clusters_encoding_error_characters(#[case] input: char, #[case] expected_family: &str) { + assert_eq!(encoding_error_family(input), expected_family); + } + + #[rstest::rstest] + #[case::nfkc_comparison_becomes_exact( + "Ⅲ", + "same", + Some("same"), + "same", + &[], + "nfkc_input_equivalent -> exact" + )] + #[case::encoding_error_becomes_pending( + "Ⅳ장", + "expected", + Some("different"), + "different", + &[], + "encoding_error -> encoded_mismatch_pending_rule_review" + )] + #[case::compound_encoding_error_remains( + "Ⅳ㈜", + "expected", + None, + "different", + &['㈜'], + "encoding_error -> unsupported_character_review" + )] + fn reconstructs_rule_36_observed_transition( + #[case] input: &str, + #[case] expected: &str, + #[case] actual: Option<&str>, + #[case] nfkc_actual: &str, + #[case] singleton_unsupported_characters: &[char], + #[case] expected_transition: &str, + ) { + let encoded = EncodedCase { + located: LocatedCase { + shard: "synthetic.json".to_string(), + index: 1, + case: CorpusCase { + input: input.to_string(), + unicode: expected.to_string(), + }, + }, + actual: actual.map_or_else( + || Err("another unsupported symbol".to_string()), + |value| Ok(value.to_string()), + ), + nfc_actual: None, + nfkc_actual: Some(Ok(nfkc_actual.to_string())), + singleton_unsupported_characters: singleton_unsupported_characters.to_vec(), + }; + + assert_eq!( + rule_36_observed_transition(&encoded), + Some(expected_transition) + ); + } + + #[rstest::rstest] + #[case::singleton_explained( + &['㈜'], + PrimaryClass::UnsupportedCharacterReview, + Reason::UnsupportedCharacterReview + )] + #[case::unclassified( + &[], + PrimaryClass::UnclassifiedEncodingErrorReview, + Reason::UnclassifiedEncodingErrorReview + )] + fn encoding_error_primary_requires_independent_pdf_evidence( + #[case] singleton_unsupported_characters: &[char], + #[case] expected_primary: PrimaryClass, + #[case] expected_reason: Reason, + ) { + let encoded = EncodedCase { + located: LocatedCase { + shard: "synthetic.json".to_string(), + index: 1, + case: CorpusCase { + input: "입력".to_string(), + unicode: "expected".to_string(), + }, + }, + actual: Err("encoding failed".to_string()), + nfc_actual: None, + nfkc_actual: None, + singleton_unsupported_characters: singleton_unsupported_characters.to_vec(), + }; + + assert_eq!( + classify(&encoded, &BTreeSet::new()), + (expected_primary, expected_reason) + ); + } + + #[test] + fn whitespace_normalization_does_not_change_braille_cells() { + assert_eq!(normalized_braille_whitespace("⠁ ⠃"), "⠁⠀⠃"); + } + + #[rstest::rstest] + #[case::different_cells("⠁", "⠃", 0, "U+2801 ⠁ -> U+2803 ⠃")] + #[case::expected_ended("", "⠃", 0, " -> U+2803 ⠃")] + #[case::actual_ended("⠁", "", 0, "U+2801 ⠁ -> ")] + fn formats_first_difference_transition_key( + #[case] expected: &str, + #[case] actual: &str, + #[case] index: usize, + #[case] transition: &str, + ) { + assert_eq!(cell_transition_key(expected, actual, index), transition); + } + + #[rstest::rstest] + #[case::korean_suffix("값은 3.14이다.", vec!["3.14이다."])] + #[case::roman_identifier("GPT-3.5보다", vec!["GPT-3.5보다"])] + #[case::unit_and_punctuation("구간(1.0km), 종료", vec!["구간(1.0km),"])] + #[case::multiple_points("주소 1.2.3 확인", vec!["1.2.3"])] + #[case::period_not_between_digits("제3. 항목", vec![])] + #[case::leading_decimal("값 .48", vec![])] + fn detects_decimal_words(#[case] input: &str, #[case] expected: Vec<&str>) { + let actual = decimal_word_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[test] + fn locates_decimal_word_in_current_korean_context_output() { + let input = "수치는 34.3리터(L)이다."; + let actual = braillify::encode_to_unicode(input).expect("decimal probe must encode"); + let ranges = korean_context_signature_ranges(input, &actual, &decimal_word_spans(input), 0); + + assert_eq!(ranges.len(), 1); + assert!(ranges[0].start < ranges[0].end); + assert!(ranges[0].end <= actual.chars().count()); + } + + #[rstest::rstest] + #[case::energy("용량 13GWh 규모", vec![("13GWh", "GWh")])] + #[case::distance("구간 0.73km", vec![("0.73km", "km")])] + #[case::mass("필로폰 968g 등", vec![("968g", "g")])] + #[case::ambiguous_variable("값 3x", vec![("3x", "x")])] + #[case::letter_prefix("GPT3 모델", vec![])] + #[case::alphanumeric_suffix("13GWh2", vec![])] + fn detects_compact_numeric_ascii_suffixes( + #[case] input: &str, + #[case] expected: Vec<(&str, &str)>, + ) { + let actual = compact_numeric_ascii_suffix_spans(input) + .into_iter() + .map(|span| { + ( + &input[span.start_byte..span.end_byte], + compact_numeric_ascii_suffix(span, input), + ) + }) + .collect::>(); + assert_eq!(actual, expected); + } + + #[test] + fn locates_compact_numeric_ascii_suffix_and_entry_boundary_in_current_output() { + let input = "용량은 13GWh 규모다."; + let actual = braillify::encode_to_unicode(input).expect("compact suffix probe must encode"); + let spans = compact_numeric_ascii_suffix_spans(input); + let ranges = korean_context_signature_ranges(input, &actual, &spans, 1); + let signature = korean_context_signature(&input[spans[0].start_byte..spans[0].end_byte]) + .expect("compact suffix signature must encode"); + let signature_start_byte = actual + .find(&signature) + .expect("current output must contain compact suffix signature"); + let signature_start = actual[..signature_start_byte].chars().count(); + + assert_eq!(ranges.len(), 1); + assert_eq!(ranges[0].start + 1, signature_start); + assert_eq!(ranges[0].end, signature_start + signature.chars().count()); + } + + #[rstest::rstest] + #[case::kilogram_parenthesis("상자(20kg)당", vec!["20kg)"])] + #[case::metre_quote("길이는 3m”라고", vec!["3m”"])] + #[case::ordinary_unit_boundary("무게는 3kg이다", vec![])] + #[case::ambiguous_suffix("값은 3x)이다", vec![])] + #[case::forced_slash_boundary("속도는 3m/시", vec![])] + fn detects_rule69_ascii_units_before_terminator_skipping_symbols( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = rule69_ascii_unit_before_terminator_skipping_symbol_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[test] + fn roman_indicator_moves_before_capital_word_indicator() { + assert_eq!(roman_before_capital_order("⠠⠠⠴⠁⠃"), "⠴⠠⠠⠁⠃"); + } + + #[rstest::rstest] + #[case::hca("HCA(Home Connectivity Alliance)", true)] + #[case::embedded_in_korean("협회 HCA(Home Connectivity Alliance)는", true)] + #[case::lowercase_headword("Hca(Home Connectivity Alliance)", false)] + #[case::single_word_parenthetical("HCA(Alliance)", false)] + #[case::unclosed_parenthetical("HCA(Home Connectivity Alliance", false)] + #[case::operator_inside("AB(C + D)", false)] + #[case::subscript_inside("AB(C D_1)", false)] + #[case::nested_parenthetical("AB(C (D E))", false)] + fn detects_uppercase_roman_headword_expansion(#[case] input: &str, #[case] expected: bool) { + assert_eq!(has_uppercase_roman_headword_expansion(input), expected); + } + + #[rstest::rstest] + #[case::person_label("학생 A(14)양", vec!["A(14)"])] + #[case::standalone("A(1)", vec!["A(1)"])] + #[case::multiple("A(11)과 B(15)", vec!["A(11)", "B(15)"])] + #[case::lowercase("a(14)", vec![])] + #[case::multi_capital("AB(14)", vec![])] + #[case::empty_parenthetical("A()", vec![])] + #[case::letter_argument("A(x)", vec![])] + #[case::ascii_suffix("A(14)b", vec![])] + fn detects_single_capital_parenthesized_digits( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = single_capital_parenthesized_digit_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[test] + fn rejects_parenthesized_digits_after_an_ascii_letter() { + let input = std::hint::black_box(String::from("AB(14)")); + assert!(single_capital_parenthesized_digit_spans(&input).is_empty()); + } + + #[rstest::rstest] + #[case::single_capital("미 F-35 전투기", vec!["F-35"])] + #[case::multi_capital("육군 AH-64 헬기", vec!["AH-64"])] + #[case::rule_35_shape("수능 D-100일", vec!["D-100"])] + #[case::multiple("F-35와 AH-64", vec!["F-35", "AH-64"])] + #[case::lowercase("x-1", vec![])] + #[case::missing_digits("F-", vec![])] + #[case::unicode_minus("F−35", vec![])] + #[case::ascii_suffix("F-35A", vec![])] + fn detects_uppercase_roman_hyphen_digits(#[case] input: &str, #[case] expected: Vec<&str>) { + let actual = uppercase_roman_hyphen_digit_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[test] + fn locates_hyphen_digit_run_through_mixed_korean_routing() { + let input = "한글 F-35 전투기"; + let spans = uppercase_roman_hyphen_digit_spans(input); + let actual = braillify::encode_to_unicode(input).expect("probe must encode"); + let ranges = korean_context_signature_ranges(input, &actual, &spans, 1); + + assert_eq!(spans.len(), 1); + assert_eq!(&input[spans[0].start_byte..spans[0].end_byte], "F-35"); + assert_eq!(ranges.len(), 1); + assert!(ranges[0].start < ranges[0].end); + } + + #[rstest::rstest] + #[case::rule35_kf94("요즘에는 KF94 마스크가 필수입니다.", vec!["KF94"])] + #[case::rule35_mp4("새로운 MP4 Player를 출시했다.", vec!["MP4"])] + #[case::rule35_d100("2023학년도 수능 D-100일 학습 전략", vec!["D-100"])] + #[case::corpus_complex_identifier("항공기 E-4B 나이트워치", vec!["E-4B"])] + #[case::decimal_identifier("사업 LINC3.0 참여", vec!["LINC3.0"])] + #[case::trailing_comma("브랜드 C27, 도넛킬러", vec!["C27,"])] + #[case::unicode_minus_control("기종 F−35", vec![])] + fn detects_uppercase_alphanumeric_roman_digit_sequences( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = uppercase_alphanumeric_roman_digit_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[test] + fn localizes_current_complex_identifier_entry_without_using_expected() { + let input = "항공기로 불리는 E-4B 나이트워치"; + let spans = uppercase_alphanumeric_roman_digit_spans(input); + let actual = braillify::encode_to_unicode(input).expect("identifier probe must encode"); + let ranges = current_engine_input_entry_ranges(input, &actual, &spans, 2); + + assert_eq!(spans.len(), 1); + assert_eq!(ranges.len(), 1); + assert_eq!(actual.chars().nth(ranges[0].start), Some('⠀')); + } + + #[rstest::rstest] + #[case::mixed_particle_before_expansion( + "Matter와 HCA(Home Connectivity Alliance) 표준", + vec!["HCA"] + )] + #[case::another_korean_suffix("Device는 ABC(Alpha Beta Company) 규격", vec!["ABC"])] + #[case::korean_only_previous("기기와 HCA(Home Connectivity Alliance) 표준", vec![])] + #[case::roman_only_previous("Matter HCA(Home Connectivity Alliance) 표준", vec![])] + #[case::no_space_boundary("Matter와HCA(Home Connectivity Alliance)", vec![])] + #[case::single_parenthetical_word("Matter와 HCA(Alliance)", vec![])] + fn detects_mixed_roman_korean_boundary_before_headword_expansion( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = mixed_roman_korean_before_headword_expansion_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[test] + fn locates_later_headword_instead_of_earlier_roman_entry() { + let input = "한글 Matter와 HCA(Home Connectivity Alliance) 표준"; + let spans = mixed_roman_korean_before_headword_expansion_spans(input); + let actual = braillify::encode_to_unicode(input).expect("probe must encode"); + let ranges = current_engine_signature_ranges(input, &actual, &spans, 1); + let first_roman_entry = actual + .chars() + .position(|cell| cell == '⠴') + .expect("Matter must have an earlier Roman entry"); + + assert_eq!(spans.len(), 1); + assert_eq!(&input[spans[0].start_byte..spans[0].end_byte], "HCA"); + assert_eq!(ranges.len(), 1); + assert!(first_roman_entry < ranges[0].start); + } + + #[rstest::rstest] + #[case::parenthesized_initialism("업무협약(MOU)을", vec!["MOU"])] + #[case::standalone_word("SOUTH KOREA", vec!["SOUTH"])] + #[case::multiple_runs("MOU와 YOUTH", vec!["MOU", "YOUTH"])] + #[case::lowercase("Mou", vec![])] + #[case::mixed_case("MoU", vec![])] + #[case::no_ou("WHO", vec![])] + #[case::digit_boundary("1MOU", vec![])] + #[case::identifier_suffix("MOU2", vec![])] + fn detects_allcaps_roman_runs_containing_ou(#[case] input: &str, #[case] expected: Vec<&str>) { + let actual = allcaps_roman_runs_containing_ou(input) + .into_iter() + .map(|run| &input[run.start_byte..run.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::internal_pair("WSTS HUST OST USTR", vec!["WSTS", "HUST", "OST", "USTR"])] + #[case::lowercase("West", vec![])] + #[case::mixed_case("WStS", vec![])] + #[case::no_st("WHO", vec![])] + #[case::digit_prefix("1WSTS", vec![])] + #[case::digit_suffix("WSTS2", vec![])] + fn detects_allcaps_roman_runs_containing_st(#[case] input: &str, #[case] expected: Vec<&str>) { + let actual = allcaps_roman_runs_containing_st(input) + .into_iter() + .map(|run| &input[run.start_byte..run.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::official_and_corpus_shapes("DAR AR ARS START", vec!["DAR", "AR", "ARS", "START"])] + #[case::lowercase("Ar", vec![])] + #[case::mixed_case("aR", vec![])] + #[case::no_ar("WHO", vec![])] + #[case::digit_prefix("1AR", vec![])] + #[case::digit_suffix("ARS2", vec![])] + fn detects_allcaps_roman_runs_containing_ar(#[case] input: &str, #[case] expected: Vec<&str>) { + let actual = allcaps_roman_runs_containing_ar(input) + .into_iter() + .map(|run| &input[run.start_byte..run.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + /// UEB 10.12.1-10.12.2 supply both outcomes: `DAR` is pronounced as + /// separate letters, whereas `START` uses the `ar` groupsign. + #[rstest::rstest] + #[case::separate_letters("DAR", "⠠⠠⠙⠁⠗")] + #[case::contracted_acronym("START", "⠠⠠⠌⠜⠞")] + fn full_encoder_preserves_official_ueb_ar_controls( + #[case] input: &str, + #[case] expected: &str, + ) { + assert_eq!(braillify::encode_to_unicode(input).unwrap(), expected); + } + + #[rstest::rstest] + #[case::initialisms("OED LED GED", vec!["OED", "LED", "GED"])] + #[case::lowercase("Ed", vec![])] + #[case::mixed_case("eD", vec![])] + #[case::no_ed("WHO", vec![])] + #[case::digit_prefix("1LED", vec![])] + #[case::digit_suffix("LED2", vec![])] + fn detects_allcaps_roman_runs_containing_ed(#[case] input: &str, #[case] expected: Vec<&str>) { + let actual = allcaps_roman_runs_containing_ed(input) + .into_iter() + .map(|run| &input[run.start_byte..run.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + /// UEB 10.12.1-10.12.2 supply both `ed` outcomes for abbreviations. + #[rstest::rstest] + #[case::separate_letters("OED", "⠠⠠⠕⠑⠙")] + #[case::contracted_abbreviation("BEd", "⠠⠃⠠⠫")] + fn full_encoder_preserves_official_ueb_ed_controls( + #[case] input: &str, + #[case] expected: &str, + ) { + assert_eq!(braillify::encode_to_unicode(input).unwrap(), expected); + } + + #[rstest::rstest] + #[case::parenthetical_then_comma("인공지능(AI), ChatGTP", vec!["ChatGTP"])] + #[case::parenthetical_then_plain("액티브(H) ETF", vec!["ETF"])] + #[case::parenthetical_then_quoted("다목적선(MPV) ‘HMM 울산호’", vec!["HMM"])] + #[case::quoted_then_quoted("‘아리송(ARISONG)’, ‘Boyfriend’", vec!["Boyfriend"])] + #[case::ordinary_roman_words("Los Angeles", vec![])] + #[case::unclosed_parenthetical("인공지능(AI ChatGTP", vec![])] + #[case::nonroman_enclosure("항목(가) ETF", vec![])] + #[case::no_whitespace("인공지능(AI),ChatGTP", vec![])] + #[case::numeric_next("액티브(H) 3ETF", vec![])] + fn detects_roman_run_after_closed_roman_enclosure( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = roman_run_after_closed_roman_enclosure_spans(input) + .into_iter() + .map(|run| &input[run.start_byte..run.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::common_initialisms("M&A P&G R&D", vec!["M&A", "P&G", "R&D"])] + #[case::ueb_examples("AT&T B&B", vec!["AT&T", "B&B"])] + #[case::multiple_ampersands("A&B&C", vec!["A&B&C"])] + #[case::spaced_symbol("Marks & Spencer", vec![])] + #[case::korean_segments("가&나", vec![])] + #[case::empty_segment("A&&B", vec![])] + #[case::digit_continuation("R&D2", vec![])] + fn detects_attached_ascii_roman_ampersand_spans( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = attached_ascii_roman_ampersand_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::official_ueb_one_sided("&c", vec!["&c"])] + #[case::after_korean("과학&ICT", vec!["&ICT"])] + #[case::after_punctuation("(참고)&Ref", vec!["&Ref"])] + #[case::already_owned_two_sided("A&B", vec![])] + #[case::digit_left_boundary("2&K", vec![])] + #[case::digit_right_continuation("한국&K2", vec![])] + #[case::empty_segment("한국&&K", vec![])] + fn detects_ampersand_before_attached_ascii_roman_segment( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = ampersand_before_attached_ascii_roman_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::official_name("O'Hara", vec!["O'Hara"])] + #[case::official_contraction("DON'T", vec!["DON'T"])] + #[case::official_possessive("THAT'S", vec!["THAT'S"])] + #[case::detached_quotes("rock 'n' roll", vec![])] + #[case::measurement_mark("6' 2", vec![])] + #[case::korean_single_quotes("‘가’", vec![])] + fn detects_ascii_apostrophe_between_ascii_letter_runs( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = ascii_internal_apostrophe_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::numeric_list("17, 16, 15", vec![", 1", ", 1"])] + #[case::attached_rule41_number("9,375명", vec![])] + #[case::roman_prose("A, B", vec![])] + #[case::non_numeric_right_side("17, sixteen", vec![])] + fn detects_only_spaced_commas_between_ascii_digit_runs( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = spaced_comma_between_ascii_digit_run_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + // 제69항 붙임 2의 `%p` 단위에만 한정한 입력 구조 진단이다. + #[case::complete_percent_point_units("0.7%p, 0.5%p", vec!["0.7%p, 0.5%p"])] + #[case::right_unit_missing("0.7%p, 0.5%", vec![])] + #[case::comma_not_followed_by_space("0.7%p,0.5%p", vec![])] + fn detects_only_complete_percent_point_unit_lists( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = percent_point_unit_list_comma_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[test] + fn localizes_pdf_spaced_numeric_list_comma_after_rule41_fix() { + // 2024 Korean-rules PDF physical p.209. + let input = "제5열 버튼(3, 7 혹은 S)"; + let actual = braillify::encode_to_unicode(input).expect("numeric-list probe must encode"); + let ranges = spaced_numeric_list_comma_actual_ranges(input, &actual); + + assert_eq!(ranges.len(), 1); + assert!( + ranges + .iter() + .all(|range| actual.chars().nth(range.start) == Some('⠐')) + ); + } + + #[rstest::rstest] + #[case::allcaps("과학&ICT")] + #[case::single_capital("한국&K리츠")] + fn localizes_current_rule71_exit_before_attached_roman(#[case] input: &str) { + let actual = braillify::encode_to_unicode(input).expect("ampersand probe must encode"); + let ranges = ampersand_before_ascii_roman_boundary_ranges(input, &actual); + + assert!(ranges.iter().any(|range| { + actual.chars().skip(range.start).take(3).collect::() == "⠴⠈⠯" + })); + } + + #[rstest::rstest] + #[case::at_and_t("AT&T", "⠠⠠⠁⠞⠈⠯⠠⠞")] + #[case::b_and_b("B&B", "⠠⠃⠈⠯⠠⠃")] + #[case::and_c("&c (etc)", "⠈⠯⠉⠀⠐⠣⠑⠞⠉⠐⠜")] + fn full_encoder_matches_ueb_3_1_1_ampersand_examples( + #[case] input: &str, + #[case] expected: &str, + ) { + assert_eq!(braillify::encode_to_unicode(input).as_deref(), Ok(expected)); + } + + #[rstest::rstest] + #[case::official_at_and_t("AT&T", vec!["AT&T"])] + #[case::official_b_and_b("B&B", vec!["B&B"])] + #[case::multiple_uppercase_segments("M&A&R", vec!["M&A&R"])] + #[case::lowercase_segment_excluded("R&d", vec![])] + #[case::spaced_excluded("R & D", vec![])] + #[case::digit_continuation_excluded("R&D3", vec![])] + #[case::korean_attached_excluded("가(R&D)", vec![])] + #[case::whitespace_token_start("가 R&D", vec!["R&D"])] + fn detects_complete_uppercase_ampersand_segments( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = uppercase_ascii_ampersand_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[test] + fn localizes_uppercase_ampersand_in_korean_context() { + let input = "가 R&D 나"; + let actual = braillify::encode_to_unicode(input).expect("ampersand probe must encode"); + let ranges = korean_context_signature_ranges( + input, + &actual, + &uppercase_ascii_ampersand_spans(input), + 0, + ); + let signature = korean_context_signature("R&D").expect("signature must encode"); + + assert_eq!(ranges.len(), 1); + assert_eq!( + actual + .chars() + .skip(ranges[0].start) + .take(ranges[0].len()) + .collect::(), + signature + ); + } + + #[rstest::rstest] + #[case::ampersand("R&D", vec!["R&D"])] + #[case::attached_korean_suffix("S&P는", vec!["S&P는"])] + #[case::alphanumeric_restart("A1B", vec!["A1B"])] + #[case::pure_letters_excluded("PURE", vec![])] + #[case::korean_prefix_excluded("가(R&D)", vec![])] + #[case::lowercase_excluded("R&d", vec![])] + #[case::trailing_digit_excluded("MP3", vec![])] + #[case::trailing_korean_excluded("KDI에", vec![])] + fn detects_former_capitals_word_nonletter_change_scope( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = capitals_word_nonletter_change_scope_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::whole_shortform("가(WD) 나", vec!["WD"])] + #[case::longer_prefixes("PDS LLM GDP", vec!["PDS", "LLM", "GDP"])] + #[case::ueb_examples("ALT NEC LLC", vec!["ALT", "NEC", "LLC"])] + #[case::shortform_much("MCH", vec!["MCH"])] + #[case::noncolliding_controls("US KBS", vec![])] + #[case::alphanumeric_excluded("O4O Li2S V2X", vec![])] + fn detects_allcaps_shortform_prefix_collisions( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = allcaps_shortform_prefix_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::mixed_identifiers("O4O Li2S V2X", vec!["O4O", "Li2S", "V2X"])] + #[case::multiple_boundaries("A1B2C", vec!["A1B2C"])] + #[case::lowercase_after_digit("240mg", vec![])] + #[case::hyphenated_identifier("U-ENTER", vec![])] + fn detects_uppercase_after_digit_in_roman_sequence( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = roman_uppercase_after_digit_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::uppercase_segments("U-ENTER CD-ROM", vec!["U-ENTER", "CD-ROM"])] + #[case::digit_after_hyphen("F-35", vec![])] + #[case::lowercase_after_hyphen("U-enter", vec![])] + fn detects_uppercase_after_hyphen_in_roman_sequence( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = roman_uppercase_after_hyphen_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::pdf_complete_letters_sequence("CD-ROM", vec!["CD-ROM"])] + #[case::single_capital_control("Around-U", vec![])] + #[case::mixed_prefix_control("Ko-LLM", vec![])] + #[case::digit_hyphen_control("F-35", vec![])] + fn detects_only_pure_allcaps_hyphen_multi_allcaps_engine_boundary( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = pure_allcaps_hyphen_multi_allcaps_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::single_capital("오는 하쿠토-R 미션", vec!["하쿠토-R"])] + #[case::multi_capital("기장-KBO 야구센터", vec!["기장-KBO"])] + #[case::uppercase_sequence("한-UAE 협력", vec!["한-UAE"])] + #[case::normalized_roman_numeral("천궁-Ⅱ 미사일", vec!["천궁-Ⅱ"])] + #[case::lowercase_word("봄은 온다-life goes on", vec!["온다-life"])] + #[case::single_lowercase_math_control("값-x 계산", vec![])] + #[case::explicit_operator_control("값-X+1 계산", vec![])] + #[case::digit_control("한-3 단계", vec![])] + fn detects_attached_korean_to_roman_hyphen_boundary( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = korean_to_roman_hyphen_boundary_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::after_korean_word("연구단, A-STAR 방문", vec!["A-STAR"])] + #[case::after_ascii_word("research A-STAR 방문", vec![])] + #[case::without_whitespace("연구단,A-STAR 방문", vec![])] + #[case::digit_hyphen_is_separate("연구단 F-35 방문", vec![])] + #[case::numeric_leading_run("지표 2-CE 결과", vec![])] + fn detects_hyphenated_roman_word_after_korean_word( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = roman_hyphenated_word_after_korean_word_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::mixed_case_headword("시스템 ccNC(connected car) 탑재", vec!["ccNC"])] + #[case::allcaps_headword("줌 URL(ID : 3) 입력", vec!["URL"])] + #[case::after_ascii_word("system URL(ID) 입력", vec![])] + #[case::single_letter("시스템 A(x) 입력", vec![])] + #[case::nested_parenthetical("시스템 URL(ID(x)) 입력", vec![])] + fn detects_parenthetical_roman_headword_after_korean_word( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = roman_parenthetical_headword_after_korean_word_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::attached_structure("퀀텀닷(QD)-OLED 패널", vec!["(QD)-OLED"])] + #[case::without_korean_prefix("(QD)-OLED 패널", vec![])] + #[case::mixed_case_body("퀀텀닷(Qd)-OLED 패널", vec![])] + #[case::single_cap_suffix("퀀텀닷(QD)-O 패널", vec![])] + fn detects_korean_prefixed_parenthetical_hyphen_suffix( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = korean_prefixed_roman_parenthetical_hyphen_suffix_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::mixed_then_allcaps("Neo QLED는", vec!["QLED"])] + #[case::allcaps_pair("DO DREAM)", vec!["DREAM"])] + #[case::pdf_capital_passage("WELCOME TO KOREA", vec!["TO", "KOREA"])] + #[case::punctuation_break("Neo. QLED는", vec![])] + #[case::mixed_case_second("Neo Qled는", vec![])] + #[case::korean_previous("한글 QLED는", vec![])] + fn detects_consecutive_roman_uppercase_word_reentry( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = consecutive_roman_uppercase_word_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::pdf_los_angeles( + "그녀는 Los Angeles의 한인 타운에 살고 있다.", + vec!["Los Angeles"] + )] + #[case::pdf_table_of_contents( + "Table of Contents", + vec!["Table of", "of Contents"] + )] + #[case::parenthesized_name("(Max Anderson)", vec!["Max Anderson"])] + #[case::allcaps_pair("(EU CSRD)", vec!["EU CSRD"])] + #[case::slash_separated("ESS /VPP", vec![])] + #[case::korean_words("한인 타운", vec![])] + fn detects_consecutive_ascii_roman_word_boundaries( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = consecutive_ascii_roman_word_boundary_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::pdf_rule_29("그녀는 Los Angeles의 한인 타운에 살고 있다.", vec!['⠀'])] + #[case::corpus_parenthesized_name( + "LCK 글로벌 중계진은 지난 해와 마찬가지로 ‘아틀러스(Atlus)’ 맥스 앤더슨(Max Anderson), ‘발데스(Valdes)’ 브랜든 발데스(Brendan Valdes), ‘울프(Wolf)’ 울프 슈뢰더(Wolf Schroeder)와 ‘크로니클러(Chronicler)’ 모리츠 뮈센(Maurits Meeusen)이", + vec!['⠀', '⠀', '⠀', '⠀'] + )] + fn locates_current_consecutive_ascii_roman_boundary_cell( + #[case] input: &str, + #[case] expected_markers: Vec, + ) { + let actual = braillify::encode_to_unicode(input).expect("Roman boundary probe must encode"); + let ranges = consecutive_ascii_roman_boundary_actual_ranges(input, &actual); + let markers = ranges + .iter() + .map(|range| { + actual + .chars() + .nth(range.start) + .expect("localized boundary must exist") + }) + .collect::>(); + + assert_eq!( + markers, expected_markers, + "the locator must identify each occurrence-specific current boundary" + ); + } + + #[rstest::rstest] + #[case::pdf_rule_34("링컨(Lincoln)은", vec!["(Lincoln)"])] + #[case::after_digit("웹3(Web3)", vec!["(Web3)"])] + #[case::after_whitespace("전시회 (Moulding Expo)", vec!["(Moulding Expo)"])] + #[case::after_quote("선언’(Washington Declaration)", vec!["(Washington Declaration)"])] + #[case::function_call("f(x)", vec![])] + #[case::standalone_math_control("(x)", vec!["(x)"])] + #[case::nested_math_control("(f(x))", vec![])] + #[case::korean_first_body("(한글 AI)", vec![])] + fn detects_closed_roman_parenthetical_after_non_ascii_letter_boundary( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = roman_parenthetical_after_nonletter_boundary_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[test] + fn full_encoder_does_not_reenter_before_consecutive_uppercase_word() { + let input = "가(NEW YORK)"; + let spans = consecutive_roman_uppercase_word_spans(input); + let actual = braillify::encode_to_unicode(input).expect("Roman reentry probe must encode"); + let ranges = roman_entry_signature_ranges(input, &actual, &spans, 0); + + assert_eq!(spans.len(), 1); + assert!(actual.contains("⠦⠄⠴⠠⠠")); + assert!(ranges.is_empty()); + } + + #[rstest::rstest] + #[case::shortform_prefix("가(WD) 나", allcaps_shortform_prefix_spans("가(WD) 나"))] + #[case::numeric_continuation("가(Li2S) 나", roman_uppercase_after_digit_spans("가(Li2S) 나"))] + #[case::hyphen_continuation( + "가(U-ENTER) 나", + roman_uppercase_after_hyphen_spans("가(U-ENTER) 나") + )] + fn locates_grade1_cohort_signature_in_korean_context( + #[case] input: &str, + #[case] spans: Vec, + ) { + let actual = braillify::encode_to_unicode(input).expect("grade-1 probe must encode"); + let ranges = roman_entry_signature_ranges(input, &actual, &spans, 1); + + assert_eq!(ranges.len(), 1); + assert!(ranges[0].start < ranges[0].end); + assert!(ranges[0].end <= actual.chars().count()); + } + + #[rstest::rstest] + #[case::hyphenated_word( + "연구단, A-STAR 방문", + roman_hyphenated_word_after_korean_word_spans("연구단, A-STAR 방문") + )] + #[case::parenthetical_headword( + "시스템 ccNC(connected car) 탑재", + roman_parenthetical_headword_after_korean_word_spans("시스템 ccNC(connected car) 탑재") + )] + #[case::attached_parenthetical_suffix( + "퀀텀닷(QD)-OLED 패널", + korean_prefixed_roman_parenthetical_hyphen_suffix_spans("퀀텀닷(QD)-OLED 패널") + )] + fn locates_roman_entry_residual_boundary(#[case] input: &str, #[case] spans: Vec) { + let actual = braillify::encode_to_unicode(input).expect("Roman entry probe must encode"); + let ranges = current_engine_input_entry_ranges(input, &actual, &spans, 2); + + assert_eq!(spans.len(), 1); + assert_eq!(ranges.len(), 1); + assert!(ranges[0].start < ranges[0].end); + assert!(ranges[0].end <= actual.chars().count()); + } + + #[rstest::rstest] + #[case::short_digit_exact_control("웹3(Web3)", '⠦')] + #[case::after_whitespace("전시회 (Moulding Expo)", '⠴')] + #[case::after_quote("선언’(Washington Declaration)", '⠦')] + fn locates_nonletter_parenthetical_opening_in_current_output( + #[case] input: &str, + #[case] expected_opening: char, + ) { + let spans = roman_parenthetical_after_nonletter_boundary_spans(input); + let actual = braillify::encode_to_unicode(input).expect("parenthetical probe must encode"); + let ranges = current_engine_parenthetical_entry_ranges(input, &actual, &spans, 3); + + assert_eq!(spans.len(), 1); + assert_eq!(ranges.len(), 1); + assert_eq!(actual.chars().nth(ranges[0].start), Some(expected_opening)); + } + + #[rstest::rstest] + #[case::attached_digit_suffix( + "안랩은 블록체인 자회사 안랩블록체인컴퍼니가 웹(Web)3 지갑인 ‘ABC 월렛(Wallet)’ 모바일 버전을 구글플레이와 애플 앱스토어에 출시했다고 10일 발표했다." + )] + #[case::attached_hyphen_suffix( + "인천시는 경제성 향상을 위해 유정복 인천시장의 민선8기 1호 공약인 제물포르네상스와 3기 신도시인 광명·시흥 공공주택지구 등 신규 개발계획을 반영하고, 수도권광역급행철도(GTX)-D Y자(인천공항행)와 연계 방안 등을 중점 검토한다." + )] + #[case::attached_middle_dot_suffix( + "10~11일에는 지역 주민들과 함께 하는 전야제를 포함해 아주대 50년사 출판 기념보고회, 인공지능(AI)·6G 융합 콜로키움 시리즈가 열린다." + )] + fn localizes_current_rule34_boundary_before_roman_parenthetical(#[case] input: &str) { + let spans = roman_parenthetical_after_nonletter_boundary_spans(input); + let actual = braillify::encode_to_unicode(input).expect("parenthetical probe must encode"); + let ranges = current_engine_parenthetical_leading_boundary_ranges(input, &actual, &spans); + + assert!(!spans.is_empty()); + assert!(!ranges.is_empty()); + assert!(ranges.iter().any(|range| { + actual + .chars() + .skip(range.start) + .take(range.len()) + .collect::() + .contains("⠦⠄⠴") + })); + assert!(!ranges.iter().any(|range| { + actual + .chars() + .skip(range.start) + .take(range.len()) + .collect::() + .contains("⠀⠀⠦") + })); + } + + #[test] + fn locates_allcaps_ou_signature_in_complete_output() { + let input = "업무협약(MOU)을 체결했다."; + let actual = braillify::encode_to_unicode(input).expect("probe must encode"); + let ranges = allcaps_ou_actual_ranges(input, &actual); + + assert_eq!(ranges.len(), 1); + assert!(ranges[0].start < ranges[0].end); + assert!(ranges[0].end <= actual.chars().count()); + } + + #[test] + fn locates_allcaps_st_signature_in_complete_output() { + let input = "통계기구(WSTS)에 따르면"; + let actual = braillify::encode_to_unicode(input).expect("probe must encode"); + let ranges = allcaps_st_actual_ranges(input, &actual); + + assert_eq!(ranges.len(), 1); + assert!(ranges[0].start < ranges[0].end); + assert!(ranges[0].end <= actual.chars().count()); + } + + #[rstest::rstest] + #[case::augmented_reality("증강현실(AR)")] + #[case::automated_response("자동응답시스템(ARS)")] + fn localizes_allcaps_ar_signature_in_complete_output(#[case] input: &str) { + let actual = braillify::encode_to_unicode(input).expect("probe must encode"); + let ranges = allcaps_ar_actual_ranges(input, &actual); + + assert_eq!(ranges.len(), 1); + assert!(ranges[0].start < ranges[0].end); + assert!(ranges[0].end <= actual.chars().count()); + } + + #[rstest::rstest] + #[case::light_emitting_diode("마이크로 LED")] + #[case::education_database("데이터베이스(GED)")] + fn localizes_allcaps_ed_signature_in_complete_output(#[case] input: &str) { + let actual = braillify::encode_to_unicode(input).expect("probe must encode"); + let ranges = allcaps_ed_actual_ranges(input, &actual); + + assert_eq!(ranges.len(), 1); + assert!(ranges[0].start < ranges[0].end); + assert!(ranges[0].end <= actual.chars().count()); + } + + #[rstest::rstest] + #[case::parenthetical_then_comma("인공지능(AI), ChatGTP")] + #[case::parenthetical_then_plain("액티브(H) ETF")] + #[case::parenthetical_then_quoted("다목적선(MPV) ‘HMM 울산호’")] + #[case::quoted_then_quoted("‘아리송(ARISONG)’, ‘Boyfriend’")] + fn localizes_roman_after_closed_enclosure_signature(#[case] input: &str) { + let actual = braillify::encode_to_unicode(input).expect("probe must encode"); + let ranges = roman_after_closed_enclosure_actual_ranges(input, &actual); + + assert_eq!(ranges.len(), 1); + assert!(ranges[0].start < ranges[0].end); + assert!(ranges[0].end <= actual.chars().count()); + } + + #[rstest::rstest] + #[case::merger("인수·합병(M&A) 시장")] + #[case::brand("브랜드(P&G) 편입")] + #[case::research("연구·개발(R&D)을 추진")] + fn resolved_attached_roman_ampersand_has_no_boundary_signature(#[case] input: &str) { + let actual = braillify::encode_to_unicode(input).expect("ampersand probe must encode"); + let ranges = attached_roman_ampersand_boundary_ranges(input, &actual); + + assert!(ranges.is_empty()); + assert!(actual.contains("⠈⠯")); + assert!(!actual.contains("⠲⠴⠈⠯")); + assert!(!actual.contains("⠈⠯⠲⠴")); + } + + #[rstest::rstest] + #[case::standalone("새로운 DRX 브랜드", true)] + #[case::inside_parentheses("엠디(MD), SNS", true)] + #[case::expansion_headword("HCA(Home Connectivity Alliance)", false)] + #[case::single_capital("점 A가 있다", false)] + #[case::mixed_case("SmartThings Hub", false)] + #[case::alphanumeric("O4O 시스템", false)] + #[case::chemical_formula("PETCO2이다", false)] + #[case::lowercase("web service", false)] + fn detects_standalone_uppercase_roman_word(#[case] input: &str, #[case] expected: bool) { + assert_eq!(has_standalone_uppercase_roman_word(input), expected); + } + + #[rstest::rstest] + #[case::acronym("최고운영책임자(COO)", true)] + #[case::organization("국가안전보장회의(NSC)를", true)] + #[case::chemical_formula("일산화탄소(CO)는", true)] + #[case::space_before_parenthesis("책임자 (COO)", false)] + #[case::roman_prefix("HCA(COO)", false)] + #[case::mixed_case("책임자(Ceo)", false)] + #[case::digit_inside("규격(CO2)", false)] + #[case::unclosed("책임자(COO", false)] + fn detects_korean_prefixed_allcaps_parenthetical(#[case] input: &str, #[case] expected: bool) { + assert_eq!(has_korean_prefixed_allcaps_parenthetical(input), expected); + } + + #[rstest::rstest] + #[case::pdf_example("링컨(Lincoln)은", vec!["(Lincoln)"])] + #[case::allcaps_annotation("엠디(MD),", vec!["(MD)"])] + #[case::roman_suffix("폐쇄회로(CC)TV", vec!["(CC)"])] + #[case::alphanumeric("표기(O4O)는", vec!["(O4O)"])] + #[case::space_in_body("표기(Home Alliance)는", vec![])] + #[case::operator_in_body("수식(x+1)은", vec![])] + #[case::roman_prefix("HCA(Home)는", vec![])] + #[case::space_before_open("표기 (MD)는", vec![])] + #[case::unclosed("표기(MD", vec![])] + fn detects_korean_prefixed_closed_roman_annotations( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = korean_prefixed_closed_roman_annotation_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[test] + fn locates_rule_34_opening_after_the_real_korean_prefix() { + let input = "앞말 링컨(Lincoln)은"; + let actual = braillify::encode_to_unicode(input).expect("rule-34 probe must encode"); + let ranges = korean_prefixed_annotation_opening_ranges(input, &actual); + + assert_eq!(ranges.len(), 1); + assert_eq!(actual.chars().nth(ranges[0].start), Some('⠦')); + assert_eq!(actual.chars().nth(ranges[0].start + 1), Some('⠄')); + } + + #[test] + fn classifies_only_the_rule_34_three_cell_reference_order_as_corpus_suspect() { + let input = "링컨(Lincoln)은"; + let actual = braillify::encode_to_unicode(input).expect("rule-34 probe must encode"); + let opening = korean_prefixed_annotation_opening_ranges(input, &actual) + .into_iter() + .next() + .expect("opening must be localized"); + let mut expected = actual.chars().collect::>(); + expected.splice(opening.start..opening.start + 3, ['⠴', '⠐', '⠣']); + let encoded = EncodedCase { + located: LocatedCase { + shard: "synthetic.json".to_string(), + index: 1, + case: CorpusCase { + input: input.to_string(), + unicode: expected.into_iter().collect(), + }, + }, + actual: Ok(actual), + nfc_actual: None, + nfkc_actual: None, + singleton_unsupported_characters: Vec::new(), + }; + + assert!(is_rule_34_reference_order_contradiction(&encoded)); + assert_eq!( + classify(&encoded, &BTreeSet::new()), + ( + PrimaryClass::CorpusSuspect, + Reason::Rule34RomanIndicatorBeforeOpeningParenthesis + ) + ); + } + + #[test] + fn classifies_korean_cells_for_an_enclosed_roman_ellipsis_as_corpus_suspect() { + let input = "제목(Love Is…)까지"; + let actual = braillify::encode_to_unicode(input).expect("Roman ellipsis probe must encode"); + assert!(actual.contains("⠲⠲⠲")); + let expected = actual.replacen("⠲⠲⠲", "⠠⠠⠠", 1); + let encoded = EncodedCase { + located: LocatedCase { + shard: "synthetic.json".to_string(), + index: 1, + case: CorpusCase { + input: input.to_string(), + unicode: expected, + }, + }, + actual: Ok(actual), + nfc_actual: None, + nfkc_actual: None, + singleton_unsupported_characters: Vec::new(), + }; + + assert!(is_roman_ellipsis_reference_contradiction(&encoded)); + assert_eq!( + classify(&encoded, &BTreeSet::new()), + ( + PrimaryClass::CorpusSuspect, + Reason::RomanEllipsisUsesKoreanCellsInRomanEnclosure + ) + ); + } + + #[rstest::rstest] + #[case::round_shortform("가 PDS(설명) 나", false, true)] + #[case::square_shortform("가 PDS[설명] 나", false, true)] + #[case::curly_shortform("가 PDS{설명} 나", false, true)] + #[case::standing_sequence("가 PDS 설명", false, false)] + #[case::plain_initialism("가 KBS(설명) 나", false, false)] + #[case::unrelated_difference("가 PDS(설명) 나", true, false)] + fn recognizes_only_extra_grade1_before_nonstanding_opening_parenthesis( + #[case] input: &str, + #[case] add_unrelated_difference: bool, + #[case] expected_result: bool, + ) { + let actual = braillify::encode_to_unicode(input).expect("grade-1 probe must encode"); + let capitals = actual + .find("⠠⠠") + .expect("probe must contain a capitals-word indicator"); + let mut expected = format!("{}⠰{}", &actual[..capitals], &actual[capitals..]); + if add_unrelated_difference { + expected.push('⠁'); + } + let encoded = EncodedCase { + located: LocatedCase { + shard: "synthetic.json".to_string(), + index: 1, + case: CorpusCase { + input: input.to_string(), + unicode: expected, + }, + }, + actual: Ok(actual), + nfc_actual: None, + nfkc_actual: None, + singleton_unsupported_characters: Vec::new(), + }; + + assert_eq!( + is_ueb_grade1_before_nonstanding_opening_parenthesis_reference_contradiction(&encoded), + expected_result + ); + } + + #[test] + fn classifies_extra_grade1_before_nonstanding_opening_parenthesis_as_corpus_suspect() { + let input = "가 PDS(설명) 나"; + let actual = braillify::encode_to_unicode(input).expect("grade-1 probe must encode"); + let capitals = actual + .find("⠠⠠") + .expect("probe must contain a capitals-word indicator"); + let expected = format!("{}⠰{}", &actual[..capitals], &actual[capitals..]); + let encoded = EncodedCase { + located: LocatedCase { + shard: "synthetic.json".to_string(), + index: 1, + case: CorpusCase { + input: input.to_string(), + unicode: expected, + }, + }, + actual: Ok(actual), + nfc_actual: None, + nfkc_actual: None, + singleton_unsupported_characters: Vec::new(), + }; + + assert_eq!( + classify(&encoded, &BTreeSet::new()), + ( + PrimaryClass::CorpusSuspect, + Reason::UebGrade1BeforeNonstandingOpeningParenthesis + ) + ); + } + + #[rstest::rstest] + #[case::caution_wet_paint( + "CAUTION: WET PAINT!", + "⠠⠠⠠⠉⠁⠥⠞⠊⠕⠝⠒⠀⠺⠑⠞⠀⠏⠁⠊⠝⠞⠖⠠⠄", + "⠠⠠⠉⠁⠥⠞⠊⠕⠝⠒⠀⠠⠠⠺⠑⠞⠀⠠⠠⠏⠁⠊⠝⠞⠖", + true + )] + #[case::bbc_africa_news( + "THE BBC AFRICA NEWS", + "⠠⠠⠠⠞⠓⠑⠀⠃⠃⠉⠀⠁⠋⠗⠊⠉⠁⠀⠝⠑⠺⠎⠠⠄", + "⠠⠠⠞⠓⠑⠀⠠⠠⠃⠃⠉⠀⠠⠠⠁⠋⠗⠊⠉⠁⠀⠠⠠⠝⠑⠺⠎", + true + )] + #[case::self_made_man( + "A SELF-MADE MAN", + "⠠⠠⠠⠁⠀⠎⠑⠇⠋⠤⠍⠁⠙⠑⠀⠍⠁⠝⠠⠄", + "⠠⠁⠀⠠⠠⠎⠑⠇⠋⠤⠍⠁⠙⠑⠀⠠⠠⠍⠁⠝", + true + )] + #[case::only_two_sequences("WET PAINT!", "⠠⠠⠠⠺⠑⠞⠀⠏⠁⠊⠝⠞⠖⠠⠄", "⠠⠠⠺⠑⠞⠀⠠⠠⠏⠁⠊⠝⠞⠖", false)] + #[case::unrelated_content_difference( + "CAUTION: WET PAINT!", + "⠠⠠⠠⠉⠁⠥⠞⠊⠕⠝⠒⠀⠺⠑⠞⠀⠏⠁⠊⠝⠞⠖⠠⠄", + "⠠⠠⠉⠁⠥⠞⠊⠕⠝⠒⠀⠠⠠⠺⠑⠞⠀⠠⠠⠏⠁⠊⠝⠭⠖", + false + )] + fn recognizes_only_complete_ueb_capitals_passage_marker_substitutions( + #[case] input: &str, + #[case] actual: &str, + #[case] expected: &str, + #[case] expected_result: bool, + ) { + let encoded = EncodedCase { + located: LocatedCase { + shard: "synthetic.json".to_string(), + index: 1, + case: CorpusCase { + input: input.to_string(), + unicode: expected.to_string(), + }, + }, + actual: Ok(actual.to_string()), + nfc_actual: None, + nfkc_actual: None, + singleton_unsupported_characters: Vec::new(), + }; + + assert_eq!( + is_ueb_capitalized_passage_reference_contradiction(&encoded), + expected_result + ); + } + + #[test] + fn classifies_separate_capital_words_for_a_ueb_passage_as_corpus_suspect() { + let encoded = EncodedCase { + located: LocatedCase { + shard: "synthetic.json".to_string(), + index: 1, + case: CorpusCase { + input: "A SELF-MADE MAN".to_string(), + unicode: "⠠⠁⠀⠠⠠⠎⠑⠇⠋⠤⠍⠁⠙⠑⠀⠠⠠⠍⠁⠝".to_string(), + }, + }, + actual: Ok("⠠⠠⠠⠁⠀⠎⠑⠇⠋⠤⠍⠁⠙⠑⠀⠍⠁⠝⠠⠄".to_string()), + nfc_actual: None, + nfkc_actual: None, + singleton_unsupported_characters: Vec::new(), + }; + + assert_eq!( + classify(&encoded, &BTreeSet::new()), + ( + PrimaryClass::CorpusSuspect, + Reason::UebCapitalizedPassageWrittenAsSeparateCapitalWords + ) + ); + } + + #[rstest::rstest] + #[case::embedded_in_korean("AI·SW교육", true)] + #[case::standalone("DRX·SNS", true)] + #[case::lowercase("a·b", false)] + #[case::single_letters("A·B", false)] + #[case::korean("가·나", false)] + #[case::numeric("3·1 운동", false)] + #[case::space_separated("AI · SW", false)] + #[case::alphanumeric_boundary("1AI·SW2", false)] + fn detects_allcaps_roman_middle_dot_runs(#[case] input: &str, #[case] expected: bool) { + assert_eq!(has_allcaps_roman_middle_dot_runs(input), expected); + } + + #[rstest::rstest] + #[case::roman_korean("신작 PC·모바일", vec!["PC·모"])] + #[case::roman_roman("AI·SW교육", vec!["AI·SW"])] + #[case::mixed_case("기관(Fed·연준)", vec!["Fed·연"])] + #[case::korean_only("온·오프라인", vec![])] + #[case::numeric("3·1 운동", vec![])] + #[case::spaced("AI · SW", vec![])] + fn detects_roman_run_before_middle_dot_boundary( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = roman_run_before_middle_dot_boundary_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::roman_korean("신작 PC·모바일", true)] + #[case::multi_letter_roman_continuation("AI·SW교육", false)] + #[case::mixed_case("기관(Fed·연준)", true)] + fn localizes_only_current_terminator_before_middle_dot( + #[case] input: &str, + #[case] expected_terminator: bool, + ) { + let actual = braillify::encode_to_unicode(input).expect("probe must encode"); + let ranges = roman_middle_dot_boundary_actual_ranges(input, &actual); + + assert_eq!(ranges.len(), usize::from(expected_terminator)); + if let Some(range) = ranges.first() { + assert_eq!(range.end - range.start, 1); + assert_eq!(actual.chars().nth(range.start), Some('⠲')); + assert!(range.end <= actual.chars().count()); + } + } + + #[rstest::rstest] + #[case::acronym_annotation("FAA항공정보(NOTAMS)", vec!["FAA항"])] + #[case::mixed_case_word("e스포츠", vec!["e스"])] + #[case::domain_control("www.대통령.kr", vec![])] + #[case::spaced_pdf_control("What is 김치 in English?", vec![])] + fn detects_only_attached_roman_to_korean_boundaries( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = attached_ascii_roman_to_korean_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::rule29_after_korean_majority_narrowing( + "이와 관련해서 FAA는 “FAA항공정보(NOTAMS) 업데이트에 관여한다”고 말했다.", + '⠲', + 1 + )] + #[case::current_rule29_terminator("종목 e스포츠", '⠲', 1)] + fn localizes_current_attached_roman_to_korean_marker( + #[case] input: &str, + #[case] first_marker: char, + #[case] marker_cells: usize, + ) { + let actual = braillify::encode_to_unicode(input).expect("boundary probe must encode"); + let ranges = attached_ascii_roman_to_korean_actual_ranges(input, &actual); + + assert!(!ranges.is_empty()); + assert!( + ranges + .iter() + .all(|range| range.end - range.start == marker_cells) + ); + assert!( + ranges + .iter() + .all(|range| actual.chars().nth(range.start) == Some(first_marker)) + ); + } + + #[rstest::rstest] + #[case::korean_majority_sandwich( + "이와 관련해서 FAA는 “FAA항공정보(NOTAMS) 업데이트에 관여한다”고 말했다.", + vec!["항공정보"] + )] + #[case::pdf_domain_control("대통령실의 누리집 주소는 www.대통령.kr이다.", vec![])] + #[case::english_majority_pdf_control("What is 김치 in English?", vec![])] + fn detects_rule39_narrowed_scope_without_using_reference_output( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = korean_majority_roman_sandwich_non_domain_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[rstest::rstest] + #[case::plus("양(+)극", vec!['+'])] + #[case::hyphen_minus("음(-)극", vec!['-'])] + #[case::unicode_minus("음(−)극", vec!['−'])] + #[case::times("항(×)목", vec!['×'])] + #[case::division("항(÷)목", vec!['÷'])] + #[case::equals("항(=)목", vec!['='])] + #[case::standalone("(+) 전극", vec![])] + #[case::space_inside("양( + )극", vec![])] + fn detects_inline_parenthesized_operators( + #[case] input: &str, + #[case] expected_operators: Vec, + ) { + assert_eq!( + inline_parenthesized_operators(input) + .into_iter() + .map(|candidate| candidate.operator) + .collect::>(), + expected_operators + ); + } + + #[test] + fn locates_current_engine_parenthesized_operator_output() { + let input = "양(+)극"; + let actual = braillify::encode_to_unicode(input).expect("probe must encode"); + let ranges = inline_parenthesized_operator_actual_ranges(input, &actual); + + assert_eq!(ranges.len(), 1); + assert!(ranges[0].start < ranges[0].end); + assert!(ranges[0].end <= actual.chars().count()); + } + + #[rstest::rstest] + #[case::program_name("도전+(플러스)", vec!["+(플러스)"])] + #[case::descriptive_gloss("자립+(더하기) 프로젝트", vec!["+(더하기)"])] + #[case::multiple("디즈니+(플러스), 애플+(플러스)", vec!["+(플러스)", "+(플러스)"])] + #[case::roman_prefix("TV+(플러스)", vec![])] + #[case::spaced_plus("도전 +(플러스)", vec![])] + #[case::roman_body("도전+(plus)", vec![])] + #[case::empty_body("도전+()", vec![])] + fn detects_attached_plus_parenthesized_korean_gloss( + #[case] input: &str, + #[case] expected: Vec<&str>, + ) { + let actual = attached_plus_parenthesized_korean_gloss_spans(input) + .into_iter() + .map(|span| &input[span.start_byte..span.end_byte]) + .collect::>(); + assert_eq!(actual, expected); + } + + #[test] + fn locates_current_engine_attached_plus_gloss_output() { + let input = "과 도전+(플러스) 프로그램"; + let actual = braillify::encode_to_unicode(input).expect("probe must encode"); + let ranges = attached_plus_parenthesized_korean_gloss_actual_ranges(input, &actual); + + assert_eq!(ranges.len(), 1); + assert!(ranges[0].start < ranges[0].end); + assert!(ranges[0].end <= actual.chars().count()); + } + + #[rstest::rstest] + #[case::tight("△보성군", 1)] + #[case::embedded("목록 △교과전형", 1)] + #[case::spaced("△ 보성군", 0)] + #[case::repeated_omission("△△ 종목", 0)] + #[case::square_bullet("□2021", 0)] + fn detects_tight_triangle_before_korean(#[case] input: &str, #[case] expected: usize) { + assert_eq!(tight_triangle_positions(input).len(), expected); + } + + #[rstest::rstest] + #[case::leading("△보성군")] + #[case::embedded("목록 △교과전형")] + #[case::after_roman_context("MOU 협약 뒤 △항목")] + fn locates_tight_triangle_and_first_korean_output(#[case] input: &str) { + let actual = braillify::encode_to_unicode(input).expect("probe must encode"); + let ranges = tight_triangle_actual_ranges(input, &actual); + + assert_eq!(ranges.len(), 1); + assert!(ranges[0].start < ranges[0].end); + assert!(ranges[0].end <= actual.chars().count()); + } + + #[rstest::rstest] + #[case::corpus_attached_suffix("성장을 하고있다.", true)] + #[case::pdf_printed_space("그림을 그리고 있다.", false)] + #[case::independent_suffix("있다.", false)] + fn detects_only_attached_korean_auxiliary_itda(#[case] input: &str, #[case] expected: bool) { + assert_eq!(has_attached_korean_auxiliary_itda(input), expected); + } + + #[test] + fn aggregates_headword_expansion_outcomes_without_reclassification() { + let cases = [ + ("HCA(Home Connectivity Alliance)", "exact", "exact"), + ("WHO(World Health Organization)", "expected", "actual"), + ] + .into_iter() + .enumerate() + .map(|(offset, (input, expected, actual))| { + let located = LocatedCase { + shard: "synthetic.json".to_string(), + index: offset + 1, + case: CorpusCase { + input: input.to_string(), + unicode: expected.to_string(), + }, + }; + let encoded = EncodedCase { + located: located.clone(), + actual: Ok(actual.to_string()), + nfc_actual: None, + nfkc_actual: None, + singleton_unsupported_characters: Vec::new(), + }; + (located, encoded) + }) + .collect::>(); + let (located, encoded): (Vec<_>, Vec<_>) = cases.into_iter().unzip(); + + let report = analyze(located, encoded, 5); + let stats = report + .pending_rule_review_clusters + .get(UPPERCASE_ROMAN_HEADWORD_EXPANSION) + .unwrap(); + + assert_eq!((stats.candidates, stats.exact, stats.mismatch), (2, 1, 1)); + assert_eq!(stats.conflicting_reference_cases, 0); + assert_eq!(stats.mismatch_primary_classes["pending_rule_review"], 1); + assert_eq!(stats.samples["mismatch"][0].expected_excerpt, "expected"); + assert_eq!(stats.samples["mismatch"][0].actual_excerpt, "actual"); + assert_eq!(stats.samples["mismatch"][0].first_difference_cell, Some(0)); + assert_eq!(report.primary_classes["exact"], 1); + assert_eq!(report.primary_classes["pending_rule_review"], 1); + } +} diff --git a/libs/braillify/src/encoder.rs b/libs/braillify/src/encoder.rs index a14e275a..6678df48 100644 --- a/libs/braillify/src/encoder.rs +++ b/libs/braillify/src/encoder.rs @@ -109,7 +109,13 @@ impl Encoder { token_engine.register(Box::new( rules::token_rules::historical_gloss_spacing::HistoricalGlossSpacingRule, )); + token_engine.register(Box::new( + rules::token_rules::normalize::NormalizeAsciiAngleBrackets, + )); token_engine.register(Box::new(rules::token_rules::normalize::NormalizeEllipsis)); + token_engine.register(Box::new( + rules::korean::rule_72::Rule72AttachedMarkerTokenRule, + )); // PDF 한국어 제33항 — 학술 인용 형식 year-suffix token (1998a,, 1998b;). token_engine.register(Box::new( rules::token_rules::rule_33_citation::Rule33CitationYearSuffixRule, diff --git a/libs/braillify/src/english_logic.rs b/libs/braillify/src/english_logic.rs index f31b0899..1056fa9a 100644 --- a/libs/braillify/src/english_logic.rs +++ b/libs/braillify/src/english_logic.rs @@ -78,6 +78,141 @@ fn is_ascii_letter_or_digit(ch: Option) -> bool { ch.is_some_and(|c| c.is_ascii_alphanumeric()) } +/// Whether a following print item starts with a number and then crosses +/// directly into Korean text, without an intervening Roman unit/identifier. +/// +/// Korean rule 33 changes a comma to the Korean punctuation sign only at an +/// actual Roman-to-Korean boundary. Looking for Korean anywhere later in the +/// whitespace-delimited item is too broad: in `173cm, 68kg의`, the comma is +/// followed first by the Roman measurement `68kg`, and the particle `의` is a +/// later boundary. Numeric grouping/decimal punctuation remains part of the +/// numeric prefix (`1,000년`, `3.5년`). +fn begins_numeric_then_korean(chars: impl Iterator) -> bool { + let mut chars = chars.peekable(); + if !chars.peek().is_some_and(char::is_ascii_digit) { + return false; + } + + while let Some(ch) = chars.next() { + if ch.is_ascii_digit() + || (matches!(ch, ',' | '.') && chars.peek().is_some_and(char::is_ascii_digit)) + { + continue; + } + return utils::is_korean_char(ch); + } + + false +} + +/// Returns whether `index` is an ampersand inside a complete sequence of +/// non-empty ASCII-letter segments joined by `&`. Korean rule 35 allows the +/// resulting Roman text to continue directly into digits and later Roman +/// letters, as in the official `MP4 Player` example. A leading digit remains +/// outside this predicate because the cited sequence begins with Roman text. +pub(crate) fn is_attached_ascii_roman_ampersand(word_chars: &[char], index: usize) -> bool { + if word_chars.get(index) != Some(&'&') + || index == 0 + || index + 1 >= word_chars.len() + || !word_chars[index - 1].is_ascii_alphabetic() + || !word_chars[index + 1].is_ascii_alphabetic() + { + return false; + } + + let mut start = index; + while start > 0 && (word_chars[start - 1].is_ascii_alphabetic() || word_chars[start - 1] == '&') + { + start -= 1; + } + let mut end = index + 1; + while end < word_chars.len() + && (word_chars[end].is_ascii_alphanumeric() || word_chars[end] == '&') + { + end += 1; + } + + let segment = &word_chars[start..end]; + segment.first().is_some_and(|ch| ch.is_ascii_alphabetic()) + && segment.last().is_some_and(|ch| ch.is_ascii_alphanumeric()) + && segment.iter().enumerate().all(|(offset, ch)| { + *ch != '&' + || (offset > 0 + && offset + 1 < segment.len() + && segment[offset - 1].is_ascii_alphabetic() + && segment[offset + 1].is_ascii_alphabetic()) + }) + && (start == 0 || !word_chars[start - 1].is_ascii_alphanumeric()) + && (end == word_chars.len() || !word_chars[end].is_ascii_alphanumeric()) +} + +/// Returns whether `index` is an asterisk inside a complete sequence of +/// non-empty ASCII-alphanumeric segments, each containing a Roman letter. +/// UEB 3.3.1 says that the asterisk follows its UEB form regardless of meaning +/// and gives `M*A*S*H` as an attached Roman example. Korean rules 32 and 35 +/// keep the resulting UEB text, including directly adjacent digits, in one +/// Roman section. Requiring a Roman letter in every segment keeps numeric +/// `2*3`, detached asterisks, Korean text, and empty segments out of this rule. +pub(crate) fn is_attached_ascii_roman_asterisk(word_chars: &[char], index: usize) -> bool { + if word_chars.get(index) != Some(&'*') || index == 0 || index + 1 >= word_chars.len() { + return false; + } + + let mut start = index; + while start > 0 + && (word_chars[start - 1].is_ascii_alphanumeric() || word_chars[start - 1] == '*') + { + start -= 1; + } + let mut end = index + 1; + while end < word_chars.len() + && (word_chars[end].is_ascii_alphanumeric() || word_chars[end] == '*') + { + end += 1; + } + + let segment = &word_chars[start..end]; + segment + .split(|ch| *ch == '*') + .all(|part| !part.is_empty() && part.iter().any(|ch| ch.is_ascii_alphabetic())) + && segment.first().is_some_and(|ch| ch.is_ascii_alphabetic()) + && (start == 0 || !word_chars[start - 1].is_ascii_alphanumeric()) + && (end == word_chars.len() || !word_chars[end].is_ascii_alphanumeric()) +} + +/// Returns whether `index` is the one-sided ampersand at the beginning of a +/// complete attached ASCII-letter segment. UEB 3.1.1 prints `&c` without a +/// boundary between the ampersand and `c`; Korean rule 35 then permits an +/// attached numeric/Roman continuation. A left ASCII alphanumeric or another +/// ampersand is excluded so the existing two-sided `A&B` rule retains ownership. +pub(crate) fn is_ampersand_before_attached_ascii_roman_segment( + word_chars: &[char], + index: usize, +) -> bool { + if word_chars.get(index) != Some(&'&') + || !word_chars + .get(index + 1) + .is_some_and(|ch| ch.is_ascii_alphabetic()) + || index + .checked_sub(1) + .and_then(|i| word_chars.get(i)) + .is_some_and(|previous| previous.is_ascii_alphanumeric() || *previous == '&') + { + return false; + } + + let mut end = index + 1; + while word_chars + .get(end) + .is_some_and(|ch| ch.is_ascii_alphanumeric()) + { + end += 1; + } + word_chars + .get(end) + .is_none_or(|next| !next.is_ascii_alphanumeric()) +} + fn is_digital_notation_symbol(symbol: char) -> bool { matches!(symbol, '/' | '@' | '#' | '.' | '_' | ':') } @@ -141,6 +276,41 @@ pub(crate) fn next_ascii_letter_or_digit( false } +/// Korean rule 46's `BMI(체질량 지수)` example assigns the attached, closed +/// parenthesis to Korean punctuation even though it follows a Roman run. Scan +/// the complete balanced enclosure because its Korean content can begin after +/// a number, a Roman expansion, or a print-space boundary. Pure Roman/number +/// enclosures remain eligible for rule 32's UEB punctuation. +fn closed_parenthesis_contains_korean( + word_chars: &[char], + index: usize, + remaining_words: &[&str], +) -> bool { + let mut depth = 1usize; + let mut contains_korean = false; + let tail = word_chars + .iter() + .skip(index + 1) + .copied() + .chain(remaining_words.iter().flat_map(|word| word.chars())); + + for ch in tail { + match ch { + '(' => depth += 1, + ')' => { + depth -= 1; + if depth == 0 { + return contains_korean; + } + } + _ if utils::is_korean_char(ch) => contains_korean = true, + _ => {} + } + } + + false +} + #[allow(clippy::too_many_arguments)] /// 괄호/쉼표가 영어 점자로 이어져야 하는지 판정한다. /// - '(' 는 뒤에 올 문자가 ASCII 영숫자여야 하고, 앞은 한글이 아니어야 한다. @@ -149,6 +319,7 @@ pub(crate) fn next_ascii_letter_or_digit( pub(crate) fn should_render_symbol_as_english( english_indicator: bool, is_english: bool, + is_english_majority: bool, parenthesis_stack: &[bool], symbol: char, word_chars: &[char], @@ -170,24 +341,107 @@ pub(crate) fn should_render_symbol_as_english( remaining_words.first().and_then(|w| w.chars().next()) }; + // A non-English closing enclosure is a hard Roman-section boundary. The + // look-behind helpers deliberately skip punctuation for attached UEB runs, + // but must not reach through that boundary and pull a following version or + // identifier mark (`(XBB).1.5`) back into the closed Roman section. + if !is_english && prev_char.is_some_and(|ch| matches!(ch, ')' | ']' | '}')) { + return false; + } + match symbol { - '(' => is_ascii_letter_or_digit(next_char) && !prev_char.is_some_and(utils::is_korean_char), + '(' => { + (is_english_majority + || !closed_parenthesis_contains_korean(word_chars, index, remaining_words)) + && is_ascii_letter_or_digit(next_char) + && !prev_char.is_some_and(utils::is_korean_char) + } ')' => parenthesis_stack.last().copied().unwrap_or(false), + // UEB 3.1.1 prints an ampersand without ending and restarting + // grade-1 mode in attached Roman forms such as AT&T and B&B. Use + // a complete ASCII-letter run so spaced prose, Hangul, and outer + // alphanumeric continuations keep their existing routes. + '&' => is_attached_ascii_roman_ampersand(word_chars, index), + // UEB 3.3.1 explicitly keeps the general-purpose asterisk inside the + // attached Roman example `M*A*S*H`. Preserve that one Roman section; + // Korean Rule 60 continues to own standalone and non-Roman asterisks. + '*' => is_attached_ascii_roman_asterisk(word_chars, index), + // UEB 8.4.2 keeps the apostrophe inside the Roman word in its + // `O'Hara`, `DON'T`, and `THAT'S` examples. Capitals-word mode may + // terminate at this nonalphabetic symbol, but the surrounding Roman + // section does not. Detached quotes and digit measurement marks stay + // on their existing punctuation routes. + '\'' => { + prev_char.is_some_and(|ch| ch.is_ascii_alphabetic()) + && word_chars + .get(index + 1) + .is_some_and(|ch| ch.is_ascii_alphabetic()) + } + // UEB 7.3 writes the single-character ellipsis as three full stops. + // Keep that UEB punctuation only while the Roman run visibly + // continues or closes an enclosure. A Unicode ellipsis followed + // directly by Korean remains the Rule-53 Korean middle-dot ellipsis. + '…' => { + is_english + && (next_ascii_letter_or_digit(word_chars, index, remaining_words) + || matches!(next_char, Some(')' | ']' | '}' | '”' | '’' | '」' | '』'))) + } ',' => { if !is_english { return false; } + let next_word_is_digit_led_korean = if index + 1 < word_chars.len() { + begins_numeric_then_korean(word_chars[index + 1..].iter().copied()) + } else { + remaining_words + .first() + .is_some_and(|word| begins_numeric_then_korean(word.chars())) + }; + if next_word_is_digit_led_korean { + // Korean rule 33: punctuation whose UEB and Korean cells + // differ is written as Korean punctuation at a Roman-to- + // Korean boundary. Limit the whole-token lookahead to a + // digit-led Korean word: Roman-led mixed words such as + // `LG유플러스` continue the Roman list and are not this case. + return false; + } + + let prev_roman = prev_ascii_letter_or_digit(word_chars, index) + || prev_char + .is_some_and(crate::rules::korean::rule_69::is_compatibility_unit_presentation); + let next_roman = next_ascii_letter_or_digit(word_chars, index, remaining_words) + || next_char + .is_some_and(crate::rules::korean::rule_69::is_compatibility_unit_presentation); + + prev_roman && next_roman + } + '-' => { let prev_ascii = prev_ascii_letter_or_digit(word_chars, index); let next_ascii = next_ascii_letter_or_digit(word_chars, index, remaining_words); - - prev_ascii && next_ascii + let roman_started_before_hyphen = word_chars[..index] + .iter() + .rev() + .take_while(|ch| ch.is_ascii_alphanumeric() || **ch == '-') + .any(|ch| ch.is_ascii_alphabetic()); + + // Korean rule 35 keeps a Roman-led identifier (`CV3-AD685`, + // `N-79-20`) in one Roman/number chain across its hyphens. A + // number-led item (`0-Zone`, `777-300ER`) has not entered Roman + // mode yet: its first Roman indicator belongs immediately before + // the first letter, never before an earlier hyphen. + prev_ascii && next_ascii && (is_english || roman_started_before_hyphen) } - '/' | '@' | '#' | '.' | '_' | ':' | '-' => { + '/' | '@' | '#' | '.' | '_' | ':' => { let prev_ascii = prev_ascii_letter_or_digit(word_chars, index); let next_ascii = next_ascii_letter_or_digit(word_chars, index, remaining_words); (prev_ascii && next_ascii) + // Korean rules 29/32/35: `Alpha : Beta` is one Roman + // section. The print tokenizer makes the colon a standalone + // word; an already-open section proves its left Roman item, + // while this lookahead proves the right Roman/number item. + || (symbol == ':' && is_english && word_chars == [':'] && next_ascii) || (symbol == '/' && prev_char == Some('/') && next_ascii) || (symbol == '/' && next_char == Some('/') && prev_ascii) } @@ -205,7 +459,16 @@ pub(crate) fn should_keep_english_mode_for_symbol( return false; } - should_render_symbol_as_english(true, true, &[], symbol, word_chars, index, remaining_words) + should_render_symbol_as_english( + true, + true, + false, + &[], + symbol, + word_chars, + index, + remaining_words, + ) } #[cfg(test)] @@ -283,39 +546,56 @@ mod tests { assert_eq!(next_ascii_letter_or_digit(&word, idx, remaining), expected); } - #[test] - fn should_render_symbol_as_english_for_parentheses() { - let opener: Vec = "(Hello".chars().collect(); - assert!(should_render_symbol_as_english( - true, - false, - &[], - '(', - &opener, - 0, - &[] - )); - - let korean_before: Vec = "가(".chars().collect(); - assert!(!should_render_symbol_as_english( - true, - false, - &[], - '(', - &korean_before, - 1, - &["A"] - )); - - assert!(!should_render_symbol_as_english( - false, - false, - &[], - '(', - &opener, - 0, - &[] - )); + #[rstest::rstest] + #[case::pure_roman("(Hello)", 0, &[], true, false, false, true)] + #[case::korean_before("가(", 1, &["A)"], true, false, false, false)] + #[case::indicator_disabled("(Hello)", 0, &[], false, false, false, false)] + #[case::official_rule_46_shape("BMI(체질량", 3, &["지수)"], true, true, false, false)] + #[case::roman_then_korean_body( + "SDV(Software", + 3, + &["Defined", "Vehicle,", "소프트웨어", "중심)"], + true, + true, + false, + false + )] + #[case::pure_roman_body("ABC(def)", 3, &[], true, true, false, true)] + #[case::pure_number_body("BSI(73)", 3, &[], true, true, false, true)] + #[case::unclosed_body("ABC(def", 3, &["한글"], true, true, false, true)] + #[case::nested_korean_body( + "BIT(BT(바이오)+IT(정보))", + 3, + &[], + true, + true, + false, + false + )] + #[case::rule_39_english_majority("(Korean:", 0, &["반찬)"], true, true, true, true)] + fn should_render_symbol_as_english_for_opening_parenthesis( + #[case] input: &str, + #[case] index: usize, + #[case] remaining_words: &[&str], + #[case] english_indicator: bool, + #[case] is_english: bool, + #[case] is_english_majority: bool, + #[case] expected: bool, + ) { + let word = input.chars().collect::>(); + assert_eq!( + should_render_symbol_as_english( + english_indicator, + is_english, + is_english_majority, + &[], + '(', + &word, + index, + remaining_words, + ), + expected, + ); } /// `should_render_symbol_as_english` for ')' — paren stack top 만 본다. @@ -328,7 +608,7 @@ mod tests { ) { let closer: Vec = ")".chars().collect(); assert_eq!( - should_render_symbol_as_english(true, true, &[stack_top], ')', &closer, 0, &[]), + should_render_symbol_as_english(true, true, false, &[stack_top], ')', &closer, 0, &[],), expected, ); } @@ -336,6 +616,7 @@ mod tests { /// `should_render_symbol_as_english` for ',' — 양쪽 ASCII + 영어 컨텍스트 둘 다 필요. #[rstest::rstest] #[case::both_ascii_in_english_mode("A,B", true, true)] + #[case::compatibility_unit_in_english_mode("㎿,30", true, true)] #[case::not_in_english_mode("A,B", false, false)] #[case::korean_neighbor("가,B", true, false)] fn should_render_symbol_as_english_for_comma_requires_ascii_neighbors( @@ -345,11 +626,198 @@ mod tests { ) { let word: Vec = input.chars().collect(); assert_eq!( - should_render_symbol_as_english(true, is_english, &[], ',', &word, 1, &[]), + should_render_symbol_as_english(true, is_english, false, &[], ',', &word, 1, &[],), + expected + ); + } + + #[rstest::rstest] + #[case::roman_led_chain("CV3-AD685", 3, false, true)] + #[case::roman_led_numeric_chain("N-79-20", 4, false, true)] + #[case::number_led_word("0-Zone", 1, false, false)] + #[case::number_led_suffix("777-300ER", 3, false, false)] + #[case::after_closed_enclosure("(GTX)-C", 5, false, false)] + fn hyphen_enters_roman_punctuation_only_after_a_roman_run( + #[case] input: &str, + #[case] index: usize, + #[case] is_english: bool, + #[case] expected: bool, + ) { + let chars = input.chars().collect::>(); + assert_eq!( + should_render_symbol_as_english(true, is_english, false, &[], '-', &chars, index, &[],), + expected + ); + } + + #[test] + fn punctuation_after_closed_korean_enclosure_does_not_reenter_roman_mode() { + let chars = "(XBB).1.5".chars().collect::>(); + assert!(!should_render_symbol_as_english( + true, + false, + false, + &[false], + '.', + &chars, + 5, + &[], + )); + } + + /// Korean rule 33 classifies a comma before a digit-led Korean word from + /// the complete following token. A Roman-led mixed word remains Roman + /// context at the boundary. + #[rstest::rstest] + #[case::digit_led_korean("2000년대", false)] + #[case::grouped_digit_led_korean("2,000년대", false)] + #[case::decimal_digit_led_korean("3.5년", false)] + #[case::pure_number("2000", true)] + #[case::roman_word("Beta", true)] + #[case::roman_led_mixed_word("LG유플러스", true)] + #[case::numeric_roman_unit_before_korean_particle("68kg의", true)] + fn comma_before_next_word_uses_narrow_digit_led_korean_context( + #[case] next_word: &str, + #[case] expected: bool, + ) { + let word = ['A', ',']; + assert_eq!( + should_render_symbol_as_english(true, true, false, &[], ',', &word, 1, &[next_word],), expected ); } + /// UEB 8.4.2 keeps a word-internal apostrophe inside the Roman section. + #[rstest::rstest] + #[case::official_name("O'Hara", true)] + #[case::official_contraction("DON'T", true)] + #[case::official_possessive("THAT'S", true)] + #[case::detached_open("'word", false)] + #[case::detached_close("word'", false)] + #[case::measurement("6'2", false)] + fn internal_apostrophe_requires_ascii_letters_on_both_sides( + #[case] input: &str, + #[case] expected: bool, + ) { + let word = input.chars().collect::>(); + let index = word.iter().position(|ch| *ch == '\'').unwrap(); + assert_eq!( + should_render_symbol_as_english(true, true, false, &[], '\'', &word, index, &[],), + expected, + ); + } + + #[test] + fn apostrophe_does_not_join_the_next_whitespace_delimited_word() { + let word = "Guitar'".chars().collect::>(); + assert!(!should_render_symbol_as_english( + true, + true, + false, + &[], + '\'', + &word, + word.len() - 1, + &["Listening"], + )); + } + + /// UEB 3.1.1 keeps attached Roman segments on both sides of `&` in the + /// same mode. Korean rule 35 additionally keeps a trailing number/Roman + /// continuation in that section. The left boundary still excludes a + /// number-led sequence because the cited section begins with Roman text. + #[rstest::rstest] + #[case::official_at_and_t("AT&T", true, true)] + #[case::official_b_and_b("B&B", true, true)] + #[case::spaced("A & B", true, false)] + #[case::hangul_left("가&B", true, false)] + #[case::hangul_right("A&나", true, false)] + #[case::digit_neighbor("3&B", true, false)] + #[case::digit_outer_left("3A&B", true, false)] + #[case::rule35_digit_suffix("A&B3", true, true)] + #[case::rule35_digit_then_roman_suffix("A&B3C", true, true)] + #[case::ampersand_after_digit("A&B3&C", true, false)] + #[case::multiple_ampersands("A&B&C", true, true)] + #[case::empty_segment("A&&B", true, false)] + #[case::no_roman_indicator("AT&T", false, false)] + fn attached_ampersand_requires_complete_ascii_roman_run( + #[case] input: &str, + #[case] english_indicator: bool, + #[case] expected: bool, + ) { + let word: Vec = input.chars().collect(); + let index = word.iter().position(|ch| *ch == '&').unwrap(); + assert_eq!( + should_render_symbol_as_english( + english_indicator, + true, + false, + &[], + '&', + &word, + index, + &[], + ), + expected, + ); + } + + /// UEB 3.3.1 uses one uninterrupted UEB run for official `M*A*S*H`. + /// Korean rules 32/35 preserve the same form inside a Korean document, + /// while numeric multiplication, detached marks, and empty segments remain + /// outside the Roman-asterisk grammar. + #[rstest::rstest] + #[case::official_mash_first("M*A*S*H", 1, true, true)] + #[case::official_mash_middle("M*A*S*H", 3, true, true)] + #[case::official_mash_last("M*A*S*H", 5, true, true)] + #[case::roman_number_chain("A1*B2", 2, true, true)] + #[case::number_led("2*A", 1, true, false)] + #[case::digit_only_right_segment("A*2", 1, true, false)] + #[case::empty_segment("A**B", 1, true, false)] + #[case::hangul_segment("가*A", 1, true, false)] + #[case::detached("A * B", 2, true, false)] + #[case::no_roman_indicator("M*A*S*H", 1, false, false)] + fn attached_asterisk_requires_complete_ascii_roman_segments( + #[case] input: &str, + #[case] index: usize, + #[case] english_indicator: bool, + #[case] expected: bool, + ) { + let word = input.chars().collect::>(); + assert_eq!( + should_render_symbol_as_english( + english_indicator, + true, + false, + &[], + '*', + &word, + index, + &[], + ), + expected, + ); + } + + #[rstest::rstest] + #[case::official_and_c("&c", true)] + #[case::official_at_and_t("AT&T", false)] + #[case::official_b_and_b("B&B", false)] + #[case::official_spaced("Marks & Spencer", false)] + #[case::rule35_digit_suffix("&P500", true)] + #[case::digit_without_roman_segment("&500", false)] + fn one_sided_ampersand_requires_complete_right_roman_segment( + #[case] input: &str, + #[case] expected: bool, + ) { + let word = input.chars().collect::>(); + let index = word.iter().rposition(|ch| *ch == '&').unwrap(); + assert_eq!( + is_ampersand_before_attached_ascii_roman_segment(&word, index), + expected, + ); + } + /// `has_digital_notation_signature` — `//`, `@`, `#` 강한 마커 또는 /// underscore + digital marker 조합은 true, 단순 underscore는 false. #[rstest::rstest] diff --git a/libs/braillify/src/ipa.rs b/libs/braillify/src/ipa.rs index 10705d04..5baf0f77 100644 --- a/libs/braillify/src/ipa.rs +++ b/libs/braillify/src/ipa.rs @@ -6,7 +6,7 @@ use crate::rules::context::EncodingMode; use crate::{encode, english, utils, with_encoder}; pub(crate) fn is_ipa_phonetic_symbol(c: char) -> bool { - matches!(c, 'θ' | 'ə' | 'æ' | 'ŋ' | 'ː') + matches!(c, 'θ' | 'ə' | 'æ' | 'ŋ' | 'ː' | 'ˑ') } /// PDF 제38항 자동 감지 — input의 묶음 패턴 안 IPA 음운 기호로 IPA 컨텍스트 추론. @@ -210,6 +210,7 @@ pub(crate) fn encode_ipa_char(ch: char) -> Option> { match ch { 'ə' => Some(vec![34]), // ⠢ (점 2+6) 'ː' => Some(vec![18]), // ⠒ (점 2+5) — 장음 표시 + 'ˑ' => Some(vec![16, 2]), // ⠐⠂ — 반장음 부호 (IPA 제2장) 'θ' => Some(vec![40, 57]), // ⠨⠹ (점 4+6, 점 1+4+5+6) 'ŋ' => Some(vec![43]), // ⠫ (점 1+2+4+6) 'æ' => Some(vec![41]), // ⠩ (점 1+4+6) @@ -263,6 +264,12 @@ mod tests { assert_eq!(encode_ipa_char(ch), Some(expected)); } + #[test] + fn encodes_korean_ipa_half_length_mark() { + assert_eq!(encode_ipa_char('ˑ'), Some(cells("⠐⠂"))); + assert!(detect_ipa_context("[aˑ]")); + } + #[test] fn bracket_open_flushes_prior_korean_and_strips_english_terminator() { let bracket_first = encode_ipa("[θ]").expect("initial IPA bracket should encode"); diff --git a/libs/braillify/src/lib.rs b/libs/braillify/src/lib.rs index 1b429370..e968e270 100644 --- a/libs/braillify/src/lib.rs +++ b/libs/braillify/src/lib.rs @@ -1,5 +1,54 @@ use std::{borrow::Cow, cell::RefCell}; +/// Small, semantic-neutral predicates shared with the NIKL analysis example. +/// Keeping them here lets the ordinary library test target verify analyzer +/// input boundaries without making the whole example a coverage target. +#[doc(hidden)] +pub mod corpus_analysis { + /// Whether a corpus filename belongs to the deterministic sentence shards. + pub fn is_sentence_corpus_shard_name(name: &str) -> bool { + name.starts_with("sentence_") && name.ends_with(".json") + } + + /// Whether the Unicode scalar immediately before `byte_index` is an ASCII + /// letter or digit. Callers provide a boundary from `str::char_indices`. + pub fn has_ascii_alphanumeric_before(input: &str, byte_index: usize) -> bool { + input + .get(..byte_index) + .unwrap_or_default() + .chars() + .next_back() + .is_some_and(|previous| previous.is_ascii_alphanumeric()) + } + + #[cfg(test)] + mod tests { + use super::*; + + #[rstest::rstest] + #[case::sentence_json("sentence_000.json", true)] + #[case::wrong_prefix("document_000.json", false)] + #[case::wrong_extension("sentence_000.txt", false)] + fn classifies_sentence_corpus_shard_names(#[case] name: &str, #[case] expected: bool) { + assert_eq!(is_sentence_corpus_shard_name(name), expected); + } + + #[rstest::rstest] + #[case::start_of_input("A(14)", 0, false)] + #[case::ascii_letter("BA(14)", 1, true)] + #[case::ascii_digit("1A(14)", 1, true)] + #[case::korean_scalar("가A(14)", 3, false)] + #[case::non_scalar_boundary("가A(14)", 1, false)] + fn detects_ascii_alphanumeric_immediately_before_boundary( + #[case] input: &str, + #[case] byte_index: usize, + #[case] expected: bool, + ) { + assert_eq!(has_ascii_alphanumeric_before(input, byte_index), expected); + } + } +} + mod char_shortcut; pub(crate) mod char_struct; #[cfg(feature = "cli")] @@ -47,6 +96,7 @@ mod test_helpers { pub result: Vec, pub prev_word: String, pub remaining_words: Vec, + pub roman_section_continues_from_previous_word: bool, } impl CtxOwned { @@ -67,6 +117,7 @@ mod test_helpers { result: Vec::new(), prev_word: String::new(), remaining_words: Vec::new(), + roman_section_continues_from_previous_word: false, } } @@ -76,6 +127,13 @@ mod test_helpers { self } + /// Mark this print word as a continuation of an already active Roman + /// section. + pub(crate) fn with_roman_section_continuation(mut self) -> Self { + self.roman_section_continues_from_previous_word = true; + self + } + /// Builder: set the `remaining_words` field that the borrowed `RuleContext` /// exposes. Stores owned strings so the borrowed context can outlive call sites. pub(crate) fn with_remaining_words(mut self, words: I) -> Self @@ -112,6 +170,8 @@ mod test_helpers { }), is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: self + .roman_section_continues_from_previous_word, skip_count: &mut self.skip_count, state: &mut self.state, result: &mut self.result, @@ -256,6 +316,130 @@ fn normalize_math_alphanumeric_string(text: &str) -> Cow<'_, str> { Cow::Owned(text.chars().map(normalize_math_alphanumeric_char).collect()) } +fn may_normalize_roman_numeral_presentation(c: char) -> bool { + (0x2160..=0x217f).contains(&(c as u32)) +} + +fn may_normalize_parenthesized_hangul_presentation(c: char) -> bool { + (0x3200..=0x321e).contains(&(c as u32)) +} + +fn may_normalize_word_separator_middle_dot(c: char) -> bool { + c == '\u{2e31}' +} + +fn pure_roman_compatibility_unit_decomposition(c: char) -> Option> { + use unicode_normalization::UnicodeNormalization; + + let parts = + crate::rules::korean::rule_69::compatibility_unit_decomposition(c).or_else(|| { + crate::rules::korean::rule_68::is_rule_68_symbol(c) + .then(|| std::iter::once(c).nfkc().collect()) + })?; + parts.iter().all(char::is_ascii_alphabetic).then_some(parts) +} + +/// Korean Braille rule 36 transcribes a Roman numeral with its corresponding +/// Roman letters. Unicode U+2160–U+217F are presentation forms whose NFKC +/// decomposition is exactly that Roman-letter spelling (`Ⅱ` → `II`). Normalize +/// only this block; the existing rule-36 token logic remains responsible for +/// numeral validity, case indicators, context, and Roman termination. +fn normalize_roman_numeral_presentation<'a>(text: Cow<'a, str>) -> Cow<'a, str> { + use unicode_normalization::UnicodeNormalization; + + let mut out = String::with_capacity(text.len()); + for ch in text.chars() { + if may_normalize_roman_numeral_presentation(ch) { + out.extend(std::iter::once(ch).nfkc()); + } else { + out.push(ch); + } + } + Cow::Owned(out) +} + +/// Unicode U+3200-U+321E are compatibility presentation forms whose visible +/// content is ordinary Hangul enclosed by literal parentheses (`㈜` -> `(주)`, +/// `㈔` -> `(사)`). The Korean braille standard already defines both the +/// enclosed Hangul and the parentheses; expanding the presentation form lets +/// those existing rules own the transcription without assigning a new braille +/// symbol to each Unicode glyph. +fn normalize_parenthesized_hangul_presentation<'a>(text: Cow<'a, str>) -> Cow<'a, str> { + use unicode_normalization::UnicodeNormalization; + + let mut out = String::with_capacity(text.len()); + for ch in text.chars() { + if may_normalize_parenthesized_hangul_presentation(ch) { + out.extend(std::iter::once(ch).nfkc()); + } else { + out.push(ch); + } + } + Cow::Owned(out) +} + +/// U+2E31 WORD SEPARATOR MIDDLE DOT is a visible presentation of a word +/// boundary, not U+00B7 MIDDLE DOT punctuation. Preserve that semantic +/// distinction by expanding it to one ordinary print space before tokenization; +/// the existing Korean spacing rules then emit one blank braille cell. +fn normalize_word_separator_middle_dot<'a>(text: Cow<'a, str>) -> Cow<'a, str> { + let mut out = String::with_capacity(text.len()); + for ch in text.chars() { + if may_normalize_word_separator_middle_dot(ch) { + out.push(' '); + } else { + out.push(ch); + } + } + Cow::Owned(out) +} + +/// Korean Braille rule 69 assigns Unicode compatibility unit glyphs the +/// transcription of their semantic Roman spelling. When such a glyph is +/// immediately combined with ordinary Roman letters (`㎾h` -> `kWh`) or with +/// another unit component (`W/㎏` -> `W/kg`), encoding it as a self-contained +/// symbol would incorrectly close and reopen the Roman section at the Unicode +/// code-point boundary. +/// +/// Expand only unit glyphs whose complete NFKC decomposition consists of Roman +/// letters *and* which are joined to another Roman unit component. Standalone +/// compatibility units retain their dedicated rule-68/69 encoding. +/// Compatibility forms containing a slash or an exponent (`㎧`, `㎥`) and +/// non-unit compatibility characters remain untouched. +fn normalize_pure_roman_compatibility_units<'a>(text: Cow<'a, str>) -> Cow<'a, str> { + let chars = text.chars().collect::>(); + let mut out = String::with_capacity(text.len()); + for (index, ch) in chars.iter().copied().enumerate() { + let is_roman_unit_component = |candidate: char| { + candidate.is_ascii_alphabetic() + || candidate == 'μ' + || pure_roman_compatibility_unit_decomposition(candidate).is_some() + }; + let directly_joined = index + .checked_sub(1) + .and_then(|previous| chars.get(previous)) + .is_some_and(|previous| is_roman_unit_component(*previous)) + || chars + .get(index + 1) + .is_some_and(|next| is_roman_unit_component(*next)); + let joined_through_slash = (index >= 2 + && matches!(chars[index - 1], '/' | '\u{2044}' | '\u{2215}') + && is_roman_unit_component(chars[index - 2])) + || (index + 2 < chars.len() + && matches!(chars[index + 1], '/' | '\u{2044}' | '\u{2215}') + && is_roman_unit_component(chars[index + 2])); + + if (directly_joined || joined_through_slash) + && let Some(parts) = pure_roman_compatibility_unit_decomposition(ch) + { + out.extend(parts); + } else { + out.push(ch); + } + } + Cow::Owned(out) +} + /// Default-route whole expressions that contain math-only relational/grouping /// glyphs which cannot be encoded correctly one space-separated token at a time. /// @@ -276,7 +460,12 @@ fn default_math_expression_needs_whole_route(text: &str) -> bool { let has_operand = chars.iter().any(|c| c.is_ascii_alphanumeric()); has_operand && chars.iter().enumerate().any(|(i, c)| match *c { - '→' | '←' | '↗' | '↘' | '↑' | '↓' | '△' | '□' => true, + // 수학 제32·33항의 합동/기하 연산 기호도 양쪽 변수를 포함한 + // 하나의 수식이다. 공백 단위 token 경로로 나누면 뒤쪽 대문자 + // 변수가 국어 제29항의 로마자 연속으로 오인될 수 있다. + '→' | '←' | '↗' | '↘' | '↑' | '↓' | '△' | '□' | '≅' | '▷' | '◁' => { + true + } // 수학 제34/37항 hat/bar 결합부호는 단일 문자 operand에 붙는다 // (`x̂`, `x̄`, `p̂`, `2̄.3010`). NFD 분해된 악센트 단어(`maître` → // `mai`+◌̂+`tre`)처럼 결합부호가 3글자 이상 단어 내부에 있으면 @@ -318,6 +507,10 @@ fn combining_mark_on_single_letter(chars: &[char], i: usize) -> bool { #[derive(Clone, Copy, Default)] struct NormalizationTriggers { has_math_alphanumeric: bool, + has_roman_numeral_presentation: bool, + has_parenthesized_hangul_presentation: bool, + has_word_separator_middle_dot: bool, + has_pure_roman_compatibility_unit: bool, has_decomposable_latin: bool, has_negation_combiner: bool, has_vector_mark: bool, @@ -331,6 +524,12 @@ impl NormalizationTriggers { let mut triggers = Self::default(); for c in text.chars() { triggers.has_math_alphanumeric |= may_normalize_math_alphanumeric(c); + triggers.has_roman_numeral_presentation |= may_normalize_roman_numeral_presentation(c); + triggers.has_parenthesized_hangul_presentation |= + may_normalize_parenthesized_hangul_presentation(c); + triggers.has_word_separator_middle_dot |= may_normalize_word_separator_middle_dot(c); + triggers.has_pure_roman_compatibility_unit |= + pure_roman_compatibility_unit_decomposition(c).is_some(); triggers.has_decomposable_latin |= may_decompose_accented_latin(c); triggers.has_negation_combiner |= c == '\u{0338}'; triggers.has_vector_mark |= is_vector_mark(c); @@ -722,6 +921,26 @@ pub fn encode_with_options(text: &str, options: &EncodeOptions) -> Result Result= 3 + && text.starts_with('$') + && text.ends_with('$') + && text.matches('$').count() == 2 + { + let inner = &text[1..text.len() - 1]; + return crate::rules::token_rules::latex_math::encode_latex_math_bytes_with_context( + inner, + math_context, + ); + } + let chars: Vec = text.chars().collect(); // PDF 수학 제12항: 단일 ASCII lowercase = 영자표시 ⠴(52) + 알파벳 점자. @@ -1947,6 +2185,134 @@ mod coverage_targeted_tests { assert_eq!(normalize_math_alphanumeric_char(input), expected); } + #[rstest::rstest] + #[case::upper_one("Ⅰ", "I")] + #[case::upper_two("Ⅱ", "II")] + #[case::upper_seven("Ⅶ", "VII")] + #[case::lower_four("ⅳ", "iv")] + #[case::embedded("제Ⅲ장", "제III장")] + fn normalizes_roman_numeral_presentation(#[case] input: &str, #[case] expected: &str) { + assert_eq!( + normalize_roman_numeral_presentation(Cow::Borrowed(input)), + expected + ); + } + + /// Rule 36 spells Roman numerals with Roman letters. Presentation forms must + /// therefore enter the same existing encoder path in spaced, attached, + /// particle-adjacent, and lower-case contexts. + #[rstest::rstest] + #[case::pdf_sentence( + "가영이는 미적분학 Ⅱ 과목을 수강하고 있다.", + "가영이는 미적분학 II 과목을 수강하고 있다." + )] + #[case::attached_chapter("제Ⅲ장", "제III장")] + #[case::adjacent_particle("Ⅶ을", "VII을")] + #[case::lowercase_indicator("ⅳ를", "iv를")] + fn unicode_roman_numeral_matches_ascii_rule_36_path( + #[case] presentation: &str, + #[case] ascii: &str, + ) { + assert_eq!(encode_to_unicode(presentation), encode_to_unicode(ascii)); + } + + #[test] + fn roman_numeral_normalization_leaves_other_nfkc_characters_unchanged() { + let input = "ↀ㈜"; + assert_eq!( + normalize_roman_numeral_presentation(Cow::Borrowed(input)), + input + ); + } + + #[rstest::rstest] + #[case::parenthesized_jamo("㈀", "(ᄀ)")] + #[case::parenthesized_syllable("㈎", "(가)")] + #[case::incorporated_association("㈔", "(사)")] + #[case::incorporated_company("㈜", "(주)")] + #[case::afternoon("㈞", "(오후)")] + fn normalizes_parenthesized_hangul_presentation(#[case] input: &str, #[case] expected: &str) { + assert_eq!( + normalize_parenthesized_hangul_presentation(Cow::Borrowed(input)), + expected + ); + } + + /// The compatibility glyph carries no independent braille semantics: its + /// expanded print-equivalent must follow the ordinary Korean parenthesis + /// and Hangul rules in every surrounding position. + #[rstest::rstest] + #[case::association_prefix("㈔한국", "(사)한국")] + #[case::company_prefix("㈜한빛", "(주)한빛")] + #[case::attached_company_suffix("한빛㈜", "한빛(주)")] + fn parenthesized_hangul_presentation_matches_expanded_print( + #[case] presentation: &str, + #[case] expanded: &str, + ) { + assert_eq!(encode_to_unicode(presentation), encode_to_unicode(expanded)); + } + + #[test] + fn normalizes_word_separator_middle_dot_to_print_space() { + assert_eq!( + normalize_word_separator_middle_dot(Cow::Borrowed("인증⸱실천⸱교육")), + "인증 실천 교육" + ); + } + + #[test] + fn word_separator_middle_dot_matches_visible_word_spacing() { + assert_eq!( + encode_to_unicode("인증⸱실천⸱교육"), + encode_to_unicode("인증 실천 교육") + ); + } + + #[rstest::rstest] + #[case::kilowatt_hour("㎾h", "kWh")] + #[case::milli_sievert("m㏜", "mSv")] + #[case::watt_per_kilogram("W/㎏", "W/kg")] + #[case::kilogram_carbon_equivalent("㎏CO2eq", "kgCO2eq")] + #[case::milligram_per_gram("㎎/g", "mg/g")] + fn normalizes_pure_roman_compatibility_unit_components( + #[case] input: &str, + #[case] expected: &str, + ) { + assert_eq!( + normalize_pure_roman_compatibility_units(Cow::Borrowed(input)), + expected + ); + } + + #[rstest::rstest] + #[case::superscript("㎥")] + #[case::quotient("㎧")] + #[case::standalone_hectare("㏊")] + #[case::non_unit_compatibility_abbreviation("㏚")] + fn pure_roman_unit_normalization_preserves_other_compatibility_forms(#[case] input: &str) { + assert_eq!( + normalize_pure_roman_compatibility_units(Cow::Borrowed(input)), + input + ); + } + + /// Rule 69: a compatibility unit presentation and its semantic Roman + /// spelling are one unit section even when joined to another component. + #[rstest::rstest] + #[case::kilowatt_hour("용량은 1㎾h이다", "용량은 1kWh이다")] + #[case::milli_sievert("선량은 1m㏜보다 낮다", "선량은 1mSv보다 낮다")] + #[case::watt_per_kilogram("기준은 4.0W/㎏이다", "기준은 4.0W/kg이다")] + fn compound_compatibility_units_match_semantic_roman_spelling( + #[case] presentation: &str, + #[case] expanded: &str, + ) { + assert_eq!( + encode_to_unicode(presentation), + encode_to_unicode(expanded), + "presentation={presentation:?}" + ); + } + #[test] fn normalize_math_alphanumeric_runtime_block_offset() { let input = std::hint::black_box('\u{1D44F}'); @@ -2115,6 +2481,19 @@ mod coverage_targeted_tests { assert!(result.is_ok()); } + /// 수학 제32·33항: 수학 전용 관계 기호 양쪽의 대문자는 하나의 수식 + /// 안의 변수다. 뒤쪽 변수를 국어 로마자 연속 항목으로 보아 ⠰를 붙이지 않는다. + #[rstest::rstest] + #[case::congruence("A ≅ B", "⠠⠁⠀⠈⠔⠒⠒⠀⠠⠃")] + #[case::right_geometric_operation("G ▷ N", "⠠⠛⠀⠸⠜⠀⠠⠝")] + #[case::left_geometric_operation("N ◁ G", "⠠⠝⠀⠸⠣⠀⠠⠛")] + fn default_route_keeps_math_relation_operands_in_one_expression( + #[case] input: &str, + #[case] expected: &str, + ) { + assert_eq!(encode_to_unicode(input).as_deref(), Ok(expected)); + } + /// Math mode — multi-char expression with spaces around operators. /// Covers the whitespace-cleaning loop (lines 777-790). #[test] diff --git a/libs/braillify/src/rules/context.rs b/libs/braillify/src/rules/context.rs index 7c957021..0f26671c 100644 --- a/libs/braillify/src/rules/context.rs +++ b/libs/braillify/src/rules/context.rs @@ -87,6 +87,11 @@ pub struct EncoderState { pub needs_english_continuation: bool, /// Rule 35 chain: English followed by digits may resume English without indicators pub roman_number_chain: bool, + /// The active Roman section is an English phrase rather than metalinguistic + /// Roman material in a Korean sentence. Korean rule 37 keeps its six lower + /// wordsigns expanded in Korean context, while UEB 10.5 applies in an + /// independently recognizable English phrase. + pub roman_section_is_english_context: bool, /// Stack tracking whether parentheses were opened in English context pub parenthesis_stack: Vec, /// Currently in a number sequence (수표 already emitted) @@ -122,6 +127,7 @@ impl EncoderState { has_processed_word: false, needs_english_continuation: false, roman_number_chain: false, + roman_section_is_english_context: false, parenthesis_stack: Vec::new(), is_number: false, is_big_english: false, @@ -178,6 +184,13 @@ pub struct RuleContext<'a> { pub is_all_uppercase: bool, /// Whether ASCII letters start at index 0 pub ascii_starts_at_beginning: bool, + /// Whether a Roman section was already active before the current print word. + /// + /// Korean rule 37 suppresses whole-word contractions only for the first + /// Roman word after the Roman indicator. The emitter may enter Roman mode + /// before character rules run, so `state.is_english` alone cannot distinguish + /// that first word from a later word in the same section. + pub roman_section_continues_from_previous_word: bool, /// Skip count — rules can set this to skip subsequent characters pub skip_count: &'a mut usize, /// Shared mutable encoder state diff --git a/libs/braillify/src/rules/emit.rs b/libs/braillify/src/rules/emit.rs index 69ee7d52..19192e56 100644 --- a/libs/braillify/src/rules/emit.rs +++ b/libs/braillify/src/rules/emit.rs @@ -19,6 +19,112 @@ struct WordContext<'a> { remaining_words: &'a [&'a str], } +/// Rule 29/35: a following print word which begins with Roman text or a number +/// continues the same Roman section across its intervening print space. +fn next_word_starts_roman_or_number(remaining_words: &[&str]) -> bool { + remaining_words + .first() + .and_then(|word| word.chars().next()) + .is_some_and(|ch| ch.is_ascii_alphanumeric()) +} + +fn is_opening_english_phrase_enclosure(ch: char) -> bool { + matches!(ch, '(' | '[' | '{' | '‘' | '“' | '"') +} + +fn is_closing_english_phrase_enclosure(ch: char) -> bool { + matches!(ch, ')' | ']' | '}' | '’' | '”' | '"') +} + +/// Decide whether the Roman section beginning in `tokens[start_index]` has an +/// independently visible English-phrase context. +/// +/// Korean rule 37 expands the six UEB lower wordsigns when Roman material is +/// mentioned inside Korean prose. The NIKL's rule consultation distinguishes +/// that case from an English title or sentence, where UEB 10.5 applies. We use +/// only print structure available to a plain-text encoder: at least two Roman +/// print words plus either sentence/title capitalization or a paired-enclosure +/// boundary. A lowercase metalinguistic list such as the rule-37 attachment +/// (`be, his, was, were의 ...`) therefore remains Korean context. +fn roman_section_has_english_phrase_context(tokens: &[Token<'_>], start_index: usize) -> bool { + let mut started = false; + let mut roman_word_count = 0usize; + let mut has_uppercase = false; + let mut has_enclosure_boundary = false; + + for token in tokens.iter().skip(start_index) { + let Token::Word(word) = token else { + if matches!(token, Token::Space(_) | Token::Mode(_)) { + continue; + } + if started { + break; + } + continue; + }; + + let scan_start = if started { + let Some(first_script) = word + .chars + .iter() + .position(|ch| ch.is_ascii_alphanumeric() || crate::utils::is_korean_char(*ch)) + else { + continue; + }; + if crate::utils::is_korean_char(word.chars[first_script]) { + break; + } + let Some(first_roman) = word.chars[first_script..] + .iter() + .position(|ch| ch.is_ascii_alphabetic()) + .map(|offset| first_script + offset) + else { + continue; + }; + first_roman + } else { + let Some(first_roman) = word.chars.iter().position(|ch| ch.is_ascii_alphabetic()) + else { + continue; + }; + first_roman + }; + + if !started { + has_enclosure_boundary |= word.chars[..scan_start] + .iter() + .rev() + .take_while(|ch| !ch.is_ascii_alphanumeric() && !crate::utils::is_korean_char(**ch)) + .any(|ch| is_opening_english_phrase_enclosure(*ch)); + } + + let section_end = word.chars[scan_start..] + .iter() + .position(|ch| crate::utils::is_korean_char(*ch)) + .map_or(word.chars.len(), |offset| scan_start + offset); + let roman_slice = &word.chars[scan_start..section_end]; + // `scan_start` is selected from an ASCII alphabetic position above, so + // this slice necessarily contains at least that Roman letter. + let last_roman = roman_slice + .iter() + .rposition(|ch| ch.is_ascii_alphabetic()) + .expect("Roman section starts at an ASCII alphabetic character"); + + started = true; + roman_word_count += 1; + has_uppercase |= roman_slice.iter().any(|ch| ch.is_ascii_uppercase()); + has_enclosure_boundary |= roman_slice[last_roman + 1..] + .iter() + .any(|ch| is_closing_english_phrase_enclosure(*ch)); + + if section_end < word.chars.len() { + break; + } + } + + roman_word_count >= 2 && (has_uppercase || has_enclosure_boundary) +} + /// 토큰의 byte 슬라이스가 한글표(⠸⠷) 점형과 일치하는지. fn is_hangul_wrap_start(token: &Token<'_>) -> bool { matches!(token, Token::PreEncoded(bytes) if bytes.as_slice() == HANGUL_WRAP_START_BYTES) @@ -88,6 +194,187 @@ fn token_is_math_word(token: Option<&Token<'_>>) -> bool { } } +/// Find the word governed by a run of UEB grade-1/capital mode markers. +/// +/// Korean rule 29 requires the roman indicator before the roman text, while +/// rule 28 appendix places UEB capitalization indicators immediately before +/// the capitalized roman word. Token rewriting may discover capitalization +/// before the character emitter discovers a new roman section, so the emitter +/// must establish roman mode before it emits these UEB prefix markers. +fn roman_word_after_prefix<'a>( + tokens: &'a [Token<'a>], + prefix_index: usize, +) -> Option<&'a WordToken<'a>> { + for token in tokens.iter().skip(prefix_index + 1) { + match token { + Token::Mode( + ModeEvent::Grade1Indicator | ModeEvent::CapsWord | ModeEvent::CapsPassageStart, + ) => continue, + Token::Word(word) => return Some(word), + _ => return None, + } + } + None +} + +fn current_word_at_or_after<'a>( + tokens: &'a [Token<'a>], + index: usize, +) -> Option<&'a WordToken<'a>> { + for token in tokens.iter().skip(index) { + match token { + Token::Mode(_) => continue, + Token::Word(word) => return Some(word), + _ => return None, + } + } + None +} + +fn is_separated_from_previous_word(tokens: &[Token<'_>], index: usize) -> bool { + let mut saw_space = false; + for token in tokens[..index].iter().rev() { + match token { + Token::Mode(_) => {} + Token::Space(_) => saw_space = true, + Token::Word(_) => return saw_space, + _ => return false, + } + } + false +} + +fn previous_word_index_before(tokens: &[Token<'_>], index: usize) -> Option { + tokens[..index] + .iter() + .rposition(|token| matches!(token, Token::Word(_))) +} + +fn matching_group_open(close: char) -> Option { + match close { + ')' => Some('('), + ']' => Some('['), + '}' => Some('{'), + '’' => Some('‘'), + '”' => Some('“'), + '〉' => Some('〈'), + '》' => Some('《'), + '」' => Some('「'), + '』' => Some('『'), + '】' => Some('【'), + '〕' => Some('〔'), + '〗' => Some('〖'), + '〙' => Some('〘'), + '〛' => Some('〚'), + _ => None, + } +} + +fn closed_enclosure_before_contains_ascii(tokens: &[Token<'_>], index: usize) -> bool { + let Some(previous_index) = previous_word_index_before(tokens, index) else { + return false; + }; + let Token::Word(previous) = &tokens[previous_index] else { + unreachable!("previous_word_index_before returns a Word token"); + }; + + let mut end = previous.chars.len(); + while end > 0 && matches!(previous.chars[end - 1], ',' | ':' | ';' | '.' | '!' | '?') { + end -= 1; + } + let Some(&closer) = previous.chars.get(end.saturating_sub(1)) else { + return false; + }; + let Some(opener) = matching_group_open(closer) else { + return false; + }; + + let mut nesting = 1usize; + let mut contains_ascii = false; + for token_index in (0..=previous_index).rev() { + let Token::Word(word) = &tokens[token_index] else { + if matches!(tokens[token_index], Token::Space(_) | Token::Mode(_)) { + continue; + } + return false; + }; + let word_end = if token_index == previous_index { + end - 1 + } else { + word.chars.len() + }; + for &ch in word.chars[..word_end].iter().rev() { + if ch == closer { + nesting += 1; + } else if ch == opener { + nesting -= 1; + if nesting == 0 { + return contains_ascii; + } + } else if ch.is_ascii_alphabetic() { + contains_ascii = true; + } + } + } + false +} + +/// Rule 34's closing enclosure ends the enclosed Roman item without a Roman +/// terminator. If the next whitespace-delimited item starts directly with +/// Roman text, rule 29 opens a new Roman section. A following enclosure is +/// excluded so the official rule-32 list `(a), (e), (i)` remains one Roman +/// section and may use UEB grade-1 indicators for its single letters. +fn starts_new_roman_section_after_closed_enclosure(tokens: &[Token<'_>], index: usize) -> bool { + if !is_separated_from_previous_word(tokens, index) { + return false; + } + let Some(current) = current_word_at_or_after(tokens, index) else { + return false; + }; + + let current_starts_roman = current + .chars + .iter() + .copied() + .find(|ch| !matches!(ch, '‘' | '“' | '\'' | '"')) + .is_some_and(|ch| ch.is_ascii_alphabetic()); + if !current_starts_roman { + return false; + } + closed_enclosure_before_contains_ascii(tokens, index) +} + +fn enter_roman_before_ueb_prefix( + tokens: &[Token<'_>], + prefix_index: usize, + event: ModeEvent, + state: &mut EncoderState, + result: &mut Vec, +) { + let is_ueb_prefix = matches!( + event, + ModeEvent::Grade1Indicator | ModeEvent::CapsWord | ModeEvent::CapsPassageStart + ); + let roman_word = + roman_word_after_prefix(tokens, prefix_index).filter(|word| word.meta.starts_with_ascii); + + if is_ueb_prefix + && state.english_indicator + && !state.is_english + && let Some(word) = roman_word + { + // Use the shared rule-29/35 transition so a capital word after a number + // in the same roman section (`KBS 1 TV`) resumes without a second + // roman indicator, while a genuinely new section receives one. + roman_mode::enter_english_if_starting( + state, + &word.chars, + word.meta.has_ascii_alphabetic, + result, + ); + } +} + /// PDF 수학 — `Word(math)+Space+Word(=/==/관계)+Space+Word(math)` 패턴에서 /// 등호 양옆 Space 토큰을 묵음 처리한다. 점역 결과는 `expr⠒⠒expr`로 인접한다. fn is_math_operator_space_suppression<'a>(tokens: &'a [Token<'a>], space_idx: usize) -> bool { @@ -170,7 +457,35 @@ pub fn emit(ir: &mut DocumentIR, char_engine: &mut RuleEngine) -> Result result.push(0); } } - Token::Mode(event) => emit_mode_event(*event, &mut ir.state, &mut result), + Token::Mode(event) => { + let starts_new_roman_section = + starts_new_roman_section_after_closed_enclosure(&ir.tokens, idx); + let event = if *event == ModeEvent::EnterEnglishContinue && starts_new_roman_section + { + ModeEvent::EnterEnglish + } else { + *event + }; + if starts_new_roman_section { + ir.state.needs_english_continuation = false; + } + let opens_fresh_roman_section = !ir.state.is_english + && !ir.state.roman_number_chain + && !ir.state.needs_english_continuation + && matches!( + event, + ModeEvent::EnterEnglish + | ModeEvent::Grade1Indicator + | ModeEvent::CapsWord + | ModeEvent::CapsPassageStart + ); + if opens_fresh_roman_section { + ir.state.roman_section_is_english_context = + roman_section_has_english_phrase_context(&ir.tokens, idx); + } + enter_roman_before_ueb_prefix(&ir.tokens, idx, event, &mut ir.state, &mut result); + emit_mode_event(event, &mut ir.state, &mut result); + } Token::Fraction(frac) => { if let Some(ref w) = frac.whole { result.extend(fraction::encode_mixed_fraction( @@ -238,10 +553,230 @@ fn word_context<'a>(word_texts: &'a [&'a str], word_index: usize) -> WordContext } } +/// Whether the next word token is separated from the current word by print +/// whitespace. Token rewrites can insert mode/pre-encoded tokens between the +/// two words, so inspect the whole intervening token span rather than only the +/// immediate successor. +fn has_space_before_next_word(tokens: &[Token<'_>], token_index: usize) -> bool { + let mut saw_space = false; + for token in tokens.iter().skip(token_index + 1) { + match token { + Token::Space(_) => saw_space = true, + Token::Word(_) => return saw_space, + _ => {} + } + } + false +} + +fn matching_group_close(ch: char) -> Option { + match ch { + '(' => Some(')'), + '[' => Some(']'), + '{' => Some('}'), + '〈' => Some('〉'), + '《' => Some('》'), + '「' => Some('」'), + '『' => Some('』'), + '【' => Some('】'), + '〔' => Some('〕'), + '〖' => Some('〗'), + '〘' => Some('〙'), + '〚' => Some('〛'), + '‘' => Some('’'), + '“' => Some('”'), + _ => None, + } +} + +/// Rule 33's printed `Umm ...이라고` example treats a whitespace-separated +/// ellipsis as the punctuation ending the Roman run, so the preceding Roman +/// terminator is still omitted. Accept the Unicode ellipsis forms and the +/// three-full-stop print spelling as the same punctuation grammar. +fn word_starts_with_rule_33_ellipsis(word: &WordToken<'_>) -> bool { + matches!(word.chars.first(), Some('…' | '⋯')) || word.chars.starts_with(&['.', '.', '.']) +} + +/// Korean rules 29, 32, and 35: print whitespace around a colon does not split +/// a Roman section when the colon is followed by another Roman/number item. +/// The tokenizer represents `Alpha : Beta` as three words, so prove the item +/// after the standalone colon before treating the colon as UEB punctuation. +fn spaced_colon_connects_roman_items(tokens: &[Token<'_>], colon_index: usize) -> bool { + let Some(Token::Word(colon)) = tokens.get(colon_index) else { + return false; + }; + if colon.chars.as_slice() != [':'] { + return false; + } + + tokens + .iter() + .skip(colon_index + 1) + .find_map(|token| match token { + Token::Space(_) | Token::Mode(_) => None, + Token::Word(word) => Some( + word.chars + .iter() + .find(|ch| ch.is_ascii_alphanumeric() || crate::utils::is_korean_char(**ch)) + .is_some_and(|ch| ch.is_ascii_alphanumeric()), + ), + _ => Some(false), + }) + .unwrap_or(false) +} + +/// UEB 3.1.1 and Korean rule 29 keep an ampersand inside a spaced Roman name +/// or phrase (`Marks & Spencer`, `Scan & Solution`). The tokenizer makes the +/// ampersand its own word, so prove a Roman word on both sides before allowing +/// it to bridge the current Roman section. A right-hand word may be attached +/// directly to the sign (`Mining &Development`). +fn spaced_ampersand_connects_roman_words(tokens: &[Token<'_>], ampersand_index: usize) -> bool { + let Some(Token::Word(ampersand)) = tokens.get(ampersand_index) else { + return false; + }; + if ampersand.chars.first() != Some(&'&') { + return false; + } + + let left_is_roman = tokens[..ampersand_index] + .iter() + .rev() + .find_map(|token| match token { + Token::Space(_) | Token::Mode(_) => None, + Token::Word(word) => Some( + word.chars + .iter() + .rev() + .find(|ch| ch.is_ascii_alphanumeric() || crate::utils::is_korean_char(**ch)) + .is_some_and(|ch| ch.is_ascii_alphanumeric()), + ), + _ => Some(false), + }) + .unwrap_or(false); + if !left_is_roman { + return false; + } + + if ampersand + .chars + .get(1) + .is_some_and(|ch| ch.is_ascii_alphabetic()) + { + return true; + } + if ampersand.chars.len() != 1 { + return false; + } + + tokens + .iter() + .skip(ampersand_index + 1) + .find_map(|token| match token { + Token::Space(_) | Token::Mode(_) => None, + Token::Word(word) => Some( + word.chars + .iter() + .find(|ch| ch.is_ascii_alphanumeric() || crate::utils::is_korean_char(**ch)) + .is_some_and(|ch| ch.is_ascii_alphabetic()), + ), + _ => Some(false), + }) + .unwrap_or(false) +} + +/// Rule 29 keeps consecutive Roman/number text in one section even across +/// print spaces. A separated enclosure continues that section only when the +/// *complete* enclosure is Roman/number text. This distinguishes +/// `GRI (Global Reporting Initiative)` from `Poison (모래성)` and from a mixed +/// gloss such as `TVB (Television - 전시광파유한공사)`. +fn separated_symbol_continues_roman_section(tokens: &[Token<'_>], token_index: usize) -> bool { + let next_word = tokens + .iter() + .enumerate() + .skip(token_index + 1) + .find_map(|(index, token)| match token { + Token::Word(word) => Some((index, word)), + _ => None, + }); + let Some((next_word_index, next_word)) = next_word else { + return false; + }; + + if next_word.chars.first() == Some(&'&') + && spaced_ampersand_connects_roman_words(tokens, next_word_index) + { + return true; + } + + if word_starts_with_rule_33_ellipsis(next_word) { + return true; + } + + if spaced_colon_connects_roman_items(tokens, next_word_index) { + return true; + } + + // Rule 35: punctuation may introduce a numeric continuation (`'23`). + if next_word + .chars + .iter() + .find(|ch| ch.is_ascii_alphanumeric() || crate::utils::is_korean_char(**ch)) + .is_some_and(char::is_ascii_digit) + { + return true; + } + + let Some(opening) = next_word.chars.first().copied() else { + return false; + }; + let Some(closing) = matching_group_close(opening) else { + return false; + }; + + let mut depth = 0usize; + let mut saw_roman_or_number = false; + let mut saw_korean = false; + for token in tokens.iter().skip(token_index + 1) { + match token { + Token::Space(_) | Token::Mode(_) => continue, + Token::Word(word) => { + for ch in word.chars.iter().copied() { + if ch == opening { + depth += 1; + continue; + } + if ch == closing { + // The first scanned character is the matching opener, + // and the function returns as soon as that level closes. + depth -= 1; + if depth == 0 { + return saw_roman_or_number && !saw_korean; + } + continue; + } + if depth > 0 { + saw_roman_or_number |= ch.is_ascii_alphanumeric(); + saw_korean |= crate::utils::is_korean_char(ch); + } + } + } + _ => return false, + } + } + false +} + fn emit_mode_event(event: ModeEvent, state: &mut EncoderState, result: &mut Vec) { match event { ModeEvent::EnterEnglish => { - result.push(52); + // Korean rule 29 uses one Roman section for consecutive Roman + // text. Token-level capitalization can discover a later word and + // request entry again after the character emitter has already kept + // that section open; make the explicit event idempotent at the + // authoritative emit-state boundary. + if !state.is_english { + result.push(52); + } state.is_english = true; state.needs_english_continuation = false; state.roman_number_chain = false; @@ -284,6 +819,7 @@ fn apply_core_encoding_rules( is_all_uppercase: bool, has_korean_char: bool, ascii_starts_at_beginning: bool, + roman_section_continues_from_previous_word: bool, state: &mut EncoderState, skip_count: &mut usize, remaining_words: &[&str], @@ -299,6 +835,7 @@ fn apply_core_encoding_rules( has_korean_char, is_all_uppercase, ascii_starts_at_beginning, + roman_section_continues_from_previous_word, skip_count, state, result, @@ -315,12 +852,13 @@ fn apply_inter_character_rules( is_all_uppercase: bool, has_korean_char: bool, ascii_starts_at_beginning: bool, + roman_section_continues_from_previous_word: bool, state: &mut EncoderState, skip_count: &mut usize, remaining_words: &[&str], prev_word: &str, result: &mut Vec, -) -> Result<(), String> { +) -> Result { let mut ctx = RuleContext { word_chars, index, @@ -330,12 +868,12 @@ fn apply_inter_character_rules( has_korean_char, is_all_uppercase, ascii_starts_at_beginning, + roman_section_continues_from_previous_word, skip_count, state, result, }; - engine.apply_phase(Phase::InterCharacter, &mut ctx)?; - Ok(()) + engine.apply_phase(Phase::InterCharacter, &mut ctx) } fn emit_word( @@ -349,6 +887,7 @@ fn emit_word( ) -> Result<(), String> { let prev_word = context.prev_word; let remaining_words = context.remaining_words; + let next_word_is_separated = has_space_before_next_word(all_tokens, token_index); // 다음 비공백 토큰이 한글표(⠸⠷)이면 영어 모드를 끊지 않는다 (제39항). let next_is_hangul_wrap = next_non_space_is_hangul_wrap_start(all_tokens, token_index); // 직전 비공백 토큰이 한글 종료표(⠸⠾)이면 이 토큰의 시작 문장부호도 @@ -366,15 +905,42 @@ fn emit_word( let has_ascii_alphabetic = meta.has_ascii_alphabetic; if word_chars.first().is_some_and(|ch| ch.is_ascii_digit()) - && let Some((numeric, unit, consumed)) = parse_numeric_ascii_unit_prefix(word_chars) + && !state.is_english + && let Some((numeric, mut unit, consumed)) = parse_numeric_ascii_unit_prefix(word_chars) && consumed == word_chars.len() { + let continues_roman_section = next_word_starts_roman_or_number(remaining_words); + if continues_roman_section && unit.last() == Some(&crate::unicode::decode_unicode('⠲')) + { + unit.pop(); + } let mut encoded = crate::encode(&numeric)?; encoded.extend(unit); result.extend(encoded); + state.is_english = continues_roman_section; + state.needs_english_continuation = false; return Ok(()); } + if starts_new_roman_section_after_closed_enclosure(all_tokens, token_index) { + state.needs_english_continuation = false; + } + + // Korean Rule 35 keeps a Roman-led alphanumeric chain in the same + // Roman section across whitespace (`MP4 Player`). While the final + // digit temporarily leaves `is_english` false, `roman_number_chain` + // records that the next Roman word is a continuation rather than a new + // Rule-37 entry word. + let roman_section_continues_from_previous_word = + state.is_english || state.roman_number_chain; + let starts_fresh_roman_section = !roman_section_continues_from_previous_word + && !state.needs_english_continuation + && has_ascii_alphabetic; + if starts_fresh_roman_section { + state.roman_section_is_english_context = + roman_section_has_english_phrase_context(all_tokens, token_index); + } + // English entry (제28/35/39항) — 로마자표/연속표 emit + 영어 모드 전환. roman_mode::enter_english_if_starting(state, word_chars, has_ascii_alphabetic, result); @@ -401,6 +967,10 @@ fn emit_word( CharType::Number(_) => { roman_mode::exit_english_for_roman_number_chain(state); } + CharType::MathSymbol('+') + if crate::rules::token_rules::math_expression::is_roman_plus_identifier( + word_chars, + ) => {} CharType::Symbol(sym) => { // 한글 wrap 직후의 첫 디지털 표기 기호(. / @ # _ : -)는 // 영어 컨텍스트의 연속으로 본다. 예) "www.대통령.kr"에서 @@ -424,9 +994,12 @@ fn emit_word( if prev_wrap_eng_continuation || next_wrap_eng_continuation + || (*sym == '&' + && spaced_ampersand_connects_roman_words(all_tokens, token_index)) || english_logic::should_render_symbol_as_english( state.english_indicator, state.is_english, + state.doc_summary.is_english_majority, &state.parenthesis_stack, *sym, word_chars, @@ -448,7 +1021,7 @@ fn emit_word( } else { roman_mode::exit_english( state, - english_logic::should_request_continuation(*sym), + *sym != ')' && english_logic::should_request_continuation(*sym), ); } } @@ -463,12 +1036,35 @@ fn emit_word( if state.roman_number_chain && !state.is_english { match &char_type { CharType::English(_) => { - // PDF — roman_number_chain 안 digit 뒤 letter는 영어 연속 표지(⠰)를 - // 부착해 letter임을 명시한다 (digit과 혼동 방지). - result.push(48); + // Korean rule 35 keeps adjacent Roman letters and digits in + // one Roman section. Under UEB 6.5.2, lowercase a-j still + // need grade 1 after a digit because their cells are numeric; + // a capital indicator or a lowercase k-z cell is sufficient + // for every other Roman letter class. + if matches!(*c, 'a'..='j') { + result.push(crate::rules::korean::rule_29::ENGLISH_CONTINUATION); + } roman_mode::resume_english_from_roman_number_chain(state); } CharType::Number(_) => {} + CharType::MathSymbol('+') + if crate::rules::token_rules::math_expression::is_roman_plus_identifier( + word_chars, + ) => + { + roman_mode::resume_english_from_roman_number_chain(state); + } + CharType::Symbol(symbol) + if crate::rules::korean::rule_69::is_compatibility_unit_presentation( + *symbol, + ) || (*symbol == '-' + && word_chars + .get(i + 1) + .is_some_and(|next| next.is_ascii_alphanumeric())) + || (*symbol == '*' + && english_logic::is_attached_ascii_roman_asterisk( + word_chars, i, + )) => {} _ => { state.roman_number_chain = false; } @@ -494,6 +1090,7 @@ fn emit_word( is_all_uppercase, has_korean_char, ascii_starts_at_beginning, + roman_section_continues_from_previous_word, state, &mut skip_count, remaining_words, @@ -522,6 +1119,7 @@ fn emit_word( is_all_uppercase, has_korean_char, ascii_starts_at_beginning, + roman_section_continues_from_previous_word, state, &mut skip_count, remaining_words, @@ -565,8 +1163,13 @@ fn emit_word( || crate::symbol_shortcut::is_symbol_char(ch) || crate::utils::is_korean_char(ch)) }); + let starts_with_roman_letter = next_word + .chars() + .find(|ch| ch.is_ascii_alphabetic() || crate::utils::is_korean_char(*ch)) + .is_some_and(|ch| ch.is_ascii_alphabetic()); let is_single_letter_word = ascii_letters.len() == 1 && !next_word.chars().any(|ch| ch.is_ascii_digit()) + && starts_with_roman_letter && !has_invalid_symbol; if is_single_letter_word @@ -578,7 +1181,22 @@ fn emit_word( match next_type { CharType::English(_) | CharType::Number(_) => {} CharType::Symbol(sym) => { - if state.english_indicator + let separated_continuation = next_word_is_separated + && separated_symbol_continues_roman_section( + all_tokens, + token_index, + ); + // Rule 33/34 terminator omission applies when the + // punctuation is attached to the Roman run. If the + // print has whitespace first (`Poison (모래성)`), + // Rule 29 closes the Roman run before that space. + if next_word_is_separated && !separated_continuation { + result.push(50); + roman_mode::exit_english(state, false); + } else if separated_continuation && sym == '&' { + // A standalone ampersand joining Roman words is + // itself part of the current Roman section. + } else if state.english_indicator && state.is_english && english_logic::is_english_symbol(sym) { @@ -698,6 +1316,15 @@ mod tests { engine } + fn word_token(text: &'static str) -> Token<'static> { + let chars = text.chars().collect::>(); + Token::Word(WordToken { + text: Cow::Borrowed(text), + chars: chars.clone(), + meta: super::super::token::WordMeta::from_chars(&chars), + }) + } + /// Helper: round-trip test via emit(parse(text)) == encode(text) fn assert_round_trip(text: &str) { let mut ir = DocumentIR::parse(text, english_indicator(text)); @@ -717,6 +1344,273 @@ mod tests { ); } + #[rstest::rstest] + #[case::capitalized_parenthetical("논문(Frontiers in Drug Delivery)에", "Frontiers", true)] + #[case::capitalized_unenclosed("Nuclear Week in Parliament에 참석했다.", "Nuclear", true)] + #[case::lowercase_enclosed("제목(plain words in context)이다.", "plain", true)] + #[case::rule_37_metalinguistic_list("be, his, was, were의 약자를 바르게 쓰시오.", "be,", false)] + #[case::single_roman_annotation("논문(Cell)이 발표됐다.", "논문(Cell)이", false)] + #[case::numeric_word_before_phrase("123 Alpha Beta", "123", true)] + #[case::numeric_word_inside_phrase("Alpha 123 Beta", "Alpha", true)] + #[case::korean_word_ends_phrase("Alpha 한국 Beta", "Alpha", false)] + fn recognizes_structural_english_phrase_context( + #[case] input: &str, + #[case] first_roman_word: &str, + #[case] expected: bool, + ) { + let ir = DocumentIR::parse(input, true); + let start_index = ir + .tokens + .iter() + .position( + |token| matches!(token, Token::Word(word) if word.text.contains(first_roman_word)), + ) + .expect("test phrase must have a first Roman word"); + + assert_eq!( + roman_section_has_english_phrase_context(&ir.tokens, start_index), + expected + ); + } + + #[test] + fn english_phrase_scan_handles_non_text_boundaries() { + let tokens = vec![ + word_token("Alpha"), + Token::Space(SpaceKind::Regular), + word_token("Beta"), + Token::PreEncoded(vec![1]), + ]; + let leading_boundary = vec![ + Token::PreEncoded(vec![1]), + word_token("Alpha"), + Token::Space(SpaceKind::Regular), + word_token("Beta"), + ]; + + assert!(roman_section_has_english_phrase_context(&tokens, 0)); + assert!(roman_section_has_english_phrase_context( + &leading_boundary, + 0 + )); + } + + #[test] + fn previous_word_separation_scan_crosses_mode_markers() { + let tokens = vec![ + word_token("Alpha"), + Token::Space(SpaceKind::Regular), + Token::Mode(ModeEvent::CapsWord), + ]; + + assert!(is_separated_from_previous_word(&tokens, tokens.len())); + assert!(!is_separated_from_previous_word(&tokens, 1)); + } + + #[test] + fn current_word_lookup_handles_hard_boundaries_and_end_of_stream() { + assert!(current_word_at_or_after(&[Token::PreEncoded(vec![1])], 0).is_none()); + assert!(current_word_at_or_after(&[], 0).is_none()); + } + + #[rstest::rstest] + #[case::parenthesis('(', ')')] + #[case::square_bracket('[', ']')] + #[case::curly_brace('{', '}')] + #[case::single_quote('‘', '’')] + #[case::double_quote('“', '”')] + #[case::single_angle('〈', '〉')] + #[case::double_angle('《', '》')] + #[case::corner_bracket('「', '」')] + #[case::white_corner_bracket('『', '』')] + #[case::lenticular_bracket('【', '】')] + #[case::tortoise_shell_bracket('〔', '〕')] + #[case::white_lenticular_bracket('〖', '〗')] + #[case::white_tortoise_shell_bracket('〘', '〙')] + #[case::white_square_bracket('〚', '〛')] + fn enclosure_delimiter_pairs_are_bidirectional(#[case] opening: char, #[case] closing: char) { + assert_eq!(matching_group_open(closing), Some(opening)); + assert_eq!(matching_group_close(opening), Some(closing)); + } + + #[rstest::rstest] + #[case::no_previous_word(None, false)] + #[case::empty_previous(Some(""), false)] + #[case::punctuation_only_previous(Some("..."), false)] + #[case::nested_enclosure(Some("((A))"), true)] + #[case::missing_opener(Some("A)"), false)] + fn closed_enclosure_scans_only_a_complete_ascii_group( + #[case] previous: Option<&'static str>, + #[case] expected: bool, + ) { + let tokens = previous.map_or_else( + || vec![Token::Space(SpaceKind::Regular)], + |text| vec![word_token(text)], + ); + + assert_eq!( + closed_enclosure_before_contains_ascii(&tokens, tokens.len()), + expected + ); + } + + #[test] + fn new_section_probe_handles_a_mode_prefix_without_a_current_word() { + let tokens = vec![ + word_token("(A)"), + Token::Space(SpaceKind::Regular), + Token::Mode(ModeEvent::CapsWord), + ]; + + assert!(!starts_new_roman_section_after_closed_enclosure(&tokens, 2)); + } + + #[test] + fn spaced_colon_rejects_non_words_and_non_textual_right_boundaries() { + assert!(!spaced_colon_connects_roman_items( + &[Token::Space(SpaceKind::Regular)], + 0 + )); + let tokens = vec![ + word_token(":"), + Token::Space(SpaceKind::Regular), + Token::PreEncoded(vec![1]), + ]; + assert!(!spaced_colon_connects_roman_items(&tokens, 0)); + } + + #[rstest::rstest] + #[case::non_word_at_index("non_word")] + #[case::non_textual_left_boundary("non_text_left")] + #[case::non_roman_left_word("korean_left")] + #[case::malformed_attached_suffix("long_ampersand")] + #[case::non_textual_right_boundary("non_text_right")] + fn spaced_ampersand_rejects_incomplete_roman_neighbors(#[case] scenario: &str) { + let (tokens, index) = match scenario { + "non_word" => (vec![Token::Space(SpaceKind::Regular)], 0), + "non_text_left" => ( + vec![ + Token::PreEncoded(vec![1]), + Token::Space(SpaceKind::Regular), + word_token("&"), + ], + 2, + ), + "korean_left" => ( + vec![ + word_token("한국"), + Token::Space(SpaceKind::Regular), + word_token("&"), + ], + 2, + ), + "long_ampersand" => ( + vec![ + word_token("Alpha"), + Token::Space(SpaceKind::Regular), + word_token("&?"), + ], + 2, + ), + "non_text_right" => ( + vec![ + word_token("Alpha"), + Token::Space(SpaceKind::Regular), + word_token("&"), + Token::Space(SpaceKind::Regular), + Token::PreEncoded(vec![1]), + ], + 2, + ), + _ => unreachable!("unknown fixture"), + }; + + assert!(!spaced_ampersand_connects_roman_words(&tokens, index)); + } + + #[rstest::rstest] + #[case::no_following_word("no_next", false)] + #[case::empty_following_word("empty", false)] + #[case::nested_complete_group("nested", true)] + #[case::non_textual_group_body("non_text", false)] + #[case::unclosed_group("unclosed", false)] + fn separated_symbol_requires_a_complete_roman_group( + #[case] scenario: &str, + #[case] expected: bool, + ) { + let tokens = match scenario { + "no_next" => vec![word_token("Alpha")], + "empty" => vec![ + word_token("Alpha"), + Token::Space(SpaceKind::Regular), + word_token(""), + ], + "nested" => vec![ + word_token("Alpha"), + Token::Space(SpaceKind::Regular), + word_token("((Beta))"), + ], + "non_text" => vec![ + word_token("Alpha"), + Token::Space(SpaceKind::Regular), + word_token("("), + Token::PreEncoded(vec![1]), + ], + "unclosed" => vec![ + word_token("Alpha"), + Token::Space(SpaceKind::Regular), + word_token("(Beta"), + ], + _ => unreachable!("unknown fixture"), + }; + + assert_eq!( + separated_symbol_continues_roman_section(&tokens, 0), + expected + ); + } + + #[test] + fn slash_forces_a_terminator_before_leaving_roman_mode() { + let mut ir = DocumentIR::parse("ABC/한글", true); + let mut engine = make_char_engine(); + + let output = emit(&mut ir, &mut engine).expect("mixed Roman/Korean word must encode"); + + assert!(output.contains(&crate::unicode::decode_unicode('⠲'))); + assert!(!ir.state.is_english); + } + + #[test] + fn forced_symbol_between_adjacent_word_tokens_terminates_roman_mode() { + let tokens = vec![word_token("ABC"), word_token("/")]; + let Token::Word(word) = &tokens[0] else { + unreachable!("fixture begins with a word") + }; + let remaining_words = ["/"]; + let mut state = EncoderState::new(true); + state.is_english = true; + let mut engine = make_char_engine(); + let mut result = Vec::new(); + + emit_word( + word, + 0, + &mut state, + &mut engine, + &tokens, + WordContext { + prev_word: "", + remaining_words: &remaining_words, + }, + &mut result, + ) + .expect("Roman word must encode"); + + assert_eq!(result.last(), Some(&50)); + assert!(!state.is_english); + } + // ── Step 1-3: Basic token tests ── /// `emit` 결과가 `encode()` 와 byte-identical 한지 (round-trip) 다양한 @@ -769,6 +1663,292 @@ mod tests { assert_eq!(out, vec![52, 48, 32, 32, 32, 32, 32, 32, 4, 48]); } + /// Korean rules 28 appendix and 29: in Korean prose the roman indicator + /// precedes the UEB capital-word indicator (`0,,KTX`, not `,,0KTX`). + #[test] + fn roman_indicator_precedes_capital_prefix_without_explicit_entry_token() { + let chars = "KTX".chars().collect::>(); + let mut ir = DocumentIR { + tokens: vec![ + Token::Mode(ModeEvent::CapsWord), + Token::Word(WordToken { + text: Cow::Borrowed("KTX"), + chars: chars.clone(), + meta: super::super::token::WordMeta::from_chars(&chars), + }), + ], + state: EncoderState::new(true), + }; + let mut engine = make_char_engine(); + + let out = emit(&mut ir, &mut engine).unwrap(); + + assert!(out.starts_with(&[52, 32, 32])); + } + + #[test] + fn explicit_roman_entry_is_not_duplicated_before_capital_prefix() { + let chars = "KTX".chars().collect::>(); + let mut ir = DocumentIR { + tokens: vec![ + Token::Mode(ModeEvent::EnterEnglish), + Token::Mode(ModeEvent::CapsWord), + Token::Word(WordToken { + text: Cow::Borrowed("KTX"), + chars: chars.clone(), + meta: super::super::token::WordMeta::from_chars(&chars), + }), + ], + state: EncoderState::new(true), + }; + let mut engine = make_char_engine(); + + let out = emit(&mut ir, &mut engine).unwrap(); + + assert!(out.starts_with(&[52, 32, 32])); + assert_eq!(out.iter().filter(|byte| **byte == 52).count(), 1); + } + + /// Korean rule 29: consecutive Roman text shares one Roman section. A + /// token-level rediscovery of capitalization must not emit a second entry + /// when the final character emitter is still in that section. + #[test] + fn repeated_explicit_roman_entry_is_idempotent_in_active_section() { + let new_chars = "NEW".chars().collect::>(); + let york_chars = "YORK".chars().collect::>(); + let mut ir = DocumentIR { + tokens: vec![ + Token::Mode(ModeEvent::EnterEnglish), + Token::Mode(ModeEvent::CapsWord), + Token::Word(WordToken { + text: Cow::Borrowed("NEW"), + chars: new_chars.clone(), + meta: super::super::token::WordMeta::from_chars(&new_chars), + }), + Token::Space(SpaceKind::Regular), + Token::Mode(ModeEvent::EnterEnglish), + Token::Mode(ModeEvent::CapsWord), + Token::Word(WordToken { + text: Cow::Borrowed("YORK"), + chars: york_chars.clone(), + meta: super::super::token::WordMeta::from_chars(&york_chars), + }), + ], + state: EncoderState::new(true), + }; + let mut engine = make_char_engine(); + + let out = emit(&mut ir, &mut engine).unwrap(); + + assert_eq!(out.iter().filter(|byte| **byte == 52).count(), 1); + } + + /// Rules 29 and 33: whitespace closes the Roman run before a following + /// parenthetical; Rule 34's omission is only for an attached enclosure. + #[rstest::rstest] + #[case::ordinary_word("Poison (모래성)", "⠝⠲⠀")] + #[case::mixed_roman_korean_gloss("그룹 TVB (Television - 전시광파유한공사)", "⠃⠲⠀")] + #[case::roman_number_chain("8PM (최초)", "⠍⠲⠀")] + fn spaced_parenthetical_follows_a_closed_roman_run( + #[case] input: &str, + #[case] expected_boundary: &str, + ) { + let actual = crate::encode_to_unicode(input).expect("input must encode"); + assert!( + actual.contains(expected_boundary), + "missing Rule 29 terminator at spaced boundary: {actual}" + ); + } + + /// Rules 29 and 34: an enclosure closes its own Roman item without a + /// terminator, but a following unenclosed Roman item starts a new section. + #[rstest::rstest] + #[case::capitalized_after_comma("가는 설명(ABC), Next 나다", "⠠⠴⠐⠀⠴⠠⠝")] + #[case::all_caps_after_parenthesis("가는 설명(ABC) XYZ 나다", "⠠⠴⠀⠴⠠⠠⠭")] + #[case::quoted_title_after_parenthesis("가는 설명(ABC) ‘Title’ 나다", "⠠⠴⠀⠠⠦⠴⠠⠞")] + #[case::multiword_parenthesis("가는 설명(Alpha Beta) XYZ 나다", "⠠⠴⠀⠴⠠⠠⠭")] + #[case::multiword_quote("가는 ‘Alpha Beta’, ‘Title’ 나다", "⠄⠐⠀⠠⠦⠴⠠⠞")] + #[case::mixed_quote("가는 ‘설명 ABC’, XYZ 나다", "⠴⠄⠐⠀⠴⠠⠠⠭")] + fn unenclosed_roman_after_closed_enclosure_starts_a_new_section( + #[case] input: &str, + #[case] expected_boundary: &str, + ) { + let actual = crate::encode_to_unicode(input).expect("input must encode"); + assert!( + actual.contains(expected_boundary), + "missing new Rule-29 Roman section: {actual}" + ); + } + + /// Rule 32's official sequence keeps the successively enclosed single + /// letters in one Roman section; `e` still takes its UEB grade-1 marker. + #[test] + fn successive_enclosed_single_letters_remain_one_roman_section() { + let actual = crate::encode_to_unicode("모음에는 (a), (e), (i)가 있다.") + .expect("official Rule-32 example must encode"); + + assert_eq!( + actual + .chars() + .filter(|cell| *cell == crate::unicode::encode_unicode(52)) + .count(), + 1 + ); + assert!(actual.contains("⠐⠣⠰⠑⠐⠜")); + } + + /// Rules 29, 34, and 35 keep a pure Roman enclosure or a following number + /// inside the active Roman section. + #[rstest::rstest] + #[case::pure_roman_expansion("기준 GRI (Global Reporting Initiative) Standards", "⠊⠲⠀")] + #[case::roman_parenthetical("노래 Back for More (with Anitta)", "⠍⠲⠀")] + #[case::number_continuation("대회 May Circuit '23에서", "⠞⠲⠀")] + #[case::ascii_ellipsis("머뭇거리며 Umm ...이라고 말했다", "⠍⠍⠲⠀")] + #[case::unicode_ellipsis("머뭇거리며 Umm …이라고 말했다", "⠍⠍⠲⠀")] + #[case::midline_ellipsis("머뭇거리며 Umm ⋯이라고 말했다", "⠍⠍⠲⠀")] + fn separated_roman_or_number_continuation_stays_in_the_section( + #[case] input: &str, + #[case] forbidden_boundary: &str, + ) { + let actual = crate::encode_to_unicode(input).expect("input must encode"); + assert!( + !actual.contains(forbidden_boundary), + "Roman section closed too early: {actual}" + ); + } + + /// Korean rules 29, 32, and 35: a standalone print colon between Roman + /// items is UEB punctuation inside one Roman section, even when spaces + /// surround it. + #[rstest::rstest] + #[case::capitalized_words("가 Alpha : Beta 나")] + #[case::uppercase_and_number("가 URL : 393 나")] + #[case::mixed_case_and_number("가 Id : 7 나")] + fn spaced_colon_between_roman_items_remains_inside_section(#[case] input: &str) { + let actual = crate::encode_to_unicode(input).expect("input must encode"); + assert!( + actual.contains("⠀⠒⠀"), + "colon was not rendered as UEB punctuation: {actual}" + ); + assert!( + !actual.contains("⠲⠀⠐⠂⠀⠴"), + "colon split one Roman section: {actual}" + ); + } + + /// UEB 3.1.1 and Korean rule 29: a spaced ampersand connecting Roman words + /// neither closes the section before itself nor starts a new one after it. + #[rstest::rstest] + #[case::official_name("가 Marks & Spencer 나")] + #[case::technical_phrase("가 3D Scan & Solution 나")] + #[case::attached_right_word("가 EV Mining &Development 나")] + fn spaced_ampersand_bridges_one_roman_section(#[case] input: &str) { + let ir = DocumentIR::parse(input, true); + let ampersand_index = ir + .tokens + .iter() + .position( + |token| matches!(token, Token::Word(word) if word.chars.first() == Some(&'&')), + ) + .expect("ampersand token"); + assert!( + spaced_ampersand_connects_roman_words(&ir.tokens, ampersand_index), + "test input must contain a structurally Roman ampersand" + ); + + let actual = crate::encode_to_unicode(input).expect("Roman phrase must encode"); + assert!(actual.contains("⠈⠯"), "ampersand missing: {actual}"); + assert!( + !actual.contains("⠲⠀⠈⠯") && !actual.contains("⠈⠯⠀⠴"), + "ampersand split the Roman section: {actual}" + ); + } + + /// Rule 29: a Korean word containing one embedded Roman letter is not a + /// standalone one-letter Roman continuation. + #[rstest::rstest] + #[case::parenthesized_letter("KODEX 골드선물(H)", "⠭⠲⠀")] + #[case::korean_word_with_letter("ABB FIA 포뮬러E", "⠁⠲⠀")] + #[case::following_model_name("SUV 모델X", "⠧⠲⠀")] + fn korean_word_with_one_roman_letter_closes_the_previous_section( + #[case] input: &str, + #[case] expected_boundary: &str, + ) { + let actual = crate::encode_to_unicode(input).expect("input must encode"); + assert!( + actual.contains(expected_boundary), + "missing Roman terminator before Korean word: {actual}" + ); + } + + /// Rule 29: a separated one-letter Roman name remains part of the same + /// Roman section when it starts the next print word. A directly attached + /// Korean suffix or gloss does not change that Roman-first boundary. + #[rstest::rstest] + #[case::korean_particle("Global X가")] + #[case::korean_classifier("WBC B조")] + #[case::korean_gloss("DAY6 Young K(영케이)")] + fn roman_initial_single_letter_with_korean_suffix_continues_section(#[case] input: &str) { + let actual = crate::encode_to_unicode(input).expect("input must encode"); + assert!( + actual.contains("⠀⠰"), + "separated Roman initial did not continue the active section: {actual}" + ); + } + + /// Korean rule 35 PDF example: numbers do not split a roman section, so + /// the later capital word resumes without another roman indicator. + #[test] + fn capital_prefix_after_roman_number_chain_does_not_reenter_roman_mode() { + let out = encode("KBS 1 TV 좀 켜 주세요.").unwrap(); + + assert_eq!(out.iter().filter(|byte| **byte == 52).count(), 1); + } + + /// UEB 5.6.1/6.5.1-6.5.2 through the rule-29/35 character route. `A` is only + /// the Roman-chain routing scaffold for the PDF's exact `3b`, `3B`, and `3m` + /// suffixes; those cases directly cover lowercase a-j grade 1, capitalization, + /// and unmarked lowercase k-z. Comparison starts immediately after the route's + /// single Roman indicator, so another occurrence cannot satisfy the assertion. + #[rstest::rstest] + #[case::braille4all("Braille4All", "⠠⠃⠗⠁⠊⠇⠇⠑⠼⠙⠠⠁⠇⠇")] + #[case::m4g("M4G", "⠠⠍⠼⠙⠠⠛")] + #[case::w1n("W1N", "⠠⠺⠼⠁⠠⠝")] + #[case::lower_a_to_j("A3b", "⠠⠁⠼⠉⠰⠃")] + #[case::uppercase("A3B", "⠠⠁⠼⠉⠠⠃")] + #[case::lower_k_to_z("A3m", "⠠⠁⠼⠉⠍")] + fn numeric_grade1_mode_continues_into_pdf_roman_examples( + #[case] surface: &str, + #[case] expected_ueb: &str, + ) { + let output = encode(&format!("가({surface})")).unwrap(); + let expected_ueb = expected_ueb + .chars() + .map(crate::unicode::decode_unicode) + .collect::>(); + let roman_start = output + .iter() + .position(|cell| *cell == crate::rules::korean::rule_29::ROMAN_INDICATOR) + .expect("Korean wrapper must enter one Roman section") + + 1; + + assert_eq!( + output.get(roman_start..roman_start + expected_ueb.len()), + Some(expected_ueb.as_slice()) + ); + } + + /// UEB 6.5.2 full-encoder controls for the three Roman letter classes after + /// a digit: lowercase a-j needs grade 1, capitals use their capital indicator, + /// and lowercase k-z needs no additional indicator. + #[rstest::rstest] + #[case::lower_a_to_j("3b", "⠼⠉⠰⠃")] + #[case::uppercase("3B", "⠼⠉⠠⠃")] + #[case::lower_k_to_z("3m", "⠼⠉⠍")] + fn numeric_grade1_letter_class_pdf_controls(#[case] input: &str, #[case] expected: &str) { + assert_eq!(crate::encode_to_unicode(input).unwrap(), expected); + } + #[test] fn fraction_token_encodes() { let mut ir = DocumentIR { diff --git a/libs/braillify/src/rules/engine.rs b/libs/braillify/src/rules/engine.rs index 96f0f5f6..53567614 100644 --- a/libs/braillify/src/rules/engine.rs +++ b/libs/braillify/src/rules/engine.rs @@ -204,6 +204,7 @@ mod tests { has_korean_char: true, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count: &mut skip, state: &mut state, result: &mut result, @@ -421,6 +422,7 @@ mod tests { has_korean_char: false, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count: &mut skip, state: &mut state, result: &mut result, @@ -455,6 +457,7 @@ mod tests { has_korean_char: false, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count: &mut skip, state: &mut state, result: &mut result, diff --git a/libs/braillify/src/rules/english_shortform.rs b/libs/braillify/src/rules/english_shortform.rs index 24bb740d..c6c6d87b 100644 --- a/libs/braillify/src/rules/english_shortform.rs +++ b/libs/braillify/src/rules/english_shortform.rs @@ -1,147 +1,120 @@ //! English shortform collision detection (UEB 5.7.2 + 10.9). //! -//! When an all-uppercase ASCII word is point-encoded as `⠠⠠xy...`, the trailing -//! cells are identical to the corresponding lowercase shortform abbreviation. To -//! prevent the contraction reading (e.g. `⠠⠠⠉⠙` could otherwise be read as the -//! capitalised word "COULD"), the Grade-1 indicator (`⠰`) must be inserted before -//! the capital indicator. +//! When an all-uppercase ASCII letters-sequence is point-encoded as +//! `⠠⠠xy...`, its cells can be identical to a shortform or to the beginning of +//! a longer word containing one. To prevent that reading (for example `CD` as +//! "could", or the official `LLC` as "little" + `c`), the Grade-1 indicator +//! (`⠰`) must be inserted before the capital indicator. //! //! Reference: 통일영어점자 규정 제3판 //! - §5.7.2: 약자(축어 포함)와의 혼동 방지를 위한 1급 점자 모드 -//! - §10.9: 축어(shortform) 목록 (부록 1) -//! -//! Only **pure-letter** shortforms (whose braille cells map one-to-one to a-z) -//! can collide. Shortforms that embed contractions like `ch` (⠡), `sh` (⠩), `st` -//! (⠌), `th` (⠹), `ou` (⠳), or `con` (⠒) are NOT pure-letter, so their uppercase -//! acronyms (e.g. "MCH" → ⠠⠠⠍⠉⠓) cannot be confused with the shortform reading -//! and do not require the Grade-1 indicator. - -use std::collections::HashSet; -use std::sync::OnceLock; - -/// All pure-letter multi-letter shortforms from UEB Appendix 1 (lowercase form). -/// These cause collision with all-uppercase acronyms of the same letters. -const PURE_LETTER_SHORTFORMS: &[&str] = &[ - // a-series (10.9: about, above, according, ...) - "ab", "abv", "ac", "acr", "af", "afn", "afw", "ag", "al", "alm", "alr", "alt", - "alw", // b-series (10.9: because, before, behind, below, ...) - "bc", "bf", "bh", "bl", "bn", "brl", "bs", "bt", "by", // c-series - "cd", // could - // d-series - "dcl", "dclg", "dcv", "dcvg", // e-series - "ei", // either - // f-series - "fri", "fst", // g-series - "gd", "grt", // h-series - "hm", "hmf", "hrf", // i-series - "imm", // l-series - "ll", "lr", // m-series - "myf", // n-series - "nec", "nei", // p-series - "pd", "perh", // q-series - "qk", // r-series - "rcv", "rcvg", "rjc", "rjcg", // s-series - "sd", // t-series - "td", "tgr", "tm", "tn", // w-series - "wd", // x-series - "xf", "xs", // y-series - "yr", "yrf", "yrvs", -]; +//! - §10.9: 축어(shortform) 목록과 10.9.2-10.9.5의 긴 단어 조건 -fn shortform_set() -> &'static HashSet<&'static str> { - static CACHE: OnceLock> = OnceLock::new(); - CACHE.get_or_init(|| PURE_LETTER_SHORTFORMS.iter().copied().collect()) -} - -/// Returns `true` if the given ASCII word (already verified all-uppercase) collides -/// with a multi-letter shortform when emitted as `⠠⠠letters`. The Grade-1 indicator -/// `⠰` must be inserted before the CapsWord/CapsPassage marker in that case. +/// Returns `true` if the given initial ASCII letters-sequence collides with a +/// shortform under UEB 10.9.7 or with a permitted longer shortform reading under +/// 10.9.8. The Grade-1 indicator `⠰` must precede the capitalization marker. /// /// Single-letter words are excluded (UEB §10.1 single-letter alphabetic word signs /// require their own "독립적으로 사용된 경우" analysis handled elsewhere). pub fn requires_grade1_indicator(uppercase_word: &str) -> bool { - if uppercase_word.len() < 2 { - return false; - } - if !uppercase_word.chars().all(|c| c.is_ascii_alphabetic()) { - return false; + super::english_ueb::rule_10_9::requires_grade1_at_word_start(uppercase_word) +} + +/// UEB 2.6.1-2.6.3 boundary after a letters-sequence. +/// +/// A grade-1 symbol used for shortform disambiguation is valid only when the +/// letters-sequence is standing alone (10.9.7), or is the initial sequence of a +/// longer alphabetic word (10.9.8). Callers use this after an all-capitals ASCII +/// run, so any nonletter suffix must satisfy the standing-alone boundary. A +/// Korean syllable starts the next code span and is likewise a hard boundary for +/// the embedded Roman sequence. Digits, slash, plus, and an opening grouping +/// sign are deliberately excluded (`CD47`, `CD/ATM`, `NEIS+`, `LLM(SLM)`). +pub fn permits_grade1_boundary_after_run(suffix: &[char]) -> bool { + for &ch in suffix { + // UEB 2.6.1 makes a hyphen or dash a boundary in its own right. Do + // not scan through it into the next segment: the official `CD-ROM` + // requires grade 1 for `CD` even though another Roman segment follows. + if matches!( + ch, + '-' | '\u{2010}' | '\u{2011}' | '\u{2012}' | '\u{2013}' | '\u{2014}' + ) { + return true; + } + if matches!(ch as u32, 0x3131..=0x3163 | 0xAC00..=0xD7A3) + || matches!(ch, '\u{00b7}' | '\u{30fb}') + { + return true; + } + if !matches!( + ch, + ',' | ';' + | ':' + | '.' + | '\u{2026}' + | '!' + | '?' + | ')' + | ']' + | '}' + | '\'' + | '"' + | '\u{2019}' + | '\u{201d}' + ) { + return false; + } } - let lowered = uppercase_word.to_ascii_lowercase(); - shortform_set().contains(lowered.as_str()) + true } #[cfg(test)] mod tests { use super::*; - #[test] - fn cd_collides_with_could() { - assert!(requires_grade1_indicator("CD")); + #[rstest::rstest] + #[case::empty_boundary("", true)] + #[case::official_cd_rom_boundary("-ROM", true)] + #[case::closing_then_korean(")은", true)] + #[case::adjacent_digit("47", false)] + #[case::slash_continuation("/ATM", false)] + #[case::plus_continuation("+", false)] + #[case::opening_group("(SLM)", false)] + #[case::comma_before_attached_letters(",ABC", false)] + fn grade1_boundary_follows_ueb_standing_alone_rules( + #[case] suffix: &str, + #[case] expected: bool, + ) { + assert_eq!( + permits_grade1_boundary_after_run(&suffix.chars().collect::>()), + expected + ); } - #[test] - fn hm_collides_with_him() { - assert!(requires_grade1_indicator("HM")); - } - - #[test] - fn td_collides_with_today() { - assert!(requires_grade1_indicator("TD")); - } - - #[test] - fn wd_collides_with_would() { - assert!(requires_grade1_indicator("WD")); - } - - #[test] - fn lp_does_not_collide() { - // L = like, P = people are single-letter alphabetic wordsigns; - // their concatenation is not a multi-letter shortform. - assert!(!requires_grade1_indicator("LP")); - } - - #[test] - fn kbs_does_not_collide() { - assert!(!requires_grade1_indicator("KBS")); - } - - #[test] - fn mp_does_not_collide() { - assert!(!requires_grade1_indicator("MP")); - } - - #[test] - fn tv_does_not_collide() { - assert!(!requires_grade1_indicator("TV")); - } - - #[test] - fn sns_does_not_collide() { - assert!(!requires_grade1_indicator("SNS")); - } - - #[test] - fn single_letter_excluded() { - assert!(!requires_grade1_indicator("C")); - assert!(!requires_grade1_indicator("A")); - } - - #[test] - fn non_ascii_excluded() { - assert!(!requires_grade1_indicator("É")); - assert!(!requires_grade1_indicator("C1")); - } - - #[test] - fn case_insensitive_input() { - // Function expects already-uppercase but should still match if lowercase given. - assert!(requires_grade1_indicator("cd")); + #[rstest::rstest] + #[case::complete_cd("CD", true)] + #[case::complete_hm("HM", true)] + #[case::groupsign_fst("FST", true)] + #[case::groupsign_shd("SHD", true)] + #[case::official_llc_prefix("LLC", true)] + #[case::good_prefix("GDP", true)] + #[case::added_s("SDS", true)] + #[case::because_needs_be("BC", false)] + #[case::about_unlisted_suffix("ABBA", false)] + #[case::little_before_vowel("LLAMA", false)] + #[case::plain_initialism("KBS", false)] + #[case::single_letter("C", false)] + #[case::non_ascii("É", false)] + #[case::alphanumeric("C1", false)] + fn detects_complete_and_word_initial_shortform_confusion( + #[case] input: &str, + #[case] expected: bool, + ) { + assert_eq!(requires_grade1_indicator(input), expected); } #[test] - fn runtime_shortform_lookup_uses_lowercase_key() { - let word = std::hint::black_box("CD"); + fn case_insensitive_runtime_input() { + let word = std::hint::black_box("cd"); assert!(requires_grade1_indicator(word)); } diff --git a/libs/braillify/src/rules/english_ueb/engine.rs b/libs/braillify/src/rules/english_ueb/engine.rs index b091d8cb..c58c6ceb 100644 --- a/libs/braillify/src/rules/english_ueb/engine.rs +++ b/libs/braillify/src/rules/english_ueb/engine.rs @@ -193,6 +193,106 @@ impl EnglishUebEngine { Self { contractions } } + /// Encode one Roman word embedded in Korean text according to Korean rule 37. + /// + /// At a rule-37 Roman entry, whole-word signs and shortforms are suppressed + /// while UEB multi-letter groupsigns remain. The restriction applies only to + /// the English word immediately preceded by the Roman indicator, so later + /// standalone words in that section (and a rule-39 return to English-dominant + /// context) use ordinary UEB wordsigns and shortforms. Keeping both paths on + /// the same contraction engine makes rule 10 preference and morphology gates + /// identical. Roman mode transitions remain the Korean engine's job. + #[expect( + clippy::too_many_arguments, + reason = "the independent UEB context flags mirror distinct rule gates" + )] + pub(crate) fn encode_korean_word( + &self, + chars: &[char], + suppress_caps: bool, + prepend_grade1_indicator: bool, + standing_alone: bool, + word_initial: bool, + digit_adjacent: bool, + numeric_grade1_active: bool, + apostrophe_joined_lexeme: bool, + ) -> Option> { + let mut out = Vec::new(); + if prepend_grade1_indicator { + out.push(GRADE1); + } + + let lower: Vec = chars.iter().flat_map(|ch| ch.to_lowercase()).collect(); + + // UEB 5.6.1-5.6.2 and 6.5.3: a numeric indicator establishes grade-1 + // mode through a resumed Roman letters-sequence, so no contraction may + // follow an internal number (`Kep1er`). A number-first Korean token + // instead inserts rule 29's Roman indicator before its letters. In that + // latter shape, UEB 10.4.2 still spells a complete `ch/sh/th/wh/ou/st` + // sequence because its one-cell groupsign would be read as a word. + let complete_strong_sequence_would_be_word = digit_adjacent + && !word_initial + && matches!( + lower.as_slice(), + ['c', 'h'] | ['s', 'h'] | ['t', 'h'] | ['w', 'h'] | ['o', 'u'] | ['s', 't'] + ); + if numeric_grade1_active || complete_strong_sequence_would_be_word { + match classify_caps(chars) { + _ if suppress_caps => {} + Some(Caps::None) => {} + Some(Caps::Single) => out.push(CAPITAL), + Some(Caps::Word) => out.extend([CAPITAL, CAPITAL]), + None => { + for &ch in chars { + if ch.is_ascii_uppercase() { + out.push(CAPITAL); + } + out.push(crate::english::encode_english(ch.to_ascii_lowercase()).ok()?); + } + return Some(out); + } + } + for &ch in chars { + out.push(crate::english::encode_english(ch.to_ascii_lowercase()).ok()?); + } + return Some(out); + } + + let lower_word: String = lower.iter().collect(); + if !standing_alone && super::rule_10_5::wordsign(&lower_word).is_some() { + if !suppress_caps { + let caps = classify_caps(chars)?; + let indicator_count = + usize::from(caps != Caps::None) + usize::from(caps == Caps::Word); + out.extend(std::iter::repeat_n(CAPITAL, indicator_count)); + } + out.extend(super::rule_10_9::encode_korean_groupsigns( + &lower, + &self.contractions, + word_initial, + word_initial, + )?); + return Some(out); + } + self.encode_word_with_apostrophe_lexeme( + chars, + WordContext { + standing_alone, + upper_usable: standing_alone, + shortform_usable: standing_alone, + allow_longer_shortforms: standing_alone, + lower_usable: standing_alone, + suppress_caps, + word_initial, + restricted_prefix_boundary: word_initial, + digit_adjacent, + }, + apostrophe_joined_lexeme, + &mut out, + )?; + Some(out) + } + /// Encode a token stream. Returns `None` if any token is unsupported /// (a number, a symbol, or a mixed-case word), so the legacy path — which /// handles those — takes over. `explicit_english` is true only under an @@ -797,6 +897,115 @@ impl EnglishUebEngine { mod test_support { use super::*; + /// Korean rule 37's official `Can you ...` and UEB 10.5.1's official + /// `... YOU CAN ...` exercise every capitals classification while Korean + /// mode spells the wordsign through the shared UEB engine. + #[rstest::rstest] + #[case::lowercase("you", 0)] + #[case::initial_capital("Can", 1)] + #[case::capitals_word("CAN", 2)] + fn korean_non_standalone_wordsign_preserves_capitals_extent( + #[case] input: &str, + #[case] expected_capitals: usize, + ) { + let chars = input.chars().collect::>(); + let encoded = EnglishUebEngine::new() + .encode_korean_word(&chars, false, false, false, true, false, false, false) + .expect("ASCII Roman word must encode"); + + assert_eq!( + encoded.iter().take_while(|cell| **cell == CAPITAL).count(), + expected_capitals + ); + } + + /// Korean rule 37 suppresses contractions only in the English word directly + /// preceded by the Roman indicator. Later standalone words use ordinary UEB + /// shortforms (UEB 10.9). + #[rstest::rstest] + #[case::good("good", "⠛⠙")] + #[case::little("little", "⠇⠇")] + #[case::today("today", "⠞⠙")] + fn korean_roman_section_continuation_uses_shortforms( + #[case] input: &str, + #[case] expected: &str, + ) { + let chars = input.chars().collect::>(); + let encoded = EnglishUebEngine::new() + .encode_korean_word(&chars, false, false, true, true, false, false, false) + .expect("ASCII Roman word must encode"); + + assert_eq!(encoded, cells(expected)); + } + + #[test] + fn korean_rule_37_entry_still_suppresses_shortform() { + let chars = "good".chars().collect::>(); + let encoded = EnglishUebEngine::new() + .encode_korean_word(&chars, false, false, false, true, false, false, false) + .expect("ASCII Roman word must encode"); + + assert_eq!(encoded, cells("⠛⠕⠕⠙")); + } + + #[rstest::rstest] + #[case::ordinal_st("st", "⠎⠞")] + #[case::ordinal_th("th", "⠞⠓")] + #[case::unit_year("yr", "⠽⠗")] + #[case::capital_ordinal("ST", "⠠⠠⠎⠞")] + fn numeric_grade1_mode_spells_korean_context_letters( + #[case] input: &str, + #[case] expected: &str, + ) { + let chars = input.chars().collect::>(); + let encoded = EnglishUebEngine::new() + .encode_korean_word(&chars, false, false, false, false, true, true, false) + .expect("ASCII Roman suffix must encode"); + + assert_eq!(encoded, cells(expected)); + } + + #[rstest::rstest] + #[case::still_wordsign_collision("st", "⠎⠞")] + #[case::out_wordsign_collision("ou", "⠕⠥")] + #[case::er_is_not_a_wordsign("er", "⠻")] + #[case::gh_is_not_a_wordsign("gh", "⠣")] + fn number_first_roman_entry_applies_strong_groupsign_word_collision( + #[case] input: &str, + #[case] expected: &str, + ) { + let chars = input.chars().collect::>(); + let encoded = EnglishUebEngine::new() + .encode_korean_word(&chars, false, false, false, false, true, false, false) + .expect("ASCII Roman suffix must encode"); + + assert_eq!(encoded, cells(expected)); + } + + #[rstest::rstest] + #[case::mixed_capitalization(false, "⠁⠠⠃")] + #[case::caps_suppressed(true, "⠁⠃")] + fn numeric_grade1_mode_spells_mixed_case_letters( + #[case] suppress_caps: bool, + #[case] expected: &str, + ) { + let chars = "aB".chars().collect::>(); + let encoded = EnglishUebEngine::new() + .encode_korean_word( + &chars, + suppress_caps, + false, + false, + false, + true, + true, + false, + ) + .expect("mixed-case numeric continuation must encode"); + + assert_eq!(encoded, cells(expected)); + } + pub(super) fn enc(text: &str) -> Option> { super::super::try_encode(text) } diff --git a/libs/braillify/src/rules/english_ueb/engine/caps.rs b/libs/braillify/src/rules/english_ueb/engine/caps.rs index dd2926f4..b85edce1 100644 --- a/libs/braillify/src/rules/english_ueb/engine/caps.rs +++ b/libs/braillify/src/rules/english_ueb/engine/caps.rs @@ -81,6 +81,10 @@ pub(super) fn chemical_formula_caps(chars: &[char]) -> bool { chars.len() >= 2 && !matches!(chars, ['C', 'O']) && chars.iter().all(|c| matches!(c, 'C' | 'H' | 'O')) + // A repeated element is written with a subscript in chemical notation + // (`H₂O`, `CO₂`), not by repeating its capital letter. Thus corporate + // initialisms such as `COO`/`CCO` must remain ordinary capitals words. + && !chars.windows(2).any(|pair| pair[0] == pair[1]) } pub(super) fn encode_letters_literal(chars: &[char]) -> Option> { @@ -549,7 +553,7 @@ pub(super) fn is_letter_pronounced_initialism(chars: &[char]) -> bool { matches!( word.as_str(), "WHO" | "OED" | "US" | "IT" | "MSH" | "DAR" | "EST" | "TEN" | "POW" | "FRS" - ) + ) || super::super::pronunciation::cmudict::has_unambiguous_letter_name_pronunciation(chars) } /// §8.6.3 vs §8.8.2 dispatch: whether a lowercase tail after a capitals-word run @@ -867,6 +871,19 @@ mod tests { assert_eq!(enc(text), Some(cells(expected))); } + #[rstest::rstest] + #[case::hydroxide(&['O', 'H'], true)] + #[case::three_distinct_elements(&['C', 'H', 'O'], true)] + #[case::company_co(&['C', 'O'], false)] + #[case::chief_operating_officer(&['C', 'O', 'O'], false)] + #[case::chief_commercial_officer(&['C', 'C', 'O'], false)] + fn identifies_unsubscripted_chemical_capital_runs( + #[case] chars: &[char], + #[case] expected: bool, + ) { + assert_eq!(chemical_formula_caps(chars), expected); + } + /// §15.3.2: in level-change tone notation, an up/down-step arrow printed before /// a word is followed by a braille space and the under-word bullet indicator. /// The tone reading needs a tone-notation context (several level arrows in the diff --git a/libs/braillify/src/rules/english_ueb/engine/encode_word.rs b/libs/braillify/src/rules/english_ueb/engine/encode_word.rs index c8f6dfad..6b67ebd3 100644 --- a/libs/braillify/src/rules/english_ueb/engine/encode_word.rs +++ b/libs/braillify/src/rules/english_ueb/engine/encode_word.rs @@ -651,7 +651,7 @@ macro_rules! encode_word_arm { { lower_usable = true; } - $engine.encode_word( + $engine.encode_word_with_apostrophe_lexeme( $chars, WordContext { standing_alone, @@ -675,6 +675,7 @@ macro_rules! encode_word_arm { digit_adjacent: matches!(prev, Some(EnglishToken::Number(_))) || matches!(next, Some(EnglishToken::Number(_))), }, + apostrophe_joined_recorded_token_word($tokens, $i), &mut $out, )?; $prev_was_number = false; diff --git a/libs/braillify/src/rules/english_ueb/engine/tokens.rs b/libs/braillify/src/rules/english_ueb/engine/tokens.rs index 685956f7..41182cb5 100644 --- a/libs/braillify/src/rules/english_ueb/engine/tokens.rs +++ b/libs/braillify/src/rules/english_ueb/engine/tokens.rs @@ -124,6 +124,44 @@ pub(super) fn token_plain_chars(tokens: &[EnglishToken]) -> Vec { chars } +/// Whether the word token at `i` belongs to an apostrophe-separated sequence +/// whose elided spelling is a recorded lexical word (`O'PENing` → `opening`). +/// This is pronunciation evidence for §10.12.1: the capital run is part of a +/// spoken word rather than a sequence of separately pronounced initials. +pub(super) fn apostrophe_joined_recorded_token_word(tokens: &[EnglishToken], i: usize) -> bool { + let Some(EnglishToken::Word(current)) = tokens.get(i) else { + return false; + }; + + let apostrophe_at = |index: usize| { + matches!( + tokens.get(index), + Some(EnglishToken::Symbol('\'' | '\u{2019}')) + ) + }; + let word_at = |index: usize| matches!(tokens.get(index), Some(EnglishToken::Word(_))); + + let mut start = i; + while start >= 2 && apostrophe_at(start - 1) && word_at(start - 2) { + start -= 2; + } + let mut end = i; + while end + 2 < tokens.len() && apostrophe_at(end + 1) && word_at(end + 2) { + end += 2; + } + if start == end { + return false; + } + + // `start`/`end` are grown only across alternating Word/Apostrophe tokens. + // Reuse the exhaustive token flattener instead of repeating an unreachable + // defensive match for the already-proven slice shape. + let run_start = token_plain_chars(&tokens[start..i]).len(); + let joined = token_plain_chars(&tokens[start..=end]); + let run_end = run_start + current.len(); + super::super::pronunciation::apostrophe_elided_recorded_word_at(&joined, run_start, run_end) +} + pub(super) fn token_plain_chars_preserve_word_division(tokens: &[EnglishToken]) -> Vec { let mut chars = Vec::new(); for token in tokens { @@ -933,6 +971,14 @@ mod tests { assert_eq!(styled_column_gap(&tokens, 0), None); } + #[test] + fn apostrophe_lexeme_lookup_requires_a_word_at_the_requested_index() { + assert!(!apostrophe_joined_recorded_token_word( + &[EnglishToken::Space], + 0 + )); + } + #[test] fn push_spatial_char_renders_line_arrow() { // §16 spatial mode: a line arrow (`→`) renders via its two-cell arrow sign. diff --git a/libs/braillify/src/rules/english_ueb/engine/word_methods.rs b/libs/braillify/src/rules/english_ueb/engine/word_methods.rs index cab42b83..703aba35 100644 --- a/libs/braillify/src/rules/english_ueb/engine/word_methods.rs +++ b/libs/braillify/src/rules/english_ueb/engine/word_methods.rs @@ -6,6 +6,19 @@ impl EnglishUebEngine { chars: &[char], ctx: WordContext, out: &mut Vec, + ) -> Option<()> { + self.encode_word_with_apostrophe_lexeme(chars, ctx, false, out) + } + + /// Encode a word with evidence that an adjacent apostrophe-separated run + /// reconstructs one lexical word. Only the mixed-case capitals prefix uses + /// this evidence; ordinary words follow [`Self::encode_word`] unchanged. + pub(super) fn encode_word_with_apostrophe_lexeme( + &self, + chars: &[char], + ctx: WordContext, + apostrophe_joined_lexeme: bool, + out: &mut Vec, ) -> Option<()> { let WordContext { standing_alone, @@ -63,7 +76,12 @@ impl EnglishUebEngine { return Some(()); } if !suppress_caps && classify_caps(chars).is_none() { - return self.encode_mixed_case(chars, allow_longer_shortforms, out); + return self.encode_mixed_case( + chars, + allow_longer_shortforms, + apostrophe_joined_lexeme, + out, + ); } if shortform_usable && super::super::rule_10_9::is_pure_shortform_abbreviation(&word) { out.push(GRADE1); @@ -186,6 +204,7 @@ impl EnglishUebEngine { &self, chars: &[char], allow_longer_shortforms: bool, + apostrophe_joined_lexeme: bool, out: &mut Vec, ) -> Option<()> { if allow_longer_shortforms && let Some(boundary) = initial_caps_shortform_boundary(chars) { @@ -194,7 +213,12 @@ impl EnglishUebEngine { out.extend([CAPITAL, CAPITAL]); out.extend(cells); out.extend([CAPITAL, decode_unicode('⠄')]); - self.encode_mixed_case(&chars[boundary..], allow_longer_shortforms, out)?; + self.encode_mixed_case( + &chars[boundary..], + allow_longer_shortforms, + apostrophe_joined_lexeme, + out, + )?; return Some(()); } let camel_subunit_start = camel_title_subunit_after_caps_prefix(chars); @@ -214,7 +238,12 @@ impl EnglishUebEngine { allow_longer_shortforms, )?, ); - self.encode_mixed_case(&chars[subunit_start..], allow_longer_shortforms, out)?; + self.encode_mixed_case( + &chars[subunit_start..], + allow_longer_shortforms, + apostrophe_joined_lexeme, + out, + )?; return Some(()); } let initial_caps = chars.iter().take_while(|c| c.is_uppercase()).count(); @@ -355,8 +384,28 @@ impl EnglishUebEngine { return Some(()); } out.extend([CAPITAL, CAPITAL]); - for c in &chars[..initial_caps] { - out.push(crate::english::encode_english(c.to_ascii_lowercase()).ok()?); + if apostrophe_joined_lexeme { + // §10.6.8: `en`/`in` and the other permitted groupsigns remain + // available inside capital mode when the letters belong to an + // ordinarily pronounced word. §10.12.1 initialisms retain the + // literal path because they have no lexical-word evidence. + let lower_prefix: Vec = chars[..initial_caps] + .iter() + .map(|c| c.to_ascii_lowercase()) + .collect(); + out.extend( + super::super::rule_10_9::encode_with_optional_longer_shortforms( + &lower_prefix, + &self.contractions, + false, + false, + false, + )?, + ); + } else { + for c in &chars[..initial_caps] { + out.push(crate::english::encode_english(c.to_ascii_lowercase()).ok()?); + } } out.extend([CAPITAL, decode_unicode('⠄')]); let lower: Vec = suffix.iter().flat_map(|c| c.to_lowercase()).collect(); @@ -513,6 +562,10 @@ mod tests { #[case::bachelor_science("BSc", "⠠⠃⠠⠎⠉")] #[case::megahertz("MHz", "⠠⠍⠠⠓⠵")] #[case::potassium_chloride("KCl", "⠠⠅⠠⠉⠇")] + // §10.6.8/§10.12.1: the apostrophe-elided spelling is the recorded word + // `opening`, so `PEN` is a capitalised word segment, not initials, and keeps + // the `en` groupsign inside capitals-word mode. + #[case::apostrophe_joined_lexeme("O'PENing", "⠠⠕⠄⠠⠠⠏⠢⠠⠄⠬")] #[case::chemical_subscript("HOCH₂", "⠠⠓⠠⠕⠠⠉⠠⠓⠰⠢⠼⠃")] fn encodes_mixed_case_words_8_2(#[case] text: &str, #[case] expected: &str) { assert_eq!(enc(text), Some(cells(expected))); @@ -800,6 +853,7 @@ mod tests { .encode_mixed_case( &['f', 'o', 'u', 'n', 'D', 'A', 't', 'i', 'o', 'n'], true, + false, &mut out, ) .unwrap(); @@ -814,6 +868,7 @@ mod tests { .encode_mixed_case( &['f', 'o', 'u', 'n', 'D', 'A', 't', 'i', 'o', 'n'], true, + false, &mut out, ) .unwrap(); @@ -906,4 +961,28 @@ mod tests { ); assert!(!out.is_empty()); } + + #[test] + fn standing_all_caps_longer_shortform_collision_gets_grade1() { + // UEB 5.7.2/10.9.8: LLC begins with the `little` shortform cells but is + // not itself a complete pure-letter shortform abbreviation. + let ctx = WordContext { + standing_alone: true, + upper_usable: true, + shortform_usable: true, + allow_longer_shortforms: true, + lower_usable: true, + suppress_caps: false, + word_initial: true, + restricted_prefix_boundary: true, + digit_adjacent: false, + }; + let mut out = Vec::new(); + + EnglishUebEngine::new() + .encode_word(&['L', 'L', 'C'], ctx, &mut out) + .expect("ASCII acronym must encode"); + + assert!(out.starts_with(&[GRADE1, CAPITAL, CAPITAL])); + } } diff --git a/libs/braillify/src/rules/english_ueb/engine/words.rs b/libs/braillify/src/rules/english_ueb/engine/words.rs index 2a0ebbeb..d5427a97 100644 --- a/libs/braillify/src/rules/english_ueb/engine/words.rs +++ b/libs/braillify/src/rules/english_ueb/engine/words.rs @@ -674,10 +674,10 @@ mod tests { #[test] fn encodes_standalone_shortform_collision_with_grade1() { // §8.7: an all-caps word that collides with a multi-letter shortform yet - // is NOT itself a pure shortform abbreviation (`BC` shares letters with - // the `bc`="because" wordsign) takes a grade-1 indicator before the caps - // marker so it reads as literal letters. - let out = enc("BC").expect("should encode"); + // is not itself a complete shortform abbreviation (`LLC` begins with the + // `ll`="little" shortform) takes a grade-1 indicator before the caps marker + // so it reads as literal letters. UEB 10.9.8 gives `LLC` as the example. + let out = enc("LLC").expect("should encode"); assert_eq!(out.first(), Some(&GRADE1)); } } diff --git a/libs/braillify/src/rules/english_ueb/korean_context.rs b/libs/braillify/src/rules/english_ueb/korean_context.rs index 5854a096..1e789dc9 100644 --- a/libs/braillify/src/rules/english_ueb/korean_context.rs +++ b/libs/braillify/src/rules/english_ueb/korean_context.rs @@ -259,6 +259,27 @@ mod tests { assert_eq!(got, expected); } + /// In an open Roman span, UEB §10.5's restricted `con` lower groupsign can + /// be reached after earlier letters; the strong-sign cascade does not own it. + #[test] + fn matches_restricted_lower_groupsign_inside_open_roman_word() { + let chars = std::hint::black_box("reconsider") + .chars() + .collect::>(); + let matched = match_korean_prefix(KoreanPrefixInput { + word: &chars, + pos: 2, + wrap_active: true, + is_all_uppercase: false, + at_entry: false, + standalone_wordsign: false, + }) + .expect("internal con groupsign must match"); + + assert_eq!(matched.cells, vec![decode_unicode('⠒')]); + assert_eq!(matched.consumed, 3); + } + #[rstest::rstest] #[case::alphabetic_you("you", decode_unicode('⠽'), 3)] #[case::strong_this("this", decode_unicode('⠹'), 4)] diff --git a/libs/braillify/src/rules/english_ueb/pronunciation/cmudict.rs b/libs/braillify/src/rules/english_ueb/pronunciation/cmudict.rs index de33fefb..398f95ec 100644 --- a/libs/braillify/src/rules/english_ueb/pronunciation/cmudict.rs +++ b/libs/braillify/src/rules/english_ueb/pronunciation/cmudict.rs @@ -45,6 +45,74 @@ pub fn is_recorded_word(word: &str) -> bool { INDEX.contains_key(word) } +/// Whether CMUdict supplies sufficiently specific evidence that an uppercase +/// abbreviation is pronounced as letter names. +/// +/// UEB §10.12.1 suppresses contractions when an abbreviation is pronounced as +/// letters. Case-folded dictionary headwords alone cannot establish that +/// (`LED` otherwise collides with lexical *led*), so compare a pronunciation +/// variant against the concatenated ARPABET names of the printed capitals. An +/// For three or more letters, every recorded pronunciation must be the exact +/// letter-name sequence; this keeps word-pronounced acronyms such as `ASEAN` +/// on UEB's contract-when-uncertain fallback when the dictionary records both +/// readings. For a two-capital abbreviation, an exact two-letter reading is +/// already specific evidence for the printed abbreviation even when the same +/// case-folded headword also has a one-word homograph (`AI` versus *ai*). +/// Unknown abbreviations and longer mixed-pronunciation entries return false. +pub fn has_unambiguous_letter_name_pronunciation(chars: &[char]) -> bool { + if chars.len() < 2 || !chars.iter().all(|ch| ch.is_ascii_uppercase()) { + return false; + } + + const LETTER_PHONES: &[&[&str]; 26] = &[ + &["EY"], + &["B", "IY"], + &["S", "IY"], + &["D", "IY"], + &["IY"], + &["EH", "F"], + &["JH", "IY"], + &["EY", "CH"], + &["AY"], + &["JH", "EY"], + &["K", "EY"], + &["EH", "L"], + &["EH", "M"], + &["EH", "N"], + &["OW"], + &["P", "IY"], + &["K", "Y", "UW"], + &["AA", "R"], + &["EH", "S"], + &["T", "IY"], + &["Y", "UW"], + &["V", "IY"], + &["D", "AH", "B", "AH", "L", "Y", "UW"], + &["EH", "K", "S"], + &["W", "AY"], + &["Z", "IY"], + ]; + + let expected: Vec<&str> = chars + .iter() + // The guard above proves every character is in `A..=Z`. + .flat_map(|letter| LETTER_PHONES[*letter as usize - 'A' as usize]) + .copied() + .collect(); + let key: String = chars.iter().map(|ch| ch.to_ascii_lowercase()).collect(); + INDEX.get(key.as_str()).is_some_and(|variants| { + let is_letter_name_variant = |variant: &&str| { + variant + .split_whitespace() + .map(|phone| phone.trim_end_matches(['0', '1', '2'])) + .eq(expected.iter().copied()) + }; + !variants.is_empty() + && (variants.iter().all(is_letter_name_variant) + || (chars.len() == 2 && variants.iter().any(is_letter_name_variant))) + }) +} + /// Looks up ARPABET pronunciations from the embedded CMUdict. pub struct CmuDictProvider; @@ -136,4 +204,18 @@ mod tests { // A head with no phones → None. assert_eq!(parse_cmudict_line("word "), None); } + + #[rstest::rstest] + #[case::ged_is_ambiguous("GED", false)] + #[case::ai_two_letter_abbreviation("AI", true)] + #[case::cc_two_letter_abbreviation("CC", true)] + #[case::asean_mixed_pronunciation("ASEAN", false)] + #[case::ofc_initialism("OFC", true)] + #[case::lexical_led_only("LED", false)] + #[case::unknown_mou("MOU", false)] + #[case::lowercase_is_not_capitals("ged", false)] + fn detects_unambiguous_letter_name_pronunciations(#[case] text: &str, #[case] expected: bool) { + let chars: Vec = text.chars().collect(); + assert_eq!(has_unambiguous_letter_name_pronunciation(&chars), expected); + } } diff --git a/libs/braillify/src/rules/english_ueb/pronunciation/mod.rs b/libs/braillify/src/rules/english_ueb/pronunciation/mod.rs index 6a190532..84a5e0ed 100644 --- a/libs/braillify/src/rules/english_ueb/pronunciation/mod.rs +++ b/libs/braillify/src/rules/english_ueb/pronunciation/mod.rs @@ -15,6 +15,65 @@ pub mod aligner; pub mod classifier; pub mod cmudict; +/// Decide whether an apostrophe-separated print sequence is one recorded +/// lexical word when the apostrophe is elided. +/// +/// UEB §10.12.1 suppresses contractions when capitals are letters pronounced +/// separately, while §10.6.8 retains `en`/`in` inside an ordinarily pronounced +/// word. A stylised spelling such as `O'PENing` is split into two parser runs, +/// so the case pattern of `PENing` alone cannot distinguish those situations. +/// Requiring the complete adjacent sequence (`opening`) to be in CMUdict gives +/// pronunciation evidence without recognising any particular corpus phrase. +/// The caller supplies the current ASCII run so unrelated quote punctuation is +/// never absorbed into the lookup. +pub(crate) fn apostrophe_elided_recorded_word_at( + chars: &[char], + run_start: usize, + run_end: usize, +) -> bool { + if run_start >= run_end + || run_end > chars.len() + || !chars[run_start..run_end] + .iter() + .all(|ch| ch.is_ascii_alphabetic()) + { + return false; + } + + let is_apostrophe = |ch: char| matches!(ch, '\'' | '\u{2019}'); + let is_member = |ch: char| ch.is_ascii_alphabetic() || is_apostrophe(ch); + + let mut start = run_start; + while start > 0 && is_member(chars[start - 1]) { + start -= 1; + } + let mut end = run_end; + while end < chars.len() && is_member(chars[end]) { + end += 1; + } + + let segment = &chars[start..end]; + if !segment.iter().any(|ch| is_apostrophe(*ch)) { + return false; + } + if segment.iter().enumerate().any(|(index, ch)| { + is_apostrophe(*ch) + && (index == 0 + || index + 1 == segment.len() + || !segment[index - 1].is_ascii_alphabetic() + || !segment[index + 1].is_ascii_alphabetic()) + }) { + return false; + } + + let normalized: String = segment + .iter() + .filter(|ch| ch.is_ascii_alphabetic()) + .map(|ch| ch.to_ascii_lowercase()) + .collect(); + cmudict::is_recorded_word(&normalized) +} + /// One ARPABET phoneme: its base symbol (e.g. `B`, `AH`, `N`) and, for vowels, /// the lexical stress (0 = unstressed, 1 = primary, 2 = secondary). In CMUdict /// only vowels carry a stress digit, so `stress.is_some()` identifies a vowel. @@ -102,4 +161,25 @@ mod tests { assert_eq!(ph.stress, None); assert!(!ph.is_vowel()); } + + #[rstest::rstest] + #[case::straight_apostrophe("O'PENing", 2, 8, true)] + #[case::curly_apostrophe("O\u{2019}PENing", 2, 8, true)] + #[case::no_join("PENing", 0, 6, false)] + #[case::unknown_elision("rock'n", 5, 6, false)] + #[case::empty_requested_run("abc", 1, 1, false)] + #[case::run_end_out_of_bounds("abc", 0, 4, false)] + #[case::requested_run_contains_nonletter("a1c", 0, 3, false)] + fn classifies_apostrophe_elided_lexical_words( + #[case] text: &str, + #[case] run_start: usize, + #[case] run_end: usize, + #[case] expected: bool, + ) { + let chars: Vec = text.chars().collect(); + assert_eq!( + apostrophe_elided_recorded_word_at(&chars, run_start, run_end), + expected + ); + } } diff --git a/libs/braillify/src/rules/english_ueb/rule_10_9.rs b/libs/braillify/src/rules/english_ueb/rule_10_9.rs index 3ca0a74b..f4dc14b2 100644 --- a/libs/braillify/src/rules/english_ueb/rule_10_9.rs +++ b/libs/braillify/src/rules/english_ueb/rule_10_9.rs @@ -43,9 +43,81 @@ pub fn whole_word_cells(word: &str) -> Option> { /// A literal all-letter abbreviation that collides with a pure-letter shortform /// needs a grade-1 indicator before normal letter encoding (§10.9.7). pub fn is_pure_shortform_abbreviation(word: &str) -> bool { + let letters = word.chars().collect::>(); + if letters.len() < 2 || !letters.iter().all(char::is_ascii_lowercase) { + return false; + } + let literal_cells = korean_letter_sequence_cells(&letters); + SHORTFORMS .values() - .any(|abbr| abbr.chars().all(|ch| ch.is_ascii_lowercase()) && *abbr == word) + .any(|notation| notation_cells(notation).as_deref() == Some(literal_cells.as_slice())) +} + +/// UEB 5.7.2 and 10.9.7-10.9.8: returns whether an ASCII letters-sequence at +/// the beginning of a word needs a grade-1 symbol before its capitalization +/// indicator so it cannot be read as a shortform, or as the beginning of a +/// longer word containing one. +/// +/// This compares cells produced by the ordinary rule-37 groupsign encoder with +/// the complete shortform table. Consequently sequences containing groupsigns +/// are handled without a second hand-maintained alias list: `FST` collides with +/// `first` (`f` + `st`), `SHD` with `should`, while `BC` does not collide with +/// `because` (`be` + `c`). For a proper prefix, the existing 10.9.2-10.9.5 +/// longer-word grammar decides whether that shortform reading is actually +/// permitted; this is why the official `LLC` is guarded but `LLAMA` is not. +pub fn requires_grade1_at_word_start(letters: &str) -> bool { + let lower = letters.to_ascii_lowercase(); + let chars = lower.chars().collect::>(); + if chars.len() < 2 || !letters.chars().all(|ch| ch.is_ascii_alphabetic()) { + return false; + } + + for end in 2..=chars.len() { + let prefix_cells = korean_letter_sequence_cells(&chars[..end]); + for (shortform, notation) in SHORTFORMS.entries() { + if notation_cells(notation).as_deref() != Some(prefix_cells.as_slice()) { + continue; + } + if end == chars.len() { + return true; + } + + let suffix = &chars[end..]; + // §10.9.5 admits an added `s` for every base shortform except + // `abouts`, `almosts`, and `hims`. + if suffix == ['s'] && !matches!(*shortform, "about" | "almost" | "him") { + return true; + } + + let hypothetical = shortform + .chars() + .chain(suffix.iter().copied()) + .collect::>(); + if longer_use_allowed(&hypothetical, 0, shortform) { + return true; + } + } + } + false +} + +/// Produce the cells that the real Korean-rule-37 Roman body encoder would emit +/// for a lowercase ASCII letters-sequence. Grade-1 collision detection must use +/// this exact path: a default [`ContractionEngine`] contains no registered rules +/// and would consequently miss cell-equivalent sequences such as `fst` (`f` + +/// the `st` groupsign) and `shd` (the `sh` groupsign + `d`). +fn korean_letter_sequence_cells(letters: &[char]) -> Vec { + super::span::encode_korean_word( + letters, true, // capitalization indicators are compared separately + false, // do not recursively prepend grade 1 + false, // rule 37 suppresses whole-word signs on Roman entry + true, // the sequence begins at a Roman word boundary + false, // no adjacent digit in a pure letters-sequence + false, // no numeric grade-1 mode in a pure letters-sequence + false, // not split by an apostrophe + ) + .expect("a lowercase ASCII letters-sequence must be encodable") } /// Encode a word as the §10.10.2 cell-minimising contraction sequence. @@ -77,6 +149,7 @@ pub fn encode_with_longer_shortforms( false, true, false, + false, ) } @@ -96,6 +169,30 @@ pub fn encode_with_optional_longer_shortforms( false, allow_longer_shortforms, false, + false, + ) +} + +/// Korean rule 37 word body: use UEB multi-letter groupsigns, but do not let a +/// groupsign that is also a lower wordsign consume the entire first Roman word. +/// This is a structural gate, not a word-output table: inner groupsigns such as +/// `en` in `enough` remain available. +pub(crate) fn encode_korean_groupsigns( + word: &[char], + contractions: &ContractionEngine, + suppress_initial_ing: bool, + restricted_prefix_boundary: bool, +) -> Option> { + encode_with_constraints( + word, + contractions, + suppress_initial_ing, + restricted_prefix_boundary, + None, + false, + false, + false, + true, ) } @@ -119,6 +216,7 @@ pub fn encode_anglicised_word( false, true, true, + false, ) } @@ -159,6 +257,7 @@ pub fn encode_with_division( first_line_has_upper_prefix, true, false, + false, )?); return Some(out); } @@ -171,6 +270,7 @@ pub fn encode_with_division( first_line_has_upper_prefix, true, false, + false, ) } @@ -187,6 +287,7 @@ fn encode_with_constraints( first_line_has_upper_prefix: bool, allow_longer_shortforms: bool, relax_shortforms: bool, + suppress_whole_word_wordsign: bool, ) -> Option> { let n = word.len(); // §10.11.1: a contraction must not bridge the seam of a compound word. Look up @@ -240,6 +341,7 @@ fn encode_with_constraints( first_line_has_upper_prefix, allow_longer_shortforms, relax_shortforms, + suppress_whole_word_wordsign, ) { let next = pos + consumed; let total = cells.len() + cost[next]; @@ -294,6 +396,7 @@ fn candidate_moves( first_line_has_upper_prefix: bool, allow_longer_shortforms: bool, relax_shortforms: bool, + suppress_whole_word_wordsign: bool, ) -> Vec<(Vec, usize, u16)> { let mut moves = Vec::new(); // §10.9 longer-word shortform placement (preferred on a cost tie → priority 0). @@ -316,6 +419,17 @@ fn candidate_moves( } let protected_here = inside_protected[pos]; for m in contractions.matches_at(word, pos) { + // Korean rule 37: immediately after the Roman indicator, a lower + // wordsign is written with alphabet/multi-letter groupsigns instead. + // Reject only a contraction consuming the complete wordsign; inner + // groupsigns remain candidates (`enough` keeps `en` and `gh`). + if suppress_whole_word_wordsign + && pos == 0 + && m.consumed == word.len() + && super::rule_10_5::wordsign(&word.iter().collect::()).is_some() + { + continue; + } // §10.11.1: a GROUPSIGN must not bridge a compound-word seam — // `an[t·h]ill`, `cart[·h]orse`, `nor[the]ast` spell the bridging digraph // out. An initial-letter contraction (§10.7 `upon`, priority 55) and a @@ -854,6 +968,7 @@ mod tests { false, false, false, + false, ); assert!(moves.iter().all(|(cells, consumed, _)| { *consumed != pattern.len() || cells != &vec![decode_unicode('⠆')] @@ -982,6 +1097,7 @@ mod tests { false, false, false, + false, ); assert!(moves.iter().any(|(cells, consumed, priority)| { @@ -1175,6 +1291,7 @@ mod tests { false, false, false, + false, ); assert!(moves.iter().any(|(cells_, consumed, priority)| { *cells_ == cells("⠼⠮") && *consumed == 1 && *priority == u16::MAX @@ -1196,6 +1313,7 @@ mod tests { false, true, false, + false, ); assert_eq!(result, None); } diff --git a/libs/braillify/src/rules/english_ueb/span.rs b/libs/braillify/src/rules/english_ueb/span.rs index 70bf66df..4a893c21 100644 --- a/libs/braillify/src/rules/english_ueb/span.rs +++ b/libs/braillify/src/rules/english_ueb/span.rs @@ -14,6 +14,10 @@ use super::korean_context::{KoreanPrefixInput, match_korean_prefix}; use crate::english::encode_english; +use std::sync::LazyLock; + +static KOREAN_WORD_ENGINE: LazyLock = + LazyLock::new(super::engine::EnglishUebEngine::new); /// One unit of Korean-context English output: a 제37항-restricted UEB contraction /// when one begins at `input.pos`, otherwise the single §28 letter cell. @@ -29,6 +33,35 @@ pub(crate) struct KoreanSpanUnit { pub(crate) contracted: bool, } +/// Encode a complete ASCII Roman run in Korean context with the shared UEB +/// contraction engine. Korean rule 37 disables wordsigns and shortforms while +/// retaining multi-letter groupsigns; the engine entry point enforces that gate. +#[expect( + clippy::too_many_arguments, + reason = "the wrapper preserves the engine's independent UEB rule gates" +)] +pub(crate) fn encode_korean_word( + chars: &[char], + suppress_caps: bool, + prepend_grade1_indicator: bool, + standing_alone: bool, + word_initial: bool, + digit_adjacent: bool, + numeric_grade1_active: bool, + apostrophe_joined_lexeme: bool, +) -> Option> { + KOREAN_WORD_ENGINE.encode_korean_word( + chars, + suppress_caps, + prepend_grade1_indicator, + standing_alone, + word_initial, + digit_adjacent, + numeric_grade1_active, + apostrophe_joined_lexeme, + ) +} + /// Encode the English unit beginning at `input.pos` to UEB cells. /// /// Returns `Err` only when the position is not an encodable English letter diff --git a/libs/braillify/src/rules/korean/rule_18.rs b/libs/braillify/src/rules/korean/rule_18.rs index ae98451a..a54ce7b7 100644 --- a/libs/braillify/src/rules/korean/rule_18.rs +++ b/libs/braillify/src/rules/korean/rule_18.rs @@ -201,6 +201,7 @@ mod tests { has_korean_char: true, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count, state, result, diff --git a/libs/braillify/src/rules/korean/rule_23.rs b/libs/braillify/src/rules/korean/rule_23.rs index 2dfdb67f..ec0886a5 100644 --- a/libs/braillify/src/rules/korean/rule_23.rs +++ b/libs/braillify/src/rules/korean/rule_23.rs @@ -181,6 +181,7 @@ mod tests { has_korean_char: false, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count: &mut skip_count, state: &mut state, result: &mut result, @@ -229,6 +230,7 @@ mod tests { has_korean_char: false, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count: &mut skip_count, state: &mut state, result: &mut result, @@ -281,6 +283,7 @@ mod tests { has_korean_char: false, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count: &mut skip_count, state: &mut state, result: &mut result, diff --git a/libs/braillify/src/rules/korean/rule_25.rs b/libs/braillify/src/rules/korean/rule_25.rs index df397054..41bbf29a 100644 --- a/libs/braillify/src/rules/korean/rule_25.rs +++ b/libs/braillify/src/rules/korean/rule_25.rs @@ -134,6 +134,7 @@ mod tests { has_korean_char: false, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count: &mut skip_count, state: &mut state, result: &mut result, diff --git a/libs/braillify/src/rules/korean/rule_27.rs b/libs/braillify/src/rules/korean/rule_27.rs index 8a55f0d7..c782ffe4 100644 --- a/libs/braillify/src/rules/korean/rule_27.rs +++ b/libs/braillify/src/rules/korean/rule_27.rs @@ -26,10 +26,18 @@ fn is_historical_context_word(word: &str) -> bool { let code = c as u32; (0xE000..=0xF8FF).contains(&code) || (0x4E00..=0x9FFF).contains(&code) - || matches!(c, ':' | '〔' | '〕') + || matches!(c, ':' | '〮' | '〯' | '〔' | '〕') }) } +fn is_geoseong(c: char) -> bool { + matches!(c, '·' | '〮') +} + +fn is_sangseong(c: char) -> bool { + matches!(c, ':' | '〯') +} + fn has_historical_context(ctx: &RuleContext) -> bool { if is_historical_context_word(&ctx.word_chars.iter().collect::()) { return true; @@ -44,10 +52,19 @@ fn has_historical_context(ctx: &RuleContext) -> bool { } fn is_middle_korean_geoseong(ctx: &RuleContext) -> bool { - if !matches!(ctx.char_type, CharType::Symbol('·')) { + let CharType::Symbol(c) = ctx.char_type else { + return false; + }; + if !is_geoseong(*c) { return false; } + // U+302E HANGUL SINGLE DOT TONE MARK is semantically unambiguous. Unlike + // U+00B7, it can never be the modern middle-dot punctuation of Rule 49. + if *c == '〮' { + return true; + } + // 단독 입력 `·`은 한국어 점자에서 두 가지 의미를 가진다: // - 일반 한국어(가운뎃점, 제49항): ⠐⠆ — rule_49가 처리 // - 중세국어(거성, 제27항): ⠸⠂ — 이 규칙이 처리 @@ -67,7 +84,7 @@ fn is_middle_korean_geoseong(ctx: &RuleContext) -> bool { } fn is_middle_korean_particle_geoseong(ctx: &RuleContext) -> bool { - matches!(ctx.char_type, CharType::Symbol('·')) + matches!(ctx.char_type, CharType::Symbol(c) if is_geoseong(*c)) && ctx.state.current_mode() == EncodingMode::MiddleKorean && ctx.next_char() == Some('에') } @@ -80,7 +97,7 @@ fn is_inline_gloss_separator(ctx: &RuleContext) -> bool { } fn is_middle_korean_sangseong(ctx: &RuleContext) -> bool { - matches!(ctx.char_type, CharType::Symbol(':')) + matches!(ctx.char_type, CharType::Symbol(c) if is_sangseong(*c)) } pub struct Rule27; @@ -99,7 +116,8 @@ impl BrailleRule for Rule27 { } fn matches(&self, ctx: &RuleContext) -> bool { - let is_potential_tone_mark = matches!(ctx.char_type, CharType::Symbol('·' | ':')); + let is_potential_tone_mark = + matches!(ctx.char_type, CharType::Symbol(c) if is_geoseong(*c) || is_sangseong(*c)); if !is_potential_tone_mark { return false; } @@ -118,15 +136,15 @@ impl BrailleRule for Rule27 { match c { '·' if is_inline_gloss_separator(ctx) => {} - '·' if is_middle_korean_particle_geoseong(ctx) => { + c if is_geoseong(*c) && is_middle_korean_particle_geoseong(ctx) => { ctx.emit(0); ctx.emit_slice(&GEOSEONG); } - '·' if ctx.state.current_mode() == EncodingMode::MiddleKorean => { + c if is_geoseong(*c) && ctx.state.current_mode() == EncodingMode::MiddleKorean => { ctx.emit_slice(&GEOSEONG); } - '·' if is_middle_korean_geoseong(ctx) => ctx.emit_slice(&GEOSEONG), - ':' => ctx.emit_slice(&SANGSEONG), + c if is_geoseong(*c) && is_middle_korean_geoseong(ctx) => ctx.emit_slice(&GEOSEONG), + c if is_sangseong(*c) => ctx.emit_slice(&SANGSEONG), _ => return Ok(RuleResult::Skip), } @@ -153,6 +171,14 @@ mod tests { let _ = Rule27.matches(&ctx); } + #[test] + fn geoseong_predicate_rejects_non_symbol_context() { + let mut owned = crate::test_helpers::CtxOwned::for_text("A", false); + let ctx = owned.ctx_at(0); + + assert!(!is_middle_korean_geoseong(&ctx)); + } + /// 제27항 — `has_historical_context` returns true when current word contains /// a hanja character (CJK Unified). Exercises lines 33-35 (own-word branch). #[test] @@ -213,6 +239,19 @@ mod tests { assert_eq!(owned.result, GEOSEONG.to_vec()); } + #[rstest::rstest] + #[case::single_dot('〮', GEOSEONG.to_vec())] + #[case::double_dot('〯', SANGSEONG.to_vec())] + fn unicode_hangul_tone_marks_use_rule_27_cells(#[case] input: char, #[case] expected: Vec) { + let mut owned = crate::test_helpers::CtxOwned::for_text(&input.to_string(), false); + let mut ctx = owned.ctx_at(0); + + let outcome = Rule27.apply(&mut ctx).unwrap(); + + assert!(matches!(outcome, RuleResult::Consumed)); + assert_eq!(owned.result, expected); + } + /// rule_27 line 106 — `_ => return Ok(Skip)` fallback for non-· non-: symbol char. #[test] fn rule27_apply_skip_for_unrelated_symbol() { diff --git a/libs/braillify/src/rules/korean/rule_28.rs b/libs/braillify/src/rules/korean/rule_28.rs index e681e4d7..8e7341d5 100644 --- a/libs/braillify/src/rules/korean/rule_28.rs +++ b/libs/braillify/src/rules/korean/rule_28.rs @@ -8,10 +8,16 @@ //! Reference: 2024 Korean Braille Standard, Chapter 4, Section 10, Article 28 use crate::char_struct::CharType; +use crate::english_logic::requires_single_letter_continuation; use crate::rules::RuleMeta; use crate::rules::context::RuleContext; +use crate::rules::english_shortform::{ + permits_grade1_boundary_after_run, requires_grade1_indicator, +}; use crate::rules::english_ueb::korean_context::KoreanPrefixInput; -use crate::rules::english_ueb::span::encode_korean_unit; +use crate::rules::english_ueb::span::{encode_korean_unit, encode_korean_word}; +use crate::rules::english_ueb::standing_alone::lower_wordsign_usable; +use crate::rules::english_ueb::token::EnglishToken; use crate::rules::traits::{BrailleRule, Phase, RuleResult}; pub static META: RuleMeta = RuleMeta { @@ -75,6 +81,16 @@ impl BrailleRule for Rule28 { return Ok(RuleResult::Skip); }; + // At index 0 the emitter may already have emitted the Roman indicator, + // so use its pre-entry snapshot. For an ASCII run later in a mixed + // print word, the live mode accurately says whether this run continues + // an existing Roman section or starts a fresh one. + let continuing_roman_section = if ctx.index == 0 { + ctx.roman_section_continues_from_previous_word + } else { + ctx.state.is_english + }; + // Enter English mode (로마자표 / 연속표) // 제39항 영어 주도 문서에서는 영자표시/연속표를 emit하지 않는다. if ctx.state.english_indicator @@ -88,6 +104,182 @@ impl BrailleRule for Rule28 { } } + // 제37항: a Roman section in Korean text spells the word with UEB + // alphabet signs and multi-letter groupsigns, while suppressing UEB + // whole-word contractions. Encode each contiguous ASCII letter run in + // one pass so the shared UEB preference/morphology algorithm can choose + // contractions across the whole word. Lowercase apostrophe continuations + // retain the legacy position-aware path because they are not fresh word + // starts. An uppercase continuation is encoded as a run so UEB 8.4.2 can + // restart capitals mode after the nonalphabetic apostrophe. + let starts_ascii_run = c.is_ascii_alphabetic() + && ctx + .index + .checked_sub(1) + .and_then(|index| ctx.word_chars.get(index)) + .is_none_or(|previous| !previous.is_ascii_alphabetic()); + let follows_apostrophe = ctx + .index + .checked_sub(1) + .and_then(|index| ctx.word_chars.get(index)) + .is_some_and(|previous| matches!(previous, '\'' | '\u{2019}')); + if starts_ascii_run && (!follows_apostrophe || c.is_ascii_uppercase()) { + let run_end = ctx.index + + ctx.word_chars[ctx.index..] + .iter() + .take_while(|ch| ch.is_ascii_alphabetic()) + .count(); + let run = &ctx.word_chars[ctx.index..run_end]; + // The token rule pre-emits capitals-word mode only for the initial + // uppercase letters-sequence. UEB 8.4.2 ends that mode at a + // nonletter, so a later run (the final `T` in official `AT&T`) + // must produce its own capitalization indicator. + let caps_already_emitted = ctx.state.triple_big_english + || (ctx.index == 0 + && ctx.is_all_uppercase + && ctx.word_len() >= 2 + && ctx.ascii_starts_at_beginning); + let word_initial = ctx.index == 0 + || ctx.word_chars.get(ctx.index - 1).is_some_and(|previous| { + crate::utils::is_korean_char(*previous) + || matches!( + previous, + '(' | '[' + | '{' + | '\u{2018}' + | '\u{201c}' + | '"' + | '-' + | '\u{2010}' + | '\u{2011}' + | '\u{2012}' + | '\u{2013}' + | '\u{2014}' + ) + }); + let run_is_all_uppercase = run.iter().all(|ch| ch.is_ascii_uppercase()); + let is_standing_alone_ordinary_run = !run_is_all_uppercase + && word_initial + && permits_grade1_boundary_after_run(&ctx.word_chars[run_end..]); + // Rule 37's PDF example, "그는 Can you help me?라고 도움을 요청했다.", + // suppresses a whole-word sign for the first Roman word (`Can`) but retains + // the UEB wordsign for the following `you`. Rule 29 keeps consecutive + // Roman words in the same section, so every complete ordinary-cased word + // after the first Roman word has the same continuation status, including + // the final word of a phrase. UEB capitalization does not suppress a + // wordsign, hence Title-case `Like`/`This` follows the same rule. All-caps + // runs remain excluded because Rule 10.12.1 initialisms and emphasized + // words have the same surface form and require pronunciation semantics. + // Rule 39's "What is 김치 in English?" resumes the surrounding English + // passage after Korean, so the persistent English-dominant gate retains + // the resumed `in` wordsign. Neither gate depends on a corpus reference. + let whole_print_word = ctx.index == 0 && run_end == ctx.word_chars.len(); + let wrap_wordsign = ctx.state.english_dominant_wrap_active && whole_print_word; + // 제37항 붙임: these six words are spelled with alphabet signs and + // applicable groupsigns even when they occur later in the Roman + // section immediately before its terminator. Other continuation + // words, such as official `you` in `Can you help me?`, retain their + // ordinary UEB wordsign. + let lower_run = run + .iter() + .map(|ch| ch.to_ascii_lowercase()) + .collect::(); + let is_lower_wordsign = matches!( + lower_run.as_str(), + "be" | "enough" | "his" | "in" | "was" | "were" + ); + let rule_37_korean_context_exception = !ctx.state.english_dominant_wrap_active + && !ctx.state.roman_section_is_english_context + && is_lower_wordsign; + // UEB 10.5 gives lower wordsigns a stricter boundary than ordinary + // standing-alone wordsigns. In particular, a hyphen, dash, quote, + // or lower punctuation cell touching either side forces spelling. + // Reuse the English engine's boundary predicate instead of treating + // Rule 28's general grade-1 boundary as sufficient (`In-house`). + let previous_boundary = ctx + .index + .checked_sub(1) + .and_then(|index| ctx.word_chars.get(index)) + .copied() + .map(EnglishToken::Symbol); + let next_boundary = ctx + .word_chars + .get(run_end) + .copied() + .map(EnglishToken::Symbol); + let lower_wordsign_boundary_permits = !is_lower_wordsign + || lower_wordsign_usable(previous_boundary.as_ref(), next_boundary.as_ref()); + let standalone_wordsign = is_standing_alone_ordinary_run + && (wrap_wordsign || continuing_roman_section) + && !rule_37_korean_context_exception + && lower_wordsign_boundary_permits; + let digit_adjacent = ctx + .index + .checked_sub(1) + .and_then(|index| ctx.word_chars.get(index)) + .is_some_and(|ch| ch.is_ascii_digit()) + || ctx + .word_chars + .get(run_end) + .is_some_and(|ch| ch.is_ascii_digit()); + let numeric_grade1_active = ctx + .index + .checked_sub(1) + .and_then(|index| ctx.word_chars.get(index)) + .is_some_and(|ch| ch.is_ascii_digit()) + && ctx.word_chars[..ctx.index.saturating_sub(1)] + .iter() + .any(|ch| ch.is_ascii_alphabetic()); + // UEB 5.7.1-5.7.2 and 5.8.1: grade 1 precedes the capitalization + // marker when a standing letter/letters-sequence would otherwise be + // read as an alphabetic wordsign or shortform. A bare one-letter + // Rule 28 specimen (`K`) keeps the PDF's plain alphabet cell, while + // the same letter in running text (`K-POP`, `ARIRANG K방산Fn`) is a + // UEB 5.7.1 standing letter. Multi-letter uppercase tokens have + // already had their capitalization mode emitted by + // `UppercasePassageRule`; an adjacent digit is not a standing-alone + // boundary, whereas a hyphen or dash explicitly is (UEB 2.6.1). + let uppercase_run = run.iter().collect::(); + let entire_isolated_rule_28_specimen = ctx.index == 0 + && run_end == ctx.word_chars.len() + && ctx.prev_word.is_empty() + && ctx.remaining_words.is_empty(); + let single_letter_wordsign_collision = run.len() == 1 + && requires_single_letter_continuation(run[0]) + && ctx.index == 0 + && ctx.roman_section_continues_from_previous_word + && !entire_isolated_rule_28_specimen; + let shortform_collision = requires_grade1_indicator(&uppercase_run); + let prepend_grade1_indicator = !caps_already_emitted + && word_initial + && !digit_adjacent + && run.iter().all(|ch| ch.is_ascii_uppercase()) + && permits_grade1_boundary_after_run(&ctx.word_chars[run_end..]) + && (single_letter_wordsign_collision || shortform_collision); + let apostrophe_joined_lexeme = + crate::rules::english_ueb::pronunciation::apostrophe_elided_recorded_word_at( + ctx.word_chars, + ctx.index, + run_end, + ); + if let Some(cells) = encode_korean_word( + run, + caps_already_emitted, + prepend_grade1_indicator, + standalone_wordsign, + word_initial, + digit_adjacent, + numeric_grade1_active, + apostrophe_joined_lexeme, + ) { + ctx.emit_slice(&cells); + *ctx.skip_count = run.len().saturating_sub(1); + ctx.state.is_english = true; + ctx.state.needs_english_continuation = false; + return Ok(RuleResult::Consumed); + } + } + // Uppercase indicators (single/consecutive uppercase run) if (!ctx.is_all_uppercase || ctx.word_len() < 2 || !ctx.ascii_starts_at_beginning) && !ctx.state.is_big_english @@ -139,7 +331,9 @@ impl BrailleRule for Rule28 { #[cfg(test)] mod tests { use super::*; + use crate::rules::context::EncodingMode; use crate::unicode::decode_unicode; + use crate::{EncodeOptions, encode_to_unicode, encode_with_options}; /// 제28항 — 영문자 점역. 소문자/대문자 모두 동일 점형으로 인코딩. #[rstest::rstest] @@ -173,6 +367,286 @@ mod tests { assert_eq!(uppercase_indicators(single, is_word, run), expected); } + /// 제37항 PDF examples: Korean-context Roman words suppress whole-word + /// contractions while retaining their applicable multi-letter groupsigns. + #[rstest::rstest] + #[case::initial_letter_groupsign("every", &[52, 16, 17, 61, 50])] + #[case::lower_and_strong_groupsigns("enough", &[52, 34, 51, 35, 50])] + #[case::strong_contraction_inside_word("rather", &[52, 23, 1, 46, 23, 50])] + #[case::entry_lower_wordsign_spelled_as_letters("in", &[52, 10, 29, 50])] + fn korean_roman_words_share_ueb_groupsign_algorithm( + #[case] input: &str, + #[case] expected: &[u8], + ) { + let options = EncodeOptions { + default_mode: Some(EncodingMode::Korean), + }; + assert_eq!(encode_with_options(input, &options).unwrap(), expected); + } + + /// 제37항 PDF 문장 전체를 공개 encoder로 통과시켜, 첫 Roman 어절의 + /// complete wordsign 억제와 뒤따르는 Roman phrase 경로를 함께 검증한다. + #[test] + fn rule_37_official_sentence_uses_shared_roman_engine() { + assert_eq!( + encode_to_unicode("그는 Can you help me?라고 도움을 요청했다.").unwrap(), + "⠈⠪⠉⠵⠀⠴⠠⠉⠁⠝⠀⠽⠀⠓⠑⠇⠏⠀⠍⠑⠦⠐⠣⠈⠥⠀⠊⠥⠍⠢⠮⠀⠬⠰⠻⠚⠗⠌⠊⠲" + ); + } + + /// Rule 37 limits its whole-word-contraction suppression to the Roman word + /// immediately following the indicator. Apostrophe punctuation in that + /// first word and a Korean suffix attached to the final word do not start a + /// second Roman section, so subsequent `do` and `this` retain UEB wordsigns. + #[test] + fn rule_37_continuation_survives_apostrophe_and_attached_korean_suffix() { + assert_eq!( + encode_to_unicode("그는 Let's do this라고 말했다.").unwrap(), + "⠈⠪⠉⠵⠀⠴⠠⠇⠑⠞⠄⠎⠀⠙⠀⠹⠲⠐⠣⠈⠥⠀⠑⠂⠚⠗⠌⠊⠲" + ); + } + + /// UEB 5.7.2/5.8.1/10.9.7 complete-shortform handling through the complete + /// Korean encoder. Every Roman surface comes directly from the PDF examples + /// (`CD`, `ALT`, `NEC`); the Korean wrapper exercises only rule 28/29/34 routing. + #[rstest::rstest] + #[case::standing_alone_could("가(CD)", "⠫⠦⠄⠴⠰⠠⠠⠉⠙⠠⠴")] + #[case::alt_example("가(ALT)", "⠫⠦⠄⠴⠰⠠⠠⠁⠇⠞⠠⠴")] + #[case::nec_example("가(NEC)", "⠫⠦⠄⠴⠰⠠⠠⠝⠑⠉⠠⠴")] + fn attached_allcaps_complete_shortform_uses_grade1( + #[case] input: &str, + #[case] expected: &str, + ) { + assert_eq!(crate::encode_to_unicode(input).unwrap(), expected); + } + + /// UEB 2.6.1-2.6.3 boundaries must be enforced on the Rule28 path as well as + /// on whitespace tokens. A leading quote forces this path because the token + /// itself no longer starts with ASCII; `CD`/`LLC` are official 10.9 examples. + #[rstest::rstest] + #[case::closing_quote("‘CD’", true)] + #[case::attached_after_korean("가CD", true)] + #[case::non_shortform_after_korean("가KBS", false)] + #[case::digit_after_hyphen("5-CD-678", true)] + #[case::closing_group_before_korean_middle_dot("(CD)·현금", true)] + #[case::adjacent_digit("‘CD47", false)] + #[case::slash_continuation("‘CD/ATM", false)] + #[case::opening_group_after_sequence("‘LLC(회사)", false)] + fn noninitial_ascii_run_respects_grade1_boundary(#[case] input: &str, #[case] expected: bool) { + let encoded = crate::encode(input).unwrap(); + assert_eq!( + encoded + .windows(3) + .any(|window| window == [48, UPPERCASE_SINGLE, UPPERCASE_SINGLE]), + expected + ); + } + + /// UEB 5.7.1/5.8.1: a single capital wordsign letter standing in running + /// text needs grade 1 before its capital indicator. The rule is structural: + /// the following boundary may be whitespace, a hyphen, or a Korean code + /// boundary. `a`, `i`, and `o` are excluded by the shared UEB predicate. + #[rstest::rstest] + #[case::roman_number_chain("가 X5 M 나", 'm')] + #[case::hyphen_bounded("가 EAFF E-1 나", 'e')] + #[case::korean_code_boundary("가 ARIRANG K방산Fn 나", 'k')] + fn running_single_capital_wordsign_letter_uses_grade1( + #[case] input: &str, + #[case] letter: char, + ) { + let encoded = crate::encode(input).unwrap(); + let letter = crate::english::encode_english(letter).unwrap(); + + assert!(encoded.windows(3).any(|window| { + window + == [ + crate::rules::korean::rule_29::ENGLISH_CONTINUATION, + UPPERCASE_SINGLE, + letter, + ] + })); + } + + /// Korean Rule 28's alphabet table is a specimen, not running contracted + /// English. Its isolated capital letters therefore retain the plain Rule + /// 28 form without a UEB grade-1 prefix. + #[test] + fn isolated_rule_28_capital_specimen_stays_plain() { + assert_eq!(crate::encode_to_unicode("K").as_deref(), Ok("⠠⠅")); + } + + /// UEB 8.4.2 keeps an internal apostrophe in the Roman letters-sequence but + /// terminates capitals-word mode at that nonalphabetic symbol. The Roman + /// surfaces are official UEB examples; the neutral Korean wrapper exercises + /// Rule 28/29 routing. Korean Rule 37 still suppresses the `that` wordsign in + /// `THAT'S`, so its initial run retains the permitted `th` groupsign instead. + #[rstest::rstest] + #[case::official_name("가 O'Hara 나", "⠫⠀⠴⠠⠕⠄⠠⠓⠜⠁⠲⠀⠉")] + #[case::official_contraction("가 DON'T 나", "⠫⠀⠴⠠⠠⠙⠕⠝⠄⠠⠞⠲⠀⠉")] + #[case::official_possessive("가 THAT'S 나", "⠫⠀⠴⠠⠠⠹⠁⠞⠄⠠⠎⠲⠀⠉")] + #[case::official_two_letter_suffix("가 SHE'LL 나", "⠫⠀⠴⠠⠠⠩⠑⠄⠠⠠⠇⠇⠲⠀⠉")] + fn korean_wrapper_restarts_capitals_after_internal_apostrophe( + #[case] input: &str, + #[case] expected: &str, + ) { + assert_eq!(crate::encode_to_unicode(input).as_deref(), Ok(expected)); + } + + /// UEB 10.6.8 keeps `en` inside a capitals word when the letters belong to + /// an ordinarily pronounced word. Removing the internal apostrophe yields + /// recorded `opening`, which distinguishes this emphasis from a 10.12.1 + /// initialism while exercising the Korean Rule 28/37 wrapper. + #[test] + fn apostrophe_elided_lexeme_contracts_inside_capitals_word() { + assert_eq!( + crate::encode_to_unicode("가 O'PENing 나").as_deref(), + Ok("⠫⠀⠴⠠⠕⠄⠠⠠⠏⠢⠠⠄⠬⠲⠀⠉") + ); + } + + #[test] + fn english_dominant_wrap_resumes_ueb_wordsigns_after_korean_span() { + let mut owned = crate::test_helpers::CtxOwned::for_text("in", true); + owned.state.is_english = true; + owned.state.english_dominant_wrap_active = true; + let mut ctx = owned.ctx_at(0); + + assert!(matches!( + Rule28.apply(&mut ctx).unwrap(), + RuleResult::Consumed + )); + assert_eq!(owned.result, vec![20]); + } + + /// A document-level rule-39 wrap signal must not turn each separate Roman + /// annotation inside a mixed Korean print word into a continuation. Each + /// parenthesized item below begins a fresh rule-37 Roman section and is + /// therefore spelled, even when its surface is also a UEB wordsign. + #[rstest::rstest] + #[case::titlecase_us( + "(Us)", + 1, + &[52, 32, decode_unicode('⠥'), decode_unicode('⠎')] + )] + #[case::lowercase_it("(it)", 1, &[52, decode_unicode('⠊'), decode_unicode('⠞')])] + fn mixed_print_word_starts_fresh_rule_37_section( + #[case] input: &str, + #[case] index: usize, + #[case] expected: &[u8], + ) { + let mut owned = crate::test_helpers::CtxOwned::for_text(input, true); + owned.state.english_dominant_wrap_active = true; + let mut ctx = owned.ctx_at(index); + + assert!(matches!( + Rule28.apply(&mut ctx).unwrap(), + RuleResult::Consumed + )); + assert_eq!(owned.result, expected); + } + + /// Rule 37's PDF sentence `Can you help me?` permits wordsigns after the + /// first Roman word. Rule 29 keeps the final Roman word in that same section, + /// and UEB capitalization leaves the wordsign itself unchanged. + #[rstest::rstest] + #[case::interior_lowercase("you", "Can", &[decode_unicode('⠽')])] + #[case::final_titlecase("This", "Like", &[32, decode_unicode('⠹')])] + #[case::final_lowercase("will", "Boys", &[decode_unicode('⠺')])] + fn rule_37_continuation_word_uses_standalone_wordsign( + #[case] input: &str, + #[case] previous: &str, + #[case] expected: &[u8], + ) { + let mut owned = crate::test_helpers::CtxOwned::for_text(input, true) + .with_prev_word(previous) + .with_roman_section_continuation(); + owned.state.is_english = true; + let mut ctx = owned.ctx_at(0); + + assert!(matches!( + Rule28.apply(&mut ctx).unwrap(), + RuleResult::Consumed + )); + assert_eq!(owned.result, expected); + } + + /// 제37항 붙임: these words stay expanded throughout a Korean-context + /// Roman section, including immediately before the Roman terminator. + #[rstest::rstest] + #[case::be("be", "⠃⠑")] + #[case::enough("enough", "⠢⠳⠣")] + #[case::his("his", "⠓⠊⠎")] + #[case::in_word("in", "⠊⠝")] + #[case::was("was", "⠺⠁⠎")] + #[case::were("were", "⠺⠻⠑")] + fn rule_37_terminator_exceptions_remain_expanded(#[case] input: &str, #[case] expected: &str) { + let mut owned = crate::test_helpers::CtxOwned::for_text(input, true) + .with_prev_word("Can") + .with_roman_section_continuation(); + owned.state.is_english = true; + let mut ctx = owned.ctx_at(0); + + assert!(matches!( + Rule28.apply(&mut ctx).unwrap(), + RuleResult::Consumed + )); + assert_eq!( + owned.result, + expected.chars().map(decode_unicode).collect::>() + ); + } + + /// The NIKL's rule consultation distinguishes Korean metalinguistic Roman + /// material from a visibly English phrase. In the latter context UEB 10.5 + /// applies to all six lower wordsigns, even though the surrounding document + /// is Korean. + #[rstest::rstest] + #[case::be_word("be", '⠆')] + #[case::enough_word("enough", '⠢')] + #[case::his_word("his", '⠦')] + #[case::in_word("in", '⠔')] + #[case::was_word("was", '⠴')] + #[case::were_word("were", '⠶')] + fn english_phrase_uses_ueb_lower_wordsigns(#[case] word: &str, #[case] wordsign: char) { + let input = format!("제목(Alpha {word} Omega)이다."); + let actual = encode_to_unicode(&input).expect("English phrase must encode"); + let expected = format!("⠀{wordsign}⠀"); + + assert!( + actual.contains(&expected), + "missing UEB lower wordsign in English phrase: {actual}" + ); + } + + #[test] + fn english_phrase_context_survives_a_preceding_capitals_passage() { + let actual = + encode_to_unicode("제목 ‘2023 SHINHWA WDJ FANPARTY COME TO LIFE in TAIPEI’는 끝이다.") + .expect("capitalized English title must encode"); + + assert!( + actual.contains("⠀⠔⠀"), + "caps-passage mode prefix lost the English phrase context: {actual}" + ); + } + + /// UEB 10.5: a lower wordsign touching a hyphen is not usable even when the + /// surrounding Roman section is clearly an English title. + #[test] + fn english_phrase_spells_lower_wordsign_touching_hyphen() { + let actual = encode_to_unicode("제목(Alpha In-house Teams)이다.") + .expect("hyphenated English phrase must encode"); + + assert!( + actual.contains("⠀⠠⠊⠝⠤"), + "hyphen-adjacent `In` must remain expanded: {actual}" + ); + assert!( + !actual.contains("⠀⠠⠔⠤"), + "hyphen-adjacent `In` must not use its lower wordsign: {actual}" + ); + } + #[test] fn apply_skips_non_korean() { let mut owned = crate::test_helpers::CtxOwned::for_text("A", false); @@ -210,6 +684,7 @@ mod tests { has_korean_char: false, is_all_uppercase: false, ascii_starts_at_beginning: true, + roman_section_continues_from_previous_word: false, skip_count: &mut skip, state: &mut state, result: &mut out, @@ -219,6 +694,22 @@ mod tests { let _ = outcome; } + /// A rule invocation that resumes inside an ASCII run must restart a + /// capitals indicator and stop its extent at the following lowercase + /// letter. Normal full-word routing skips over this position in one pass; + /// this direct check preserves the defensive continuation behavior. + #[test] + fn uppercase_continuation_stops_before_following_lowercase_letter() { + let mut owned = crate::test_helpers::CtxOwned::for_text("aBc", false); + owned.state.is_english = true; + let mut ctx = owned.ctx_at(1); + + let outcome = Rule28.apply(&mut ctx).unwrap(); + + assert!(matches!(outcome, RuleResult::Consumed)); + assert_eq!(owned.result.first(), Some(&UPPERCASE_SINGLE)); + } + /// rule_28 line 64 — `let-else return Skip` for non-English ctx. #[test] fn rule28_apply_skip_for_non_english_ctx() { diff --git a/libs/braillify/src/rules/korean/rule_40.rs b/libs/braillify/src/rules/korean/rule_40.rs index 96f663e8..7fd99ac6 100644 --- a/libs/braillify/src/rules/korean/rule_40.rs +++ b/libs/braillify/src/rules/korean/rule_40.rs @@ -63,10 +63,11 @@ impl BrailleRule for Rule40 { // (rule_69.rs:174-181 matches() + 184-196 apply() 참조) if !ctx.state.is_number { - // 제43항: skip prefix after continuation characters (. or ,) - let needs_prefix = ctx - .prev_char() - .is_none_or(|prev| !is_number_continuation(prev)); + // 제43항: 마침표/쉼표가 *숫자 사이*에 있을 때에만 뒤 수표를 + // 생략한다. `M.2`, `No.1`, `2만,4142`처럼 문장 부호의 왼쪽이 + // 숫자가 아닌 경우에는 제40항에 따라 새 수표를 적는다. + let needs_prefix = + !is_number_continuation(ctx.word_chars, ctx.index, ctx.state.english_indicator); if needs_prefix { ctx.emit(NUMBER_INDICATOR); // 제61항: apostrophe/right single quote before number emits ⠄ after 수표 @@ -85,10 +86,28 @@ impl BrailleRule for Rule40 { } } -/// Check if the previous character is a continuation character (. or ,) -/// that should suppress the number indicator on the next digit. -pub fn is_number_continuation(prev: char) -> bool { - prev == '.' || prev == ',' +/// Return whether the digit at `index` follows `digit + (. or ,)`. +/// +/// 제43항의 적용 조건은 문장 부호 자체가 아니라 그 문장 부호가 두 숫자 +/// 사이에 놓였는지이다. 따라서 로마자나 한글 뒤의 마침표/쉼표는 새 숫자 +/// 묶음의 수표를 생략하지 않는다. +pub fn is_number_continuation(word_chars: &[char], index: usize, in_korean_document: bool) -> bool { + if index == 0 || !matches!(word_chars[index - 1], '.' | ',') { + return false; + } + + if in_korean_document { + return index >= 2 && word_chars[index - 2].is_numeric(); + } + + // UEB 6.3.1: numeric mode continues through a sequence of full stops or + // commas. It can therefore span `4..7`, but it was never established in + // an identifier such as `M.2`. + word_chars[..index] + .iter() + .rev() + .find(|ch| !matches!(ch, '.' | ',')) + .is_some_and(|ch| ch.is_numeric()) } #[cfg(test)] @@ -110,14 +129,51 @@ mod tests { assert!(encode_digit('a').is_err()); } - /// `is_number_continuation` — `.` / `,` 만 숫자 흐름에 포함. + /// 제43항 — `.` / `,`가 실제로 숫자 사이에 있을 때만 숫자 흐름에 포함. + #[rstest::rstest] + #[case::korean_decimal("3.9", 2, true, true)] + #[case::korean_grouped("1,000", 2, true, true)] + #[case::korean_repeated_period("4..7", 3, true, false)] + #[case::ueb_repeated_period("4..7", 3, false, true)] + #[case::roman_period("M.2", 2, false, false)] + #[case::roman_period_in_korean("M.2", 2, true, false)] + #[case::roman_comma("X,1", 2, true, false)] + #[case::korean_comma("2만,4142", 3, true, false)] + #[case::leading_period(".47", 1, false, false)] + #[case::hyphen("3-4", 2, false, false)] + #[case::first_digit("7", 0, false, false)] + fn continuation_chars( + #[case] input: &str, + #[case] index: usize, + #[case] in_korean_document: bool, + #[case] expected: bool, + ) { + assert_eq!( + is_number_continuation( + &input.chars().collect::>(), + index, + in_korean_document, + ), + expected + ); + } + + /// 제35항/제40항/제43항 — 로마자 뒤 마침표는 숫자 사이의 소수점이 + /// 아니므로 뒤 숫자에는 수표를 새로 적는다. #[rstest::rstest] - #[case::period('.', true)] - #[case::comma(',', true)] - #[case::space(' ', false)] - #[case::hyphen('-', false)] - fn continuation_chars(#[case] ch: char, #[case] expected: bool) { - assert_eq!(is_number_continuation(ch), expected); + #[case::capital_identifier("가 M.2 나", "⠍⠲⠼⠃")] + #[case::all_caps_identifier("가 NO.1 나", "⠕⠲⠼⠁")] + #[case::korean_before_comma("가 2만,4142명 나", "⠑⠒⠐⠼⠙")] + #[case::ueb_multiple_periods("4..7", "⠼⠙⠲⠲⠛")] + fn non_numeric_left_side_does_not_suppress_number_indicator( + #[case] input: &str, + #[case] expected_fragment: &str, + ) { + let actual = crate::encode_to_unicode(input).expect("input must encode"); + assert!( + actual.contains(expected_fragment), + "missing rule-40 number indicator in {actual}" + ); } /// PDF 제40항 + 제69항 — numeric prefix followed by ASCII unit (kg, cm, etc.) diff --git a/libs/braillify/src/rules/korean/rule_41.rs b/libs/braillify/src/rules/korean/rule_41.rs index 8501ffec..aceeb27a 100644 --- a/libs/braillify/src/rules/korean/rule_41.rs +++ b/libs/braillify/src/rules/korean/rule_41.rs @@ -1,8 +1,9 @@ -//! 제41항 — 숫자 또는 로마자 구간에서 쉼표는 ⠂(2)으로 적는다. +//! 제41항 — 숫자 사이에 붙어 나오는 쉼표는 ⠂(2)으로 적는다. //! -//! When a comma appears between digits (e.g., "1,000") or between ASCII letters -//! and alphanumeric characters, it uses the numeric comma ⠂ instead of the -//! standard Korean comma ⠐. +//! When a comma is attached between digits (e.g., "1,000"), it uses the numeric +//! comma ⠂ instead of the standard Korean comma ⠐. A whitespace boundary means +//! the comma is ordinary punctuation under rule 49, not an attached numeric +//! comma under this rule. //! //! Reference: 2024 Korean Braille Standard, Chapter 5, Section 11, Article 41 @@ -15,7 +16,7 @@ pub static META: RuleMeta = RuleMeta { subsection: None, name: "numeric_comma", standard_ref: "2024 Korean Braille Standard, Ch.5 Sec.11 Art.41", - description: "Comma between digits/letters uses ⠂ (2) instead of standard comma", + description: "Attached comma within a numeric/ASCII sequence uses ⠂ (2)", }; /// Numeric comma braille code. @@ -23,7 +24,7 @@ const NUMERIC_COMMA: u8 = 2; // ⠂ /// Plugin struct for the rule engine. /// -/// Handles comma encoding in numeric/English context. +/// Handles attached comma encoding in numeric/English context. /// Runs before generic punctuation (rule_49) to intercept commas. pub struct Rule41; @@ -49,7 +50,9 @@ impl BrailleRule for Rule41 { } let (has_numeric_prefix, has_ascii_prefix) = scan_prefix(ctx.word_chars, ctx.index); - let next_char = get_next_char(ctx); + // 제41항의 "붙어 나오는" 경계만 본다. `remaining_words`까지 + // 건너뛰면 `1, 2`의 일반 쉼표를 숫자 쉼표로 오분류한다. + let next_char = ctx.word_chars.get(ctx.index + 1).copied(); let next_is_digit = next_char.is_some_and(|ch| ch.is_ascii_digit()); let next_is_ascii = next_char.is_some_and(|ch| ch.is_ascii_alphabetic()); let next_is_alphanumeric = next_is_digit || next_is_ascii; @@ -78,15 +81,6 @@ fn scan_prefix(word_chars: &[char], index: usize) -> (bool, bool) { } } -/// Get the next character (within word or from next word). -fn get_next_char(ctx: &RuleContext) -> Option { - if ctx.index + 1 < ctx.word_chars.len() { - Some(ctx.word_chars[ctx.index + 1]) - } else { - ctx.remaining_words.first().and_then(|w| w.chars().next()) - } -} - #[cfg(test)] mod tests { use super::*; @@ -121,7 +115,7 @@ mod tests { assert!(!Rule41.matches(&ctx)); } - /// 제41항 — 숫자/로마자 구간 안의 쉼표는 숫자 쉼표 규칙이 잡는다. + /// 제41항 숫자 쉼표와 UEB의 같은-token 로마자 쉼표 경로. #[rstest::rstest] #[case::between_digits("1,000", 1)] #[case::between_ascii_letters("A,B", 1)] @@ -133,6 +127,45 @@ mod tests { assert!(Rule41.matches(&ctx)); } + #[rstest::rstest] + // PDF physical p.209: `제5열 버튼(3, 7 혹은 S)`. + #[case::music_button_list("3,", "7")] + // PDF physical p.142: `1/3, 2/3의 길이`. + #[case::music_fraction_list("1/3,", "2/3의")] + fn rule41_does_not_cross_whitespace_token_boundaries( + #[case] current_word: &str, + #[case] next_word: &str, + ) { + let mut owned = crate::test_helpers::CtxOwned::for_text(current_word, false) + .with_remaining_words([next_word]); + let comma_index = current_word + .chars() + .position(|ch| ch == ',') + .expect("test word must contain a comma"); + let ctx = owned.ctx_at(comma_index); + + assert!(!Rule41.matches(&ctx)); + } + + /// 제41항/제49항 PDF 예제를 전체 인코더로 통과시켜 붙은 숫자 쉼표와 + /// 일반 한글 쉼표의 서로 다른 셀을 함께 고정한다. + #[rstest::rstest] + #[case::rule41_grouped_number("9,375명", '⠂')] + #[case::rule41_verse_reference("창세기 12,1-9", '⠂')] + #[case::rule49_korean_list("근면, 검소, 협동은 우리 겨레의 미덕이다.", '⠐')] + fn full_encoder_preserves_pdf_comma_boundaries( + #[case] input: &str, + #[case] expected_comma: char, + ) { + let comma_byte = input.find(',').expect("PDF example must contain comma"); + let prefix = crate::encode_to_unicode(&input[..comma_byte]).expect("prefix must encode"); + let actual = crate::encode_to_unicode(input).expect("PDF example must encode"); + let comma_cell = actual.chars().nth(prefix.chars().count()); + + assert!(actual.starts_with(&prefix)); + assert_eq!(comma_cell, Some(expected_comma)); + } + /// rule_41 line 75 — `j -= 1;` when prev char is a space (continues backward scan). #[test] fn scan_prefix_skips_space_then_finds_digit() { diff --git a/libs/braillify/src/rules/korean/rule_44.rs b/libs/braillify/src/rules/korean/rule_44.rs index f772aecf..710fa5a7 100644 --- a/libs/braillify/src/rules/korean/rule_44.rs +++ b/libs/braillify/src/rules/korean/rule_44.rs @@ -22,6 +22,14 @@ pub static META: RuleMeta = RuleMeta { /// Choseong characters that could be confused with digit braille patterns. const CONFUSABLE_CHOSEONG: [char; 7] = ['ㄴ', 'ㄷ', 'ㅁ', 'ㅋ', 'ㅌ', 'ㅍ', 'ㅎ']; +pub(crate) fn is_number_confusable_korean_char(ch: char) -> bool { + matches!( + CharType::new(ch), + Ok(CharType::Korean(korean)) + if CONFUSABLE_CHOSEONG.contains(&korean.cho) || ch == '운' + ) +} + /// Plugin struct for the rule engine. /// /// Inserts a space (code 0) before Korean syllables with confusable choseong @@ -46,10 +54,10 @@ impl BrailleRule for Rule44 { if !ctx.state.is_number { return false; } - let CharType::Korean(korean) = ctx.char_type else { + let CharType::Korean(_) = ctx.char_type else { return false; }; - CONFUSABLE_CHOSEONG.contains(&korean.cho) || ctx.current_char() == '운' + is_number_confusable_korean_char(ctx.current_char()) } fn apply(&self, ctx: &mut RuleContext) -> Result { @@ -83,6 +91,24 @@ mod tests { } } + #[rstest::rstest] + #[case::nieun("는", true)] + #[case::digeut("당", true)] + #[case::mieum("명", true)] + #[case::kieuk("칸", true)] + #[case::tieut("톤", true)] + #[case::pieup("평", true)] + #[case::hieuh("항", true)] + #[case::un_abbreviation("운", true)] + #[case::vowel_initial("이다", false)] + #[case::non_korean("A", false)] + fn detects_number_confusable_following_korean(#[case] input: &str, #[case] expected: bool) { + assert_eq!( + is_number_confusable_korean_char(input.chars().next().unwrap()), + expected + ); + } + #[test] fn meta_is_correct() { assert_eq!(META.section, "44"); @@ -109,6 +135,7 @@ mod tests { has_korean_char: true, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count: &mut skip, state: &mut state, result: &mut out, diff --git a/libs/braillify/src/rules/korean/rule_49.rs b/libs/braillify/src/rules/korean/rule_49.rs index f5c034d3..6d57efaa 100644 --- a/libs/braillify/src/rules/korean/rule_49.rs +++ b/libs/braillify/src/rules/korean/rule_49.rs @@ -293,6 +293,7 @@ mod tests { has_korean_char: false, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count: &mut skip_count, state: &mut state, result: &mut result, @@ -321,6 +322,7 @@ mod tests { has_korean_char: false, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count: &mut skip_count, state: &mut state, result: &mut result, diff --git a/libs/braillify/src/rules/korean/rule_53.rs b/libs/braillify/src/rules/korean/rule_53.rs index 37f535da..b712160e 100644 --- a/libs/braillify/src/rules/korean/rule_53.rs +++ b/libs/braillify/src/rules/korean/rule_53.rs @@ -121,6 +121,7 @@ mod tests { has_korean_char: false, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count, state, result, diff --git a/libs/braillify/src/rules/korean/rule_57.rs b/libs/braillify/src/rules/korean/rule_57.rs index 3359f22c..5f58837a 100644 --- a/libs/braillify/src/rules/korean/rule_57.rs +++ b/libs/braillify/src/rules/korean/rule_57.rs @@ -135,6 +135,7 @@ mod tests { has_korean_char: false, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count: &mut skip, state: &mut state, result: &mut out, @@ -161,6 +162,7 @@ mod tests { has_korean_char: false, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count: &mut skip, state: &mut state, result: &mut out, @@ -188,6 +190,7 @@ mod tests { has_korean_char: false, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count: &mut skip, state: &mut state, result: &mut out, @@ -215,6 +218,7 @@ mod tests { has_korean_char: false, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count: &mut skip, state: &mut state, result: &mut out, diff --git a/libs/braillify/src/rules/korean/rule_68.rs b/libs/braillify/src/rules/korean/rule_68.rs index d7e18bb2..1935af95 100644 --- a/libs/braillify/src/rules/korean/rule_68.rs +++ b/libs/braillify/src/rules/korean/rule_68.rs @@ -40,13 +40,26 @@ fn encode_unicode_cells(unicode: &str) -> Vec { } fn should_insert_separator_after_symbol(ctx: &RuleContext) -> bool { - matches!(ctx.current_char(), '㎡') && matches!(ctx.next_char(), Some('는' | '은')) + matches!(ctx.current_char(), '㎡') + && ctx + .next_char() + .is_some_and(super::rule_44::is_number_confusable_korean_char) } pub fn is_rule_68_symbol(c: char) -> bool { MAPPINGS.iter().any(|(candidate, _)| *candidate == c) } +/// Return the PDF-defined cells for a single Rule 68 symbol. Rule 69 reuses +/// this owning-rule encoding when a supported compatibility unit has a pure +/// ASCII NFKC spelling (for example, the `ha` spelling of `㏊`). +pub(crate) fn encode_rule_68_symbol(c: char) -> Option> { + MAPPINGS + .iter() + .find(|(candidate, _)| *candidate == c) + .map(|(_, unicode)| encode_unicode_cells(unicode)) +} + fn is_superscript_symbol(c: char) -> bool { matches!(c, '⁺' | '⁻') } @@ -214,17 +227,21 @@ impl BrailleRule for Rule68 { return Ok(RuleResult::Consumed); } - let Some((_, unicode)) = MAPPINGS - .iter() - .find(|(candidate, _)| *candidate == ctx.current_char()) - else { + let Some(mut encoded) = encode_rule_68_symbol(ctx.current_char()) else { return Ok(RuleResult::Skip); }; - let encoded = encode_unicode_cells(unicode); + let is_roman_unit = matches!(ctx.current_char(), '㎡' | '㏊'); + let continues = is_roman_unit + && super::rule_69::adjust_roman_unit_boundary(ctx, ctx.index + 1, &mut encoded); ctx.emit_slice(&encoded); if should_insert_separator_after_symbol(ctx) { ctx.emit(0); } + if is_roman_unit { + ctx.state.is_english = continues; + ctx.state.needs_english_continuation = false; + ctx.state.roman_number_chain = false; + } Ok(RuleResult::Consumed) } } @@ -256,6 +273,21 @@ fn is_digit_grade_plus_notation(word: &[char], index: usize) -> bool { mod tests { use super::*; + #[rstest::rstest] + #[case::official_particle("10,000㎡는", true)] + #[case::confusable_counter("3.3㎡당", true)] + #[case::vowel_initial_predicate("3.3㎡이다", false)] + fn square_metre_separates_only_number_confusable_korean( + #[case] input: &str, + #[case] expects_separator: bool, + ) { + let actual = crate::encode_to_unicode(input).unwrap(); + let unit = "⠴⠍⠘⠼⠃"; + let unit_end = actual.find(unit).expect("square-metre cells") + unit.len(); + let follows_with_space = actual[unit_end..].starts_with('⠀'); + assert_eq!(follows_with_space, expects_separator, "input={input}"); + } + #[test] fn is_rule_68_symbol_recognises_each_entry() { for (c, _) in MAPPINGS { @@ -424,6 +456,7 @@ mod tests { has_korean_char: false, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count: &mut skip, state: &mut state, result: &mut out, @@ -451,6 +484,7 @@ mod tests { has_korean_char: true, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count: &mut skip, state: &mut state, result: &mut out, @@ -481,6 +515,7 @@ mod tests { has_korean_char: false, is_all_uppercase: true, ascii_starts_at_beginning: true, + roman_section_continues_from_previous_word: false, skip_count: &mut skip, state: &mut state, result: &mut out, @@ -508,6 +543,7 @@ mod tests { has_korean_char: false, is_all_uppercase: false, ascii_starts_at_beginning: false, + roman_section_continues_from_previous_word: false, skip_count: &mut skip, state: &mut state, result: &mut out, @@ -566,6 +602,7 @@ mod tests { has_korean_char: false, is_all_uppercase: true, ascii_starts_at_beginning: true, + roman_section_continues_from_previous_word: false, skip_count: &mut skip, state: &mut state, result: &mut out, diff --git a/libs/braillify/src/rules/korean/rule_69.rs b/libs/braillify/src/rules/korean/rule_69.rs index 722dde5d..a06432d6 100644 --- a/libs/braillify/src/rules/korean/rule_69.rs +++ b/libs/braillify/src/rules/korean/rule_69.rs @@ -1,8 +1,13 @@ +use std::collections::BTreeMap; +use std::sync::OnceLock; + use crate::char_struct::CharType; use crate::rules::RuleMeta; -use crate::rules::context::RuleContext; +use crate::rules::context::{EncoderState, RuleContext}; +use crate::rules::english_ueb::span::encode_korean_word; use crate::rules::korean::rule_29::{ENGLISH_CONTINUATION, ROMAN_INDICATOR}; use crate::rules::traits::{BrailleRule, Phase, RuleResult}; +use unicode_normalization::UnicodeNormalization; pub static META: RuleMeta = RuleMeta { section: "69", @@ -13,11 +18,6 @@ pub static META: RuleMeta = RuleMeta { }; const SINGLE_MAPPINGS: &[(char, &str)] = &[ - ('㎎', "⠴⠍⠛"), - ('㎗', "⠙⠇⠲"), - ('㎠', "⠉⠍⠘⠼⠃"), - ('㎞', "⠴⠅⠍⠲"), - ('㎒', "⠴⠠⠍⠠⠓⠵⠲"), ('Ω', "⠴⠠⠨⠺⠲"), ('%', "⠴⠏"), ('‰', "⠴⠏⠍"), @@ -34,13 +34,28 @@ const ASCII_UNIT_MAPPINGS: &[(&str, &str)] = &[ ("kg", "⠴⠅⠛⠲"), ("in", "⠴⠊⠝⠲"), ("mm", "⠴⠍⠍⠲"), - ("min", "⠍⠔⠲"), - ("cal", "⠴⠉⠁⠇"), + ("min", "⠴⠍⠔⠲"), + ("cal", "⠴⠉⠁⠇⠲"), ("GB", "⠴⠠⠠⠛⠃⠲"), ("m", "⠴⠍⠲"), ("h", "⠴⠓⠲"), ]; +/// Roman unit symbols printed in the Rule 69 / science-braille unit tables +/// which do not all have a Unicode square-unit presentation form. Their cells +/// are derived through the ordinary Rule 37 letter encoder below, rather than +/// duplicated here as an input-to-output lookup table. +const PDF_ASCII_UNIT_SYMBOLS: &[&str] = &[ + "yard", "sec", "dyn", "kgf", "mmHg", "erg", "HP", "dB", "Hz", "pH", "hPa", +]; + +/// SI prefixes are case-sensitive. This set is used only as the grammar for a +/// complete measured-unit suffix; it never reclassifies a separated Roman word. +const SI_PREFIXES: &[&str] = &[ + "q", "r", "y", "z", "a", "f", "p", "n", "u", "m", "c", "d", "da", "h", "k", "M", "G", "T", "P", + "E", "Z", "Y", "R", "Q", +]; + const PERCENT_ABBREVIATION_MAPPINGS: &[(&str, &str)] = &[("%ile", "⠴⠏⠞"), ("%p", "⠴⠏⠏")]; const SEPARATED_SYMBOLS: &[char] = &['%', '‰', '°', '℃', '℉']; @@ -52,13 +67,167 @@ fn encode_unicode_cells(unicode: &str) -> Vec { .collect() } +/// Unicode contains compatibility presentation forms for Roman unit symbols: +/// CJK square units (`㎏` → `kg`, `㎓` → `GHz`, `㎥` → `m3`) and the letterlike +/// litre sign (`ℓ` → `l`). Rules 68/69 define the transcription from the +/// semantic Roman unit, so recognize these families from their compatibility +/// decomposition instead of assigning input-specific braille cells. Japanese +/// square words and other CJK compatibility characters are rejected by the +/// code-point ranges and component grammar. +pub(crate) fn compatibility_unit_decomposition(c: char) -> Option> { + // Unicode CJK Compatibility contains non-unit square abbreviations too + // (`㏑` ln, `㏒` log, `㏚` PR). Keep the accepted ranges to scientific and + // measurement symbols; the component grammar is an additional guard, not + // the sole evidence that a square abbreviation is a unit. + let is_unit_codepoint = c == 'ℓ' + || matches!( + c as u32, + 0x3371..=0x337a + | 0x3380..=0x33c6 + | 0x33c8..=0x33cc + | 0x33ce..=0x33d0 + | 0x33d3..=0x33d9 + | 0x33db..=0x33df + | 0x33ff + ); + if !is_unit_codepoint || super::rule_68::is_rule_68_symbol(c) { + return None; + } + let parts = c.to_string().nfkc().collect::>(); + (parts.iter().any(|part| part.is_ascii_alphabetic()) + && parts.iter().all(|part| { + part.is_ascii_alphabetic() + || matches!(part, '2' | '3' | '/' | '\u{2044}' | '\u{2215}' | 'μ') + })) + .then_some(parts) +} + +/// Unicode presentation forms whose printed meaning is a Roman measurement +/// unit. Rule 68 owns `㎡` and `㏊`; the remaining forms are decoded by this +/// module from their compatibility decomposition. Keeping this predicate at +/// the shared Roman/number state boundary prevents an intervening glyph from +/// breaking a section that began with Roman text and continued through digits. +pub(crate) fn is_compatibility_unit_presentation(c: char) -> bool { + matches!(c, '㎡' | '㏊') || compatibility_unit_decomposition(c).is_some() +} + +/// Rule 69 delegates only to rule 37's multi-letter groupsigns. This is not +/// ordinary UEB word encoding: whole-word signs and shortforms are disabled, +/// and a lower groupsign cannot consume the whole entry run (`in` is spelled +/// `i`-`n`, while the same `in` may contract inside `min`). +fn encode_rule_69_unit_letters(letters: &[char]) -> Result, String> { + match encode_korean_word(letters, false, false, false, true, false, false, false) { + Some(encoded) => Ok(encoded), + None => Err(format!( + "cannot encode rule 69 Roman unit letters: {}", + letters.iter().collect::() + )), + } +} + +fn encode_compatibility_unit( + parts: &[char], + needs_roman_indicator: bool, + needs_roman_terminator: bool, +) -> Result, String> { + let mut encoded = Vec::new(); + if needs_roman_indicator { + encoded.push(ROMAN_INDICATOR); + } + + let mut index = 0usize; + while index < parts.len() { + match parts[index] { + 'μ' => { + encoded.extend(encode_unicode_cells("⠨⠍")); + index += 1; + } + '2' | '3' => { + encoded.extend(encode_unicode_cells("⠘⠼")); + encoded.push(crate::number::encode_number(parts[index])?); + index += 1; + } + '/' | '\u{2044}' | '\u{2215}' => { + encoded.extend(encode_unicode_cells("⠸⠌")); + index += 1; + } + ch if ch.is_ascii_alphabetic() => { + let end = index + + parts[index..] + .iter() + .take_while(|part| part.is_ascii_alphabetic()) + .count(); + let letters = &parts[index..end]; + let unit = encode_rule_69_unit_letters(letters)?; + encoded.extend(unit); + index = end; + } + unsupported => { + return Err(format!( + "unsupported compatibility unit component: U+{:04X}", + unsupported as u32 + )); + } + } + } + + // Rule 68's superscript closes the compact unit without a Roman terminator + // (`㎡` → `0m^#b`). Otherwise rule 69 terminates the Roman unit unless the + // same Roman unit chain continues through a slash. + if needs_roman_terminator && !matches!(parts.last(), Some('2' | '3')) { + encoded.push(crate::unicode::decode_unicode('⠲')); + } + Ok(encoded) +} + +fn is_roman_unit_component(ch: char) -> bool { + ch.is_ascii_alphabetic() || ch == 'μ' || compatibility_unit_decomposition(ch).is_some() +} + +fn roman_unit_chain_continues_before(ctx: &RuleContext) -> bool { + ctx.index >= 2 + && ctx.word_chars.get(ctx.index - 1) == Some(&'/') + && ctx + .word_chars + .get(ctx.index - 2) + .is_some_and(|previous| is_roman_unit_component(*previous)) +} + +fn roman_unit_chain_continues_after(ctx: &RuleContext) -> bool { + ctx.word_chars.get(ctx.index + 1) == Some(&'/') + && ctx + .word_chars + .get(ctx.index + 2) + .is_some_and(|next| is_roman_unit_component(*next)) +} + pub fn is_rule_69_symbol(c: char) -> bool { - SINGLE_MAPPINGS.iter().any(|(candidate, _)| *candidate == c) || c == 'μ' + SINGLE_MAPPINGS.iter().any(|(candidate, _)| *candidate == c) + || c == 'μ' + || compatibility_unit_decomposition(c).is_some() } fn is_numeric_or_unit_context(ctx: &RuleContext) -> bool { - ctx.prev_char() - .is_some_and(|prev| prev.is_ascii_digit() || matches!(prev, '/' | 'μ')) + let mut numeric_start = ctx.index; + while numeric_start > 0 + && (ctx.word_chars[numeric_start - 1].is_ascii_digit() + || matches!(ctx.word_chars[numeric_start - 1], ',' | '.')) + { + numeric_start -= 1; + } + let compact_numeric_prefix = numeric_start < ctx.index + && ctx.word_chars[numeric_start..ctx.index] + .iter() + .any(char::is_ascii_digit) + && numeric_start + .checked_sub(1) + .and_then(|index| ctx.word_chars.get(index)) + .is_none_or(|previous| !previous.is_ascii_alphabetic()); + + compact_numeric_prefix + || ctx + .prev_char() + .is_some_and(|prev| matches!(prev, '/' | 'μ')) || ctx.prev_word.chars().next().is_some() && ctx .prev_word @@ -108,15 +277,213 @@ fn chars_start_with_ascii(tail: &[char], s: &str) -> bool { s.bytes().zip(tail.iter()).all(|(b, c)| (b as char) == *c) } +fn encode_ascii_unit_letters(spelling: &[char]) -> Option> { + encode_compatibility_unit(spelling, true, true).ok() +} + +/// `Wh` and `Ah` are products of the Rule-69 Roman unit symbols watt/ampere +/// and hour. Accept every case-sensitive SI-prefixed form (`mAh`, `kWh`, +/// `GWh`, ...), rather than enumerating values observed in a corpus. +fn is_si_prefixed_electrical_hour_unit(spelling: &str) -> bool { + let Some(head) = spelling.strip_suffix('h') else { + return false; + }; + let Some(base) = head.chars().last() else { + return false; + }; + if !matches!(base, 'A' | 'W') { + return false; + } + let prefix = &head[..head.len() - base.len_utf8()]; + prefix.is_empty() || SI_PREFIXES.contains(&prefix) +} + +/// The litre symbol may be printed as either `l` or `L`; an SI prefix retains +/// its case (`dL`, `mL`, `kL`, ...). Rule 69's printed `㎗` example owns the +/// decilitre semantics, while this grammar preserves the case of an ASCII +/// spelling instead of copying the compatibility character's lowercase NFKC. +fn is_si_prefixed_litre_unit(spelling: &str) -> bool { + let Some(base) = spelling.chars().last() else { + return false; + }; + if !matches!(base, 'l' | 'L') { + return false; + } + let prefix = &spelling[..spelling.len() - base.len_utf8()]; + prefix.is_empty() || SI_PREFIXES.contains(&prefix) +} + +/// Rule 69 prints `GB` as its storage-unit example. Treat the same `B` unit +/// with another case-sensitive SI prefix as one unit symbol, rather than +/// enumerating each storage capacity. A bare `B` remains ambiguous with a +/// Roman letter and therefore is not selected by this automatic prose route. +fn is_si_prefixed_byte_unit(spelling: &str) -> bool { + let Some(prefix) = spelling.strip_suffix('B') else { + return false; + }; + !prefix.is_empty() && SI_PREFIXES.contains(&prefix) +} + +fn standard_ascii_unit_candidate(tail: &[char]) -> Option<(Vec, usize)> { + let consumed = tail + .iter() + .take_while(|ch| ch.is_ascii_alphabetic()) + .count(); + if consumed == 0 { + return None; + } + let spelling = tail[..consumed].iter().collect::(); + if !PDF_ASCII_UNIT_SYMBOLS.contains(&spelling.as_str()) + && !is_si_prefixed_electrical_hour_unit(&spelling) + && !is_si_prefixed_litre_unit(&spelling) + && !is_si_prefixed_byte_unit(&spelling) + { + return None; + } + Some((encode_ascii_unit_letters(&tail[..consumed])?, consumed)) +} + +/// ASCII spellings that are canonically exposed by the same Unicode +/// compatibility-unit family already accepted above. This derives the unit +/// lexicon from semantic unit code points instead of maintaining a second +/// corpus-shaped list (`㎞` -> `km`, `㎎` -> `mg`, `㎾` -> `kW`, ...). +fn compatibility_ascii_unit_candidate(glyph: char) -> Option<(String, Vec)> { + let parts = glyph.to_string().nfkc().collect::>(); + if !parts.iter().all(char::is_ascii_alphabetic) { + return None; + } + let encoded = if compatibility_unit_decomposition(glyph).is_some() { + encode_compatibility_unit(&parts, true, true).ok()? + } else { + super::rule_68::encode_rule_68_symbol(glyph)? + }; + Some((parts.into_iter().collect(), encoded)) +} + +fn compatibility_ascii_unit_owners() -> BTreeMap)>> { + let mut by_spelling = BTreeMap::)>>::new(); + for glyph in (0x3300..=0x33ff).filter_map(char::from_u32) { + if let Some((spelling, encoded)) = compatibility_ascii_unit_candidate(glyph) { + by_spelling + .entry(spelling) + .or_default() + .push((glyph, encoded)); + } + } + by_spelling +} + +fn retain_unambiguous_ascii_unit_spellings( + owners_by_spelling: BTreeMap)>>, +) -> Vec<(String, Vec)> { + let mut spellings = owners_by_spelling + .into_iter() + .filter_map(|(spelling, owners)| { + let first = &owners.first()?.1; + owners + .iter() + .all(|(_, encoded)| encoded == first) + .then(|| (spelling, first.clone())) + }) + .collect::>(); + spellings.sort_by(|left, right| { + right + .0 + .len() + .cmp(&left.0.len()) + .then_with(|| left.0.cmp(&right.0)) + }); + spellings +} + +fn compatibility_ascii_unit_spellings() -> &'static [(String, Vec)] { + static SPELLINGS: OnceLock)>> = OnceLock::new(); + SPELLINGS + .get_or_init(|| retain_unambiguous_ascii_unit_spellings(compatibility_ascii_unit_owners())) +} + pub(crate) fn encode_ascii_unit(word: &[char], index: usize) -> Option<(Vec, usize)> { let tail = &word[index..]; - for (unit, unicode) in ASCII_UNIT_MAPPINGS { - if !chars_start_with_ascii(tail, unit) { - continue; + let explicit = ASCII_UNIT_MAPPINGS + .iter() + .filter(|(unit, _)| chars_start_with_ascii(tail, unit)) + .max_by_key(|(unit, _)| unit.len()) + .map(|(unit, unicode)| (encode_unicode_cells(unicode), unit.len())); + let standard = standard_ascii_unit_candidate(tail); + + match (explicit, standard) { + (Some(explicit), Some(standard)) if explicit.1 == standard.1 => { + (explicit.0 == standard.0).then_some(explicit) } - return Some((encode_unicode_cells(unicode), unit.len())); + (Some(explicit), Some(standard)) if explicit.1 < standard.1 => Some(standard), + (Some(explicit), _) => Some(explicit), + (None, standard) => standard, } - None +} + +/// Numeric-compact Rule 69 path. Compatibility-derived spellings are limited +/// to this measured boundary so an unrelated English word after a separated +/// number cannot become a unit merely because it starts with a unit spelling. +fn encode_numeric_ascii_unit(word: &[char], index: usize) -> Option<(Vec, usize)> { + let tail = &word[index..]; + let explicit = encode_ascii_unit(word, index); + let derived = compatibility_ascii_unit_spellings() + .iter() + .filter(|(unit, _)| chars_start_with_ascii(tail, unit)) + .max_by_key(|(unit, _)| unit.len()); + + if let Some((encoded, consumed)) = explicit { + match derived { + Some((candidate, derived_encoded)) if consumed == candidate.len() => { + return (encoded.as_slice() == derived_encoded.as_slice()) + .then_some((encoded, consumed)); + } + Some((candidate, _)) if consumed < candidate.len() => {} + _ => return Some((encoded, consumed)), + } + } + + let (unit, encoded) = derived?; + Some((encoded.clone(), unit.len())) +} + +fn encode_complete_numeric_ascii_unit(word: &[char], index: usize) -> Option<(Vec, usize)> { + let (encoded, consumed) = encode_numeric_ascii_unit(word, index)?; + if word + .get(index + consumed) + .is_some_and(|ch| ch.is_ascii_alphabetic()) + { + return None; + } + Some((encoded, consumed)) +} + +/// Length of a complete Rule-69 ASCII unit beginning at `index`. +/// +/// Token-level capitalization uses this predicate to leave a separated +/// uppercase unit (`5 GB`, `350 PB`) to Rule 69. Otherwise it would emit a +/// Roman/capital prefix before the character rule emits the unit's own prefix. +pub(crate) fn complete_ascii_unit_len(word: &[char], index: usize) -> Option { + encode_complete_numeric_ascii_unit(word, index).map(|(_, consumed)| consumed) +} + +pub(crate) fn is_ascii_unit_chain_slash(word: &[char], index: usize) -> bool { + if word.get(index) != Some(&'/') || index == 0 { + return false; + } + + let left_start = (0..index) + .rev() + .take_while(|position| word[*position].is_ascii_alphabetic()) + .last() + .unwrap_or(index); + let left_len = index.saturating_sub(left_start); + let left_is_complete = left_len > 0 + && encode_complete_numeric_ascii_unit(word, left_start) + .is_some_and(|(_, consumed)| consumed == left_len); + let right_is_complete = encode_complete_numeric_ascii_unit(word, index + 1).is_some(); + + left_is_complete && right_is_complete } fn encode_percent_abbreviation(word: &[char], index: usize) -> Option<(Vec, usize)> { @@ -147,10 +514,81 @@ pub(crate) fn parse_numeric_ascii_unit_prefix(word: &[char]) -> Option<(String, } let numeric = word[..numeric_len].iter().collect::(); - let (unit, consumed) = encode_ascii_unit(word, numeric_len)?; + let (unit, consumed) = encode_complete_numeric_ascii_unit(word, numeric_len)?; Some((numeric, unit, numeric_len + consumed)) } +fn numeric_component_len(word: &[char], start: usize) -> usize { + let len = word[start..] + .iter() + .take_while(|ch| ch.is_ascii_digit() || matches!(ch, ',' | '.')) + .count(); + if word[start..start + len].iter().any(char::is_ascii_digit) { + len + } else { + 0 + } +} + +/// Parse a complete Rule-69 measurement expression which starts with a +/// number, including a range (`3.5~8.5m`), a Roman-unit quotient +/// (`240mg/dL`), or a Rule-50 middle-dot list whose every member owns a +/// complete unit (`256GB·512GB·1TB`). This is a routing predicate only: the +/// ordinary Korean character rules still emit every number, range sign, +/// middle dot, slash and unit. +/// +/// Requiring every slash component and the final expression to contain a +/// recognized Rule-69 unit keeps general fractions, dates, model numbers and +/// arbitrary ASCII suffixes outside this route. +pub(crate) fn parse_numeric_ascii_unit_expression(word: &[char]) -> Option { + let mut cursor = 0usize; + let mut saw_unit = false; + let mut current_component_requires_unit = false; + + loop { + let numeric_len = numeric_component_len(word, cursor); + if numeric_len == 0 { + return None; + } + cursor += numeric_len; + + let mut component_has_unit = false; + if let Some((_, unit_len)) = encode_complete_numeric_ascii_unit(word, cursor) { + cursor += unit_len; + saw_unit = true; + component_has_unit = true; + } + + while component_has_unit && word.get(cursor) == Some(&'/') { + let unit_start = cursor + 1; + let Some((_, unit_len)) = encode_complete_numeric_ascii_unit(word, unit_start) else { + break; + }; + cursor = unit_start + unit_len; + saw_unit = true; + component_has_unit = true; + } + + if current_component_requires_unit && !component_has_unit { + return None; + } + + if word.get(cursor).is_some_and(|ch| matches!(ch, '~' | '∼')) { + cursor += 1; + current_component_requires_unit = false; + continue; + } + if word.get(cursor) == Some(&'·') && component_has_unit { + cursor += 1; + current_component_requires_unit = true; + continue; + } + break; + } + + saw_unit.then_some(cursor) +} + fn trim_recent_english_indicator(result: &mut Vec) { if result .last() @@ -160,6 +598,111 @@ fn trim_recent_english_indicator(result: &mut Vec) { } } +/// Rules 33/34 override rule 69's ordinary trailing Roman terminator when a +/// listed Korean punctuation mark or an enclosing mark closes the Roman run; +/// rule 35 likewise omits it when an attached digit continues the Roman/number +/// chain. +/// Unit encoders include their ordinary terminator so standalone/end/Korean +/// boundaries stay unchanged; this helper applies only at the actual following +/// input boundary. +fn omit_roman_terminator_before_boundary( + encoded: &mut Vec, + word: &[char], + boundary_index: usize, +) { + let skips_for_punctuation = word + .get(boundary_index) + .is_some_and(|symbol| crate::english_logic::should_skip_terminator_for_symbol(*symbol)); + let continues_through_slash = word.get(boundary_index) == Some(&'/') + && word + .get(boundary_index + 1) + .is_some_and(|next| is_roman_unit_component(*next)); + let continues_into_number = word + .get(boundary_index) + .is_some_and(|next| next.is_ascii_digit()); + if (skips_for_punctuation || continues_through_slash || continues_into_number) + && encoded.last() == Some(&crate::unicode::decode_unicode('⠲')) + { + encoded.pop(); + } +} + +/// A comma after a Roman unit remains inside the same Roman section when the +/// next print item begins with another Roman/numeric item. Rule 33 switches +/// to the Korean comma only at an actual Roman-to-Korean boundary; a later +/// Korean particle does not retroactively change the comma in a measurement +/// list such as `173cm, 68kg의`. +fn roman_unit_comma_continues_section(ctx: &RuleContext, boundary_index: usize) -> bool { + // Rule 68's superscript cell closes a compact square/cubic unit without a + // Roman terminator. The following comma is therefore Korean punctuation, + // and a later unit starts a fresh Roman section. + let closes_with_superscript = ctx.current_char() == '㎡' + || compatibility_unit_decomposition(ctx.current_char()) + .is_some_and(|parts| matches!(parts.last(), Some('2' | '3'))); + if closes_with_superscript { + return false; + } + + ctx.word_chars.get(boundary_index) == Some(&',') + && crate::english_logic::should_render_symbol_as_english( + ctx.state.english_indicator, + true, + ctx.state.doc_summary.is_english_majority, + &ctx.state.parenthesis_stack, + ',', + ctx.word_chars, + boundary_index, + ctx.remaining_words, + ) +} + +/// Rules 29 and 35 keep a separated following Roman/number word in the same +/// section. This is the cross-word counterpart of +/// [`omit_roman_terminator_before_boundary`], whose look-ahead is intentionally +/// limited to the current print word. +fn roman_unit_continues_into_next_word(ctx: &RuleContext, boundary_index: usize) -> bool { + boundary_index == ctx.word_chars.len() + && ctx + .remaining_words + .first() + .and_then(|word| word.chars().next()) + .is_some_and(|ch| ch.is_ascii_alphanumeric()) +} + +fn omit_trailing_roman_terminator(encoded: &mut Vec) { + if encoded.last() == Some(&crate::unicode::decode_unicode('⠲')) { + encoded.pop(); + } +} + +/// Rule 35 keeps a Roman unit directly following a number in the already-open +/// Roman section. The number temporarily places the emitter in +/// `roman_number_chain`; in that state the unit's self-contained Rule-69 entry +/// marker would be a duplicate. +fn omit_unit_entry_in_open_roman_number_chain(encoded: &mut Vec, state: &EncoderState) { + if state.roman_number_chain && encoded.first() == Some(&ROMAN_INDICATOR) { + encoded.remove(0); + } +} + +/// Apply the common Rule 29/33/34/35 boundary behavior to a self-contained +/// Roman unit encoding and report whether the section continues into a later +/// print word. Rule 68 reuses this for its two Roman unit presentations. +pub(crate) fn adjust_roman_unit_boundary( + ctx: &RuleContext, + boundary_index: usize, + encoded: &mut Vec, +) -> bool { + omit_unit_entry_in_open_roman_number_chain(encoded, ctx.state); + omit_roman_terminator_before_boundary(encoded, ctx.word_chars, boundary_index); + let comma_continues = roman_unit_comma_continues_section(ctx, boundary_index); + let separated_continues = roman_unit_continues_into_next_word(ctx, boundary_index); + if separated_continues { + omit_trailing_roman_terminator(encoded); + } + comma_continues || separated_continues +} + fn should_insert_separator_after_symbol(symbol: char, next: Option) -> bool { SEPARATED_SYMBOLS.contains(&symbol) && next.is_some_and(crate::utils::is_korean_char) } @@ -186,19 +729,24 @@ impl BrailleRule for Rule69 { || matches!(ctx.char_type, CharType::English(_) if (is_numeric_or_unit_context(ctx) || (ctx.index == 0 && word_looks_like_unit_chain(ctx.word_chars))) - && encode_ascii_unit(ctx.word_chars, ctx.index).is_some()) + && encode_complete_numeric_ascii_unit(ctx.word_chars, ctx.index).is_some()) } fn apply(&self, ctx: &mut RuleContext) -> Result { if matches!(ctx.char_type, CharType::Number(_)) && ctx.index == 0 - && let Some((numeric, unit, consumed)) = parse_numeric_ascii_unit_prefix(ctx.word_chars) + && let Some((numeric, mut unit, consumed)) = + parse_numeric_ascii_unit_prefix(ctx.word_chars) { + let continues = adjust_roman_unit_boundary(ctx, consumed, &mut unit); let mut encoded = crate::encode(&numeric)?; encoded.extend(unit); ctx.emit_slice(&encoded); - ctx.state.is_english = false; + ctx.state.is_english = continues; ctx.state.needs_english_continuation = false; + if continues { + ctx.state.roman_number_chain = false; + } *ctx.skip_count = consumed.saturating_sub(1); return Ok(RuleResult::Consumed); } @@ -206,12 +754,20 @@ impl BrailleRule for Rule69 { if matches!(ctx.char_type, CharType::English(_)) && (is_numeric_or_unit_context(ctx) || (ctx.index == 0 && word_looks_like_unit_chain(ctx.word_chars))) - && let Some((encoded, consumed)) = encode_ascii_unit(ctx.word_chars, ctx.index) + && let Some((mut encoded, consumed)) = + encode_complete_numeric_ascii_unit(ctx.word_chars, ctx.index) { + let continues = adjust_roman_unit_boundary(ctx, ctx.index + consumed, &mut encoded); + if roman_unit_chain_continues_before(ctx) && encoded.first() == Some(&ROMAN_INDICATOR) { + encoded.remove(0); + } trim_recent_english_indicator(ctx.result); ctx.emit_slice(&encoded); - ctx.state.is_english = false; + ctx.state.is_english = continues; ctx.state.needs_english_continuation = false; + if continues { + ctx.state.roman_number_chain = false; + } *ctx.skip_count = consumed.saturating_sub(1); return Ok(RuleResult::Consumed); } @@ -249,6 +805,12 @@ impl BrailleRule for Rule69 { encoded.extend(encode_unicode_cells("⠍")); } + omit_roman_terminator_before_boundary( + &mut encoded, + ctx.word_chars, + ctx.index + consumed, + ); + ctx.emit_slice(&encoded); ctx.state.is_english = false; ctx.state.needs_english_continuation = false; @@ -256,6 +818,33 @@ impl BrailleRule for Rule69 { return Ok(RuleResult::Consumed); } + if let Some(parts) = compatibility_unit_decomposition(ctx.current_char()) { + let continues_from_previous = + roman_unit_chain_continues_before(ctx) || ctx.state.roman_number_chain; + let continues_within_word = roman_unit_chain_continues_after(ctx); + let separated_continues = roman_unit_continues_into_next_word(ctx, ctx.index + 1); + let continues_after = continues_within_word || separated_continues; + let mut encoded = + encode_compatibility_unit(&parts, !continues_from_previous, !continues_after)?; + let continues = adjust_roman_unit_boundary(ctx, ctx.index + 1, &mut encoded); + ctx.emit_slice(&encoded); + if matches!(parts.last(), Some('2' | '3')) + && ctx + .next_char() + .is_some_and(super::rule_44::is_number_confusable_korean_char) + { + ctx.emit(0); + } + // Same-word compatibility-unit chains are emitted component by + // component and must retain their existing closed state. Only a + // continuation across a print space needs to survive into the next + // Word token. + ctx.state.is_english = continues; + ctx.state.needs_english_continuation = false; + ctx.state.roman_number_chain = false; + return Ok(RuleResult::Consumed); + } + // `matches()` guard `is_rule_69_symbol(c)` is a `SINGLE_MAPPINGS` lookup, // so reaching here without the prior μ/ASCII-unit/`%`-shortcut paths // means the char is guaranteed to be in `SINGLE_MAPPINGS`. @@ -263,7 +852,8 @@ impl BrailleRule for Rule69 { .iter() .find(|(candidate, _)| *candidate == ctx.current_char()) .expect("matches() guarantees the char is in SINGLE_MAPPINGS"); - let encoded = encode_unicode_cells(unicode); + let mut encoded = encode_unicode_cells(unicode); + omit_roman_terminator_before_boundary(&mut encoded, ctx.word_chars, ctx.index + 1); ctx.emit_slice(&encoded); if should_insert_separator_after_symbol(ctx.current_char(), ctx.next_char()) { ctx.emit(0); @@ -275,8 +865,14 @@ impl BrailleRule for Rule69 { #[cfg(test)] mod tests { use super::{ - Rule69, encode_ascii_unit, encode_percent_abbreviation, parse_numeric_ascii_unit_prefix, - word_looks_like_unit_chain, + Rule69, compatibility_ascii_unit_owners, compatibility_unit_decomposition, + encode_ascii_unit, encode_compatibility_unit, encode_complete_numeric_ascii_unit, + encode_numeric_ascii_unit, encode_percent_abbreviation, encode_rule_69_unit_letters, + encode_unicode_cells, is_ascii_unit_chain_slash, is_si_prefixed_byte_unit, + is_si_prefixed_electrical_hour_unit, is_si_prefixed_litre_unit, + omit_roman_terminator_before_boundary, omit_trailing_roman_terminator, + parse_numeric_ascii_unit_expression, parse_numeric_ascii_unit_prefix, + retain_unambiguous_ascii_unit_spellings, word_looks_like_unit_chain, }; #[rstest::rstest] @@ -290,6 +886,580 @@ mod tests { assert_eq!(word_looks_like_unit_chain(&chars), expected); } + #[rstest::rstest] + #[case::kilogram('㎏', "kg")] + #[case::gigahertz('㎓', "GHz")] + #[case::cubic_metre('㎥', "m3")] + #[case::metres_per_second('㎧', "m∕s")] + #[case::milliwatt('㎽', "mW")] + #[case::kilowatt('㎾', "kW")] + #[case::sievert('㏜', "Sv")] + #[case::litre('ℓ', "l")] + fn decomposes_compatibility_unit_symbols(#[case] input: char, #[case] expected: &str) { + assert_eq!( + compatibility_unit_decomposition(input), + Some(expected.chars().collect()) + ); + } + + /// Unicode CJK Compatibility names distinguish the accepted SQUARE IU + /// (U+337A) from non-unit square abbreviations LN, LOG, and PR. In + /// particular, U+33DA is SQUARE PR, not SQUARE IU. + #[rstest::rstest] + #[case::international_unit('㍺', Some("IU"))] + #[case::natural_logarithm('㏑', None)] + #[case::logarithm('㏒', None)] + #[case::public_relations('㏚', None)] + fn accepts_only_unit_semantics(#[case] input: char, #[case] expected: Option<&str>) { + assert_eq!( + compatibility_unit_decomposition(input), + expected.map(|text| text.chars().collect()) + ); + } + + const ACCEPTED_GLYPHS: &str = "㍱㍲㍳㍴㍵㍶㍷㍸㍹㍺㎀㎁㎂㎃㎄㎅㎆㎇㎈㎉㎊㎋㎌㎍㎎㎏㎐㎑㎒㎓㎔㎕㎖㎗㎘㎙㎚㎛㎜㎝㎞㎟㎠㎢㎣㎤㎥㎦㎧㎨㎩㎪㎫㎬㎭㎮㎯㎰㎱㎲㎳㎴㎵㎶㎷㎸㎹㎺㎻㎼㎽㎾㎿㏃㏄㏅㏆㏈㏉㏋㏌㏎㏏㏐㏓㏔㏕㏖㏗㏙㏛㏜㏝㏞㏟㏿"; + + #[test] + fn accepted_compatibility_unit_set_is_stable() { + let actual = (0x3300..=0x33ff) + .filter_map(char::from_u32) + .filter(|ch| compatibility_unit_decomposition(*ch).is_some()) + .collect::(); + + assert_eq!(actual, ACCEPTED_GLYPHS); + } + + #[test] + fn every_accepted_compatibility_unit_encodes_without_panicking() { + // Generated property check: the set identity is asserted separately, + // while this loop only proves that every accepted decomposition and + // each of its ASCII letter runs reaches the fallible Rule 69 encoder. + for glyph in ACCEPTED_GLYPHS.chars() { + let parts = compatibility_unit_decomposition(glyph).unwrap(); + let mut index = 0usize; + while index < parts.len() { + if !parts[index].is_ascii_alphabetic() { + index += 1; + continue; + } + let end = index + + parts[index..] + .iter() + .take_while(|part| part.is_ascii_alphabetic()) + .count(); + encode_rule_69_unit_letters(&parts[index..end]).unwrap(); + index = end; + } + encode_compatibility_unit(&parts, true, true).unwrap(); + } + } + + #[test] + fn compatibility_unit_encoder_rejects_components_outside_its_grammar() { + let error = encode_compatibility_unit(&['?'], true, true).unwrap_err(); + + assert_eq!(error, "unsupported compatibility unit component: U+003F"); + } + + /// The compatibility-unit grammar sends only ASCII-letter runs here. + /// Rejecting a numeric component directly keeps that defensive contract + /// observable without weakening the accepted Rule 68/69 glyph set. + #[rstest::rstest] + #[case::single_letter("m", true)] + #[case::multi_letter_unit("min", true)] + #[case::non_letter_component("1", false)] + fn unit_letter_encoder_accepts_only_roman_letter_runs( + #[case] input: &str, + #[case] expected_ok: bool, + ) { + let letters = input.chars().collect::>(); + let result = encode_rule_69_unit_letters(&letters); + + assert_eq!(result.is_ok(), expected_ok); + if !expected_ok { + assert_eq!( + result.unwrap_err(), + "cannot encode rule 69 Roman unit letters: 1" + ); + } + } + + #[test] + fn every_rule_68_or_69_ascii_derivation_matches_every_owner_glyph() { + for (spelling, owners) in compatibility_ascii_unit_owners() { + let first = &owners[0].1; + for (glyph, owner_encoding) in &owners { + assert_eq!( + owner_encoding, first, + "conflicting owner cells for NFKC spelling {spelling:?}: U+{:04X}", + *glyph as u32 + ); + + let chars = spelling.chars().collect::>(); + let (derived, consumed) = encode_numeric_ascii_unit(&chars, 0) + .unwrap_or_else(|| { + panic!( + "unambiguous ASCII compatibility-unit spelling {spelling:?} from U+{:04X} must be recognized", + *glyph as u32 + ) + }); + assert_eq!(consumed, spelling.len(), "partial match for {spelling}"); + assert_eq!( + &derived, owner_encoding, + "derived cells differ from owner U+{:04X} for {spelling}", + *glyph as u32 + ); + + let ascii_input = format!("값은 1{spelling}이다"); + let glyph_input = format!("값은 1{glyph}이다"); + assert_eq!( + crate::encode_to_unicode(&ascii_input).unwrap(), + crate::encode_to_unicode(&glyph_input).unwrap(), + "full encoder differs for {spelling} and owner U+{:04X}", + *glyph as u32 + ); + } + } + } + + #[test] + fn conflicting_nfkc_owner_cells_are_excluded_instead_of_first_wins() { + let owners = std::collections::BTreeMap::from([ + ( + "safe".to_string(), + vec![('A', vec![1, 2]), ('B', vec![1, 2])], + ), + ("conflict".to_string(), vec![('C', vec![3]), ('D', vec![4])]), + ]); + + let resolved = retain_unambiguous_ascii_unit_spellings(owners); + + assert!(resolved.iter().any(|(spelling, _)| spelling == "safe")); + assert!(resolved.iter().all(|(spelling, _)| spelling != "conflict")); + } + + #[rstest::rstest] + #[case::kilometre("80km", "80㎞")] + #[case::pdf_milligram("160mg", "160㎎")] + #[case::numeric_invariance_milligram("240mg", "240㎎")] + #[case::kilowatt("30kW", "30㎾")] + #[case::megahertz("96.7MHz", "96.7㎒")] + #[case::hectare("15.2ha", "15.2㏊")] + fn compact_ascii_units_match_supported_compatibility_forms( + #[case] ascii: &str, + #[case] compatibility: &str, + ) { + let ascii = format!("값은 {ascii}이다"); + let compatibility = format!("값은 {compatibility}이다"); + assert_eq!( + crate::encode_to_unicode(&ascii).unwrap(), + crate::encode_to_unicode(&compatibility).unwrap() + ); + } + + #[rstest::rstest] + #[case::letter_after_digit("3m", "⠼⠉⠍")] + #[case::letter_after_decimal_punctuation("4.m", "⠼⠙⠲⠍")] + fn pure_english_ambiguous_suffixes_remain_on_ueb_path( + #[case] input: &str, + #[case] expected: &str, + ) { + assert_eq!(crate::encode_to_unicode(input).unwrap(), expected); + } + + #[rstest::rstest] + #[case::longest_derived("30mW", 4)] + #[case::hectare_derived("15.2ha", 6)] + #[case::compound_kilowatt_hour("30kWh", 5)] + #[case::compound_milliampere_hour("900mAh", 6)] + #[case::pdf_millimetres_of_mercury("140mmHg", 7)] + #[case::reject_partial_suffix("30kWhours", 0)] + fn parses_only_complete_compatibility_derived_units( + #[case] input: &str, + #[case] expected_consumed: usize, + ) { + let chars = input.chars().collect::>(); + assert_eq!( + parse_numeric_ascii_unit_prefix(&chars).map_or(0, |(_, _, consumed)| consumed), + expected_consumed + ); + } + + #[rstest::rstest] + #[case::simple_unit("180cm", 5)] + #[case::range_with_final_unit("3.5~8.5m", 8)] + #[case::unicode_range_with_final_unit("3∼5kg", 5)] + #[case::compound_unit_quotient("240mg/dL", 8)] + #[case::middle_dot_unit_list("256GB·512GB·1TB", 15)] + #[case::middle_dot_mixed_units("3kg·4cm", 7)] + #[case::range_without_unit("3.5~8.5", 0)] + #[case::fraction_with_units_as_operands("3m/4m", 2)] + #[case::middle_dot_missing_left_unit("3·4kg", 0)] + #[case::middle_dot_missing_right_unit("3kg·4", 0)] + #[case::middle_dot_numeric_list("54·55·56", 0)] + #[case::unknown_ascii_suffix("3.5~8.5models", 0)] + fn recognizes_only_complete_numeric_unit_expressions( + #[case] input: &str, + #[case] expected_consumed: usize, + ) { + let chars = input.chars().collect::>(); + assert_eq!( + parse_numeric_ascii_unit_expression(&chars).unwrap_or(0), + expected_consumed + ); + } + + #[rstest::rstest] + #[case::ascii_range("범위는 3.5~8.5m이다", "⠼⠉⠲⠑⠈⠔⠼⠓⠲⠑⠴⠍⠲")] + #[case::unicode_range("범위는 3∼5kg이다", "⠼⠉⠈⠔⠼⠑⠴⠅⠛⠲")] + fn numeric_unit_ranges_stay_on_korean_number_and_unit_rules( + #[case] input: &str, + #[case] expected_segment: &str, + ) { + let actual = crate::encode_to_unicode(input).unwrap(); + assert!( + actual.contains(expected_segment), + "missing standard range {expected_segment:?} in {actual:?}" + ); + } + + #[rstest::rstest] + #[case::storage_capacities("256GB·512GB·1TB")] + #[case::mixed_measurements("3kg·4cm")] + fn separated_middle_dot_unit_lists_stay_on_korean_rules(#[case] expression: &str) { + let separated = crate::encode_to_unicode(&format!("가는 {expression} 나다")).unwrap(); + let attached = crate::encode_to_unicode(&format!("가는 {expression}이다")).unwrap(); + let segment = attached + .strip_prefix(&crate::encode_to_unicode("가는 ").unwrap()) + .and_then(|tail| tail.strip_suffix(&crate::encode_to_unicode("이다").unwrap())) + .expect("attached control must contain the measurement segment"); + + assert!( + separated.contains(segment), + "measurement segment {segment:?} was rerouted in {separated:?}" + ); + } + + #[rstest::rstest] + #[case::pdf_gigabyte("가는 5 GB 나다", "⠫⠉⠵⠀⠼⠑⠀⠴⠠⠠⠛⠃⠲⠀⠉⠊")] + #[case::petabyte("가는 5 PB 나다", "⠫⠉⠵⠀⠼⠑⠀⠴⠠⠠⠏⠃⠲⠀⠉⠊")] + #[case::terabyte("가는 5 TB 나다", "⠫⠉⠵⠀⠼⠑⠀⠴⠠⠠⠞⠃⠲⠀⠉⠊")] + fn separated_uppercase_units_emit_one_roman_capital_prefix( + #[case] input: &str, + #[case] expected: &str, + ) { + assert_eq!(crate::encode_to_unicode(input).as_deref(), Ok(expected)); + } + + /// Rules 29 and 35 omit a unit's Roman terminator when another separated + /// Roman/number item follows. Exercise both ASCII and compatibility-unit + /// Rule-69 paths and the Roman-word continuation path. + #[rstest::rstest] + #[case::ascii_unit_before_number("가는 12km 3구간", "⠫⠉⠵⠀⠼⠁⠃⠴⠅⠍⠀⠼⠉⠈⠍⠫⠒")] + #[case::compatibility_unit_before_number("가는 8.4㎞ 2구간", "⠫⠉⠵⠀⠼⠓⠲⠙⠴⠅⠍⠀⠼⠃⠈⠍⠫⠒")] + #[case::ascii_unit_before_roman("가는 1TB SSD 나다", "⠫⠉⠵⠀⠼⠁⠴⠠⠠⠞⠃⠀⠠⠠⠎⠎⠙⠲⠀⠉⠊")] + fn separated_rule_69_unit_continues_roman_number_section( + #[case] input: &str, + #[case] expected: &str, + ) { + assert_eq!(crate::encode_to_unicode(input).as_deref(), Ok(expected)); + } + + #[test] + fn ascii_unit_quotient_preserves_printed_case_in_one_roman_section() { + let actual = crate::encode_to_unicode("수치는 240mg/dL이다").unwrap(); + assert!( + actual.contains("⠼⠃⠙⠚⠴⠍⠛⠸⠌⠙⠠⠇⠲"), + "unexpected Rule-69 quotient: {actual}" + ); + } + + #[rstest::rstest] + #[case::decilitre("dL", true)] + #[case::millilitre_lower_l("ml", true)] + #[case::millilitre_upper_l("mL", true)] + #[case::litre("L", true)] + #[case::word_ending_l("model", false)] + #[case::invalid_prefix("xL", false)] + #[case::empty_spelling("", false)] + fn recognizes_case_preserving_si_litre_symbols(#[case] spelling: &str, #[case] expected: bool) { + assert_eq!(is_si_prefixed_litre_unit(spelling), expected); + } + + #[rstest::rstest] + #[case::pdf_gigabyte("GB", true)] + #[case::terabyte("TB", true)] + #[case::megabyte("MB", true)] + #[case::bare_letter("B", false)] + #[case::wrong_base_case("Gb", false)] + #[case::unknown_prefix("xB", false)] + fn recognizes_si_prefixed_byte_units(#[case] spelling: &str, #[case] expected: bool) { + assert_eq!(is_si_prefixed_byte_unit(spelling), expected); + } + + #[rstest::rstest] + #[case::milligram_per_decilitre("240mg/dL", 5, true)] + #[case::calorie_per_minute("cal/min", 3, true)] + #[case::fraction_operands("3m/4m", 2, false)] + #[case::arbitrary_letters("F/N", 1, false)] + fn recognizes_only_slashes_between_complete_unit_components( + #[case] input: &str, + #[case] slash_index: usize, + #[case] expected: bool, + ) { + let chars = input.chars().collect::>(); + assert_eq!( + is_ascii_unit_chain_slash(&chars, slash_index), + expected, + "input={input}" + ); + } + + #[rstest::rstest] + #[case::watt_hour("Wh", true)] + #[case::gigawatt_hour("GWh", true)] + #[case::milliampere_hour("mAh", true)] + #[case::decaampere_hour("daAh", true)] + #[case::missing_hour("GW", false)] + #[case::invalid_base_before_hour("mh", false)] + #[case::unknown_prefix("xWh", false)] + #[case::wrong_case("gWh", false)] + fn recognizes_si_prefixed_electrical_hour_units( + #[case] spelling: &str, + #[case] expected: bool, + ) { + assert_eq!(is_si_prefixed_electrical_hour_unit(spelling), expected); + } + + #[test] + fn trailing_roman_terminator_is_removed_when_section_continues() { + let terminator = crate::unicode::decode_unicode('⠲'); + let mut encoded = vec![1, terminator]; + + omit_trailing_roman_terminator(&mut encoded); + + assert_eq!(encoded, vec![1]); + } + + #[test] + fn continuing_ascii_unit_clears_a_prior_roman_number_chain() { + use crate::rules::traits::BrailleRule; + + let mut owned = crate::test_helpers::CtxOwned::for_text("GB", true) + .with_prev_word("5") + .with_remaining_words(["SSD"]); + owned.state.roman_number_chain = true; + let mut ctx = owned.ctx_at(0); + + let outcome = Rule69.apply(&mut ctx).expect("Rule 69 unit must encode"); + + assert!(matches!( + outcome, + crate::rules::traits::RuleResult::Consumed + )); + assert!(!ctx.state.roman_number_chain); + assert!(ctx.state.is_english); + } + + /// Rule 69 and its science-braille unit table: a complete Roman-written + /// unit is one section, including its ordinary entry/exit indicators. + #[rstest::rstest] + #[case::minute("90min이다", "⠼⠊⠚⠴⠍⠔⠲⠕⠊")] + #[case::millimetres_of_mercury("140mmHg이다", "⠼⠁⠙⠚⠴⠍⠍⠠⠓⠛⠲⠕⠊")] + #[case::kilogram_force("75.5kgf이다", "⠼⠛⠑⠲⠑⠴⠅⠛⠋⠲⠕⠊")] + #[case::gigawatt_hour("13GWh이다", "⠼⠁⠉⠴⠠⠛⠠⠺⠓⠲⠕⠊")] + #[case::milliampere_hour("900mAh이다", "⠼⠊⠚⠚⠴⠍⠠⠁⠓⠲⠕⠊")] + fn compact_standard_units_form_one_roman_section(#[case] input: &str, #[case] expected: &str) { + assert_eq!(crate::encode_to_unicode(input).unwrap(), expected); + } + + #[rstest::rstest] + #[case::korean_thousand("1천59ha", "1천59㏊")] + #[case::korean_ten_thousand("3만433ha", "3만433㏊")] + fn mixed_korean_numbers_keep_the_complete_ascii_unit( + #[case] ascii: &str, + #[case] compatibility: &str, + ) { + assert_eq!( + crate::encode_to_unicode(ascii).unwrap(), + crate::encode_to_unicode(compatibility).unwrap() + ); + } + + #[test] + fn complete_unit_matching_rejects_an_ascii_word_with_a_unit_prefix() { + let chars = "harmony".chars().collect::>(); + assert!(encode_complete_numeric_ascii_unit(&chars, 0).is_none()); + } + + #[rstest::rstest] + #[case::inch('㏌', "in")] + #[case::centimetre('㎝', "cm")] + #[case::millimetre('㎜', "mm")] + #[case::gigabyte('㎇', "GB")] + fn compatibility_units_match_existing_ascii_unit_spelling( + #[case] glyph: char, + #[case] ascii: &str, + ) { + let ascii_chars = ascii.chars().collect::>(); + let expected = encode_ascii_unit(&ascii_chars, 0) + .expect("existing rule 69 ASCII unit") + .0; + let decomposition = compatibility_unit_decomposition(glyph).unwrap(); + let actual = encode_compatibility_unit(&decomposition, true, true).unwrap(); + assert_eq!(actual, expected); + } + + /// Rules 68/69: a compatibility presentation form follows the same general + /// Roman-unit and superscript algorithm as its Unicode decomposition. + #[rstest::rstest] + #[case::kilogram("㎏", "⠴⠅⠛⠲")] + #[case::gigahertz("㎓", "⠴⠠⠛⠠⠓⠵⠲")] + #[case::cubic_metre("㎥", "⠴⠍⠘⠼⠉")] + #[case::milliwatt("㎽", "⠴⠍⠠⠺⠲")] + #[case::kilowatt("㎾", "⠴⠅⠠⠺⠲")] + #[case::sievert("㏜", "⠴⠠⠎⠧⠲")] + #[case::litre("ℓ", "⠴⠇⠲")] + fn encodes_compatibility_unit_symbols(#[case] input: &str, #[case] expected: &str) { + assert_eq!(crate::encode_to_unicode(input).unwrap(), expected); + } + + /// Rules 29, 33, 34 and 35 depend on the semantic Roman unit, not on + /// whether print used ordinary letters or a Unicode compatibility glyph. + #[rstest::rstest] + #[case::comma_separated_megawatts("수치는 12.5㎿, 30㎿이다", "수치는 12.5MW, 30MW이다")] + #[case::unit_after_roman_number_chain("용량은 Lidocaine 5㎖이다", "용량은 Lidocaine 5ml이다")] + #[case::unit_after_separated_roman_number_chain("대역은 5G 28㎓이다", "대역은 5G 28GHz이다")] + #[case::rule_68_unit_before_closing_parenthesis("면적은(141㏊)이다", "면적은(141ha)이다")] + fn compatibility_unit_presentations_match_semantic_roman_spelling( + #[case] presentation: &str, + #[case] expanded: &str, + ) { + assert_eq!( + crate::encode_to_unicode(presentation).unwrap(), + crate::encode_to_unicode(expanded).unwrap(), + "presentation={presentation:?}" + ); + } + + #[test] + fn superscript_closed_compatibility_units_restart_after_a_comma() { + let actual = crate::encode_to_unicode("농도는 29㎍/㎥, 16㎍/㎥이다").unwrap(); + assert!( + actual.contains("⠍⠘⠼⠉⠐⠀⠼⠁⠋⠴⠨⠍"), + "square/cubic unit must close before Korean comma: {actual:?}" + ); + } + + #[rstest::rstest] + #[case::confusable_counter("3㎠당", true)] + #[case::vowel_initial_predicate("3㎠이다", false)] + fn compatibility_unit_superscript_separates_only_number_confusable_korean( + #[case] input: &str, + #[case] expects_separator: bool, + ) { + let actual = crate::encode_to_unicode(input).unwrap(); + let unit = "⠴⠉⠍⠘⠼⠃"; + let unit_end = actual.find(unit).expect("square-centimetre cells") + unit.len(); + let follows_with_space = actual[unit_end..].starts_with('⠀'); + assert_eq!(follows_with_space, expects_separator, "input={input}"); + } + + #[test] + fn slash_after_korean_starts_a_new_roman_unit_chain() { + let encoded = crate::encode_to_unicode("시/㎏").unwrap(); + assert!( + encoded.ends_with("⠸⠌⠴⠅⠛⠲"), + "the Roman indicator must not be suppressed after a Korean component: {encoded}" + ); + } + + /// Exact PDF examples exercise both Roman-unit continuation through `/` + /// and termination before a slash followed by a Korean unit. + #[rstest::rstest] + #[case::milligram_per_decilitre("160㎎/㎗", "⠼⠁⠋⠚⠴⠍⠛⠸⠌⠙⠇⠲")] + #[case::calorie_per_square_centimetre_per_minute("cal/㎠/min", "⠴⠉⠁⠇⠸⠌⠉⠍⠘⠼⠃⠸⠌⠍⠔⠲")] + #[case::megahertz("96.7 ㎒", "⠼⠊⠋⠲⠛⠀⠴⠠⠍⠠⠓⠵⠲")] + #[case::kilometres_per_hour("80 ㎞/시", "⠼⠓⠚⠀⠴⠅⠍⠲⠸⠌⠠⠕")] + fn preserves_pdf_unit_examples(#[case] input: &str, #[case] expected: &str) { + assert_eq!(crate::encode_to_unicode(input).unwrap(), expected); + } + + /// Rules 33/34/35/69: the ordinary unit terminator is omitted when the + /// actual following boundary is a standard punctuation/enclosing mark or + /// an attached digit continuing the Roman/number chain. These are + /// full-encoder checks, including numeric-prefix routing. + #[rstest::rstest] + #[case::kilogram_in_parentheses("상자(20kg)당", "⠼⠃⠚⠴⠅⠛⠠⠴", "⠴⠅⠛⠲⠠⠴")] + #[case::centimetre_before_korean_comma("키는 173cm, 몸무게는", "⠼⠁⠛⠉⠴⠉⠍⠐", "⠴⠉⠍⠲⠐")] + #[case::centimetre_before_next_measurement("키 173cm, 68kg", "⠼⠁⠛⠉⠴⠉⠍⠂", "⠴⠉⠍⠲⠂")] + #[case::metre_before_sentence_period("비거리 130m.", "⠼⠁⠉⠚⠴⠍⠲", "⠴⠍⠲⠲")] + #[case::compatibility_kilogram_in_parentheses("상자(20㎏)당", "⠼⠃⠚⠴⠅⠛⠠⠴", "⠴⠅⠛⠲⠠⠴")] + #[case::metre_between_numbers("기록은 2m36이다", "⠼⠃⠴⠍⠼⠉⠋", "⠴⠍⠲⠼")] + #[case::metre_between_larger_numbers("기록은 57m57이다", "⠼⠑⠛⠴⠍⠼⠑⠛", "⠴⠍⠲⠼")] + #[case::compatibility_kilometre_before_number("거리는 2㎞30이다", "⠼⠃⠴⠅⠍⠼⠉⠚", "⠴⠅⠍⠲⠼")] + fn omits_unit_terminator_at_standard_override_boundary( + #[case] input: &str, + #[case] expected_segment: &str, + #[case] forbidden_segment: &str, + ) { + let actual = crate::encode_to_unicode(input).unwrap(); + assert!( + actual.contains(expected_segment), + "missing rule-33/34 unit boundary {expected_segment:?} in {actual:?}" + ); + assert!( + !actual.contains(forbidden_segment), + "unexpected rule-69 terminator at rule-33/34 boundary {forbidden_segment:?} in {actual:?}" + ); + } + + /// Rules 29, 33, 35 and 69: a comma between consecutive measurements is + /// UEB punctuation inside one Roman section. The second unit therefore + /// does not repeat the Roman indicator, even when a Korean particle is + /// attached after that unit. + #[rstest::rstest] + #[case::particle_after_second_unit("키는 173cm, 68kg의 차이다", "⠼⠁⠛⠉⠴⠉⠍⠂⠀⠼⠋⠓⠅⠛⠲")] + #[case::second_unit_at_end("키는 173cm, 68kg", "⠼⠁⠛⠉⠴⠉⠍⠂⠀⠼⠋⠓⠅⠛⠲")] + #[case::nanometre_list("공정은 5nm, 1nm는 다르다", "⠼⠑⠴⠝⠍⠂⠀⠼⠁⠝⠍⠲")] + fn keeps_comma_separated_measurements_in_one_roman_section( + #[case] input: &str, + #[case] expected_segment: &str, + ) { + let actual = crate::encode_to_unicode(input).unwrap(); + assert!( + actual.contains(expected_segment), + "missing continuous Roman measurement list {expected_segment:?} in {actual:?}" + ); + } + + /// Rule 69 remains the default outside the rule-33/34 override. End of + /// input, a following Korean syllable, and forced slash boundaries retain + /// the ordinary Roman terminator. + #[rstest::rstest] + #[case::end_of_input("180cm", "⠴⠉⠍⠲")] + #[case::calorie_at_end("열량은 3cal", "⠴⠉⠁⠇⠲")] + #[case::before_korean("1m는", "⠴⠍⠲")] + #[case::before_forced_slash("3m/시", "⠴⠍⠲⠸⠌")] + fn retains_unit_terminator_at_ordinary_rule_69_boundary( + #[case] input: &str, + #[case] expected_segment: &str, + ) { + let actual = crate::encode_to_unicode(input).unwrap(); + assert!( + actual.contains(expected_segment), + "missing ordinary rule-69 unit boundary {expected_segment:?} in {actual:?}" + ); + } + + #[test] + fn boundary_helper_does_not_remove_non_terminator_cells() { + let word = "kg)".chars().collect::>(); + let mut encoded = encode_unicode_cells("⠴⠅⠛"); + omit_roman_terminator_before_boundary(&mut encoded, &word, 2); + assert_eq!(encoded, encode_unicode_cells("⠴⠅⠛")); + } + #[test] fn parses_compact_number_unit_word() { let chars: Vec = "180cm".chars().collect(); diff --git a/libs/braillify/src/rules/korean/rule_70.rs b/libs/braillify/src/rules/korean/rule_70.rs index 34349a84..250f2991 100644 --- a/libs/braillify/src/rules/korean/rule_70.rs +++ b/libs/braillify/src/rules/korean/rule_70.rs @@ -56,8 +56,17 @@ impl BrailleRule for Rule70 { else { return Ok(RuleResult::Skip); }; + // 제70항은 화살표의 앞뒤를 한 칸씩 띄도록 명시한다. 묵자에 + // 공백이 이미 있으면 별도 Space token이 담당하므로, 같은 token + // 안에 인접 문자가 있을 때만 누락된 한 칸을 보충한다. + if ctx.index > 0 && ctx.result.last() != Some(&0) { + ctx.emit(0); + } let encoded = encode_unicode_cells(unicode); ctx.emit_slice(&encoded); + if ctx.index + 1 < ctx.word_len() { + ctx.emit(0); + } Ok(RuleResult::Consumed) } } @@ -76,6 +85,17 @@ mod tests { assert_eq!(encode_unicode_cells(expected), encode_enclosed_arrow(input)); } + /// 제70항 — 화살표 앞뒤 한 칸은 묵자 공백 유무와 무관하게 보장하며, + /// 이미 띄어 쓴 공식 예제에는 공백을 중복하지 않는다. + #[rstest::rstest] + #[case::tight_both_sides("부산→서울", "⠘⠍⠇⠒⠀⠒⠕⠀⠠⠎⠯")] + #[case::tight_right_side("←행주대교", "⠪⠒⠀⠚⠗⠶⠨⠍⠊⠗⠈⠬")] + #[case::tight_left_side("거래량↓", "⠈⠎⠐⠗⠐⠜⠶⠀⠘⠒⠕")] + #[case::already_spaced("부산 → 서울", "⠘⠍⠇⠒⠀⠒⠕⠀⠠⠎⠯")] + fn enforces_one_blank_around_arrow(#[case] input: &str, #[case] expected: &str) { + assert_eq!(crate::encode_to_unicode(input).as_deref(), Ok(expected)); + } + #[test] fn apply_skips_non_korean() { let mut owned = crate::test_helpers::CtxOwned::for_text("A", false); diff --git a/libs/braillify/src/rules/korean/rule_71.rs b/libs/braillify/src/rules/korean/rule_71.rs index 6cbf6c77..4db73246 100644 --- a/libs/braillify/src/rules/korean/rule_71.rs +++ b/libs/braillify/src/rules/korean/rule_71.rs @@ -16,6 +16,7 @@ const MAPPINGS: &[(char, &str)] = &[ ('^', "⠈⠢"), ('#', "⠸⠹"), ('|', "⠸⠳"), + ('│', "⠸⠳"), ('\\', "⠸⠡"), ('&', "⠈⠯"), ('§', "⠘⠎"), @@ -47,6 +48,20 @@ fn should_wrap_information_symbol(ctx: &RuleContext) -> bool { prev_has_korean || next_has_korean } +/// UEB 3.1.1 writes `&` directly between attached ASCII-letter segments +/// (for example, AT&T and B&B). The surrounding Roman section already owns +/// the mode indicators, so Rule 71 must emit only the ampersand cells there. +fn is_attached_roman_ampersand(ctx: &RuleContext) -> bool { + crate::english_logic::is_attached_ascii_roman_ampersand(ctx.word_chars, ctx.index) +} + +fn begins_attached_roman_segment(ctx: &RuleContext) -> bool { + crate::english_logic::is_ampersand_before_attached_ascii_roman_segment( + ctx.word_chars, + ctx.index, + ) +} + pub fn is_rule_71_symbol(c: char) -> bool { MAPPINGS.iter().any(|(candidate, _)| *candidate == c) } @@ -102,7 +117,26 @@ impl BrailleRule for Rule71 { let mut encoded = Vec::new(); if should_wrap_information_symbol(ctx) + && ctx.current_char() == '&' + && begins_attached_roman_segment(ctx) + { + // Korean rules 29/32/71 and UEB 3.1.1 `&c`: the ambiguous + // ampersand opens the Roman section, but the attached ASCII-letter + // segment owns its eventual terminator. Do not close and re-enter + // between the two printed-adjacent items. + if !ctx.state.is_english { + if ctx.state.english_dominant_no_indicator { + ctx.state.is_english = true; + ctx.state.needs_english_continuation = false; + ctx.state.roman_number_chain = false; + } else { + crate::rules::roman_mode::enter_english(ctx.state, ctx.result); + } + } + encoded = encode_unicode_cells(unicode); + } else if should_wrap_information_symbol(ctx) && matches!(ctx.current_char(), '&' | '¶' | '©' | '®' | '™') + && !is_attached_roman_ampersand(ctx) { encoded.push(crate::unicode::decode_unicode('⠴')); encoded.extend(encode_unicode_cells(unicode)); @@ -110,7 +144,19 @@ impl BrailleRule for Rule71 { } else { encoded = encode_unicode_cells(unicode); } + + // U+2502 is the Unicode box-drawing presentation of a vertical line + // segment. Korean Rule 71 assigns the same cells as `|`, while UEB + // 16.4.3 requires a vertical line segment to be surrounded by spaces. + // Insert only missing intra-token boundaries; ordinary Token::Space + // already owns whitespace printed around a standalone line. + if ctx.current_char() == '│' && ctx.prev_char().is_some() { + ctx.emit(0); + } ctx.emit_slice(&encoded); + if ctx.current_char() == '│' && ctx.next_char().is_some() { + ctx.emit(0); + } Ok(RuleResult::Consumed) } } @@ -134,40 +180,127 @@ mod tests { let _ = Rule71.matches(&ctx); } - /// 제71항 — § 정보 기호가 직후 숫자를 만나면 종료표(⠲) 생략 (line 84-86). + /// 제71항 붙임 `헌법§1①` covers the digit continuation that omits ⠲; + /// the end/non-digit controls cover the ordinary wrapped terminator branch. + #[rstest::rstest] + #[case::official_digit_continuation("헌법§1①", "⠴⠘⠎")] + #[case::word_end("헌법§", "⠴⠘⠎⠲")] + #[case::non_digit_continuation("헌법§A", "⠴⠘⠎⠲")] + fn section_sign_wrapper_terminator_boundary(#[case] input: &str, #[case] expected: &str) { + let section_index = input.chars().position(|ch| ch == '§').unwrap(); + let mut owned = crate::test_helpers::CtxOwned::for_text(input, false); + let mut ctx = owned.ctx_at(section_index); + + let outcome = Rule71.apply(&mut ctx).unwrap(); + + assert!(matches!(outcome, RuleResult::Consumed)); + assert_eq!(ctx.result.as_slice(), encode_unicode_cells(expected)); + } + + /// The Korean Rule 71 encoder owns the ampersand cells even while the + /// surrounding Roman section remains open. These are the official UEB + /// 3.1.1 surface forms, not corpus-derived examples. + #[rstest::rstest] + #[case::official_at_and_t("AT&T")] + #[case::official_b_and_b("B&B")] + fn attached_roman_ampersand_emits_bare_rule_71_cells(#[case] input: &str) { + let mut owned = crate::test_helpers::CtxOwned::for_text(input, false); + let ampersand_index = input.chars().position(|ch| ch == '&').unwrap(); + let mut ctx = owned.ctx_at(ampersand_index); + + let outcome = Rule71.apply(&mut ctx).unwrap(); + + assert!(matches!(outcome, RuleResult::Consumed)); + assert_eq!(ctx.result.as_slice(), encode_unicode_cells("⠈⠯")); + } + + /// UEB 3.1.1's official `&c` surface exercises the Korean Rule-71 wrapper + /// state directly: the Roman indicator precedes `&`, and the section stays + /// open for the attached `c` rather than emitting a terminator/re-entry. #[test] - fn rule71_section_sign_before_digit_omits_terminator() { - let word: Vec = "§1".chars().collect(); - let ct = CharType::Symbol('§'); - let mut skip = 0usize; - let mut state = crate::rules::context::EncoderState::new(false); - let mut out = Vec::new(); - let mut ctx = RuleContext { - word_chars: &word, - index: 0, - char_type: &ct, - prev_word: "", - remaining_words: &[], - has_korean_char: false, - is_all_uppercase: false, - ascii_starts_at_beginning: false, - skip_count: &mut skip, - state: &mut state, - result: &mut out, - }; + fn one_sided_official_ampersand_opens_and_keeps_roman_section() { + let mut owned = crate::test_helpers::CtxOwned::for_text("&c", true); + let mut ctx = owned.ctx_at(0); + + let outcome = Rule71.apply(&mut ctx).unwrap(); + + assert!(matches!(outcome, RuleResult::Consumed)); + assert_eq!(ctx.result.as_slice(), encode_unicode_cells("⠴⠈⠯")); + assert!(ctx.state.is_english); + } + + #[test] + fn attached_ampersand_resumes_indicator_free_english_dominant_context() { + let mut owned = crate::test_helpers::CtxOwned::for_text("&c", true); + owned.state.english_dominant_no_indicator = true; + let mut ctx = owned.ctx_at(0); + let outcome = Rule71.apply(&mut ctx).unwrap(); + assert!(matches!(outcome, RuleResult::Consumed)); - // No ⠲ terminator because next char is a digit - assert!(!out.contains(&crate::unicode::decode_unicode('⠲'))); + assert_eq!(ctx.result.as_slice(), encode_unicode_cells("⠈⠯")); + assert!(ctx.state.is_english); + assert!(!ctx.state.needs_english_continuation); + assert!(!ctx.state.roman_number_chain); + } + + /// Full-encoder controls reproduce the two UEB 3.1.1 examples exactly. + #[rstest::rstest] + #[case::official_at_and_t("AT&T", "⠠⠠⠁⠞⠈⠯⠠⠞")] + #[case::official_b_and_b("B&B", "⠠⠃⠈⠯⠠⠃")] + fn full_encoder_preserves_official_ueb_ampersand_examples( + #[case] input: &str, + #[case] expected: &str, + ) { + assert_eq!(crate::encode_to_unicode(input).unwrap(), expected); + } + + /// UEB 3.1.1 keeps `AT&T` in one Roman section and Korean rule 35 keeps + /// the directly following digit in that same section. The two rules must + /// compose without a terminator/re-entry around the ampersand or digit. + #[test] + fn full_encoder_keeps_ampersand_roman_number_chain() { + assert_eq!( + crate::encode_to_unicode("가 AT&T3 나").unwrap(), + "⠫⠀⠴⠠⠠⠁⠞⠈⠯⠠⠞⠼⠉⠀⠉" + ); + } + + /// UEB 8.4.2 ends capitals word mode at the nonalphabetic ampersand. + /// Wrapping the official UEB 3.1.1 examples in neutral Korean text proves + /// that the mixed-document rule-28/29 path restarts capitalization for the + /// next ASCII-letter segment while keeping one Roman section. + #[rstest::rstest] + #[case::official_at_and_t("가 AT&T 나", "⠫⠀⠴⠠⠠⠁⠞⠈⠯⠠⠞⠲⠀⠉")] + #[case::official_b_and_b("가 B&B 나", "⠫⠀⠴⠠⠃⠈⠯⠠⠃⠲⠀⠉")] + fn korean_wrapper_preserves_ampersand_capitalization_extent( + #[case] input: &str, + #[case] expected: &str, + ) { + assert_eq!(crate::encode_to_unicode(input).as_deref(), Ok(expected)); + } + + #[rstest::rstest] + #[case::standalone("│", "|")] + #[case::spaced("저자 │ 홍길동", "저자 | 홍길동")] + #[case::attached("제작│감독", "제작 | 감독")] + fn box_drawing_vertical_line_matches_rule_71_print_form( + #[case] presentation: &str, + #[case] standard_print: &str, + ) { + assert_eq!( + crate::encode_to_unicode(presentation), + crate::encode_to_unicode(standard_print) + ); } - /// rule_71:85 — § followed by NON-digit (or end of input) appends ⠲ terminator. + /// Korean Rule 71's spaced Hangul example remains an independently + /// delimited information symbol after the attached-Roman exception. #[test] - fn rule71_section_symbol_followed_by_non_digit_appends_terminator() { - // Encode "§A" — next char is letter, not digit → ⠲ appended at line 85. - let result = crate::encode("§A"); - assert!(result.is_ok()); - // Also: § alone (no next char) → no digit → ⠲ appended. - let _ = crate::encode("§"); + fn full_encoder_preserves_official_korean_spaced_ampersand_example() { + assert_eq!( + crate::encode_to_unicode("종이접기 & 클레이아트").unwrap(), + "⠨⠿⠕⠨⠎⠃⠈⠕⠀⠴⠈⠯⠲⠀⠋⠮⠐⠝⠕⠣⠓⠪", + ); } } diff --git a/libs/braillify/src/rules/korean/rule_72.rs b/libs/braillify/src/rules/korean/rule_72.rs index 493a9383..e9318847 100644 --- a/libs/braillify/src/rules/korean/rule_72.rs +++ b/libs/braillify/src/rules/korean/rule_72.rs @@ -1,6 +1,10 @@ +use std::borrow::Cow; + use crate::char_struct::CharType; use crate::rules::RuleMeta; -use crate::rules::context::RuleContext; +use crate::rules::context::{EncodingMode, RuleContext}; +use crate::rules::token::{SpaceKind, Token, WordMeta, WordToken}; +use crate::rules::token_rule::{TokenAction, TokenPhase, TokenRule}; use crate::rules::traits::{BrailleRule, Phase, RuleResult}; pub static META: RuleMeta = RuleMeta { @@ -15,6 +19,8 @@ const MAPPINGS: &[(char, &str)] = &[ ('○', "⠸⠴"), ('□', "⠸⠶"), ('△', "⠸⠬"), + ('▲', "⠸⠬"), + ('▴', "⠸⠬"), ('•', "⠸⠲"), ('◎', "⠸⠴⠴"), ('▣', "⠸⠶⠶"), @@ -31,6 +37,116 @@ pub fn is_rule_72_symbol(c: char) -> bool { MAPPINGS.iter().any(|(candidate, _)| *candidate == c) } +fn is_vertex_decoration(c: char) -> bool { + matches!(c, '\'' | '′' | '″' | '\u{2034}') || ('\u{2070}'..='\u{209f}').contains(&c) +} + +fn consume_triangle_name(chars: &[char], start: usize) -> Option { + if chars.get(start) != Some(&'△') { + return None; + } + + let mut index = start + 1; + for _ in 0..3 { + if !chars.get(index).is_some_and(char::is_ascii_uppercase) { + return None; + } + index += 1; + while chars.get(index).is_some_and(|c| is_vertex_decoration(*c)) { + index += 1; + } + } + Some(index) +} + +/// 수학 점자 제40·42·43항의 삼각형 이름은 `△` 뒤에 꼭짓점 대문자 +/// 세 개를 붙여 쓴다 (`△ABC`, `△A′B′C′`). 합동·닮음 관계로 같은 +/// 형태가 이어지는 경우도 제72항 글머리 기호로 재해석하지 않는다. +fn is_triangle_geometry_expression(chars: &[char]) -> bool { + let Some(mut index) = consume_triangle_name(chars, 0) else { + return false; + }; + + loop { + if index == chars.len() { + return true; + } + if chars[index..] + .iter() + .all(|c| matches!(*c, ',' | '.' | ';' | '?' | '!')) + { + return true; + } + if !matches!(chars[index], '=' | '≡' | '≅' | '∼' | '∽' | '≈') { + return false; + } + index += 1; + let Some(next) = consume_triangle_name(chars, index) else { + return false; + }; + index = next; + } +} + +fn owned_word(text: String) -> Token<'static> { + let chars = text.chars().collect::>(); + let meta = WordMeta::from_chars(&chars); + Token::Word(WordToken { + text: Cow::Owned(text), + chars, + meta, + }) +} + +/// 제72항 글머리 기호가 항목 내용에 붙은 일반 텍스트를, 수식 판정보다 +/// 먼저 `기호 + 한 칸 + 내용`으로 복원한다. 수학 제40·42·43항 문법은 +/// 위의 구조 판정으로 제외한다. +pub struct Rule72AttachedMarkerTokenRule; + +impl TokenRule for Rule72AttachedMarkerTokenRule { + fn phase(&self) -> TokenPhase { + TokenPhase::Normalization + } + + fn priority(&self) -> u16 { + 90 + } + + fn apply<'a>( + &self, + tokens: &[Token<'a>], + index: usize, + _state: &mut crate::rules::context::EncoderState, + ) -> Result, String> { + let Some(Token::Word(word)) = tokens.get(index) else { + return Ok(TokenAction::Noop); + }; + let Some(marker) = word.chars.first().copied() else { + return Ok(TokenAction::Noop); + }; + if !matches!(marker, '△' | '▲' | '▴') || word.chars.len() == 1 || word.meta.has_korean + { + return Ok(TokenAction::Noop); + } + // 제57항의 반복 가림표는 하나의 묶음이다. 첫 `△`를 제72항의 + // 글머리 기호로 떼어 내면 문자 규칙이 반복 개수를 볼 수 없으므로, + // 같은 표지가 연속될 때에는 원래 토큰을 그대로 둔다. + if word.chars.get(1) == Some(&marker) { + return Ok(TokenAction::Noop); + } + if marker == '△' && is_triangle_geometry_expression(&word.chars) { + return Ok(TokenAction::Noop); + } + + let rest = word.chars[1..].iter().collect::(); + Ok(TokenAction::ReplaceMany(vec![ + owned_word(marker.to_string()), + Token::Space(SpaceKind::Regular), + owned_word(rest), + ])) + } +} + pub struct Rule72; impl BrailleRule for Rule72 { @@ -57,10 +173,25 @@ impl BrailleRule for Rule72 { return Ok(RuleResult::Skip); } + // 명시적인 사물부호 문맥(제49항)과 수학 제40항의 `△ABC`는 + // 제72항의 동형 글머리 기호보다 우선한다. + if matches!(ctx.state.current_mode(), EncodingMode::ObjectSymbol) + || is_triangle_geometry_expression(&ctx.word_chars[ctx.index..]) + { + return Ok(RuleResult::Skip); + } + + // 일반 텍스트 추출 과정에서 `△항목`, `△R&D`, `△2025`처럼 글머리 + // 기호와 항목 내용의 경계가 사라질 수 있다. 제72항 공식 예제처럼 + // 둘 사이 한 칸을 복원하되, 문자 종류를 열거하지 않고 비공백 내용이 + // 실제로 이어지는지만 판정한다. + let tight_before_content = matches!(current, '△' | '▲' | '▴') + && ctx.next_char().is_some_and(|c| !c.is_whitespace()); let contextual_marker = ctx.word_len() == 1 || ctx .next_char() .is_some_and(|c| c.is_whitespace() || matches!(c, '(' | '\'' | '"')) + || tight_before_content || matches!(current, '◎' | '▣'); if !contextual_marker { return Ok(RuleResult::Skip); @@ -72,6 +203,9 @@ impl BrailleRule for Rule72 { }; let encoded = encode_unicode_cells(unicode); ctx.emit_slice(&encoded); + if tight_before_content { + ctx.emit(0); + } Ok(RuleResult::Consumed) } } @@ -100,6 +234,8 @@ mod tests { #[case::square('□', "⠸⠶")] #[case::triangle('△', "⠸⠬")] #[case::bullet('•', "⠸⠲")] + #[case::filled_triangle('▲', "⠸⠬")] + #[case::small_filled_triangle('▴', "⠸⠬")] #[case::double_circle('◎', "⠸⠴⠴")] #[case::filled_square('▣', "⠸⠶⠶")] fn apply_encodes_placeholder_markers(#[case] input: char, #[case] expected: &str) { @@ -109,6 +245,90 @@ mod tests { assert_eq!(output, encode_unicode_cells(expected)); } + #[rstest::rstest] + #[case::outline("△문화", "△ 문화")] + #[case::filled("▲문화", "△ 문화")] + #[case::small_filled("▴문화", "△ 문화")] + #[case::roman_item("목록은 △R&D이다", "목록은 △ R&D이다")] + #[case::numeric_item("목록은 △2025년이다", "목록은 △ 2025년이다")] + #[case::quoted_item("목록은 △‘첫째’이다", "목록은 △ ‘첫째’이다")] + #[case::roman_token("목록은 △AI", "목록은 △ AI")] + #[case::numeric_token("목록은 △2025", "목록은 △ 2025")] + fn attached_triangle_list_markers_supply_the_rule_72_boundary( + #[case] input: &str, + #[case] standard_print: &str, + ) { + assert_eq!( + crate::encode_to_unicode(input), + crate::encode_to_unicode(standard_print) + ); + } + + #[test] + fn repeated_triangle_stays_grouped_for_rule_57() { + assert_eq!(crate::encode_to_unicode("△△").unwrap(), "⠸⠬⠬⠇"); + } + + #[test] + fn list_marker_after_attached_previous_item_still_supplies_right_boundary() { + let mut owned = crate::test_helpers::CtxOwned::for_text("첫째△둘째", false); + let mut ctx = owned.ctx_at(2); + + let outcome = Rule72.apply(&mut ctx).unwrap(); + + assert!(matches!(outcome, RuleResult::Consumed)); + assert_eq!(&*ctx.result, &encode_unicode_cells("⠸⠬⠀")); + } + + #[test] + fn math_triangle_name_is_not_reinterpreted_as_a_list_marker() { + let mut owned = crate::test_helpers::CtxOwned::for_text("△ABC", false); + let mut ctx = owned.ctx_at(0); + + let outcome = Rule72.apply(&mut ctx).unwrap(); + + assert!(matches!(outcome, RuleResult::Skip)); + assert!(ctx.result.is_empty()); + } + + #[rstest::rstest] + #[case::plain("△ABC")] + #[case::primed("△A′B′C′")] + #[case::congruent("△ABC≡△DEF")] + #[case::similar_primed("△ABC∽△A′B′C′")] + #[case::trailing_punctuation("△ABC.")] + fn recognizes_math_triangle_grammar(#[case] input: &str) { + let chars = input.chars().collect::>(); + assert!(is_triangle_geometry_expression(&chars)); + } + + #[rstest::rstest] + #[case::acronym_with_gloss("△UAM(도심항공교통)")] + #[case::brand_with_digits("△G3930P")] + #[case::numeric_item("△2025")] + #[case::incomplete_second_triangle("△ABC=△AB")] + fn attached_list_items_do_not_match_triangle_geometry(#[case] input: &str) { + let chars = input.chars().collect::>(); + assert!(!is_triangle_geometry_expression(&chars)); + } + + #[test] + fn attached_marker_rule_ignores_an_empty_word_token() { + let tokens = vec![Token::Word(WordToken { + text: Cow::Borrowed(""), + chars: Vec::new(), + meta: WordMeta::from_chars(&[]), + })]; + let mut state = crate::rules::context::EncoderState::new(false); + + assert!(matches!( + Rule72AttachedMarkerTokenRule + .apply(&tokens, 0, &mut state) + .expect("empty input is a no-op"), + TokenAction::Noop + )); + } + #[test] fn detects_double_circle_placeholder_symbol() { assert!(is_rule_72_symbol('◎')); diff --git a/libs/braillify/src/rules/korean/rule_english_symbol.rs b/libs/braillify/src/rules/korean/rule_english_symbol.rs index 3b8133ae..5c9df60d 100644 --- a/libs/braillify/src/rules/korean/rule_english_symbol.rs +++ b/libs/braillify/src/rules/korean/rule_english_symbol.rs @@ -9,6 +9,7 @@ use crate::char_struct::CharType; use crate::english_logic; use crate::rules::RuleMeta; use crate::rules::context::RuleContext; +use crate::rules::english_ueb::rule_5_7::is_wordsign_letter; use crate::rules::traits::{BrailleRule, Phase, RuleResult}; use crate::symbol_shortcut; use crate::utils; @@ -23,6 +24,30 @@ pub static META: RuleMeta = RuleMeta { pub struct RuleEnglishSymbol; +/// Korean rules 28 and 32 + UEB 5.7.1: inside an already-open Roman +/// section, a one-letter ASCII segment after a hyphen needs the continuation / +/// grade-1 cell when it could be read as an alphabetic wordsign. Multi-letter +/// segments such as `pop`, `ray`, and `Case` do not take this indicator. Rule +/// 28's Roman indicator already establishes the first segment (`K-pop`), while +/// rule 36's `v-x` demonstrates the indicator on the later single-letter +/// segment. A directly attached ASCII digit remains part of the same rule-35 +/// Roman-number sequence and therefore is not a single-letter segment. +fn hyphen_suffix_requires_grade1(word_chars: &[char], hyphen_index: usize) -> bool { + let Some(suffix) = word_chars.get(hyphen_index + 1..) else { + return false; + }; + let suffix_len = suffix + .iter() + .take_while(|ch| ch.is_ascii_alphabetic()) + .count(); + + suffix_len == 1 + && is_wordsign_letter(suffix[0]) + && suffix + .get(suffix_len) + .is_none_or(|ch| !ch.is_ascii_alphanumeric()) +} + impl BrailleRule for RuleEnglishSymbol { fn meta(&self) -> &'static RuleMeta { &META @@ -45,9 +70,29 @@ impl BrailleRule for RuleEnglishSymbol { return Ok(RuleResult::Skip); }; + // 한글 점자 제43항·제48항: ASCII 숫자 사이의 마침표는 숫자 흐름의 + // 소수점이다. 같은 어절 뒤쪽에 로마자가 있다는 이유만으로 이 위치에서 + // 로마자 모드에 재진입하면 `⠴⠲`가 되어 수표 뒤의 올바른 `⠲` 앞에 + // 불필요한 로마자표가 붙는다. 이 기호는 아래의 일반 한글 문장부호 + // 규칙이 처리하도록 넘기고, 접미사 종류에는 관여하지 않는다. + if *sym == '.' + && ctx.prev_char().is_some_and(|ch| ch.is_ascii_digit()) + && ctx.next_char().is_some_and(|ch| ch.is_ascii_digit()) + { + return Ok(RuleResult::Continue); + } + + // Rule 69 [붙임 3]: a slash joining two complete Roman-written unit + // components stays in that measurement chain. Do not let the generic + // English-symbol route insert a second Roman indicator before `/`. + if *sym == '/' && super::rule_69::is_ascii_unit_chain_slash(ctx.word_chars, ctx.index) { + return Ok(RuleResult::Continue); + } + let mut use_english_symbol = english_logic::should_render_symbol_as_english( ctx.state.english_indicator, ctx.state.is_english, + ctx.state.doc_summary.is_english_majority, &ctx.state.parenthesis_stack, *sym, ctx.word_chars, @@ -55,6 +100,26 @@ impl BrailleRule for RuleEnglishSymbol { ctx.remaining_words, ); + // Korean rules 34 and 54: when a Korean prose item (optionally ending + // in an attached Arabic number) introduces a Roman explanation, the + // Korean opening parenthesis is written before the Roman indicator. + // A parenthesis reached while a Roman section is already active stays + // UEB punctuation (`ABC(def)`), as does ordinary function notation. + if *sym == '(' && !ctx.state.is_english { + let prefix = &ctx.word_chars[..ctx.index]; + let prefix_contains_korean = prefix.iter().any(|ch| utils::is_korean_char(*ch)); + let numeric_prefix = !prefix.is_empty() + && prefix.iter().any(char::is_ascii_digit) + && prefix.iter().all(|ch| { + ch.is_ascii_digit() + || matches!(*ch, '.' | ',' | '\'' | '’' | '"' | '”' | '‘' | '“') + }); + let previous_word_is_korean = ctx.prev_word.chars().any(utils::is_korean_char); + if prefix_contains_korean || (numeric_prefix && previous_word_is_korean) { + use_english_symbol = false; + } + } + // 제39항 영-한 wrap context: 단어 끝의 영어 모드 유지 가능 기호(. , : ;) // 다음에 한글 어절(wrap 대상)이 이어지면 그 기호를 영어 점자로 처리한다. // 예) "(Korean:" 끝의 ':'은 다음 wrap된 "반찬" 직전이므로 영어 점자 ⠒. @@ -83,23 +148,30 @@ impl BrailleRule for RuleEnglishSymbol { let can_use_english_symbol = ctx.state.is_english || has_ascii_alphabetic; if ctx.state.english_indicator && can_use_english_symbol && use_english_symbol { - if !ctx.state.is_english && !ctx.state.needs_english_continuation { + if !ctx.state.is_english + && !ctx.state.needs_english_continuation + && !ctx.state.roman_number_chain + { ctx.emit(52); ctx.state.is_english = true; ctx.state.needs_english_continuation = false; } - if let Some(encoded) = symbol_shortcut::encode_english_char_symbol_shortcut(*sym) { + let encoded = if *sym == '\'' { + // `use_english_symbol` is true here only for an ASCII apostrophe + // immediately between ASCII letters. Keep that narrow UEB 8.4.2 + // role local instead of making detached straight quotes globally + // eligible for the UEB apostrophe cell. + crate::rules::english_ueb::rule_7::encode_punctuation(*sym) + } else { + symbol_shortcut::encode_english_char_symbol_shortcut(*sym) + }; + if let Some(encoded) = encoded { ctx.emit_slice(&encoded); - if *sym == '-' && ctx.state.is_english { - // 다음 글자가 숫자이면 수표(⠼)가 emit되므로 연속표(⠰)는 - // 불필요하다 (제35항 D-100 같은 영문-숫자 인접 패턴). - let next_is_digit = ctx - .word_chars - .get(ctx.index + 1) - .is_some_and(|c| c.is_ascii_digit()); - if !next_is_digit { - ctx.emit(crate::rules::korean::rule_29::ENGLISH_CONTINUATION); - } + if *sym == '-' + && ctx.state.is_english + && hyphen_suffix_requires_grade1(ctx.word_chars, ctx.index) + { + ctx.emit(crate::rules::korean::rule_29::ENGLISH_CONTINUATION); } return Ok(RuleResult::Consumed); } @@ -123,6 +195,79 @@ impl BrailleRule for RuleEnglishSymbol { mod tests { use super::*; + #[rstest::rstest] + #[case::rule_36_single_x("v-x", true)] + #[case::single_x_before_korean("v-x쪽", true)] + #[case::single_t_before_korean_annotation("CAR-T(카티)", true)] + #[case::non_wordsign_i("v-i", false)] + #[case::multi_letter_pop("K-pop", false)] + #[case::multi_letter_case("Title-Case", false)] + #[case::letter_abutting_digit("v-x1", false)] + #[case::multi_letter_shortform_candidate("CD-AB", false)] + fn grade1_after_hyphen_is_limited_to_a_standing_single_letter( + #[case] input: &str, + #[case] expected: bool, + ) { + let chars = input.chars().collect::>(); + let hyphen_index = chars + .iter() + .position(|ch| *ch == '-') + .expect("fixture must contain a hyphen"); + + assert_eq!( + hyphen_suffix_requires_grade1(&chars, hyphen_index), + expected + ); + } + + #[test] + fn hyphen_suffix_lookup_rejects_an_out_of_bounds_index() { + assert!(!hyphen_suffix_requires_grade1(&['A'], 1)); + } + + /// Korean rules 28/29/32: the Roman indicator establishes the first + /// one-letter segment, and a multi-letter segment after the hyphen starts + /// directly with its UEB letters. In particular, no continuation/grade-1 + /// cell is inserted between the hyphen and the lowercase word. + #[rstest::rstest] + #[case::k_pop("가 K-pop 나", "⠫⠀⠴⠠⠅⠤⠏⠕⠏⠲⠀⠉")] + #[case::x_ray("가 X-ray 나", "⠫⠀⠴⠠⠭⠤⠗⠁⠽⠲⠀⠉")] + #[case::k_water("가 K-water 나", "⠫⠀⠴⠠⠅⠤⠺⠁⠞⠻⠲⠀⠉")] + fn hyphenated_roman_word_has_no_spurious_post_hyphen_indicator( + #[case] input: &str, + #[case] expected: &str, + ) { + assert_eq!(crate::encode_to_unicode(input).as_deref(), Ok(expected)); + } + + /// Korean rule 35: the official `D-100` establishes that adjacent Roman + /// letters, digits, and identifier hyphens form one chain. The Roman + /// indicator stays at the first Roman letter; a number-led item opens its + /// Roman section only when its first Roman letter is reached. + #[rstest::rstest] + #[case::roman_led_multi_segment("가 CV3-AD685 나", "⠫⠀⠴⠠⠠⠉⠧⠼⠉⠤⠠⠠⠁⠙⠼⠋⠓⠑⠀⠉")] + #[case::number_led_word("가 0-Zone 나", "⠫⠀⠼⠚⠤⠴⠠⠵⠐⠕⠲⠀⠉")] + #[case::roman_led_repeated_numeric_segments("가 N-79-20 나", "⠫⠀⠴⠠⠝⠤⠼⠛⠊⠤⠼⠃⠚⠀⠉")] + fn rule_35_places_the_roman_indicator_at_the_first_roman_letter( + #[case] input: &str, + #[case] expected: &str, + ) { + assert_eq!(crate::encode_to_unicode(input).as_deref(), Ok(expected)); + } + + /// Korean rules 29 and 34: a Korean closing parenthesis ends its enclosed + /// Roman section. Later Roman text therefore starts with a new Roman + /// indicator, even when a comma, number, or identifier hyphen intervenes. + #[rstest::rstest] + #[case::numeric_item_after_comma("가 액세스(FWA), 5G 나", "⠫⠀⠗⠁⠠⠝⠠⠪⠦⠄⠴⠠⠠⠋⠺⠁⠠⠴⠐⠀⠼⠑⠴⠠⠛⠲⠀⠉")] + #[case::hyphenated_item_after_enclosure("가(GTX)-C 나", "⠫⠦⠄⠴⠠⠠⠛⠞⠭⠠⠴⠤⠴⠠⠉⠲⠀⠉")] + fn korean_parenthesis_does_not_leak_english_continuation( + #[case] input: &str, + #[case] expected: &str, + ) { + assert_eq!(crate::encode_to_unicode(input).as_deref(), Ok(expected)); + } + #[test] fn apply_exercise() { let mut owned = crate::test_helpers::CtxOwned::for_text("A", false); @@ -166,4 +311,81 @@ mod tests { assert!(ctx.state.parenthesis_stack.is_empty()); } + + #[test] + fn decimal_point_between_digits_is_not_an_english_entry_symbol() { + let mut owned = crate::test_helpers::CtxOwned::for_text("3.5P", false); + let mut ctx = owned.ctx_at(1); + + let outcome = RuleEnglishSymbol.apply(&mut ctx).unwrap(); + + assert_eq!(outcome, RuleResult::Continue); + assert!(owned.result.is_empty()); + } + + #[rstest::rstest] + #[case::percent_with_later_roman("42.2%포인트(P)", "42.2")] + #[case::korean_unit_with_annotation("34.3리터(L)", "34.3")] + #[case::two_decimals_with_arrow("99.8→99.4", "99.8")] + #[case::roman_identifier("GPT-3.5", "3.5")] + fn decimal_subsequence_matches_the_standalone_rule_48_encoding( + #[case] input: &str, + #[case] decimal: &str, + ) { + let actual = crate::encode_to_unicode(input).expect("mixed decimal context must encode"); + let standalone = + crate::encode_to_unicode(decimal).expect("standalone rule-48 decimal must encode"); + + assert!( + actual.contains(&standalone), + "input={input}, decimal={decimal}, actual={actual}, standalone={standalone}" + ); + } + + /// UEB 5.7.2 prints `CD-ROM` with one grade-1 indicator before the complete + /// letters-sequence and no second grade-1 indicator after the hyphen. This + /// full-encoder wrapper exercises the Korean rule-29 character route rather + /// than the standalone-English token route used by the standard PDF case. + #[test] + fn korean_wrapper_keeps_pdf_cd_rom_as_one_grade1_letters_sequence() { + let output = crate::encode("가(CD-ROM)나").expect("Korean wrapper must encode"); + let expected_ueb = "⠰⠠⠠⠉⠙⠤⠠⠠⠗⠕⠍" + .chars() + .map(crate::unicode::decode_unicode) + .collect::>(); + let roman_start = output + .iter() + .position(|cell| *cell == crate::rules::korean::rule_29::ROMAN_INDICATOR) + .expect("Korean wrapper must enter one Roman section") + + 1; + + assert_eq!( + output.get(roman_start..roman_start + expected_ueb.len()), + Some(expected_ueb.as_slice()) + ); + } + + /// UEB 3.3.1 writes the official `M*A*S*H` example with the UEB asterisk + /// and no mode boundary around any of its attached marks. Korean rule 32 + /// adds only the outer Roman indicator and terminator in mixed text. + #[test] + fn korean_wrapper_keeps_official_mash_in_one_roman_section() { + let official = crate::encode_to_unicode("M*A*S*H").unwrap(); + assert_eq!(official, "⠠⠍⠐⠔⠠⠁⠐⠔⠠⠎⠐⠔⠠⠓"); + assert_eq!( + crate::encode_to_unicode("가 M*A*S*H 나").as_deref(), + Ok("⠫⠀⠴⠠⠍⠐⠔⠠⠁⠐⠔⠠⠎⠐⠔⠠⠓⠲⠀⠉") + ); + } + + /// UEB 7.3: a Unicode ellipsis closing Roman content is equivalent to the + /// three-full-stop print spelling, even when Korean text owns the outer + /// parenthesis. + #[test] + fn roman_ellipsis_before_a_closing_parenthesis_uses_ueb_cells() { + assert_eq!( + crate::encode_to_unicode("문구(I AM…)이다"), + crate::encode_to_unicode("문구(I AM...)이다") + ); + } } diff --git a/libs/braillify/src/rules/korean/rule_math.rs b/libs/braillify/src/rules/korean/rule_math.rs index 51fa40c5..1e593c7c 100644 --- a/libs/braillify/src/rules/korean/rule_math.rs +++ b/libs/braillify/src/rules/korean/rule_math.rs @@ -18,11 +18,257 @@ pub static META: RuleMeta = RuleMeta { description: "Math symbols with Korean spacing rules", }; -/// Korean particles (josa) that should NOT have spacing before them. -const JOSA: &[&str] = &["과", "와", "이다", "하고", "이랑", "와", "랑", "아니다"]; +/// Korean particles or copulas that do not form the right-hand operand of an +/// Article 46 expression by themselves. +const NON_OPERAND_KOREAN_SUFFIXES: &[&str] = + &["과", "와", "의", "이다", "하고", "이랑", "랑", "아니다"]; pub struct RuleMath; +fn matching_opening_delimiter(ch: char) -> Option { + match ch { + ')' => Some('('), + ']' => Some('['), + '}' => Some('{'), + '〉' => Some('〈'), + '》' => Some('《'), + '」' => Some('「'), + '』' => Some('『'), + '】' => Some('【'), + '〕' => Some('〔'), + '〗' => Some('〖'), + '〙' => Some('〘'), + '〛' => Some('〚'), + _ => None, + } +} + +fn matching_closing_delimiter(ch: char) -> Option { + match ch { + '(' => Some(')'), + '[' => Some(']'), + '{' => Some('}'), + '〈' => Some('〉'), + '《' => Some('》'), + '「' => Some('」'), + '『' => Some('』'), + '【' => Some('】'), + '〔' => Some('〕'), + '〖' => Some('〗'), + '〘' => Some('〙'), + '〚' => Some('〛'), + _ => None, + } +} + +fn is_operand_separator(ch: char) -> bool { + matches!( + ch, + '+' | '-' + | '−' + | '×' + | '÷' + | '=' + | '<' + | '>' + | '≤' + | '≥' + | '≠' + | ',' + | ';' + | ':' + | '!' + | '?' + | '…' + | '\'' + | '"' + | '‘' + | '’' + | '“' + | '”' + ) +} + +/// Finds the syntactic operand immediately to the left of an Article 46 sign. +/// A balanced annotation remains part of its operand (`레트로(RETRO)`), while +/// an unmatched opening delimiter is a hard boundary (`기업(+5p)`). +fn left_operand(chars: &[char], operator_index: usize) -> &[char] { + let mut start = operator_index; + let mut openings = Vec::new(); + + for index in (0..operator_index).rev() { + let ch = chars[index]; + if let Some(opening) = matching_opening_delimiter(ch) { + openings.push(opening); + start = index; + continue; + } + if matching_closing_delimiter(ch).is_some() { + if openings.last() == Some(&ch) { + openings.pop(); + start = index; + continue; + } + break; + } + if openings.is_empty() && is_operand_separator(ch) { + break; + } + start = index; + } + + &chars[start..operator_index] +} + +/// Finds the syntactic operand immediately to the right of an Article 46 sign. +/// Balanced annotations and numeric unit notation stay inside the operand, but +/// the next top-level sign or unmatched closing delimiter ends it. +fn right_operand(chars: &[char], operator_index: usize) -> &[char] { + let mut end = operator_index + 1; + let mut closings = Vec::new(); + + for (index, ch) in chars.iter().copied().enumerate().skip(operator_index + 1) { + if let Some(closing) = matching_closing_delimiter(ch) { + closings.push(closing); + end = index + 1; + continue; + } + if matching_opening_delimiter(ch).is_some() { + if closings.last() == Some(&ch) { + closings.pop(); + end = index + 1; + continue; + } + break; + } + if closings.is_empty() && is_operand_separator(ch) { + break; + } + end = index + 1; + } + + &chars[operator_index + 1..end] +} + +fn first_korean_run(chars: &[char]) -> Option { + let start = chars.iter().position(|ch| utils::is_korean_char(*ch))?; + let end = chars[start..] + .iter() + .position(|ch| !utils::is_korean_char(*ch)) + .map_or(chars.len(), |offset| start + offset); + Some(chars[start..end].iter().collect()) +} + +fn rule_46_requires_padding(ctx: &RuleContext) -> bool { + let left_is_korean_operand = left_operand(ctx.word_chars, ctx.index) + .iter() + .any(|ch| utils::is_korean_char(*ch)); + let right_is_non_suffix_korean_operand = + first_korean_run(right_operand(ctx.word_chars, ctx.index)) + .is_some_and(|run| !NON_OPERAND_KOREAN_SUFFIXES.contains(&run.as_str())); + + left_is_korean_operand && right_is_non_suffix_korean_operand +} + +/// U+002D is both HYPHEN-MINUS, so its braille meaning has to be inferred from +/// syntax. Treat it as the Article 45 subtraction/minus sign only when the +/// surrounding token makes that role explicit. In particular, a leading +/// signed number is a minus, while phone numbers, dates, ranges and identifiers +/// such as `02-799-1000` and `A-3` remain hyphenated. +fn is_semantic_ascii_minus(ctx: &RuleContext) -> bool { + if ctx.current_char() != '-' { + return false; + } + + let next_starts_number = ctx.next_char().is_some_and(|next| { + next.is_ascii_digit() + || (next == '.' + && ctx + .word_chars + .get(ctx.index + 2) + .is_some_and(char::is_ascii_digit)) + }); + let unary_boundary = ctx.prev_char().is_none_or(|prev| { + matches!( + prev, + '(' | '[' + | '{' + | '〈' + | '《' + | '「' + | '『' + | '【' + | '〔' + | '〖' + | '〘' + | '〚' + | '‘' + | '“' + | '\'' + | '"' + | ',' + | ':' + | ';' + | '=' + | '+' + | '×' + | '÷' + | '<' + | '>' + | '≤' + | '≥' + | '≠' + ) + }); + if next_starts_number && unary_boundary { + return true; + } + + // A sign cited by itself inside a matched delimiter is an operator, as in + // the common polarity notation `양(+)극·음(-)극`. A hyphen joining text has + // operands on the same side and therefore cannot have this shape. + let isolated_operator = matches!( + (ctx.prev_char(), ctx.next_char()), + (Some('('), Some(')')) + | (Some('['), Some(']')) + | (Some('{'), Some('}')) + | (Some('〈'), Some('〉')) + | (Some('《'), Some('》')) + | (Some('「'), Some('」')) + | (Some('『'), Some('』')) + | (Some('【'), Some('】')) + | (Some('〔'), Some('〕')) + | (Some('〖'), Some('〗')) + | (Some('〘'), Some('〙')) + | (Some('〚'), Some('〛')) + | (Some('‘'), Some('’')) + | (Some('“'), Some('”')) + | (Some('\''), Some('\'')) + | (Some('"'), Some('"')) + ); + if isolated_operator { + return true; + } + + // Article 46's printed example `5개-3개=2개` contains Hangul, so the + // token-level mathematics parser deliberately leaves it to Korean rules. + // A second explicit operator disambiguates the inner U+002D from a range. + let has_other_math_operator = ctx.word_chars.iter().enumerate().any(|(index, ch)| { + index != ctx.index + && matches!( + ch, + '+' | '=' | '×' | '÷' | '<' | '>' | '≤' | '≥' | '≠' | '−' + ) + }); + let prev_ends_operand = ctx.prev_char().is_some_and(|prev| { + prev.is_alphanumeric() + || utils::is_korean_char(prev) + || matches!(prev, ')' | ']' | '}' | '〉' | '》' | '」' | '』' | '】') + }); + + prev_ends_operand && next_starts_number && has_other_math_operator +} + impl BrailleRule for RuleMath { fn meta(&self) -> &'static RuleMeta { &META @@ -34,52 +280,51 @@ impl BrailleRule for RuleMath { fn matches(&self, ctx: &RuleContext) -> bool { matches!(ctx.char_type, CharType::MathSymbol(_)) + || (matches!(ctx.char_type, CharType::Symbol('-')) && is_semantic_ascii_minus(ctx)) } fn apply(&self, ctx: &mut RuleContext) -> Result { - let CharType::MathSymbol(c) = ctx.char_type else { - return Ok(RuleResult::Skip); + let c = match ctx.char_type { + CharType::MathSymbol(c) => *c, + CharType::Symbol('-') if is_semantic_ascii_minus(ctx) => '\u{2212}', + _ => return Ok(RuleResult::Skip), }; - // PDF 제46항 — 사칙연산 기호(+, −, ×, ÷, =) 띄어쓰기 규칙. - // 좌·우가 모두 "한글이 포함된 식"일 때에만 기호 앞뒤를 한 칸씩 띄어 쓴다. + // UEB §3.17 + Korean rules 29/35: a plus that belongs to a Roman + // product/grade identifier stays inside that Roman section and uses + // the UEB general-symbol cells ⠐⠖. The token-level grammar has already + // rejected completed sums and the ambiguous one-letter `A+` shape. + if c == '+' + && ctx.state.english_indicator + && ctx.state.is_english + && crate::rules::token_rules::math_expression::is_roman_plus_identifier(ctx.word_chars) + { + let encoded = crate::rules::english_ueb::rule_3::encode_symbol(c) + .ok_or_else(|| "UEB plus sign must be defined".to_string())?; + ctx.emit_slice(&encoded); + return Ok(RuleResult::Consumed); + } + + // PDF 제46항 — 사칙연산 기호(+, −, ×, ÷, =)가 한글 사이에 + // 나올 때에만 기호 앞뒤를 한 칸씩 띄어 쓴다. // // 판정: - // - 좌측 segment: 단어 시작부터 현재 기호 직전까지의 chars. 한글 포함 여부. - // - 우측 segment: 현재 기호 직후부터 단어 끝까지의 chars 중 **선행 비한글을 건너뛴 - // 첫 한글 묶음**. (예: `3.14이다` → `이다`; `3개=2개` → `개`) - // - 우측 묶음이 비어 있거나 JOSA(조사: 과/와/이다/하고/이랑/랑/아니다 등)이면 + // - 바로 인접한 피연산자 범위 안에 한글이 각각 있어야 한다. + // - 괄호 속 로마자·한글 주석은 그 피연산자에 포함한다. + // 예: `레트로(RETRO)+뉴트로(NEWTRO)`. + // - 괄호 경계나 다른 연산 기호를 넘어 문법적으로 무관한 한글은 + // 찾지 않는다. 예: `기업(+5p)의`, `행사(1+1)이다`. + // - 우측 한글 묶음이 비어 있거나 조사·서술격 표현(과/와/의/이다 등)이면 // 기호 양쪽을 띄어쓰지 않는다. // 예: `반지름×3.14이다` → `이다`는 JOSA → 띄어쓰지 않음. // 예: `5개−3개=2개` → `개`는 JOSA가 아님 → 띄어씀. - let prev_has_korean = ctx.word_chars[..ctx.index] - .iter() - .any(|c| utils::is_korean_char(*c)); - - let next_korean_is_non_josa = { - let mut korean = Vec::new(); - for wc in &ctx.word_chars[ctx.index + 1..] { - if utils::is_korean_char(*wc) { - korean.push(*wc); - } else if !korean.is_empty() { - break; - } - } - if korean.is_empty() { - false - } else { - let s: String = korean.into_iter().collect(); - !JOSA.contains(&s.as_str()) - } - }; - - let pad_spaces = prev_has_korean && next_korean_is_non_josa; + let pad_spaces = rule_46_requires_padding(ctx); if pad_spaces { ctx.emit(0); } - let encoded = math_symbol_shortcut::encode_char_math_symbol_shortcut(*c)?; + let encoded = math_symbol_shortcut::encode_char_math_symbol_shortcut(c)?; ctx.emit_slice(encoded); if pad_spaces { @@ -94,6 +339,24 @@ impl BrailleRule for RuleMath { mod tests { use super::*; + #[rstest::rstest] + #[case::parenthesis('(', ')')] + #[case::square_bracket('[', ']')] + #[case::curly_brace('{', '}')] + #[case::single_angle('〈', '〉')] + #[case::double_angle('《', '》')] + #[case::corner_bracket('「', '」')] + #[case::white_corner_bracket('『', '』')] + #[case::lenticular_bracket('【', '】')] + #[case::tortoise_shell_bracket('〔', '〕')] + #[case::white_lenticular_bracket('〖', '〗')] + #[case::white_tortoise_shell_bracket('〘', '〙')] + #[case::white_square_bracket('〚', '〛')] + fn delimiter_pairs_are_bidirectional(#[case] opening: char, #[case] closing: char) { + assert_eq!(matching_opening_delimiter(closing), Some(opening)); + assert_eq!(matching_closing_delimiter(opening), Some(closing)); + } + #[test] fn apply_exercise() { let mut owned = crate::test_helpers::CtxOwned::for_text("A", false); @@ -120,4 +383,135 @@ mod tests { assert!(owned.result.starts_with(&[0])); assert!(owned.result.ends_with(&[0])); } + + #[rstest::rstest] + #[case::plus('+')] + #[case::times('×')] + #[case::division('÷')] + #[case::equals('=')] + fn parenthesized_math_symbol_does_not_gain_inner_spaces(#[case] operator: char) { + let input = format!("가({operator})나"); + let mut owned = crate::test_helpers::CtxOwned::for_text(&input, false); + let mut ctx = owned.ctx_at(2); + + let outcome = RuleMath.apply(&mut ctx).expect("math rule should apply"); + + assert!(matches!(outcome, RuleResult::Consumed)); + assert!(!owned.result.is_empty()); + assert_ne!(owned.result.first(), Some(&0)); + assert_ne!(owned.result.last(), Some(&0)); + } + + #[rstest::rstest] + #[case::service("TV+")] + #[case::alphanumeric_product("HDR10+")] + #[case::mixed_case_service("U+tv")] + fn roman_terminal_plus_uses_ueb_general_symbol(#[case] identifier: &str) { + let output = crate::encode(&format!("가 {identifier} 나")) + .expect("Roman product identifier must encode"); + let ueb_plus = crate::rules::english_ueb::rule_3::encode_symbol('+') + .expect("UEB plus must be defined"); + + assert!( + output + .windows(ueb_plus.len()) + .any(|cells| cells == ueb_plus), + "identifier={identifier}" + ); + } + + #[rstest::rstest] + #[case::plus_math_symbol("양", "+", "극")] + #[case::ascii_hyphen_minus_symbol("음", "-", "극")] + fn full_encoder_preserves_tight_parenthesized_operator( + #[case] left: &str, + #[case] operator: &str, + #[case] right: &str, + ) { + let input = format!("{left}({operator}){right}"); + let expected = [left, &format!("({operator})"), right] + .into_iter() + .map(|part| crate::encode_to_unicode(part).expect("component must encode")) + .collect::>() + .concat(); + + assert_eq!( + crate::encode_to_unicode(&input).expect("full input must encode"), + expected + ); + } + + #[rstest::rstest] + #[case::signed_integer("-3", 0, true)] + #[case::parenthesized_signed_decimal("(-3.5)", 1, true)] + #[case::quoted_negative_quantity("‘-2배’", 1, true)] + #[case::pdf_phone_number("02-799-1000", 2, false)] + #[case::identifier_suffix("A-3", 1, false)] + #[case::calendar_date("2024-09-03", 4, false)] + #[case::non_hyphen_character("A", 0, false)] + fn ascii_hyphen_minus_is_disambiguated_by_syntax( + #[case] input: &str, + #[case] index: usize, + #[case] expected: bool, + ) { + let mut owned = crate::test_helpers::CtxOwned::for_text(input, false); + let ctx = owned.ctx_at(index); + assert_eq!(is_semantic_ascii_minus(&ctx), expected, "input={input}"); + } + + #[rstest::rstest] + #[case::korean_words("나루+배", 2, true)] + #[case::korean_numeric_units("5개-3개", 2, true)] + #[case::percentage_noun_operand("팬=51%지분", 1, true)] + #[case::roman_annotation_on_left("레트로(RETRO)+뉴트로", 10, true)] + #[case::korean_annotations_on_both_sides("AI(인공지능)+DX(디지털전환)", 8, true)] + #[case::mixed_script_right_operand("밀레니얼+Z세대", 4, true)] + #[case::signed_parenthetical("기업(+5p)의", 3, false)] + #[case::numeric_sum("행사(1+1)이다", 4, false)] + #[case::brand_particle("디즈니+와", 3, false)] + #[case::roman_variable_left("T+3일", 1, false)] + #[case::particle_after_annotated_roman("OPEC(석유수출국기구)+의", 13, false)] + #[case::quoted_suffix("저소음+’", 3, false)] + fn rule_46_padding_depends_on_actual_operands( + #[case] input: &str, + #[case] index: usize, + #[case] expected: bool, + ) { + let mut owned = crate::test_helpers::CtxOwned::for_text(input, false); + let ctx = owned.ctx_at(index); + assert_eq!(rule_46_requires_padding(&ctx), expected, "input={input}"); + } + + #[rstest::rstest] + #[case::negative_percentage("-2.73%를", "−2.73%를")] + #[case::negative_unit("체급(-67kg)은", "체급(−67kg)은")] + #[case::parenthesized_polarity("음(-)극", "음(−)극")] + #[case::article_46_equation("5개-3개=2개", "5개−3개=2개")] + fn semantic_ascii_minus_matches_explicit_unicode_minus( + #[case] ascii: &str, + #[case] explicit: &str, + ) { + assert!(matches!( + crate::char_struct::CharType::new('-').expect("hyphen-minus must classify"), + crate::char_struct::CharType::Symbol('-') + )); + assert_eq!( + crate::encode_to_unicode(ascii).expect("ASCII expression must encode"), + crate::encode_to_unicode(explicit).expect("Unicode expression must encode"), + "input={ascii}" + ); + } + + #[rstest::rstest] + #[case::pdf_phone_number("02-799-1000")] + #[case::identifier_suffix("A-3")] + #[case::calendar_date("2024-09-03")] + fn non_operator_hyphens_do_not_become_minus(#[case] input: &str) { + let explicit_minus = input.replacen('-', "−", 1); + assert_ne!( + crate::encode_to_unicode(input).expect("hyphenated input must encode"), + crate::encode_to_unicode(&explicit_minus).expect("minus variant must encode"), + "input={input}" + ); + } } diff --git a/libs/braillify/src/rules/math/encoder/symbol_rule.rs b/libs/braillify/src/rules/math/encoder/symbol_rule.rs index a3f694e8..15a738a6 100644 --- a/libs/braillify/src/rules/math/encoder/symbol_rule.rs +++ b/libs/braillify/src/rules/math/encoder/symbol_rule.rs @@ -76,20 +76,6 @@ impl MathTokenRule for MathSymbolRule { let _ = rule_26::is_reserved_rule_26(); let _ = rule_22::NTH_ROOT_INDEX_MARKER; - let prev_is_variable_or_upper = matches!( - rule_12::prev_non_space(tokens, index), - Some(MathToken::Variable(_) | MathToken::UpperVariable(_)) - ); - let next_is_upper = matches!( - Self::next_non_space(tokens, index + 1), - Some(MathToken::UpperVariable(_)) - ); - if *c == '\u{00AC}' && index > 0 && prev_is_variable_or_upper && next_is_upper { - result.push(40); - state.prev_was_number = false; - return Ok(MathTokenResult::Consumed(1)); - } - if *c == '\u{FF03}' && matches!( Self::next_non_space(tokens, index + 1), @@ -482,25 +468,31 @@ mod tests { // ---------------- Specialised prefix arms ---------------- - /// `A¬B` — ¬ (U+00AC) sandwiched between two UpperVariables hits the - /// negation prefix arm at lines 59-73. Encoded byte 40 is pushed. + /// Math rule 61: a negation sign keeps its complete two-cell mapping + /// between uppercase variables. #[test] fn negation_between_upper_variables() { let result = enc("A\u{00AC}B"); - assert!(!result.is_empty(), "A¬B must encode"); - // Compare against pattern WITHOUT the matching neighbours to ensure - // a different code path was taken. - let other = enc("\u{00AC}B"); - assert_ne!(result, other, "A¬B (sandwiched) must differ from ¬B"); + let negation = crate::math_symbol_shortcut::encode_char_math_symbol_shortcut('\u{00AC}') + .expect("rule-61 negation must be mapped"); + assert!( + result + .windows(negation.len()) + .any(|cells| cells == negation) + ); } - /// `A¬ B` with a leading lower variable instead of upper still triggers - /// the prev=Variable arm of the match (line 63). + /// The same rule applies between a lowercase and uppercase variable. #[test] fn negation_between_lower_and_upper_variable() { - // `a¬B` — prev is Variable('a'), next is UpperVariable('B'). let result = enc("a\u{00AC}B"); - assert!(!result.is_empty(), "a¬B must encode"); + let negation = crate::math_symbol_shortcut::encode_char_math_symbol_shortcut('\u{00AC}') + .expect("rule-61 negation must be mapped"); + assert!( + result + .windows(negation.len()) + .any(|cells| cells == negation) + ); } /// `#B` — FF03 fullwidth hash + UpperVariable hits lines 75-96. diff --git a/libs/braillify/src/rules/roman_mode.rs b/libs/braillify/src/rules/roman_mode.rs index fc66e7a5..ed6ef64c 100644 --- a/libs/braillify/src/rules/roman_mode.rs +++ b/libs/braillify/src/rules/roman_mode.rs @@ -16,6 +16,9 @@ pub(crate) fn exit_english(state: &mut EncoderState, needs_continuation: bool) { state.is_english = false; state.needs_english_continuation = needs_continuation; state.roman_number_chain = false; + if !needs_continuation { + state.roman_section_is_english_context = false; + } } /// 영어 모드로 진입하며 로마자표 ⠴ (또는 직전 종료 후 연속표 ⠰)를 emit한다. @@ -32,7 +35,8 @@ pub(crate) fn enter_english(state: &mut EncoderState, result: &mut Vec) { /// 제35항 — 로마자+숫자 연결(`D-100` 등)을 위해 영어 모드를 잠시 내려놓는다. pub(crate) fn exit_english_for_roman_number_chain(state: &mut EncoderState) { - exit_english(state, false); + state.is_english = false; + state.needs_english_continuation = false; state.roman_number_chain = true; } diff --git a/libs/braillify/src/rules/token.rs b/libs/braillify/src/rules/token.rs index 7c9e84b0..880311f8 100644 --- a/libs/braillify/src/rules/token.rs +++ b/libs/braillify/src/rules/token.rs @@ -37,11 +37,24 @@ impl WordMeta { .iter() .any(|ch| (0xAC00..=0xD7A3).contains(&(*ch as u32))); let ascii_letter_count = chars.iter().filter(|ch| ch.is_ascii_alphabetic()).count(); - let uppercase_count = chars.iter().filter(|ch| ch.is_ascii_uppercase()).count(); let has_ascii_alphabetic = ascii_letter_count > 0; let starts_with_ascii = chars.first().is_some_and(char::is_ascii_alphabetic); - let is_all_uppercase = ascii_letter_count >= 2 && ascii_letter_count == uppercase_count; + // UEB 8.4.1-8.4.2 applies the capitals-word indicator to the initial + // uppercase letters-sequence and terminates it at the first nonletter. + // Korean rule 35's `MP3` therefore still pre-emits capitals-word mode + // for `MP`, while UEB's `B&B` does not do so for its one-letter prefix. + // Mixed-case forms stay on the span path, which can emit the required + // capitals terminator before a lowercase continuation (`TVOntario`). + let initial_uppercase_count = chars + .iter() + .take_while(|ch| ch.is_ascii_uppercase()) + .count(); + let all_ascii_letters_uppercase = chars + .iter() + .filter(|ch| ch.is_ascii_alphabetic()) + .all(|ch| ch.is_ascii_uppercase()); + let is_all_uppercase = initial_uppercase_count >= 2 && all_ascii_letters_uppercase; WordMeta { has_korean, @@ -266,6 +279,22 @@ mod tests { assert!(meta.is_all_uppercase); } + /// UEB 8.4.2 terminates capitals word mode at a nonalphabetic symbol. + /// The official UEB 3.1.1/8.4 and Korean rule-35 examples distinguish a + /// multi-letter initial sequence from a one-letter initial sequence. + #[rstest::rstest] + #[case::official_at_and_t("AT&T", true)] + #[case::official_b_and_b("B&B", false)] + #[case::official_mixed_case("TVOntario", false)] + #[case::rule35_mp3("MP3", true)] + fn word_meta_scopes_capitals_word_to_initial_letters_sequence( + #[case] input: &str, + #[case] expected: bool, + ) { + let chars = input.chars().collect::>(); + assert_eq!(WordMeta::from_chars(&chars).is_all_uppercase, expected); + } + #[test] fn word_meta_mixed() { let chars: Vec = "A한b".chars().collect(); diff --git a/libs/braillify/src/rules/token_rules/english_dominant_korean_wrap.rs b/libs/braillify/src/rules/token_rules/english_dominant_korean_wrap.rs index b43c55cc..ad76baa5 100644 --- a/libs/braillify/src/rules/token_rules/english_dominant_korean_wrap.rs +++ b/libs/braillify/src/rules/token_rules/english_dominant_korean_wrap.rs @@ -79,6 +79,13 @@ fn is_punct_only(chars: &[char]) -> bool { .all(|c| !c.is_ascii_alphabetic() && !is_korean_char(*c) && !c.is_ascii_digit()) } +fn same_token_rule39_context_allowed( + is_english_majority: bool, + dot_delimited_domain_label: bool, +) -> bool { + is_english_majority || dot_delimited_domain_label +} + /// 같은 토큰 내에서 좌측을 거슬러 처음 만나는 letter가 ASCII 영문인지. /// 한글을 먼저 만나거나, 영문도 한글도 없으면 false. fn same_token_left_is_english(left_chars: &[char]) -> bool { @@ -267,6 +274,27 @@ fn count_script_words(tokens: &[Token<'_>]) -> (usize, usize) { (english_words, korean_words) } +/// Rule 39 applies to a Roman-main sentence, not merely a Korean sentence that +/// contains a long Roman citation. The first lexical script is a stable matrix- +/// language signal: the PDF's Roman-main examples begin in Roman script, while +/// its Korean domain-name example is handled by the same-token domain rule. +fn document_begins_in_roman_script(tokens: &[Token<'_>]) -> bool { + tokens.iter().find_map(|token| { + let Token::Word(word) = token else { + return None; + }; + word.chars.iter().find_map(|ch| { + if ch.is_ascii_alphabetic() { + Some(true) + } else if is_korean_char(*ch) { + Some(false) + } else { + None + } + }) + }) == Some(true) +} + /// Compute all document-level English-Korean predicates once per encode call. pub fn compute_document_summary(tokens: &[Token<'_>]) -> DocumentSummary { let candidates = scan_english_context_candidates(tokens); @@ -275,9 +303,11 @@ pub fn compute_document_summary(tokens: &[Token<'_>]) -> DocumentSummary { } let (english_words, korean_words) = count_script_words(tokens); - let is_english_majority = english_words >= korean_words.max(1); + let is_roman_main = + document_begins_in_roman_script(tokens) && english_words >= korean_words.max(1); + let is_english_majority = is_roman_main; let is_english_dominant = - english_words >= 10 && english_words >= korean_words.saturating_mul(5); + is_roman_main && english_words >= 10 && english_words >= korean_words.saturating_mul(5); let has_english_context_for_korean = candidates.has_same_token_context || (candidates.has_boundary_candidate && is_english_majority); @@ -295,8 +325,9 @@ pub fn compute_document_summary(tokens: &[Token<'_>]) -> DocumentSummary { /// 예: "김치", "반찬)". 인접 word token이 모두 영어이면서 _문서 전체가 영어 다수_ /// 일 때만 wrap. (한글 주도 문장에 영어가 끼인 경우는 wrap 대상 아님.) /// 2. **양쪽 토큰 내부** — segment의 양쪽이 같은 토큰 내 영어 letter로 둘러싸였다. -/// 예: "www.대통령.kr"의 "대통령". 양쪽이 영어 letter이면 wrap. -/// (단일 단어 내부 패턴은 문서 비율과 무관하게 항상 적용한다.) +/// 문서가 영어 다수이면 wrap한다. 한국어 주도 문장에서는 제39항을 임의로 +/// 확장하지 않고, 공식 예제 `www.대통령.kr`처럼 양쪽이 점으로 구분된 도메인 +/// label만 구조적으로 보존한다. /// /// 두 케이스가 _혼합_된 경우(한쪽은 token boundary, 다른 쪽은 same-token letter)는 /// 영어 어절 + 한국어 조사/어미 결합(예: "be는")일 가능성이 높으므로 wrap하지 않는다. @@ -317,7 +348,12 @@ fn segment_in_english_context_with_majority<'a>( return boundary_segment_wrap(tokens, token_index, is_english_majority); } if !left_at_boundary && !right_at_boundary { - return same_token_left_is_english(left_slice) && same_token_right_is_english(right_slice); + let has_roman_on_both_sides = + same_token_left_is_english(left_slice) && same_token_right_is_english(right_slice); + let dot_delimited_domain_label = + left_slice.last() == Some(&'.') && right_slice.first() == Some(&'.'); + return has_roman_on_both_sides + && same_token_rule39_context_allowed(is_english_majority, dot_delimited_domain_label); } false } @@ -510,6 +546,33 @@ mod tests { // The "123" word's first_script_char Some('1') hits `_ => {}` (not counted). } + #[rstest::rstest] + #[case::roman_main("2024 What is 김치 in English?", true)] + #[case::korean_main_with_long_roman_citation( + "익수다의 주요 ADC 프로그램은 IKS012(Anti-Folate Receptor Alpha (FRa)) ADC와 함께한다.", + false + )] + #[case::korean_domain_exception("대통령실 주소는 www.대통령.kr이다.", false)] + fn detects_the_matrix_script_from_the_first_lexical_script( + #[case] input: &str, + #[case] expected: bool, + ) { + let ir = crate::rules::token::DocumentIR::parse(input, true); + assert_eq!(document_begins_in_roman_script(&ir.tokens), expected); + } + + #[test] + fn korean_main_sentence_does_not_become_rule_39_from_a_long_roman_citation() { + let input = + "익수다의 주요 ADC 프로그램은 IKS012(Anti-Folate Receptor Alpha (FRa)) ADC와 함께한다."; + let ir = crate::rules::token::DocumentIR::parse(input, true); + + let summary = compute_document_summary(&ir.tokens); + + assert!(!summary.is_english_majority); + assert!(!summary.is_english_dominant); + } + /// english_dominant_korean_wrap:311 — `(true, true) =>` arm of the boundary /// match. Korean segment fills the whole word (both slices empty/punct-only), /// AND prev/next tokens are English-only words. is_english_majority required true. @@ -534,22 +597,47 @@ mod tests { ); } - /// english_dominant_korean_wrap:316 — `(false, false) =>` arm of the boundary - /// match. Korean segment is sandwiched within same-token English letters. - /// Both `left_at_boundary` and `right_at_boundary` are false because the - /// surrounding chars include English letters. + /// Rule 39's same-token branch still requires a Roman-main document, except + /// for the PDF's dot-delimited `www.대통령.kr` domain structure. #[test] - fn segment_within_same_token_english_letters() { - // "www.대통령.kr" — Korean chars '대통령' surrounded by 'w'/'k' letters - // (separated by '.'). same_token_*_is_english returns true on both sides. + fn pdf_domain_label_wraps_without_english_majority() { let token = word("www.대통령.kr"); let tokens = vec![token.clone()]; let kor_word = unwrap_word(&tokens[0]); let result = build_wrapped_replacement(kor_word, &tokens, 0, false); - // Inner same-token English context should wrap regardless of majority. + assert!(result.is_some()); + } + + #[rstest::rstest] + #[case::english_majority(true, false, true)] + #[case::dot_delimited_domain(false, true, true)] + #[case::neither(false, false, false)] + fn same_token_rule39_gate_requires_dominance_or_domain( + #[case] is_english_majority: bool, + #[case] dot_delimited_domain_label: bool, + #[case] expected: bool, + ) { + assert_eq!( + same_token_rule39_context_allowed(is_english_majority, dot_delimited_domain_label), + expected + ); + } + + #[rstest::rstest] + #[case::question("What is 김치 in English?")] + #[case::domain("대통령실의 누리집 주소는 www.대통령.kr이다.")] + #[case::definition( + "Banchan (Korean: 반찬) are small side dishes served along with cooked rice in Korean cuisine." + )] + fn full_encoder_preserves_rule39_pdf_controls(#[case] input: &str) { + let actual = crate::encode_to_unicode(input).expect("rule 39 PDF example must encode"); assert!( - result.is_some(), - "Korean segment within same-token English letters should wrap" + actual.contains("⠸⠷"), + "missing Korean opening marker: {actual}" + ); + assert!( + actual.contains("⠸⠾"), + "missing Korean closing marker: {actual}" ); } diff --git a/libs/braillify/src/rules/token_rules/latex_math.rs b/libs/braillify/src/rules/token_rules/latex_math.rs index e9dc73b9..11fd49b0 100644 --- a/libs/braillify/src/rules/token_rules/latex_math.rs +++ b/libs/braillify/src/rules/token_rules/latex_math.rs @@ -120,6 +120,26 @@ mod tests { assert!(result.contains('\u{00B2}')); } + /// 수학 제6항의 원 둘레 공식: Korean grouped operands and repeated + /// `\\times` commands must survive LaTeX normalization as one expression. + #[test] + fn korean_grouped_operands_with_repeated_latex_times_encode() { + let inner = "(원의 둘레)=(반지름)\\times 2\\times 3.14"; + assert_eq!(strip_latex_to_math(inner), "(원의 둘레)=(반지름)×2×3.14"); + + assert!(encode_latex_math_bytes_with_context(inner, MathContext::default()).is_ok()); + assert!(crate::encode(&format!("${inner}$")).is_ok()); + assert!( + crate::encode_with_options( + &format!("${inner}$"), + &crate::EncodeOptions { + default_mode: Some(crate::rules::context::EncodingMode::Math), + }, + ) + .is_ok() + ); + } + #[test] fn test_strip_subscript() { let result = strip_latex_to_math("x_{2}"); diff --git a/libs/braillify/src/rules/token_rules/math_expression.rs b/libs/braillify/src/rules/token_rules/math_expression.rs index 9fe4a38f..2751ab75 100644 --- a/libs/braillify/src/rules/token_rules/math_expression.rs +++ b/libs/braillify/src/rules/token_rules/math_expression.rs @@ -33,8 +33,23 @@ impl TokenRule for MathExpressionTokenRule { } } +/// Shared character-emission predicate for a Roman identifier whose `+` must +/// use the UEB general-symbol cells rather than the Korean math plus cell. +pub(crate) fn is_roman_plus_identifier(chars: &[char]) -> bool { + apply::is_korean_prose_roman_plus_identifier(chars) + || apply::has_korean_prefix_roman_plus_annotation(chars) + || apply::has_korean_prefix_terminal_roman_plus_identifier(chars) +} + #[cfg(test)] mod tests { + use super::apply::{ + has_korean_prefix_roman_hyphen_suffix, has_korean_prefix_roman_plus_annotation, + has_korean_prefix_terminal_roman_plus_identifier, is_korean_prose_acronym_parenthetical, + is_korean_prose_roman_hyphen_identifier, is_korean_prose_roman_number_identifier, + is_korean_prose_roman_plus_identifier, is_korean_prose_roman_slash_identifier, + is_korean_prose_single_letter_slash_phrase, + }; use super::detect::is_math_expression; use super::helpers::*; use super::*; @@ -72,6 +87,296 @@ mod tests { assert!(!is_math_expression(&chars, "hello")); } + /// Korean rules 28/29/34/35: Roman code/compound surfaces must remain on + /// the Roman path in Korean prose, while lowercase algebra and one-letter + /// subtraction stay math-owned. + #[rstest::rstest] + #[case::official_roman_number("D-100", true)] + #[case::capital_code("AB-12", true)] + #[case::capitalised_compound("Title-Case", true)] + #[case::enclosed_code("(ABC)-D", true)] + #[case::single_capital_lexical_prefix("K-pop", true)] + #[case::single_capital_common_term("X-ray", true)] + #[case::single_capital_brand_prefix("K-water", true)] + #[case::single_lowercase_brand_prefix("k-water", true)] + #[case::mixed_case_digit_code("pH-1", true)] + #[case::decimal_model_code("GPT-3.5", true)] + #[case::lowercase_algebra("x-1", false)] + #[case::uppercase_subtraction("A-B", false)] + #[case::function_expression("F(x-1)", false)] + fn korean_prose_hyphen_identifier_grammar(#[case] input: &str, #[case] expected: bool) { + assert_eq!( + is_korean_prose_roman_hyphen_identifier(&input.chars().collect::>()), + expected + ); + } + + /// Korean rules 29/35 keep a Roman-led name and its adjoining number in + /// one Roman section. A one-letter algebraic variable remains math-owned. + #[rstest::rstest] + #[case::media_generation("Web3.0", true)] + #[case::model_version("GPT3.5", true)] + #[case::audio_format("MP3", true)] + #[case::mixed_case_measure("pH7", true)] + #[case::single_letter_variable("x2", false)] + #[case::number_first("3ab", false)] + #[case::plain_decimal("3.14", false)] + #[case::separator_not_between_digits("Web.3", false)] + fn korean_prose_roman_number_identifier_grammar(#[case] input: &str, #[case] expected: bool) { + assert_eq!( + is_korean_prose_roman_number_identifier(&input.chars().collect::>()), + expected + ); + } + + #[rstest::rstest] + #[case::enclosed_code("한글(ABC)-D", true)] + #[case::roman_gloss_then_code("한글(Title)-AB", true)] + #[case::korean_then_single_capital("하쿠토-R", true)] + #[case::korean_then_initialism("기장-KBO", true)] + #[case::korean_then_lowercase_word("온다-life", true)] + #[case::korean_then_alphanumeric_label("대신-Y2HC", true)] + #[case::lowercase_algebra("한글(x-1)", false)] + #[case::uppercase_subtraction("한글(A-B)", false)] + #[case::korean_then_lowercase_variable("값-x", false)] + #[case::korean_then_explicit_expression("값-X+1", false)] + #[case::korean_then_number("한-3", false)] + fn korean_prefix_roman_hyphen_suffix_grammar(#[case] input: &str, #[case] expected: bool) { + assert_eq!( + has_korean_prefix_roman_hyphen_suffix(&input.chars().collect::>()), + expected + ); + } + + /// Korean rules 29 and 33: a Korean/Roman hyphen boundary stays attached, + /// uses the Korean hyphen cell, and opens one Roman section after it. + #[rstest::rstest] + #[case::single_capital("하쿠토-R", "⠚⠋⠍⠓⠥⠤⠴⠠⠗⠲")] + #[case::initialism("기장-KBO", "⠈⠕⠨⠶⠤⠴⠠⠠⠅⠃⠕⠲")] + #[case::country_initialism("한-UAE", "⠚⠒⠤⠴⠠⠠⠥⠁⠑⠲")] + fn korean_to_roman_hyphen_boundary_stays_prose(#[case] input: &str, #[case] expected: &str) { + assert_eq!(crate::encode_to_unicode(input).as_deref(), Ok(expected)); + } + + #[rstest::rstest] + #[case::standards_bodies("ISO/IEC", true)] + #[case::market_pair("WEMIX/KRW", true)] + #[case::aircraft_family("F-5E/F", true)] + #[case::roman_model_family("NVMe/TCP", true)] + #[case::single_letter_fraction("F/N", false)] + #[case::algebraic_fraction("A/B", false)] + #[case::numeric_fraction("1/2", false)] + #[case::equation_context("X≈F/N", false)] + fn korean_prose_slash_identifier_grammar(#[case] input: &str, #[case] expected: bool) { + assert_eq!( + is_korean_prose_roman_slash_identifier(&input.chars().collect::>()), + expected + ); + } + + #[rstest::rstest] + #[case::hardware_wallet("H/W Wallet", true)] + #[case::relapsed_refractory_cancer("폐암(R/R ES-SCLC)에서", true)] + #[case::lowercase_math_description("F/N ratio", false)] + #[case::explicit_equation("X≈F/N Result", false)] + #[case::isolated_fraction("F/N", false)] + fn single_letter_slash_phrase_requires_a_capital_led_roman_continuation( + #[case] input: &str, + #[case] expected: bool, + ) { + let ir = crate::rules::token::DocumentIR::parse(input, true); + let index = ir + .tokens + .iter() + .position(|token| matches!(token, Token::Word(word) if word.chars.contains(&'/'))) + .expect("probe must contain a slash word"); + let Token::Word(word) = &ir.tokens[index] else { + unreachable!("selected token must be a word"); + }; + + assert_eq!( + is_korean_prose_single_letter_slash_phrase(&ir.tokens, index, &word.chars), + expected + ); + } + + #[rstest::rstest] + #[case::hardware_wallet("가 H/W Wallet 나", "⠫⠀⠴⠠⠓⠸⠌⠠⠺⠀⠠⠺⠁⠇⠇⠑⠞⠲⠀⠉")] + #[case::relapsed_refractory_cancer("가 R/R ES-SCLC 나", "⠫⠀⠴⠠⠗⠸⠌⠠⠗⠀⠠⠠⠑⠎⠤⠠⠠⠎⠉⠇⠉⠲⠀⠉")] + #[case::official_math_rule_29("X ≈ F/N", "⠠⠭⠀⠈⠔⠈⠔⠀⠠⠋⠸⠌⠠⠝")] + fn slash_phrase_respects_korean_rule_29_without_stealing_math_rule_29( + #[case] input: &str, + #[case] expected: &str, + ) { + assert_eq!(crate::encode_to_unicode(input).as_deref(), Ok(expected)); + } + + #[rstest::rstest] + #[case::service("TV+", true)] + #[case::safety_grade("TSP+", true)] + #[case::alphanumeric_product("HDR10+", true)] + #[case::numeric_parenthetical("ATC+(20017936)", true)] + #[case::mixed_case_service("U+tv", true)] + #[case::attached_korean_particle("XYZ+는", true)] + #[case::number_led_identifier("24K+", true)] + #[case::identifier_separator("Model.Name+", true)] + #[case::repeated_terminal_plus("UV++++", true)] + #[case::contextual_single_letter_grade("A+(우수)", true)] + #[case::ascii_single_letter_expression("A+(B)", false)] + #[case::one_letter_terminal_identifier("A+", true)] + #[case::completed_sum("AB+C", false)] + #[case::chemical_expression("SmBa0.5-xCo2O5+d", false)] + #[case::lexical_compound("Dog+Yoga", true)] + #[case::lowercase_math_functions("sin+cos", false)] + #[case::korean_service("U+유모바일", true)] + #[case::mixed_script_korean_service("U+한글tv", true)] + #[case::one_letter_korean_addition("A+나", true)] + fn korean_prose_plus_identifier_grammar(#[case] input: &str, #[case] expected: bool) { + assert_eq!( + is_korean_prose_roman_plus_identifier(&input.chars().collect::>()), + expected + ); + } + + #[rstest::rstest] + #[case::closed_lexical_gloss("도가(Dog+Yoga)", true)] + #[case::attached_particle("워케이션(Work+Vacation)은", true)] + #[case::single_letter_terminal_label("등급(A+)은", true)] + #[case::middle_dot_chained_identifier("상품(Service+)·후속(Next+)는", true)] + #[case::math_body("공식(A+B)은", false)] + #[case::unclosed("도가(Dog+Yoga", false)] + fn korean_prefix_plus_annotation_grammar(#[case] input: &str, #[case] expected: bool) { + assert_eq!( + has_korean_prefix_roman_plus_annotation(&input.chars().collect::>()), + expected + ); + } + + #[rstest::rstest] + #[case::roman_suffix("한글TV+는", true)] + #[case::numeric_roman_suffix("한글7GB+는", true)] + #[case::completed_sum("한글A+B는", false)] + #[case::all_capital_internal_ambiguity("한글X+U는", false)] + fn korean_prefix_terminal_plus_suffix_grammar(#[case] input: &str, #[case] expected: bool) { + assert_eq!( + has_korean_prefix_terminal_roman_plus_identifier(&input.chars().collect::>()), + expected + ); + } + + #[rstest::rstest] + #[case::acronym_expansion("ABC(Alpha", true)] + #[case::alphanumeric_acronym("S2E(System)", true)] + #[case::single_math_function("f(x)", false)] + #[case::operator_body("AB(x+1)", false)] + fn korean_prose_acronym_parenthetical_grammar(#[case] input: &str, #[case] expected: bool) { + assert_eq!( + is_korean_prose_acronym_parenthetical(&input.chars().collect::>()), + expected + ); + } + + /// Korean rules 29, 35 and 54 compose the same anonymized-person label + /// regardless of whether the following honorific is attached or spaced. + #[rstest::rstest] + #[case::adult("A(27)", "씨는")] + #[case::minor_male("B(11)", "군에게")] + #[case::minor_female("C(16)", "양의")] + #[case::elected_official("A(31)", "도의원을")] + #[case::professor("B(61)", "교수를")] + #[case::judge("C(54)", "부장판사에게")] + fn spaced_anonymized_person_label_uses_korean_prose_composition( + #[case] label: &str, + #[case] honorific: &str, + ) { + let input = format!("{label} {honorific}"); + let mut expected = encode_anonymized_person_label(&label.chars().collect::>()) + .expect("valid anonymized-person label"); + expected.push(0); + expected.extend(crate::encode(honorific).expect("Korean honorific must encode")); + + assert_eq!( + crate::encode(&input).expect("prose label must encode"), + expected + ); + } + + #[test] + fn spaced_function_value_is_not_an_anonymized_person_label() { + let input = "A(14) 값"; + let tokens = crate::rules::token::DocumentIR::parse(input, true).tokens; + assert!(!super::apply::next_word_begins_korean_prose_label_context( + &tokens, 0 + )); + } + + #[rstest::rstest] + #[case::lower_list_item("(x)", false)] + #[case::upper_list_item("(A)", false)] + #[case::plain_parenthesized_word("(abc)", false)] + fn standalone_parenthesized_inputs_keep_baseline_detector_result( + #[case] input: &str, + #[case] expected: bool, + ) { + let chars: Vec = input.chars().collect(); + assert_eq!(is_math_expression(&chars, input), expected, "input={input}"); + } + + #[rstest::rstest] + #[case::addition("(x+1)")] + #[case::fraction("(a/b)")] + #[case::subscript("(x₁)")] + fn parenthesized_explicit_expressions_keep_existing_math_result(#[case] input: &str) { + let chars: Vec = input.chars().collect(); + assert!(is_math_expression(&chars, input)); + } + + #[test] + fn rule_34_bare_roman_parenthesis_has_exact_particle_suffix() { + let bare = crate::encode("링컨(Lincoln)").expect("bare rule 34 form must encode"); + let attached = + crate::encode("링컨(Lincoln)은").expect("particle-attached rule 34 form must encode"); + let particle = crate::encode("은").expect("particle must encode"); + + assert_eq!( + attached.strip_prefix(bare.as_slice()), + Some(particle.as_slice()) + ); + } + + #[rstest::rstest] + #[case::comma("링컨(Lincoln)", "링컨(Lincoln),", ",")] + #[case::period("링컨(Lincoln)", "링컨(Lincoln).", ".")] + fn rule_54_punctuation_follows_closed_roman_parenthesis_without_resplitting( + #[case] bare_input: &str, + #[case] with_punctuation: &str, + #[case] punctuation: &str, + ) { + let bare = crate::encode(bare_input).expect("bare rule 34 form must encode"); + let punctuated = + crate::encode(with_punctuation).expect("punctuated rule 54 form must encode"); + let punctuation = crate::encode(punctuation).expect("punctuation must encode"); + + assert_eq!( + punctuated.strip_prefix(bare.as_slice()), + Some(punctuation.as_slice()) + ); + } + + #[test] + fn rule_34_alphanumeric_o4o_uses_the_same_bare_and_particle_path() { + let bare = crate::encode("표기(O4O)").expect("alphanumeric Roman form must encode"); + let attached = + crate::encode("표기(O4O)는").expect("particle-attached Roman form must encode"); + let particle = crate::encode("는").expect("particle must encode"); + + assert_eq!( + attached.strip_prefix(bare.as_slice()), + Some(particle.as_slice()) + ); + assert!(!bare.windows(2).any(|cells| cells == [0, 0])); + } + #[test] fn test_is_math_with_superscript() { let chars: Vec = "x²".chars().collect(); @@ -208,6 +513,21 @@ mod tests { assert!(split_mixed_math_word(&word, 2, MathContext::default()).is_none()); } + #[rstest::rstest] + #[case::rule_34_particle("링컨(Lincoln)은")] + #[case::rule_54_comma("링컨(Lincoln),")] + #[case::alphanumeric_roman("표기(O4O).")] + fn split_mixed_math_word_keeps_korean_prefixed_closed_roman_annotation(#[case] input: &str) { + let chars: Vec = input.chars().collect(); + let word = crate::rules::token::WordToken { + text: Cow::Borrowed(input), + chars: chars.clone(), + meta: WordMeta::from_chars(&chars), + }; + + assert!(split_mixed_math_word(&word, 0, MathContext::default()).is_none()); + } + fn enc(input: &str) -> Vec { crate::encode(input).unwrap_or_default() } @@ -237,14 +557,28 @@ mod tests { assert!(!is_combining_math_mark('a')); } - #[test] - fn is_middle_dot_numeric_word_paths() { - let chars: Vec = "1·2".chars().collect(); - assert!(is_middle_dot_numeric_word(&chars)); - let chars: Vec = "ab".chars().collect(); - assert!(!is_middle_dot_numeric_word(&chars)); - let chars: Vec = "".chars().collect(); - assert!(!is_middle_dot_numeric_word(&chars)); + #[rstest::rstest] + #[case::single_middle_dot("1·2", true)] + #[case::multiple_middle_dots("2017·2018·2019·2021", true)] + #[case::trailing_comma("4·5,", true)] + #[case::letters("ab", false)] + #[case::empty("", false)] + fn is_middle_dot_numeric_word_paths(#[case] input: &str, #[case] expected: bool) { + let chars: Vec = input.chars().collect(); + assert_eq!(is_middle_dot_numeric_word(&chars), expected); + } + + #[rstest::rstest] + #[case::numeric_fraction("1/3", true)] + #[case::year_range("2023/2024", true)] + #[case::leading_decimal(".515", true)] + #[case::decimal_range("1.77~5.72", true)] + #[case::middle_dot_years("2017·2018·2019·2021", true)] + #[case::signed_number("-3", false)] + #[case::algebra("1/3+x", false)] + fn korean_prose_numeric_notation_paths(#[case] input: &str, #[case] expected: bool) { + let chars: Vec = input.chars().collect(); + assert_eq!(is_korean_prose_numeric_notation(&chars), expected); } #[test] diff --git a/libs/braillify/src/rules/token_rules/math_expression/apply.rs b/libs/braillify/src/rules/token_rules/math_expression/apply.rs index 5a2c5066..9daa02ff 100644 --- a/libs/braillify/src/rules/token_rules/math_expression/apply.rs +++ b/libs/braillify/src/rules/token_rules/math_expression/apply.rs @@ -107,6 +107,674 @@ fn is_consecutive_ascii_letter_run(chars: &[char]) -> bool { .all(|pair| u32::from(pair[1]) == u32::from(pair[0]) + 1) } +/// Whether the characters attached after a Roman closing parenthesis belong +/// to ordinary prose rather than an alphanumeric/math continuation. +/// +/// Korean rule 34 explicitly attaches the Korean particle in +/// `링컨(Lincoln)은`. The same boundary applies to a multiword Roman expansion: +/// Korean text and sentence punctuation after `)` must remain on the prose +/// path, while a digit or an ASCII letter keeps the token eligible for math. +fn is_roman_parenthetical_prose_trailer(chars: impl Iterator) -> bool { + chars.into_iter().all(|ch| { + is_korean_char(ch) + || matches!( + ch, + ',' | '.' | ';' | ':' | '!' | '?' | '·' | '\'' | '"' | '’' | '”' + ) + }) +} + +fn is_roman_hyphen(ch: char) -> bool { + matches!( + ch, + '-' | '\u{2010}' | '\u{2011}' | '\u{2012}' | '\u{2013}' | '\u{2014}' + ) +} + +fn trim_roman_identifier_edge(chars: &[char]) -> &[char] { + let mut start = 0usize; + let mut end = chars.len(); + while start < end + && matches!( + chars[start], + '\'' | '"' | '‘' | '“' | '〈' | '《' | '「' | '『' + ) + { + start += 1; + } + while start < end + && matches!( + chars[end - 1], + ',' | '.' | ';' | ':' | '!' | '?' | '\'' | '"' | '’' | '”' | '〉' | '》' | '」' | '』' + ) + { + end -= 1; + } + &chars[start..end] +} + +fn is_decimal_separator_between_digits(chars: &[char], index: usize) -> bool { + matches!(chars.get(index), Some('.' | ',')) + && index > 0 + && chars.get(index - 1).is_some_and(char::is_ascii_digit) + && chars.get(index + 1).is_some_and(char::is_ascii_digit) +} + +/// A Roman-led alphanumeric identifier in ordinary Korean prose, such as +/// `MP3`, `Web3.0`, or `GPT3.5`. +/// +/// Korean rules 29 and 35 keep an adjoining Roman letters-sequence and number +/// in the Roman section. A decimal point/comma is accepted only between two +/// digits. Requiring at least two Roman letters keeps a bare algebraic shape +/// such as `x2` on the mathematical route; explicit math mode is rejected by +/// the caller as an additional boundary. +pub(super) fn is_korean_prose_roman_number_identifier(chars: &[char]) -> bool { + let chars = trim_roman_identifier_edge(chars); + if chars.len() < 3 || !chars.first().is_some_and(char::is_ascii_alphabetic) { + return false; + } + + let mut letter_count = 0usize; + let mut has_digit = false; + for (index, ch) in chars.iter().enumerate() { + if ch.is_ascii_alphabetic() { + letter_count += 1; + } else if ch.is_ascii_digit() { + has_digit = true; + } else if !is_decimal_separator_between_digits(chars, index) { + return false; + } + } + + letter_count >= 2 && has_digit +} + +/// A print token whose numeric prefix is immediately followed by Roman +/// letters, such as `50bp`, `3.1p`, `1st`, or `3x3`. +/// +/// Korean rules 29 and 35 transcribe the Roman run and the adjoining number +/// compositionally. The same print shape can denote algebra in isolation, so +/// this predicate describes only the token grammar; the caller additionally +/// requires Korean prose context and rejects explicit/cued mathematics. +fn is_korean_prose_numeric_roman_identifier(chars: &[char]) -> bool { + let chars = trim_roman_identifier_edge(chars); + let mut index = 0usize; + let mut previous_was_digit = false; + + while let Some(&ch) = chars.get(index) { + if ch.is_ascii_digit() { + previous_was_digit = true; + index += 1; + continue; + } + if matches!(ch, ',' | '.') + && previous_was_digit + && chars.get(index + 1).is_some_and(char::is_ascii_digit) + { + previous_was_digit = false; + index += 1; + continue; + } + break; + } + + index > 0 + && chars.get(index).is_some_and(char::is_ascii_alphabetic) + && chars[index..].iter().all(char::is_ascii_alphanumeric) +} + +/// Korean articles 28, 29, 34 and 35 make an ASCII identifier in Korean prose +/// Roman text unless the caller selected math mode or the print contains an +/// unambiguous mathematical operator. A hyphen alone is not such a signal: +/// the standard's `D-100` is explicitly Roman+number, and UEB treats hyphenated +/// Roman compounds as one letters-sequence context. +/// +/// The surface remains ambiguous for algebra such as `x-1`. Keep a narrow, +/// script-based default here: digit-bearing identifiers must begin with a +/// capital Roman letter or have at least two letters in the leading segment; +/// letter-only compounds need either a capitalised segment of at least two +/// letters or the lexical `K-pop`/`x-axis` shape of one-letter prefix followed +/// by a lowercase word. Thus ordinary lowercase `x-1` and uppercase `A-B` +/// stay on the math path, while model/code and lexical-compound shapes use +/// rules 28-35. +pub(super) fn is_korean_prose_roman_hyphen_identifier(chars: &[char]) -> bool { + let chars = trim_roman_identifier_edge(chars); + if chars.is_empty() { + return false; + } + + // Rule 34 enclosure followed by a Roman continuation, e.g. `(ABC)-D`. + let core = if chars.first() == Some(&'(') { + let Some(close) = chars.iter().position(|ch| *ch == ')') else { + return false; + }; + let enclosed = &chars[1..close]; + if enclosed.len() < 2 + || !enclosed.iter().all(char::is_ascii_uppercase) + || !chars.get(close + 1).is_some_and(|ch| is_roman_hyphen(*ch)) + { + return false; + } + &chars[1..] + } else { + chars + }; + + // A parenthetical expansion after the identifier is Roman prose only when + // the current fragment starts with letters again. `F(x-1)` therefore + // remains math, while a hyphenated acronym followed by a word expansion is + // allowed to continue through subsequent whitespace tokens. + let identifier_end = core.iter().position(|ch| *ch == '(').unwrap_or(core.len()); + if identifier_end < core.len() { + let body = &core[identifier_end + 1..]; + if body.is_empty() || !body.iter().all(char::is_ascii_alphabetic) { + return false; + } + } + let identifier = &core[..identifier_end]; + + if !identifier.iter().any(|ch| is_roman_hyphen(*ch)) + || !identifier.iter().enumerate().all(|(index, ch)| { + ch.is_ascii_alphanumeric() + || is_roman_hyphen(*ch) + || *ch == ')' + || is_decimal_separator_between_digits(identifier, index) + }) + { + return false; + } + + let segments = identifier.split(|ch| is_roman_hyphen(*ch)); + let mut has_digit = false; + let mut first_ascii_letter = None; + let mut has_capitalised_word_segment = false; + let mut first_segment_letter_count = 0usize; + let mut first_segment_is_single_letter = false; + let mut has_later_lowercase_lexical_segment = false; + for (segment_index, raw_segment) in segments.enumerate() { + let segment = raw_segment + .iter() + .copied() + .filter(|ch| ch.is_ascii_alphanumeric()) + .collect::>(); + if segment.is_empty() { + return false; + } + has_digit |= segment.iter().any(char::is_ascii_digit); + first_ascii_letter = + first_ascii_letter.or_else(|| segment.iter().copied().find(char::is_ascii_alphabetic)); + let letter_count = segment.iter().filter(|ch| ch.is_ascii_alphabetic()).count(); + has_capitalised_word_segment |= letter_count >= 2 + && segment + .iter() + .find(|ch| ch.is_ascii_alphabetic()) + .is_some_and(|ch| ch.is_ascii_uppercase()); + if segment_index == 0 { + first_segment_letter_count = letter_count; + first_segment_is_single_letter = segment.len() == 1 && letter_count == 1; + } else if first_segment_is_single_letter { + has_later_lowercase_lexical_segment |= + letter_count >= 2 && segment.iter().all(char::is_ascii_lowercase); + } + } + + if has_digit { + first_segment_letter_count >= 2 + || first_ascii_letter.is_some_and(|ch| ch.is_ascii_uppercase()) + } else { + has_capitalised_word_segment + || (first_segment_is_single_letter && has_later_lowercase_lexical_segment) + } +} + +/// Roman identifier joined by a solidus in ordinary Korean prose. +/// +/// The solidus is shared by UEB Roman text and mathematical division. Keep +/// the mathematical one-letter fraction shapes (`A/B`, `F/N`) on the math +/// route, and recognize only identifier-like forms that start with a capital +/// and contain at least one multi-character alphanumeric segment. This covers +/// standard prose abbreviations and model families such as `ISO/IEC` and +/// `F-5E/F` without changing an explicitly selected math context. +pub(super) fn is_korean_prose_roman_slash_identifier(chars: &[char]) -> bool { + let chars = trim_roman_identifier_edge(chars); + if chars.is_empty() + || !chars.first().is_some_and(|ch| ch.is_ascii_uppercase()) + || !chars.contains(&'/') + || !chars.iter().all(|ch| { + ch.is_ascii_alphanumeric() || *ch == '/' || is_roman_hyphen(*ch) || *ch == '.' + }) + { + return false; + } + + let mut has_letter = false; + let mut has_multi_character_segment = false; + for segment in chars.split(|ch| *ch == '/') { + if segment.is_empty() || segment.iter().all(|ch| is_roman_hyphen(*ch) || *ch == '.') { + return false; + } + has_letter |= segment.iter().any(char::is_ascii_alphabetic); + has_multi_character_segment |= segment + .iter() + .filter(|ch| ch.is_ascii_alphanumeric()) + .count() + >= 2; + } + has_letter && has_multi_character_segment +} + +/// A single-letter solidus initialism can be distinguished from mathematical +/// division when it begins a capital-led multi-letter Roman phrase, such as +/// `H/W Wallet` or `R/R ES-SCLC`. Korean rule 29 keeps consecutive Roman words +/// in one section, while an isolated `F/N` remains on the math path used by the +/// official mathematics rule 29 example. +pub(super) fn is_korean_prose_single_letter_slash_phrase( + tokens: &[Token<'_>], + index: usize, + chars: &[char], +) -> bool { + let has_strong_math_symbol = chars.iter().any(|ch| { + math_symbol_shortcut::is_math_symbol_char(*ch) + && !matches!(*ch, '\u{00B7}' | '\u{22C5}' | '/' | '_') + }); + if has_strong_math_symbol { + return false; + } + + let has_single_letter_slash_run = (0..chars.len()).any(|start| { + if !chars[start].is_ascii_uppercase() + || start + .checked_sub(1) + .and_then(|before| chars.get(before)) + .is_some_and(|ch| ch.is_ascii_alphanumeric() || *ch == '/') + { + return false; + } + + let mut cursor = start + 1; + let mut slash_count = 0usize; + while chars.get(cursor) == Some(&'/') + && chars + .get(cursor + 1) + .is_some_and(|ch| ch.is_ascii_uppercase()) + { + slash_count += 1; + cursor += 2; + } + + slash_count > 0 + && !chars + .get(cursor) + .is_some_and(|ch| ch.is_ascii_alphanumeric() || *ch == '/') + }); + if !has_single_letter_slash_run { + return false; + } + + let Some(next_word) = next_word_skip_space(tokens, index + 1) else { + return false; + }; + let mut next_roman = next_word + .chars + .iter() + .copied() + .skip_while(|ch| matches!(*ch, '\'' | '"' | '‘' | '“' | '(' | '[' | '{')) + .take_while(|ch| ch.is_ascii_alphanumeric() || is_roman_hyphen(*ch)); + let Some(first) = next_roman.next() else { + return false; + }; + first.is_ascii_uppercase() + && next_roman.filter(char::is_ascii_alphabetic).count() + + usize::from(first.is_ascii_alphabetic()) + >= 2 +} + +fn is_roman_identifier_head_separator(chars: &[char], index: usize) -> bool { + matches!( + chars.get(index), + Some('.' | '/' | '-' | '‐' | '‑' | '‒' | '–' | '—') + ) && index > 0 + && chars[index - 1].is_ascii_alphanumeric() + && chars + .get(index + 1) + .is_some_and(char::is_ascii_alphanumeric) +} + +/// A Roman identifier ending in one or more plus signs. +/// +/// The head may combine Roman letters with adjoining digits and the ordinary +/// identifier separators already covered by Korean rules 29/32/35. A head of +/// two or more alphanumerics is structurally terminal (`TV+`, `24K+`), and a +/// repeated plus is likewise not a completed binary addition (`C++`). A +/// one-letter `A+` is terminal in ordinary prose unless a visible right operand +/// follows; explicit mathematics is rejected by the caller before this rule. +fn is_terminal_roman_plus_core(core: &[char], allow_single_letter: bool) -> bool { + let plus_start = core + .iter() + .rposition(|ch| *ch != '+') + .map_or(0, |index| index + 1); + if plus_start == 0 || plus_start == core.len() { + return false; + } + + let head = &core[..plus_start]; + let plus_count = core.len() - plus_start; + if head.contains(&'+') + || !head.iter().enumerate().all(|(index, ch)| { + ch.is_ascii_alphanumeric() || is_roman_identifier_head_separator(head, index) + }) + || !head.iter().any(char::is_ascii_alphabetic) + { + return false; + } + + let alphanumeric_count = head.iter().filter(|ch| ch.is_ascii_alphanumeric()).count(); + alphanumeric_count >= 2 + || plus_count >= 2 + || (allow_single_letter && head.len() == 1 && head[0].is_ascii_uppercase()) +} + +fn is_attached_plus_prose_trailer_char(ch: char) -> bool { + is_korean_char(ch) + || matches!( + ch, + '(' | ')' + | '[' + | ']' + | '{' + | '}' + | ',' + | '.' + | ';' + | ':' + | '!' + | '?' + | '\'' + | '"' + | '‘' + | '’' + | '“' + | '”' + | '〈' + | '〉' + | '《' + | '》' + | '「' + | '」' + | '『' + | '』' + ) +} + +fn is_terminal_plus_closer_char(ch: char) -> bool { + matches!( + ch, + ')' | ']' + | '}' + | ',' + | '.' + | ';' + | ':' + | '!' + | '?' + | '\'' + | '"' + | '’' + | '”' + | '〉' + | '》' + | '」' + | '』' + ) +} + +/// Roman product, service, or lexical compound using a plus sign. +/// +/// A completed mathematical addition necessarily has a right operand, whereas +/// a terminal `+` is a common part of a Roman identifier (`TV+`, `HDR10+`). A +/// Korean particle or annotation may be attached directly after that core, and +/// repeated plus signs remain part of the same identifier. A one-letter +/// terminal form stays on this prose path unless a parenthesized ASCII operand +/// completes the expression; explicit math mode remains math-owned. +/// +/// A plus between capital-led Roman words is likewise lexical when at least one +/// side has a lowercase letter and two or more letters (`Dog+Yoga`). That +/// orthographic signal deliberately excludes all-capital algebra-like surfaces +/// such as `AB+C` and lowercase function sums such as `sin+cos`. Finally, a +/// single capital immediately followed by `+` and attached Hangul +/// (`U+유모바일`) is a Roman brand prefix followed by Korean text; Article 46 +/// would require spaces around a genuine Korean addition. +pub(super) fn is_korean_prose_roman_plus_identifier(chars: &[char]) -> bool { + let chars = trim_roman_identifier_edge(chars); + if chars.is_empty() { + return false; + } + + let roman_end = chars + .iter() + .take_while(|ch| { + ch.is_ascii_alphanumeric() + || **ch == '+' + || matches!(**ch, '.' | '/' | '-' | '‐' | '‑' | '‒' | '–' | '—') + }) + .count(); + let core = &chars[..roman_end]; + let trailer = &chars[roman_end..]; + let korean_led_mixed_trailer = trailer.first().is_some_and(|ch| is_korean_char(*ch)) + && trailer + .iter() + .all(|ch| ch.is_ascii_alphanumeric() || is_attached_plus_prose_trailer_char(*ch)); + let trailer_is_prose = trailer.is_empty() + || trailer.first() == Some(&'(') + || korean_led_mixed_trailer + || trailer + .iter() + .copied() + .all(is_attached_plus_prose_trailer_char); + if !trailer_is_prose { + return false; + } + + let has_korean_trailer = trailer.iter().any(|ch| is_korean_char(*ch)); + let allow_single_letter = trailer.is_empty() + || has_korean_trailer + || trailer.iter().copied().all(is_terminal_plus_closer_char); + if is_terminal_roman_plus_core(core, allow_single_letter) { + return true; + } + + if !core.first().is_some_and(|ch| ch.is_ascii_uppercase()) { + return false; + } + + if !core.contains(&'+') { + return false; + } + + if !core.iter().all(|ch| ch.is_ascii_alphabetic() || *ch == '+') { + return false; + } + + let segments = core.split(|ch| *ch == '+').collect::>(); + segments.len() >= 2 + && segments.iter().all(|segment| !segment.is_empty()) + && segments.iter().any(|segment| { + segment.len() >= 2 + && segment.iter().any(char::is_ascii_lowercase) + && segment.iter().all(char::is_ascii_alphabetic) + }) +} + +/// A Korean word may immediately introduce a parenthesized Roman lexical +/// compound (`도가(Dog+Yoga)`). Prove the Korean prefix and a closed Roman +/// body, then reuse the same plus grammar. Text following the close must be +/// ordinary Korean prose or punctuation, never another ASCII operand. +pub(super) fn has_korean_prefix_roman_plus_annotation(chars: &[char]) -> bool { + chars.iter().enumerate().any(|(start, ch)| { + if !ch.is_ascii_alphabetic() || !chars[..start].iter().any(|prefix| is_korean_char(*prefix)) + { + return false; + } + + let suffix = &chars[start..]; + let Some(close) = suffix.iter().position(|candidate| *candidate == ')') else { + return false; + }; + let trailer = &suffix[close + 1..]; + close > 0 + && (is_korean_prose_roman_plus_identifier(&suffix[..close]) + || is_terminal_roman_plus_core(&suffix[..close], true)) + && (is_roman_parenthetical_prose_trailer(trailer.iter().copied()) + || trailer + .first() + .is_some_and(|ch| is_korean_char(*ch) || *ch == '·')) + }) +} + +/// A Korean lexical prefix may attach directly to a terminal Roman identifier +/// (`한글TV+는`). Once the first Roman/digit run after Korean is found, reuse +/// the same terminal-plus grammar. A later operand after an earlier plus is not +/// a new start, so `한글A+B` remains math-owned. +pub(super) fn has_korean_prefix_terminal_roman_plus_identifier(chars: &[char]) -> bool { + chars.iter().enumerate().any(|(start, ch)| { + ch.is_ascii_alphanumeric() + && chars[..start].iter().any(|prefix| is_korean_char(*prefix)) + && start + .checked_sub(1) + .and_then(|index| chars.get(index)) + .is_none_or(|previous| !previous.is_ascii_alphanumeric() && *previous != '+') + && is_korean_prose_roman_plus_identifier(&chars[start..]) + }) +} + +/// A Korean word may attach directly to a hyphenated Roman identifier in two +/// directions: an enclosed Roman run can continue after a hyphen +/// (`한글(ABC)-D`), or the Korean run itself can be followed by a Roman +/// label (`하쿠토-R`, `기장-KBO`). Korean rule 33 proves that `-` at a +/// Korean/Roman boundary is punctuation rather than mathematical subtraction; +/// rules 29 and 35 then own the Roman run. +/// +/// A single lowercase letter remains ambiguous algebra (`값-x`), and an +/// explicit operator after the Roman start remains math-owned (`값-x+1`). +pub(super) fn has_korean_prefix_roman_hyphen_suffix(chars: &[char]) -> bool { + for (index, ch) in chars.iter().enumerate() { + if ch.is_ascii_alphabetic() + && chars[..index] + .iter() + .any(|prefix| crate::utils::is_korean_char(*prefix)) + && is_korean_prose_roman_hyphen_identifier(&chars[index..]) + { + return true; + } + } + + chars.windows(3).enumerate().any(|(index, window)| { + if !crate::utils::is_korean_char(window[0]) + || window[1] != '-' + || !window[2].is_ascii_alphabetic() + { + return false; + } + + let roman_tail = &chars[index + 2..]; + let identifier_len = roman_tail + .iter() + .take_while(|ch| ch.is_ascii_alphanumeric()) + .count(); + let letter_count = roman_tail[..identifier_len] + .iter() + .filter(|ch| ch.is_ascii_alphabetic()) + .count(); + let identifier_is_unambiguous = window[2].is_ascii_uppercase() || letter_count >= 2; + let has_explicit_math_operator = roman_tail.iter().any(|ch| { + matches!( + *ch, + '+' | '−' + | '×' + | '÷' + | '=' + | '<' + | '>' + | '≤' + | '≥' + | '≠' + | '≈' + | '^' + | '_' + | '/' + | '*' + | '|' + | '∈' + | '∉' + | '⊂' + | '⊃' + | '∧' + | '∨' + ) + }); + + identifier_is_unambiguous && !has_explicit_math_operator + }) +} + +/// Whether a spaced `A(31)`-shaped label is followed by ordinary Korean prose. +/// +/// A print-space plus a Korean person role (`도의원`, `교수`, `부장판사`) +/// resolves the same function-notation ambiguity as an honorific does. The +/// explicit mathematical value/product cues remain on the math route. This +/// predicate deliberately requires a real source space and an all-Korean next +/// word; attached particles are handled by the narrower label splitter. +pub(super) fn next_word_begins_korean_prose_label_context( + tokens: &[Token<'_>], + index: usize, +) -> bool { + if !matches!(tokens.get(index + 1), Some(Token::Space(_))) + || next_word_starts_with_math_value_cue(tokens, index) + { + return false; + } + + next_indexed_word_skip_space(tokens, index + 1).is_some_and(|(next_index, word)| { + next_index > index + 1 + && word.chars.iter().any(|ch| is_korean_char(*ch)) + && word + .chars + .iter() + .all(|ch| is_korean_char(*ch) || matches!(*ch, ',' | '.' | '!' | '?')) + }) +} + +/// Rule 34 parenthetical Roman prose headed by a multi-character acronym. +/// Requiring at least two alphanumeric head characters and rejecting math +/// operators keeps `f(x)` / `A(x+1)` in the math engine. +pub(super) fn is_korean_prose_acronym_parenthetical(chars: &[char]) -> bool { + let chars = trim_roman_identifier_edge(chars); + let Some(open) = chars.iter().position(|ch| *ch == '(') else { + return false; + }; + let head = &chars[..open]; + if head.len() < 2 + || !head.iter().all(char::is_ascii_alphanumeric) + || !head.iter().any(char::is_ascii_uppercase) + { + return false; + } + + let after_open = &chars[open + 1..]; + let close = after_open.iter().position(|ch| *ch == ')'); + let body = close.map_or(after_open, |index| &after_open[..index]); + if body.is_empty() + || !body + .iter() + .all(|ch| ch.is_ascii_alphanumeric() || is_roman_hyphen(*ch)) + { + return false; + } + close.is_none_or(|index| { + is_roman_parenthetical_prose_trailer(after_open[index + 1..].iter().copied()) + }) +} + fn has_ascii_letter_korean_math_suffix(chars: &[char]) -> bool { if chars.len() < 3 { return false; @@ -144,6 +812,323 @@ fn prev_word_is_math_product_cue(tokens: &[Token<'_>], index: usize) -> bool { .is_some_and(|word| word.text.as_ref() == "곱") } +/// Returns true when `word` is the final fragment of a whitespace-split, +/// closed Roman parenthetical whose earlier fragments contain letters only. +/// +/// Korean rules 29 and 34 keep consecutive Roman words in one Roman section +/// and omit its terminator before the closing parenthesis. UEB 9.7.1 likewise +/// prints multiword prose inside one paired parenthesis. The token parser keeps +/// the spaces as separate tokens, so the final `Letters)` fragment must not be +/// mistaken for a standalone mathematical expression merely because it has a +/// closing bracket. A lowercase/mixed-case ASCII letter immediately before the +/// opening parenthesis is excluded so function-call syntax such as `f(x)` +/// remains math-owned; a complete all-capitals initialism (`WTO(World ...),`) +/// is the ordinary rule-29/34 prose form. +fn is_multiword_closed_roman_parenthetical_tail( + tokens: &[Token<'_>], + index: usize, + word: &WordToken<'_>, +) -> bool { + let Some(close) = word.chars.iter().position(|ch| *ch == ')') else { + return false; + }; + let body = &word.chars[..close]; + let trailing = &word.chars[close + 1..]; + if body.is_empty() + || !body.iter().all(char::is_ascii_alphabetic) + || !is_roman_parenthetical_prose_trailer(trailing.iter().copied()) + { + return false; + } + + let mut cursor = index.checked_sub(1); + while let Some(i) = cursor { + match tokens.get(i) { + Some(Token::Space(_)) => cursor = i.checked_sub(1), + Some(Token::Word(previous)) => { + let previous_text = previous.text.as_ref(); + if let Some(open) = previous_text.rfind('(') { + let before = &previous_text[..open]; + let after = &previous_text[open + 1..]; + if after.is_empty() || !after.chars().all(|ch| ch.is_ascii_alphabetic()) { + return false; + } + let before_is_initialism = before.chars().count() >= 2 + && before.chars().all(|ch| ch.is_ascii_uppercase()); + if before + .chars() + .next_back() + .is_some_and(|ch| ch.is_ascii_alphabetic()) + && !before_is_initialism + { + return false; + } + return before.find(['(', ')']).is_none(); + } + if previous_text.chars().all(|ch| ch.is_ascii_alphabetic()) { + cursor = i.checked_sub(1); + } else { + return false; + } + } + _ => return false, + } + } + false +} + +/// Returns true when `word` begins a closed, multiword Roman expansion headed +/// by a complete all-capitals abbreviation. +/// +/// Korean rules 29 and 34 make this ordinary Roman prose: the headword starts +/// a Roman section, and the spaces inside the paired parenthesis do not split +/// that section. The narrow grammar excludes single variables, digits, +/// operators, nested brackets, and an alphanumeric continuation after `)` so +/// mathematical expressions remain owned by the math parser. +fn is_multiword_closed_roman_parenthetical_head( + tokens: &[Token<'_>], + index: usize, + word: &WordToken<'_>, +) -> bool { + let text = word.text.as_ref(); + let Some(open) = text.find('(') else { + return false; + }; + let head = &text[..open]; + let first_body_word = &text[open + 1..]; + if head.chars().count() < 2 + || !head.chars().all(|ch| ch.is_ascii_uppercase()) + || first_body_word.is_empty() + || !first_body_word.chars().all(|ch| ch.is_ascii_alphabetic()) + { + return false; + } + + let mut cursor = index + 1; + let mut body_words = 1usize; + loop { + let mut saw_space = false; + while matches!(tokens.get(cursor), Some(Token::Space(_))) { + saw_space = true; + cursor += 1; + } + if !saw_space { + return false; + } + let Some(Token::Word(next)) = tokens.get(cursor) else { + return false; + }; + let next_text = next.text.as_ref(); + if let Some(close) = next_text.find(')') { + let final_body_word = &next_text[..close]; + let trailing = &next_text[close + 1..]; + body_words += 1; + return body_words >= 2 + && !final_body_word.is_empty() + && final_body_word.chars().all(|ch| ch.is_ascii_alphabetic()) + && is_roman_parenthetical_prose_trailer(trailing.chars()); + } + if next_text.is_empty() || !next_text.chars().all(|ch| ch.is_ascii_alphabetic()) { + return false; + } + body_words += 1; + cursor += 1; + } +} + +/// Returns whether `index` belongs to a complete prose parenthetical that is +/// attached directly to Korean text. +/// +/// Korean rules 34 and 54 keep the Korean parenthesis outside the enclosed +/// Roman section (`링컨(Lincoln)은`: `⠦⠄⠴...⠠⠴`). The generic mathematics +/// detector must therefore not take ownership merely because the enclosed +/// text is an all-capitals identifier, an alphanumeric name, or a decimal. +/// This scan covers a parenthetical split across whitespace tokens as well as +/// a digit immediately following its close (`용어(Web)3`). +/// +/// A one-letter variable and an expression carrying an unambiguous operator +/// remain math-owned. This is the structural distinction between the rule-34 +/// prose form and ordinary function/expression notation such as `함수(x+1)`. +fn is_within_attached_korean_prose_parenthetical(tokens: &[Token<'_>], index: usize) -> bool { + #[derive(Clone, Copy)] + struct Opening { + token_index: usize, + char_index: usize, + attached_to_korean_prose: bool, + } + + fn enclosed_chars( + tokens: &[Token<'_>], + opening: Opening, + close_token_index: usize, + close_char_index: usize, + ) -> Vec { + let mut body = Vec::new(); + for (token_index, token) in tokens + .iter() + .enumerate() + .take(close_token_index + 1) + .skip(opening.token_index) + { + match token { + Token::Word(word) => { + let start = if token_index == opening.token_index { + opening.char_index + 1 + } else { + 0 + }; + let end = if token_index == close_token_index { + close_char_index + } else { + word.chars.len() + }; + if start <= end && end <= word.chars.len() { + body.extend_from_slice(&word.chars[start..end]); + } + } + Token::Space(_) => body.push(' '), + Token::Mode(_) => {} + Token::Fraction(_) | Token::PreEncoded(_) => return Vec::new(), + } + } + body + } + + fn is_prose_body(body: &[char]) -> bool { + let body = body + .iter() + .copied() + .skip_while(|ch| ch.is_whitespace()) + .collect::>(); + let body = body + .iter() + .copied() + .rev() + .skip_while(|ch| ch.is_whitespace()) + .collect::>() + .into_iter() + .rev() + .collect::>(); + if body.is_empty() || body.iter().any(|ch| matches!(*ch, '(' | ')')) { + return false; + } + + // Operators which cannot be ordinary punctuation or part of a Roman + // identifier make the enclosure an explicit mathematical expression. + if body.iter().any(|ch| { + matches!( + *ch, + '=' | '<' + | '>' + | '≤' + | '≥' + | '≠' + | '≈' + | '≡' + | '×' + | '÷' + | '√' + | '∑' + | '∏' + | '∫' + | '∈' + | '∉' + | '⊂' + | '⊃' + | '^' + | '_' + ) + }) { + return false; + } + + // A Korean explanation inside an attached parenthesis is prose. Its + // embedded Roman/numeric fragments are still handled compositionally + // by rules 28-35 after this token rule declines the whole expression. + if body.iter().any(|ch| is_korean_char(*ch)) { + return true; + } + + let numeric_annotation = body.iter().any(char::is_ascii_digit) + && body.iter().all(|ch| { + ch.is_ascii_digit() + || ch.is_whitespace() + || matches!(*ch, '.' | ',' | '%' | '‰' | '+' | '-' | '−' | '~') + }); + if numeric_annotation { + return true; + } + + let ascii_alphanumeric_count = body.iter().filter(|ch| ch.is_ascii_alphanumeric()).count(); + let has_ascii_letter = body.iter().any(char::is_ascii_alphabetic); + ascii_alphanumeric_count >= 2 + && has_ascii_letter + && body.iter().all(|ch| { + ch.is_ascii_alphanumeric() + || ch.is_whitespace() + || matches!( + *ch, + ',' | '.' + | ':' + | ';' + | '\'' + | '’' + | '-' + | '‐' + | '‑' + | '‒' + | '–' + | '—' + | '/' + | '&' + | '·' + | '⋅' + ) + }) + } + + let mut openings = Vec::::new(); + for (token_index, token) in tokens.iter().enumerate() { + let Token::Word(word) = token else { + continue; + }; + for (char_index, ch) in word.chars.iter().copied().enumerate() { + match ch { + '(' => openings.push(Opening { + token_index, + char_index, + attached_to_korean_prose: { + let prefix = &word.chars[..char_index]; + let prefix_contains_korean = prefix.iter().any(|ch| is_korean_char(*ch)); + let numeric_prefix = !prefix.is_empty() + && prefix.iter().any(char::is_ascii_digit) + && prefix.iter().all(|ch| { + ch.is_ascii_digit() + || matches!(*ch, '.' | ',' | '\'' | '’' | '"' | '”' | '‘' | '“') + }); + prefix_contains_korean + || (numeric_prefix && has_adjacent_korean_word(tokens, token_index)) + }, + }), + ')' => { + let Some(opening) = openings.pop() else { + continue; + }; + if opening.attached_to_korean_prose + && opening.token_index <= index + && index <= token_index + && is_prose_body(&enclosed_chars(tokens, opening, token_index, char_index)) + { + return true; + } + } + _ => {} + } + } + } + false +} + /// Walks backward from `index - 1`, skipping `Space`, returning whether the /// preceding content is a math-letter Word or a math-context PreEncoded. fn prev_is_math_context_for_ellipsis(tokens: &[Token<'_>], index: usize) -> bool { @@ -240,13 +1225,36 @@ fn prev_prev_is_math_or_mixed_context(tokens: &[Token<'_>], index: usize) -> boo false } -/// Detect a Word that is exactly the logic XOR symbol `⊻` (U+22BB). +/// Detect one unambiguous set/logic symbol from math rules 60-61. /// -/// PDF 수학 — `A ⊻ B` 패턴에서 양쪽 대문자를 math 변수로 처리하기 위해 사용. -pub(super) fn is_logic_symbol_word(word: &crate::rules::token::WordToken<'_>) -> bool { - word.chars - .first() - .is_some_and(|c| word.chars.len() == 1 && matches!(*c, '⊻')) +/// These Unicode signs are not Roman-prose punctuation. A separated adjacent +/// capital therefore remains a math variable instead of entering UEB grade-1 +/// text (`A ¬ B`, `{x | x ∈ R}`). +pub(super) fn is_set_or_logic_symbol_word(word: &crate::rules::token::WordToken<'_>) -> bool { + word.chars.first().is_some_and(|c| { + word.chars.len() == 1 + && matches!( + *c, + '¬' | '∈' + | '∋' + | '∉' + | '∌' + | '⊂' + | '⊃' + | '⊄' + | '⊅' + | '∪' + | '∩' + | '∀' + | '∃' + | '∄' + | '∧' + | '∨' + | '⊻' + | '⇒' + | '⇔' + ) + }) } /// PDF — Compute leading spaces for a math token inserted at `index` based on @@ -287,6 +1295,69 @@ pub(super) fn run<'a>( let text = word.text.as_ref(); + // Preserve the more specific anonymized-person grammar before the general + // rule-34 prose-parenthetical guard below. A Korean name fragment may be + // attached before the Roman initial (`모A(61)씨`), so this must split and + // retain that prefix rather than merely declining whole-token math. + if state.english_indicator + && !state.math_mode_active + && let Some(replacement) = split_anonymized_person_label(&word.chars) + { + return Ok(TokenAction::ReplaceMany(replacement)); + } + + if is_multiword_closed_roman_parenthetical_head(tokens, index, word) + || is_multiword_closed_roman_parenthetical_tail(tokens, index, word) + || is_within_attached_korean_prose_parenthetical(tokens, index) + { + return Ok(TokenAction::Noop); + } + + // Korean rules 29, 35, 54: in anonymized-person prose, encode the Roman + // initial, Korean parentheses and age compositionally even when the + // following Korean honorific/role is separated by a print-space. The + // following word is deliberately left as its own token so source spacing + // is preserved. + if state.english_indicator + && !state.math_mode_active + && next_word_begins_korean_prose_label_context(tokens, index) + && let Some(encoded) = encode_anonymized_person_label(&word.chars) + { + return Ok(TokenAction::Replace(Token::PreEncoded(encoded))); + } + + // In ordinary Korean prose, rules 28-35 own structurally Roman identifiers. + // Do this before the generic `letter + operator` math detector: ASCII '-' is + // both a math minus candidate and the hyphen used by the official `D-100`. + if state.english_indicator + && !state.math_mode_active + && (is_korean_prose_roman_hyphen_identifier(&word.chars) + || is_korean_prose_roman_number_identifier(&word.chars) + || is_korean_prose_roman_slash_identifier(&word.chars) + || is_korean_prose_single_letter_slash_phrase(tokens, index, &word.chars) + || is_korean_prose_roman_plus_identifier(&word.chars) + || has_korean_prefix_roman_plus_annotation(&word.chars) + || has_korean_prefix_terminal_roman_plus_identifier(&word.chars) + || has_korean_prefix_roman_hyphen_suffix(&word.chars) + || is_korean_prose_acronym_parenthetical(&word.chars)) + { + return Ok(TokenAction::Noop); + } + + // Korean rules 29 and 35 also own a number immediately followed by a Roman + // letters-sequence in ordinary Korean prose. Keep an isolated `3ab` on the + // mathematical route, and preserve explicit math mode plus the established + // Korean `곱`/`값` cues for genuinely mathematical uses. + if state.english_indicator + && !state.math_mode_active + && has_adjacent_korean_word(tokens, index) + && is_korean_prose_numeric_roman_identifier(&word.chars) + && !prev_word_is_math_product_cue(tokens, index) + && !next_word_starts_with_math_value_cue(tokens, index) + { + return Ok(TokenAction::Noop); + } + // PDF 수학 제60/61항 — `a ≲ b:`, `p ⊻ q:` 같이 단일 letter + 관계기호 + 단일 // letter + 콜론 패턴의 inline math expression. 콜론 이전까지를 하나의 math // expression으로 병합해 인코딩한다 (letter들이 산문 quote-wrap되지 않도록). @@ -532,9 +1603,13 @@ pub(super) fn run<'a>( } } - // Numeric middle-dot forms in Korean prose (e.g. 3·1 운동) should stay non-math, - // while standalone numeric expressions like 6·9 should be routed to math. - if is_middle_dot_numeric_word(&word.chars) && has_adjacent_korean_word(tokens, index) { + // Korean Rules 43, 47 [appendix], 48, and 50: numeric punctuation in prose + // (`3·1 운동`, `1/3 규모`, `.515로`, `1.7~2.4 사이`) remains on the + // ordinary number/punctuation path. A standalone expression keeps using + // the math engine because there is no adjacent Korean prose context. + if (is_middle_dot_numeric_word(&word.chars) || is_korean_prose_numeric_notation(&word.chars)) + && has_adjacent_korean_word(tokens, index) + { return Ok(TokenAction::Noop); } @@ -554,12 +1629,52 @@ pub(super) fn run<'a>( } } - // Logical symbols separated by spaces should still treat uppercase letters as variables. + // Math rules 60-61: process a separated right-hand capital while the + // set/logic sign is still a Word token. Once the sign becomes PreEncoded, + // neighbour lookup intentionally stops at that boundary and the capital + // would otherwise fall through to UEB prose (and gain a grade-1 marker). + // + // The right token may retain non-alphanumeric punctuation or a Korean + // suffix (`R}`, `P는`), but another ASCII letter/digit means it is a Roman + // word or identifier rather than one mathematical variable (`Road`). + if is_set_or_logic_symbol_word(word) + && let Some((right_index, right_word)) = next_indexed_word_skip_space(tokens, index + 1) + && right_word + .chars + .first() + .is_some_and(char::is_ascii_uppercase) + && right_word.chars[1..] + .iter() + .all(|ch| crate::utils::is_korean_char(*ch) || !ch.is_ascii_alphanumeric()) + { + let symbol = math_symbol_shortcut::encode_char_math_symbol_shortcut(word.chars[0])?; + let upper = right_word.chars[0]; + let code = crate::english::encode_english(upper.to_ascii_lowercase())?; + let mut replacement: Vec> = vec![Token::PreEncoded(symbol.to_vec())]; + replacement.extend(tokens[index + 1..right_index].iter().cloned()); + replacement.push(Token::PreEncoded(vec![32, code])); + + if right_word.chars.len() > 1 { + let suffix = right_word.chars[1..].iter().collect::(); + replacement.push(build_word_token(suffix)); + } + + return Ok(TokenAction::ReplaceRange( + right_index + 1 - index, + replacement, + )); + } + + // Set/logic symbols separated by spaces still own adjacent uppercase math + // variables. Emit the capital indicator here so the later UEB token rules + // cannot reinterpret the one-letter variable as an alphabetic wordsign. if word.chars.len() == 1 && word.chars[0].is_ascii_uppercase() { let (prev, next) = prev_next_words(tokens, index); - if prev.is_some_and(is_logic_symbol_word) || next.is_some_and(is_logic_symbol_word) { + if prev.is_some_and(is_set_or_logic_symbol_word) + || next.is_some_and(is_set_or_logic_symbol_word) + { let code = crate::english::encode_english(word.chars[0].to_ascii_lowercase())?; - return Ok(TokenAction::Replace(Token::PreEncoded(vec![code]))); + return Ok(TokenAction::Replace(Token::PreEncoded(vec![32, code]))); } } @@ -837,6 +1952,188 @@ pub(super) fn run<'a>( #[cfg(test)] mod tests { use super::*; + + #[rstest::rstest] + #[case::ueb_multiword_parenthetical("plays (such as Romeo and Juliet)", true)] + #[case::initialism_prefixed_comma("WTO(World Tourism Organization),", true)] + #[case::korean_particle_after_parenthesis("설명(Home Connectivity Alliance)를", true)] + #[case::korean_particle_after_quote("설명(Home Connectivity Alliance)’를", true)] + #[case::ueb_letter_list("(q, r)", false)] + #[case::math_function("f(x)", false)] + #[case::operator_interrupts_prose_run("(x + y)", false)] + #[case::no_closing_parenthesis("Romeo Juliet", false)] + #[case::function_with_spaced_argument("f(x y)", false)] + #[case::missing_opening_parenthesis("Romeo Juliet)", false)] + #[case::invalid_trailing_digit("(Romeo Juliet)1", false)] + #[case::digit_in_final_fragment("(Romeo Juliet2)", false)] + #[case::digit_after_opening("(2Romeo Juliet)", false)] + #[case::digit_in_earlier_fragment("(Romeo2 Juliet)", false)] + #[case::nonletter_earlier_without_opening("Romeo2 Juliet More)", false)] + #[case::nested_opening_before_fragment("((Romeo Juliet)", false)] + #[case::closing_before_opening(")(Romeo Juliet)", false)] + fn recognizes_only_complete_multiword_roman_parenthetical_tails( + #[case] input: &str, + #[case] expected: bool, + ) { + let ir = crate::rules::token::DocumentIR::parse(input, true); + let index = ir + .tokens + .iter() + .rposition(|token| matches!(token, Token::Word(_))) + .expect("probe must contain a word"); + let Token::Word(word) = &ir.tokens[index] else { + unreachable!("selected token must be a word"); + }; + + assert_eq!( + is_multiword_closed_roman_parenthetical_tail(&ir.tokens, index, word), + expected + ); + + if expected { + let mut state = EncoderState::new(false); + assert!(matches!( + run(&ir.tokens, index, &mut state).unwrap(), + TokenAction::Noop + )); + } + } + + #[rstest::rstest] + #[case::initialism_expansion("HCA(Home Connectivity Alliance)", true)] + #[case::punctuated_expansion("TB(Top View Battle),", true)] + #[case::korean_particle("HCA(Home Connectivity Alliance)를", true)] + #[case::quoted_korean_particle("HCA(Home Connectivity Alliance)’를", true)] + #[case::single_capital_head("A(Home Connectivity Alliance)", false)] + #[case::mixed_case_head("HCa(Home Connectivity Alliance)", false)] + #[case::single_word_body("HCA(Alliance)", false)] + #[case::digit_in_body("HCA(Home Connectivity2 Alliance)", false)] + #[case::operator_in_body("HCA(Home + Alliance)", false)] + #[case::nested_parenthesis("HCA((Home Connectivity Alliance))", false)] + #[case::alphanumeric_trailer("HCA(Home Connectivity Alliance)1", false)] + #[case::unclosed_expansion("HCA(Home Connectivity Alliance", false)] + fn recognizes_only_complete_allcaps_multiword_roman_expansion_heads( + #[case] input: &str, + #[case] expected: bool, + ) { + let ir = crate::rules::token::DocumentIR::parse(input, true); + let index = ir + .tokens + .iter() + .position(|token| matches!(token, Token::Word(_))) + .expect("probe must contain a word"); + let Token::Word(word) = &ir.tokens[index] else { + unreachable!("selected token must be a word"); + }; + + assert_eq!( + is_multiword_closed_roman_parenthetical_head(&ir.tokens, index, word), + expected + ); + + if expected { + let mut state = EncoderState::new(false); + assert!(matches!( + run(&ir.tokens, index, &mut state).unwrap(), + TokenAction::Noop + )); + } + } + + #[rstest::rstest] + #[case::roman_followed_by_digit("용어(Web)3", 1)] + #[case::roman_then_korean_explanation("기관(KRISS, 원장)", 2)] + #[case::numeric_annotation("최고치(2126.14)", 1)] + #[case::multiword_roman_name("전환(DT·Digital Transformation)", 2)] + #[case::korean_numeric_name("용어2(Version Two)", 2)] + #[case::year_with_roman_explanation("보고서 2023(MWC 2023)", 2)] + #[case::single_variable("함수(x)", 0)] + #[case::lowercase_expression("함수(x+1)", 0)] + #[case::uppercase_expression("식(A+B)", 0)] + #[case::separated_function("함수 f(x)", 0)] + fn recognizes_attached_korean_prose_parenthetical_span( + #[case] input: &str, + #[case] expected_matching_words: usize, + ) { + let ir = crate::rules::token::DocumentIR::parse(input, true); + let matching_indices = ir + .tokens + .iter() + .enumerate() + .filter_map(|(index, token)| { + matches!(token, Token::Word(_)) + .then(|| is_within_attached_korean_prose_parenthetical(&ir.tokens, index)) + .is_some_and(|matches| matches) + .then_some(index) + }) + .collect::>(); + + assert_eq!(matching_indices.len(), expected_matching_words); + for index in matching_indices { + let mut state = EncoderState::new(false); + assert!(matches!( + run(&ir.tokens, index, &mut state).unwrap(), + TokenAction::Noop + )); + } + } + + /// Korean rules 34 and 54 put the Korean opening parenthesis before the + /// Roman indicator; the math route instead starts with a two-cell prose + /// separator. Exercise each accepted body class at the public boundary. + #[rstest::rstest] + #[case::roman_followed_by_digit("용어(Web)3")] + #[case::roman_then_korean_explanation("기관(KRISS, 원장)")] + #[case::multiword_roman_name("전환(DT·Digital Transformation)")] + #[case::korean_numeric_name("용어2(Version Two)")] + #[case::year_with_roman_explanation("보고서 2023(MWC 2023)")] + fn attached_korean_prose_parentheses_keep_rule_34_order(#[case] input: &str) { + let encoded = crate::encode_to_unicode(input).expect("input must encode"); + assert!( + encoded.contains("⠦⠄⠴"), + "Korean opening parenthesis must precede Roman entry: {encoded}" + ); + } + + #[test] + fn attached_korean_name_keeps_specialized_anonymized_person_path() { + let ir = crate::rules::token::DocumentIR::parse("모A(61)씨", true); + let index = ir + .tokens + .iter() + .position(|token| matches!(token, Token::Word(_))) + .expect("fixture must contain a word"); + let mut state = EncoderState::new(true); + + assert!(matches!( + run(&ir.tokens, index, &mut state).unwrap(), + TokenAction::ReplaceMany(_) + )); + assert!( + crate::encode_to_unicode("모A(61)씨") + .expect("fixture must encode") + .contains("⠴⠠⠁⠦⠄⠼⠋⠁⠠⠴") + ); + } + + /// Decimal-context spacing recognizes each structural marker independently: + /// the parser sentinel, the Rule 12 ellipsis, and a combining math mark. + #[rstest::rstest] + #[case::unit_separator("a\u{001f}b", "ab", true)] + #[case::midline_ellipsis("a⋯b", "ab", true)] + #[case::combining_mark("ab", "a\u{0305}", true)] + #[case::plain_expression("a+b", "a+b", false)] + fn detects_decimal_context_spacing_markers( + #[case] text: &str, + #[case] chars: &str, + #[case] expected: bool, + ) { + assert_eq!( + needs_decimal_context_spacing(text, &chars.chars().collect::>()), + expected + ); + } + use crate::rules::token::{SpaceKind, WordMeta, WordToken}; use std::borrow::Cow; @@ -859,6 +2156,159 @@ mod tests { Token::Space(SpaceKind::Regular) } + #[rstest::rstest] + #[case::empty_segment("ISO//IEC")] + #[case::punctuation_only_segment("ISO/-./IEC")] + fn roman_slash_identifier_rejects_incomplete_segments(#[case] input: &str) { + assert!(!is_korean_prose_roman_slash_identifier( + &input.chars().collect::>() + )); + } + + #[test] + fn single_letter_slash_phrase_requires_letters_in_the_following_word() { + let tokens = vec![word_tok("H/W"), space_tok(), word_tok("((")]; + let chars = "H/W".chars().collect::>(); + + assert!(!is_korean_prose_single_letter_slash_phrase( + &tokens, 0, &chars + )); + } + + #[test] + fn multiword_parenthetical_tail_stops_at_a_non_word_boundary() { + let tokens = vec![ + Token::PreEncoded(vec![1]), + space_tok(), + word_tok("Alliance)"), + ]; + let Token::Word(tail) = &tokens[2] else { + unreachable!("fixture ends in a word") + }; + + assert!(!is_multiword_closed_roman_parenthetical_tail( + &tokens, 2, tail + )); + } + + #[test] + fn multiword_parenthetical_head_stops_at_a_non_word_boundary() { + let tokens = vec![ + word_tok("HCA(Home"), + space_tok(), + Token::PreEncoded(vec![1]), + ]; + let Token::Word(head) = &tokens[0] else { + unreachable!("fixture begins with a word") + }; + + assert!(!is_multiword_closed_roman_parenthetical_head( + &tokens, 0, head + )); + } + + #[test] + fn attached_prose_parenthetical_rejects_a_preencoded_body() { + let tokens = vec![word_tok("한국("), Token::PreEncoded(vec![1]), word_tok(")")]; + + assert!(!is_within_attached_korean_prose_parenthetical(&tokens, 1)); + } + + #[test] + fn attached_prose_parenthetical_ignores_mode_tokens_in_its_body() { + let tokens = vec![ + word_tok("한국("), + Token::Mode(crate::rules::token::ModeEvent::EnterEnglish), + word_tok("Web)"), + ]; + + assert!(is_within_attached_korean_prose_parenthetical(&tokens, 1)); + } + + #[rstest::rstest] + #[case::enclosed_roman_continuation("한글(ABC)-D", true)] + #[case::korean_prefix_before_initialism("기장-KBO", true)] + #[case::lowercase_math_variable("값-x", false)] + fn korean_roman_hyphen_suffix_is_classified_structurally( + #[case] input: &str, + #[case] expected: bool, + ) { + assert_eq!( + has_korean_prefix_roman_hyphen_suffix(&input.chars().collect::>()), + expected + ); + } + + #[rstest::rstest] + #[case::compact_unit("50bp", true)] + #[case::decimal_prefix("3.1p", true)] + #[case::ordinal("1st", true)] + #[case::mixed_case_name("25Project", true)] + #[case::digit_after_letter("3x3", true)] + #[case::capital_suffix("6G", true)] + #[case::trailing_punctuation("50bp,", true)] + #[case::letter_first("MP3", false)] + #[case::operator("3a+b", false)] + #[case::solidus("3/4", false)] + #[case::punctuation_before_letter("3.a", false)] + #[case::number_only("3", false)] + #[case::letters_only("abc", false)] + #[case::korean_suffix("3한", false)] + fn recognizes_numeric_prefix_roman_identifier_grammar( + #[case] text: &str, + #[case] expected: bool, + ) { + assert_eq!( + is_korean_prose_numeric_roman_identifier(&text.chars().collect::>()), + expected + ); + } + + #[rstest::rstest] + #[case::compact_unit("가는 50bp 인상", "50bp", true)] + #[case::decimal_prefix("가는 3.1p 표본", "3.1p", true)] + #[case::ordinal("가는 1st 항목", "1st", true)] + #[case::mixed_case_name("가는 25Project 자료", "25Project", true)] + #[case::digit_after_letter("가는 3x3 배열", "3x3", true)] + #[case::isolated_expression("3ab", "3ab", false)] + #[case::previous_product_cue("곱 3ab 결과", "3ab", false)] + #[case::next_value_cue("식은 3ab 값을", "3ab", false)] + #[case::explicit_latex("가는 $3ab$ 식", "$3ab$", false)] + fn numeric_roman_route_respects_korean_prose_and_math_context( + #[case] input: &str, + #[case] target: &str, + #[case] expected_noop: bool, + ) { + let ir = crate::rules::token::DocumentIR::parse(input, true); + let index = ir + .tokens + .iter() + .position(|token| matches!(token, Token::Word(word) if word.text.as_ref() == target)) + .expect("target word must be tokenized as one word"); + let mut state = EncoderState::new(true); + + assert_eq!( + matches!( + run(&ir.tokens, index, &mut state).unwrap(), + TokenAction::Noop + ), + expected_noop + ); + } + + /// The complete token-rule path must preserve both defensive boundaries: + /// an unsupported mixed-math glyph falls through, and a leading space with + /// no preceding math token is not treated as mixed-math continuation. + #[test] + fn unsupported_mixed_expression_after_leading_space_falls_through() { + let tokens = vec![space_tok(), word_tok("√분산🚀")]; + let mut state = EncoderState::new(false); + + let action = run(&tokens, 1, &mut state).unwrap(); + + assert!(matches!(action, TokenAction::Noop)); + } + // ---------- Direct tests on extracted helpers ---------- /// `prev_next_words` returns (None, None) for an out-of-range index. @@ -939,21 +2389,35 @@ mod tests { assert!(!next_word_starts_with_math_value_cue(&tokens, 0)); } - /// `is_logic_symbol_word` — XOR(⊻) 단독 토큰만 true, 그 외는 false. - /// Kills: `-> false`, `!=` mutations. + /// Only one complete rule-60/61 set or logic sign is accepted. #[rstest::rstest] #[case::xor_alone("⊻", true)] - #[case::wedge_alone("∧", false)] + #[case::wedge_alone("∧", true)] + #[case::membership_alone("∈", true)] + #[case::negation_alone("¬", true)] + #[case::ascii_plus("+", false)] #[case::xor_then_letter("⊻x", false)] #[case::empty_word("", false)] - fn is_logic_symbol_word_matches_only_xor(#[case] text: &'static str, #[case] expected: bool) { + fn set_or_logic_symbol_word_is_complete(#[case] text: &'static str, #[case] expected: bool) { let chars: Vec = text.chars().collect(); let word = WordToken { text: Cow::Borrowed(text), meta: WordMeta::from_chars(&chars), chars, }; - assert_eq!(is_logic_symbol_word(&word), expected); + assert_eq!(is_set_or_logic_symbol_word(&word), expected); + } + + /// Math rules 60-61: spaces do not turn a capital operand into UEB prose. + #[rstest::rstest] + #[case::upper_negation("A ¬ B", "⠠⠁⠀⠈⠔⠀⠠⠃")] + #[case::mixed_case_negation("p ¬ Q", "⠏⠀⠈⠔⠀⠠⠟")] + #[case::set_builder_membership("{x | x ∈ R}", "⠦⠂⠭⠀⠸⠳⠀⠭⠀⠖⠀⠠⠗⠐⠴")] + fn spaced_set_and_logic_operands_stay_math_variables( + #[case] input: &str, + #[case] expected: &str, + ) { + assert_eq!(crate::encode_to_unicode(input).as_deref(), Ok(expected)); } // ----- Lines 66-110: `a ≲ b:` colon-suffix math merge ----- diff --git a/libs/braillify/src/rules/token_rules/math_expression/detect.rs b/libs/braillify/src/rules/token_rules/math_expression/detect.rs index fcda3d30..dcb29578 100644 --- a/libs/braillify/src/rules/token_rules/math_expression/detect.rs +++ b/libs/braillify/src/rules/token_rules/math_expression/detect.rs @@ -100,6 +100,32 @@ pub(super) fn is_math_expression(chars: &[char], text: &str) -> bool { return false; } + // Korean rules 33 and 35: a letter-led Roman/number identifier remains + // Roman text when ordinary prose punctuation follows it. The punctuation + // alone must not turn `MP3`-shaped text into a mathematical expression. + if let Some((trailing, core)) = chars.split_last() + && matches!(*trailing, ',' | ';' | ':' | '.') + && core.first().is_some_and(|ch| ch.is_ascii_alphabetic()) + && core.iter().any(|ch| ch.is_ascii_digit()) + && core.iter().all(|ch| ch.is_ascii_alphanumeric()) + { + return false; + } + + // PDF 제33·34·69항: 숫자+로마자 단위와 바로 뒤의 종료표 생략 문장부호는 + // 수식이 아니라 하나의 국어 문장 내 단위 표기다. 일반 operator/symbol 판정보다 + // 먼저 배제해야 `173cm,` 같은 토큰이 comma 때문에 수식 경로로 우회하지 않는다. + if let Some(consumed) = + crate::rules::korean::rule_69::parse_numeric_ascii_unit_expression(chars) + && (consumed == chars.len() + || (consumed + 1 == chars.len() + && chars.get(consumed).is_some_and(|symbol| { + crate::english_logic::should_skip_terminator_for_symbol(*symbol) + }))) + { + return false; + } + // Slash-only numeric tokens: 2-part (N/M) is a fraction expression for any digit count; // 3-or-more parts (e.g. 2024/12/31) is a date/range and stays non-math. if !has_letters && chars.contains(&'/') && chars.iter().all(|c| c.is_ascii_digit() || *c == '/') @@ -242,13 +268,6 @@ pub(super) fn is_math_expression(chars: &[char], text: &str) -> bool { // Digit-then-letter transition at start of word (like "3ab" → math multiplication) // But NOT letter-then-digit (like "MP3" which is NOT math) if chars.len() >= 2 && chars[0].is_ascii_digit() { - // PDF 제69항: 숫자+단위 (180cm, 5kg, 1in 등)은 math가 아닌 단위 표기로 처리. - if let Some((_, _, consumed)) = - crate::rules::korean::rule_69::parse_numeric_ascii_unit_prefix(chars) - && consumed == chars.len() - { - return false; - } // PDF 제33항 — 학술 인용 형식: `YYYYa`, `YYYYa,`, `YYYYa;` (4자리+년도+단일 // 알파벳 suffix + 구두점). 이런 토큰은 수학 곱셈이 아닌 영어 모드 인용 // 표기이므로 math expression이 아니다. @@ -305,4 +324,34 @@ mod tests { let chars: Vec = "3}".chars().collect(); assert!(super::is_math_expression(&chars, "3}")); } + + /// Rules 33/34/69: punctuation which suppresses a Roman terminator remains + /// attached to the compact unit token without turning the unit into math. + #[rstest::rstest] + #[case::ordinary_unit("180cm", false)] + #[case::derived_kilometre("80km", false)] + #[case::derived_milligram("240mg", false)] + #[case::derived_kilowatt("30kW", false)] + #[case::derived_megahertz("96.7MHz", false)] + #[case::derived_hectare("15.2ha", false)] + #[case::unit_before_comma("173cm,", false)] + #[case::unit_before_closing_parenthesis("20kg)", false)] + #[case::unit_before_period("130m.", false)] + #[case::ambiguous_product_before_comma("3ab,", true)] + fn compact_rule69_unit_with_boundary_is_not_math(#[case] input: &str, #[case] expected: bool) { + let chars = input.chars().collect::>(); + assert_eq!(super::is_math_expression(&chars, input), expected); + } + + /// Rules 33/35: trailing prose punctuation preserves the same non-math + /// classification as the underlying letter-led Roman/number identifier. + #[rstest::rstest] + #[case::comma("MP3,")] + #[case::mixed_case_comma("RX350h,")] + #[case::colon("A4:")] + #[case::period("KF94.")] + fn roman_number_identifier_with_prose_punctuation_is_not_math(#[case] input: &str) { + let chars = input.chars().collect::>(); + assert!(!super::is_math_expression(&chars, input)); + } } diff --git a/libs/braillify/src/rules/token_rules/math_expression/helpers.rs b/libs/braillify/src/rules/token_rules/math_expression/helpers.rs index 32d70b0f..acfabeed 100644 --- a/libs/braillify/src/rules/token_rules/math_expression/helpers.rs +++ b/libs/braillify/src/rules/token_rules/math_expression/helpers.rs @@ -62,12 +62,45 @@ pub(super) fn is_middle_dot_numeric_word(chars: &[char]) -> bool { .iter() .filter(|c| matches!(**c, '\u{00B7}' | '\u{22C5}')) .count(); - if middle_dot_count != 1 { + if middle_dot_count == 0 { return false; } - chars + chars.iter().all(|c| { + c.is_ascii_digit() + || matches!( + *c, + '\u{00B7}' | '\u{22C5}' | '\u{2212}' | '-' | ',' | ';' | ':' + ) + }) +} + +/// Numeric notation which is written as ordinary Korean prose rather than as +/// a standalone mathematical expression. +/// +/// Korean Braille Rules 43, 47 [appendix], 48, and 50 keep the print order of +/// numeric slashes, decimal points, ranges, and middle-dot lists. Routing +/// these tokens through the math-expression layer only adds mathematical +/// delimiters; the character rules already emit the required repeated number +/// signs after `/`, `~`, and `·`. +pub(super) fn is_korean_prose_numeric_notation(chars: &[char]) -> bool { + let has_digit = chars.iter().any(|c| c.is_ascii_digit()); + let has_prose_separator = chars .iter() - .all(|c| c.is_ascii_digit() || matches!(*c, '\u{00B7}' | '\u{22C5}' | '\u{2212}' | '-')) + .any(|c| matches!(*c, '.' | '/' | '~' | '\u{00B7}' | '\u{22C5}')); + let starts_with_signed_minus = chars + .first() + .is_some_and(|c| matches!(*c, '-' | '\u{2212}')); + + has_digit + && has_prose_separator + && !starts_with_signed_minus + && chars.iter().all(|c| { + c.is_ascii_digit() + || matches!( + *c, + '.' | ',' | '-' | '\u{2212}' | '~' | '/' | '\u{00B7}' | '\u{22C5}' | ';' | ':' + ) + }) } pub(super) fn adjacent_korean_word_flags(tokens: &[Token<'_>], index: usize) -> (bool, bool) { @@ -345,6 +378,262 @@ fn build_korean_prefix_math_suffix(prefix: String, bytes: Vec) -> Vec Option { + if !chars.get(start).is_some_and(char::is_ascii_uppercase) || chars.get(start + 1) != Some(&'(') + { + return None; + } + + let mut cursor = start + 2; + let digit_start = cursor; + while chars.get(cursor).is_some_and(char::is_ascii_digit) { + cursor += 1; + } + if cursor == digit_start { + return None; + } + + match chars.get(cursor) { + Some('대') => cursor += 1, + Some('·' | 'ㆍ') + if chars + .get(cursor + 1) + .is_some_and(|ch| matches!(*ch, '여' | '남')) => + { + cursor += 2; + } + _ => {} + } + + (chars.get(cursor) == Some(&')')).then_some(cursor + 1) +} + +fn starts_anonymized_person_marker(chars: &[char], index: usize) -> bool { + chars + .get(index) + .is_some_and(|marker| matches!(*marker, '씨' | '군' | '양')) +} + +/// Whether the label is immediately followed by a Korean human-role noun. +/// +/// The role stem is separated from an attached case particle and classified +/// by its productive title/rank ending (`-사`, `-병`, `-감`, `-관`, `-장`, +/// `-원`). This covers ranks and occupations without enumerating corpus +/// phrases. Mathematical nouns such as `함수`, `변수`, and `값` do not have +/// one of these endings, so `A(14)함수는` remains on the math path. +fn starts_attached_korean_person_role(chars: &[char], index: usize) -> bool { + let suffix = chars[index..] + .iter() + .take_while(|ch| is_korean_char(**ch)) + .collect::(); + if suffix.is_empty() { + return false; + } + + const PARTICLES: &[&str] = &[ + "에게서", + "으로", + "에게", + "께서", + "에서", + "까지", + "부터", + "처럼", + "보다", + "라고", + "이라", + "이랑", + "하고", + "께", + "의", + "이", + "가", + "은", + "는", + "을", + "를", + "와", + "과", + "에", + "도", + "로", + ]; + let stem = PARTICLES + .iter() + .find_map(|particle| suffix.strip_suffix(particle)) + .unwrap_or(&suffix); + + stem.chars().count() >= 2 + && stem + .chars() + .last() + .is_some_and(|ending| matches!(ending, '사' | '병' | '감' | '관' | '장' | '원')) +} + +fn attached_korean_suffix_text(chars: &[char], index: usize) -> String { + chars[index..] + .iter() + .take_while(|ch| is_korean_char(**ch)) + .collect() +} + +fn starts_animate_dative_particle(chars: &[char], index: usize) -> bool { + attached_korean_suffix_text(chars, index).starts_with("에게") +} + +/// A Korean personal name can be printed directly before an anonymizing Roman +/// label, for example `조너선M(41)이`. In that structure a following case +/// particle resolves the `M(41)` function-notation ambiguity. Require a +/// three-syllable-or-longer attached Korean prefix; ordinary mathematical +/// heads such as `함수A(14)는` therefore remain math-owned. +fn has_attached_korean_name_and_case_particle(chars: &[char], start: usize, end: usize) -> bool { + let korean_prefix_len = chars[..start] + .iter() + .rev() + .take_while(|ch| is_korean_char(**ch)) + .count(); + if korean_prefix_len < 3 { + return false; + } + + matches!( + attached_korean_suffix_text(chars, end).as_str(), + "이" | "가" | "은" | "는" | "을" | "를" | "와" | "과" | "의" | "에" + ) +} + +/// A list can defer its person marker to the final member, as in +/// `B(60)·C(41)씨`. Every preceding age label is still Roman prose. Only a +/// middle-dot chain whose eventual member has an explicit person marker is +/// accepted, so an algebraic `A(1)·B(2)` remains mathematical. +fn anonymized_person_chain_has_marker(chars: &[char], mut cursor: usize) -> bool { + while chars.get(cursor) == Some(&'·') { + let next_start = cursor + 1; + let Some(next_end) = anonymized_person_label_end(chars, next_start) else { + return false; + }; + if starts_anonymized_person_marker(chars, next_end) { + return true; + } + cursor = next_end; + } + false +} + +fn anonymized_person_label_span(chars: &[char]) -> Option<(usize, usize)> { + let mut index = 0usize; + while index < chars.len() { + if !chars[index].is_ascii_uppercase() + || index + .checked_sub(1) + .and_then(|previous| chars.get(previous)) + .is_some_and(|previous| previous.is_ascii_alphanumeric()) + || chars.get(index + 1) != Some(&'(') + { + index += 1; + continue; + } + + if let Some(end) = anonymized_person_label_end(chars, index) + && (starts_anonymized_person_marker(chars, end) + || starts_attached_korean_person_role(chars, end) + || starts_animate_dative_particle(chars, end) + || has_attached_korean_name_and_case_particle(chars, index, end) + || anonymized_person_chain_has_marker(chars, end)) + { + return Some((index, end)); + } + index += 1; + } + None +} + +pub(super) fn encode_anonymized_person_label(chars: &[char]) -> Option> { + let (&letter, _) = chars.split_first()?; + if anonymized_person_label_end(chars, 0) != Some(chars.len()) { + return None; + } + + let mut encoded = vec![ + crate::rules::korean::rule_29::ROMAN_INDICATOR, + crate::rules::korean::rule_28::UPPERCASE_SINGLE, + crate::english::encode_english(letter).ok()?, + ]; + let parenthetical = chars[1..].iter().collect::(); + encoded.extend(crate::encode(&parenthetical).ok()?); + Some(encoded) +} + +pub(super) fn split_anonymized_person_label(chars: &[char]) -> Option>> { + let mut replacement = Vec::new(); + let mut cursor = 0usize; + let mut found = false; + + while cursor < chars.len() { + let Some((relative_start, relative_end)) = anonymized_person_label_span(&chars[cursor..]) + else { + break; + }; + let start = cursor + relative_start; + let end = cursor + relative_end; + let encoded = encode_anonymized_person_label(&chars[start..end])?; + if start > cursor { + replacement.push(build_word_token(chars[cursor..start].iter().collect())); + } + replacement.push(Token::PreEncoded(encoded)); + cursor = end; + found = true; + } + + if !found { + return None; + } + if cursor < chars.len() { + replacement.push(build_word_token(chars[cursor..].iter().collect())); + } + Some(replacement) +} + +/// Recognize only the suffix shape used after an already-confirmed Korean +/// prefix. Korean rule 34's PDF example is `링컨(Lincoln)은`: Roman text may +/// be enclosed in a bracket without a Roman terminator. Rule 54 requires the +/// bracket to attach to its contents, so a comma or period after the closing +/// bracket does not turn that Roman annotation into mathematics. +/// +/// This predicate is intentionally not part of the global math detector, whose +/// existing results for standalone `(x)`, `(A)`, and `(abc)` stay unchanged. +/// The caller below must first prove that all preceding characters are Korean. +fn is_closed_roman_annotation_suffix(chars: &[char]) -> bool { + if chars.first() != Some(&'(') { + return false; + } + + let Some(close) = chars.iter().position(|c| *c == ')') else { + return false; + }; + let body = &chars[1..close]; + let trailing = &chars[close + 1..]; + + !body.is_empty() + && body.iter().any(|c| c.is_ascii_alphabetic()) + && body + .iter() + .all(|c| c.is_ascii_alphanumeric() || matches!(*c, '-' | '\'' | '.')) + && trailing + .iter() + .all(|c| matches!(*c, ',' | '.' | ';' | ':' | '!' | '?' | '\'' | '"')) +} + pub(super) fn split_mixed_math_word( word: &crate::rules::token::WordToken<'_>, leading_delimiter_len: usize, @@ -354,6 +643,10 @@ pub(super) fn split_mixed_math_word( return None; } + if let Some(replacement) = split_anonymized_person_label(&word.chars) { + return Some(replacement); + } + let chars = &word.chars; let len = chars.len(); @@ -391,6 +684,9 @@ pub(super) fn split_mixed_math_word( if !prefix_all_korean || !suffix_no_korean { return None; } + if is_closed_roman_annotation_suffix(suffix_chars) { + return None; + } let suffix_text: String = suffix_chars.iter().collect(); let suffix_is_math = is_mixed_math_expression(suffix_chars, &suffix_text) || is_math_expression(suffix_chars, &suffix_text); @@ -409,9 +705,96 @@ pub(super) fn split_mixed_math_word( #[cfg(test)] mod tests { use super::*; + + #[rstest::rstest] + #[case::official_rule_34_annotation("(Lincoln)", true)] + #[case::missing_opening("Lincoln)", false)] + #[case::missing_closing("(Lincoln", false)] + #[case::empty_body("()", false)] + fn recognizes_only_closed_roman_annotation_suffixes( + #[case] input: &str, + #[case] expected: bool, + ) { + assert_eq!( + is_closed_roman_annotation_suffix(&input.chars().collect::>()), + expected + ); + } use crate::rules::math::math_token_rule::MathContext; use crate::rules::token::SpaceKind; + #[rstest::rstest] + #[case::adult("A(54)씨는", true)] + #[case::minor_male("B(17)군에게", true)] + #[case::minor_female("C(16)양은", true)] + #[case::gender_annotation("A(41·여)씨는", true)] + #[case::age_decade("B(30대)씨는", true)] + #[case::korean_name_prefix("김모A(41)씨", true)] + #[case::military_rank("A(21)상병을", true)] + #[case::police_rank("B(42)경사가", true)] + #[case::occupation("C(47)원사에게", true)] + #[case::animate_dative("A(30)에게", true)] + #[case::attached_korean_name("조너선M(41)이", true)] + #[case::math_function_particle("A(14)는", false)] + #[case::attached_math_function("함수A(14)는", false)] + #[case::math_function_noun("A(14)함수는", false)] + #[case::non_honorific_syllable("A(14)시는", false)] + #[case::missing_digits("A()씨", false)] + fn recognizes_only_anonymized_person_labels(#[case] input: &str, #[case] expected: bool) { + assert_eq!( + anonymized_person_label_span(&input.chars().collect::>()).is_some(), + expected + ); + } + + #[rstest::rstest] + #[case::adult("A(54)씨는")] + #[case::minor_male("B(17)군에게")] + #[case::minor_female("C(16)양은")] + #[case::gender_annotation("A(41·여)씨는")] + #[case::age_decade("B(30대)씨는")] + fn anonymized_person_labels_use_korean_prose_cells(#[case] input: &str) { + let chars = input.chars().collect::>(); + let word = WordToken { + text: Cow::Borrowed(input), + chars: chars.clone(), + meta: WordMeta::from_chars(&chars), + }; + + let replacement = split_mixed_math_word(&word, 0, MathContext::default()) + .expect("honorific resolves the function/prose ambiguity"); + assert!(matches!( + replacement.as_slice(), + [Token::PreEncoded(_), Token::Word(_)] + )); + let Token::PreEncoded(label) = &replacement[0] else { + unreachable!(); + }; + let (start, end) = anonymized_person_label_span(&chars).expect("label span"); + let expected = encode_anonymized_person_label(&chars[start..end]).expect("label cells"); + assert_eq!(label, &expected); + } + + #[rstest::rstest] + #[case::deferred_marker("B(60)·C(41)씨", 2)] + #[case::child_markers("B(6)군·C(3)양이", 2)] + #[case::three_people("A(41)씨·B(28)씨와C(27)씨가", 3)] + fn splits_every_anonymized_person_label_in_one_token( + #[case] input: &str, + #[case] expected_labels: usize, + ) { + let chars = input.chars().collect::>(); + let replacement = split_anonymized_person_label(&chars).expect("person-label list"); + + assert_eq!( + replacement + .iter() + .filter(|token| matches!(token, Token::PreEncoded(_))) + .count(), + expected_labels + ); + } + /// helpers:235 — `try_encode_math_slice` fallback to `crate::encode` when /// math encoder fails. Use `f(~)`: passes `has_function_call` candidacy /// (1-letter + `(`) and is_math_expression, but math encoder rejects `~`. @@ -431,6 +814,33 @@ mod tests { assert!(result.is_none()); } + /// The PDF's `√분산` form is a mixed expression because the radical is + /// directly attached to Korean text; it must produce a concrete cell sequence. + #[test] + fn try_encode_mixed_math_slice_encodes_valid_expression() { + let chars = std::hint::black_box("√분산").chars().collect::>(); + let result = try_encode_mixed_math_slice(&chars, MathContext::default()); + + assert!(result.is_some()); + } + + #[test] + fn mixed_fraction_detects_korean_inside_the_parenthesized_operand() { + let text = "2/(삼+오)"; + let chars = text.chars().collect::>(); + + assert!(is_mixed_math_expression(&chars, text)); + } + + #[test] + fn anonymized_person_chain_rejects_invalid_or_unmarked_following_labels() { + let invalid = "A(1)·not".chars().collect::>(); + assert!(!anonymized_person_chain_has_marker(&invalid, 4)); + + let unmarked = "A(1)·B(2)·C(3)".chars().collect::>(); + assert!(!anonymized_person_chain_has_marker(&unmarked, 4)); + } + #[test] fn try_encode_mixed_math_prefix_encodes_math_prefix_before_korean_suffix() { let prefix: Vec = "x²".chars().collect(); diff --git a/libs/braillify/src/rules/token_rules/middle_dot_spacing.rs b/libs/braillify/src/rules/token_rules/middle_dot_spacing.rs index 4be011c9..63e353e0 100644 --- a/libs/braillify/src/rules/token_rules/middle_dot_spacing.rs +++ b/libs/braillify/src/rules/token_rules/middle_dot_spacing.rs @@ -1,8 +1,69 @@ -use crate::rules::token::Token; +use std::borrow::Cow; + +use crate::rules::token::{Token, WordMeta, WordToken}; use crate::rules::token_rule::{TokenAction, TokenPhase, TokenRule}; pub struct MiddleDotSpacingRule; +fn previous_word<'a, 'b>(tokens: &'b [Token<'a>], index: usize) -> Option<&'b WordToken<'a>> { + tokens[..index] + .iter() + .rev() + .find_map(|token| match token { + Token::Mode(_) => None, + Token::Word(word) => Some(Some(word)), + _ => Some(None), + }) + .flatten() +} + +fn next_word<'a, 'b>(tokens: &'b [Token<'a>], index: usize) -> Option<(usize, &'b WordToken<'a>)> { + tokens + .iter() + .enumerate() + .skip(index + 1) + .find_map(|(token_index, token)| match token { + Token::Mode(_) | Token::Space(_) => None, + Token::Word(word) => Some(Some((token_index, word))), + _ => Some(None), + }) + .flatten() +} + +/// Rules 51 and 59 attach a Korean colon/semicolon to the item on its left. +/// A spaced colon between two Roman/number items remains UEB print spacing, +/// so require a Korean item on either side of the punctuation boundary. +fn space_precedes_korean_colon_or_semicolon( + tokens: &[Token<'_>], + index: usize, + previous: &WordToken<'_>, +) -> bool { + let Some((punctuation_index, punctuation)) = next_word(tokens, index) else { + return false; + }; + if !punctuation + .chars + .first() + .is_some_and(|symbol| matches!(symbol, ':' | ';')) + || punctuation.chars.len() != 1 + { + return false; + } + + previous + .chars + .iter() + .rev() + .find(|ch| ch.is_ascii_alphanumeric() || crate::utils::is_korean_char(**ch)) + .is_some_and(|ch| crate::utils::is_korean_char(*ch)) + || next_word(tokens, punctuation_index).is_some_and(|(_, word)| { + word.chars + .iter() + .find(|ch| ch.is_ascii_alphanumeric() || crate::utils::is_korean_char(**ch)) + .is_some_and(|ch| crate::utils::is_korean_char(*ch)) + }) +} + impl TokenRule for MiddleDotSpacingRule { fn phase(&self) -> TokenPhase { TokenPhase::PostWord @@ -18,17 +79,48 @@ impl TokenRule for MiddleDotSpacingRule { index: usize, _state: &mut crate::rules::context::EncoderState, ) -> Result, String> { + // Merge a one-sided editorial space at the token boundary so the + // middle dot is encoded with the same character context as canonical + // `정치·경제`, not merely emitted as an adjacent second word. + if let Some(Token::Word(left)) = tokens.get(index) + && matches!(tokens.get(index + 1), Some(Token::Space(_))) + && let Some(Token::Word(right)) = tokens.get(index + 2) + && (left.chars.last() == Some(&'·') || right.chars.first() == Some(&'·')) + { + let text = format!("{}{}", left.text, right.text); + let chars = text.chars().collect::>(); + return Ok(TokenAction::ReplaceRange( + 3, + vec![Token::Word(WordToken { + text: Cow::Owned(text), + chars: chars.clone(), + meta: WordMeta::from_chars(&chars), + })], + )); + } + let Some(Token::Space(_)) = tokens.get(index) else { return Ok(TokenAction::Noop); }; - let Some(Token::Word(prev)) = index.checked_sub(1).and_then(|i| tokens.get(i)) else { + let Some(prev) = previous_word(tokens, index) else { return Ok(TokenAction::Noop); }; - let Some(Token::Word(next)) = tokens.get(index + 1) else { + let Some((_, next)) = next_word(tokens, index) else { return Ok(TokenAction::Noop); }; + // Korean rule 50: the middle dot is attached on both sides. Its print + // source sometimes contains editorial spaces, but the braille spacing + // is still canonicalized by the rule. + if prev.chars.last() == Some(&'·') || next.chars.first() == Some(&'·') { + return Ok(TokenAction::ReplaceMany(vec![])); + } + + if space_precedes_korean_colon_or_semicolon(tokens, index, prev) { + return Ok(TokenAction::ReplaceMany(vec![])); + } + let prev_text = prev.text.as_ref(); let next_text = next.text.as_ref(); @@ -45,3 +137,46 @@ impl TokenRule for MiddleDotSpacingRule { Ok(TokenAction::Noop) } } + +#[cfg(test)] +mod tests { + use super::*; + + /// Korean rules 50, 51, and 59 determine braille spacing even when the + /// print source contains editorial spaces around the punctuation. + #[rstest::rstest] + #[case::middle_dot_both_sides("정치 · 경제", "정치·경제")] + #[case::middle_dot_left("정치 ·경제", "정치·경제")] + #[case::middle_dot_right("정치· 경제", "정치·경제")] + #[case::korean_colon("제목 : 내용", "제목: 내용")] + #[case::roman_to_korean_colon("WHO : 세계", "WHO: 세계")] + #[case::korean_semicolon("채소 ; 과일", "채소; 과일")] + fn canonical_korean_punctuation_spacing(#[case] spaced: &str, #[case] canonical: &str) { + assert_eq!(crate::encode(spaced), crate::encode(canonical)); + } + + /// Rule 32 leaves print spacing inside a Roman section to UEB. A Korean + /// prefix earlier in the token does not turn `FAPAS : Food` into a Korean + /// colon boundary because the immediately preceding item is Roman. + #[test] + fn attached_roman_item_preserves_space_before_ueb_colon() { + assert_ne!( + crate::encode("설명(FAPAS : Food)"), + crate::encode("설명(FAPAS: Food)") + ); + } + + #[test] + fn colon_spacing_probe_returns_false_when_no_punctuation_word_follows() { + let mut ir = crate::rules::token::DocumentIR::parse("한국", false); + ir.tokens + .push(Token::Space(crate::rules::token::SpaceKind::Regular)); + let Token::Word(previous) = &ir.tokens[0] else { + unreachable!("fixture begins with a word") + }; + + assert!(!space_precedes_korean_colon_or_semicolon( + &ir.tokens, 1, previous + )); + } +} diff --git a/libs/braillify/src/rules/token_rules/normalize.rs b/libs/braillify/src/rules/token_rules/normalize.rs index 06b48b89..231ddd3f 100644 --- a/libs/braillify/src/rules/token_rules/normalize.rs +++ b/libs/braillify/src/rules/token_rules/normalize.rs @@ -3,6 +3,161 @@ use std::borrow::Cow; use crate::rules::token::{Token, WordToken}; use crate::rules::token_rule::{TokenAction, TokenPhase, TokenRule}; +/// Normalize the ASCII glyph substitutes `<` and `>` to the single angle +/// brackets defined by Korean Braille Standard Article 49 when they form a +/// balanced prose enclosure in a Korean document. +/// +/// U+003C/U+003E still retain their mathematical comparison meaning in an +/// actual relation (`xz`). News and publishing text commonly substitutes +/// the ASCII glyphs for U+3008/U+3009 around titles, including enclosures that +/// span print spaces, so the decision has to be document-wide rather than +/// character-local. +pub struct NormalizeAsciiAngleBrackets; + +#[derive(Clone, Copy)] +struct FlatChar { + token_index: usize, + char_index: usize, + ch: char, +} + +fn flattened_chars(tokens: &[Token<'_>]) -> Vec { + let mut flattened = Vec::new(); + for (token_index, token) in tokens.iter().enumerate() { + match token { + Token::Word(word) => flattened.extend(word.chars.iter().copied().enumerate().map( + |(char_index, ch)| FlatChar { + token_index, + char_index, + ch, + }, + )), + Token::Space(_) => flattened.push(FlatChar { + token_index, + char_index: usize::MAX, + ch: ' ', + }), + Token::Fraction(_) | Token::Mode(_) | Token::PreEncoded(_) => {} + } + } + flattened +} + +fn is_simple_relation_operand(chars: &[FlatChar]) -> bool { + let visible = chars + .iter() + .filter(|item| !item.ch.is_whitespace()) + .map(|item| item.ch) + .collect::>(); + if visible.is_empty() || !visible.iter().all(char::is_ascii_alphanumeric) { + return false; + } + + visible.iter().all(char::is_ascii_digit) + || (visible.iter().all(char::is_ascii_alphabetic) && visible.len() <= 2) +} + +fn is_chained_comparison(flattened: &[FlatChar], open: usize, close: usize) -> bool { + let left = flattened[..open] + .iter() + .rev() + .find(|item| !item.ch.is_whitespace()) + .map(|item| item.ch); + let right = flattened[close + 1..] + .iter() + .find(|item| !item.ch.is_whitespace()) + .map(|item| item.ch); + + left.is_some_and(|ch| ch.is_ascii_alphanumeric()) + && right.is_some_and(|ch| ch.is_ascii_alphanumeric()) + && is_simple_relation_operand(&flattened[open + 1..close]) +} + +fn ascii_angle_replacements(tokens: &[Token<'_>]) -> Vec<(usize, usize, char)> { + if !tokens + .iter() + .any(|token| matches!(token, Token::Word(word) if word.meta.has_korean)) + { + return Vec::new(); + } + + let flattened = flattened_chars(tokens); + let mut openings = Vec::new(); + let mut replacements = Vec::new(); + + for (index, item) in flattened.iter().enumerate() { + match item.ch { + '<' | '〈' => openings.push(index), + '>' | '〉' => { + let Some(open) = openings.pop() else { + continue; + }; + if open + 1 == index || is_chained_comparison(&flattened, open, index) { + continue; + } + + let opening = flattened[open]; + if opening.ch == '<' { + replacements.push((opening.token_index, opening.char_index, '〈')); + } + if item.ch == '>' { + replacements.push((item.token_index, item.char_index, '〉')); + } + } + _ => {} + } + } + + replacements +} + +impl TokenRule for NormalizeAsciiAngleBrackets { + fn phase(&self) -> TokenPhase { + TokenPhase::Normalization + } + + fn priority(&self) -> u16 { + 90 + } + + fn apply<'a>( + &self, + tokens: &[Token<'a>], + index: usize, + _state: &mut crate::rules::context::EncoderState, + ) -> Result, String> { + let Some(Token::Word(word)) = tokens.get(index) else { + return Ok(TokenAction::Noop); + }; + if !word.chars.iter().any(|ch| matches!(ch, '<' | '>')) { + return Ok(TokenAction::Noop); + } + + let replacements = ascii_angle_replacements(tokens); + let mut chars = word.chars.clone(); + let mut changed = false; + for (_, char_index, replacement) in replacements + .into_iter() + .filter(|(token_index, _, _)| *token_index == index) + { + if let Some(ch) = chars.get_mut(char_index) { + *ch = replacement; + changed = true; + } + } + if !changed { + return Ok(TokenAction::Noop); + } + + let normalized = chars.iter().collect::(); + Ok(TokenAction::Replace(Token::Word(WordToken { + text: Cow::Owned(normalized), + meta: crate::rules::token::WordMeta::from_chars(&chars), + chars, + }))) + } +} + pub struct NormalizeEllipsis; impl TokenRule for NormalizeEllipsis { @@ -42,3 +197,57 @@ impl TokenRule for NormalizeEllipsis { }))) } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::rules::token::{DocumentIR, SpaceKind}; + use crate::rules::token_engine::TokenRuleEngine; + + fn normalize(input: &str) -> String { + let mut ir = DocumentIR::parse(input, false); + let mut engine = TokenRuleEngine::new(); + engine.register(Box::new(NormalizeAsciiAngleBrackets)); + engine + .apply_all(&mut ir.tokens, &mut ir.state) + .expect("normalization must succeed"); + + ir.tokens + .iter() + .map(|token| match token { + Token::Word(word) => word.chars.iter().collect::(), + Token::Space(SpaceKind::Regular) => " ".to_string(), + Token::Fraction(_) | Token::Mode(_) | Token::PreEncoded(_) => String::new(), + }) + .collect() + } + + #[rstest::rstest] + #[case::title_at_start("<제목>을 읽다", "〈제목〉을 읽다")] + #[case::attached_title("책<긴 제목>이다", "책〈긴 제목〉이다")] + #[case::score_tiebreak("경기 7-6<7-3> 2-6", "경기 7-6〈7-3〉 2-6")] + #[case::roman_title("영화 이다", "영화 〈Das Boot〉이다")] + #[case::comparison_chain("식 xz이다", "식 xz이다")] + #[case::single_comparison("식 x Option<(&str, &str)> { - for suffix in AUX_VERB_SUFFIXES { - if let Some(prefix) = text.strip_suffix(suffix) - && !prefix.is_empty() - && prefix.chars().any(crate::utils::is_korean_char) - { - return Some((prefix, *suffix)); - } - } - None -} - impl TokenRule for KoreanAuxiliaryVerbSpacingRule { fn phase(&self) -> TokenPhase { TokenPhase::Normalization } fn priority(&self) -> u16 { - 50 // Word_shortcut(100)·LaTeX(110+)보다 먼저 분리 + 50 // Registry compatibility; no normalization is performed. } fn apply<'a>( &self, - tokens: &[Token<'a>], - index: usize, + _tokens: &[Token<'a>], + _index: usize, _state: &mut crate::rules::context::EncoderState, ) -> Result, String> { - let Some(Token::Word(word)) = tokens.get(index) else { - return Ok(TokenAction::Noop); - }; - - if !word.meta.has_korean { - return Ok(TokenAction::Noop); - } - - let text = word.text.as_ref(); - let Some((prefix, suffix)) = split_aux_verb(text) else { - return Ok(TokenAction::Noop); - }; - - let prefix_owned = prefix.to_string(); - let suffix_owned = suffix.to_string(); - let prefix_chars: Vec = prefix_owned.chars().collect(); - let suffix_chars: Vec = suffix_owned.chars().collect(); - - Ok(TokenAction::ReplaceMany(vec![ - Token::Word(WordToken { - text: Cow::Owned(prefix_owned), - chars: prefix_chars.clone(), - meta: WordMeta::from_chars(&prefix_chars), - }), - Token::Space(SpaceKind::Regular), - Token::Word(WordToken { - text: Cow::Owned(suffix_owned), - chars: suffix_chars.clone(), - meta: WordMeta::from_chars(&suffix_chars), - }), - ])) + Ok(TokenAction::Noop) } } @@ -132,3 +77,30 @@ impl TokenRule for AsteriskSpacingRule { Ok(TokenAction::ReplaceMany(replacement)) } } + +#[cfg(test)] +mod tests { + /// 제49항은 묵자의 띄어쓰기를 따르며, 각 spaced 입력은 PDF에 그대로 + /// 실린 예제다. 대응 attached 입력에서는 없는 공백을 새로 만들지 않는다. + #[rstest::rstest] + #[case::rule18("그림을 그리고 있다.", "그림을 그리고있다.")] + #[case::rule29( + "그녀는 Los Angeles의 한인 타운에 살고 있다.", + "그녀는 Los Angeles의 한인 타운에 살고있다." + )] + #[case::rule36( + "가영이는 미적분학 II 과목을 수강하고 있다.", + "가영이는 미적분학 II 과목을 수강하고있다." + )] + fn full_encoder_preserves_printed_auxiliary_spacing_only( + #[case] spaced: &str, + #[case] attached: &str, + ) { + let spaced_output = crate::encode_to_unicode(spaced).expect("PDF example must encode"); + let attached_output = crate::encode_to_unicode(attached).expect("control must encode"); + let spaced_blanks = spaced_output.chars().filter(|cell| *cell == '⠀').count(); + let attached_blanks = attached_output.chars().filter(|cell| *cell == '⠀').count(); + + assert_eq!(spaced_blanks, attached_blanks + 1); + } +} diff --git a/libs/braillify/src/rules/token_rules/uppercase_passage.rs b/libs/braillify/src/rules/token_rules/uppercase_passage.rs index 83732bc6..145a6103 100644 --- a/libs/braillify/src/rules/token_rules/uppercase_passage.rs +++ b/libs/braillify/src/rules/token_rules/uppercase_passage.rs @@ -1,4 +1,6 @@ -use crate::rules::english_shortform::requires_grade1_indicator; +use crate::rules::english_shortform::{ + permits_grade1_boundary_after_run, requires_grade1_indicator, +}; use crate::rules::token::{ModeEvent, Token, WordToken}; use crate::rules::token_rule::{TokenAction, TokenPhase, TokenRule}; @@ -37,8 +39,153 @@ fn next_two_words<'a>( (first, second) } -fn is_ascii_word(word: &WordToken) -> bool { - word.text.chars().all(|c| c.is_ascii_alphabetic()) +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +struct CapitalizedGroup { + /// First affected ASCII capital. Opening punctuation before this position + /// is outside capitals mode (UEB §8.5 placement). + start: usize, + /// First character outside the affected symbols-sequence. A Korean gloss + /// or closing quote attached to the final word begins here. + end: usize, +} + +fn is_opening_passage_punctuation(ch: char) -> bool { + matches!( + ch, + '\'' | '"' | '\u{2018}' | '\u{201c}' | '(' | '[' | '{' | '〈' | '《' | '「' | '『' + ) +} + +fn is_closing_passage_quote(ch: char) -> bool { + matches!( + ch, + '\'' | '"' | '\u{2019}' | '\u{201d}' | '〉' | '》' | '」' | '』' + ) +} + +/// Locate one whitespace-delimited capitalised symbols-sequence. +/// +/// `DocumentIR` deliberately preserves print whitespace and therefore keeps +/// punctuation and a Korean gloss attached to the same `WordToken` (`‘BET`, +/// `ME’이다`, `COSMO(코스모)`). UEB §8.5 places the passage indicators inside +/// opening/closing punctuation and before a following non-Roman gloss, so the +/// token rule needs the precise affected slice instead of asking whether the +/// entire token is ASCII. +fn capitalized_group(word: &WordToken<'_>) -> Option { + if word.chars.iter().any(char::is_ascii_lowercase) { + return None; + } + + let start = word.chars.iter().position(char::is_ascii_uppercase)?; + if !word.chars[..start] + .iter() + .copied() + .all(is_opening_passage_punctuation) + { + return None; + } + + let last_capital = word.chars.iter().rposition(char::is_ascii_uppercase)?; + if word.chars[start..=last_capital] + .iter() + .any(|ch| crate::utils::is_korean_char(*ch)) + { + return None; + } + + let mut end = word.chars.len(); + for index in last_capital + 1..word.chars.len() { + let ch = word.chars[index]; + let opens_attached_korean_gloss = matches!(ch, '(' | '[' | '{') + && word.chars[index + 1..] + .iter() + .any(|next| crate::utils::is_korean_char(*next)); + if crate::utils::is_korean_char(ch) + || is_closing_passage_quote(ch) + || opens_attached_korean_gloss + { + end = index; + break; + } + } + + Some(CapitalizedGroup { start, end }) +} + +fn owned_word<'a>(chars: &[char]) -> Token<'a> { + let text = chars.iter().collect::(); + Token::Word(WordToken { + text: std::borrow::Cow::Owned(text), + chars: chars.to_vec(), + meta: crate::rules::token::WordMeta::from_chars(chars), + }) +} + +/// A separated uppercase unit is emitted atomically by Korean rule 69, +/// including its Roman and capitalization indicators. Do not pre-emit the +/// generic UEB word prefix for the same letters. +fn is_separated_rule_69_unit(tokens: &[Token<'_>], index: usize, word: &WordToken<'_>) -> bool { + let Some(previous) = prev_word(tokens, index) else { + return false; + }; + let previous_is_number = previous.chars.iter().any(char::is_ascii_digit) + && previous + .chars + .iter() + .all(|ch| ch.is_ascii_digit() || matches!(ch, ',' | '.')); + previous_is_number + && crate::rules::korean::rule_69::complete_ascii_unit_len(&word.chars, 0).is_some() +} + +fn is_single_capital_comma_item(word: &WordToken<'_>) -> bool { + word.chars.first().is_some_and(char::is_ascii_uppercase) + && word.chars.get(1) == Some(&',') + && word + .chars + .iter() + .filter(|ch| ch.is_ascii_uppercase()) + .count() + == 1 +} + +/// 수학 제12항 [붙임 1]의 국어 문장 안 로마자 변수 나열은 각 변수를 +/// 독립된 로마자 항목으로 점역한다 (`세 점 A, B, C가 있다.`). 겉모양만 +/// 보면 UEB 8.5의 세 대문자 symbols-sequence와 같으므로, 앞의 국어 문맥과 +/// 마지막 변수에 붙은 국어 조사를 함께 확인해 대문자 구절로 오인하지 않는다. +/// 문자의 이름은 열거하지 않고 동일한 단일 대문자 콤마 나열 전체에 적용한다. +fn is_korean_math_letter_list_start( + tokens: &[Token<'_>], + index: usize, + word: &WordToken<'_>, + upcoming_first: Option<&WordToken<'_>>, + upcoming_second: Option<&WordToken<'_>>, +) -> bool { + let previous_is_korean = + prev_word(tokens, index).is_some_and(|previous| previous.meta.has_korean); + let Some(first) = upcoming_first else { + return false; + }; + let Some(second) = upcoming_second else { + return false; + }; + let second_group = capitalized_group(second); + let second_has_one_capital = second + .chars + .iter() + .filter(|ch| ch.is_ascii_uppercase()) + .count() + == 1; + let second_has_attached_korean = second_group.is_some_and(|group| { + second.chars[group.end..] + .iter() + .any(|ch| crate::utils::is_korean_char(*ch)) + }); + + previous_is_korean + && is_single_capital_comma_item(word) + && is_single_capital_comma_item(first) + && second_has_one_capital + && second_has_attached_korean } impl TokenRule for UppercasePassageRule { @@ -66,40 +213,93 @@ impl TokenRule for UppercasePassageRule { let (upcoming_first, upcoming_second) = next_two_words(tokens, index); let word_len = word.chars.len(); let ascii_starts_at_beginning = word.meta.starts_with_ascii; + let capitalized = capitalized_group(word); let needs_inline_entry = state.english_indicator && !state.is_english && word.meta.has_ascii_alphabetic - && ascii_starts_at_beginning; + && capitalized.is_some(); + + let upcoming_first_group = upcoming_first.and_then(capitalized_group); + let upcoming_second_group = upcoming_second.and_then(capitalized_group); + let is_korean_math_letter_list = + is_korean_math_letter_list_start(tokens, index, word, upcoming_first, upcoming_second); + let can_start_passage = capitalized.is_some_and(|group| group.end == word_len) + && upcoming_first + .zip(upcoming_first_group) + .is_some_and(|(next, group)| group.start == 0 && group.end == next.chars.len()) + && upcoming_second_group.is_some_and(|group| group.start == 0) + && !is_korean_math_letter_list + && !is_separated_rule_69_unit(tokens, index, word); - if word.meta.is_all_uppercase && !state.triple_big_english && ascii_starts_at_beginning { + if can_start_passage && !state.triple_big_english { + let group = capitalized.expect("passage start has a capitalized group"); + let mut replacement = Vec::new(); + if group.start > 0 { + replacement.push(owned_word(&word.chars[..group.start])); + } if needs_inline_entry { let entry = if state.needs_english_continuation { ModeEvent::EnterEnglishContinue } else { ModeEvent::EnterEnglish }; - prefix.push(Token::Mode(entry)); + replacement.push(Token::Mode(entry)); state.is_english = true; state.needs_english_continuation = false; } - let prev_ascii = prev_word(tokens, index).is_some_and(is_ascii_word); - let can_start_passage = (!state.has_processed_word || !prev_ascii) - && upcoming_first.is_some_and(is_ascii_word) - && upcoming_second.is_some_and(is_ascii_word); + // UEB §5.7.2 + §10.9: inspect the initial maximal ASCII-capital + // letters-sequence rather than the entire whitespace token. Korean + // text or punctuation attached after that run is its boundary, not + // part of the UEB shortform-collision decision (`AC밀란`, `CD,`). + let uppercase_run_len = word + .chars + .iter() + .skip(group.start) + .take_while(|ch| ch.is_ascii_uppercase()) + .count(); + let uppercase_run_end = group.start + uppercase_run_len; + let uppercase_run = word.chars[group.start..uppercase_run_end] + .iter() + .collect::(); + let needs_grade1 = permits_grade1_boundary_after_run(&word.chars[uppercase_run_end..]) + && requires_grade1_indicator(&uppercase_run); + if needs_grade1 { + replacement.push(Token::Mode(ModeEvent::Grade1Indicator)); + } + replacement.push(Token::Mode(ModeEvent::CapsPassageStart)); + replacement.push(owned_word(&word.chars[group.start..])); + state.triple_big_english = true; + state.has_processed_word = true; + return Ok(TokenAction::ReplaceMany(replacement)); + } - // UEB §5.7.2 + §10.9: prepend Grade-1 indicator (⠰) when the uppercase - // letters spell a multi-letter shortform (e.g. CD = "could"). This forces - // literal letter reading and prevents shortform mis-interpretation. - let needs_grade1 = requires_grade1_indicator(word.text.as_ref()); - if can_start_passage { - if needs_grade1 { - prefix.push(Token::Mode(ModeEvent::Grade1Indicator)); - } - prefix.push(Token::Mode(ModeEvent::CapsPassageStart)); - state.triple_big_english = true; - } else if word_len >= 2 { + if word.meta.is_all_uppercase + && !state.triple_big_english + && ascii_starts_at_beginning + && !is_separated_rule_69_unit(tokens, index, word) + { + if needs_inline_entry { + let entry = if state.needs_english_continuation { + ModeEvent::EnterEnglishContinue + } else { + ModeEvent::EnterEnglish + }; + prefix.push(Token::Mode(entry)); + state.is_english = true; + state.needs_english_continuation = false; + } + + let uppercase_run_len = word + .chars + .iter() + .take_while(|ch| ch.is_ascii_uppercase()) + .count(); + let uppercase_run = word.chars[..uppercase_run_len].iter().collect::(); + let needs_grade1 = permits_grade1_boundary_after_run(&word.chars[uppercase_run_len..]) + && requires_grade1_indicator(&uppercase_run); + if word_len >= 2 { if needs_grade1 { prefix.push(Token::Mode(ModeEvent::Grade1Indicator)); } @@ -107,10 +307,23 @@ impl TokenRule for UppercasePassageRule { } } - let next_is_ascii = upcoming_first.is_some_and(is_ascii_word); - if state.triple_big_english && !next_is_ascii { - suffix.push(Token::Mode(ModeEvent::CapsPassageEnd)); + let next_continues_passage = upcoming_first_group.is_some_and(|group| group.start == 0); + if state.triple_big_english && !next_continues_passage { state.triple_big_english = false; + + if let Some(group) = capitalized + && group.start == 0 + && group.end < word_len + { + let replacement = vec![ + owned_word(&word.chars[..group.end]), + Token::Mode(ModeEvent::CapsPassageEnd), + owned_word(&word.chars[group.end..]), + ]; + state.has_processed_word = true; + return Ok(TokenAction::ReplaceMany(replacement)); + } + suffix.push(Token::Mode(ModeEvent::CapsPassageEnd)); } if !state.has_processed_word { @@ -145,6 +358,27 @@ mod tests { }) } + fn spaced_words(words: &[&str]) -> Vec> { + let mut tokens = Vec::with_capacity(words.len().saturating_mul(2).saturating_sub(1)); + for (index, value) in words.iter().enumerate() { + if index > 0 { + tokens.push(Token::Space(SpaceKind::Regular)); + } + tokens.push(word(value)); + } + tokens + } + + fn replacement_words<'a>(tokens: &'a [Token<'_>]) -> Vec<&'a str> { + tokens + .iter() + .filter_map(|token| match token { + Token::Word(word) => Some(word.text.as_ref()), + _ => None, + }) + .collect() + } + /// uppercase_passage:78 — `EnterEnglishContinue` arm fires when /// `state.needs_english_continuation` is true at the moment of inline entry. /// Direct apply with hand-crafted state. @@ -192,24 +426,164 @@ mod tests { assert!(found, "expected EnterEnglish Mode token"); } - /// uppercase_passage:98 — Grade1Indicator pushed for shortform-colliding word - /// (e.g. "CD" = "could") at passage start. - #[test] - fn uppercase_passage_grade1_indicator_for_shortform_direct() { + /// UEB 2.6 + 5.7.2 + 10.9.7-10.9.8: a shortform-confusable sequence gets + /// grade 1 only at a permitted standing-alone/code boundary. `CD` and `LLC` + /// are the rulebook's official shortform-confusion examples. + #[rstest::rstest] + #[case::bare_cd("CD", true)] + #[case::llc_before_closing_group("LLC)", true)] + #[case::llc_before_korean_code_span("LLC회사", true)] + #[case::cd_before_digit("CD47", false)] + #[case::cd_before_slash("CD/ATM", false)] + #[case::neither_s_before_plus("NEIS+", false)] + #[case::little_m_before_opening_group("LLM(SLM)", false)] + fn uppercase_passage_grade1_respects_letters_sequence_boundary( + #[case] input: &str, + #[case] expected: bool, + ) { let r = UppercasePassageRule; let mut state = EncoderState::new(false); state.english_indicator = true; state.is_english = false; - let tokens = vec![ - word("CD"), - Token::Space(SpaceKind::Regular), - word("ABC"), - Token::Space(SpaceKind::Regular), - word("DEF"), - ]; + let tokens = vec![word(input)]; let action = r.apply(&tokens, 0, &mut state).unwrap(); let found = matches!(action, TokenAction::ReplaceMany(ref ts) if ts.iter().any(|t| matches!(t, Token::Mode(ModeEvent::Grade1Indicator)))); - assert!(found, "expected Grade1Indicator Mode token"); + assert_eq!(found, expected); + } + + #[test] + fn capitalized_group_rejects_korean_inside_the_capital_extent() { + let Token::Word(word) = word("A한B") else { + unreachable!("helper always builds a word") + }; + + assert_eq!(capitalized_group(&word), None); + } + + #[test] + fn shortform_collision_before_capitals_passage_gets_grade1() { + let rule = UppercasePassageRule; + let tokens = spaced_words(&["CD", "EF", "GH"]); + let mut state = EncoderState::new(false); + state.english_indicator = true; + + let TokenAction::ReplaceMany(replacement) = + rule.apply(&tokens, 0, &mut state).expect("passage starts") + else { + panic!("expected a passage-start replacement") + }; + + assert!(replacement.windows(2).any(|window| matches!( + window, + [ + Token::Mode(ModeEvent::Grade1Indicator), + Token::Mode(ModeEvent::CapsPassageStart) + ] + ))); + } + + #[rstest::rstest] + #[case::pdf_gigabyte("GB")] + #[case::petabyte("PB")] + #[case::terabyte("TB")] + fn separated_uppercase_units_do_not_preemit_ueb_modes(#[case] unit: &str) { + let r = UppercasePassageRule; + let mut state = EncoderState::new(false); + state.english_indicator = true; + let tokens = vec![word("5"), Token::Space(SpaceKind::Regular), word(unit)]; + + assert!(matches!( + r.apply(&tokens, 2, &mut state).unwrap(), + TokenAction::Noop + )); + } + + /// UEB 8.5.2-8.5.3: three or more capitalised symbols-sequences use one + /// passage indicator, and the terminator immediately follows the final + /// affected sequence. These are official 2024 UEB examples with Korean + /// boundary punctuation/glosses attached to exercise the mixed-script + /// tokenisation used by `DocumentIR`. + #[rstest::rstest] + #[case::caution_with_quote( + &["‘CAUTION:", "WET", "PAINT!’이다."], + &["‘", "CAUTION:"], + &["PAINT!", "’이다."] + )] + #[case::bbc_news_with_quote( + &["“THE", "BBC", "AFRICA", "NEWS”이다."], + &["“", "THE"], + &["NEWS", "”이다."] + )] + #[case::self_made_man_with_gloss( + &["A", "SELF-MADE", "MAN(남자)이다."], + &["A"], + &["MAN", "(남자)이다."] + )] + fn capitalized_passage_respects_attached_mixed_script_boundaries( + #[case] words: &[&str], + #[case] expected_start_words: &[&str], + #[case] expected_end_words: &[&str], + ) { + let rule = UppercasePassageRule; + let tokens = spaced_words(words); + let mut state = EncoderState::new(false); + state.english_indicator = true; + + let TokenAction::ReplaceMany(start) = rule + .apply(&tokens, 0, &mut state) + .expect("official capitalised passage must start") + else { + panic!("expected a passage-start replacement"); + }; + assert_eq!(replacement_words(&start), expected_start_words); + assert_eq!( + start + .iter() + .filter(|token| matches!(token, Token::Mode(ModeEvent::CapsPassageStart))) + .count(), + 1 + ); + assert!( + !start + .iter() + .any(|token| matches!(token, Token::Mode(ModeEvent::CapsWord))) + ); + assert!(state.triple_big_english); + + let last_index = tokens.len() - 1; + let TokenAction::ReplaceMany(end) = rule + .apply(&tokens, last_index, &mut state) + .expect("official capitalised passage must terminate") + else { + panic!("expected a passage-end replacement"); + }; + assert_eq!(replacement_words(&end), expected_end_words); + assert!(matches!( + end.get(1), + Some(Token::Mode(ModeEvent::CapsPassageEnd)) + )); + assert!(!state.triple_big_english); + } + + /// 수학 제12항 [붙임 1]: 국어 문장 안에서 콤마로 나열한 단일 대문자 + /// 변수는 UEB 대문자 구절이 아니라 각각의 로마자 항목으로 유지한다. + #[test] + fn korean_math_letter_list_does_not_start_capitals_passage() { + let ir = crate::rules::token::DocumentIR::parse("세 점 A, B, C가 있다.", true); + let index = ir + .tokens + .iter() + .position(|token| matches!(token, Token::Word(word) if word.text == "A,")) + .expect("official example must contain its first Roman variable"); + let mut state = EncoderState::new(true); + + assert!(matches!( + UppercasePassageRule + .apply(&ir.tokens, index, &mut state) + .expect("letter list classification must succeed"), + TokenAction::Noop + )); + assert!(!state.triple_big_english); } } diff --git a/libs/braillify/src/symbol_shortcut.rs b/libs/braillify/src/symbol_shortcut.rs index 878ff291..e5d42f53 100644 --- a/libs/braillify/src/symbol_shortcut.rs +++ b/libs/braillify/src/symbol_shortcut.rs @@ -13,6 +13,9 @@ static SHORTCUT_MAP: phf::Map = phf_map! { '\u{F000}' => &[decode_unicode('⠸'), decode_unicode('⠦'), decode_unicode('⠦'), decode_unicode('⠄')], '…' => &[decode_unicode('⠠'), decode_unicode('⠠'), decode_unicode('⠠')], '⋯' => &[decode_unicode('⠠'), decode_unicode('⠠'), decode_unicode('⠠')], + // 제53항 [다만] — 점 개수를 밝혀야 하는 줄임표는 묵자의 점 수만큼 + // ⠠을 적는다. U+2025 TWO DOT LEADER visibly carries two points. + '‥' => &[decode_unicode('⠠'), decode_unicode('⠠')], '!' => &[decode_unicode('⠖')], '.' => &[decode_unicode('⠲')], ',' => &[decode_unicode('⠐')], @@ -69,6 +72,9 @@ static SHORTCUT_MAP: phf::Map = phf_map! { '□' => &[decode_unicode('⠸'),decode_unicode('⠶'), decode_unicode('⠇')], '•' => &[decode_unicode('⠸'),decode_unicode('⠲')], 'ː' => &[decode_unicode('⠠'), decode_unicode('⠄')], + // 국제음성기호 제2장 — U+02D1 MODIFIER LETTER HALF TRIANGULAR COLON, + // 반장음 부호. IPA 문맥 밖에서도 이 Unicode scalar is unambiguous. + 'ˑ' => &[decode_unicode('⠐'), decode_unicode('⠂')], '〃' => &[decode_unicode('⠴'), decode_unicode('⠴')], // PDF 제60항 [붙임 1] — 참조 기호 ※ (U+203B). '※' => &[decode_unicode('⠸'), decode_unicode('⠔')], @@ -81,7 +87,7 @@ static SHORTCUT_MAP: phf::Map = phf_map! { /// gates *which* symbols are English-eligible and does NOT duplicate the point /// shapes. (Whether a given `:`/`,` is actually rendered English in 제39항 영-한 /// wrap context is decided by `english_logic::should_render_symbol_as_english`.) -const ENGLISH_SYMBOL_CHARS: [char; 5] = ['(', ')', ',', '-', ':']; +const ENGLISH_SYMBOL_CHARS: [char; 6] = ['(', ')', ',', '-', ':', '…']; pub fn encode_char_symbol_shortcut(text: char) -> Result<&'static [u8], String> { if let Some(code) = SHORTCUT_MAP.get(&text) { @@ -128,6 +134,7 @@ mod test { #[case('\'')] #[case('~')] #[case('…')] + #[case('‥')] #[case('!')] #[case('.')] #[case(',')] @@ -143,10 +150,20 @@ mod test { #[case('①')] #[case('ⓐ')] #[case('₩')] + #[case('ˑ')] pub fn test_is_symbol_char(#[case] ch: char) { assert!(is_symbol_char(ch)); } + #[rstest::rstest] + #[case::two_dot_leader('‥', "⠠⠠")] + #[case::ipa_half_length('ˑ', "⠐⠂")] + fn extended_standard_marks_have_pdf_defined_cells(#[case] input: char, #[case] expected: &str) { + let actual = encode_char_symbol_shortcut(input).unwrap(); + let expected = expected.chars().map(decode_unicode).collect::>(); + assert_eq!(actual, expected); + } + #[rstest::rstest] #[case::enclosed_jamo('㉠')] #[case::currency_dollar('$')] diff --git a/libs/braillify/tests/snapshots/coverage_extra__ellipsis_numbers.snap b/libs/braillify/tests/snapshots/coverage_extra__ellipsis_numbers.snap index 050d9bd2..cc721319 100644 --- a/libs/braillify/tests/snapshots/coverage_extra__ellipsis_numbers.snap +++ b/libs/braillify/tests/snapshots/coverage_extra__ellipsis_numbers.snap @@ -3,5 +3,5 @@ source: libs/braillify/tests/coverage_extra.rs expression: rendered --- input = "1, 2, 3, ..., 10" -unicode = ok: "⠼⠁⠂⠀⠼⠃⠂⠀⠼⠉⠐⠀⠲⠲⠲⠐⠀⠼⠁⠚" -bytes = ok: [60, 1, 2, 0, 60, 3, 2, 0, 60, 9, 16, 0, 50, 50, 50, 16, 0, 60, 1, 26] +unicode = ok: "⠼⠁⠐⠀⠼⠃⠐⠀⠼⠉⠐⠀⠲⠲⠲⠐⠀⠼⠁⠚" +bytes = ok: [60, 1, 16, 0, 60, 3, 16, 0, 60, 9, 16, 0, 50, 50, 50, 16, 0, 60, 1, 26] diff --git a/libs/braillify/tests/snapshots/coverage_extra__korean_math_squared.snap b/libs/braillify/tests/snapshots/coverage_extra__korean_math_squared.snap index beda7e13..6c600545 100644 --- a/libs/braillify/tests/snapshots/coverage_extra__korean_math_squared.snap +++ b/libs/braillify/tests/snapshots/coverage_extra__korean_math_squared.snap @@ -3,5 +3,5 @@ source: libs/braillify/tests/coverage_extra.rs expression: rendered --- input = "변수 a^2 와 b^2" -unicode = ok: "⠘⠡⠠⠍⠀⠴⠁⠲⠈⠢⠼⠃⠀⠸⠷⠧⠸⠾⠀⠃⠲⠈⠢⠼⠃" -bytes = ok: [24, 33, 32, 13, 0, 52, 1, 50, 8, 34, 60, 3, 0, 56, 55, 39, 56, 62, 0, 3, 50, 8, 34, 60, 3] +unicode = ok: "⠘⠡⠠⠍⠀⠴⠁⠲⠈⠢⠼⠃⠀⠧⠀⠴⠃⠲⠈⠢⠼⠃" +bytes = ok: [24, 33, 32, 13, 0, 52, 1, 50, 8, 34, 60, 3, 0, 39, 0, 52, 3, 50, 8, 34, 60, 3] diff --git a/libs/braillify/tests/snapshots/coverage_extra__negation_ab.snap b/libs/braillify/tests/snapshots/coverage_extra__negation_ab.snap index d9be7b62..e3c57e10 100644 --- a/libs/braillify/tests/snapshots/coverage_extra__negation_ab.snap +++ b/libs/braillify/tests/snapshots/coverage_extra__negation_ab.snap @@ -3,5 +3,5 @@ source: libs/braillify/tests/coverage_extra.rs expression: rendered --- input = "A¬B" -unicode = ok: "⠁⠨⠃" -bytes = ok: [1, 40, 3] +unicode = ok: "⠁⠈⠔⠃" +bytes = ok: [1, 8, 20, 3] diff --git a/libs/braillify/tests/snapshots/coverage_extra__negation_pq.snap b/libs/braillify/tests/snapshots/coverage_extra__negation_pq.snap index fb666244..e5a02d93 100644 --- a/libs/braillify/tests/snapshots/coverage_extra__negation_pq.snap +++ b/libs/braillify/tests/snapshots/coverage_extra__negation_pq.snap @@ -3,5 +3,5 @@ source: libs/braillify/tests/coverage_extra.rs expression: rendered --- input = "p¬Q" -unicode = ok: "⠏⠨⠟" -bytes = ok: [15, 40, 31] +unicode = ok: "⠏⠈⠔⠟" +bytes = ok: [15, 8, 20, 31] diff --git a/libs/braillify/tests/snapshots/coverage_extra__negation_xy.snap b/libs/braillify/tests/snapshots/coverage_extra__negation_xy.snap index 4efee896..5069c662 100644 --- a/libs/braillify/tests/snapshots/coverage_extra__negation_xy.snap +++ b/libs/braillify/tests/snapshots/coverage_extra__negation_xy.snap @@ -3,5 +3,5 @@ source: libs/braillify/tests/coverage_extra.rs expression: rendered --- input = "X¬Y" -unicode = ok: "⠭⠨⠽" -bytes = ok: [45, 40, 61] +unicode = ok: "⠭⠈⠔⠽" +bytes = ok: [45, 8, 20, 61] diff --git a/libs/braillify/tests/snapshots2/coverage_extra2__eng_dom_korean_in_middle.snap b/libs/braillify/tests/snapshots2/coverage_extra2__eng_dom_korean_in_middle.snap index e6d62be4..47ad045b 100644 --- a/libs/braillify/tests/snapshots2/coverage_extra2__eng_dom_korean_in_middle.snap +++ b/libs/braillify/tests/snapshots2/coverage_extra2__eng_dom_korean_in_middle.snap @@ -3,5 +3,5 @@ source: libs/braillify/tests/coverage_extra2.rs expression: rendered --- input = "Hello 안녕 World" -unicode = ok: "⠴⠠⠓⠑⠇⠇⠕⠀⠸⠷⠣⠒⠉⠻⠸⠾⠀⠠⠺⠕⠗⠇⠙⠲" -bytes = ok: [52, 32, 19, 17, 7, 7, 21, 0, 56, 55, 35, 18, 9, 59, 56, 62, 0, 32, 58, 21, 23, 7, 25, 50] +unicode = ok: "⠴⠠⠓⠑⠇⠇⠕⠀⠸⠷⠣⠒⠉⠻⠸⠾⠀⠠⠸⠺⠲" +bytes = ok: [52, 32, 19, 17, 7, 7, 21, 0, 56, 55, 35, 18, 9, 59, 56, 62, 0, 32, 56, 58, 50] diff --git a/libs/braillify/tests/snapshots2/coverage_extra2__eng_dom_long_with_uppercase.snap b/libs/braillify/tests/snapshots2/coverage_extra2__eng_dom_long_with_uppercase.snap index fcf64c57..526b8493 100644 --- a/libs/braillify/tests/snapshots2/coverage_extra2__eng_dom_long_with_uppercase.snap +++ b/libs/braillify/tests/snapshots2/coverage_extra2__eng_dom_long_with_uppercase.snap @@ -3,5 +3,5 @@ source: libs/braillify/tests/coverage_extra2.rs expression: rendered --- input = "API와 SDK를 사용해서 ABCDE 작업을 한다" -unicode = ok: "⠴⠠⠠⠁⠏⠊⠲⠧⠀⠠⠠⠴⠎⠙⠅⠲⠐⠮⠀⠇⠬⠶⠚⠗⠠⠎⠀⠠⠠⠴⠁⠃⠉⠙⠑⠲⠀⠨⠁⠎⠃⠮⠀⠚⠒⠊" -bytes = ok: [52, 32, 32, 1, 15, 10, 50, 39, 0, 32, 32, 52, 14, 25, 5, 50, 16, 46, 0, 7, 44, 54, 26, 23, 32, 14, 0, 32, 32, 52, 1, 3, 9, 25, 17, 50, 0, 40, 1, 14, 3, 46, 0, 26, 18, 10] +unicode = ok: "⠴⠠⠠⠁⠏⠊⠲⠧⠀⠴⠠⠠⠎⠙⠅⠲⠐⠮⠀⠇⠬⠶⠚⠗⠠⠎⠀⠴⠠⠠⠁⠃⠉⠙⠑⠲⠀⠨⠁⠎⠃⠮⠀⠚⠒⠊" +bytes = ok: [52, 32, 32, 1, 15, 10, 50, 39, 0, 52, 32, 32, 14, 25, 5, 50, 16, 46, 0, 7, 44, 54, 26, 23, 32, 14, 0, 52, 32, 32, 1, 3, 9, 25, 17, 50, 0, 40, 1, 14, 3, 46, 0, 26, 18, 10] diff --git a/libs/braillify/tests/snapshots2/coverage_extra2__eng_dom_sentence_period.snap b/libs/braillify/tests/snapshots2/coverage_extra2__eng_dom_sentence_period.snap index e9d4fb9c..d2ecbc15 100644 --- a/libs/braillify/tests/snapshots2/coverage_extra2__eng_dom_sentence_period.snap +++ b/libs/braillify/tests/snapshots2/coverage_extra2__eng_dom_sentence_period.snap @@ -4,5 +4,5 @@ assertion_line: 304 expression: rendered --- input = "The quick brown fox. 매우 빠르다." -unicode = ok: "⠴⠠⠮⠀⠟⠥⠊⠉⠅⠀⠃⠗⠪⠝⠀⠋⠕⠭⠲⠀⠑⠗⠍⠀⠠⠘⠐⠪⠊⠲" -bytes = ok: [52, 32, 46, 0, 31, 37, 10, 9, 5, 0, 3, 23, 42, 29, 0, 11, 21, 45, 50, 0, 17, 23, 13, 0, 32, 24, 16, 42, 10, 50] +unicode = ok: "⠴⠠⠮⠀⠟⠅⠀⠃⠗⠪⠝⠀⠋⠕⠭⠲⠀⠑⠗⠍⠀⠠⠘⠐⠪⠊⠲" +bytes = ok: [52, 32, 46, 0, 31, 5, 0, 3, 23, 42, 29, 0, 11, 21, 45, 50, 0, 17, 23, 13, 0, 32, 24, 16, 42, 10, 50] diff --git a/libs/braillify/tests/snapshots2/coverage_extra2__neg_upper_upper.snap b/libs/braillify/tests/snapshots2/coverage_extra2__neg_upper_upper.snap index 100631ef..f3cc1dd1 100644 --- a/libs/braillify/tests/snapshots2/coverage_extra2__neg_upper_upper.snap +++ b/libs/braillify/tests/snapshots2/coverage_extra2__neg_upper_upper.snap @@ -3,5 +3,5 @@ source: libs/braillify/tests/coverage_extra2.rs expression: rendered --- input = "A¬B" -unicode = ok: "⠁⠨⠃" -bytes = ok: [1, 40, 3] +unicode = ok: "⠁⠈⠔⠃" +bytes = ok: [1, 8, 20, 3] diff --git a/libs/braillify/tests/snapshots2/coverage_extra2__neg_var_upper.snap b/libs/braillify/tests/snapshots2/coverage_extra2__neg_var_upper.snap index 13734add..48471590 100644 --- a/libs/braillify/tests/snapshots2/coverage_extra2__neg_var_upper.snap +++ b/libs/braillify/tests/snapshots2/coverage_extra2__neg_var_upper.snap @@ -3,5 +3,5 @@ source: libs/braillify/tests/coverage_extra2.rs expression: rendered --- input = "x¬B" -unicode = ok: "⠭⠨⠃" -bytes = ok: [45, 40, 3] +unicode = ok: "⠭⠈⠔⠃" +bytes = ok: [45, 8, 20, 3] diff --git a/libs/braillify/tests/snapshots2/coverage_extra2__roman_mixed.snap b/libs/braillify/tests/snapshots2/coverage_extra2__roman_mixed.snap index 62bb1fd2..213d67c8 100644 --- a/libs/braillify/tests/snapshots2/coverage_extra2__roman_mixed.snap +++ b/libs/braillify/tests/snapshots2/coverage_extra2__roman_mixed.snap @@ -3,5 +3,5 @@ source: libs/braillify/tests/coverage_extra2.rs expression: rendered --- input = "VIII과 IX" -unicode = ok: "⠴⠠⠠⠧⠊⠊⠊⠲⠈⠧⠀⠠⠠⠴⠊⠭⠲" -bytes = ok: [52, 32, 32, 39, 10, 10, 10, 50, 8, 39, 0, 32, 32, 52, 10, 45, 50] +unicode = ok: "⠴⠠⠠⠧⠊⠊⠊⠲⠈⠧⠀⠴⠠⠠⠊⠭⠲" +bytes = ok: [52, 32, 32, 39, 10, 10, 10, 50, 8, 39, 0, 52, 32, 32, 10, 45, 50] diff --git a/test_cases/korean/rule_47.json b/test_cases/korean/rule_47.json index f196d202..ce7a19e0 100644 --- a/test_cases/korean/rule_47.json +++ b/test_cases/korean/rule_47.json @@ -59,7 +59,7 @@ "jeomsarang": "" }, { - "input": "지구 표면의 2/3는 바다로 덮여있다.", + "input": "지구 표면의 2/3는 바다로 덮여 있다.", "internal": ".o@m`d+e*w`#b_/#c`cz`^i\"u`is4:`o/i4", "expected": "402181302544173358060356126090953024101637010145049021121050", "unicode": "⠨⠕⠈⠍⠀⠙⠬⠑⠡⠺⠀⠼⠃⠸⠌⠼⠉⠀⠉⠵⠀⠘⠊⠐⠥⠀⠊⠎⠲⠱⠀⠕⠌⠊⠲", diff --git a/uv.lock b/uv.lock index 7abe1eee..a27ec581 100644 --- a/uv.lock +++ b/uv.lock @@ -11,7 +11,7 @@ members = [ [[package]] name = "braillify" -version = "2.1.0" +version = "2.1.2" source = { editable = "packages/python" } [[package]]