Add the four double-byte CJK pages behind an on-by-default dbcs feature
Code pages 932 (Shift-JIS), 936 (GBK), 949 (Unified Hangul), and 950 (Big5), generated from the Unicode Consortium's Microsoft WINDOWS tables. Each page carries a three-state single-byte table (map / lead / undefined) plus sorted pair arrays binary-searched in both directions. Where several codes decode to one scalar (932's NEC/IBM overlap, 950's duplicated box-drawing rows), the encoder's winner is resolved from Microsoft's own bestfit WCTABLE — used only for duplicate resolution, never to import lossy best-fit mappings. The full decode direction was cross-validated against Python's cp932/936/949/950 codecs: zero mismatches over 60k+ pairs. Decode gains TruncatedPair and UndefinedPair failure modes; decode_lossy emits one U+FFFD per failed pair. Both feature configurations are clippy-clean and tested.
This commit is contained in:
@@ -2,123 +2,353 @@
|
||||
# /// script
|
||||
# requires-python = ">=3.10"
|
||||
# ///
|
||||
"""Generate src/tables.rs from the Microsoft mapping tables in data/.
|
||||
"""Generate src/tables/ from the Microsoft mapping tables in data/.
|
||||
|
||||
The inputs are the canonical Microsoft code page tables published by the
|
||||
Unicode Consortium (https://www.unicode.org/Public/MAPPINGS/VENDORS/MICSFT/PC/).
|
||||
Each output table is a full 256-entry ``[Option<char>; 256]``: MS-DOS pages are
|
||||
not uniformly ASCII in the low half (CP864 maps 0x25 to ARABIC PERCENT SIGN)
|
||||
and several pages leave byte values undefined (CP857/864/869/874).
|
||||
Unicode Consortium (https://www.unicode.org/Public/MAPPINGS/VENDORS/MICSFT/,
|
||||
``PC/`` for the single-byte DOS pages and ``WINDOWS/`` for the double-byte
|
||||
CJK pages).
|
||||
|
||||
Single-byte pages become full 256-entry ``[Option<char>; 256]`` tables:
|
||||
MS-DOS pages are not uniformly ASCII in the low half (CP864 maps 0x25 to
|
||||
ARABIC PERCENT SIGN) and several pages leave byte values undefined
|
||||
(CP857/864/869/874).
|
||||
|
||||
Double-byte pages become a 256-entry single-byte/lead-byte table plus a pair
|
||||
of sorted ``(code, scalar)`` arrays for binary search. Where several codes
|
||||
decode to one scalar (CP932's NEC/IBM overlap, CP950's box-drawing rows),
|
||||
the encoder's choice is resolved from Microsoft's own ``bestfit*.txt``
|
||||
WCTABLE — used *only* to pick among duplicate candidates, never to import
|
||||
lossy best-fit mappings.
|
||||
|
||||
Run from the crate root: ``uv run tools/gen_tables.py``
|
||||
(needs ``rustfmt`` on PATH).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import subprocess
|
||||
from collections.abc import Sequence
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
CODE_PAGES: Sequence[int] = (
|
||||
SINGLE_BYTE_PAGES: Sequence[int] = (
|
||||
437, 737, 775, 850, 852, 855, 857, 860,
|
||||
861, 862, 863, 864, 865, 866, 869, 874,
|
||||
)
|
||||
DOUBLE_BYTE_PAGES: Sequence[int] = (932, 936, 949, 950)
|
||||
|
||||
DATA_DIR = Path(__file__).resolve().parent.parent / "data"
|
||||
OUTPUT = Path(__file__).resolve().parent.parent / "src" / "tables.rs"
|
||||
OUTPUT_DIR = Path(__file__).resolve().parent.parent / "src" / "tables"
|
||||
|
||||
HEADER = """\
|
||||
//! Generated code page tables — do not edit by hand.
|
||||
GENERATED_NOTE = """\
|
||||
//! Generated by `tools/gen_tables.py` — do not edit by hand.
|
||||
//!
|
||||
//! Source: the Unicode Consortium's Microsoft mapping tables
|
||||
//! (`VENDORS/MICSFT/PC/CP*.TXT`), vendored in `data/` and regenerated with
|
||||
//! `tools/gen_tables.py`. Each table maps every byte value `0x00..=0xFF` to
|
||||
//! its Unicode scalar, or `None` where the code page leaves the byte
|
||||
//! undefined.
|
||||
|
||||
//! Source: the Unicode Consortium's Microsoft mapping tables, vendored in
|
||||
//! `data/`.
|
||||
"""
|
||||
|
||||
SURROGATE_RANGE = range(0xD800, 0xE000)
|
||||
|
||||
|
||||
class TableError(Exception):
|
||||
"""A mapping table failed to parse or violated an invariant."""
|
||||
|
||||
|
||||
def parse_table(path: Path) -> list[int | None]:
|
||||
"""Parse one CP*.TXT into a 256-entry byte-to-scalar table.
|
||||
@dataclass
|
||||
class CodePageTable:
|
||||
"""One parsed Microsoft mapping table.
|
||||
|
||||
Args:
|
||||
path: The mapping table file.
|
||||
Attributes:
|
||||
single: Byte value to Unicode scalar for single-byte codes.
|
||||
leads: Byte values marked ``DBCS LEAD BYTE`` (empty for single-byte
|
||||
pages).
|
||||
pairs: 16-bit lead<<8|trail code to Unicode scalar.
|
||||
listed_bytes: Every byte value the file listed, mapped or not.
|
||||
"""
|
||||
|
||||
Returns:
|
||||
A list indexed by byte value; ``None`` marks an undefined byte.
|
||||
single: dict[int, int]
|
||||
leads: set[int]
|
||||
pairs: dict[int, int]
|
||||
listed_bytes: set[int]
|
||||
|
||||
|
||||
def read_mapping_lines(path: Path) -> list[str]:
|
||||
"""Read a mapping file and return its data lines (those starting 0x).
|
||||
|
||||
Raises:
|
||||
TableError: If the file is missing or non-ASCII, a byte is out of
|
||||
range or listed twice, a byte is never listed, or two bytes map
|
||||
to the same scalar (the encoder requires a bijection).
|
||||
TableError: If the file is missing or non-ASCII (the lone stray byte
|
||||
in canonical data is a DOS EOF marker, 0x1A, which is still ASCII).
|
||||
"""
|
||||
try:
|
||||
# The lone stray byte in canonical data is a DOS EOF marker (0x1A),
|
||||
# which is still ASCII; anything else should fail loudly.
|
||||
text = path.read_text(encoding="ascii")
|
||||
except (OSError, UnicodeDecodeError) as error:
|
||||
raise TableError(f"{path.name}: unreadable mapping table") from error
|
||||
table: list[int | None] = [None] * 256
|
||||
return [line.strip() for line in text.splitlines() if line.strip().startswith("0x")]
|
||||
|
||||
|
||||
def parse_table(path: Path) -> CodePageTable:
|
||||
"""Parse one CP*.TXT into single-byte, lead-byte, and pair mappings.
|
||||
|
||||
Raises:
|
||||
TableError: If a code is listed twice, a single byte is out of range,
|
||||
a scalar is a surrogate or beyond the BMP, or byte 0x00 appears
|
||||
as a DBCS trail (which would break NUL-terminated decoding).
|
||||
"""
|
||||
table = CodePageTable(single={}, leads=set(), pairs={}, listed_bytes=set())
|
||||
seen: set[int] = set()
|
||||
for raw_line in text.splitlines():
|
||||
line = raw_line.strip()
|
||||
if not line.startswith("0x"):
|
||||
continue
|
||||
for line in read_mapping_lines(path):
|
||||
columns = line.split("\t")
|
||||
byte = int(columns[0], 16)
|
||||
if byte > 0xFF:
|
||||
raise TableError(f"{path.name}: byte {byte:#x} out of range")
|
||||
if byte in seen:
|
||||
raise TableError(f"{path.name}: byte {byte:#04x} listed twice")
|
||||
seen.add(byte)
|
||||
if len(columns) >= 2 and columns[1].strip().startswith("0x"):
|
||||
table[byte] = int(columns[1], 16)
|
||||
if len(seen) != 256:
|
||||
missing = sorted(set(range(256)) - seen)
|
||||
raise TableError(f"{path.name}: bytes never listed: {missing}")
|
||||
defined = [scalar for scalar in table if scalar is not None]
|
||||
if len(defined) != len(set(defined)):
|
||||
raise TableError(f"{path.name}: table is not a bijection")
|
||||
code = int(columns[0], 16)
|
||||
if code in seen:
|
||||
raise TableError(f"{path.name}: code {code:#06x} listed twice")
|
||||
seen.add(code)
|
||||
if code > 0xFFFF:
|
||||
raise TableError(f"{path.name}: code {code:#x} out of range")
|
||||
if code <= 0xFF:
|
||||
table.listed_bytes.add(code)
|
||||
mapped = len(columns) >= 2 and columns[1].strip().startswith("0x")
|
||||
if not mapped:
|
||||
if code <= 0xFF and "DBCS LEAD BYTE" in line:
|
||||
table.leads.add(code)
|
||||
continue
|
||||
scalar = int(columns[1], 16)
|
||||
if scalar in SURROGATE_RANGE or scalar > 0xFFFF:
|
||||
raise TableError(f"{path.name}: scalar {scalar:#06x} not a BMP char")
|
||||
if code <= 0xFF:
|
||||
table.single[code] = scalar
|
||||
else:
|
||||
if code & 0xFF == 0:
|
||||
raise TableError(f"{path.name}: NUL trail byte in {code:#06x}")
|
||||
table.pairs[code] = scalar
|
||||
unmarked_leads = {code >> 8 for code in table.pairs} - table.leads
|
||||
if unmarked_leads:
|
||||
raise TableError(
|
||||
f"{path.name}: pairs under bytes not marked DBCS LEAD BYTE: "
|
||||
f"{[hex(lead) for lead in sorted(unmarked_leads)]}"
|
||||
)
|
||||
return table
|
||||
|
||||
|
||||
def rust_entry(scalar: int | None) -> str:
|
||||
"""Render one table slot as a Rust ``Option<char>`` literal."""
|
||||
if scalar is None:
|
||||
return "None"
|
||||
def parse_wctable(path: Path) -> dict[int, int]:
|
||||
"""Parse the WCTABLE (Unicode to codepage) section of a bestfit file.
|
||||
|
||||
Raises:
|
||||
TableError: If the file is missing or has no WCTABLE section.
|
||||
"""
|
||||
try:
|
||||
# Comments embed raw codepage-encoded glyph bytes; latin-1 accepts
|
||||
# any byte and the data lines we match are pure ASCII.
|
||||
text = path.read_text(encoding="latin-1")
|
||||
except OSError as error:
|
||||
raise TableError(f"{path.name}: unreadable bestfit table") from error
|
||||
_, marker, rest = text.partition("WCTABLE")
|
||||
if not marker:
|
||||
raise TableError(f"{path.name}: no WCTABLE section")
|
||||
section = rest.split("ENDCODEPAGE")[0]
|
||||
entry = re.compile(r"^0x([0-9a-fA-F]{4})\s+0x([0-9a-fA-F]+)")
|
||||
table: dict[int, int] = {}
|
||||
for line in section.splitlines():
|
||||
if match := entry.match(line.strip()):
|
||||
table[int(match.group(1), 16)] = int(match.group(2), 16)
|
||||
return table
|
||||
|
||||
|
||||
def check_bijection(path_name: str, table: CodePageTable) -> None:
|
||||
"""Require every scalar to have exactly one code (single-byte pages).
|
||||
|
||||
Raises:
|
||||
TableError: If two codes map to the same scalar.
|
||||
"""
|
||||
scalars = list(table.single.values()) + list(table.pairs.values())
|
||||
if len(scalars) != len(set(scalars)):
|
||||
raise TableError(f"{path_name}: table is not a bijection")
|
||||
|
||||
|
||||
def resolve_encode(
|
||||
path_name: str, table: CodePageTable, wctable: dict[int, int]
|
||||
) -> dict[int, int]:
|
||||
"""Choose one code per scalar, resolving duplicates via the WCTABLE.
|
||||
|
||||
Args:
|
||||
path_name: The mapping file name, for error messages.
|
||||
table: The parsed decode-direction table.
|
||||
wctable: Microsoft's Unicode-to-codepage table for this page.
|
||||
|
||||
Returns:
|
||||
Scalar to pair-code for the double-byte portion only (single-byte
|
||||
codes never collide with pairs in these tables, which is asserted).
|
||||
|
||||
Raises:
|
||||
TableError: If a duplicate set is not settled by the WCTABLE, mixes
|
||||
single-byte and pair codes (the Rust encoder tries the single
|
||||
table first, so a pair winner would be unreachable), or has a
|
||||
single-byte winner that is not the lowest byte (the Rust encoder
|
||||
scans upward, so any other winner would be shadowed).
|
||||
"""
|
||||
candidates: dict[int, list[int]] = {}
|
||||
for code, scalar in list(table.single.items()) + list(table.pairs.items()):
|
||||
candidates.setdefault(scalar, []).append(code)
|
||||
encode: dict[int, int] = {}
|
||||
for scalar, codes in candidates.items():
|
||||
if len(codes) == 1:
|
||||
chosen = codes[0]
|
||||
else:
|
||||
if any(c <= 0xFF for c in codes) and any(c > 0xFF for c in codes):
|
||||
raise TableError(
|
||||
f"{path_name}: U+{scalar:04X} mixes single and pair codes "
|
||||
f"({[hex(c) for c in codes]})"
|
||||
)
|
||||
chosen = wctable.get(scalar, -1)
|
||||
if chosen not in codes:
|
||||
raise TableError(
|
||||
f"{path_name}: WCTABLE does not settle duplicate scalar "
|
||||
f"U+{scalar:04X} (candidates {[hex(c) for c in codes]})"
|
||||
)
|
||||
if chosen <= 0xFF and chosen != min(codes):
|
||||
raise TableError(
|
||||
f"{path_name}: U+{scalar:04X} single-byte winner "
|
||||
f"{chosen:#04x} is not the lowest candidate"
|
||||
)
|
||||
if chosen <= 0xFF:
|
||||
continue # single-byte encoding is derived from the single table
|
||||
encode[scalar] = chosen
|
||||
return encode
|
||||
|
||||
|
||||
def rust_char(scalar: int) -> str:
|
||||
"""Render a Unicode scalar as a Rust char literal."""
|
||||
ch = chr(scalar)
|
||||
if ch.isprintable() and ch not in ("'", "\\"):
|
||||
return f"Some('{ch}')"
|
||||
return f"Some('\\u{{{scalar:04X}}}')"
|
||||
return f"'{ch}'"
|
||||
return f"'\\u{{{scalar:04X}}}'"
|
||||
|
||||
|
||||
def render_table(code_page: int, table: Sequence[int | None]) -> str:
|
||||
"""Render one code page as a Rust constant, eight entries per line."""
|
||||
def render_single_byte_page(code_page: int, table: CodePageTable) -> str:
|
||||
"""Render one single-byte code page as an ``[Option<char>; 256]``."""
|
||||
lines = [
|
||||
f"/// Code page {code_page}, byte value to Unicode scalar.",
|
||||
f"pub(crate) const CP{code_page}: [Option<char>; 256] = [",
|
||||
]
|
||||
for row_start in range(0, 256, 8):
|
||||
row = ", ".join(rust_entry(table[i]) for i in range(row_start, row_start + 8))
|
||||
lines.append(f" {row}, // {row_start:#04x}-{row_start + 7:#04x}")
|
||||
entries = []
|
||||
for byte in range(row_start, row_start + 8):
|
||||
scalar = table.single.get(byte)
|
||||
entries.append("None" if scalar is None else f"Some({rust_char(scalar)})")
|
||||
lines.append(f" {', '.join(entries)}, // {row_start:#04x}-{row_start + 7:#04x}")
|
||||
lines.append("];\n")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def render_pair_array(name: str, doc: str, pairs: Sequence[tuple[int, int]]) -> str:
|
||||
"""Render a sorted ``[(u16, char); N]`` static, eight entries per line."""
|
||||
lines = [f"/// {doc}", f"static {name}: [(u16, char); {len(pairs)}] = ["]
|
||||
for row_start in range(0, len(pairs), 8):
|
||||
row = pairs[row_start : row_start + 8]
|
||||
rendered = ", ".join(f"({code:#06x}, {rust_char(scalar)})" for code, scalar in row)
|
||||
lines.append(f" {rendered},")
|
||||
lines.append("];\n")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def render_encode_array(name: str, doc: str, pairs: Sequence[tuple[int, int]]) -> str:
|
||||
"""Render a sorted ``[(char, u16); N]`` static, eight entries per line."""
|
||||
lines = [f"/// {doc}", f"static {name}: [(char, u16); {len(pairs)}] = ["]
|
||||
for row_start in range(0, len(pairs), 8):
|
||||
row = pairs[row_start : row_start + 8]
|
||||
rendered = ", ".join(f"({rust_char(scalar)}, {code:#06x})" for scalar, code in row)
|
||||
lines.append(f" {rendered},")
|
||||
lines.append("];\n")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def render_double_byte_page(
|
||||
code_page: int, table: CodePageTable, encode: dict[int, int]
|
||||
) -> str:
|
||||
"""Render one DBCS page as a module body with single/decode/encode tables."""
|
||||
single_lines = [
|
||||
"/// Single-byte slots: mapped char, DBCS lead byte, or undefined.",
|
||||
"const SINGLE: [SingleEntry; 256] = [",
|
||||
]
|
||||
for row_start in range(0, 256, 8):
|
||||
entries = []
|
||||
for byte in range(row_start, row_start + 8):
|
||||
if byte in table.leads:
|
||||
entries.append("L")
|
||||
elif (scalar := table.single.get(byte)) is not None:
|
||||
entries.append(f"M({rust_char(scalar)})")
|
||||
else:
|
||||
entries.append("U")
|
||||
single_lines.append(
|
||||
f" {', '.join(entries)}, // {row_start:#04x}-{row_start + 7:#04x}"
|
||||
)
|
||||
single_lines.append("];\n")
|
||||
|
||||
decode_pairs = sorted(table.pairs.items())
|
||||
encode_pairs = sorted(encode.items())
|
||||
return "\n".join(
|
||||
[
|
||||
GENERATED_NOTE,
|
||||
"use super::SingleEntry::{self, Lead as L, Map as M, Undefined as U};",
|
||||
"use super::DbcsPage;",
|
||||
"",
|
||||
f"/// Code page {code_page}.",
|
||||
f"pub(crate) static CP{code_page}: DbcsPage = DbcsPage {{",
|
||||
" single: &SINGLE,",
|
||||
" decode: &DECODE,",
|
||||
" encode: &ENCODE,",
|
||||
"};",
|
||||
"",
|
||||
"\n".join(single_lines),
|
||||
render_pair_array(
|
||||
"DECODE",
|
||||
"Every defined pair, sorted by 16-bit code (lead<<8 | trail).",
|
||||
decode_pairs,
|
||||
),
|
||||
render_encode_array(
|
||||
"ENCODE",
|
||||
"One pair per scalar, sorted by scalar; duplicates resolved per "
|
||||
"Microsoft's bestfit WCTABLE.",
|
||||
encode_pairs,
|
||||
),
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
"""Regenerate src/tables.rs from every table in CODE_PAGES."""
|
||||
rendered = [HEADER]
|
||||
for code_page in CODE_PAGES:
|
||||
"""Regenerate src/tables/ from every configured code page."""
|
||||
outputs: list[Path] = []
|
||||
|
||||
single_rendered = [GENERATED_NOTE]
|
||||
for code_page in SINGLE_BYTE_PAGES:
|
||||
table = parse_table(DATA_DIR / f"CP{code_page}.TXT")
|
||||
rendered.append(render_table(code_page, table))
|
||||
OUTPUT.write_text("\n".join(rendered), encoding="utf-8")
|
||||
subprocess.run(["rustfmt", "--edition", "2021", str(OUTPUT)], check=True)
|
||||
print(f"wrote {OUTPUT} ({len(CODE_PAGES)} tables)")
|
||||
if table.listed_bytes != set(range(256)):
|
||||
missing = sorted(set(range(256)) - table.listed_bytes)
|
||||
raise TableError(f"CP{code_page}: bytes never listed: {missing}")
|
||||
if table.leads or table.pairs:
|
||||
raise TableError(f"CP{code_page}: unexpected DBCS content")
|
||||
check_bijection(f"CP{code_page}", table)
|
||||
single_rendered.append(render_single_byte_page(code_page, table))
|
||||
single_path = OUTPUT_DIR / "single.rs"
|
||||
single_path.write_text("\n".join(single_rendered), encoding="utf-8")
|
||||
outputs.append(single_path)
|
||||
|
||||
for code_page in DOUBLE_BYTE_PAGES:
|
||||
table = parse_table(DATA_DIR / f"CP{code_page}.TXT")
|
||||
wctable = parse_wctable(DATA_DIR / f"bestfit{code_page}.txt")
|
||||
encode = resolve_encode(f"CP{code_page}", table, wctable)
|
||||
page_path = OUTPUT_DIR / f"cp{code_page}.rs"
|
||||
page_path.write_text(
|
||||
render_double_byte_page(code_page, table, encode), encoding="utf-8"
|
||||
)
|
||||
outputs.append(page_path)
|
||||
|
||||
subprocess.run(
|
||||
["rustfmt", "--edition", "2021", *map(str, outputs)], check=True
|
||||
)
|
||||
for path in outputs:
|
||||
print(f"wrote {path}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
Reference in New Issue
Block a user