Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 11 additions & 3 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

7 changes: 5 additions & 2 deletions Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@ members = [
"rio-vt",
"rio-backend",
"rio-grapheme-width",
"rio-unicode",
"rio-window",
"rio-notifier",
"frontends/rioterm",
Expand Down Expand Up @@ -51,8 +52,10 @@ parking_lot = { version = "0.12.5", features = [
rustc-hash = "2.1.3"
smallvec = { version = "1.15.2", default-features = false }

# unicode-width = "0.2.0"
unicode-width = { package = "unicode-width-16", version = "0.1.0" }
# One Unicode version for the whole tree: width tables and grapheme
# segmentation both come from rio-unicode (currently Unicode 17),
# replacing the alacritty unicode-width fork that froze at Unicode 16.
unicode-width = { package = "rio-unicode", path = "rio-unicode" }
simdutf = "0.7.0"
base64 = "0.23.0"
image_rs = { package = "image", version = "0.25.10", default-features = false, features = [
Expand Down
1 change: 1 addition & 0 deletions rio-unicode/.gitignore
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
*.txt
23 changes: 23 additions & 0 deletions rio-unicode/Cargo.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
[package]
name = "rio-unicode"
version = "0.1.0"
edition = "2021"
description = """
Unicode tables for rio, on a single Unicode version: character width
(vendored from unicode-width 0.1.14, tables regenerated for Unicode 17)
and grapheme cluster breaks (UAX #29 state machine). One crate so the
grid's widths and the renderer's segmentation can never disagree on
what Unicode says.
"""
license = "MIT OR Apache-2.0"
repository = "https://github.com/raphamorim/rio"

[features]
default = ["cjk"]
cjk = []
# Vendored from unicode-width; enables nightly benches upstream.
bench = []

[dev-dependencies]
unicode-segmentation = "1.13"
unicode-width-16 = "0.1.0"
2 changes: 2 additions & 0 deletions rio-unicode/scripts/.gitignore
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
*.txt
tables.rs
175 changes: 175 additions & 0 deletions rio-unicode/scripts/grapheme.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,175 @@
#!/usr/bin/env python3
"""Generate rio-unicode's grapheme-cluster-break tables.

Reads (from the working directory, as fetched by unicode.py's runs):
- GraphemeBreakProperty.txt (UAX #29 GCB classes)
- emoji-data.txt (Extended_Pictographic)
- DerivedCoreProperties.txt (InCB=Consonant/Linker/Extend, for GB9c)
- GraphemeBreakTest.txt (conformance vectors)

Emits:
- ../src/grapheme_tables.rs (class ranges + Unicode version)
- ../tests/grapheme_conformance.rs (every UCD test vector)

Each codepoint gets exactly one class; InCB values refine the GCB
Extend/ZWJ/Other classes so the GB9c state machine can run on classes
alone.
"""
import re
from collections import defaultdict

def parse_props(path, wanted_prop=None):
"""Yield (lo, hi, value) from a UCD file."""
out = []
with open(path, encoding="utf-8") as f:
for line in f:
line = line.split("#", 1)[0].strip()
if not line:
continue
fields = [x.strip() for x in line.split(";")]
cps, value = fields[0], fields[1]
if wanted_prop is not None:
if value != wanted_prop:
continue
value = fields[2] if len(fields) > 2 else value
if ".." in cps:
lo, hi = (int(x, 16) for x in cps.split(".."))
else:
lo = hi = int(cps, 16)
out.append((lo, hi, value))
return out

gcb = {}
for lo, hi, v in parse_props("GraphemeBreakProperty.txt"):
for cp in range(lo, hi + 1):
gcb[cp] = v

extpic = set()
for lo, hi, v in parse_props("emoji-data.txt"):
if v == "Extended_Pictographic":
extpic.update(range(lo, hi + 1))

incb = {}
with open("DerivedCoreProperties.txt", encoding="utf-8") as f:
for line in f:
line = line.split("#", 1)[0].strip()
if not line or "; InCB;" not in line:
continue
cps, _, value = [x.strip() for x in line.split(";")]
if ".." in cps:
lo, hi = (int(x, 16) for x in cps.split(".."))
else:
lo = hi = int(cps, 16)
for cp in range(lo, hi + 1):
incb[cp] = value

CLASSES = ["Other", "CR", "LF", "Control", "Extend", "ExtendIncb", "Linker",
"Zwj", "RegionalIndicator", "Prepend", "SpacingMark",
"L", "V", "T", "LV", "LVT", "ExtPic", "Consonant"]

def classify(cp):
g = gcb.get(cp, "Other")
ic = incb.get(cp)
if g == "CR": return "CR"
if g == "LF": return "LF"
if g == "Control": return "Control"
if g == "ZWJ": return "Zwj"
if g == "Extend":
if ic == "Linker": return "Linker"
if ic == "Extend": return "ExtendIncb"
return "Extend"
if g == "Regional_Indicator": return "RegionalIndicator"
if g == "Prepend": return "Prepend"
if g == "SpacingMark": return "SpacingMark"
if g in ("L", "V", "T", "LV", "LVT"): return g
# g == Other:
if cp in extpic: return "ExtPic"
if ic == "Consonant": return "Consonant"
return "Other"

# Build merged ranges over all assigned-relevant codepoints.
ranges = []
prev_class = None
start = None
for cp in range(0x110000):
c = classify(cp)
if c != prev_class:
if prev_class is not None and prev_class != "Other":
ranges.append((start, cp - 1, prev_class))
prev_class = c
start = cp
if prev_class is not None and prev_class != "Other":
ranges.append((start, 0x10FFFF, prev_class))

version = "17.0.0"
with open("GraphemeBreakProperty.txt", encoding="utf-8") as f:
m = re.search(r"-(\d+\.\d+\.\d+)\.txt", f.readline())
if m:
version = m.group(1)

with open("../src/grapheme_tables.rs", "w", encoding="utf-8") as out:
out.write("// Generated by scripts/grapheme.py — do not edit.\n")
out.write(f"// Unicode {version}: GraphemeBreakProperty + Extended_Pictographic + InCB.\n\n")
out.write("use super::grapheme::GraphemeClass;\n\n")
v = version.split(".")
out.write("/// The Unicode version of the grapheme-break data; always equal\n")
out.write("/// to the width tables' [`UNICODE_VERSION`](crate::UNICODE_VERSION).\n")
out.write(f"pub const GRAPHEME_UNICODE_VERSION: (u8, u8, u8) = ({v[0]}, {v[1]}, {v[2]});\n\n")
out.write("/// Sorted, non-overlapping ranges; codepoints not listed are\n")
out.write("/// `GraphemeClass::Other`.\n")
out.write("pub(crate) static GRAPHEME_CLASS_RANGES: &[(u32, u32, GraphemeClass)] = &[\n")
for lo, hi, c in ranges:
out.write(f" (0x{lo:X}, 0x{hi:X}, GraphemeClass::{c}),\n")
out.write("];\n")
print(f"grapheme_tables.rs: {len(ranges)} ranges (Unicode {version})")

# Conformance vectors.
tests = []
with open("GraphemeBreakTest.txt", encoding="utf-8") as f:
for line in f:
line = line.split("#", 1)[0].strip()
if not line:
continue
# Format: ÷ 0020 × 0308 ÷ 0020 ÷
parts = line.split()
chars = []
breaks = [] # break BEFORE char i (excluding sot), plus eot ignored
expect = [] # byte offsets where clusters start
s = ""
for tok in parts[1:-1]: # drop leading ÷ and trailing ÷/×
if tok == "÷":
breaks.append(True)
elif tok == "×":
breaks.append(False)
else:
chars.append(int(tok, 16))
# Reconstruct cluster strings.
clusters = []
cur = ""
for i, cp in enumerate(chars):
ch = chr(cp)
if i > 0 and breaks[i - 1]:
clusters.append(cur)
cur = ""
cur += ch
clusters.append(cur)
tests.append((chars, clusters))

def rs_str(s):
return '"' + "".join(f"\\u{{{ord(ch):X}}}" for ch in s) + '"'

with open("../tests/grapheme_conformance.rs", "w", encoding="utf-8") as out:
out.write("// Generated by scripts/grapheme.py from GraphemeBreakTest.txt — do not edit.\n")
out.write(f"// Unicode {version}: every UCD conformance vector.\n\n")
out.write("use rio_unicode::grapheme::Graphemes;\n\n")
out.write("#[test]\nfn ucd_grapheme_break_test() {\n")
out.write(" let cases: &[(&str, &[&str])] = &[\n")
for chars, clusters in tests:
full = "".join(chr(c) for c in chars)
out.write(f" ({rs_str(full)}, &[{', '.join(rs_str(c) for c in clusters)}]),\n")
out.write(" ];\n")
out.write(" for (i, (input, expected)) in cases.iter().enumerate() {\n")
out.write(" let got: Vec<&str> = Graphemes::new(input).collect();\n")
out.write(" assert_eq!(&got[..], *expected, \"case {i}: {input:?}\");\n")
out.write(" }\n}\n")
print(f"grapheme_conformance.rs: {len(tests)} vectors")
Loading
Loading