std::text::unicodeUnicode character classification and UTF-8 string helpers.
Provides codepoint-level predicates mirroring Go's unicode package, plus
UTF-8-aware string helpers for counting and iterating Unicode codepoints.
All classification functions operate on codepoint values (i64), not raw
bytes. Codepoints are produced by unicode.codepoint_at(s, i) (the
codepoint at position i) or by iterating unicode.runes(s).
The underlying char_count_utf8 and codepoint_at_utf8 methods are
exposed on string by std.string (bound to hew_string_char_count and
hew_string_char_at_utf8 in hew-runtime).
import std.text.unicode;
fn main() {
// Char classification
let cp = unicode.codepoint_at("café", 3) handle _ { 0 }; // 233 ('é')
println(unicode.is_lower(cp)); // true
println(unicode.is_upper(cp)); // false
println(unicode.to_upper(cp)); // 201 ('É')
// UTF-8 helpers
println(unicode.rune_count("café")); // 4
let cps = unicode.runes("日本");
println(cps.len()); // 2
}
is_valid_runeReturn true when cp is a valid Unicode scalar value.
import std.text.unicode;
fn main() {
println(unicode.is_valid_rune(65)); // true ('A')
println(unicode.is_valid_rune(0x10FFFF)); // true (highest scalar)
println(unicode.is_valid_rune(0xD800)); // false (surrogate)
println(unicode.is_valid_rune(-1)); // false
}
is_upperTest whether the Unicode codepoint cp is an uppercase letter.
Follows Unicode general category Lu (Uppercase Letter).
import std.text.unicode;
fn main() {
println(unicode.is_upper(65)); // true ('A')
println(unicode.is_upper(97)); // false ('a')
println(unicode.is_upper(201)); // true ('É', U+00C9)
}
is_lowerTest whether the Unicode codepoint cp is a lowercase letter.
Follows Unicode general category Ll (Lowercase Letter).
import std.text.unicode;
fn main() {
println(unicode.is_lower(97)); // true ('a')
println(unicode.is_lower(65)); // false ('A')
println(unicode.is_lower(233)); // true ('é', U+00E9)
}
is_digitTest whether the Unicode codepoint cp is a decimal digit (0-9).
Matches ASCII digits only (U+0030-U+0039). Arabic-Indic and other Unicode digit forms are not included.
import std.text.unicode;
fn main() {
println(unicode.is_digit(48)); // true ('0')
println(unicode.is_digit(57)); // true ('9')
println(unicode.is_digit(65)); // false ('A')
}
is_spaceTest whether the Unicode codepoint cp is a whitespace character.
Matches Unicode White_Space property: ASCII spaces/tabs/newlines plus Unicode-specific whitespace codepoints (U+00A0 NO-BREAK SPACE, etc.).
import std.text.unicode;
fn main() {
println(unicode.is_space(32)); // true (space)
println(unicode.is_space(9)); // true (\t)
println(unicode.is_space(65)); // false ('A')
}
is_letterTest whether the Unicode codepoint cp is an alphabetic letter.
Matches Unicode Alphabetic property, which includes Latin, CJK, Arabic, Devanagari, and all other script letters.
import std.text.unicode;
fn main() {
println(unicode.is_letter(65)); // true ('A')
println(unicode.is_letter(0x65E5)); // true ('日')
println(unicode.is_letter(48)); // false ('0')
}
is_alnumTest whether the Unicode codepoint cp is a letter or decimal digit.
Equivalent to is_letter(cp) || is_digit(cp). Matches Unicode
Alphabetic or Decimal_Number properties.
import std.text.unicode;
fn main() {
println(unicode.is_alnum(65)); // true ('A')
println(unicode.is_alnum(48)); // true ('0')
println(unicode.is_alnum(32)); // false (space)
}
is_punctTest whether cp is punctuation.
import std.text.unicode;
fn main() {
println(unicode.is_punct(33)); // true ('!')
println(unicode.is_punct(0x3002)); // true (IDEOGRAPHIC FULL STOP)
println(unicode.is_punct(65)); // false ('A')
}
Punctuation is the Unicode general category group P* — Pc, Pd, Ps,
Pe, Pi, Pf, and Po — decided by the Unicode tables the runtime is
pinned to. Returns false for values that are not Unicode scalars.
to_upperConvert the Unicode codepoint cp to its uppercase equivalent.
Returns the first codepoint of the Unicode to-uppercase mapping. For
codepoints with no uppercase form (digits, punctuation, already-uppercase
letters), returns cp unchanged. Invalid codepoints are returned
unchanged.
import std.text.unicode;
fn main() {
println(unicode.to_upper(97)); // 65 ('a' -> 'A')
println(unicode.to_upper(65)); // 65 ('A' unchanged)
println(unicode.to_upper(233)); // 201 ('é' -> 'É')
}
to_lowerConvert the Unicode codepoint cp to its lowercase equivalent.
Returns the first codepoint of the Unicode to-lowercase mapping. For
codepoints with no lowercase form, returns cp unchanged. Invalid
codepoints are returned unchanged.
import std.text.unicode;
fn main() {
println(unicode.to_lower(65)); // 97 ('A' -> 'a')
println(unicode.to_lower(97)); // 97 ('a' unchanged)
println(unicode.to_lower(201)); // 233 ('É' -> 'é')
}
to_titleConvert cp to titlecase.
For scalar mappings currently exposed here, titlecase is the same single-codepoint mapping as uppercase.
import std.text.unicode;
fn main() {
println(unicode.to_title(97)); // 65 ('a' -> 'A')
println(unicode.to_title(233)); // 201 ('é' -> 'É')
println(unicode.to_title(49)); // 49 ('1' unchanged)
}
codepoint_atReturn the Unicode codepoint at codepoint-index i, or an error.
import std.text.unicode;
fn main() {
match unicode.codepoint_at("café", 3) {
.Ok(cp) => println(cp), // 233
.Err(message) => println(message),
}
match unicode.codepoint_at("hi", 2) {
.Ok(cp) => println(cp),
.Err(message) => println(message), // index out of range
}
}
rune_countReturn the number of Unicode codepoints (runes) in string s.
This equals s.len(): both count Unicode codepoints. Use s.byte_len()
for the UTF-8 byte length. Strings containing multi-byte sequences
(accented letters, CJK, emoji) have fewer codepoints than bytes.
import std.text.unicode;
fn main() {
println(unicode.rune_count("hello")); // 5
println(unicode.rune_count("café")); // 4
println(unicode.rune_count("日本語")); // 3
println(unicode.rune_count("")); // 0
}
rune_lenReturn the number of UTF-8 bytes needed to encode cp, or an error.
import std.text.unicode;
fn main() {
match unicode.rune_len(65) {
.Ok(length) => println(length), // 1
.Err(message) => println(message),
}
match unicode.rune_len(0x1F642) {
.Ok(length) => println(length), // 4
.Err(message) => println(message),
}
match unicode.rune_len(-1) {
.Ok(length) => println(length),
.Err(message) => println(message), // invalid codepoint
}
}
runesReturn all Unicode codepoints in s as a Vec<i64>.
Each element is a Unicode scalar value (codepoint). The length of the
returned vec equals rune_count(s). Order matches the original string.
import std.text.unicode;
fn main() {
let cps = unicode.runes("AB");
// cps == [65, 66]
let cps2 = unicode.runes("日");
// cps2.get(0) == 0x65E5
let empty = unicode.runes("");
// empty.len() == 0
}