package core:unicode/utf8

⌘K
Ctrl+K
or
/

    Overview

    Procedures and constants to support text-encoding in the UTF-8 character encoding.

    Packages 1

    utf8stringA convenient and efficient way to index strings by Unicode code point (rune) rather than byte.

    Types

    Accept_Range ¶

    Accept_Range :: struct {
    	lo: u8,
    	hi: u8,
    }

    Grapheme ¶

    Grapheme :: struct {
    	text:       string,
    	// The text of the grapheme, a slice of the string it was decoded from.
    	byte_index: int,
    	rune_index: int,
    	width:      int,
    }

    Grapheme_Cluster_Sequence ¶

    Grapheme_Cluster_Sequence :: enum int {
    	None, 
    	Indic, 
    	Emoji, 
    	Regional, 
    }

    Grapheme_Iterator ¶

    Grapheme_Iterator :: struct {
    	str:                        string,
    	curr_offset:                int,
    	grapheme_count:             int,
    	// The number of graphemes in the string
    	rune_count:                 int,
    	// The number of runes in the string
    	width:                      int,
    	// The widrth of the string in number of monospace cells
    	last_rune:                  rune,
    	last_rune_breaks_forward:   bool,
    	last_grapheme_count:        int,
    	bypass_next_rune:           bool,
    	regional_indicator_counter: int,
    	current_sequence:           Grapheme_Cluster_Sequence,
    	continue_sequence:          bool,
    	current_grapheme:           Grapheme,
    	continue_grapheme:          bool,
    	current_cluster_width:      int,
    	current_cluster_ri_count:   int,
    }

    Constants

    HICB ¶

    HICB :: 0b1011_1111

    LOCB ¶

    LOCB :: 0b1000_0000
     

    The default lowest and highest continuation byte.

    MASK2 ¶

    MASK2 :: 0b0001_1111

    MASK3 ¶

    MASK3 :: 0b0000_1111

    MASK4 ¶

    MASK4 :: 0b0000_0111

    MASKX ¶

    MASKX :: 0b0011_1111

    MAX_RUNE ¶

    MAX_RUNE :: '\U0010ffff'

    RUNE1_MAX ¶

    RUNE1_MAX :: 1 << 7 - 1

    RUNE2_MAX ¶

    RUNE2_MAX :: 1 << 11 - 1

    RUNE3_MAX ¶

    RUNE3_MAX :: 1 << 16 - 1

    RUNE_ERROR ¶

    RUNE_ERROR :: '\ufffd'

    SURROGATE_HIGH_MAX ¶

    SURROGATE_HIGH_MAX :: 0xdbff
     

    A high/leading surrogate is in range SURROGATE_MIN..SURROGATE_HIGH_MAX, A low/trailing surrogate is in range SURROGATE_LOW_MIN..SURROGATE_MAX.

    SURROGATE_LOW_MIN ¶

    SURROGATE_LOW_MIN :: 0xdc00

    SURROGATE_MAX ¶

    SURROGATE_MAX :: 0xdfff

    SURROGATE_MIN ¶

    SURROGATE_MIN :: 0xd800

    T1 ¶

    T1 :: 0b0000_0000

    T2 ¶

    T2 :: 0b1100_0000

    T3 ¶

    T3 :: 0b1110_0000

    T4 ¶

    T4 :: 0b1111_0000

    T5 ¶

    T5 :: 0b1111_1000

    TX ¶

    TX :: 0b1000_0000

    ZERO_WIDTH_JOINER ¶

    ZERO_WIDTH_JOINER :: unicode.ZERO_WIDTH_JOINER

    Variables

    accept_sizes ¶

    accept_sizes: [256]u8 = …

    Procedures

    decode_grapheme_clusters ¶

    @(require_results)
    decode_grapheme_clusters :: proc(
    	str:             string, 
    	track_graphemes: bool = true, 
    	allocator := context.allocator, 
    ) -> (graphemes: [dynamic]Grapheme, grapheme_count: int, rune_count: int, width: int) {…}
     

    Decode the individual graphemes in a UTF-8 string.

    Allocates Using Provided Allocator

    Inputs:

    • str The input string.
    • track_graphemes Whether or not to allocate and return graphemes with extra data about each grapheme.
    • allocator (default: context.allocator)

    Returns:

    • graphemes Extra data about each grapheme.
    • grapheme_count The number of graphemes in the string.
    • rune_count The number of runes in the string.
    • width The width of the string in number of monospace cells.

    decode_grapheme_iterate ¶

    @(require_results)
    decode_grapheme_iterate :: proc(it: ^Grapheme_Iterator) -> (text: string, grapheme: Grapheme, ok: bool) {…}

    decode_grapheme_iterator_make ¶

    @(require_results)
    decode_grapheme_iterator_make :: proc(str: string) -> (it: Grapheme_Iterator) {…}

    decode_last_rune_in_bytes ¶

    @(require_results)
    decode_last_rune_in_bytes :: proc "contextless" (s: []u8) -> (rune, int) {…}

    decode_last_rune_in_string ¶

    @(require_results)
    decode_last_rune_in_string :: proc "contextless" (s: string) -> (rune, int) {…}

    decode_rune_in_bytes ¶

    @(require_results)
    decode_rune_in_bytes :: proc "contextless" (s: []u8) -> (rune, int) {…}

    decode_rune_in_string ¶

    @(require_results)
    decode_rune_in_string :: proc "contextless" (s: string) -> (rune, int) {…}

    encode_rune ¶

    @(require_results)
    encode_rune :: proc "contextless" (c: rune) -> ([4]u8, int) {…}

    full_rune_in_bytes ¶

    @(require_results)
    full_rune_in_bytes :: proc "contextless" (b: []u8) -> bool {…}
     

    full_rune_in_bytes reports if the bytes in b begin with a full utf-8 encoding of a rune or not An invalid encoding is considered a full rune since it will convert as an error rune of width 1 (RUNE_ERROR)

    full_rune_in_string ¶

    @(require_results)
    full_rune_in_string :: proc "contextless" (s: string) -> bool {…}
     

    full_rune_in_string reports if the bytes in s begin with a full utf-8 encoding of a rune or not An invalid encoding is considered a full rune since it will convert as an error rune of width 1 (RUNE_ERROR)

    grapheme_count ¶

    @(require_results)
    grapheme_count :: proc(str: string) -> (graphemes, runes, width: int) {…}
     

    Count the individual graphemes in a UTF-8 string.

    Inputs:

    • str The input string.

    Returns:

    • graphemes The number of graphemes in the string.
    • runes The number of runes in the string.
    • width The width of the string in number of monospace cells.

    is_emoji_extended_pictographic ¶

    is_emoji_extended_pictographic :: unicode.is_emoji_extended_pictographic
     

    Extended_Pictographic

    is_gcb_extend_class ¶

    is_gcb_extend_class :: unicode.is_gcb_extend_class
     

    For grapheme text segmentation, from Unicode TR 29 Rev 43:

    Grapheme_Extend = Yes, or
    Emoji_Modifier = Yes
    
    This includes:
    General_Category = Nonspacing_Mark
    General_Category = Enclosing_Mark
    U+200C ZERO WIDTH NON-JOINER
    
    plus a few General_Category = Spacing_Mark needed for canonical equivalence.
    

    is_gcb_prepend_class ¶

    is_gcb_prepend_class :: unicode.is_gcb_prepend_class
     

    For grapheme text segmentation, from Unicode TR 29 Rev 43:

    Indic_Syllabic_Category = Consonant_Preceding_Repha, or
    Indic_Syllabic_Category = Consonant_Prefixed, or
    Prepended_Concatenation_Mark = Yes
    

    is_hangul_syllable_leading ¶

    is_hangul_syllable_leading :: unicode.is_hangul_syllable_leading
     

    Hangul_Syllable_Type=Leading_Jamo

    is_hangul_syllable_lv ¶

    is_hangul_syllable_lv :: unicode.is_hangul_syllable_lv
     

    Hangul_Syllable_Type=LV_Syllable

    is_hangul_syllable_lvt ¶

    is_hangul_syllable_lvt :: unicode.is_hangul_syllable_lvt
     

    Hangul_Syllable_Type=LVT_Syllable

    is_hangul_syllable_trailing ¶

    is_hangul_syllable_trailing :: unicode.is_hangul_syllable_trailing
     

    Hangul_Syllable_Type=Trailing_Jamo

    is_hangul_syllable_vowel ¶

    is_hangul_syllable_vowel :: unicode.is_hangul_syllable_vowel
     

    Hangul_Syllable_Type=Vowel_Jamo

    is_indic_conjunct_break_consonant ¶

    is_indic_conjunct_break_consonant :: unicode.is_indic_conjunct_break_consonant
     

    Indic_Conjunct_Break=Consonant

    is_indic_conjunct_break_extend ¶

    is_indic_conjunct_break_extend :: unicode.is_indic_conjunct_break_extend
     

    Indic_Conjunct_Break=Extend

    is_indic_conjunct_break_linker ¶

    is_indic_conjunct_break_linker :: unicode.is_indic_conjunct_break_linker
     

    Indic_Conjunct_Break=Linker

    is_regional_indicator ¶

    is_regional_indicator :: unicode.is_regional_indicator
     

    Regional_Indicator

    is_spacing_mark ¶

    is_spacing_mark :: unicode.is_spacing_mark
     

    General_Category=Spacing_Mark

    normalized_east_asian_width ¶

    normalized_east_asian_width :: unicode.normalized_east_asian_width
     

    Return values:

    • 2 if East_Asian_Width=F or W, or
    • 0 if non-printable / zero-width, or
    • 1 in all other cases.

    rune_at ¶

    @(require_results)
    rune_at :: proc "contextless" (s: string, byte_index: int) -> rune {…}

    rune_at_pos ¶

    @(require_results)
    rune_at_pos :: proc "contextless" (s: string, pos: int) -> rune {…}

    rune_count_in_bytes ¶

    @(require_results)
    rune_count_in_bytes :: proc "contextless" (s: []u8) -> int {…}

    rune_count_in_string ¶

    @(require_results)
    rune_count_in_string :: proc(s: string) -> int {…}

    rune_offset ¶

    @(require_results)
    rune_offset :: proc "contextless" (s: string, pos: int, start: int = 0) -> int {…}
     

    Returns the byte position of rune at position pos in s with an optional start byte position. Returns -1 if it runs out of the string.

    rune_size ¶

    @(require_results)
    rune_size :: proc "contextless" (r: rune) -> int {…}

    rune_start ¶

    @(require_results)
    rune_start :: proc "contextless" (b: u8) -> bool {…}

    rune_string_at_pos ¶

    @(require_results)
    rune_string_at_pos :: proc "contextless" (s: string, pos: int) -> string {…}

    runes_to_string ¶

    @(require_results)
    runes_to_string :: proc(
    	runes:     []rune, 
    	allocator := context.allocator, 
    ) -> (s: string, err: runtime.Allocator_Error) #optional_ok {…}

    string_to_runes ¶

    @(require_results)
    string_to_runes :: proc(
    	s:         string, 
    	allocator := context.allocator, 
    ) -> (runes: []rune, err: runtime.Allocator_Error) #optional_ok {…}

    valid_rune ¶

    @(require_results)
    valid_rune :: proc "contextless" (r: rune) -> bool {…}

    valid_string ¶

    @(require_results)
    valid_string :: proc "contextless" (s: string) -> bool {…}

    Procedure Groups

    full_rune ¶

    full_rune :: proc{
    	full_rune_in_bytes,
    	full_rune_in_string,
    }
    
     

    full_rune reports if the bytes in b begin with a full utf-8 encoding of a rune or not An invalid encoding is considered a full rune since it will convert as an error rune of width 1 (RUNE_ERROR)

    Source Files

    Generation Information

    Generated with odin version dev-2026-10 (vendor "odin") Windows_amd64 @ 2026-10-06 16:45:10.733986900 +0000 UTC