Counting codepoints and grapheme in utf8

This commit is contained in:
Ahmad Abdul Rehman committed 2026-03-20 11:13:45 +05:00
1 parent 2ae208fc20
commit 4d34344a4d
5 files changed
+505

No files matched your search

@@ -0,0 +1,82 @@
format PE console
entry start
include 'win32a.inc' ;Adjust according to your own path
section '.data' data readable writeable
teststr db 0xC2, 0xA2, 0xE2, 0x82, 0xAC, 0xF0, 0x90, 0x90, 0xB7, 0
msg db 'Total codepoints: %d', 10, 0
err_msg1 db 'Error: Malformed UTF-8.[bad rune]', 10, 0
err_msg2 db 'Error: Malformed UTF-8.[overlong]', 10, 0
section '.text' code readable executable
start:
push teststr
call utflen
add esp, 4
cinvoke printf, msg, eax
invoke ExitProcess, 0
; Input: teststr (string)
; Output: returns count of codepoints
; ------------------------------------------------------------------------------------
utflen:
push ebp
mov ebp, esp
push ebx
push esi
push edi
mov esi, [ebp+8]
xor ecx, ecx
.loop:
movzx eax, byte [esi]
test al, al
jz .end
call charntorune
test ebx, ebx
jz .advance_one
call overlong_check
test eax, eax
jnz .advance_n
inc ecx
.advance_n:
add esi, ebx
jmp .loop
.advance_one:
inc esi
jmp .loop
.end:
mov eax, ecx ;
pop edi
pop esi
pop ebx
pop ebp
ret
include "utf8_decode.inc"
section '.idata' import data readable
library kernel32, 'kernel32.dll', \
msvcrt, 'msvcrt.dll'
import kernel32, ExitProcess, 'ExitProcess'
import msvcrt, printf, 'printf'
@@ -0,0 +1,174 @@
format PE console
entry start
include 'win32a.inc' ; Set your own path
; Properties
prop_other = 0
prop_cr = 1
prop_lf = 2
prop_control = 3
prop_extend = 4
prop_zwj = 5
prop_spacingmark = 6
prop_RI = 7 ; Regional Indicator-Special Case
section '.data' data readable writeable
;Testcase here
teststr db 0xF0, 0x9F, 0x87, 0xB5, 0xF0, 0x9F, 0x87, 0xB0, 0xF0, 0x9F, 0x87, 0xA6, 0
msg db 'Total graphemes: %d', 10, 0
err_msg db 'Error: Malformed sequence.', 10, 0
err_msg1 = err_msg
err_msg2 = err_msg
; Mask for Other-->Extend(4), ZWJ(5), SpacingMark(6) = (1<<4)|(1<<5)|(1<<6) = 0x70 - Precomputed values!
dont_break:
dd 0x00000070 ; 0:Other
dd 0x00000004 ; 1: CR
dd 0x00000000 ; 2: LF
dd 0x00000000 ; 3: Control
dd 0x00000070 ; 4: Extend
dd 0x00000070 ; 5: ZWJ
dd 0x00000070 ; 6: Spacingmark
dd 0x00000070 ; 7: RI -> RI to RI is handled by the explicit state machine below
; Intervals for quick lookup
;quick note, this is oversimplification of the actual intervals of properties- these are just made for the sake of simpilicity
intervals:
dd 0x000D, 0x000D, prop_cr
dd 0x000A, 0x000A, prop_lf
dd 0x0000, 0x001F, prop_control
dd 0x0300, 0x036F, prop_extend
dd 0x200D, 0x200D, prop_zwj
dd 0x093E, 0x094C, prop_spacingmark
dd 0x1F1E6, 0x1F1FF, prop_RI
dd 0xFFFFFFFF, 0, 0
section '.text' code readable executable
start:
push teststr
call count_graphemes
add esp, 4
cinvoke printf, msg, eax
invoke ExitProcess, 0
;--------------------
count_graphemes:
push ebp
mov ebp, esp
sub esp, 8 ; [ebp-4] = storing previous propertry,[ebp-8] = ri_odd flag
push ebx
push esi
push edi
mov esi, [ebp+8]
xor ecx, ecx
mov dword [ebp-4], prop_control
mov dword [ebp-8], 0
.loop:
movzx eax, byte [esi]
test al, al
jz .done
call charntorune
cmp edx, 0xFFFD
je .handle_error
call overlong_check
test eax, eax
jnz .handle_error
push edx
call .get_prop ; eax = curr_prop
pop edx
; Indicator State Machine
; --------------------------------
cmp eax, prop_RI
jne .not_ri
; It IS a Regional Indicator. Check state memory.
cmp dword [ebp-8], 1
je .ri_glue ; State is odd. Glue it to make a flag
; State is even (this is the 1st half of a new flag). Set state to odd.
mov dword [ebp-8], 1
jmp .do_bitmask
.ri_glue:
mov dword [ebp-8], 0 ; Reset state to even (flag is complete)
jmp .glue
.not_ri:
; Chain broken by a normal character. Reset the RI state memory.
mov dword [ebp-8], 0
; Standard Stateless Bitmask
;------------------------------------------------
.do_bitmask:
mov edi, [ebp-4]
mov edi, dword [dont_break + edi*4]
bt edi, eax
jc .glue
;Break
inc ecx
.glue:
mov [ebp-4], eax ; Save current prop for comparison ahead
add esi, ebx
jmp .loop
.handle_error:
pusha
cinvoke printf, err_msg
popa
mov dword [ebp-4], prop_control
mov dword [ebp-8], 0
add esi, ebx
jmp .loop
.done:
mov eax, ecx
pop edi
pop esi
pop ebx
mov esp, ebp
pop ebp
ret
; --------------------------------------------------------------------
.get_prop:
mov edi, intervals
.search_loop:
mov eax, [edi]
cmp eax, 0xFFFFFFFF
je .not_found
cmp edx, eax
jb .next_interval
cmp edx, [edi+4]
ja .next_interval
mov eax, [edi+8]
ret
.next_interval:
add edi, 12
jmp .search_loop
.not_found:
mov eax, prop_other
ret
include "utf8_decode.inc"
section '.idata' import data readable
library kernel32, 'kernel32.dll', \
msvcrt, 'msvcrt.dll'
import kernel32, ExitProcess, 'ExitProcess'
import msvcrt, printf, 'printf'
@@ -0,0 +1,33 @@
Counting codepoints and graphemes [visible characters]
Quick note: Please update your own path for including win32a.inc to run it properly
check testcases for testing
This doc contains explanation and specification/implementation details: https://docs.google.com/document/d/1__C2qngd7kEUPYF-H4b1DijanunrZPBQwPgOs9SddOw/edit?usp=sharing
Summary
Codepoint/Rune counting:
Variable length parsing: Based on the specification, simple comparisons + bitwise operations can help in finding the number of bytes in a codepoint. The important part is to reconstruct from distributed bytes into a scalar value which is done through bitmasking + shift left operation. Refer to .charntorune for checking implementation.
Checks for malformed sequences and overlong sequences have also been implemented:
1 - missing continuation bytes
2 - invalid leading bytes
3 - verifying codepoint encoded using the minimum required bytes.
Grapheme counting:
Full implementation of grapheme counting is not done, some simplification has been done but it gives a good idea on how to approach the problem. Let's leave some work for GSOC as well.
Overview:
1-Get code point value → 2-find attribute → 3-compare it with previous codepoint attribute → 4-if it is the right match → 5-move ahead otherwise break and increment count → 6- if there is an attribute which requires history (like knowing what was there 2-3 bytes prior) then a state machine is used to cater it.
There are 3 types of properties:
1 - Standard Properties (Bitmask compatible) [CR, LF etc]: These can be handled pretty easily. They only require comparison of left and right codepoint and no history has to be maintained. Each "Left Property" gets a 32-bit integer row. Each bit in that row represents a "Right Property" that it should glue to. The CPU evaluates complex grapheme boundaries in a single clock cycle using the BT (Bit Test) instruction. If the bit is 1, the codepoints glue. If 0, they break. Essentially we have properties of left and right stored in an integer, and we are checking if the properties that left matches with exist in right. Note: matches here mean the right combination dictated by unicode rules, not literal values matching. For these cases we precompute these values and store them as they are not going to change.
2 - Hangul Syllable Properties [L, V, T, LV, LVT].
3 - State-Machine Properties [GB9C, 11, 12, 13]: They can be handled by creating a state machine (DFA). It is not difficult to implement but time taking for different rules so one of them has been implemented. Regional Indicators are handled by tracking a boolean on the stack. This ensures consecutive country code letters strictly form pairs and break into separate clusters upon a third occurrence.
Step Explanations:
1 - Get code point value: check chartorune.
2 - Find Attribute: Current Implementation is oversimplified but follows the suckless approach which is creating a data structure based on the intervals of different attributes and then running bin search to find the attribute. Currently we have a small range of intervals through which we do linear search for the sake of simplicity. In actual there are over 1000 distinct ranges.
@@ -0,0 +1,70 @@
Codepoint counting testcases:
; ---------------------------------------------------------
; Test 8: Pure ASCII
; "Hello" (5 bytes).
; Expected: 5 codepoints.
teststr db 'Hello', 0
; ---------------------------------------------------------
; Test 9: Valid Multi-byte Mix
; ¢ (2 bytes) + € (3 bytes) + 𐐷 (4 bytes).
; 0xC2,0xA2 + 0xE2,0x82,0xAC + 0xF0,0x90,0x90,0xB7
; Expected: 3 codepoints.
teststr db 0xC2, 0xA2, 0xE2, 0x82, 0xAC, 0xF0, 0x90, 0x90, 0xB7, 0
; ---------------------------------------------------------
; Test 10: Invalid Leading Byte
; 'A' + 0xFF + 'B'.
; Expected: 3 codepoints ('A', U+FFFD replacement, 'B').
teststr db 'A', 0xFF, 'B', 0
; ---------------------------------------------------------
; Test 11: Truncated Sequence (Missing continuation byte)
; 'X' + 0xE2, 0x82 (Incomplete Euro sign) + 'Y'.
; Expected: 3 or 4 codepoints depending on exact error recovery overlap.
teststr db 'X', 0xE2, 0x82, 'Y', 0
; ---------------------------------------------------------
; Test 12: Overlong Encoding (Security Check)
; 'M' + Overlong Null (0xC0, 0x80) + 'N'.
; Expected: 2 codepoints.
teststr db 'M', 0xC0, 0x80, 'N', 0
Grapheme_counting testcases:
; ---------------------------------------------------------
; Test 1: Basic ASCII
; "A", "B", "C".
; Expected: 3 graphemes.
teststr db 'ABC', 0
; ---------------------------------------------------------
; Test 2: CRLF Rule (GB3)
; Carriage Return (0x0D) + Line Feed (0x0A).
; Expected: 1 grapheme.
teststr db 0x0D, 0x0A, 0
; ---------------------------------------------------------
; Test 3: Combining Marks / Extend (GB9)
; 'e' (0x65) + Combining Acute Accent U+0301 (0xCC, 0x81).
; Expected: 1 grapheme.
teststr db 'e', 0xCC, 0x81, 0
; ---------------------------------------------------------
; Test 4: Regional Indicator Pair (GB12/GB13)
; Flag of Pakistan 🇵🇰 (P = U+1F1F5, K = U+1F1F0).
; Expected: 1 grapheme.
teststr db 0xF0, 0x9F, 0x87, 0xB5, 0xF0, 0x9F, 0x87, 0xB0, 0
; ---------------------------------------------------------
; Test 5: Broken Regional Indicator (3 RIs)
; 🇵🇰 + 🇦 (A = U+1F1E6).
; Expected: 2 graphemes.
teststr db 0xF0, 0x9F, 0x87, 0xB5, 0xF0, 0x9F, 0x87, 0xB0, 0xF0, 0x9F, 0x87, 0xA6, 0
; ---------------------------------------------------------
; Test 6: Invalid UTF-8 Structural Error
; 'A' + 0xFF (Invalid lead byte) + 'B'.
; Expected: 3 graphemes. (The 0xFF becomes a U+FFFD fallback character).
teststr db 'A', 0xFF, 'B', 0
@@ -0,0 +1,146 @@
; Returns rune(integer) of the codepoint-useful for grapheme cluster boundary
; --------------------------------------------------------------------
charntorune:
movzx eax, byte [esi]
call .utfseq ; ebx = n (size of character) can be 1,2,3,4 byte
test ebx, ebx
jz .rune_error
cmp ebx, 1
je .type_1
cmp ebx, 2
je .type_2
cmp ebx, 3
je .type_3
cmp ebx, 4
je .type_4
jmp .rune_error
.type_1:
movzx edx, al
jmp .cont_done ; no continuation loop needed for 1 byte cahr
.type_2:
movzx edx, al
and edx, 0x1F ; strip 110xxxxx
jmp .cont_loop
.type_3:
movzx edx, al
and edx, 0x0F ; strip 1110xxxx
jmp .cont_loop
.type_4:
movzx edx, al
and edx, 0x07 ; strip 11110xxx
.cont_loop:
mov edi, 1 ;consider this i
.cont_next:
cmp edi, ebx ; i < n check.
jge .cont_done
movzx eax, byte [esi+edi]
;verify continuation byte: (s[i] & 0xC0) == 0x80
mov ah, al
and ah, 0xC0
cmp ah, 0x80
jne .rune_error ; not a continuation byte
; r = (r << 6) | (s[i] & 0x3F)
shl edx, 6
and eax, 0x3F ; strip 10xxxxxx
or edx, eax
inc edi ; i++
jmp .cont_next
.cont_done:
ret
.rune_error:
pusha
cinvoke printf, err_msg1
popa
mov edx, 0xFFFD
mov ebx, 1
ret
; Input: al : leading byte
; OUTPUT: expected sequence length (1,2,3,4)- 0:if invalid
; --------------------------------------------------------------------------------------
.utfseq:
test al, 0x80 ; 0xxxxxxx-1byte
jz .seq_1
mov bl, al
and bl, 0xC0
cmp bl, 0x80 ; 10xxxxxx (invalid leader)
je .seq_0
mov bl, al
and bl, 0xE0
cmp bl, 0xC0 ; 110xxxxx-2byte
je .seq_2
mov bl, al
and bl, 0xF0
cmp bl, 0xE0 ; 1110xxxx-3byte
je .seq_3
mov bl, al
and bl, 0xF8
cmp bl, 0xF0 ; 11110xxx-4byte
je .seq_4
jmp .seq_0 ; 11111xxx (invalid)
.seq_0: mov ebx, 0
ret
.seq_1: mov ebx, 1
ret
.seq_2: mov ebx, 2
ret
.seq_3: mov ebx, 3
ret
.seq_4: mov ebx, 4
ret
; Verifies whether code follows the shortest form rule
; Output: returns 1 or 0 in eax for valid/invalid
; --------------------------------------------------------------------
overlong_check:
cmp ebx, 1
je .ol_valid ;check1
cmp ebx, 2
jne .check3
cmp edx, 0x7F ;check2
jg .ol_valid
jmp .ol_invalid
.check3:
cmp ebx, 3
jne .check4
cmp edx, 0x7FF
jg .ol_valid
jmp .ol_invalid
.check4:
cmp edx, 0xFFFF
jg .ol_valid
.ol_invalid:
pusha
cinvoke printf, err_msg2
popa
mov eax, 1
ret
.ol_valid:
xor eax, eax
ret