forked from KolibriOS/kolibrios
Counting codepoints and grapheme in utf8
This commit is contained in:
1 parent
2ae208fc20
commit
4d34344a4d
5 files changed
+505
No files matched your search
@@ -0,0 +1,82 @@
|
||||
format PE console
|
||||
entry start
|
||||
|
||||
include 'win32a.inc' ;Adjust according to your own path
|
||||
|
||||
section '.data' data readable writeable
|
||||
|
||||
|
||||
teststr db 0xC2, 0xA2, 0xE2, 0x82, 0xAC, 0xF0, 0x90, 0x90, 0xB7, 0
|
||||
msg db 'Total codepoints: %d', 10, 0
|
||||
err_msg1 db 'Error: Malformed UTF-8.[bad rune]', 10, 0
|
||||
err_msg2 db 'Error: Malformed UTF-8.[overlong]', 10, 0
|
||||
|
||||
section '.text' code readable executable
|
||||
|
||||
start:
|
||||
|
||||
push teststr
|
||||
call utflen
|
||||
add esp, 4
|
||||
|
||||
|
||||
cinvoke printf, msg, eax
|
||||
|
||||
invoke ExitProcess, 0
|
||||
|
||||
; Input: teststr (string)
|
||||
; Output: returns count of codepoints
|
||||
; ------------------------------------------------------------------------------------
|
||||
utflen:
|
||||
push ebp
|
||||
mov ebp, esp
|
||||
push ebx
|
||||
push esi
|
||||
push edi
|
||||
|
||||
mov esi, [ebp+8]
|
||||
xor ecx, ecx
|
||||
|
||||
.loop:
|
||||
movzx eax, byte [esi]
|
||||
test al, al
|
||||
jz .end
|
||||
|
||||
call charntorune
|
||||
|
||||
test ebx, ebx
|
||||
jz .advance_one
|
||||
|
||||
call overlong_check
|
||||
test eax, eax
|
||||
jnz .advance_n
|
||||
|
||||
inc ecx
|
||||
|
||||
.advance_n:
|
||||
add esi, ebx
|
||||
jmp .loop
|
||||
|
||||
.advance_one:
|
||||
inc esi
|
||||
jmp .loop
|
||||
|
||||
|
||||
.end:
|
||||
mov eax, ecx ;
|
||||
pop edi
|
||||
pop esi
|
||||
pop ebx
|
||||
pop ebp
|
||||
|
||||
ret
|
||||
|
||||
|
||||
|
||||
include "utf8_decode.inc"
|
||||
section '.idata' import data readable
|
||||
library kernel32, 'kernel32.dll', \
|
||||
msvcrt, 'msvcrt.dll'
|
||||
|
||||
import kernel32, ExitProcess, 'ExitProcess'
|
||||
import msvcrt, printf, 'printf'
|
||||
@@ -0,0 +1,174 @@
|
||||
format PE console
|
||||
entry start
|
||||
|
||||
include 'win32a.inc' ; Set your own path
|
||||
|
||||
|
||||
|
||||
; Properties
|
||||
prop_other = 0
|
||||
prop_cr = 1
|
||||
prop_lf = 2
|
||||
prop_control = 3
|
||||
prop_extend = 4
|
||||
prop_zwj = 5
|
||||
prop_spacingmark = 6
|
||||
prop_RI = 7 ; Regional Indicator-Special Case
|
||||
|
||||
section '.data' data readable writeable
|
||||
;Testcase here
|
||||
teststr db 0xF0, 0x9F, 0x87, 0xB5, 0xF0, 0x9F, 0x87, 0xB0, 0xF0, 0x9F, 0x87, 0xA6, 0
|
||||
msg db 'Total graphemes: %d', 10, 0
|
||||
err_msg db 'Error: Malformed sequence.', 10, 0
|
||||
err_msg1 = err_msg
|
||||
err_msg2 = err_msg
|
||||
|
||||
|
||||
; Mask for Other-->Extend(4), ZWJ(5), SpacingMark(6) = (1<<4)|(1<<5)|(1<<6) = 0x70 - Precomputed values!
|
||||
dont_break:
|
||||
dd 0x00000070 ; 0:Other
|
||||
dd 0x00000004 ; 1: CR
|
||||
dd 0x00000000 ; 2: LF
|
||||
dd 0x00000000 ; 3: Control
|
||||
dd 0x00000070 ; 4: Extend
|
||||
dd 0x00000070 ; 5: ZWJ
|
||||
dd 0x00000070 ; 6: Spacingmark
|
||||
dd 0x00000070 ; 7: RI -> RI to RI is handled by the explicit state machine below
|
||||
|
||||
; Intervals for quick lookup
|
||||
;quick note, this is oversimplification of the actual intervals of properties- these are just made for the sake of simpilicity
|
||||
intervals:
|
||||
dd 0x000D, 0x000D, prop_cr
|
||||
dd 0x000A, 0x000A, prop_lf
|
||||
dd 0x0000, 0x001F, prop_control
|
||||
dd 0x0300, 0x036F, prop_extend
|
||||
dd 0x200D, 0x200D, prop_zwj
|
||||
dd 0x093E, 0x094C, prop_spacingmark
|
||||
dd 0x1F1E6, 0x1F1FF, prop_RI
|
||||
|
||||
dd 0xFFFFFFFF, 0, 0
|
||||
|
||||
section '.text' code readable executable
|
||||
start:
|
||||
push teststr
|
||||
call count_graphemes
|
||||
add esp, 4
|
||||
|
||||
cinvoke printf, msg, eax
|
||||
|
||||
invoke ExitProcess, 0
|
||||
;--------------------
|
||||
count_graphemes:
|
||||
push ebp
|
||||
mov ebp, esp
|
||||
sub esp, 8 ; [ebp-4] = storing previous propertry,[ebp-8] = ri_odd flag
|
||||
push ebx
|
||||
push esi
|
||||
push edi
|
||||
mov esi, [ebp+8]
|
||||
xor ecx, ecx
|
||||
|
||||
mov dword [ebp-4], prop_control
|
||||
mov dword [ebp-8], 0
|
||||
.loop:
|
||||
movzx eax, byte [esi]
|
||||
test al, al
|
||||
jz .done
|
||||
|
||||
call charntorune
|
||||
cmp edx, 0xFFFD
|
||||
je .handle_error
|
||||
call overlong_check
|
||||
test eax, eax
|
||||
jnz .handle_error
|
||||
|
||||
push edx
|
||||
call .get_prop ; eax = curr_prop
|
||||
pop edx
|
||||
|
||||
|
||||
; Indicator State Machine
|
||||
; --------------------------------
|
||||
cmp eax, prop_RI
|
||||
jne .not_ri
|
||||
|
||||
; It IS a Regional Indicator. Check state memory.
|
||||
cmp dword [ebp-8], 1
|
||||
je .ri_glue ; State is odd. Glue it to make a flag
|
||||
|
||||
; State is even (this is the 1st half of a new flag). Set state to odd.
|
||||
mov dword [ebp-8], 1
|
||||
jmp .do_bitmask
|
||||
.ri_glue:
|
||||
mov dword [ebp-8], 0 ; Reset state to even (flag is complete)
|
||||
jmp .glue
|
||||
|
||||
.not_ri:
|
||||
; Chain broken by a normal character. Reset the RI state memory.
|
||||
mov dword [ebp-8], 0
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
; Standard Stateless Bitmask
|
||||
;------------------------------------------------
|
||||
.do_bitmask:
|
||||
mov edi, [ebp-4]
|
||||
mov edi, dword [dont_break + edi*4]
|
||||
bt edi, eax
|
||||
jc .glue
|
||||
|
||||
;Break
|
||||
inc ecx
|
||||
|
||||
.glue:
|
||||
mov [ebp-4], eax ; Save current prop for comparison ahead
|
||||
add esi, ebx
|
||||
jmp .loop
|
||||
|
||||
.handle_error:
|
||||
pusha
|
||||
cinvoke printf, err_msg
|
||||
popa
|
||||
mov dword [ebp-4], prop_control
|
||||
mov dword [ebp-8], 0
|
||||
add esi, ebx
|
||||
jmp .loop
|
||||
.done:
|
||||
mov eax, ecx
|
||||
pop edi
|
||||
pop esi
|
||||
pop ebx
|
||||
mov esp, ebp
|
||||
pop ebp
|
||||
ret
|
||||
|
||||
; --------------------------------------------------------------------
|
||||
.get_prop:
|
||||
mov edi, intervals
|
||||
.search_loop:
|
||||
mov eax, [edi]
|
||||
cmp eax, 0xFFFFFFFF
|
||||
je .not_found
|
||||
cmp edx, eax
|
||||
jb .next_interval
|
||||
cmp edx, [edi+4]
|
||||
ja .next_interval
|
||||
mov eax, [edi+8]
|
||||
ret
|
||||
.next_interval:
|
||||
add edi, 12
|
||||
jmp .search_loop
|
||||
.not_found:
|
||||
mov eax, prop_other
|
||||
ret
|
||||
|
||||
|
||||
include "utf8_decode.inc"
|
||||
section '.idata' import data readable
|
||||
library kernel32, 'kernel32.dll', \
|
||||
msvcrt, 'msvcrt.dll'
|
||||
import kernel32, ExitProcess, 'ExitProcess'
|
||||
import msvcrt, printf, 'printf'
|
||||
@@ -0,0 +1,33 @@
|
||||
Counting codepoints and graphemes [visible characters]
|
||||
|
||||
Quick note: Please update your own path for including win32a.inc to run it properly
|
||||
check testcases for testing
|
||||
|
||||
This doc contains explanation and specification/implementation details: https://docs.google.com/document/d/1__C2qngd7kEUPYF-H4b1DijanunrZPBQwPgOs9SddOw/edit?usp=sharing
|
||||
|
||||
|
||||
Summary
|
||||
Codepoint/Rune counting:
|
||||
Variable length parsing: Based on the specification, simple comparisons + bitwise operations can help in finding the number of bytes in a codepoint. The important part is to reconstruct from distributed bytes into a scalar value which is done through bitmasking + shift left operation. Refer to .charntorune for checking implementation.
|
||||
|
||||
Checks for malformed sequences and overlong sequences have also been implemented:
|
||||
1 - missing continuation bytes
|
||||
2 - invalid leading bytes
|
||||
3 - verifying codepoint encoded using the minimum required bytes.
|
||||
|
||||
Grapheme counting:
|
||||
Full implementation of grapheme counting is not done, some simplification has been done but it gives a good idea on how to approach the problem. Let's leave some work for GSOC as well.
|
||||
|
||||
Overview:
|
||||
1-Get code point value → 2-find attribute → 3-compare it with previous codepoint attribute → 4-if it is the right match → 5-move ahead otherwise break and increment count → 6- if there is an attribute which requires history (like knowing what was there 2-3 bytes prior) then a state machine is used to cater it.
|
||||
|
||||
There are 3 types of properties:
|
||||
1 - Standard Properties (Bitmask compatible) [CR, LF etc]: These can be handled pretty easily. They only require comparison of left and right codepoint and no history has to be maintained. Each "Left Property" gets a 32-bit integer row. Each bit in that row represents a "Right Property" that it should glue to. The CPU evaluates complex grapheme boundaries in a single clock cycle using the BT (Bit Test) instruction. If the bit is 1, the codepoints glue. If 0, they break. Essentially we have properties of left and right stored in an integer, and we are checking if the properties that left matches with exist in right. Note: matches here mean the right combination dictated by unicode rules, not literal values matching. For these cases we precompute these values and store them as they are not going to change.
|
||||
|
||||
2 - Hangul Syllable Properties [L, V, T, LV, LVT].
|
||||
|
||||
3 - State-Machine Properties [GB9C, 11, 12, 13]: They can be handled by creating a state machine (DFA). It is not difficult to implement but time taking for different rules so one of them has been implemented. Regional Indicators are handled by tracking a boolean on the stack. This ensures consecutive country code letters strictly form pairs and break into separate clusters upon a third occurrence.
|
||||
|
||||
Step Explanations:
|
||||
1 - Get code point value: check chartorune.
|
||||
2 - Find Attribute: Current Implementation is oversimplified but follows the suckless approach which is creating a data structure based on the intervals of different attributes and then running bin search to find the attribute. Currently we have a small range of intervals through which we do linear search for the sake of simplicity. In actual there are over 1000 distinct ranges.
|
||||
@@ -0,0 +1,70 @@
|
||||
Codepoint counting testcases:
|
||||
; ---------------------------------------------------------
|
||||
; Test 8: Pure ASCII
|
||||
; "Hello" (5 bytes).
|
||||
; Expected: 5 codepoints.
|
||||
teststr db 'Hello', 0
|
||||
|
||||
; ---------------------------------------------------------
|
||||
; Test 9: Valid Multi-byte Mix
|
||||
; ¢ (2 bytes) + € (3 bytes) + 𐐷 (4 bytes).
|
||||
; 0xC2,0xA2 + 0xE2,0x82,0xAC + 0xF0,0x90,0x90,0xB7
|
||||
; Expected: 3 codepoints.
|
||||
teststr db 0xC2, 0xA2, 0xE2, 0x82, 0xAC, 0xF0, 0x90, 0x90, 0xB7, 0
|
||||
|
||||
; ---------------------------------------------------------
|
||||
; Test 10: Invalid Leading Byte
|
||||
; 'A' + 0xFF + 'B'.
|
||||
; Expected: 3 codepoints ('A', U+FFFD replacement, 'B').
|
||||
teststr db 'A', 0xFF, 'B', 0
|
||||
|
||||
; ---------------------------------------------------------
|
||||
; Test 11: Truncated Sequence (Missing continuation byte)
|
||||
; 'X' + 0xE2, 0x82 (Incomplete Euro sign) + 'Y'.
|
||||
; Expected: 3 or 4 codepoints depending on exact error recovery overlap.
|
||||
teststr db 'X', 0xE2, 0x82, 'Y', 0
|
||||
|
||||
; ---------------------------------------------------------
|
||||
; Test 12: Overlong Encoding (Security Check)
|
||||
; 'M' + Overlong Null (0xC0, 0x80) + 'N'.
|
||||
; Expected: 2 codepoints.
|
||||
teststr db 'M', 0xC0, 0x80, 'N', 0
|
||||
|
||||
|
||||
Grapheme_counting testcases:
|
||||
|
||||
; ---------------------------------------------------------
|
||||
; Test 1: Basic ASCII
|
||||
; "A", "B", "C".
|
||||
; Expected: 3 graphemes.
|
||||
teststr db 'ABC', 0
|
||||
|
||||
; ---------------------------------------------------------
|
||||
; Test 2: CRLF Rule (GB3)
|
||||
; Carriage Return (0x0D) + Line Feed (0x0A).
|
||||
; Expected: 1 grapheme.
|
||||
teststr db 0x0D, 0x0A, 0
|
||||
|
||||
; ---------------------------------------------------------
|
||||
; Test 3: Combining Marks / Extend (GB9)
|
||||
; 'e' (0x65) + Combining Acute Accent U+0301 (0xCC, 0x81).
|
||||
; Expected: 1 grapheme.
|
||||
teststr db 'e', 0xCC, 0x81, 0
|
||||
|
||||
; ---------------------------------------------------------
|
||||
; Test 4: Regional Indicator Pair (GB12/GB13)
|
||||
; Flag of Pakistan 🇵🇰 (P = U+1F1F5, K = U+1F1F0).
|
||||
; Expected: 1 grapheme.
|
||||
teststr db 0xF0, 0x9F, 0x87, 0xB5, 0xF0, 0x9F, 0x87, 0xB0, 0
|
||||
|
||||
; ---------------------------------------------------------
|
||||
; Test 5: Broken Regional Indicator (3 RIs)
|
||||
; 🇵🇰 + 🇦 (A = U+1F1E6).
|
||||
; Expected: 2 graphemes.
|
||||
teststr db 0xF0, 0x9F, 0x87, 0xB5, 0xF0, 0x9F, 0x87, 0xB0, 0xF0, 0x9F, 0x87, 0xA6, 0
|
||||
|
||||
; ---------------------------------------------------------
|
||||
; Test 6: Invalid UTF-8 Structural Error
|
||||
; 'A' + 0xFF (Invalid lead byte) + 'B'.
|
||||
; Expected: 3 graphemes. (The 0xFF becomes a U+FFFD fallback character).
|
||||
teststr db 'A', 0xFF, 'B', 0
|
||||
@@ -0,0 +1,146 @@
|
||||
; Returns rune(integer) of the codepoint-useful for grapheme cluster boundary
|
||||
; --------------------------------------------------------------------
|
||||
charntorune:
|
||||
movzx eax, byte [esi]
|
||||
call .utfseq ; ebx = n (size of character) can be 1,2,3,4 byte
|
||||
|
||||
test ebx, ebx
|
||||
jz .rune_error
|
||||
|
||||
cmp ebx, 1
|
||||
je .type_1
|
||||
cmp ebx, 2
|
||||
je .type_2
|
||||
cmp ebx, 3
|
||||
je .type_3
|
||||
cmp ebx, 4
|
||||
je .type_4
|
||||
jmp .rune_error
|
||||
|
||||
.type_1:
|
||||
movzx edx, al
|
||||
jmp .cont_done ; no continuation loop needed for 1 byte cahr
|
||||
|
||||
.type_2:
|
||||
movzx edx, al
|
||||
and edx, 0x1F ; strip 110xxxxx
|
||||
jmp .cont_loop
|
||||
|
||||
.type_3:
|
||||
movzx edx, al
|
||||
and edx, 0x0F ; strip 1110xxxx
|
||||
jmp .cont_loop
|
||||
.type_4:
|
||||
movzx edx, al
|
||||
and edx, 0x07 ; strip 11110xxx
|
||||
|
||||
|
||||
.cont_loop:
|
||||
mov edi, 1 ;consider this i
|
||||
|
||||
.cont_next:
|
||||
cmp edi, ebx ; i < n check.
|
||||
jge .cont_done
|
||||
|
||||
movzx eax, byte [esi+edi]
|
||||
|
||||
;verify continuation byte: (s[i] & 0xC0) == 0x80
|
||||
mov ah, al
|
||||
and ah, 0xC0
|
||||
cmp ah, 0x80
|
||||
jne .rune_error ; not a continuation byte
|
||||
|
||||
; r = (r << 6) | (s[i] & 0x3F)
|
||||
shl edx, 6
|
||||
and eax, 0x3F ; strip 10xxxxxx
|
||||
or edx, eax
|
||||
|
||||
inc edi ; i++
|
||||
jmp .cont_next
|
||||
|
||||
.cont_done:
|
||||
ret
|
||||
|
||||
.rune_error:
|
||||
pusha
|
||||
cinvoke printf, err_msg1
|
||||
popa
|
||||
mov edx, 0xFFFD
|
||||
mov ebx, 1
|
||||
ret
|
||||
|
||||
|
||||
; Input: al : leading byte
|
||||
; OUTPUT: expected sequence length (1,2,3,4)- 0:if invalid
|
||||
; --------------------------------------------------------------------------------------
|
||||
.utfseq:
|
||||
test al, 0x80 ; 0xxxxxxx-1byte
|
||||
jz .seq_1
|
||||
|
||||
mov bl, al
|
||||
and bl, 0xC0
|
||||
cmp bl, 0x80 ; 10xxxxxx (invalid leader)
|
||||
je .seq_0
|
||||
|
||||
mov bl, al
|
||||
and bl, 0xE0
|
||||
cmp bl, 0xC0 ; 110xxxxx-2byte
|
||||
je .seq_2
|
||||
|
||||
mov bl, al
|
||||
and bl, 0xF0
|
||||
cmp bl, 0xE0 ; 1110xxxx-3byte
|
||||
je .seq_3
|
||||
|
||||
mov bl, al
|
||||
and bl, 0xF8
|
||||
cmp bl, 0xF0 ; 11110xxx-4byte
|
||||
je .seq_4
|
||||
|
||||
jmp .seq_0 ; 11111xxx (invalid)
|
||||
|
||||
.seq_0: mov ebx, 0
|
||||
ret
|
||||
.seq_1: mov ebx, 1
|
||||
ret
|
||||
.seq_2: mov ebx, 2
|
||||
ret
|
||||
.seq_3: mov ebx, 3
|
||||
ret
|
||||
.seq_4: mov ebx, 4
|
||||
ret
|
||||
|
||||
|
||||
; Verifies whether code follows the shortest form rule
|
||||
; Output: returns 1 or 0 in eax for valid/invalid
|
||||
; --------------------------------------------------------------------
|
||||
overlong_check:
|
||||
cmp ebx, 1
|
||||
je .ol_valid ;check1
|
||||
|
||||
cmp ebx, 2
|
||||
jne .check3
|
||||
cmp edx, 0x7F ;check2
|
||||
jg .ol_valid
|
||||
jmp .ol_invalid
|
||||
|
||||
.check3:
|
||||
cmp ebx, 3
|
||||
jne .check4
|
||||
cmp edx, 0x7FF
|
||||
jg .ol_valid
|
||||
jmp .ol_invalid
|
||||
|
||||
.check4:
|
||||
cmp edx, 0xFFFF
|
||||
jg .ol_valid
|
||||
|
||||
.ol_invalid:
|
||||
pusha
|
||||
cinvoke printf, err_msg2
|
||||
popa
|
||||
mov eax, 1
|
||||
ret
|
||||
.ol_valid:
|
||||
xor eax, eax
|
||||
ret
|
||||
Reference in new issue
Block a user