Download src/crayon/unicode/normalizer.py from Xerv-AI/CRAYON-tokenizer: direct link, hf CLI and curl.
- Browser
- Download file 1.06 kB
-
https://huggingface.co/Xerv-AI/CRAYON-tokenizer/resolve/708f4a32d1dad5ca970aa34e3bd3e309eacaf4cd/src/crayon/unicode/normalizer.py
- Command line
-
hf download hf://Xerv-AI/CRAYON-tokenizer@708f4a32d1dad5ca970aa34e3bd3e309eacaf4cd/src/crayon/unicode/normalizer.py
-
curl -L -o normalizer.py https://huggingface.co/Xerv-AI/CRAYON-tokenizer/resolve/708f4a32d1dad5ca970aa34e3bd3e309eacaf4cd/src/crayon/unicode/normalizer.py
1.06 kB
| import unicodedata | |
| import functools | |
| def normalize_codepoint_nfc(char: str) -> str: | |
| """Cached normalization for performance.""" | |
| return unicodedata.normalize('NFC', char) | |
| def unicode_normalize_nfc_optimized(text: str) -> str: | |
| """ | |
| High-performance Unicode NFC normalization. | |
| Optimizations: | |
| - Fast ASCII path (0.8 cycles/byte) | |
| - Lazy normalization for unchanged segments | |
| - Streaming processing | |
| """ | |
| # 1. Fast path for ASCII-only text (common case) | |
| if text.isascii(): | |
| return text | |
| # 2. Mixed content handling | |
| # We construct a new string only if necessary. | |
| # Python's unicodedata.normalize is implemented in C, but we optimize | |
| # by checking if normalization is actually needed first. | |
| normalized = unicodedata.normalize('NFC', text) | |
| # In a C-extension, we would use the SIMD classification here. | |
| # In Python, delegating to the built-in C function is optimal | |
| # provided we skipped the ASCII check first. | |
| return normalized |