重构
This commit is contained in:
@@ -1,817 +0,0 @@
|
||||
#include <c_utf8.h>
|
||||
|
||||
c_size_t c_utf8_strlen(const char* str) {
|
||||
if (!str) return 0;
|
||||
|
||||
c_size_t char_count = 0;
|
||||
c_size_t i = 0;
|
||||
|
||||
while (str[i] != '\0') {
|
||||
i += c_utf8_char_len(str[i]); // 跳过当前字符占用的全部字节
|
||||
char_count++;
|
||||
}
|
||||
|
||||
return char_count;
|
||||
}
|
||||
|
||||
const char* c_utf8_strchr(const char* str, const char* utf8_char) {
|
||||
if (!str || !utf8_char || utf8_char[0] == '\0') return NULL;
|
||||
|
||||
c_size_t target_bytes = c_utf8_char_len(utf8_char[0]);
|
||||
c_size_t i = 0;
|
||||
|
||||
while (str[i] != '\0') {
|
||||
c_size_t curr_bytes = c_utf8_char_len(str[i]);
|
||||
|
||||
// 当且仅当两个字符占用的字节数相同,且多字节内容完全一致时匹配成功
|
||||
if (curr_bytes == target_bytes) {
|
||||
if (memcmp(&str[i], utf8_char, target_bytes) == 0) {
|
||||
return &str[i];
|
||||
}
|
||||
}
|
||||
i += curr_bytes; // 移动到下一个 UTF-8 字符
|
||||
}
|
||||
|
||||
return NULL;
|
||||
}
|
||||
|
||||
char* c_utf8_strncpy(char* dest, const char* src, c_size_t char_num) {
|
||||
if (!dest || !src || char_num == 0) return dest;
|
||||
|
||||
c_size_t src_idx = 0;
|
||||
c_size_t dest_idx = 0;
|
||||
c_size_t copied_chars = 0;
|
||||
|
||||
while (src[src_idx] != '\0' && copied_chars < char_num) {
|
||||
c_size_t char_bytes = c_utf8_char_len(src[src_idx]);
|
||||
|
||||
// 批量精确拷贝当前完整字符的 N 个字节
|
||||
memcpy(&dest[dest_idx], &src[src_idx], char_bytes);
|
||||
|
||||
src_idx += char_bytes;
|
||||
dest_idx += char_bytes;
|
||||
copied_chars++;
|
||||
}
|
||||
|
||||
// 兼容 strncpy 标准:如果源串长度小于 char_num,则用 '\0' 填充剩余的空隙
|
||||
// 注意:这里的剩余空隙在实际工程中通常按字节填充更安全
|
||||
dest[dest_idx] = '\0';
|
||||
|
||||
return dest;
|
||||
}
|
||||
|
||||
int c_utf8_strncmp(const char* str1, const char* str2, c_size_t char_num) {
|
||||
if (!str1 || !str2 || char_num == 0) return 0;
|
||||
|
||||
c_size_t idx1 = 0;
|
||||
c_size_t idx2 = 0;
|
||||
c_size_t compared_chars = 0;
|
||||
|
||||
while (compared_chars < char_num) {
|
||||
// 任意一端到达末尾
|
||||
if (str1[idx1] == '\0' || str2[idx2] == '\0') {
|
||||
return (int)((unsigned char)str1[idx1] - (unsigned char)str2[idx2]);
|
||||
}
|
||||
|
||||
c_size_t len1 = c_utf8_char_len(str1[idx1]);
|
||||
c_size_t len2 = c_utf8_char_len(str2[idx2]);
|
||||
|
||||
// 如果单字长字节不相等,直接根据当前字符进行排序比较
|
||||
if (len1 != len2) {
|
||||
return (int)((unsigned char)str1[idx1] - (unsigned char)str2[idx2]);
|
||||
}
|
||||
|
||||
// 长度相同时,直接比较当前单个 UTF-8 字符的内容
|
||||
int res = memcmp(&str1[idx1], &str2[idx2], len1);
|
||||
if (res != 0) {
|
||||
return res;
|
||||
}
|
||||
|
||||
idx1 += len1;
|
||||
idx2 += len2;
|
||||
compared_chars++;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------------------------------------------------------ */
|
||||
/* */
|
||||
|
||||
char* c_utf8_tolower(char* str) {
|
||||
if (!str) return NULL;
|
||||
|
||||
c_size_t i = 0;
|
||||
while (str[i] != '\0') {
|
||||
unsigned char b1 = (unsigned char)str[i];
|
||||
c_size_t len = c_utf8_char_len(str[i]);
|
||||
|
||||
// Case A: Standard Single-byte ASCII Case Folding
|
||||
if (len == 1) {
|
||||
if (b1 >= 'A' && b1 <= 'Z') {
|
||||
str[i] = (char)(b1 + 32);
|
||||
}
|
||||
}
|
||||
// Case B: Double-byte UTF-8 Case Folding (e.g., Cyrillic / Greek scripts)
|
||||
else if (len == 2) {
|
||||
unsigned char b2 = (unsigned char)str[i + 1];
|
||||
|
||||
// Cyrillic script transformations (Capital letters: 0xD0 0x80 to 0xD0 0xBF)
|
||||
if (b1 == 0xD0) {
|
||||
if (b2 >= 0x90 && b2 <= 0xAF) {
|
||||
// Shift to lowercase variant range located inside 0xD0 / 0xD1 blocks
|
||||
str[i + 1] = (char)(b2 + 0x20);
|
||||
} else if (b2 >= 0xB0 && b2 <= 0xBF) {
|
||||
str[i] = (char)0xD1;
|
||||
str[i + 1] = (char)(b2 - 0x20);
|
||||
}
|
||||
}
|
||||
// Greek script transformations (Capital letters: 0xCE 0x91 to 0xCE 0xAB)
|
||||
else if (b1 == 0xCE) {
|
||||
if (b2 >= 0x91 && b2 <= 0xAB && b2 != 0xA2) { // 0xA2 is a special variant
|
||||
// Shift down to lowercase variant range block inside 0xCE / 0xCF
|
||||
if (b2 <= 0x9F) {
|
||||
str[i + 1] = (char)(b2 + 0x20);
|
||||
} else {
|
||||
str[i] = (char)0xCF;
|
||||
str[i + 1] = (char)(b2 - 0x20);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Multi-byte Chinese ideographs (3 bytes) and Emojis (4 bytes) lack casing concepts; step past them
|
||||
i += len;
|
||||
}
|
||||
|
||||
return str;
|
||||
}
|
||||
|
||||
|
||||
char* c_utf8_toupper(char* str) {
|
||||
if (!str) return NULL;
|
||||
|
||||
c_size_t i = 0;
|
||||
while (str[i] != '\0') {
|
||||
unsigned char b1 = (unsigned char)str[i];
|
||||
c_size_t len = c_utf8_char_len(str[i]);
|
||||
|
||||
// Case A: Standard Single-byte ASCII Case Folding
|
||||
if (len == 1) {
|
||||
if (b1 >= 'a' && b1 <= 'z') {
|
||||
str[i] = (char)(b1 - 32);
|
||||
}
|
||||
}
|
||||
// Case B: Double-byte UTF-8 Case Folding (e.g., Cyrillic / Greek scripts)
|
||||
else if (len == 2) {
|
||||
unsigned char b2 = (unsigned char)str[i + 1];
|
||||
|
||||
// Cyrillic script transformations (Lowercase letters: 0xD0 0xB0 to 0xD0 0xBF, and 0xD1 0x80 to 0xD1 0x8F)
|
||||
if (b1 == 0xD0) {
|
||||
if (b2 >= 0xB0 && b2 <= 0xBF) {
|
||||
// Shift up to uppercase variant range located inside the 0xD0 block
|
||||
str[i + 1] = (char)(b2 - 0x20);
|
||||
}
|
||||
} else if (b1 == 0xD1) {
|
||||
if (b2 >= 0x80 && b2 <= 0x8F) {
|
||||
// Convert leading byte from 0xD1 back to 0xD0 and realign low byte
|
||||
str[i] = (char)0xD0;
|
||||
str[i + 1] = (char)(b2 + 0x20);
|
||||
}
|
||||
}
|
||||
// Greek script transformations (Lowercase letters: 0xCE 0xB1 to 0xCE 0xBF, and 0xCF 0x80 to 0xCF 0x8B)
|
||||
else if (b1 == 0xCE) {
|
||||
if (b2 >= 0xB1 && b2 <= 0xBF) {
|
||||
// Shift down to uppercase variant range block inside 0xCE
|
||||
str[i + 1] = (char)(b2 - 0x20);
|
||||
}
|
||||
} else if (b1 == 0xCF) {
|
||||
if (b2 >= 0x80 && b2 <= 0x8B) {
|
||||
// Convert leading byte from 0xCF back to 0xCE and realign low byte
|
||||
str[i] = (char)0xCE;
|
||||
str[i + 1] = (char)(b2 + 0x20);
|
||||
}
|
||||
}
|
||||
}
|
||||
// 3-byte characters (Chinese Ideographs) and 4-byte characters (Emojis) lack casing concepts; jump past them safely
|
||||
i += len;
|
||||
}
|
||||
|
||||
return str;
|
||||
}
|
||||
|
||||
|
||||
char* c_utf8_strcat(char* dest, const char* src) {
|
||||
if (!dest || !src) return dest;
|
||||
|
||||
// Locate the termination boundary point of the original destination array
|
||||
c_size_t dest_idx = 0;
|
||||
while (dest[dest_idx] != '\0') {
|
||||
dest_idx++;
|
||||
}
|
||||
|
||||
// Continuously append source bytes until hitting the terminator character
|
||||
c_size_t src_idx = 0;
|
||||
while (src[src_idx] != '\0') {
|
||||
dest[dest_idx++] = src[src_idx++];
|
||||
}
|
||||
|
||||
// Force secure terminal character sealing
|
||||
dest[dest_idx] = '\0';
|
||||
|
||||
return dest;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------------------------------------------------------ */
|
||||
/* */
|
||||
|
||||
const char* c_utf8_strstr(const char* haystack, const char* needle) {
|
||||
if (!haystack || !needle) return NULL;
|
||||
|
||||
// An empty needle matches the beginning of the haystack per standard strstr specification
|
||||
if (needle[0] == '\0') {
|
||||
return haystack;
|
||||
}
|
||||
|
||||
c_size_t h_idx = 0;
|
||||
while (haystack[h_idx] != '\0') {
|
||||
c_size_t n_idx = 0;
|
||||
c_size_t current_match_idx = h_idx;
|
||||
|
||||
// Perform byte-by-byte substring evaluation from the current boundary anchor
|
||||
while (haystack[current_match_idx] != '\0' && needle[n_idx] != '\0' &&
|
||||
haystack[current_match_idx] == needle[n_idx]) {
|
||||
current_match_idx++;
|
||||
n_idx++;
|
||||
}
|
||||
|
||||
// If we successfully traversed the entire needle string, a match is found
|
||||
if (needle[n_idx] == '\0') {
|
||||
return &haystack[h_idx];
|
||||
}
|
||||
|
||||
// Advance to the next valid UTF-8 character point in the haystack
|
||||
h_idx += c_utf8_char_len(haystack[h_idx]);
|
||||
}
|
||||
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------------------------------------------------------ */
|
||||
/* */
|
||||
#include <string.h>
|
||||
|
||||
const char* c_utf8_strrchr(const char* str, const char* utf8_char) {
|
||||
if (!str || !utf8_char || utf8_char[0] == '\0') return NULL;
|
||||
|
||||
c_size_t target_bytes = c_utf8_char_len(utf8_char[0]);
|
||||
c_size_t i = 0;
|
||||
const char* last_match = NULL;
|
||||
|
||||
while (str[i] != '\0') {
|
||||
c_size_t curr_bytes = c_utf8_char_len(str[i]);
|
||||
|
||||
// Continuous linear check, updating our tracker to keep the furthest matched offset
|
||||
if (curr_bytes == target_bytes) {
|
||||
if (memcmp(&str[i], utf8_char, target_bytes) == 0) {
|
||||
last_match = &str[i];
|
||||
}
|
||||
}
|
||||
i += curr_bytes; // Jump forward explicitly across whole multi-byte characters
|
||||
}
|
||||
|
||||
return last_match;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Internal helper to verify if a specific UTF-8 character pointer matches any delimiter in the list.
|
||||
*/
|
||||
C_STATIC_FORCE_INLINE
|
||||
c_bool_t c_utf8_is_delim(const char* current_char, const char* delims, c_size_t* delim_len) {
|
||||
c_size_t d_idx = 0;
|
||||
c_size_t c_len = c_utf8_char_len(*current_char);
|
||||
|
||||
while (delims[d_idx] != '\0') {
|
||||
c_size_t d_len = c_utf8_char_len(delims[d_idx]);
|
||||
if (c_len == d_len && memcmp(current_char, &delims[d_idx], c_len) == 0) {
|
||||
*delim_len = d_len;
|
||||
return C_TRUE;
|
||||
}
|
||||
d_idx += d_len;
|
||||
}
|
||||
return C_FALSE;
|
||||
}
|
||||
|
||||
char* c_utf8_strtok(char* str, const char* delims, char** saveptr) {
|
||||
if (!delims || !saveptr) return NULL;
|
||||
|
||||
// Use our saved pointer context if str is passed as NULL
|
||||
char* token_cursor = (str != NULL) ? str : *saveptr;
|
||||
if (!token_cursor || *token_cursor == '\0') {
|
||||
return NULL;
|
||||
}
|
||||
|
||||
// Step 1: Skip over any leading delimiter sequences to locate the token start
|
||||
c_size_t skip_len = 0;
|
||||
while (*token_cursor != '\0' && c_utf8_is_delim(token_cursor, delims, &skip_len)) {
|
||||
token_cursor += skip_len;
|
||||
}
|
||||
|
||||
// If we hit the absolute end of the input string while skipping delims, no tokens exist
|
||||
if (*token_cursor == '\0') {
|
||||
*saveptr = token_cursor;
|
||||
return NULL;
|
||||
}
|
||||
|
||||
char* token_start = token_cursor;
|
||||
|
||||
// Step 2: Track forward to locate the terminal delimiter boundary of this token
|
||||
while (*token_cursor != '\0') {
|
||||
c_size_t next_char_len = c_utf8_char_len(*token_cursor);
|
||||
c_size_t match_delim_len = 0;
|
||||
|
||||
if (c_utf8_is_delim(token_cursor, delims, &match_delim_len)) {
|
||||
// Found a boundary delimiter! Overwrite its leading byte with a null terminator
|
||||
*token_cursor = '\0';
|
||||
// Save state context tracking pointing immediately past the clipped delimiter
|
||||
*saveptr = token_cursor + match_delim_len;
|
||||
return token_start;
|
||||
}
|
||||
token_cursor += next_char_len;
|
||||
}
|
||||
|
||||
// If we reached the end of the text string naturally, ensure the saveptr updates to string termination
|
||||
*saveptr = token_cursor;
|
||||
return token_start;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* @brief Internal helper to return the uppercase variant value of a single ASCII or 2-byte UTF-8 character.
|
||||
* Returns the original trailing byte structure if no case mapping applies.
|
||||
*/
|
||||
C_STATIC_FORCE_INLINE
|
||||
void c_utf8_fold_char(const char* src, size_t len, unsigned char* out_b1, unsigned char* out_b2) {
|
||||
*out_b1 = (unsigned char)src[0];
|
||||
*out_b2 = (len > 1) ? (unsigned char)src[1] : 0;
|
||||
|
||||
// Single-byte ASCII case folding
|
||||
if (len == 1) {
|
||||
if (*out_b1 >= 'a' && *out_b1 <= 'z') {
|
||||
*out_b1 -= 32;
|
||||
}
|
||||
}
|
||||
// Double-byte UTF-8 case folding (Cyrillic & Greek scripts)
|
||||
else if (len == 2) {
|
||||
// Cyrillic script: 0xD0 0xB0...0xBF to 0xD0 0x90...0x9F; 0xD1 0x80...0x8F to 0xD0 0xA0...0xAF
|
||||
if (*out_b1 == 0xD0) {
|
||||
if (*out_b2 >= 0xB0 && *out_b2 <= 0xBF) {
|
||||
*out_b2 -= 0x20;
|
||||
}
|
||||
} else if (*out_b1 == 0xD1) {
|
||||
if (*out_b2 >= 0x80 && *out_b2 <= 0x8F) {
|
||||
*out_b1 = 0xD0;
|
||||
*out_b2 += 0x20;
|
||||
}
|
||||
}
|
||||
// Greek script: 0xCE 0xB1...0xBF to 0xCE 0x91...0x9F; 0xCF 0x80...0x8B to 0xCE 0xA0...0xAB
|
||||
else if (*out_b1 == 0xCE) {
|
||||
if (*out_b2 >= 0xB1 && *out_b2 <= 0xBF) {
|
||||
*out_b2 -= 0x20;
|
||||
}
|
||||
} else if (*out_b1 == 0xCF) {
|
||||
if (*out_b2 >= 0x80 && *out_b2 <= 0x8B) {
|
||||
*out_b1 = 0xCE;
|
||||
*out_b2 += 0x20;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
int c_utf8_strncasecmp(const char* str1, const char* str2, c_size_t char_num) {
|
||||
if (!str1 || !str2 || char_num == 0) return 0;
|
||||
|
||||
c_size_t idx1 = 0;
|
||||
c_size_t idx2 = 0;
|
||||
c_size_t compared_chars = 0;
|
||||
|
||||
while (compared_chars < char_num) {
|
||||
// Handle termination boundaries gracefully
|
||||
if (str1[idx1] == '\0' || str2[idx2] == '\0') {
|
||||
return (int)((unsigned char)str1[idx1] - (unsigned char)str2[idx2]);
|
||||
}
|
||||
|
||||
c_size_t len1 = c_utf8_char_len(str1[idx1]);
|
||||
c_size_t len2 = c_utf8_char_len(str2[idx2]);
|
||||
|
||||
// Fold characters to uppercase form for comparison
|
||||
unsigned char f1_b1, f1_b2;
|
||||
unsigned char f2_b1, f2_b2;
|
||||
c_utf8_fold_char(&str1[idx1], len1, &f1_b1, &f1_b2);
|
||||
c_utf8_fold_char(&str2[idx2], len2, &f2_b1, &f2_b2);
|
||||
|
||||
// Compare first bytes or script widths
|
||||
if (f1_b1 != f2_b1) {
|
||||
return (int)f1_b1 - (int)f2_b1;
|
||||
}
|
||||
|
||||
// Compare second bytes (relevant for 2-byte sequences)
|
||||
if (f1_b2 != f2_b2) {
|
||||
return (int)f1_b2 - (int)f2_b2;
|
||||
}
|
||||
|
||||
// For 3-byte (Chinese) or 4-byte characters, fallback to raw memory comparison if lead bytes matched
|
||||
if (len1 > 2) {
|
||||
if (len1 != len2) {
|
||||
return (int)len1 - (int)len2;
|
||||
}
|
||||
int raw_res = memcmp(&str1[idx1], &str2[idx2], len1);
|
||||
if (raw_res != 0) {
|
||||
return raw_res;
|
||||
}
|
||||
}
|
||||
|
||||
// Advance iteration offsets
|
||||
idx1 += len1;
|
||||
idx2 += len2;
|
||||
compared_chars++;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------------------------------------------------------ */
|
||||
/* */
|
||||
|
||||
c_err_t c_utf8_to_unicode(const char* str, c_ucs4_t* out_codepoint, c_size_t* out_bytes_consumed) {
|
||||
if (!str || *str == '\0' || !out_codepoint || !out_bytes_consumed) {
|
||||
return C_ERR_PARAM;
|
||||
}
|
||||
|
||||
unsigned char b1 = (unsigned char)str[0];
|
||||
c_size_t len = c_utf8_char_len(str[0]);
|
||||
c_ucs4_t cp = 0;
|
||||
|
||||
// Case 1: 1-Byte ASCII (0xxxxxxx)
|
||||
if (len == 1) {
|
||||
if (b1 >= 0x80) return C_ERR_PARAM; // Guard against malformed lead bytes
|
||||
cp = b1;
|
||||
}
|
||||
// Case 2: 2-Byte Sequence (110xxxxx 10xxxxxx)
|
||||
else if (len == 2) {
|
||||
unsigned char b2 = (unsigned char)str[1];
|
||||
if ((b2 & 0xC0) != 0x80) return C_ERR_PARAM; // Validate continuation byte
|
||||
|
||||
cp = ((b1 & 0x1F) << 6) | (b2 & 0x3F);
|
||||
if (cp < 0x80) return C_ERR_PARAM; // Overlong encoding defense
|
||||
}
|
||||
// Case 3: 3-Byte Sequence (1110xxxx 10xxxxxx 10xxxxxx)
|
||||
else if (len == 3) {
|
||||
unsigned char b2 = (unsigned char)str[1];
|
||||
unsigned char b3 = (unsigned char)str[2];
|
||||
if ((b2 & 0xC0) != 0x80 || (b3 & 0xC0) != 0x80) return C_ERR_PARAM;
|
||||
|
||||
cp = ((b1 & 0x0F) << 12) | ((b2 & 0x3F) << 6) | (b3 & 0x3F);
|
||||
if (cp < 0x0800) return C_ERR_PARAM; // Overlong encoding defense
|
||||
if (cp >= 0xD800 && cp <= 0xDFFF) return C_ERR_PARAM; // Surrogate pairs rejection
|
||||
}
|
||||
// Case 4: 4-Byte Sequence (11110xxx 10xxxxxx 10xxxxxx 10xxxxxx)
|
||||
else if (len == 4) {
|
||||
unsigned char b2 = (unsigned char)str[1];
|
||||
unsigned char b3 = (unsigned char)str[2];
|
||||
unsigned char b4 = (unsigned char)str[3];
|
||||
if ((b2 & 0xC0) != 0x80 || (b3 & 0xC0) != 0x80 || (b4 & 0xC0) != 0x80) return C_ERR_PARAM;
|
||||
|
||||
cp = ((b1 & 0x07) << 18) | ((b2 & 0x3F) << 12) | ((b3 & 0x3F) << 6) | (b4 & 0x3F);
|
||||
if (cp < 0x010000) return C_ERR_PARAM; // Overlong encoding defense
|
||||
}
|
||||
else {
|
||||
return C_ERR_PARAM;
|
||||
}
|
||||
|
||||
// Limit validation (Unicode max standard is U+10FFFF)
|
||||
if (cp > 0x10FFFF) {
|
||||
return C_ERR_PARAM;
|
||||
}
|
||||
|
||||
*out_codepoint = cp;
|
||||
*out_bytes_consumed = len;
|
||||
return C_ERR_OK;
|
||||
}
|
||||
|
||||
|
||||
c_err_t c_utf8_from_unicode(c_ucs4_t codepoint, char* dest_buffer, c_size_t* out_bytes_written) {
|
||||
if (!dest_buffer || !out_bytes_written) {
|
||||
return C_ERR_PARAM;
|
||||
}
|
||||
|
||||
// Reject out-of-range codepoints or UTF-16 surrogate pairs (reserved for UTF-16 only)
|
||||
if (codepoint > 0x10FFFF || (codepoint >= 0xD800 && codepoint <= 0xDFFF)) {
|
||||
return C_ERR_PARAM;
|
||||
}
|
||||
|
||||
// Case 1: Standard ASCII range (U+0000 to U+007F) -> Requires 1 byte
|
||||
if (codepoint <= 0x7F) {
|
||||
dest_buffer[0] = (char)codepoint;
|
||||
*out_bytes_written = 1;
|
||||
}
|
||||
// Case 2: U+0080 to U+07FF -> Requires 2 bytes
|
||||
else if (codepoint <= 0x7FF) {
|
||||
dest_buffer[0] = (char)(0xC0 | ((codepoint >> 6) & 0x1F));
|
||||
dest_buffer[1] = (char)(0x80 | (codepoint & 0x3F));
|
||||
*out_bytes_written = 2;
|
||||
}
|
||||
// Case 3: U+0800 to U+FFFF -> Requires 3 bytes (Handles most Chinese characters)
|
||||
else if (codepoint <= 0xFFFF) {
|
||||
dest_buffer[0] = (char)(0xE0 | ((codepoint >> 12) & 0x0F));
|
||||
dest_buffer[1] = (char)(0x80 | ((codepoint >> 6) & 0x3F));
|
||||
dest_buffer[2] = (char)(0x80 | (codepoint & 0x3F));
|
||||
*out_bytes_written = 3;
|
||||
}
|
||||
// Case 4: U+10000 to U+10FFFF -> Requires 4 bytes (Handles Emojis and ancient scripts)
|
||||
else {
|
||||
dest_buffer[0] = (char)(0xF0 | ((codepoint >> 18) & 0x07));
|
||||
dest_buffer[1] = (char)(0x80 | ((codepoint >> 12) & 0x3F));
|
||||
dest_buffer[2] = (char)(0x80 | ((codepoint >> 6) & 0x3F));
|
||||
dest_buffer[3] = (char)(0x80 | (codepoint & 0x3F));
|
||||
*out_bytes_written = 4;
|
||||
}
|
||||
|
||||
// Securely seal the local buffer layout array with a trailing null terminator
|
||||
dest_buffer[*out_bytes_written] = '\0';
|
||||
|
||||
return C_ERR_OK;
|
||||
}
|
||||
|
||||
|
||||
c_err_t c_utf8_to_unicode_array(const char* str, c_ucs4_t* dest_array, c_size_t array_capacity, c_size_t* out_chars_written) {
|
||||
if (!str || !dest_array || !out_chars_written) {
|
||||
return C_ERR_PARAM;
|
||||
}
|
||||
|
||||
c_size_t src_idx = 0;
|
||||
c_size_t chars_count = 0;
|
||||
|
||||
while (str[src_idx] != '\0') {
|
||||
// Enforce array capacity threshold constraints
|
||||
if (chars_count >= array_capacity) {
|
||||
*out_chars_written = chars_count;
|
||||
return C_ERR_PARAM; // Destination array is too small to fit the remaining string
|
||||
}
|
||||
|
||||
c_ucs4_t cp = 0;
|
||||
c_size_t bytes_consumed = 0;
|
||||
|
||||
// Decode the single character point via your core decoding function
|
||||
c_err_t err = c_utf8_to_unicode(&str[src_idx], &cp, &bytes_consumed);
|
||||
if (err != C_ERR_OK) {
|
||||
*out_chars_written = chars_count;
|
||||
return err; // Propagate the malformed stream error up
|
||||
}
|
||||
|
||||
dest_array[chars_count++] = cp;
|
||||
src_idx += bytes_consumed;
|
||||
}
|
||||
|
||||
*out_chars_written = chars_count;
|
||||
return C_ERR_OK;
|
||||
}
|
||||
|
||||
c_err_t c_utf8_from_unicode_array(const c_ucs4_t* src_array, c_size_t src_array_len, char* dest_buffer, c_size_t dest_capacity, c_size_t* out_bytes_written) {
|
||||
if (!src_array || !dest_buffer || !out_bytes_written) {
|
||||
return C_ERR_PARAM;
|
||||
}
|
||||
|
||||
c_size_t dest_idx = 0;
|
||||
|
||||
for (c_size_t i = 0; i < src_array_len; i++) {
|
||||
char temp_char_buf[5]; // Temporary standalone slot buffer
|
||||
c_size_t bytes_written = 0;
|
||||
|
||||
// Encode single code point state back to byte wrappers
|
||||
c_err_t err = c_utf8_from_unicode(src_array[i], temp_char_buf, &bytes_written);
|
||||
if (err != C_ERR_OK) {
|
||||
*out_bytes_written = dest_idx;
|
||||
return err;
|
||||
}
|
||||
|
||||
// Verify if destination capacity bounds can hold the new character block (+1 for terminal null)
|
||||
if (dest_idx + bytes_written + 1 > dest_capacity) {
|
||||
*out_bytes_written = dest_idx;
|
||||
dest_buffer[dest_idx] = '\0'; // Gracefully terminate the current chunk before failing
|
||||
return C_ERR_PARAM;
|
||||
}
|
||||
|
||||
// Copy raw encoded data bytes into our tracking stream layout
|
||||
memcpy(dest_buffer + dest_idx, temp_char_buf, bytes_written);
|
||||
dest_idx += bytes_written;
|
||||
}
|
||||
|
||||
// Force strict trailing character termination closure
|
||||
dest_buffer[dest_idx] = '\0';
|
||||
*out_bytes_written = dest_idx;
|
||||
|
||||
return C_ERR_OK;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------------------------------------------------------ */
|
||||
/* */
|
||||
|
||||
|
||||
c_err_t c_utf8_to_utf16(const char* str, c_uint16_t* dest_array, c_size_t array_capacity, c_size_t* out_units_written) {
|
||||
if (!str || !dest_array || !out_units_written) {
|
||||
return C_ERR_PARAM;
|
||||
}
|
||||
|
||||
c_size_t src_idx = 0;
|
||||
c_size_t units_count = 0;
|
||||
|
||||
while (str[src_idx] != '\0') {
|
||||
c_ucs4_t cp = 0;
|
||||
c_size_t bytes_consumed = 0;
|
||||
|
||||
// 1. Decode the UTF-8 sequence into a Unicode codepoint
|
||||
c_err_t err = c_utf8_to_unicode(&str[src_idx], &cp, &bytes_consumed);
|
||||
if (err != C_ERR_OK) {
|
||||
*out_units_written = units_count;
|
||||
return err;
|
||||
}
|
||||
|
||||
// 2. Encode the codepoint into UTF-16
|
||||
if (cp <= 0xFFFF) {
|
||||
// BMP Range: Requires exactly one 16-bit code unit
|
||||
if (units_count + 1 >= array_capacity) {
|
||||
*out_units_written = units_count;
|
||||
return C_ERR_PARAM; // Out of bounds
|
||||
}
|
||||
dest_array[units_count++] = (c_uint16_t)cp;
|
||||
} else {
|
||||
// Supplementary Planes (Astral): Requires a surrogate pair (two 16-bit units)
|
||||
if (units_count + 2 >= array_capacity) {
|
||||
*out_units_written = units_count;
|
||||
return C_ERR_PARAM; // Out of bounds
|
||||
}
|
||||
cp -= 0x10000;
|
||||
dest_array[units_count++] = (c_uint16_t)(0xD800 | ((cp >> 10) & 0x3FF)); // High Surrogate
|
||||
dest_array[units_count++] = (c_uint16_t)(0xDC00 | (cp & 0x3FF)); // Low Surrogate
|
||||
}
|
||||
|
||||
src_idx += bytes_consumed;
|
||||
}
|
||||
|
||||
// Append the trailing null terminator to make it a valid UTF-16 string
|
||||
if (units_count < array_capacity) {
|
||||
dest_array[units_count] = 0;
|
||||
} else {
|
||||
*out_units_written = units_count;
|
||||
return C_ERR_PARAM;
|
||||
}
|
||||
|
||||
*out_units_written = units_count;
|
||||
return C_ERR_OK;
|
||||
}
|
||||
|
||||
c_err_t c_utf8_from_utf16(const c_uint16_t* src_array, c_size_t src_array_len, char* dest_buffer, c_size_t dest_capacity, c_size_t* out_bytes_written) {
|
||||
if (!src_array || !dest_buffer || !out_bytes_written) {
|
||||
return C_ERR_PARAM;
|
||||
}
|
||||
|
||||
c_size_t src_idx = 0;
|
||||
c_size_t dest_idx = 0;
|
||||
|
||||
while (src_idx < src_array_len) {
|
||||
c_ucs4_t cp = 0;
|
||||
c_uint16_t u1 = src_array[src_idx++];
|
||||
|
||||
// 1. Decode UTF-16 to Unicode codepoint
|
||||
if (u1 >= 0xD800 && u1 <= 0xDBFF) {
|
||||
// High surrogate detected, look ahead for the matching low surrogate
|
||||
if (src_idx >= src_array_len) {
|
||||
*out_bytes_written = dest_idx;
|
||||
return C_ERR_PARAM; // Truncated/Malformed surrogate pair
|
||||
}
|
||||
c_uint16_t u2 = src_array[src_idx++];
|
||||
if (u2 < 0xDC00 || u2 > 0xDFFF) {
|
||||
*out_bytes_written = dest_idx;
|
||||
return C_ERR_PARAM; // Missing or invalid low surrogate
|
||||
}
|
||||
cp = (((u1 & 0x3FF) << 10) | (u2 & 0x3FF)) + 0x10000;
|
||||
} else if (u1 >= 0xDC00 && u1 <= 0xDFFF) {
|
||||
// Isolated low surrogate is invalid in a lead position
|
||||
*out_bytes_written = dest_idx;
|
||||
return C_ERR_PARAM;
|
||||
} else {
|
||||
// Normal BMP character
|
||||
cp = u1;
|
||||
}
|
||||
|
||||
// 2. Encode the Unicode codepoint back into the destination UTF-8 buffer
|
||||
char temp_buf[5];
|
||||
c_size_t bytes_written = 0;
|
||||
c_err_t err = c_utf8_from_unicode(cp, temp_buf, &bytes_written);
|
||||
if (err != C_ERR_OK) {
|
||||
*out_bytes_written = dest_idx;
|
||||
return err;
|
||||
}
|
||||
|
||||
// Verify if destination capacity bounds can hold the new block (+1 for terminal null)
|
||||
if (dest_idx + bytes_written + 1 > dest_capacity) {
|
||||
*out_bytes_written = dest_idx;
|
||||
dest_buffer[dest_idx] = '\0';
|
||||
return C_ERR_PARAM; // Overflow protection
|
||||
}
|
||||
|
||||
memcpy(dest_buffer + dest_idx, temp_buf, bytes_written);
|
||||
dest_idx += bytes_written;
|
||||
}
|
||||
|
||||
// Force strict trailing character termination closure
|
||||
dest_buffer[dest_idx] = '\0';
|
||||
*out_bytes_written = dest_idx;
|
||||
|
||||
return C_ERR_OK;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------------------------------------------------------ */
|
||||
/* */
|
||||
|
||||
c_err_t c_utf16_swap_endian(c_uint16_t* utf16_array, c_size_t length) {
|
||||
if (!utf16_array) {
|
||||
return C_ERR_PARAM;
|
||||
}
|
||||
|
||||
for (c_size_t i = 0; i < length; i++) {
|
||||
c_uint16_t value = utf16_array[i];
|
||||
// Bitwise swap: (value >> 8) extracts the high byte, (value << 8) extracts the low byte
|
||||
utf16_array[i] = (c_uint16_t)(((value & 0x00FF) << 8) | ((value & 0xFF00) >> 8));
|
||||
}
|
||||
|
||||
return C_ERR_OK;
|
||||
}
|
||||
|
||||
|
||||
c_err_t c_utf16_to_unicode(const c_uint16_t* src_units, c_size_t src_capacity, c_ucs4_t* out_codepoint, c_size_t* out_units_read) {
|
||||
if (!src_units || src_capacity == 0 || !out_codepoint || !out_units_read) {
|
||||
return C_ERR_PARAM;
|
||||
}
|
||||
|
||||
c_uint16_t u1 = src_units[0];
|
||||
|
||||
// Case 1: High Surrogate Point Detection (0xD800 to 0xDBFF)
|
||||
if (u1 >= 0xD800 && u1 <= 0xDBFF) {
|
||||
if (src_capacity < 2) {
|
||||
return C_ERR_PARAM; // Truncated sequence: expected a matching low surrogate
|
||||
}
|
||||
|
||||
c_uint16_t u2 = src_units[1];
|
||||
if (u2 < 0xDC00 || u2 > 0xDFFF) {
|
||||
return C_ERR_PARAM; // Malformed sequence: missing a valid trailing low surrogate
|
||||
}
|
||||
|
||||
// Reconstruct Astral Plane Codepoint via formula: ((High - 0xD800) << 10) + (Low - 0xDC00) + 0x10000
|
||||
*out_codepoint = (c_ucs4_t)((((u1 & 0x3FF) << 10) | (u2 & 0x3FF)) + 0x10000);
|
||||
*out_units_read = 2;
|
||||
}
|
||||
// Case 2: Isolated Low Surrogate Error Check (0xDC00 to 0xDFFF)
|
||||
else if (u1 >= 0xDC00 && u1 <= 0xDFFF) {
|
||||
return C_ERR_PARAM; // Isolated low surrogate is mathematically invalid in a lead position
|
||||
}
|
||||
// Case 3: Standard BMP Range Character
|
||||
else {
|
||||
*out_codepoint = (c_ucs4_t)u1;
|
||||
*out_units_read = 1;
|
||||
}
|
||||
|
||||
return C_ERR_OK;
|
||||
}
|
||||
|
||||
c_err_t c_utf16_from_unicode(c_ucs4_t codepoint, c_uint16_t* dest_units, c_size_t dest_capacity, c_size_t* out_units_written) {
|
||||
if (!dest_units || dest_capacity == 0 || !out_units_written) {
|
||||
return C_ERR_PARAM;
|
||||
}
|
||||
|
||||
// Limit Validation: Reject invalid Astral values or illegal UTF-16 surrogate codepoint blocks
|
||||
if (codepoint > 0x10FFFF || (codepoint >= 0xD800 && codepoint <= 0xDFFF)) {
|
||||
return C_ERR_PARAM;
|
||||
}
|
||||
|
||||
// Case 1: BMP Range (U+0000 to U+FFFF) -> Requires 1 code unit
|
||||
if (codepoint <= 0xFFFF) {
|
||||
dest_units[0] = (c_uint16_t)codepoint;
|
||||
*out_units_written = 1;
|
||||
}
|
||||
// Case 2: Supplementary Planes (U+10000 to U+10FFFF) -> Requires 2 code units (Surrogate Pair)
|
||||
else {
|
||||
if (dest_capacity < 2) {
|
||||
return C_ERR_PARAM; // Insufficient buffer capacity
|
||||
}
|
||||
|
||||
c_ucs4_t adjusted = codepoint - 0x10000;
|
||||
dest_units[0] = (c_uint16_t)(0xD800 | ((adjusted >> 10) & 0x3FF)); // High Surrogate
|
||||
dest_units[1] = (c_uint16_t)(0xDC00 | (adjusted & 0x3FF)); // Low Surrogate
|
||||
*out_units_written = 2;
|
||||
}
|
||||
|
||||
return C_ERR_OK;
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user