#include "chinese_text.h" #include #include #include #include static const uint32_t CHINESE_CODEPOINTS[] = { #define CHINESE_GLYPH(codepoint) codepoint, #include "def_chinese_glyph_table.h" #undef CHINESE_GLYPH }; static size_t encode_utf8_codepoint(uint32_t codepoint, char output[4]) { if (codepoint <= 0x7F) { output[0] = (char)codepoint; return 1; } if (codepoint <= 0x7FF) { output[0] = (char)(0xC0 | (codepoint >> 6)); output[1] = (char)(0x80 | (codepoint & 0x3F)); return 2; } output[0] = (char)(0xE0 | (codepoint >> 12)); output[1] = (char)(0x80 | ((codepoint >> 6) & 0x3F)); output[2] = (char)(0x80 | (codepoint & 0x3F)); return 3; } static uint32_t decode_like_tonc(const unsigned char encoded[2]) { return ((uint32_t)(encoded[0] & 0x1F) << 6) | (encoded[1] & 0x3F); } static void test_ascii_is_unchanged(void) { char output[32]; size_t output_len = chinese_text_encode("A: +20", output, sizeof(output)); assert(output_len == strlen("A: +20")); assert(strcmp(output, "A: +20") == 0); } static void test_supported_chinese_uses_extended_glyphs(void) { char output[8]; size_t output_len = chinese_text_encode("小丑牌", output, sizeof(output)); assert(output_len == 6); assert((unsigned char)output[0] == 0xC2); assert((unsigned char)output[1] == 0x80); assert((unsigned char)output[2] == 0xC2); assert((unsigned char)output[3] == 0x81); assert((unsigned char)output[4] == 0xC2); assert((unsigned char)output[5] == 0x82); assert(output[6] == '\0'); } static void test_unsupported_unicode_becomes_question_mark(void) { char output[8]; size_t output_len = chinese_text_encode("龘", output, sizeof(output)); assert(output_len == 1); assert(strcmp(output, "?") == 0); } static void test_all_custom_glyphs_round_trip_through_tonc_utf8(void) { size_t glyph_count = sizeof(CHINESE_CODEPOINTS) / sizeof(CHINESE_CODEPOINTS[0]); assert(glyph_count <= 128); for (size_t i = 0; i < glyph_count; i++) { char input[4] = {'\0'}; size_t input_len = encode_utf8_codepoint(CHINESE_CODEPOINTS[i], input); input[input_len] = '\0'; char output[4] = {'\0'}; size_t output_len = chinese_text_encode(input, output, sizeof(output)); uint32_t expected_glyph_codepoint = 128 + i; assert(output_len == 2); assert(((unsigned char)output[0] & 0xE0) == 0xC0); assert(((unsigned char)output[1] & 0xC0) == 0x80); assert(decode_like_tonc((const unsigned char*)output) == expected_glyph_codepoint); assert(output[2] == '\0'); } } static void test_small_buffer_does_not_split_utf8_sequence(void) { char output[4] = {'x', 'x', 'x', 'x'}; size_t output_len = chinese_text_encode("小丑牌", output, sizeof(output)); assert(output_len == 2); assert((unsigned char)output[0] == 0xC2); assert((unsigned char)output[1] == 0x80); assert(output[2] == '\0'); } int main(void) { test_ascii_is_unchanged(); test_supported_chinese_uses_extended_glyphs(); test_unsupported_unicode_becomes_question_mark(); test_all_custom_glyphs_round_trip_through_tonc_utf8(); test_small_buffer_does_not_split_utf8_sequence(); printf("Chinese text tests passed.\n"); return 0; }