ascii-chat 0.11.33
Video chat in your terminal
Loading...
Searching...
No Matches
video/ascii/neon/common.c
Go to the documentation of this file.
1
12#if SIMD_SUPPORT_NEON
13#include <stdio.h>
14#include <stdlib.h>
15#include <stdint.h>
16#include <string.h>
17#include <stdarg.h>
18#include <time.h>
19#include <assert.h>
20#include <ascii-chat/atomic.h>
21#include <math.h>
22
23#include <arm_neon.h>
24
25#include <ascii-chat/common.h>
35#include <ascii-chat/log/log.h>
36
37// NEON table cache removed - performance analysis showed rebuilding (30ns) is faster than lookup (50ns)
38// Tables are now built inline when needed for optimal performance
39
40// Build NEON lookup tables inline (faster than caching - 30ns rebuild vs 50ns lookup)
41static inline void __attribute__((unused))
42build_neon_lookup_tables(utf8_palette_cache_t *utf8_cache, uint8x16x4_t *tbl, uint8x16x4_t *char_lut,
43 uint8x16x4_t *length_lut, uint8x16x4_t *char_byte0_lut, uint8x16x4_t *char_byte1_lut,
44 uint8x16x4_t *char_byte2_lut, uint8x16x4_t *char_byte3_lut) {
45 // Build NEON-specific lookup table with cache64 indices (direct mapping)
46 uint8_t cache64_indices[64];
47 for (int i = 0; i < 64; i++) {
48 cache64_indices[i] = (uint8_t)i; // Direct mapping: luminance bucket -> cache64 index
49 }
50
51 tbl->val[0] = vld1q_u8(&cache64_indices[0]);
52 tbl->val[1] = vld1q_u8(&cache64_indices[16]);
53 tbl->val[2] = vld1q_u8(&cache64_indices[32]);
54 tbl->val[3] = vld1q_u8(&cache64_indices[48]);
55
56 // Build vectorized UTF-8 lookup tables for length-aware compaction
57 uint8_t ascii_chars_lut[64]; // For ASCII fast path
58 uint8_t char_lengths[64]; // Character byte lengths
59 uint8_t char_byte0[64]; // First byte of each character
60 uint8_t char_byte1[64]; // Second byte of each character
61 uint8_t char_byte2[64]; // Third byte of each character
62 uint8_t char_byte3[64]; // Fourth byte of each character
63
64 for (int i = 0; i < 64; i++) {
65 const utf8_char_t *char_info = &utf8_cache->cache64[i];
66
67 // ASCII fast path table
68 ascii_chars_lut[i] = char_info->utf8_bytes[0];
69
70 // Length-aware compaction tables
71 char_lengths[i] = char_info->byte_len;
72 char_byte0[i] = char_info->utf8_bytes[0];
73 char_byte1[i] = char_info->byte_len > 1 ? char_info->utf8_bytes[1] : 0;
74 char_byte2[i] = char_info->byte_len > 2 ? char_info->utf8_bytes[2] : 0;
75 char_byte3[i] = char_info->byte_len > 3 ? char_info->utf8_bytes[3] : 0;
76 }
77
78 // Load all lookup tables into NEON registers
79 char_lut->val[0] = vld1q_u8(&ascii_chars_lut[0]);
80 char_lut->val[1] = vld1q_u8(&ascii_chars_lut[16]);
81 char_lut->val[2] = vld1q_u8(&ascii_chars_lut[32]);
82 char_lut->val[3] = vld1q_u8(&ascii_chars_lut[48]);
83
84 length_lut->val[0] = vld1q_u8(&char_lengths[0]);
85 length_lut->val[1] = vld1q_u8(&char_lengths[16]);
86 length_lut->val[2] = vld1q_u8(&char_lengths[32]);
87 length_lut->val[3] = vld1q_u8(&char_lengths[48]);
88
89 char_byte0_lut->val[0] = vld1q_u8(&char_byte0[0]);
90 char_byte0_lut->val[1] = vld1q_u8(&char_byte0[16]);
91 char_byte0_lut->val[2] = vld1q_u8(&char_byte0[32]);
92 char_byte0_lut->val[3] = vld1q_u8(&char_byte0[48]);
93
94 char_byte1_lut->val[0] = vld1q_u8(&char_byte1[0]);
95 char_byte1_lut->val[1] = vld1q_u8(&char_byte1[16]);
96 char_byte1_lut->val[2] = vld1q_u8(&char_byte1[32]);
97 char_byte1_lut->val[3] = vld1q_u8(&char_byte1[48]);
98
99 char_byte2_lut->val[0] = vld1q_u8(&char_byte2[0]);
100 char_byte2_lut->val[1] = vld1q_u8(&char_byte2[16]);
101 char_byte2_lut->val[2] = vld1q_u8(&char_byte2[32]);
102 char_byte2_lut->val[3] = vld1q_u8(&char_byte2[48]);
103
104 char_byte3_lut->val[0] = vld1q_u8(&char_byte3[0]);
105 char_byte3_lut->val[1] = vld1q_u8(&char_byte3[16]);
106 char_byte3_lut->val[2] = vld1q_u8(&char_byte3[32]);
107 char_byte3_lut->val[3] = vld1q_u8(&char_byte3[48]);
108}
109
110// NEON-optimized RLE detection: find run length for char+color pairs
111static inline int __attribute__((unused)) find_rle_run_length_neon(const uint8_t *char_buf, const uint8_t *color_buf,
112 int start_pos, int max_len, uint8_t target_char,
113 uint8_t target_color) {
114 int run_length = 1; // At least the starting position
115
116 // Use NEON to check multiple elements at once when possible
117 int remaining = max_len - start_pos - 1;
118 if (remaining <= 0)
119 return 1;
120
121 const uint8_t *char_ptr = &char_buf[start_pos + 1];
122 const uint8_t *color_ptr = &color_buf[start_pos + 1];
123
124 // Process in chunks of 16 for full NEON utilization
125 while (remaining >= 16) {
126 uint8x16_t chars = vld1q_u8(char_ptr);
127 uint8x16_t colors = vld1q_u8(color_ptr);
128
129 uint8x16_t char_match = vceqq_u8(chars, vdupq_n_u8(target_char));
130 uint8x16_t color_match = vceqq_u8(colors, vdupq_n_u8(target_color));
131 uint8x16_t both_match = vandq_u8(char_match, color_match);
132
133 // Use NEON min/max to find first mismatch efficiently
134 // If all elements match, min will be 0xFF, otherwise it will be 0x00
135 uint8_t min_match = vminvq_u8(both_match);
136
137 if (min_match == 0xFF) {
138 // All 16 elements match
139 run_length += 16;
140 char_ptr += 16;
141 color_ptr += 16;
142 remaining -= 16;
143 } else {
144 // Find first mismatch position using bit scan
145 uint64_t mask_lo = vgetq_lane_u64(vreinterpretq_u64_u8(both_match), 0);
146 uint64_t mask_hi = vgetq_lane_u64(vreinterpretq_u64_u8(both_match), 1);
147
148 int matches_found = 0;
149 // Check low 8 bytes first
150 for (int i = 0; i < 8; i++) {
151 if ((mask_lo >> (i * 8)) & 0xFF) {
152 matches_found++;
153 } else {
154 break;
155 }
156 }
157
158 // If all low 8 matched, check high 8 bytes
159 if (matches_found == 8) {
160 for (int i = 0; i < 8; i++) {
161 if ((mask_hi >> (i * 8)) & 0xFF) {
162 matches_found++;
163 } else {
164 break;
165 }
166 }
167 }
168
169 run_length += matches_found;
170 break; // Found mismatch, stop
171 }
172 }
173
174 // Handle remaining elements with scalar loop
175 while (remaining > 0 && *char_ptr == target_char && *color_ptr == target_color) {
176 run_length++;
177 char_ptr++;
178 color_ptr++;
179 remaining--;
180 }
181
182 return run_length;
183}
184
185// NEON helper: Check if all characters have same length
186static inline bool __attribute__((unused)) all_same_length_neon(uint8x16_t lengths, uint8_t *out_length) {
187 uint8_t first_len = vgetq_lane_u8(lengths, 0);
188 uint8x16_t first_len_vec = vdupq_n_u8(first_len);
189 uint8x16_t all_same = vceqq_u8(lengths, first_len_vec);
190
191 uint64x2_t all_same_64 = vreinterpretq_u64_u8(all_same);
192 uint64_t combined = vgetq_lane_u64(all_same_64, 0) & vgetq_lane_u64(all_same_64, 1);
193
194 if (combined == 0xFFFFFFFFFFFFFFFF) {
195 *out_length = first_len;
196 return true;
197 }
198 return false;
199}
200
201// ============================================================================
202// Vectorized Decimal Lookup Functions for NEON Color Performance
203// ============================================================================
204
205// NEON TBL lookup tables for decimal conversion (256 entries each)
206// Format: each entry has length byte + up to 3 decimal chars (4 bytes per entry)
207static uint8_t neon_decimal_table_data[256 * 4]; // 1024 bytes: [len][d1][d2][d3] per entry
208// Lifecycle for thread-safe one-time initialization (replaces C11 call_once)
209static lifecycle_t g_neon_table_lc = LIFECYCLE_INIT;
210
211// Private initialization function (called exactly once via lifecycle)
212static void do_init_neon_decimal_table(void) {
213 // Initialize g_dec3_cache first
215 init_dec3();
216 }
217
218 // Convert dec3_t cache to NEON TBL format: [len][d1][d2][d3] per 4-byte entry
219 for (int i = 0; i < 256; i++) {
220 const dec3_t *dec = &g_dec3_cache.dec3_table[i];
221 uint8_t *entry = &neon_decimal_table_data[i * 4];
222 entry[0] = dec->len; // Length (1-3)
223 entry[1] = (dec->len >= 1) ? dec->s[0] : '0'; // First digit
224 entry[2] = (dec->len >= 2) ? dec->s[1] : '0'; // Second digit
225 entry[3] = (dec->len >= 3) ? dec->s[2] : '0'; // Third digit
226 }
227}
228
229// Initialize NEON TBL decimal lookup table (called once at startup)
230// Thread-safe with lifecycle API ensuring exactly-once execution
231void init_neon_decimal_table(void) {
232 if (!lifecycle_init(&g_neon_table_lc, "neon_decimal")) {
233 return; // Already initialized
234 }
235 do_init_neon_decimal_table();
236}
237
238// TODO: Implement true NEON vectorized ANSI sequence generation using TBL + compaction
239// Following the monochrome pattern: pad sequences to uniform width, then compact null bytes
240// For now, keep the existing scalar approach to avoid breaking the build
241
242// True NEON vectorized ANSI truecolor sequence assembly - no scalar loops!
243static inline size_t __attribute__((unused))
244neon_assemble_truecolor_sequences_true_simd(uint8x16_t char_indices, uint8x16_t r_vals, uint8x16_t g_vals,
245 uint8x16_t b_vals, utf8_palette_cache_t *utf8_cache, char *output_buffer,
246 size_t buffer_capacity, bool use_background) {
247 // STREAMLINED IMPLEMENTATION: Focus on the real bottleneck - RGB->decimal conversion
248 // Key insight: ANSI sequences are too variable for effective SIMD, but TBL lookups provide major speedup
249
250 // NOTE: Ensure NEON decimal table is initialized BEFORE calling this function (done in
251 // render_ascii_neon_unified_optimized)
252
253 char *dst = output_buffer;
254
255 // Extract values for optimized scalar processing with SIMD-accelerated lookups
256 uint8_t char_idx_buf[16], r_buf[16], g_buf[16], b_buf[16];
257 vst1q_u8(char_idx_buf, char_indices);
258 vst1q_u8(r_buf, r_vals);
259 vst1q_u8(g_buf, g_vals);
260 vst1q_u8(b_buf, b_vals);
261
262 size_t total_written = 0;
263 const char *prefix = use_background ? "\033[48;2;" : "\033[38;2;";
264 const size_t prefix_len = 7;
265
266 // Optimized scalar loop with NEON TBL acceleration for RGB->decimal conversion
267 // This eliminates the expensive safe_snprintf() calls which were the real bottleneck
268 for (int i = 0; i < 16; i++) {
269 // Use NEON TBL lookups for RGB decimal conversion (major speedup!)
270 const uint8_t *r_entry = &neon_decimal_table_data[r_buf[i] * 4];
271 const uint8_t *g_entry = &neon_decimal_table_data[g_buf[i] * 4];
272 const uint8_t *b_entry = &neon_decimal_table_data[b_buf[i] * 4];
273
274 const uint8_t char_idx = char_idx_buf[i];
275 const utf8_char_t *char_info = &utf8_cache->cache64[char_idx];
276
277 // Calculate total sequence length for buffer safety
278 size_t seq_len = prefix_len + r_entry[0] + 1 + g_entry[0] + 1 + b_entry[0] + 1 + char_info->byte_len;
279 if (total_written >= buffer_capacity - seq_len) {
280 break; // Buffer safety
281 }
282
283 // Optimized assembly using TBL results (no divisions, no snprintf!)
284 memcpy(dst, prefix, prefix_len);
285 dst += prefix_len;
286
287 // RGB components using pre-computed decimal strings
288 memcpy(dst, &r_entry[1], r_entry[0]);
289 dst += r_entry[0];
290 *dst++ = ';';
291
292 memcpy(dst, &g_entry[1], g_entry[0]);
293 dst += g_entry[0];
294 *dst++ = ';';
295
296 memcpy(dst, &b_entry[1], b_entry[0]);
297 dst += b_entry[0];
298 *dst++ = 'm';
299
300 // UTF-8 character from cache
301 memcpy(dst, char_info->utf8_bytes, char_info->byte_len);
302 dst += char_info->byte_len;
303
304 total_written = dst - output_buffer;
305 }
306
307 return total_written;
308}
309
310// Min-heap management removed - no longer needed without NEON table cache
311
312// Eviction logic removed - no longer needed without NEON table cache
313
314// Continue to actual NEON functions (helper functions already defined above)
315
316// Definitions are in ascii_simd.h - just use them
317// REMOVED: #define luminance_palette g_ascii_cache.luminance_palette (causes macro expansion issues)
318
319// SIMD luma and helpers:
320
321// SIMD luminance: Y = (77R + 150G + 29B) >> 8
322static inline uint8x16_t simd_luma_neon(uint8x16_t r, uint8x16_t g, uint8x16_t b) {
323 uint16x8_t rl = vmovl_u8(vget_low_u8(r));
324 uint16x8_t rh = vmovl_u8(vget_high_u8(r));
325 uint16x8_t gl = vmovl_u8(vget_low_u8(g));
326 uint16x8_t gh = vmovl_u8(vget_high_u8(g));
327 uint16x8_t bl = vmovl_u8(vget_low_u8(b));
328 uint16x8_t bh = vmovl_u8(vget_high_u8(b));
329
330 uint32x4_t l0 = vmull_n_u16(vget_low_u16(rl), LUMA_RED);
331 uint32x4_t l1 = vmull_n_u16(vget_high_u16(rl), LUMA_RED);
332 l0 = vmlal_n_u16(l0, vget_low_u16(gl), LUMA_GREEN);
333 l1 = vmlal_n_u16(l1, vget_high_u16(gl), LUMA_GREEN);
334 l0 = vmlal_n_u16(l0, vget_low_u16(bl), LUMA_BLUE);
335 l1 = vmlal_n_u16(l1, vget_high_u16(bl), LUMA_BLUE);
336
337 uint32x4_t h0 = vmull_n_u16(vget_low_u16(rh), LUMA_RED);
338 uint32x4_t h1 = vmull_n_u16(vget_high_u16(rh), LUMA_RED);
339 h0 = vmlal_n_u16(h0, vget_low_u16(gh), LUMA_GREEN);
340 h1 = vmlal_n_u16(h1, vget_high_u16(gh), LUMA_GREEN);
341 h0 = vmlal_n_u16(h0, vget_low_u16(bh), LUMA_BLUE);
342 h1 = vmlal_n_u16(h1, vget_high_u16(bh), LUMA_BLUE);
343
344 uint16x8_t l = vcombine_u16(vrshrn_n_u32(l0, 8), vrshrn_n_u32(l1, 8));
345 uint16x8_t h = vcombine_u16(vrshrn_n_u32(h0, 8), vrshrn_n_u32(h1, 8));
346 return vcombine_u8(vqmovn_u16(l), vqmovn_u16(h));
347}
348
349// ===== SIMD helpers for 256-color quantization =====
350
351// Approximate quantize 0..255 -> 0..5 : q ≈ round(x*5/255) = (x*5 + 128)>>8
352static inline uint8x16_t q6_from_u8(uint8x16_t x) {
353 uint16x8_t xl = vmovl_u8(vget_low_u8(x));
354 uint16x8_t xh = vmovl_u8(vget_high_u8(x));
355 xl = vmlaq_n_u16(vdupq_n_u16(0), xl, 5);
356 xh = vmlaq_n_u16(vdupq_n_u16(0), xh, 5);
357 xl = vaddq_u16(xl, vdupq_n_u16(128));
358 xh = vaddq_u16(xh, vdupq_n_u16(128));
359 xl = vshrq_n_u16(xl, 8);
360 xh = vshrq_n_u16(xh, 8);
361 return vcombine_u8(vqmovn_u16(xl), vqmovn_u16(xh)); // 0..5
362}
363
364// Make 256-color index (cube vs gray). threshold: max-min < thr ⇒ gray
365#ifndef CUBE_GRAY_THRESHOLD
366#define CUBE_GRAY_THRESHOLD 10
367#endif
368
369// Apply ordered dithering to reduce color variations (creates longer runs)
370static inline uint8x16_t apply_ordered_dither(uint8x16_t color, int pixel_offset, uint8_t dither_strength) {
371 // Bayer 4x4 dithering matrix (classic ordered dithering pattern)
372 static const uint8_t bayer4x4[16] = {0, 8, 2, 10, 12, 4, 14, 6, 3, 11, 1, 9, 15, 7, 13, 5};
373
374 // Load dithering matrix into NEON register
375 const uint8x16_t dither_matrix = vld1q_u8(bayer4x4);
376
377 // Create pixel position indices for 16 consecutive pixels
378 uint8_t pos_indices[16];
379 for (int i = 0; i < 16; i++) {
380 pos_indices[i] = (pixel_offset + i) & 15; // Wrap to 4x4 matrix (0-15)
381 }
382 const uint8x16_t position_vec = vld1q_u8(pos_indices);
383
384 // Lookup dither values for each pixel position using table lookup
385 uint8x16_t dither_values = vqtbl1q_u8(dither_matrix, position_vec);
386
387 // Scale dither values by strength (0-255 range)
388 // dither_strength controls how much dithering to apply
389 uint16x8_t dither_lo = vmulq_n_u16(vmovl_u8(vget_low_u8(dither_values)), dither_strength);
390 uint16x8_t dither_hi = vmulq_n_u16(vmovl_u8(vget_high_u8(dither_values)), dither_strength);
391 dither_lo = vshrq_n_u16(dither_lo, 4); // Scale down (/16)
392 dither_hi = vshrq_n_u16(dither_hi, 4);
393 uint8x16_t scaled_dither = vcombine_u8(vqmovn_u16(dither_lo), vqmovn_u16(dither_hi));
394
395 // Apply dithering with saturation to prevent overflow
396 return vqaddq_u8(color, scaled_dither);
397}
398
399uint8x16_t palette256_index_dithered_neon(uint8x16_t r, uint8x16_t g, uint8x16_t b, int pixel_offset) {
400 // Dithering disabled in speed mode (no-op)
401 r = apply_ordered_dither(r, pixel_offset, 0);
402 g = apply_ordered_dither(g, pixel_offset + 1, 0);
403 b = apply_ordered_dither(b, pixel_offset + 2, 0);
404
405 // cube index
406 uint8x16_t R6 = q6_from_u8(r);
407 uint8x16_t G6 = q6_from_u8(g);
408 uint8x16_t B6 = q6_from_u8(b);
409
410 // idx_cube = 16 + R6*36 + G6*6 + B6 (do in 16-bit to avoid overflow)
411 uint16x8_t R6l = vmovl_u8(vget_low_u8(R6));
412 uint16x8_t R6h = vmovl_u8(vget_high_u8(R6));
413 uint16x8_t G6l = vmovl_u8(vget_low_u8(G6));
414 uint16x8_t G6h = vmovl_u8(vget_high_u8(G6));
415 uint16x8_t B6l = vmovl_u8(vget_low_u8(B6));
416 uint16x8_t B6h = vmovl_u8(vget_high_u8(B6));
417
418 uint16x8_t idxl = vmlaq_n_u16(vmulq_n_u16(R6l, 36), G6l, 6);
419 uint16x8_t idxh = vmlaq_n_u16(vmulq_n_u16(R6h, 36), G6h, 6);
420 idxl = vaddq_u16(idxl, B6l);
421 idxh = vaddq_u16(idxh, B6h);
422 idxl = vaddq_u16(idxl, vdupq_n_u16(16));
423 idxh = vaddq_u16(idxh, vdupq_n_u16(16));
424
425 // gray decision: max-min < thr ?
426 uint8x16_t maxrg = vmaxq_u8(r, g);
427 uint8x16_t minrg = vminq_u8(r, g);
428 uint8x16_t maxrgb = vmaxq_u8(maxrg, b);
429 uint8x16_t minrgb = vminq_u8(minrg, b);
430 uint8x16_t diff = vsubq_u8(maxrgb, minrgb);
431 uint8x16_t thr = vdupq_n_u8((uint8_t)CUBE_GRAY_THRESHOLD);
432 uint8x16_t is_gray = vcltq_u8(diff, thr);
433
434 // gray idx = 232 + round(Y*23/255)
435 uint8x16_t Y = simd_luma_neon(r, g, b);
436 // q23 ≈ round(Y*23/255) = (Y*23 + 128)>>8
437 uint16x8_t Yl = vmovl_u8(vget_low_u8(Y));
438 uint16x8_t Yh = vmovl_u8(vget_high_u8(Y));
439 Yl = vmlaq_n_u16(vdupq_n_u16(0), Yl, 23);
440 Yh = vmlaq_n_u16(vdupq_n_u16(0), Yh, 23);
441 Yl = vaddq_u16(Yl, vdupq_n_u16(128));
442 Yh = vaddq_u16(Yh, vdupq_n_u16(128));
443 Yl = vshrq_n_u16(Yl, 8);
444 Yh = vshrq_n_u16(Yh, 8);
445 uint16x8_t gidxl = vaddq_u16(Yl, vdupq_n_u16(232));
446 uint16x8_t gidxh = vaddq_u16(Yh, vdupq_n_u16(232));
447
448 // select gray or cube per lane
449 uint8x16_t idx_cube = vcombine_u8(vqmovn_u16(idxl), vqmovn_u16(idxh));
450 uint8x16_t idx_gray = vcombine_u8(vqmovn_u16(gidxl), vqmovn_u16(gidxh));
451 return vbslq_u8(is_gray, idx_gray, idx_cube);
452}
453
460void image_flip_horizontal_neon(image_t *image) {
461 if (!image || !image->pixels || image->w < 2) {
462 return;
463 }
464
465 // Process each row - swap pixels from both ends using NEON for faster loads/stores
466 for (int y = 0; y < image->h; y++) {
467 rgb_pixel_t *row = &image->pixels[y * image->w];
468 int width = image->w;
469
470 // NEON-accelerated swapping: process 4 pixels at a time using uint32 loads
471 // Each RGB pixel is 3 bytes, so 4 pixels = 12 bytes that can be loaded as 3x u32
472 int left_pix = 0;
473 int right_pix = width - 1;
474
475 // Fast path: swap 4-pixel groups using uint32 operations
476 while (left_pix + 3 < right_pix - 3) {
477 // Load left 4 pixels (12 bytes) as 3 uint32 values using NEON
478 uint32_t *left_ptr = (uint32_t *)&row[left_pix];
479 uint32_t *right_ptr = (uint32_t *)&row[right_pix - 3];
480
481 uint32x2_t left_0 = vld1_u32(left_ptr); // first 8 bytes
482 uint32_t left_1 = left_ptr[2]; // last 4 bytes
483
484 uint32x2_t right_0 = vld1_u32(right_ptr); // first 8 bytes
485 uint32_t right_1 = right_ptr[2]; // last 4 bytes
486
487 // Store swapped using NEON
488 vst1_u32(right_ptr, left_0);
489 right_ptr[2] = left_1;
490 vst1_u32(left_ptr, right_0);
491 left_ptr[2] = right_1;
492
493 left_pix += 4;
494 right_pix -= 4;
495 }
496
497 // Scalar cleanup for remaining pixels
498 while (left_pix < right_pix) {
499 rgb_pixel_t temp = row[left_pix];
500 row[left_pix] = row[right_pix];
501 row[right_pix] = temp;
502 left_pix++;
503 right_pix--;
504 }
505 }
506}
507
508#endif // SIMD_SUPPORT_NEON
ANSI escape sequence utilities and fast color code generation.
⚛️ Atomic operations abstraction layer with debug tracking
⚙️ Common definitions, error codes, macros, and types shared throughout the application
unsigned int uint32_t
Definition common.h:58
unsigned long long uint64_t
Definition common.h:59
unsigned char uint8_t
Definition common.h:56
global_dec3_cache_t g_dec3_cache
Global decimal cache instance.
#define LUMA_BLUE
Luminance blue coefficient (0.114 * 256 = 29)
#define LUMA_GREEN
Luminance green coefficient (0.587 * 256 = 150)
uint8_t utf8_bytes[4]
void init_dec3(void)
Initialize decimal-to-ASCII cache (0-255 → "0" to "255")
utf8_char_t cache64[64]
#define LUMA_RED
Luminance red coefficient (0.299 * 256 = 77)
Platform initialization and static synchronization helpers.
bool lifecycle_init(lifecycle_t *lc, const char *name)
Definition lifecycle.c:26
Lock-free module lifecycle state machine using pure stdatomic.
#define LIFECYCLE_INIT
Static initializer for module-global lifecycle variables (no sync primitive)
Definition lifecycle.h:68
📝 Logging API with multiple log levels and terminal output control
🔢 Mathematical Utility Functions
NEON-optimized ASCII rendering functions.
✅ Safe Integer Arithmetic and Overflow Detection
#define CUBE_GRAY_THRESHOLD
Definition sgr.c:39
Network quality metrics for a single participant.
Definition metrics.h:33
Decimal conversion cache structure (1-3 digits)
Image structure.
int w
Image width in pixels (must be > 0)
int h
Image height in pixels (must be > 0)
rgb_pixel_t * pixels
Pixel data array (width * height RGB pixels, row-major order)
RGB pixel structure.
⏱️ High-precision timing utilities using sokol_time.h and uthash
SIMD-optimized ASCII conversion interface.