ascii-chat 0.11.33
Video chat in your terminal
Loading...
Searching...
No Matches
neon/color.c
Go to the documentation of this file.
1
7#if SIMD_SUPPORT_NEON
8#include <stdio.h>
9#include <stdlib.h>
10#include <stdint.h>
11#include <string.h>
12#include <stdarg.h>
13#include <time.h>
14#include <assert.h>
15#include <ascii-chat/atomic.h>
16#include <math.h>
17
18#include <arm_neon.h>
19
20#include <ascii-chat/common.h>
31#include <ascii-chat/log/log.h>
32
33//=============================================================================
34// Optimized NEON Color Converter (based on ChatGPT reference)
35//=============================================================================
36
37// Unified optimized NEON converter (foreground/background + 256-color/truecolor)
38char *render_ascii_color_neon(const image_t *image, bool use_background, bool use_256color, const char *ascii_chars) {
39 if (!image || !image->pixels) {
40 return NULL;
41 }
42
43 const int width = image->w;
44 const int height = image->h;
45
46 if (width <= 0 || height <= 0) {
47 char *empty;
48 empty = SAFE_MALLOC(1, char *);
49 empty[0] = '\0';
50 return empty;
51 }
52
53 outbuf_t ob = {0};
54 // Estimate buffer size based on mode
55 size_t bytes_per_pixel = use_256color ? 6u : 8u; // 256-color shorter than truecolor
56
57 // Calculate buffer size with overflow checking
58 size_t height_times_width;
59 if (checked_size_mul((size_t)height, (size_t)width, &height_times_width) != ASCIICHAT_OK) {
60 log_error("Buffer size overflow: height * width overflow");
61 return NULL;
62 }
63
64 size_t pixel_data_size;
65 if (checked_size_mul(height_times_width, bytes_per_pixel, &pixel_data_size) != ASCIICHAT_OK) {
66 log_error("Buffer size overflow: (height * width) * bytes_per_pixel overflow");
67 return NULL;
68 }
69
70 size_t height_times_16;
71 if (checked_size_mul((size_t)height, 16u, &height_times_16) != ASCIICHAT_OK) {
72 log_error("Buffer size overflow: height * 16 overflow");
73 return NULL;
74 }
75
76 size_t temp;
77 if (checked_size_add(pixel_data_size, height_times_16, &temp) != ASCIICHAT_OK) {
78 log_error("Buffer size overflow: pixel_data + height*16 overflow");
79 return NULL;
80 }
81
82 if (checked_size_add(temp, 64u, &ob.cap) != ASCIICHAT_OK) {
83 log_error("Buffer size overflow: total capacity overflow");
84 return NULL;
85 }
86
87 ob.buf = SAFE_MALLOC(ob.cap ? ob.cap : 1, char *);
88 if (!ob.buf)
89 return NULL;
90
91 START_TIMER("neon_utf8_cache");
92 // Get cached UTF-8 character mappings (like monochrome function does)
93 utf8_palette_cache_t *utf8_cache = get_utf8_palette_cache(ascii_chars);
94 if (!utf8_cache) {
95 log_error("Failed to get UTF-8 palette cache for NEON color");
96 return NULL;
97 }
98 STOP_TIMER_AND_LOG_EVERY(dev, 3 * NS_PER_SEC_INT, 3 * NS_PER_MS_INT, "neon_utf8_cache",
99 "NEON_UTF8_CACHE: Complete (%.2f ms)");
100
101 START_TIMER("neon_lookup_tables");
102 // Build NEON lookup table inline (faster than caching - 30ns rebuild vs 50ns lookup)
103 uint8x16x4_t tbl, char_lut, length_lut, char_byte0_lut, char_byte1_lut, char_byte2_lut, char_byte3_lut;
104 build_neon_lookup_tables(utf8_cache, &tbl, &char_lut, &length_lut, &char_byte0_lut, &char_byte1_lut, &char_byte2_lut,
105 &char_byte3_lut);
106 STOP_TIMER_AND_LOG_EVERY(dev, 3 * NS_PER_SEC_INT, 3 * NS_PER_MS_INT, "neon_lookup_tables",
107 "NEON_LOOKUP_TABLES: Complete (%.2f ms)");
108
109 // Suppress unused variable warnings for color mode
110 (void)char_lut;
111 (void)length_lut;
112 (void)char_byte0_lut;
113 (void)char_byte1_lut;
114 (void)char_byte2_lut;
115 (void)char_byte3_lut;
116
117 START_TIMER("neon_main_loop");
118 uint64_t loop_start_ns = time_get_ns();
119
120 // Track which code path is taken
121 int chunks_256color = 0, chunks_truecolor = 0;
122
123 // PRE-INITIALIZE: Call init once before loop
124 init_neon_decimal_table();
125
126 // Process rows
127 for (int y = 0; y < height; y++) {
128 // Track current color state
129 int curR = -1, curG = -1, curB = -1;
130 int cur_color_idx = -1;
131
132 const rgb_pixel_t *row = &((const rgb_pixel_t *)image->pixels)[y * width];
133 int x = 0;
134
135 // Process 16-pixel chunks with NEON
136 while (x + 16 <= width) {
137 // Load 16 pixels: R,G,B interleaved
138 const uint8_t *p = (const uint8_t *)(row + x);
139 uint8x16x3_t pix = vld3q_u8(p); // 48 bytes
140
141 // Vector luminance: Y ≈ (77*R + 150*G + 29*B + 128) >> 8
142 uint16x8_t ylo = vmull_u8(vget_low_u8(pix.val[0]), vdup_n_u8(LUMA_RED));
143 ylo = vmlal_u8(ylo, vget_low_u8(pix.val[1]), vdup_n_u8(LUMA_GREEN));
144 ylo = vmlal_u8(ylo, vget_low_u8(pix.val[2]), vdup_n_u8(LUMA_BLUE));
145 ylo = vaddq_u16(ylo, vdupq_n_u16(LUMA_THRESHOLD));
146 ylo = vshrq_n_u16(ylo, 8);
147
148 uint16x8_t yhi = vmull_u8(vget_high_u8(pix.val[0]), vdup_n_u8(LUMA_RED));
149 yhi = vmlal_u8(yhi, vget_high_u8(pix.val[1]), vdup_n_u8(LUMA_GREEN));
150 yhi = vmlal_u8(yhi, vget_high_u8(pix.val[2]), vdup_n_u8(LUMA_BLUE));
151 yhi = vaddq_u16(yhi, vdupq_n_u16(LUMA_THRESHOLD));
152 yhi = vshrq_n_u16(yhi, 8);
153
154 uint8x16_t y8 = vcombine_u8(vmovn_u16(ylo), vmovn_u16(yhi));
155 uint8x16_t idx = vshrq_n_u8(y8, 2); // 0..63
156
157 // FAST: Use vqtbl4q_u8 to get character indices from the ramp
158 uint8x16_t char_indices = vqtbl4q_u8(tbl, idx);
159
160 if (use_256color) {
161 chunks_256color++;
162 // 256-color mode: VECTORIZED color quantization
163 uint8_t char_idx_buf[16], color_indices[16];
164 vst1q_u8(char_idx_buf, char_indices); // Character indices from SIMD lookup
165
166 // VECTORIZED: Use existing optimized 256-color quantization
167 uint8x16_t color_indices_vec = palette256_index_dithered_neon(pix.val[0], pix.val[1], pix.val[2], x);
168 vst1q_u8(color_indices, color_indices_vec);
169
170 // Emit with RLE on (UTF-8 character, color) runs using SIMD-derived indices
171 for (int i = 0; i < 16;) {
172 const uint8_t char_idx = char_idx_buf[i]; // From vqtbl4q_u8 lookup
173 const utf8_char_t *char_info = &utf8_cache->cache64[char_idx];
174 const uint8_t color_idx = color_indices[i];
175
176 // NEON-optimized RLE detection
177 const uint32_t run =
178 (uint32_t)find_rle_run_length_neon(char_idx_buf, color_indices, i, 16, char_idx, color_idx);
179
180 if (color_idx != cur_color_idx) {
181 if (use_background) {
182 emit_set_256_color_bg(&ob, color_idx);
183 } else {
184 emit_set_256_color_fg(&ob, color_idx);
185 }
186 cur_color_idx = color_idx;
187 }
188
189 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
190 if (rep_is_profitable(run)) {
191 emit_rep(&ob, run - 1);
192 } else {
193 for (uint32_t k = 1; k < run; k++) {
194 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
195 }
196 }
197 i += run;
198 }
199 } else {
200 chunks_truecolor++;
201 // VECTORIZED: Truecolor mode with full SIMD pipeline (no scalar spillover)
202 char temp_buffer[16 * 50]; // Temporary buffer for 16 ANSI sequences (up to 50 bytes each)
203 size_t vectorized_length =
204 neon_assemble_truecolor_sequences_true_simd(char_indices, pix.val[0], pix.val[1], pix.val[2], utf8_cache,
205 temp_buffer, sizeof(temp_buffer), use_background);
206
207 // Write vectorized output to main buffer
208 ob_write(&ob, temp_buffer, vectorized_length);
209 }
210 x += 16;
211 }
212
213 // Scalar tail for remaining pixels
214 for (; x < width;) {
215 const rgb_pixel_t *p = &row[x];
216 uint32_t R = p->r, G = p->g, B = p->b;
217 uint8_t Y = (uint8_t)((LUMA_RED * R + LUMA_GREEN * G + LUMA_BLUE * B + LUMA_THRESHOLD) >> 8);
218 uint8_t luma_idx = Y >> 2; // 0-63 index (matches SIMD: cache64 is indexed by luminance bucket)
219 const utf8_char_t *char_info = &utf8_cache->cache64[luma_idx];
220
221 if (use_256color) {
222 // 256-color scalar tail
223 uint8_t color_idx = rgb_to_256color((uint8_t)R, (uint8_t)G, (uint8_t)B);
224
225 int j = x + 1;
226 while (j < width) {
227 const rgb_pixel_t *q = &row[j];
228 uint32_t R2 = q->r, G2 = q->g, B2 = q->b;
229 uint8_t Y2 = (uint8_t)((LUMA_RED * R2 + LUMA_GREEN * G2 + LUMA_BLUE * B2 + LUMA_THRESHOLD) >> 8);
230 uint8_t luma_idx2 = Y2 >> 2;
231 uint8_t color_idx2 = rgb_to_256color((uint8_t)R2, (uint8_t)G2, (uint8_t)B2);
232 if (luma_idx2 != luma_idx || color_idx2 != color_idx)
233 break;
234 j++;
235 }
236 uint32_t run = (uint32_t)(j - x);
237
238 if (color_idx != cur_color_idx) {
239 if (use_background) {
240 emit_set_256_color_bg(&ob, color_idx);
241 } else {
242 emit_set_256_color_fg(&ob, color_idx);
243 }
244 cur_color_idx = color_idx;
245 }
246
247 // Emit UTF-8 character from cache
248 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
249 if (rep_is_profitable(run)) {
250 emit_rep(&ob, run - 1);
251 } else {
252 for (uint32_t k = 1; k < run; k++) {
253 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
254 }
255 }
256 x = j;
257 } else {
258 // Truecolor scalar tail with UTF-8 characters using cached lookups
259 int j = x + 1;
260 while (j < width) {
261 const rgb_pixel_t *q = &row[j];
262 uint32_t R2 = q->r, G2 = q->g, B2 = q->b;
263 uint8_t Y2 = (uint8_t)((LUMA_RED * R2 + LUMA_GREEN * G2 + LUMA_BLUE * B2 + LUMA_THRESHOLD) >> 8);
264 uint8_t luma_idx2 = Y2 >> 2; // Compare luminance buckets (matches SIMD)
265 if (luma_idx2 != luma_idx || R2 != R || G2 != G || B2 != B)
266 break;
267 j++;
268 }
269 uint32_t run = (uint32_t)(j - x);
270
271 if ((int)R != curR || (int)G != curG || (int)B != curB) {
272 if (use_background) {
274 } else {
276 }
277 curR = (int)R;
278 curG = (int)G;
279 curB = (int)B;
280 }
281
282 // Emit UTF-8 character from cache
283 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
284 if (rep_is_profitable(run)) {
285 emit_rep(&ob, run - 1);
286 } else {
287 for (uint32_t k = 1; k < run; k++) {
288 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
289 }
290 }
291 x = j;
292 }
293 }
294
295 // End row: reset SGR, add newline (except for last row)
296 emit_reset(&ob);
297 if (y < height - 1) {
298 ob_putc(&ob, '\n');
299 }
300 }
301
302 uint64_t loop_end_ns = time_get_ns();
303 uint64_t loop_time_ms = (loop_end_ns - loop_start_ns) / NS_PER_MS_INT;
304 log_dev("NEON_MAIN_LOOP_ACTUAL: %llu ms for %d rows, %d width", loop_time_ms, height, width);
305
306 // Log chunks per mode
307 log_dev(
308 "NEON_MAIN_LOOP processed %d rows x %d width = %d pixels in %llu ms (256color: %d chunks, truecolor: %d chunks)",
309 height, width, height * width, loop_time_ms, chunks_256color, chunks_truecolor);
310
311 STOP_TIMER_AND_LOG_EVERY(dev, 3 * NS_PER_SEC_INT, 5 * NS_PER_MS_INT, "neon_main_loop",
312 "NEON_MAIN_LOOP: Complete (%.2f ms)");
313
314 ob_term(&ob);
315 return ob.buf;
316}
317
318#endif
ANSI escape sequence utilities and fast color code generation.
⚛️ Atomic operations abstraction layer with debug tracking
⚙️ Common definitions, error codes, macros, and types shared throughout the application
uint8_t rgb_to_256color(uint8_t r, uint8_t g, uint8_t b)
Definition ansi.c:360
unsigned int uint32_t
Definition common.h:58
#define SAFE_MALLOC(size, cast)
Definition common.h:264
unsigned long long uint64_t
Definition common.h:59
unsigned char uint8_t
Definition common.h:56
@ ASCIICHAT_OK
Definition error_codes.h:51
#define log_dev(...)
Log a DEV message (most verbose, development only)
Definition log/log.h:534
#define log_error(...)
Log an ERROR message.
Definition log/log.h:587
#define STOP_TIMER_AND_LOG_EVERY(log_level, interval_ns, threshold_ns, timer_name, msg_fmt,...)
Stop a timer and log the result with rate limiting.
Definition time.h:647
uint64_t time_get_ns(void)
Get current monotonic time in nanoseconds.
Definition util/time.c:108
#define NS_PER_SEC_INT
Definition time.h:157
#define START_TIMER(name_fmt,...)
Start a timer with formatted name.
Definition time.h:333
#define NS_PER_MS_INT
Definition time.h:156
#define LUMA_BLUE
Luminance blue coefficient (0.114 * 256 = 29)
void emit_set_256_color_bg(outbuf_t *ob, uint8_t color_idx)
Emit 256-color background ANSI sequence.
#define LUMA_GREEN
Luminance green coefficient (0.587 * 256 = 150)
uint8_t utf8_bytes[4]
utf8_palette_cache_t * get_utf8_palette_cache(const char *ascii_chars)
Get UTF-8 palette cache for a character set.
void emit_set_256_color_fg(outbuf_t *ob, uint8_t color_idx)
Emit 256-color foreground ANSI sequence.
void ob_term(outbuf_t *ob)
Append null terminator to buffer.
#define LUMA_THRESHOLD
Luminance threshold for rounding.
void ob_putc(outbuf_t *ob, char c)
Append a character to buffer.
bool rep_is_profitable(uint32_t runlen)
Check if run-length encoding is profitable.
void emit_set_truecolor_fg(outbuf_t *ob, uint8_t r, uint8_t g, uint8_t b)
Emit truecolor foreground ANSI sequence.
void emit_rep(outbuf_t *ob, uint32_t extra)
Emit run-length encoded sequence.
void ob_write(outbuf_t *ob, const char *s, size_t n)
Append a string to buffer.
void emit_reset(outbuf_t *ob)
Emit ANSI reset sequence.
utf8_char_t cache64[64]
void emit_set_truecolor_bg(outbuf_t *ob, uint8_t r, uint8_t g, uint8_t b)
Emit truecolor background ANSI sequence.
#define LUMA_RED
Luminance red coefficient (0.299 * 256 = 77)
Platform initialization and static synchronization helpers.
Lock-free module lifecycle state machine using pure stdatomic.
📝 Logging API with multiple log levels and terminal output control
🔢 Mathematical Utility Functions
NEON-optimized ASCII rendering functions.
✅ Safe Integer Arithmetic and Overflow Detection
#define R2(v, w, x, y, z, i)
Definition sha1.c:58
Image structure.
int w
Image width in pixels (must be > 0)
int h
Image height in pixels (must be > 0)
rgb_pixel_t * pixels
Pixel data array (width * height RGB pixels, row-major order)
Dynamic output buffer (auto-expanding)
size_t cap
Buffer capacity in bytes (maximum length before reallocation)
char * buf
Buffer pointer (allocated, owned by caller, must be freed)
RGB pixel structure.
uint8_t b
Blue color component (0-255)
uint8_t g
Green color component (0-255)
uint8_t r
Red color component (0-255)
⏱️ High-precision timing utilities using sokol_time.h and uthash
SIMD-optimized ASCII conversion interface.
ARM NEON-accelerated ASCII rendering utilities (declarations)