ascii-chat 0.11.33
Video chat in your terminal
Loading...
Searching...
No Matches
sve/color.c
Go to the documentation of this file.
1
7#if SIMD_SUPPORT_SVE
8#include <stdio.h>
9#include <stdlib.h>
10#include <string.h>
11#include <stdint.h>
13#include <ascii-chat/common.h>
14#include <ascii-chat/video/ascii/common.h> // For LUMA_RED, LUMA_GREEN, LUMA_BLUE, LUMA_THRESHOLD
15#include <ascii-chat/video/ascii/output_buffer.h> // For outbuf_t, emit_*, ob_*
16
17#include <arm_sve.h>
18
20
21static inline uint8_t rgb_to_256color_sve(uint8_t r, uint8_t g, uint8_t b) {
22 return (uint8_t)(16 + 36 * (r / 51) + 6 * (g / 51) + (b / 51));
23}
24
25// Unified SVE function for all color modes (full implementation like NEON)
26
27char *render_ascii_color_sve(const image_t *image, bool use_background, bool use_256color, const char *ascii_chars) {
28 if (!image || !image->pixels) {
29 return NULL;
30 }
31
32 const int width = image->w;
33 const int height = image->h;
34
35 if (width <= 0 || height <= 0) {
36 char *empty;
37 empty = SAFE_MALLOC(1, char *);
38 empty[0] = '\0';
39 return empty;
40 }
41
42 // Use monochrome optimization for simple case
43 if (!use_background && !use_256color) {
44 return render_ascii_mono_sve(image, ascii_chars);
45 }
46
47 outbuf_t ob = {0};
48 // Estimate buffer size based on mode (copied from NEON)
49 size_t bytes_per_pixel = use_256color ? 6u : 8u; // 256-color shorter than truecolor
50
51 // Calculate buffer size with overflow checking
52 size_t height_times_width;
53 if (checked_size_mul((size_t)height, (size_t)width, &height_times_width) != ASCIICHAT_OK) {
54 log_error("Buffer size overflow: height * width overflow");
55 return NULL;
56 }
57
58 size_t pixel_data_size;
59 if (checked_size_mul(height_times_width, bytes_per_pixel, &pixel_data_size) != ASCIICHAT_OK) {
60 log_error("Buffer size overflow: (height * width) * bytes_per_pixel overflow");
61 return NULL;
62 }
63
64 size_t height_times_16;
65 if (checked_size_mul((size_t)height, 16u, &height_times_16) != ASCIICHAT_OK) {
66 log_error("Buffer size overflow: height * 16 overflow");
67 return NULL;
68 }
69
70 size_t temp;
71 if (checked_size_add(pixel_data_size, height_times_16, &temp) != ASCIICHAT_OK) {
72 log_error("Buffer size overflow: pixel_data + height*16 overflow");
73 return NULL;
74 }
75
76 if (checked_size_add(temp, 64u, &ob.cap) != ASCIICHAT_OK) {
77 log_error("Buffer size overflow: total capacity overflow");
78 return NULL;
79 }
80
81 ob.buf = SAFE_MALLOC(ob.cap ? ob.cap : 1, char *);
82 if (!ob.buf)
83 return NULL;
84
85 // Get cached UTF-8 character mappings for color rendering
86 utf8_palette_cache_t *utf8_cache = get_utf8_palette_cache(ascii_chars);
87 if (!utf8_cache) {
88 log_error("Failed to get UTF-8 palette cache for SVE color");
89 return NULL;
90 }
91
92 // Track current color state (copied from NEON)
93 int curR = -1, curG = -1, curB = -1;
94 int cur_color_idx = -1;
95
96 for (int y = 0; y < height; y++) {
97 const rgb_pixel_t *row = &((const rgb_pixel_t *)image->pixels)[y * width];
98 int x = 0;
99
100 // Process with SVE scalable vectors (adapts to hardware vector length)
101 while (x < width) {
102 svbool_t pg_active = svwhilelt_b8_s32(x, width);
103 int vec_len = svcntb_pat(SV_ALL) / 3; // Vector length in RGB pixels
104 int remaining = width - x;
105 int process_count = (remaining < vec_len) ? remaining : vec_len;
106
107 // Manual deinterleave RGB components (SVE limitation vs NEON's vld3)
108 uint8_t r_array[64], g_array[64], b_array[64]; // Max SVE vector size
109 for (int j = 0; j < process_count; j++) {
110 if (x + j < width) {
111 r_array[j] = row[x + j].r;
112 g_array[j] = row[x + j].g;
113 b_array[j] = row[x + j].b;
114 }
115 }
116
117 // Load into SVE vectors
118 svuint8_t r_vec = svld1_u8(pg_active, r_array);
119 svuint8_t g_vec = svld1_u8(pg_active, g_array);
120 svuint8_t b_vec = svld1_u8(pg_active, b_array);
121
122 // Convert to 16-bit for arithmetic
123 svuint16_t r_16 = svunpklo_u16(r_vec);
124 svuint16_t g_16 = svunpklo_u16(g_vec);
125 svuint16_t b_16 = svunpklo_u16(b_vec);
126
127 // Calculate luminance: (77*R + 150*G + 29*B + 128) >> 8
128 svuint16_t luma = svmul_n_u16_x(svptrue_b16(), r_16, LUMA_RED);
129 luma = svmla_n_u16_x(svptrue_b16(), luma, g_16, LUMA_GREEN);
130 luma = svmla_n_u16_x(svptrue_b16(), luma, b_16, LUMA_BLUE);
131 luma = svadd_n_u16_x(svptrue_b16(), luma, LUMA_THRESHOLD);
132 luma = svlsr_n_u16_x(svptrue_b16(), luma, 8);
133
134 // Store u16 luminance values (SVE1 compatible - no SVE2 narrowing intrinsics)
135 // After right-shift by 8, values are already in 0-255 range
136 uint16_t luma_temp[64];
137 svst1_u16(svptrue_b16(), luma_temp, luma);
138
139 // Convert to u8 array for ASCII lookup
140 uint8_t luma_array[64];
141 for (int j = 0; j < process_count; j++) {
142 luma_array[j] = (uint8_t)luma_temp[j];
143 }
144
145 // FAST: Use svtbl_u8 to get character indices from the ramp (SVE advantage)
146 // Convert luminance to 0-63 indices
147 svuint8_t luma_vec = svld1_u8(pg_active, luma_array); // Load luminance values
148 svuint8_t luma_idx_vec = svlsr_n_u8_x(svptrue_b8(), luma_vec, 2); // >> 2 for 0-63
149
150 // Use svtbl_u8 for fast character index lookup (scalable!)
151 svuint8_t char_lut_vec = svld1_u8(svptrue_b8(), utf8_cache->char_index_ramp);
152 svuint8_t char_indices_vec = svtbl_u8(char_lut_vec, luma_idx_vec);
153
154 uint8_t gbuf[64]; // Reuse gbuf name for compatibility
155 svst1_u8(pg_active, gbuf, char_indices_vec);
156
157 if (use_256color) {
158 // 256-color mode processing (copied from NEON logic)
159 uint8_t color_indices[64];
160 for (int i = 0; i < process_count; i++) {
161 color_indices[i] = rgb_to_256color_sve(r_array[i], g_array[i], b_array[i]);
162 }
163
164 // Emit with RLE on (glyph, color) runs (copied from NEON)
165 for (int i = 0; i < process_count;) {
166 const uint8_t char_idx = gbuf[i]; // This is now the character index
167 const utf8_char_t *char_info = &utf8_cache->cache64[char_idx];
168 const uint8_t color_idx = color_indices[i];
169
170 int j = i + 1;
171 while (j < process_count && gbuf[j] == char_idx && color_indices[j] == color_idx) {
172 j++;
173 }
174 const uint32_t run = (uint32_t)(j - i);
175
176 if (color_idx != cur_color_idx) {
177 if (use_background) {
178 emit_set_256_color_bg(&ob, color_idx);
179 } else {
180 emit_set_256_color_fg(&ob, color_idx);
181 }
182 cur_color_idx = color_idx;
183 }
184
185 // Emit UTF-8 character from cache
186 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
187 if (rep_is_profitable(run)) {
188 emit_rep(&ob, run - 1);
189 } else {
190 for (uint32_t k = 1; k < run; k++) {
191 // Emit UTF-8 character from cache
192 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
193 }
194 }
195 i = j;
196 }
197 } else {
198 // Truecolor mode processing (copied from NEON logic)
199 for (int i = 0; i < process_count;) {
200 const uint8_t char_idx = gbuf[i]; // This is now the character index
201 const utf8_char_t *char_info = &utf8_cache->cache64[char_idx];
202 const uint8_t r = r_array[i];
203 const uint8_t g = g_array[i];
204 const uint8_t b = b_array[i];
205
206 int j = i + 1;
207 while (j < process_count && gbuf[j] == char_idx && r_array[j] == r && g_array[j] == g && b_array[j] == b) {
208 j++;
209 }
210 const uint32_t run = (uint32_t)(j - i);
211
212 if (r != curR || g != curG || b != curB) {
213 if (use_background) {
214 emit_set_truecolor_bg(&ob, r, g, b);
215 } else {
216 emit_set_truecolor_fg(&ob, r, g, b);
217 }
218 curR = r;
219 curG = g;
220 curB = b;
221 }
222
223 // Emit UTF-8 character from cache
224 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
225 if (rep_is_profitable(run)) {
226 emit_rep(&ob, run - 1);
227 } else {
228 for (uint32_t k = 1; k < run; k++) {
229 // Emit UTF-8 character from cache
230 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
231 }
232 }
233 i = j;
234 }
235 }
236 x += process_count;
237 }
238
239 // Scalar tail for any remaining pixels (copied from NEON logic)
240 for (; x < width;) {
241 const rgb_pixel_t *p = &row[x];
242 uint32_t R = p->r, G = p->g, B = p->b;
243 uint8_t Y = (uint8_t)((LUMA_RED * R + LUMA_GREEN * G + LUMA_BLUE * B + LUMA_THRESHOLD) >> 8);
244 uint8_t luma_idx = Y >> 2;
245 const utf8_char_t *char_info = &utf8_cache->cache64[luma_idx];
246
247 if (use_256color) {
248 // 256-color scalar tail
249 uint8_t color_idx = rgb_to_256color_sve((uint8_t)R, (uint8_t)G, (uint8_t)B);
250
251 int j = x + 1;
252 while (j < width) {
253 const rgb_pixel_t *q = &row[j];
254 uint32_t R2 = q->r, G2 = q->g, B2 = q->b;
255 uint8_t Y2 = (uint8_t)((LUMA_RED * R2 + LUMA_GREEN * G2 + LUMA_BLUE * B2 + LUMA_THRESHOLD) >> 8);
256 uint8_t color_idx2 = rgb_to_256color_sve((uint8_t)R2, (uint8_t)G2, (uint8_t)B2);
257 if (((Y2 >> 2) != (Y >> 2)) || color_idx2 != color_idx)
258 break;
259 j++;
260 }
261 uint32_t run = (uint32_t)(j - x);
262
263 if (color_idx != cur_color_idx) {
264 if (use_background) {
265 emit_set_256_color_bg(&ob, color_idx);
266 } else {
267 emit_set_256_color_fg(&ob, color_idx);
268 }
269 cur_color_idx = color_idx;
270 }
271
272 // Emit UTF-8 character from cache
273 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
274 if (rep_is_profitable(run)) {
275 emit_rep(&ob, run - 1);
276 } else {
277 for (uint32_t k = 1; k < run; k++) {
278 // Emit UTF-8 character from cache
279 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
280 }
281 }
282 x = j;
283 } else {
284 // Truecolor scalar tail
285 int j = x + 1;
286 while (j < width) {
287 const rgb_pixel_t *q = &row[j];
288 uint32_t R2 = q->r, G2 = q->g, B2 = q->b;
289 uint8_t Y2 = (uint8_t)((77u * R2 + 150u * G2 + 29u * B2 + 128u) >> 8);
290 if (((Y2 >> 2) != (Y >> 2)) || R2 != R || G2 != G || B2 != B)
291 break;
292 j++;
293 }
294 uint32_t run = (uint32_t)(j - x);
295
296 if ((int)R != curR || (int)G != curG || (int)B != curB) {
297 if (use_background) {
299 } else {
301 }
302 curR = (int)R;
303 curG = (int)G;
304 curB = (int)B;
305 }
306
307 // Emit UTF-8 character from cache
308 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
309 if (rep_is_profitable(run)) {
310 emit_rep(&ob, run - 1);
311 } else {
312 for (uint32_t k = 1; k < run; k++) {
313 // Emit UTF-8 character from cache
314 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
315 }
316 }
317 x = j;
318 }
319 }
320
321 // End row: reset SGR, add newline (except for last row) (copied from NEON)
322 emit_reset(&ob);
323 if (y < height - 1) {
324 ob_putc(&ob, '\n');
325 }
326 curR = curG = curB = -1;
327 cur_color_idx = -1;
328 }
329
330 ob_term(&ob);
331 return ob.buf;
332}
333
334// Destroy SVE cache resources (called at program shutdown)
335void sve_caches_destroy(void) {
336 // SVE currently uses shared caches from common.c, so no specific cleanup needed
337 log_debug("SVE_CACHE: SVE caches cleaned up");
338}
339
340#endif /* SIMD_SUPPORT_SVE */
⚙️ Common definitions, error codes, macros, and types shared throughout the application
unsigned short uint16_t
Definition common.h:57
unsigned int uint32_t
Definition common.h:58
#define SAFE_MALLOC(size, cast)
Definition common.h:264
unsigned char uint8_t
Definition common.h:56
@ ASCIICHAT_OK
Definition error_codes.h:51
#define log_error(...)
Log an ERROR message.
Definition log/log.h:587
#define log_debug(...)
Log a DEBUG message.
Definition log/log.h:548
#define LUMA_BLUE
Luminance blue coefficient (0.114 * 256 = 29)
void emit_set_256_color_bg(outbuf_t *ob, uint8_t color_idx)
Emit 256-color background ANSI sequence.
#define LUMA_GREEN
Luminance green coefficient (0.587 * 256 = 150)
uint8_t utf8_bytes[4]
utf8_palette_cache_t * get_utf8_palette_cache(const char *ascii_chars)
Get UTF-8 palette cache for a character set.
void emit_set_256_color_fg(outbuf_t *ob, uint8_t color_idx)
Emit 256-color foreground ANSI sequence.
void ob_term(outbuf_t *ob)
Append null terminator to buffer.
#define LUMA_THRESHOLD
Luminance threshold for rounding.
void ob_putc(outbuf_t *ob, char c)
Append a character to buffer.
bool rep_is_profitable(uint32_t runlen)
Check if run-length encoding is profitable.
void emit_set_truecolor_fg(outbuf_t *ob, uint8_t r, uint8_t g, uint8_t b)
Emit truecolor foreground ANSI sequence.
void emit_rep(outbuf_t *ob, uint32_t extra)
Emit run-length encoded sequence.
void ob_write(outbuf_t *ob, const char *s, size_t n)
Append a string to buffer.
void emit_reset(outbuf_t *ob)
Emit ANSI reset sequence.
utf8_char_t cache64[64]
void emit_set_truecolor_bg(outbuf_t *ob, uint8_t r, uint8_t g, uint8_t b)
Emit truecolor background ANSI sequence.
#define LUMA_RED
Luminance red coefficient (0.299 * 256 = 77)
✅ Safe Integer Arithmetic and Overflow Detection
#define R2(v, w, x, y, z, i)
Definition sha1.c:58
Image structure.
int w
Image width in pixels (must be > 0)
int h
Image height in pixels (must be > 0)
rgb_pixel_t * pixels
Pixel data array (width * height RGB pixels, row-major order)
Dynamic output buffer (auto-expanding)
size_t cap
Buffer capacity in bytes (maximum length before reallocation)
char * buf
Buffer pointer (allocated, owned by caller, must be freed)
RGB pixel structure.
uint8_t b
Blue color component (0-255)
uint8_t g
Green color component (0-255)
uint8_t r
Red color component (0-255)
SVE-optimized ASCII rendering functions.
SIMD-optimized ASCII conversion interface.