ascii-chat 0.11.33
Video chat in your terminal
Loading...
Searching...
No Matches
sse2/color.c
Go to the documentation of this file.
1
7#if SIMD_SUPPORT_SSE2
8#include <stdio.h>
9#include <stdlib.h>
10#include <string.h>
11#include <stdint.h>
12
13#include <emmintrin.h>
14
17#include <ascii-chat/common.h>
20
21static inline uint8_t rgb_to_256color_sse2(uint8_t r, uint8_t g, uint8_t b) {
22 return (uint8_t)(16 + 36 * (r / 51) + 6 * (g / 51) + (b / 51));
23}
24
25// Unified SSE2 function for all color modes (full implementation like NEON)
26
27char *render_ascii_color_sse2(const image_t *image, bool use_background, bool use_256color, const char *ascii_chars) {
28 if (!image || !image->pixels) {
29 return NULL;
30 }
31
32 const int width = image->w;
33 const int height = image->h;
34
35 if (width <= 0 || height <= 0) {
36 char *empty;
37 empty = SAFE_MALLOC(1, char *);
38 empty[0] = '\0';
39 return empty;
40 }
41
42 outbuf_t ob = {0};
43 // Estimate buffer size based on mode (copied from NEON)
44 size_t bytes_per_pixel = use_256color ? 6u : 8u; // 256-color shorter than truecolor
45
46 // Calculate buffer size with overflow checking
47 size_t height_times_width;
48 if (checked_size_mul((size_t)height, (size_t)width, &height_times_width) != ASCIICHAT_OK) {
49 log_error("Buffer size overflow: height * width overflow");
50 return NULL;
51 }
52
53 size_t pixel_data_size;
54 if (checked_size_mul(height_times_width, bytes_per_pixel, &pixel_data_size) != ASCIICHAT_OK) {
55 log_error("Buffer size overflow: (height * width) * bytes_per_pixel overflow");
56 return NULL;
57 }
58
59 size_t height_times_16;
60 if (checked_size_mul((size_t)height, 16u, &height_times_16) != ASCIICHAT_OK) {
61 log_error("Buffer size overflow: height * 16 overflow");
62 return NULL;
63 }
64
65 size_t temp;
66 if (checked_size_add(pixel_data_size, height_times_16, &temp) != ASCIICHAT_OK) {
67 log_error("Buffer size overflow: pixel_data + height*16 overflow");
68 return NULL;
69 }
70
71 if (checked_size_add(temp, 64u, &ob.cap) != ASCIICHAT_OK) {
72 log_error("Buffer size overflow: total capacity overflow");
73 return NULL;
74 }
75
76 ob.buf = SAFE_MALLOC(ob.cap ? ob.cap : 1, char *);
77 if (!ob.buf)
78 return NULL;
79
80 // Get cached UTF-8 character mappings for color rendering
81 utf8_palette_cache_t *utf8_cache = get_utf8_palette_cache(ascii_chars);
82 if (!utf8_cache) {
83 log_error("Failed to get UTF-8 palette cache for SSE2 color");
84 SAFE_FREE(ob.buf);
85 return NULL;
86 }
87
88 // SSE2 doesn't have _mm_shuffle_epi8 (introduced in SSSE3), so use scalar UTF-8 cache lookup
89 // This is still much faster than the old approach since UTF-8 parsing is cached
90
91 // Track current color state (copied from NEON)
92 int curR = -1, curG = -1, curB = -1;
93 int cur_color_idx = -1;
94
95 for (int y = 0; y < height; y++) {
96 const rgb_pixel_t *row = &((const rgb_pixel_t *)image->pixels)[y * width];
97 int x = 0;
98
99 // Process 16-pixel chunks with SSE2 (full 128-bit register capacity)
100 while (x + 16 <= width) {
101 // Manual deinterleave RGB components (SSE2 limitation vs NEON's vld3q_u8)
102 uint8_t r_array[16], g_array[16], b_array[16];
103 for (int j = 0; j < 16; j++) {
104 r_array[j] = row[x + j].r;
105 g_array[j] = row[x + j].g;
106 b_array[j] = row[x + j].b;
107 }
108
109 // Load into SSE2 registers
110 __m128i r_vec = _mm_loadl_epi64((__m128i *)r_array);
111 __m128i g_vec = _mm_loadl_epi64((__m128i *)g_array);
112 __m128i b_vec = _mm_loadl_epi64((__m128i *)b_array);
113
114 // Convert to 16-bit for arithmetic
115 __m128i r_16 = _mm_unpacklo_epi8(r_vec, _mm_setzero_si128());
116 __m128i g_16 = _mm_unpacklo_epi8(g_vec, _mm_setzero_si128());
117 __m128i b_16 = _mm_unpacklo_epi8(b_vec, _mm_setzero_si128());
118
119 // Calculate luminance: (77*R + 150*G + 29*B + 128) >> 8
120 __m128i luma_r = _mm_mullo_epi16(r_16, _mm_set1_epi16(LUMA_RED));
121 __m128i luma_g = _mm_mullo_epi16(g_16, _mm_set1_epi16(LUMA_GREEN));
122 __m128i luma_b = _mm_mullo_epi16(b_16, _mm_set1_epi16(LUMA_BLUE));
123
124 __m128i luma_sum = _mm_add_epi16(luma_r, luma_g);
125 luma_sum = _mm_add_epi16(luma_sum, luma_b);
126 luma_sum = _mm_add_epi16(luma_sum, _mm_set1_epi16(LUMA_THRESHOLD));
127 luma_sum = _mm_srli_epi16(luma_sum, 8);
128
129 // Pack back to 8-bit and store
130 __m128i luminance = _mm_packus_epi16(luma_sum, _mm_setzero_si128());
131 uint8_t luma_array[8];
132 _mm_storel_epi64((__m128i *)luma_array, luminance);
133
134 // Convert to UTF-8 character indices using cached mappings
135 uint8_t char_indices[8];
136 for (int i = 0; i < 8; i++) {
137 const uint8_t luma_idx = luma_array[i] >> 2; // 0-63 index
138 char_indices[i] = luma_idx; // Direct index into cache64
139 }
140
141 if (use_256color) {
142 // 256-color mode processing (copied from NEON logic)
143 uint8_t color_indices[8];
144 for (int i = 0; i < 8; i++) {
145 color_indices[i] = rgb_to_256color_sse2(r_array[i], g_array[i], b_array[i]);
146 }
147
148 // Emit with RLE on (UTF-8 character, color) runs
149 for (int i = 0; i < 8;) { // SSE2 processes 8 pixels, not 16
150 const uint8_t char_idx = char_indices[i];
151 const utf8_char_t *char_info = &utf8_cache->cache64[char_idx];
152 const uint8_t color_idx = color_indices[i];
153
154 int j = i + 1;
155 while (j < 8 && char_indices[j] == char_idx && color_indices[j] == color_idx) {
156 j++;
157 }
158 const uint32_t run = (uint32_t)(j - i);
159
160 if (color_idx != cur_color_idx) {
161 if (use_background) {
162 emit_set_256_color_bg(&ob, color_idx);
163 } else {
164 emit_set_256_color_fg(&ob, color_idx);
165 }
166 cur_color_idx = color_idx;
167 }
168
169 // Emit UTF-8 character from cache
170 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
171 if (rep_is_profitable(run)) {
172 emit_rep(&ob, run - 1);
173 } else {
174 for (uint32_t k = 1; k < run; k++) {
175 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
176 }
177 }
178 i = j;
179 }
180 } else {
181 // Truecolor mode processing with UTF-8 characters
182 for (int i = 0; i < 8;) { // SSE2 processes 8 pixels
183 const uint8_t char_idx = char_indices[i];
184 const utf8_char_t *char_info = &utf8_cache->cache64[char_idx];
185 const uint8_t r = r_array[i];
186 const uint8_t g = g_array[i];
187 const uint8_t b = b_array[i];
188
189 int j = i + 1;
190 while (j < 8 && char_indices[j] == char_idx && r_array[j] == r && g_array[j] == g && b_array[j] == b) {
191 j++;
192 }
193 const uint32_t run = (uint32_t)(j - i);
194
195 if (r != curR || g != curG || b != curB) {
196 if (use_background) {
197 emit_set_truecolor_bg(&ob, r, g, b);
198 } else {
199 emit_set_truecolor_fg(&ob, r, g, b);
200 }
201 curR = r;
202 curG = g;
203 curB = b;
204 }
205
206 // Emit UTF-8 character from cache
207 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
208 if (rep_is_profitable(run)) {
209 emit_rep(&ob, run - 1);
210 } else {
211 for (uint32_t k = 1; k < run; k++) {
212 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
213 }
214 }
215 i = j;
216 }
217 }
218 x += 16;
219 }
220
221 // Scalar tail for remaining pixels (copied from NEON logic)
222 for (; x < width;) {
223 const rgb_pixel_t *p = &row[x];
224 uint32_t R = p->r, G = p->g, B = p->b;
225 uint8_t Y = (uint8_t)((LUMA_RED * R + LUMA_GREEN * G + LUMA_BLUE * B + LUMA_THRESHOLD) >> 8);
226 uint8_t luma_idx = Y >> 2;
227 const utf8_char_t *char_info = &utf8_cache->cache64[luma_idx];
228
229 if (use_256color) {
230 // 256-color scalar tail with UTF-8
231 uint8_t color_idx = rgb_to_256color_sse2((uint8_t)R, (uint8_t)G, (uint8_t)B);
232
233 int j = x + 1;
234 while (j < width) {
235 const rgb_pixel_t *q = &row[j];
236 uint32_t R2 = q->r, G2 = q->g, B2 = q->b;
237 uint8_t Y2 = (uint8_t)((LUMA_RED * R2 + LUMA_GREEN * G2 + LUMA_BLUE * B2 + LUMA_THRESHOLD) >> 8);
238 uint8_t luma_idx2 = Y2 >> 2;
239 uint8_t color_idx2 = rgb_to_256color_sse2((uint8_t)R2, (uint8_t)G2, (uint8_t)B2);
240 if (luma_idx2 != luma_idx || color_idx2 != color_idx)
241 break;
242 j++;
243 }
244 uint32_t run = (uint32_t)(j - x);
245
246 if (color_idx != cur_color_idx) {
247 if (use_background) {
248 emit_set_256_color_bg(&ob, color_idx);
249 } else {
250 emit_set_256_color_fg(&ob, color_idx);
251 }
252 cur_color_idx = color_idx;
253 }
254
255 // Emit UTF-8 character from cache
256 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
257 if (rep_is_profitable(run)) {
258 emit_rep(&ob, run - 1);
259 } else {
260 for (uint32_t k = 1; k < run; k++) {
261 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
262 }
263 }
264 x = j;
265 } else {
266 // Truecolor scalar tail with UTF-8
267 int j = x + 1;
268 while (j < width) {
269 const rgb_pixel_t *q = &row[j];
270 uint32_t R2 = q->r, G2 = q->g, B2 = q->b;
271 uint8_t Y2 = (uint8_t)((LUMA_RED * R2 + LUMA_GREEN * G2 + LUMA_BLUE * B2 + LUMA_THRESHOLD) >> 8);
272 uint8_t luma_idx2 = Y2 >> 2;
273 if (luma_idx2 != luma_idx || R2 != R || G2 != G || B2 != B)
274 break;
275 j++;
276 }
277 uint32_t run = (uint32_t)(j - x);
278
279 if ((int)R != curR || (int)G != curG || (int)B != curB) {
280 if (use_background) {
282 } else {
284 }
285 curR = (int)R;
286 curG = (int)G;
287 curB = (int)B;
288 }
289
290 // Emit UTF-8 character from cache
291 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
292 if (rep_is_profitable(run)) {
293 emit_rep(&ob, run - 1);
294 } else {
295 for (uint32_t k = 1; k < run; k++) {
296 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
297 }
298 }
299 x = j;
300 }
301 }
302
303 // End row: clear to EOL, reset SGR, add newline (except for last row) (copied from NEON)
304 ob_putc(&ob, '\033');
305 ob_putc(&ob, '[');
306 ob_putc(&ob, 'K');
307 emit_reset(&ob);
308 if (y < height - 1) {
309 ob_putc(&ob, '\n');
310 }
311 curR = curG = curB = -1;
312 cur_color_idx = -1;
313 }
314
315 ob_term(&ob);
316 return ob.buf;
317}
318
319// Destroy SSE2 cache resources (called at program shutdown)
320void sse2_caches_destroy(void) {
321 // SSE2 currently uses shared caches from common.c, so no specific cleanup needed
322 log_debug("SSE2_CACHE: SSE2 caches cleaned up");
323}
324
325#endif /* SIMD_SUPPORT_SSE2 */
⚙️ Common definitions, error codes, macros, and types shared throughout the application
unsigned int uint32_t
Definition common.h:58
#define SAFE_FREE(ptr)
Definition common.h:376
#define SAFE_MALLOC(size, cast)
Definition common.h:264
unsigned char uint8_t
Definition common.h:56
@ ASCIICHAT_OK
Definition error_codes.h:51
#define log_error(...)
Log an ERROR message.
Definition log/log.h:587
#define log_debug(...)
Log a DEBUG message.
Definition log/log.h:548
#define LUMA_BLUE
Luminance blue coefficient (0.114 * 256 = 29)
void emit_set_256_color_bg(outbuf_t *ob, uint8_t color_idx)
Emit 256-color background ANSI sequence.
#define LUMA_GREEN
Luminance green coefficient (0.587 * 256 = 150)
uint8_t utf8_bytes[4]
utf8_palette_cache_t * get_utf8_palette_cache(const char *ascii_chars)
Get UTF-8 palette cache for a character set.
void emit_set_256_color_fg(outbuf_t *ob, uint8_t color_idx)
Emit 256-color foreground ANSI sequence.
void ob_term(outbuf_t *ob)
Append null terminator to buffer.
#define LUMA_THRESHOLD
Luminance threshold for rounding.
void ob_putc(outbuf_t *ob, char c)
Append a character to buffer.
bool rep_is_profitable(uint32_t runlen)
Check if run-length encoding is profitable.
void emit_set_truecolor_fg(outbuf_t *ob, uint8_t r, uint8_t g, uint8_t b)
Emit truecolor foreground ANSI sequence.
void emit_rep(outbuf_t *ob, uint32_t extra)
Emit run-length encoded sequence.
void ob_write(outbuf_t *ob, const char *s, size_t n)
Append a string to buffer.
void emit_reset(outbuf_t *ob)
Emit ANSI reset sequence.
utf8_char_t cache64[64]
void emit_set_truecolor_bg(outbuf_t *ob, uint8_t r, uint8_t g, uint8_t b)
Emit truecolor background ANSI sequence.
#define LUMA_RED
Luminance red coefficient (0.299 * 256 = 77)
✅ Safe Integer Arithmetic and Overflow Detection
#define R2(v, w, x, y, z, i)
Definition sha1.c:58
SSE2-optimized ASCII rendering functions.
Image structure.
int w
Image width in pixels (must be > 0)
int h
Image height in pixels (must be > 0)
rgb_pixel_t * pixels
Pixel data array (width * height RGB pixels, row-major order)
Dynamic output buffer (auto-expanding)
size_t cap
Buffer capacity in bytes (maximum length before reallocation)
char * buf
Buffer pointer (allocated, owned by caller, must be freed)
RGB pixel structure.
uint8_t b
Blue color component (0-255)
uint8_t g
Green color component (0-255)
uint8_t r
Red color component (0-255)