ascii-chat 0.11.33
Video chat in your terminal
Loading...
Searching...
No Matches
ssse3/color.c
Go to the documentation of this file.
1
7#if SIMD_SUPPORT_SSSE3
8#include <stdio.h>
9#include <stdlib.h>
10#include <string.h>
11#include <stdint.h>
12
13#include <tmmintrin.h>
14
17#include <ascii-chat/common.h>
20
21static inline uint8_t rgb_to_256color_ssse3(uint8_t r, uint8_t g, uint8_t b) {
22 return (uint8_t)(16 + 36 * (r / 51) + 6 * (g / 51) + (b / 51));
23}
24
25// Unified SSSE3 function for all color modes (full implementation like NEON)
26
27char *render_ascii_color_ssse3(const image_t *image, bool use_background, bool use_256color, const char *ascii_chars) {
28 if (!image || !image->pixels) {
29 return NULL;
30 }
31
32 const int width = image->w;
33 const int height = image->h;
34
35 if (width <= 0 || height <= 0) {
36 char *empty;
37 empty = SAFE_MALLOC(1, char *);
38 empty[0] = '\0';
39 return empty;
40 }
41
42 // Get cached UTF-8 character mappings for color rendering
43 utf8_palette_cache_t *utf8_cache = get_utf8_palette_cache(ascii_chars);
44 if (!utf8_cache) {
45 log_error("Failed to get UTF-8 palette cache for SSSE3 color");
46 return NULL;
47 }
48
49 outbuf_t ob = {0};
50 // Estimate buffer size based on mode (copied from NEON)
51 size_t bytes_per_pixel = use_256color ? 6u : 8u; // 256-color shorter than truecolor
52
53 // Calculate buffer size with overflow checking
54 size_t height_times_width;
55 if (checked_size_mul((size_t)height, (size_t)width, &height_times_width) != ASCIICHAT_OK) {
56 log_error("Buffer size overflow: height * width overflow");
57 return NULL;
58 }
59
60 size_t pixel_data_size;
61 if (checked_size_mul(height_times_width, bytes_per_pixel, &pixel_data_size) != ASCIICHAT_OK) {
62 log_error("Buffer size overflow: (height * width) * bytes_per_pixel overflow");
63 return NULL;
64 }
65
66 size_t height_times_16;
67 if (checked_size_mul((size_t)height, 16u, &height_times_16) != ASCIICHAT_OK) {
68 log_error("Buffer size overflow: height * 16 overflow");
69 return NULL;
70 }
71
72 size_t temp;
73 if (checked_size_add(pixel_data_size, height_times_16, &temp) != ASCIICHAT_OK) {
74 log_error("Buffer size overflow: pixel_data + height*16 overflow");
75 return NULL;
76 }
77
78 if (checked_size_add(temp, 64u, &ob.cap) != ASCIICHAT_OK) {
79 log_error("Buffer size overflow: total capacity overflow");
80 return NULL;
81 }
82 ob.buf = SAFE_MALLOC(ob.cap ? ob.cap : 1, char *);
83 if (!ob.buf)
84 return NULL;
85
86 // Build SSSE3 lookup table for _mm_shuffle_epi8 (uses character indices)
87 __m128i char_lut = _mm_loadu_si128((__m128i *)utf8_cache->char_index_ramp); // Load first 16 indices
88
89 // Track current color state (copied from NEON)
90 int curR = -1, curG = -1, curB = -1;
91 int cur_color_idx = -1;
92
93 for (int y = 0; y < height; y++) {
94 const rgb_pixel_t *row = &((const rgb_pixel_t *)image->pixels)[y * width];
95 int x = 0;
96
97 // Process 16-pixel chunks with SSSE3 (full 128-bit register capacity)
98 while (x + 16 <= width) {
99 // Manual deinterleave RGB components (SSSE3 limitation vs NEON's vld3q_u8)
100 uint8_t r_array[16], g_array[16], b_array[16];
101 for (int j = 0; j < 16; j++) {
102 r_array[j] = row[x + j].r;
103 g_array[j] = row[x + j].g;
104 b_array[j] = row[x + j].b;
105 }
106
107 // Load into SSSE3 registers
108 __m128i r_vec = _mm_loadl_epi64((__m128i *)r_array);
109 __m128i g_vec = _mm_loadl_epi64((__m128i *)g_array);
110 __m128i b_vec = _mm_loadl_epi64((__m128i *)b_array);
111
112 // Convert to 16-bit for arithmetic
113 __m128i r_16 = _mm_unpacklo_epi8(r_vec, _mm_setzero_si128());
114 __m128i g_16 = _mm_unpacklo_epi8(g_vec, _mm_setzero_si128());
115 __m128i b_16 = _mm_unpacklo_epi8(b_vec, _mm_setzero_si128());
116
117 // Calculate luminance: (77*R + 150*G + 29*B + 128) >> 8
118 __m128i luma_r = _mm_mullo_epi16(r_16, _mm_set1_epi16(LUMA_RED));
119 __m128i luma_g = _mm_mullo_epi16(g_16, _mm_set1_epi16(LUMA_GREEN));
120 __m128i luma_b = _mm_mullo_epi16(b_16, _mm_set1_epi16(LUMA_BLUE));
121
122 __m128i luma_sum = _mm_add_epi16(luma_r, luma_g);
123 luma_sum = _mm_add_epi16(luma_sum, luma_b);
124 luma_sum = _mm_add_epi16(luma_sum, _mm_set1_epi16(LUMA_THRESHOLD));
125 luma_sum = _mm_srli_epi16(luma_sum, 8);
126
127 // Pack back to 8-bit and store
128 __m128i luminance = _mm_packus_epi16(luma_sum, _mm_setzero_si128());
129 uint8_t luma_array[8];
130 _mm_storel_epi64((__m128i *)luma_array, luminance);
131
132 // FAST: Use _mm_shuffle_epi8 to get character indices from the ramp (SSSE3 advantage)
133 __m128i luma_vec = _mm_loadl_epi64((__m128i *)luma_array); // Load 8 luminance values
134 __m128i luma_idx_vec = _mm_srli_epi16(_mm_unpacklo_epi8(luma_vec, _mm_setzero_si128()), 2); // >> 2 for 0-63
135 __m128i luma_idx_8bit = _mm_packus_epi16(luma_idx_vec, _mm_setzero_si128()); // Pack back to 8-bit
136
137 // Use _mm_shuffle_epi8 for fast character index lookup
138 __m128i char_indices_vec = _mm_shuffle_epi8(char_lut, luma_idx_8bit);
139
140 uint8_t char_indices[8];
141 _mm_storel_epi64((__m128i *)char_indices, char_indices_vec);
142
143 if (use_256color) {
144 // 256-color mode processing (copied from NEON logic)
145 uint8_t color_indices[8];
146 for (int i = 0; i < 8; i++) {
147 color_indices[i] = rgb_to_256color_ssse3(r_array[i], g_array[i], b_array[i]);
148 }
149
150 // Emit with RLE on (glyph, color) runs (copied from NEON)
151 for (int i = 0; i < 8;) {
152 const uint8_t char_idx = char_indices[i];
153 const utf8_char_t *char_info = &utf8_cache->cache64[char_idx];
154 const uint8_t color_idx = color_indices[i];
155
156 int j = i + 1;
157 while (j < 8 && char_indices[j] == char_idx && color_indices[j] == color_idx) {
158 j++;
159 }
160 const uint32_t run = (uint32_t)(j - i);
161
162 if (color_idx != cur_color_idx) {
163 if (use_background) {
164 emit_set_256_color_bg(&ob, color_idx);
165 } else {
166 emit_set_256_color_fg(&ob, color_idx);
167 }
168 cur_color_idx = color_idx;
169 }
170
171 // Emit UTF-8 character from cache
172 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
173 if (rep_is_profitable(run)) {
174 emit_rep(&ob, run - 1);
175 } else {
176 for (uint32_t k = 1; k < run; k++) {
177 // Emit UTF-8 character from cache
178 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
179 }
180 }
181 i = j;
182 }
183 } else {
184 // Truecolor mode processing (copied from NEON logic)
185 for (int i = 0; i < 8;) {
186 const uint8_t char_idx = char_indices[i];
187 const utf8_char_t *char_info = &utf8_cache->cache64[char_idx];
188 const uint8_t r = r_array[i];
189 const uint8_t g = g_array[i];
190 const uint8_t b = b_array[i];
191
192 int j = i + 1;
193 while (j < 8 && char_indices[j] == char_idx && r_array[j] == r && g_array[j] == g && b_array[j] == b) {
194 j++;
195 }
196 const uint32_t run = (uint32_t)(j - i);
197
198 if (r != curR || g != curG || b != curB) {
199 if (use_background) {
200 emit_set_truecolor_bg(&ob, r, g, b);
201 } else {
202 emit_set_truecolor_fg(&ob, r, g, b);
203 }
204 curR = r;
205 curG = g;
206 curB = b;
207 }
208
209 // Emit UTF-8 character from cache
210 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
211 if (rep_is_profitable(run)) {
212 emit_rep(&ob, run - 1);
213 } else {
214 for (uint32_t k = 1; k < run; k++) {
215 // Emit UTF-8 character from cache
216 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
217 }
218 }
219 i = j;
220 }
221 }
222 x += 16;
223 }
224
225 // Scalar tail for remaining pixels (copied from NEON logic)
226 for (; x < width;) {
227 const rgb_pixel_t *p = &row[x];
228 uint32_t R = p->r, G = p->g, B = p->b;
229 uint8_t Y = (uint8_t)((LUMA_RED * R + LUMA_GREEN * G + LUMA_BLUE * B + LUMA_THRESHOLD) >> 8);
230 uint8_t luma_idx = Y >> 2;
231 const utf8_char_t *char_info = &utf8_cache->cache64[luma_idx];
232
233 if (use_256color) {
234 // 256-color scalar tail
235 uint8_t color_idx = rgb_to_256color_ssse3((uint8_t)R, (uint8_t)G, (uint8_t)B);
236
237 int j = x + 1;
238 while (j < width) {
239 const rgb_pixel_t *q = &row[j];
240 uint32_t R2 = q->r, G2 = q->g, B2 = q->b;
241 uint8_t Y2 = (uint8_t)((LUMA_RED * R2 + LUMA_GREEN * G2 + LUMA_BLUE * B2 + LUMA_THRESHOLD) >> 8);
242 uint8_t color_idx2 = rgb_to_256color_ssse3((uint8_t)R2, (uint8_t)G2, (uint8_t)B2);
243 if (((Y2 >> 2) != (Y >> 2)) || color_idx2 != color_idx)
244 break;
245 j++;
246 }
247 uint32_t run = (uint32_t)(j - x);
248
249 if (color_idx != cur_color_idx) {
250 if (use_background) {
251 emit_set_256_color_bg(&ob, color_idx);
252 } else {
253 emit_set_256_color_fg(&ob, color_idx);
254 }
255 cur_color_idx = color_idx;
256 }
257
258 // Emit UTF-8 character from cache
259 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
260 if (rep_is_profitable(run)) {
261 emit_rep(&ob, run - 1);
262 } else {
263 for (uint32_t k = 1; k < run; k++) {
264 // Emit UTF-8 character from cache
265 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
266 }
267 }
268 x = j;
269 } else {
270 // Truecolor scalar tail
271 int j = x + 1;
272 while (j < width) {
273 const rgb_pixel_t *q = &row[j];
274 uint32_t R2 = q->r, G2 = q->g, B2 = q->b;
275 uint8_t Y2 = (uint8_t)((LUMA_RED * R2 + LUMA_GREEN * G2 + LUMA_BLUE * B2 + LUMA_THRESHOLD) >> 8);
276 if (((Y2 >> 2) != (Y >> 2)) || R2 != R || G2 != G || B2 != B)
277 break;
278 j++;
279 }
280 uint32_t run = (uint32_t)(j - x);
281
282 if ((int)R != curR || (int)G != curG || (int)B != curB) {
283 if (use_background) {
285 } else {
287 }
288 curR = (int)R;
289 curG = (int)G;
290 curB = (int)B;
291 }
292
293 // Emit UTF-8 character from cache
294 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
295 if (rep_is_profitable(run)) {
296 emit_rep(&ob, run - 1);
297 } else {
298 for (uint32_t k = 1; k < run; k++) {
299 // Emit UTF-8 character from cache
300 ob_write(&ob, char_info->utf8_bytes, char_info->byte_len);
301 }
302 }
303 x = j;
304 }
305 }
306
307 // End row: reset SGR, add newline (except for last row)
308 emit_reset(&ob);
309 if (y < height - 1) {
310 ob_putc(&ob, '\n');
311 }
312 curR = curG = curB = -1;
313 cur_color_idx = -1;
314 }
315
316 ob_term(&ob);
317 return ob.buf;
318}
319
320// Destroy SSSE3 cache resources (called at program shutdown)
321void ssse3_caches_destroy(void) {
322 // SSSE3 currently uses shared caches from common.c, so no specific cleanup needed
323 log_debug("SSSE3_CACHE: SSSE3 caches cleaned up");
324}
325
326#endif /* SIMD_SUPPORT_SSSE3 */
⚙️ Common definitions, error codes, macros, and types shared throughout the application
unsigned int uint32_t
Definition common.h:58
#define SAFE_MALLOC(size, cast)
Definition common.h:264
unsigned char uint8_t
Definition common.h:56
@ ASCIICHAT_OK
Definition error_codes.h:51
#define log_error(...)
Log an ERROR message.
Definition log/log.h:587
#define log_debug(...)
Log a DEBUG message.
Definition log/log.h:548
#define LUMA_BLUE
Luminance blue coefficient (0.114 * 256 = 29)
void emit_set_256_color_bg(outbuf_t *ob, uint8_t color_idx)
Emit 256-color background ANSI sequence.
#define LUMA_GREEN
Luminance green coefficient (0.587 * 256 = 150)
uint8_t utf8_bytes[4]
utf8_palette_cache_t * get_utf8_palette_cache(const char *ascii_chars)
Get UTF-8 palette cache for a character set.
void emit_set_256_color_fg(outbuf_t *ob, uint8_t color_idx)
Emit 256-color foreground ANSI sequence.
void ob_term(outbuf_t *ob)
Append null terminator to buffer.
#define LUMA_THRESHOLD
Luminance threshold for rounding.
void ob_putc(outbuf_t *ob, char c)
Append a character to buffer.
bool rep_is_profitable(uint32_t runlen)
Check if run-length encoding is profitable.
void emit_set_truecolor_fg(outbuf_t *ob, uint8_t r, uint8_t g, uint8_t b)
Emit truecolor foreground ANSI sequence.
void emit_rep(outbuf_t *ob, uint32_t extra)
Emit run-length encoded sequence.
void ob_write(outbuf_t *ob, const char *s, size_t n)
Append a string to buffer.
void emit_reset(outbuf_t *ob)
Emit ANSI reset sequence.
utf8_char_t cache64[64]
void emit_set_truecolor_bg(outbuf_t *ob, uint8_t r, uint8_t g, uint8_t b)
Emit truecolor background ANSI sequence.
#define LUMA_RED
Luminance red coefficient (0.299 * 256 = 77)
✅ Safe Integer Arithmetic and Overflow Detection
#define R2(v, w, x, y, z, i)
Definition sha1.c:58
SSSE3-optimized ASCII rendering functions.
Image structure.
int w
Image width in pixels (must be > 0)
int h
Image height in pixels (must be > 0)
rgb_pixel_t * pixels
Pixel data array (width * height RGB pixels, row-major order)
Dynamic output buffer (auto-expanding)
size_t cap
Buffer capacity in bytes (maximum length before reallocation)
char * buf
Buffer pointer (allocated, owned by caller, must be freed)
RGB pixel structure.
uint8_t b
Blue color component (0-255)
uint8_t g
Green color component (0-255)
uint8_t r
Red color component (0-255)