ascii-chat 0.11.33
Video chat in your terminal
Loading...
Searching...
No Matches
ssse3/mono.c
Go to the documentation of this file.
1
7#if SIMD_SUPPORT_SSSE3
8#include <stdio.h>
9#include <stdlib.h>
10#include <string.h>
11#include <stdint.h>
12
13#include <tmmintrin.h>
14
17#include <ascii-chat/common.h>
20
21//=============================================================================
22// Image-based API (matches NEON architecture)
23//=============================================================================
24
25// Simple monochrome ASCII function (matches scalar image_print performance)
26char *render_ascii_mono_ssse3(const image_t *image, const char *ascii_chars) {
27 if (!image || !image->pixels || !ascii_chars) {
28 return NULL;
29 }
30
31 const int h = image->h;
32 const int w = image->w;
33
34 if (h <= 0 || w <= 0) {
35 return NULL;
36 }
37
38 // Get cached UTF-8 character mappings
39 utf8_palette_cache_t *utf8_cache = get_utf8_palette_cache(ascii_chars);
40 if (!utf8_cache) {
41 log_error("Failed to get UTF-8 palette cache");
42 return NULL;
43 }
44
45 // Buffer size for UTF-8 characters
46 const size_t max_char_bytes = 4;
47
48 // Calculate buffer size with overflow checking
49 size_t w_times_bytes;
50 if (checked_size_mul((size_t)w, max_char_bytes, &w_times_bytes) != ASCIICHAT_OK) {
51 log_error("Buffer size overflow: width too large for UTF-8 encoding");
52 return NULL;
53 }
54
55 size_t w_times_bytes_plus_one;
56 if (checked_size_add(w_times_bytes, 1, &w_times_bytes_plus_one) != ASCIICHAT_OK) {
57 log_error("Buffer size overflow: width * bytes + 1 overflow");
58 return NULL;
59 }
60
61 size_t len;
62 if (checked_size_mul((size_t)h, w_times_bytes_plus_one, &len) != ASCIICHAT_OK) {
63 log_error("Buffer size overflow: height * (width * bytes + 1) overflow");
64 return NULL;
65 }
66
67 char *output;
68 output = SAFE_MALLOC(len, char *);
69
70 char *pos = output;
71 const rgb_pixel_t *pixels = (const rgb_pixel_t *)image->pixels;
72
73 // Pure SSSE3 processing - matches NEON approach
74 for (int y = 0; y < h; y++) {
75 const rgb_pixel_t *row = &pixels[y * w];
76 int x = 0;
77
78 // Process 16 pixels at a time with SSSE3 (full 128-bit register capacity)
79 for (; x + 15 < w; x += 16) {
80 // Manual deinterleave RGB components (SSSE3 limitation vs NEON's vld3q_u8)
81 uint8_t r_array[16], g_array[16], b_array[16];
82 for (int j = 0; j < 16; j++) {
83 r_array[j] = row[x + j].r;
84 g_array[j] = row[x + j].g;
85 b_array[j] = row[x + j].b;
86 }
87
88 // Process 16 pixels in two 8-pixel SSSE3 batches (same as SSE2 approach)
89 __m128i r_vec_lo = _mm_loadl_epi64((__m128i *)(r_array + 0));
90 __m128i r_vec_hi = _mm_loadl_epi64((__m128i *)(r_array + 8));
91 __m128i g_vec_lo = _mm_loadl_epi64((__m128i *)(g_array + 0));
92 __m128i g_vec_hi = _mm_loadl_epi64((__m128i *)(g_array + 8));
93 __m128i b_vec_lo = _mm_loadl_epi64((__m128i *)(b_array + 0));
94 __m128i b_vec_hi = _mm_loadl_epi64((__m128i *)(b_array + 8));
95
96 // Process first 8 pixels
97 __m128i r_16_lo = _mm_unpacklo_epi8(r_vec_lo, _mm_setzero_si128());
98 __m128i g_16_lo = _mm_unpacklo_epi8(g_vec_lo, _mm_setzero_si128());
99 __m128i b_16_lo = _mm_unpacklo_epi8(b_vec_lo, _mm_setzero_si128());
100
101 __m128i luma_r_lo = _mm_mullo_epi16(r_16_lo, _mm_set1_epi16(LUMA_RED));
102 __m128i luma_g_lo = _mm_mullo_epi16(g_16_lo, _mm_set1_epi16(LUMA_GREEN));
103 __m128i luma_b_lo = _mm_mullo_epi16(b_16_lo, _mm_set1_epi16(LUMA_BLUE));
104
105 __m128i luma_sum_lo = _mm_add_epi16(luma_r_lo, luma_g_lo);
106 luma_sum_lo = _mm_add_epi16(luma_sum_lo, luma_b_lo);
107 luma_sum_lo = _mm_add_epi16(luma_sum_lo, _mm_set1_epi16(LUMA_THRESHOLD));
108 luma_sum_lo = _mm_srli_epi16(luma_sum_lo, 8);
109
110 // Process second 8 pixels
111 __m128i r_16_hi = _mm_unpacklo_epi8(r_vec_hi, _mm_setzero_si128());
112 __m128i g_16_hi = _mm_unpacklo_epi8(g_vec_hi, _mm_setzero_si128());
113 __m128i b_16_hi = _mm_unpacklo_epi8(b_vec_hi, _mm_setzero_si128());
114
115 __m128i luma_r_hi = _mm_mullo_epi16(r_16_hi, _mm_set1_epi16(LUMA_RED));
116 __m128i luma_g_hi = _mm_mullo_epi16(g_16_hi, _mm_set1_epi16(LUMA_GREEN));
117 __m128i luma_b_hi = _mm_mullo_epi16(b_16_hi, _mm_set1_epi16(LUMA_BLUE));
118
119 __m128i luma_sum_hi = _mm_add_epi16(luma_r_hi, luma_g_hi);
120 luma_sum_hi = _mm_add_epi16(luma_sum_hi, luma_b_hi);
121 luma_sum_hi = _mm_add_epi16(luma_sum_hi, _mm_set1_epi16(LUMA_THRESHOLD));
122 luma_sum_hi = _mm_srli_epi16(luma_sum_hi, 8);
123
124 // Pack and store
125 __m128i luminance_lo = _mm_packus_epi16(luma_sum_lo, _mm_setzero_si128());
126 __m128i luminance_hi = _mm_packus_epi16(luma_sum_hi, _mm_setzero_si128());
127
128 uint8_t luma_array[16];
129 _mm_storel_epi64((__m128i *)(luma_array + 0), luminance_lo);
130 _mm_storel_epi64((__m128i *)(luma_array + 8), luminance_hi);
131
132 for (int j = 0; j < 16; j++) {
133 const utf8_char_t *char_info = &utf8_cache->cache[luma_array[j]];
134 // Optimized: Use direct assignment for single-byte ASCII characters
135 if (char_info->byte_len == 1) {
136 *pos++ = char_info->utf8_bytes[0];
137 } else {
138 // Fallback to full memcpy for multi-byte UTF-8
139 memcpy(pos, char_info->utf8_bytes, char_info->byte_len);
140 pos += char_info->byte_len;
141 }
142 }
143 }
144
145 // Handle remaining pixels with optimized scalar code
146 for (; x < w; x++) {
147 const rgb_pixel_t pixel = row[x];
148 const int luminance = (LUMA_RED * pixel.r + LUMA_GREEN * pixel.g + LUMA_BLUE * pixel.b + LUMA_THRESHOLD) >> 8;
149 const utf8_char_t *char_info = &utf8_cache->cache[luminance];
150 // Optimized: Use direct assignment for single-byte ASCII characters
151 if (char_info->byte_len == 1) {
152 *pos++ = char_info->utf8_bytes[0];
153 } else {
154 // Fallback to full memcpy for multi-byte UTF-8
155 memcpy(pos, char_info->utf8_bytes, char_info->byte_len);
156 pos += char_info->byte_len;
157 }
158 }
159
160 // Add newline (except last row)
161 if (y < h - 1) {
162 *pos++ = '\n';
163 }
164 }
165
166 // Null terminate
167 *pos = '\0';
168
169 return output;
170}
171
172// 256-color palette mapping (RGB to ANSI 256 color index) - copied from NEON
173static inline uint8_t rgb_to_256color_ssse3(uint8_t r, uint8_t g, uint8_t b) {
174 return (uint8_t)(16 + 36 * (r / 51) + 6 * (g / 51) + (b / 51));
175}
176
177// Unified SSSE3 function for all color modes (full implementation like NEON)
178
179#endif /* SIMD_SUPPORT_SSSE3 */
⚙️ Common definitions, error codes, macros, and types shared throughout the application
#define SAFE_MALLOC(size, cast)
Definition common.h:264
unsigned char uint8_t
Definition common.h:56
@ ASCIICHAT_OK
Definition error_codes.h:51
#define log_error(...)
Log an ERROR message.
Definition log/log.h:587
#define LUMA_BLUE
Luminance blue coefficient (0.114 * 256 = 29)
#define LUMA_GREEN
Luminance green coefficient (0.587 * 256 = 150)
uint8_t utf8_bytes[4]
utf8_palette_cache_t * get_utf8_palette_cache(const char *ascii_chars)
Get UTF-8 palette cache for a character set.
#define LUMA_THRESHOLD
Luminance threshold for rounding.
utf8_char_t cache[256]
#define LUMA_RED
Luminance red coefficient (0.299 * 256 = 77)
✅ Safe Integer Arithmetic and Overflow Detection
SSSE3-optimized ASCII rendering functions.
Image structure.
int w
Image width in pixels (must be > 0)
int h
Image height in pixels (must be > 0)
rgb_pixel_t * pixels
Pixel data array (width * height RGB pixels, row-major order)
RGB pixel structure.
uint8_t b
Blue color component (0-255)
uint8_t g
Green color component (0-255)
uint8_t r
Red color component (0-255)