ascii-chat 0.11.33
Video chat in your terminal
Loading...
Searching...
No Matches
sse2/mono.c
Go to the documentation of this file.
1
7#if SIMD_SUPPORT_SSE2
8#include <stdio.h>
9#include <stdlib.h>
10#include <string.h>
11#include <stdint.h>
12
13#include <emmintrin.h>
14
17#include <ascii-chat/common.h>
20
21//=============================================================================
22// Image-based API (matches NEON architecture)
23//=============================================================================
24
25// Simple monochrome ASCII function (matches scalar image_print performance)
26char *render_ascii_mono_sse2(const image_t *image, const char *ascii_chars) {
27 if (!image || !image->pixels || !ascii_chars) {
28 return NULL;
29 }
30
31 const int h = image->h;
32 const int w = image->w;
33
34 if (h <= 0 || w <= 0) {
35 return NULL;
36 }
37
38 // Get cached UTF-8 character mappings
39 utf8_palette_cache_t *utf8_cache = get_utf8_palette_cache(ascii_chars);
40 if (!utf8_cache) {
41 log_error("Failed to get UTF-8 palette cache");
42 return NULL;
43 }
44
45 // Buffer size for UTF-8 characters
46 const size_t max_char_bytes = 4;
47 const size_t len = (size_t)h * ((size_t)w * max_char_bytes + 1);
48
49 char *output;
50 output = SAFE_MALLOC(len, char *);
51
52 char *pos = output;
53 const rgb_pixel_t *pixels = (const rgb_pixel_t *)image->pixels;
54
55 // Pure SSE2 processing - matches NEON approach
56 for (int y = 0; y < h; y++) {
57 const rgb_pixel_t *row = &pixels[y * w];
58 int x = 0;
59
60 // Process 16 pixels at a time with SSE2 (full 128-bit register capacity)
61 for (; x + 15 < w; x += 16) {
62 // Manual deinterleave RGB components (SSE2 limitation vs NEON's vld3q_u8)
63 uint8_t r_array[16], g_array[16], b_array[16];
64 for (int j = 0; j < 16; j++) {
65 r_array[j] = row[x + j].r;
66 g_array[j] = row[x + j].g;
67 b_array[j] = row[x + j].b;
68 }
69
70 // Load full 16 bytes into SSE2 registers (process in two 8-pixel batches)
71 __m128i r_vec_lo = _mm_loadl_epi64((__m128i *)(r_array + 0)); // First 8 pixels
72 __m128i r_vec_hi = _mm_loadl_epi64((__m128i *)(r_array + 8)); // Second 8 pixels
73 __m128i g_vec_lo = _mm_loadl_epi64((__m128i *)(g_array + 0));
74 __m128i g_vec_hi = _mm_loadl_epi64((__m128i *)(g_array + 8));
75 __m128i b_vec_lo = _mm_loadl_epi64((__m128i *)(b_array + 0));
76 __m128i b_vec_hi = _mm_loadl_epi64((__m128i *)(b_array + 8));
77
78 // Process first 8 pixels
79 __m128i r_16_lo = _mm_unpacklo_epi8(r_vec_lo, _mm_setzero_si128());
80 __m128i g_16_lo = _mm_unpacklo_epi8(g_vec_lo, _mm_setzero_si128());
81 __m128i b_16_lo = _mm_unpacklo_epi8(b_vec_lo, _mm_setzero_si128());
82
83 __m128i luma_r_lo = _mm_mullo_epi16(r_16_lo, _mm_set1_epi16(77));
84 __m128i luma_g_lo = _mm_mullo_epi16(g_16_lo, _mm_set1_epi16(150));
85 __m128i luma_b_lo = _mm_mullo_epi16(b_16_lo, _mm_set1_epi16(29));
86
87 __m128i luma_sum_lo = _mm_add_epi16(luma_r_lo, luma_g_lo);
88 luma_sum_lo = _mm_add_epi16(luma_sum_lo, luma_b_lo);
89 luma_sum_lo = _mm_add_epi16(luma_sum_lo, _mm_set1_epi16(128));
90 luma_sum_lo = _mm_srli_epi16(luma_sum_lo, 8);
91
92 // Process second 8 pixels
93 __m128i r_16_hi = _mm_unpacklo_epi8(r_vec_hi, _mm_setzero_si128());
94 __m128i g_16_hi = _mm_unpacklo_epi8(g_vec_hi, _mm_setzero_si128());
95 __m128i b_16_hi = _mm_unpacklo_epi8(b_vec_hi, _mm_setzero_si128());
96
97 __m128i luma_r_hi = _mm_mullo_epi16(r_16_hi, _mm_set1_epi16(77));
98 __m128i luma_g_hi = _mm_mullo_epi16(g_16_hi, _mm_set1_epi16(150));
99 __m128i luma_b_hi = _mm_mullo_epi16(b_16_hi, _mm_set1_epi16(29));
100
101 __m128i luma_sum_hi = _mm_add_epi16(luma_r_hi, luma_g_hi);
102 luma_sum_hi = _mm_add_epi16(luma_sum_hi, luma_b_hi);
103 luma_sum_hi = _mm_add_epi16(luma_sum_hi, _mm_set1_epi16(128));
104 luma_sum_hi = _mm_srli_epi16(luma_sum_hi, 8);
105
106 // Pack both halves to 8-bit
107 __m128i luminance_lo = _mm_packus_epi16(luma_sum_lo, _mm_setzero_si128());
108 __m128i luminance_hi = _mm_packus_epi16(luma_sum_hi, _mm_setzero_si128());
109
110 // Store and convert to ASCII characters
111 uint8_t luma_array[16];
112 _mm_storel_epi64((__m128i *)(luma_array + 0), luminance_lo);
113 _mm_storel_epi64((__m128i *)(luma_array + 8), luminance_hi);
114
115 // Convert luminance to UTF-8 characters using optimized mappings
116 for (int j = 0; j < 16; j++) {
117 const utf8_char_t *char_info = &utf8_cache->cache[luma_array[j]];
118 // Optimized: Use direct assignment for single-byte ASCII characters
119 if (char_info->byte_len == 1) {
120 *pos++ = char_info->utf8_bytes[0];
121 } else {
122 // Fallback to full memcpy for multi-byte UTF-8
123 memcpy(pos, char_info->utf8_bytes, char_info->byte_len);
124 pos += char_info->byte_len;
125 }
126 }
127 }
128
129 // Handle remaining pixels with optimized scalar code
130 for (; x < w; x++) {
131 const rgb_pixel_t pixel = row[x];
132 const int luminance = (LUMA_RED * pixel.r + LUMA_GREEN * pixel.g + LUMA_BLUE * pixel.b + LUMA_THRESHOLD) >> 8;
133 const utf8_char_t *char_info = &utf8_cache->cache[luminance];
134 // Optimized: Use direct assignment for single-byte ASCII characters
135 if (char_info->byte_len == 1) {
136 *pos++ = char_info->utf8_bytes[0];
137 } else {
138 // Fallback to full memcpy for multi-byte UTF-8
139 memcpy(pos, char_info->utf8_bytes, char_info->byte_len);
140 pos += char_info->byte_len;
141 }
142 }
143
144 // Add clear-to-end-of-line and newline (except last row)
145 *pos++ = '\033';
146 *pos++ = '[';
147 *pos++ = 'K';
148 if (y < h - 1) {
149 *pos++ = '\n';
150 }
151 }
152
153 // Null terminate
154 *pos = '\0';
155
156 return output;
157}
158
159// 256-color palette mapping (RGB to ANSI 256 color index) - copied from NEON
160static inline uint8_t rgb_to_256color_sse2(uint8_t r, uint8_t g, uint8_t b) {
161 return (uint8_t)(16 + 36 * (r / 51) + 6 * (g / 51) + (b / 51));
162}
163
164// Unified SSE2 function for all color modes (full implementation like NEON)
165
166#endif /* SIMD_SUPPORT_SSE2 */
⚙️ Common definitions, error codes, macros, and types shared throughout the application
#define SAFE_MALLOC(size, cast)
Definition common.h:264
unsigned char uint8_t
Definition common.h:56
#define log_error(...)
Log an ERROR message.
Definition log/log.h:587
#define LUMA_BLUE
Luminance blue coefficient (0.114 * 256 = 29)
#define LUMA_GREEN
Luminance green coefficient (0.587 * 256 = 150)
uint8_t utf8_bytes[4]
utf8_palette_cache_t * get_utf8_palette_cache(const char *ascii_chars)
Get UTF-8 palette cache for a character set.
#define LUMA_THRESHOLD
Luminance threshold for rounding.
utf8_char_t cache[256]
#define LUMA_RED
Luminance red coefficient (0.299 * 256 = 77)
✅ Safe Integer Arithmetic and Overflow Detection
SSE2-optimized ASCII rendering functions.
Image structure.
int w
Image width in pixels (must be > 0)
int h
Image height in pixels (must be > 0)
rgb_pixel_t * pixels
Pixel data array (width * height RGB pixels, row-major order)
RGB pixel structure.
uint8_t b
Blue color component (0-255)
uint8_t g
Green color component (0-255)
uint8_t r
Red color component (0-255)