ascii-chat 0.11.33
Video chat in your terminal
Loading...
Searching...
No Matches
neon/mono.c
Go to the documentation of this file.
1
7#if SIMD_SUPPORT_NEON
8#include <stdio.h>
9#include <stdlib.h>
10#include <stdint.h>
11#include <string.h>
12#include <stdarg.h>
13#include <time.h>
14#include <assert.h>
15#include <ascii-chat/atomic.h>
16#include <math.h>
17
18#include <arm_neon.h>
19
20#include <ascii-chat/common.h>
31#include <ascii-chat/log/log.h>
32
33//=============================================================================
34// Simple Monochrome ASCII Function (matches scalar image_print performance)
35//=============================================================================
36
37char *render_ascii_mono_neon(const image_t *image, const char *ascii_chars) {
38 if (!image || !image->pixels || !ascii_chars) {
39 return NULL;
40 }
41
42 const int h = image->h;
43 const int w = image->w;
44
45 if (h <= 0 || w <= 0) {
46 return NULL;
47 }
48
49 // Get cached UTF-8 character mappings
50 utf8_palette_cache_t *utf8_cache = get_utf8_palette_cache(ascii_chars);
51 if (!utf8_cache) {
52 log_error("Failed to get UTF-8 palette cache");
53 return NULL;
54 }
55
56 // Build NEON lookup tables inline (faster than caching - 30ns rebuild vs 50ns lookup)
57 uint8x16x4_t tbl, char_lut, length_lut, char_byte0_lut, char_byte1_lut, char_byte2_lut, char_byte3_lut;
58 build_neon_lookup_tables(utf8_cache, &tbl, &char_lut, &length_lut, &char_byte0_lut, &char_byte1_lut, &char_byte2_lut,
59 &char_byte3_lut);
60
61 // Estimate output buffer size for UTF-8 characters
62 const size_t max_char_bytes = 4; // Max UTF-8 character size
63
64 // Calculate buffer size with overflow checking
65 size_t w_times_bytes;
66 if (checked_size_mul((size_t)w, max_char_bytes, &w_times_bytes) != ASCIICHAT_OK) {
67 log_error("Buffer size overflow: width too large for UTF-8 encoding");
68 return NULL;
69 }
70
71 size_t w_times_bytes_plus_one;
72 if (checked_size_add(w_times_bytes, 1, &w_times_bytes_plus_one) != ASCIICHAT_OK) {
73 log_error("Buffer size overflow: width * bytes + 1 overflow");
74 return NULL;
75 }
76
77 size_t len;
78 if (checked_size_mul((size_t)h, w_times_bytes_plus_one, &len) != ASCIICHAT_OK) {
79 log_error("Buffer size overflow: height * (width * bytes + 1) overflow");
80 return NULL;
81 }
82
83 // Use SIMD-aligned allocation for optimal vectorized write performance
84 char *output = SAFE_MALLOC_SIMD(len, char *);
85 if (output == NULL) {
86 return NULL; // SAFE_MALLOC_SIMD already called FATAL, but satisfy analyzer
87 }
88
89 char *pos = output;
90 const rgb_pixel_t *pixels = (const rgb_pixel_t *)image->pixels;
91
92 // Pure NEON processing - no scalar fallbacks
93 for (int y = 0; y < h; y++) {
94 const rgb_pixel_t *row = &pixels[y * w];
95 int x = 0;
96
97 // Process 16 pixels at a time with NEON
98 for (; x + 15 < w; x += 16) {
99 // Load 16 RGB pixels (48 bytes)
100 uint8x16x3_t rgb = vld3q_u8((const uint8_t *)(row + x));
101
102 // Calculate luminance for all 16 pixels: (77*R + 150*G + 29*B + 128) >> 8
103 uint16x8_t luma_lo = vmull_u8(vget_low_u8(rgb.val[0]), vdup_n_u8(LUMA_RED)); // R * 77
104 luma_lo = vmlal_u8(luma_lo, vget_low_u8(rgb.val[1]), vdup_n_u8(LUMA_GREEN)); // + G * 150
105 luma_lo = vmlal_u8(luma_lo, vget_low_u8(rgb.val[2]), vdup_n_u8(LUMA_BLUE)); // + B * 29
106 luma_lo = vaddq_u16(luma_lo, vdupq_n_u16(128)); // + 128 (rounding)
107 luma_lo = vshrq_n_u16(luma_lo, 8); // >> 8
108
109 uint16x8_t luma_hi = vmull_u8(vget_high_u8(rgb.val[0]), vdup_n_u8(LUMA_RED));
110 luma_hi = vmlal_u8(luma_hi, vget_high_u8(rgb.val[1]), vdup_n_u8(LUMA_GREEN));
111 luma_hi = vmlal_u8(luma_hi, vget_high_u8(rgb.val[2]), vdup_n_u8(LUMA_BLUE));
112 luma_hi = vaddq_u16(luma_hi, vdupq_n_u16(128));
113 luma_hi = vshrq_n_u16(luma_hi, 8);
114
115 // Convert 16-bit luminance back to 8-bit
116 uint8x16_t luminance = vcombine_u8(vmovn_u16(luma_lo), vmovn_u16(luma_hi));
117
118 // NEON optimization: Use vqtbl4q_u8 for fast character index lookup
119 // Convert luminance (0-255) to 6-bit bucket (0-63) to match scalar behavior
120 uint8x16_t luma_buckets = vshrq_n_u8(luminance, 2); // >> 2 to get 0-63 range
121 uint8x16_t char_indices = vqtbl4q_u8(tbl, luma_buckets); // 16 lookups in 1 instruction!
122
123 // VECTORIZED UTF-8 CHARACTER GENERATION: Length-aware compaction
124
125 // Step 1: Get character lengths vectorially
126 uint8x16_t char_lengths = vqtbl4q_u8(length_lut, char_indices);
127
128 // Step 2: Check if all characters have same length (vectorized check)
129 uint8_t uniform_length;
130 if (all_same_length_neon(char_lengths, &uniform_length)) {
131
132 if (uniform_length == 1) {
133 // PURE ASCII PATH: 16 characters = 16 bytes (maximum vectorization)
134 uint8x16_t ascii_output = vqtbl4q_u8(char_lut, char_indices);
135 vst1q_u8((uint8_t *)pos, ascii_output);
136 pos += 16;
137
138 } else if (uniform_length == 4) {
139 // PURE 4-BYTE UTF-8 PATH: 16 characters = 64 bytes
140 // Gather all 4 byte streams in parallel
141 uint8x16_t byte0_stream = vqtbl4q_u8(char_byte0_lut, char_indices);
142 uint8x16_t byte1_stream = vqtbl4q_u8(char_byte1_lut, char_indices);
143 uint8x16_t byte2_stream = vqtbl4q_u8(char_byte2_lut, char_indices);
144 uint8x16_t byte3_stream = vqtbl4q_u8(char_byte3_lut, char_indices);
145
146 // Interleave bytes: [char0_byte0, char0_byte1, char0_byte2, char0_byte3, char1_byte0, ...]
147 uint8x16x4_t interleaved;
148 interleaved.val[0] = byte0_stream;
149 interleaved.val[1] = byte1_stream;
150 interleaved.val[2] = byte2_stream;
151 interleaved.val[3] = byte3_stream;
152
153 // Store interleaved UTF-8 data: 64 bytes total
154 vst4q_u8((uint8_t *)pos, interleaved);
155 pos += 64;
156
157 } else if (uniform_length == 2) {
158 // PURE 2-BYTE UTF-8 PATH: 16 characters = 32 bytes (vectorized)
159 uint8x16_t byte0_stream = vqtbl4q_u8(char_byte0_lut, char_indices);
160 uint8x16_t byte1_stream = vqtbl4q_u8(char_byte1_lut, char_indices);
161
162 // Interleave: [char0_b0, char0_b1, char1_b0, char1_b1, ...]
163 uint8x16x2_t interleaved_2byte;
164 interleaved_2byte.val[0] = byte0_stream;
165 interleaved_2byte.val[1] = byte1_stream;
166
167 vst2q_u8((uint8_t *)pos, interleaved_2byte);
168 pos += 32;
169
170 } else if (uniform_length == 3) {
171 // PURE 3-BYTE UTF-8 PATH: 16 characters = 48 bytes (vectorized)
172 uint8x16_t byte0_stream = vqtbl4q_u8(char_byte0_lut, char_indices);
173 uint8x16_t byte1_stream = vqtbl4q_u8(char_byte1_lut, char_indices);
174 uint8x16_t byte2_stream = vqtbl4q_u8(char_byte2_lut, char_indices);
175
176 // Interleave: [char0_b0, char0_b1, char0_b2, char1_b0, char1_b1, char1_b2, ...]
177 uint8x16x3_t interleaved_3byte;
178 interleaved_3byte.val[0] = byte0_stream;
179 interleaved_3byte.val[1] = byte1_stream;
180 interleaved_3byte.val[2] = byte2_stream;
181
182 vst3q_u8((uint8_t *)pos, interleaved_3byte);
183 pos += 48;
184 }
185
186 } else {
187 // MIXED LENGTH PATH: SIMD shuffle mask optimization
188 // Use vqtbl4q_u8 to gather UTF-8 bytes in 4 passes, then compact with fast scalar
189
190 // Gather all UTF-8 bytes using existing lookup tables with shuffle masks
191 uint8x16_t byte0_vec = vqtbl4q_u8(char_byte0_lut, char_indices);
192 uint8x16_t byte1_vec = vqtbl4q_u8(char_byte1_lut, char_indices);
193 uint8x16_t byte2_vec = vqtbl4q_u8(char_byte2_lut, char_indices);
194 uint8x16_t byte3_vec = vqtbl4q_u8(char_byte3_lut, char_indices);
195
196 // Store gathered bytes to temporary buffers
197 uint8_t byte0_buf[16], byte1_buf[16], byte2_buf[16], byte3_buf[16];
198 vst1q_u8(byte0_buf, byte0_vec);
199 vst1q_u8(byte1_buf, byte1_vec);
200 vst1q_u8(byte2_buf, byte2_vec);
201 vst1q_u8(byte3_buf, byte3_vec);
202
203 // Fast scalar compaction: emit only valid bytes based on character lengths
204 // Store char_indices to buffer for lookup
205 uint8_t char_idx_buf[16];
206 vst1q_u8(char_idx_buf, char_indices);
207
208 for (int i = 0; i < 16; i++) {
209 const uint8_t char_idx = char_idx_buf[i];
210 const uint8_t byte_len = utf8_cache->cache64[char_idx].byte_len;
211
212 // Emit bytes based on character length (1-4 bytes)
213 *pos++ = byte0_buf[i];
214 if (byte_len > 1)
215 *pos++ = byte1_buf[i];
216 if (byte_len > 2)
217 *pos++ = byte2_buf[i];
218 if (byte_len > 3)
219 *pos++ = byte3_buf[i];
220 }
221 }
222 }
223
224 // Handle remaining pixels with optimized scalar code using 64-entry cache
225 for (; x < w; x++) {
226 const rgb_pixel_t pixel = row[x];
227 const uint8_t luminance = (LUMA_RED * pixel.r + LUMA_GREEN * pixel.g + LUMA_BLUE * pixel.b + 128) >> 8;
228 const uint8_t luma_idx = luminance >> 2; // Map 0..255 to 0..63 (same as NEON)
229 const utf8_char_t *char_info = &utf8_cache->cache64[luma_idx]; // Direct cache64 access
230 // Optimized: Use direct assignment for single-byte ASCII characters
231 if (char_info->byte_len == 1) {
232 *pos++ = char_info->utf8_bytes[0];
233 } else {
234 // Fallback to full memcpy for multi-byte UTF-8
235 memcpy(pos, char_info->utf8_bytes, char_info->byte_len);
236 pos += char_info->byte_len;
237 }
238 }
239
240 // Add newline (except last row)
241 if (y < h - 1) {
242 *pos++ = '\n';
243 }
244 }
245
246 // Null terminate
247 *pos = '\0';
248
249 return output;
250}
251
252#endif
ANSI escape sequence utilities and fast color code generation.
⚛️ Atomic operations abstraction layer with debug tracking
⚙️ Common definitions, error codes, macros, and types shared throughout the application
#define SAFE_MALLOC_SIMD(size, cast)
Definition common.h:364
unsigned char uint8_t
Definition common.h:56
@ ASCIICHAT_OK
Definition error_codes.h:51
#define log_error(...)
Log an ERROR message.
Definition log/log.h:587
#define LUMA_BLUE
Luminance blue coefficient (0.114 * 256 = 29)
#define LUMA_GREEN
Luminance green coefficient (0.587 * 256 = 150)
uint8_t utf8_bytes[4]
utf8_palette_cache_t * get_utf8_palette_cache(const char *ascii_chars)
Get UTF-8 palette cache for a character set.
utf8_char_t cache64[64]
#define LUMA_RED
Luminance red coefficient (0.299 * 256 = 77)
Platform initialization and static synchronization helpers.
Lock-free module lifecycle state machine using pure stdatomic.
📝 Logging API with multiple log levels and terminal output control
🔢 Mathematical Utility Functions
NEON-optimized ASCII rendering functions.
✅ Safe Integer Arithmetic and Overflow Detection
Image structure.
int w
Image width in pixels (must be > 0)
int h
Image height in pixels (must be > 0)
rgb_pixel_t * pixels
Pixel data array (width * height RGB pixels, row-major order)
RGB pixel structure.
uint8_t b
Blue color component (0-255)
uint8_t g
Green color component (0-255)
uint8_t r
Red color component (0-255)
⏱️ High-precision timing utilities using sokol_time.h and uthash
SIMD-optimized ASCII conversion interface.
ARM NEON-accelerated ASCII rendering utilities (declarations)