ascii-chat 0.11.33
Video chat in your terminal
Loading...
Searching...
No Matches
avx2/mono.c
Go to the documentation of this file.
1
7#if SIMD_SUPPORT_AVX2
8
9#include <stdio.h>
10#include <stdlib.h>
11#include <string.h>
12#include <stdint.h>
13#include <stdbool.h>
16#include <ascii-chat/common.h>
21
22#include <immintrin.h>
23
24// Single-pass AVX2 monochrome renderer with immediate emission
25
26char *render_ascii_mono_avx2(const image_t *image, const char *ascii_chars) {
27 if (!image || !image->pixels || !ascii_chars) {
28 return NULL;
29 }
30
31 const int h = image->h;
32 const int w = image->w;
33
34 if (h <= 0 || w <= 0) {
35 return NULL;
36 }
37
38 // Get cached UTF-8 character mappings
39 utf8_palette_cache_t *utf8_cache = get_utf8_palette_cache(ascii_chars);
40 if (!utf8_cache) {
41 log_error("Failed to get UTF-8 palette cache");
42 return NULL;
43 }
44
45 const rgb_pixel_t *pixels = (const rgb_pixel_t *)image->pixels;
46
47 // Use malloc for output buffer (will be freed by caller)
48 // Each pixel can produce: 4 bytes UTF-8 + 8 bytes RLE escape (\x1b[9999b) = 12 bytes max
49 // Plus 1 newline per row
50 size_t output_size = (size_t)h * ((size_t)w * 12 + 1);
51
52 char *output = SAFE_MALLOC(output_size, char *);
53 if (!output) {
54 log_error("Failed to allocate output buffer for AVX2 rendering");
55 return NULL;
56 }
57
58 char *pos = output;
59
60 // Process row by row for better cache locality
61 for (int y = 0; y < h; y++) {
62 const rgb_pixel_t *row_pixels = &pixels[y * w];
63 int x = 0;
64
65 // AVX2 fast path: process 32 pixels at a time
66 while (x + 31 < w) {
67 // Process 32 pixels with AVX2 using thread-local buffers
68 avx2_load_rgb32_optimized(&row_pixels[x], avx2_r_buffer, avx2_g_buffer, avx2_b_buffer);
69 avx2_compute_luminance_32(avx2_r_buffer, avx2_g_buffer, avx2_b_buffer, avx2_luminance_buffer);
70
71 // Convert to character indices and emit immediately
72 int i = 0;
73 while (i < 32) {
74 const uint8_t luma_idx = avx2_luminance_buffer[i] >> 2; // 0-63
75 const utf8_char_t *char_info = &utf8_cache->cache64[luma_idx];
76
77 // Find run length within this chunk
78 int run_end = i + 1;
79 while (run_end < 32 && x + run_end < w) {
80 const uint8_t next_luma_idx = avx2_luminance_buffer[run_end] >> 2;
81 if (next_luma_idx != luma_idx)
82 break;
83 run_end++;
84 }
85 int run = run_end - i;
86
87 // Emit UTF-8 character with RLE
88 memcpy(pos, char_info->utf8_bytes, char_info->byte_len);
89 pos += char_info->byte_len;
90
91 if (rep_is_profitable(run)) {
92 pos = emit_rle_count(pos, run - 1);
93 } else {
94 // Emit remaining characters
95 for (int k = 1; k < run; k++) {
96 memcpy(pos, char_info->utf8_bytes, char_info->byte_len);
97 pos += char_info->byte_len;
98 }
99 }
100 i = run_end;
101 }
102 x += 32;
103 }
104
105 // Scalar processing for remaining pixels (< 32)
106 while (x < w) {
107 const rgb_pixel_t *p = &row_pixels[x];
108 const int luminance = (LUMA_RED * p->r + LUMA_GREEN * p->g + LUMA_BLUE * p->b + 128) >> 8;
109 const uint8_t luma_idx = luminance >> 2; // 0-255 -> 0-63
110 const utf8_char_t *char_info = &utf8_cache->cache64[luma_idx];
111
112 // Find run length for RLE
113 int j = x + 1;
114 while (j < w) {
115 const rgb_pixel_t *next_p = &row_pixels[j];
116 const int next_luminance = (LUMA_RED * next_p->r + LUMA_GREEN * next_p->g + LUMA_BLUE * next_p->b + 128) >> 8;
117 const uint8_t next_luma_idx = next_luminance >> 2;
118 if (next_luma_idx != luma_idx)
119 break;
120 j++;
121 }
122 int run = j - x;
123
124 // Emit UTF-8 character with RLE
125 memcpy(pos, char_info->utf8_bytes, char_info->byte_len);
126 pos += char_info->byte_len;
127
128 if (rep_is_profitable(run)) {
129 pos = emit_rle_count(pos, run - 1);
130 } else {
131 for (int k = 1; k < run; k++) {
132 memcpy(pos, char_info->utf8_bytes, char_info->byte_len);
133 pos += char_info->byte_len;
134 }
135 }
136 x = j;
137 }
138
139 // Add reset sequence and newline after each row (except last)
140 *pos++ = '\x1b';
141 *pos++ = '[';
142 *pos++ = '0';
143 *pos++ = 'm';
144 if (y < h - 1) {
145 *pos++ = '\n';
146 }
147 }
148
149 *pos = '\0'; // Null terminate
150
151 return output;
152}
153
154// Single-pass AVX2 color renderer with immediate emission
155
156#endif /* SIMD_SUPPORT_AVX2 */
ANSI escape sequence utilities and fast color code generation.
AVX2-optimized ASCII rendering functions.
⚙️ Common definitions, error codes, macros, and types shared throughout the application
#define SAFE_MALLOC(size, cast)
Definition common.h:264
unsigned char uint8_t
Definition common.h:56
#define log_error(...)
Log an ERROR message.
Definition log/log.h:587
#define LUMA_BLUE
Luminance blue coefficient (0.114 * 256 = 29)
#define LUMA_GREEN
Luminance green coefficient (0.587 * 256 = 150)
uint8_t utf8_bytes[4]
utf8_palette_cache_t * get_utf8_palette_cache(const char *ascii_chars)
Get UTF-8 palette cache for a character set.
bool rep_is_profitable(uint32_t runlen)
Check if run-length encoding is profitable.
utf8_char_t cache64[64]
#define LUMA_RED
Luminance red coefficient (0.299 * 256 = 77)
✅ Safe Integer Arithmetic and Overflow Detection
ClangTool/LibTooling compatibility shim for stdbool.h.
Image structure.
int w
Image width in pixels (must be > 0)
int h
Image height in pixels (must be > 0)
rgb_pixel_t * pixels
Pixel data array (width * height RGB pixels, row-major order)
RGB pixel structure.
uint8_t b
Blue color component (0-255)
uint8_t g
Green color component (0-255)
uint8_t r
Red color component (0-255)
Shared AVX2 helper functions.
SIMD-optimized ASCII conversion interface.