ascii-chat 0.11.33
Video chat in your terminal
Loading...
Searching...
No Matches
sve/mono.c
Go to the documentation of this file.
1
7#if SIMD_SUPPORT_SVE
8#include <stdio.h>
9#include <stdlib.h>
10#include <string.h>
11#include <stdint.h>
13#include <ascii-chat/common.h>
14#include <ascii-chat/video/ascii/common.h> // For LUMA_RED, LUMA_GREEN, LUMA_BLUE, LUMA_THRESHOLD
15#include <ascii-chat/video/ascii/output_buffer.h> // For outbuf_t, emit_*, ob_*
16
17#include <arm_sve.h>
18
20
21//=============================================================================
22// Image-based API (matches NEON architecture)
23//=============================================================================
24
25// Simple monochrome ASCII function (matches scalar image_print performance)
26char *render_ascii_mono_sve(const image_t *image, const char *ascii_chars) {
27 if (!image || !image->pixels) {
28 return NULL;
29 }
30
31 const int h = image->h;
32 const int w = image->w;
33
34 if (h <= 0 || w <= 0) {
35 return NULL;
36 }
37
38 // Get cached UTF-8 character mappings
39 utf8_palette_cache_t *utf8_cache = get_utf8_palette_cache(ascii_chars);
40 if (!utf8_cache) {
41 log_error("Failed to get UTF-8 palette cache");
42 return NULL;
43 }
44
45 // Buffer size for UTF-8 characters
46 const size_t max_char_bytes = 4;
47 const size_t len = (size_t)h * ((size_t)w * max_char_bytes + 1);
48
49 char *output = SAFE_MALLOC(len, char *);
50
51 char *pos = output;
52 const rgb_pixel_t *pixels = (const rgb_pixel_t *)image->pixels;
53
54 // Pure SVE processing - matches NEON approach but with scalable vectors
55 for (int y = 0; y < h; y++) {
56 const rgb_pixel_t *row = &pixels[y * w];
57 int x = 0;
58
59 // Process pixels with SVE (scalable vector length - typically 128, 256, or 512 bits)
60 svbool_t pg = svptrue_b8(); // Predicate for all lanes
61 (void)pg; // May be unused in some code paths
62
63 while (x < w) {
64 // Calculate how many pixels we can process in this iteration
65 int remaining = w - x;
66 (void)remaining;
67 svbool_t pg_active = svwhilelt_b8_s32(x, w);
68 int vec_len = svcntb_pat(SV_ALL) / 3; // Vector length in RGB pixels (3 bytes per pixel)
69 int process_count = (remaining < vec_len) ? remaining : vec_len;
70
71 // Manual deinterleave RGB components (SVE limitation vs NEON's vld3)
72 uint8_t r_array[64], g_array[64], b_array[64]; // Max SVE vector size
73 for (int j = 0; j < process_count; j++) {
74 if (x + j < w) {
75 r_array[j] = row[x + j].r;
76 g_array[j] = row[x + j].g;
77 b_array[j] = row[x + j].b;
78 }
79 }
80
81 // Load into SVE vectors
82 svuint8_t r_vec = svld1_u8(pg_active, r_array);
83 svuint8_t g_vec = svld1_u8(pg_active, g_array);
84 svuint8_t b_vec = svld1_u8(pg_active, b_array);
85
86 // Convert to 16-bit for arithmetic
87 svuint16_t r_16 = svunpklo_u16(r_vec);
88 svuint16_t g_16 = svunpklo_u16(g_vec);
89 svuint16_t b_16 = svunpklo_u16(b_vec);
90
91 // Calculate luminance: (77*R + 150*G + 29*B + 128) >> 8
92 svuint16_t luma = svmul_n_u16_x(svptrue_b16(), r_16, LUMA_RED);
93 luma = svmla_n_u16_x(svptrue_b16(), luma, g_16, LUMA_GREEN);
94 luma = svmla_n_u16_x(svptrue_b16(), luma, b_16, LUMA_BLUE);
95 luma = svadd_n_u16_x(svptrue_b16(), luma, LUMA_THRESHOLD);
96 luma = svlsr_n_u16_x(svptrue_b16(), luma, 8);
97
98 // Store u16 luminance values (SVE1 compatible - no SVE2 narrowing intrinsics)
99 // After right-shift by 8, values are already in 0-255 range
100 uint16_t luma_temp[64];
101 svst1_u16(svptrue_b16(), luma_temp, luma);
102
103 // Convert to u8 array for ASCII lookup
104 uint8_t luma_array[64];
105 for (int j = 0; j < process_count; j++) {
106 luma_array[j] = (uint8_t)luma_temp[j];
107 }
108
109 for (int j = 0; j < process_count; j++) {
110 if (x + j < w) {
111 const utf8_char_t *char_info = &utf8_cache->cache[luma_array[j]];
112 // Optimized: Use direct assignment for single-byte ASCII characters
113 if (char_info->byte_len == 1) {
114 *pos++ = char_info->utf8_bytes[0];
115 } else {
116 // Fallback to full memcpy for multi-byte UTF-8
117 memcpy(pos, char_info->utf8_bytes, char_info->byte_len);
118 pos += char_info->byte_len;
119 }
120 }
121 }
122 x += process_count;
123 }
124
125 // Add newline (except for last row)
126 if (y < h - 1) {
127 *pos++ = '\n';
128 }
129 }
130
131 // Null terminate
132 *pos = '\0';
133
134 return output;
135}
136
137#endif /* SIMD_SUPPORT_SVE */
⚙️ Common definitions, error codes, macros, and types shared throughout the application
unsigned short uint16_t
Definition common.h:57
#define SAFE_MALLOC(size, cast)
Definition common.h:264
unsigned char uint8_t
Definition common.h:56
#define log_error(...)
Log an ERROR message.
Definition log/log.h:587
#define LUMA_BLUE
Luminance blue coefficient (0.114 * 256 = 29)
#define LUMA_GREEN
Luminance green coefficient (0.587 * 256 = 150)
uint8_t utf8_bytes[4]
utf8_palette_cache_t * get_utf8_palette_cache(const char *ascii_chars)
Get UTF-8 palette cache for a character set.
#define LUMA_THRESHOLD
Luminance threshold for rounding.
utf8_char_t cache[256]
#define LUMA_RED
Luminance red coefficient (0.299 * 256 = 77)
✅ Safe Integer Arithmetic and Overflow Detection
Image structure.
int w
Image width in pixels (must be > 0)
int h
Image height in pixels (must be > 0)
rgb_pixel_t * pixels
Pixel data array (width * height RGB pixels, row-major order)
RGB pixel structure.
uint8_t b
Blue color component (0-255)
uint8_t g
Green color component (0-255)
uint8_t r
Red color component (0-255)
SVE-optimized ASCII rendering functions.
SIMD-optimized ASCII conversion interface.