ascii-chat 0.11.33
Video chat in your terminal
Loading...
Searching...
No Matches
video/ascii/avx2/common.c
Go to the documentation of this file.
1
6#include <stdint.h>
7#include <ascii-chat/common.h>
9
10#if SIMD_SUPPORT_AVX2
11#include <immintrin.h>
12
13// Thread-local storage for AVX2 working buffers
14// These stay in L1 cache and are reused across function calls
15// Non-static for shared library compatibility (still thread-local)
16THREAD_LOCAL ALIGNED_32 uint8_t avx2_r_buffer[32];
17THREAD_LOCAL ALIGNED_32 uint8_t avx2_g_buffer[32];
18THREAD_LOCAL ALIGNED_32 uint8_t avx2_b_buffer[32];
19THREAD_LOCAL ALIGNED_32 uint8_t avx2_luminance_buffer[32];
20
21// Helper function to emit RLE repeat count (handles any count up to 9999)
22char *emit_rle_count(char *pos, uint32_t rep_count) {
23 *pos++ = '\x1b';
24 *pos++ = '[';
25
26 // Handle up to 4 digits (max 9999)
27 if (rep_count >= 1000) {
28 *pos++ = '0' + (rep_count / 1000);
29 *pos++ = '0' + ((rep_count / 100) % 10);
30 *pos++ = '0' + ((rep_count / 10) % 10);
31 *pos++ = '0' + (rep_count % 10);
32 } else if (rep_count >= 100) {
33 *pos++ = '0' + (rep_count / 100);
34 *pos++ = '0' + ((rep_count / 10) % 10);
35 *pos++ = '0' + (rep_count % 10);
36 } else if (rep_count >= 10) {
37 *pos++ = '0' + (rep_count / 10);
38 *pos++ = '0' + (rep_count % 10);
39 } else {
40 *pos++ = '0' + rep_count;
41 }
42 *pos++ = 'b';
43
44 return pos;
45}
46
47// Optimized AVX2 function to load 32 RGB pixels and separate channels
48// Uses simple loop that auto-vectorizes to VMOVDQU + VPSHUFB
49void avx2_load_rgb32_optimized(const rgb_pixel_t *__restrict pixels, uint8_t *__restrict r_out,
50 uint8_t *__restrict g_out, uint8_t *__restrict b_out) {
51 // Simple loop that compiler auto-vectorizes into efficient SIMD
52 for (int i = 0; i < 32; i++) {
53 r_out[i] = pixels[i].r;
54 g_out[i] = pixels[i].g;
55 b_out[i] = pixels[i].b;
56 }
57}
58
59// AVX2 function to compute luminance for 32 pixels
60void avx2_compute_luminance_32(const uint8_t *r_vals, const uint8_t *g_vals, const uint8_t *b_vals,
61 uint8_t *luminance_out) {
62 // Load all 32 RGB values into AVX2 registers
63 __m256i r_all = _mm256_loadu_si256((const __m256i_u *)r_vals);
64 __m256i g_all = _mm256_loadu_si256((const __m256i_u *)g_vals);
65 __m256i b_all = _mm256_loadu_si256((const __m256i_u *)b_vals);
66
67 // Process low 16 pixels with accurate coefficients (16-bit math to prevent overflow)
68 __m256i r_lo = _mm256_unpacklo_epi8(r_all, _mm256_setzero_si256());
69 __m256i g_lo = _mm256_unpacklo_epi8(g_all, _mm256_setzero_si256());
70 __m256i b_lo = _mm256_unpacklo_epi8(b_all, _mm256_setzero_si256());
71
72 __m256i luma_16_lo = _mm256_mullo_epi16(r_lo, _mm256_set1_epi16(77));
73 luma_16_lo = _mm256_add_epi16(luma_16_lo, _mm256_mullo_epi16(g_lo, _mm256_set1_epi16(150)));
74 luma_16_lo = _mm256_add_epi16(luma_16_lo, _mm256_mullo_epi16(b_lo, _mm256_set1_epi16(29)));
75 luma_16_lo = _mm256_add_epi16(luma_16_lo, _mm256_set1_epi16(128));
76 luma_16_lo = _mm256_srli_epi16(luma_16_lo, 8);
77
78 // Process high 16 pixels with accurate coefficients
79 __m256i r_hi = _mm256_unpackhi_epi8(r_all, _mm256_setzero_si256());
80 __m256i g_hi = _mm256_unpackhi_epi8(g_all, _mm256_setzero_si256());
81 __m256i b_hi = _mm256_unpackhi_epi8(b_all, _mm256_setzero_si256());
82
83 __m256i luma_16_hi = _mm256_mullo_epi16(r_hi, _mm256_set1_epi16(77));
84 luma_16_hi = _mm256_add_epi16(luma_16_hi, _mm256_mullo_epi16(g_hi, _mm256_set1_epi16(150)));
85 luma_16_hi = _mm256_add_epi16(luma_16_hi, _mm256_mullo_epi16(b_hi, _mm256_set1_epi16(29)));
86 luma_16_hi = _mm256_add_epi16(luma_16_hi, _mm256_set1_epi16(128));
87 luma_16_hi = _mm256_srli_epi16(luma_16_hi, 8);
88
89 // Pack back to 8-bit
90 __m256i luma_packed = _mm256_packus_epi16(luma_16_lo, luma_16_hi);
91
92 // After unpack and pack operations, bytes are already in correct order [0-31]
93 // luma_16_lo contains pixels 0-7 (lower 128) and 16-23 (upper 128)
94 // luma_16_hi contains pixels 8-15 (lower 128) and 24-31 (upper 128)
95 // packus produces: [0-7, 8-15] in lower 128, [16-23, 24-31] in upper 128
96 // No permute needed - this is already the correct sequential order
97 _mm256_storeu_si256((__m256i_u *)luminance_out, luma_packed);
98}
99
100#endif // SIMD_SUPPORT_AVX2
⚙️ Common definitions, error codes, macros, and types shared throughout the application
unsigned int uint32_t
Definition common.h:58
unsigned char uint8_t
Definition common.h:56
RGB pixel structure.
SIMD-optimized ASCII conversion interface.