ascii-chat 0.11.33
Video chat in your terminal
Loading...
Searching...
No Matches
analysis.h File Reference

Audio Analysis and Debugging Interface. More...

Go to the source code of this file.

Data Structures

struct  audio_analysis_stats_t
 Audio analysis statistics for sent or received audio. More...
 

Functions

int audio_analysis_init (void)
 Initialize audio analysis.
 
void audio_analysis_track_sent_sample (float sample)
 Track sent audio sample.
 
void audio_analysis_track_sent_packet (size_t size)
 Track sent packet.
 
void audio_analysis_track_received_sample (float sample)
 Track received audio sample.
 
void audio_analysis_track_received_packet (size_t size)
 Track received packet.
 
const audio_analysis_stats_t * audio_analysis_get_sent_stats (void)
 Get sent audio statistics.
 
const audio_analysis_stats_t * audio_analysis_get_received_stats (void)
 Get received audio statistics.
 
void audio_analysis_print_report (void)
 Print audio analysis report.
 
void audio_analysis_set_aec3_metrics (double echo_return_loss, double echo_return_loss_enhancement, uint64_t delay_ns)
 Set AEC3 echo cancellation metrics.
 
void audio_analysis_destroy (void)
 Cleanup audio analysis.
 

Detailed Description

Audio Analysis and Debugging Interface.

Provides audio quality analysis for troubleshooting audio issues. Tracks sent and received audio characteristics for debugging.

Author
Zachary Fogg me@zf.nosp@m.o.gg
Date
2025

Definition in file analysis.h.

Function Documentation

◆ audio_analysis_destroy()

void audio_analysis_destroy ( void  )

Cleanup audio analysis.

Definition at line 881 of file analysis.c.

881 {
882 g_analysis_enabled = false;
883
884 // Close WAV files if they were open
885 if (g_sent_wav) {
886 wav_writer_close(g_sent_wav);
887 g_sent_wav = NULL;
888 log_info("Closed sent audio WAV file");
889 }
890 if (g_received_wav) {
891 wav_writer_close(g_received_wav);
892 g_received_wav = NULL;
893 log_info("Closed received audio WAV file");
894 }
895}
#define log_info(...)
Log an INFO message.
Definition log/log.h:561
void wav_writer_close(wav_writer_t *writer)
Close WAV file and finalize header.
Definition wav_writer.c:112

References log_info, and wav_writer_close().

◆ audio_analysis_get_received_stats()

const audio_analysis_stats_t * audio_analysis_get_received_stats ( void  )

Get received audio statistics.

Returns
Pointer to analysis stats (do not free)

Definition at line 506 of file analysis.c.

506 {
507 return &g_received_stats;
508}

◆ audio_analysis_get_sent_stats()

const audio_analysis_stats_t * audio_analysis_get_sent_stats ( void  )

Get sent audio statistics.

Returns
Pointer to analysis stats (do not free)

Definition at line 502 of file analysis.c.

502 {
503 return &g_sent_stats;
504}

◆ audio_analysis_init()

int audio_analysis_init ( void  )

Initialize audio analysis.

Returns
0 on success, negative on error

Definition at line 114 of file analysis.c.

114 {
115 SAFE_MEMSET(&g_sent_stats, sizeof(g_sent_stats), 0, sizeof(g_sent_stats));
116 SAFE_MEMSET(&g_received_stats, sizeof(g_received_stats), 0, sizeof(g_received_stats));
117
118 // Reset stuttering/gap tracking
119 SAFE_MEMSET(g_received_gap_intervals_ms, sizeof(g_received_gap_intervals_ms), 0, sizeof(g_received_gap_intervals_ms));
120 g_received_gap_count = 0;
121 g_received_silence_start_sample = 0;
122 g_received_last_silence_end_sample = 0;
123 SAFE_MEMSET(g_received_packet_times_ns, sizeof(g_received_packet_times_ns), 0, sizeof(g_received_packet_times_ns));
124 g_received_packet_times_count = 0;
125 SAFE_MEMSET(g_received_packet_sizes, sizeof(g_received_packet_sizes), 0, sizeof(g_received_packet_sizes));
126 g_received_total_audio_samples = 0;
127
128 // Reset echo detection
129 SAFE_MEMSET(g_echo_buffer, sizeof(g_echo_buffer), 0, sizeof(g_echo_buffer));
130 g_echo_buffer_pos = 0;
131 g_echo_correlation_sample_count = 0;
132 for (int i = 0; i < ECHO_DELAY_COUNT; i++) {
133 g_echo_correlation_strength[i] = 0;
134 g_echo_match_count[i] = 0;
135 }
136 g_detected_echo_delay_ms = 0;
137
138 // Reset beep detection
139 SAFE_MEMSET(g_received_beep_window, sizeof(g_received_beep_window), 0, sizeof(g_received_beep_window));
140 g_received_beep_window_idx = 0;
141 g_received_beep_events = 0;
142 g_received_tonal_samples = 0;
143 g_in_beep_burst = false;
144 g_beep_burst_samples = 0;
145
146 int64_t now_us = (int64_t)time_ns_to_us(time_get_ns());
147
148 g_sent_stats.timestamp_start_ns = now_us;
149 g_received_stats.timestamp_start_ns = now_us;
150
151 g_sent_last_sample = 0.0f;
152 g_received_last_sample = 0.0f;
153 g_sent_last_packet_time_us = now_us;
154 g_received_last_packet_time_us = now_us;
155
156 // Initialize WAV file dumping if enabled
157 if (wav_dump_enabled()) {
158 char tmp[PLATFORM_MAX_PATH_LENGTH];
159 if (platform_get_temp_dir(tmp, sizeof(tmp))) {
160 char path[PLATFORM_MAX_PATH_LENGTH];
161 safe_snprintf(path, sizeof(path), "%s/sent_audio.wav", tmp);
162 g_sent_wav = wav_writer_open(path, 48000, 1);
163 safe_snprintf(path, sizeof(path), "%s/received_audio.wav", tmp);
164 g_received_wav = wav_writer_open(path, 48000, 1);
165 }
166 if (g_sent_wav) {
167 log_info("Dumping sent audio to temp dir");
168 }
169 if (g_received_wav) {
170 log_info("Dumping received audio to temp dir");
171 }
172 }
173
174 g_analysis_enabled = true;
175 log_info("Audio analysis enabled");
176 return 0;
177}
#define ECHO_DELAY_COUNT
Definition analysis.c:89
#define SAFE_MEMSET(dest, dest_size, ch, count)
Definition common.h:469
uint64_t time_get_ns(void)
Get current monotonic time in nanoseconds.
Definition util/time.c:108
int safe_snprintf(char *buffer, size_t buffer_size, const char *format,...)
Safe formatted string printing to buffer.
Definition system.c:148
bool platform_get_temp_dir(char *temp_dir, size_t path_size)
Get the system temporary directory path.
Definition util.c:74
int64_t timestamp_start_ns
Definition analysis.h:35
#define PLATFORM_MAX_PATH_LENGTH
Definition system.c:69
bool wav_dump_enabled(void)
Check if audio dumping is enabled via environment.
Definition wav_writer.c:138
wav_writer_t * wav_writer_open(const char *filepath, int sample_rate, int channels)
Open WAV file for writing.
Definition wav_writer.c:48

References ASCIICHAT_OK, ECHO_DELAY_COUNT, log_info, platform_get_temp_dir(), PLATFORM_MAX_PATH_LENGTH, SAFE_MEMSET, safe_snprintf(), time_get_ns(), audio_analysis_stats_t::timestamp_start_ns, wav_dump_enabled(), and wav_writer_open().

◆ audio_analysis_print_report()

void audio_analysis_print_report ( void  )

Print audio analysis report.

Definition at line 519 of file analysis.c.

519 {
520 if (!g_analysis_enabled) {
521 return;
522 }
523
524 int64_t now_us = (int64_t)time_ns_to_us(time_get_ns());
525
526 g_sent_stats.timestamp_end_ns = now_us;
527 g_received_stats.timestamp_end_ns = now_us;
528
529 int64_t sent_duration_ms = (g_sent_stats.timestamp_end_ns - g_sent_stats.timestamp_start_ns) / NS_PER_MS_INT;
530 int64_t recv_duration_ms = (g_received_stats.timestamp_end_ns - g_received_stats.timestamp_start_ns) / NS_PER_MS_INT;
531
532 // Calculate RMS levels
533 float sent_rms = 0.0f;
534 float recv_rms = 0.0f;
535 if (g_sent_rms_sample_count > 0) {
536 sent_rms = sqrtf(g_sent_rms_accumulator / g_sent_rms_sample_count);
537 }
538 if (g_received_rms_sample_count > 0) {
539 recv_rms = sqrtf(g_received_rms_accumulator / g_received_rms_sample_count);
540 }
541
542 log_plain("================================================================================");
543 log_plain(" AUDIO ANALYSIS REPORT ");
544 log_plain("================================================================================");
545 log_plain("SENT AUDIO (Microphone Capture):");
546 log_plain(" Duration: %lld ms", (long long)sent_duration_ms);
547 log_plain(" Total Samples: %llu", (unsigned long long)g_sent_stats.total_samples);
548 log_plain(" Peak Level: %.4f (should be < 1.0)", g_sent_stats.peak_level);
549 log_plain(" RMS Level: %.4f (audio energy/loudness)", sent_rms);
550 log_plain(" Clipping Events: %llu samples (%.2f%%)", (unsigned long long)g_sent_stats.clipping_count,
551 g_sent_stats.total_samples > 0 ? (100.0 * g_sent_stats.clipping_count / g_sent_stats.total_samples) : 0);
552 log_plain(" Silent Samples: %llu samples (%.2f%%)", (unsigned long long)g_sent_stats.silent_samples,
553 g_sent_stats.total_samples > 0 ? (100.0 * g_sent_stats.silent_samples / g_sent_stats.total_samples) : 0);
554 if (g_sent_max_silence_burst > 0) {
555 log_plain(" Max Silence Burst: %llu samples", (unsigned long long)g_sent_max_silence_burst);
556 }
557 log_plain(" Packets Sent: %u", g_sent_stats.packets_count);
558 log_plain(" Status: %s", g_sent_stats.clipping_count > 0 ? "CLIPPING DETECTED!" : "OK");
559
560 log_plain("RECEIVED AUDIO (Playback):");
561 log_plain(" Duration: %lld ms", (long long)recv_duration_ms);
562 log_plain(" Total Samples: %llu", (unsigned long long)g_received_stats.total_samples);
563 log_plain(" Peak Level: %.4f", g_received_stats.peak_level);
564 log_plain(" RMS Level: %.4f (audio energy/loudness)", recv_rms);
565 log_plain(" Clipping Events: %llu samples (%.2f%%)", (unsigned long long)g_received_stats.clipping_count,
566 g_received_stats.total_samples > 0
567 ? (100.0 * g_received_stats.clipping_count / g_received_stats.total_samples)
568 : 0);
569 log_plain(" Silent Samples: %llu samples (%.2f%%)", (unsigned long long)g_received_stats.silent_samples,
570 g_received_stats.total_samples > 0
571 ? (100.0 * g_received_stats.silent_samples / g_received_stats.total_samples)
572 : 0);
573 if (g_received_max_silence_burst > 0) {
574 log_plain(" Max Silence Burst: %llu samples", (unsigned long long)g_received_max_silence_burst);
575 }
576 double low_energy_pct =
577 g_received_stats.total_samples > 0 ? (100.0 * g_received_low_energy_samples / g_received_stats.total_samples) : 0;
578 log_plain(" Very Quiet Samples: %llu samples (%.1f%%) [amplitude < 0.05]",
579 (unsigned long long)g_received_low_energy_samples, low_energy_pct);
580 log_plain(" Packets Received: %u", g_received_stats.packets_count);
581 log_plain(" Status: %s", g_received_stats.total_samples == 0 ? "NO AUDIO RECEIVED!" : "Receiving");
582
583 log_plain("QUALITY METRICS (Scratchy/Distorted Audio Detection):");
584 log_plain("SENT:");
585 log_plain(" Jitter Events: %llu (rapid amplitude changes)", (unsigned long long)g_sent_stats.jitter_count);
586 log_plain(" Discontinuities: %llu (packet arrival gaps > 100ms)",
587 (unsigned long long)g_sent_stats.discontinuity_count);
588 log_plain(" Max Gap Between Packets: %u ms (expected ~20ms per frame)", g_sent_stats.max_gap_ns);
589
590 log_plain("RECEIVED:");
591 log_plain(" Jitter Events: %llu (rapid amplitude changes)",
592 (unsigned long long)g_received_stats.jitter_count);
593 log_plain(" Discontinuities: %llu (packet arrival gaps > 100ms)",
594 (unsigned long long)g_received_stats.discontinuity_count);
595 log_plain(" Max Gap Between Packets: %u ms (expected ~20ms per frame)", g_received_stats.max_gap_ns);
596
597 // Beep/tone artifact detection
598 if (g_received_beep_events > 0 || g_received_tonal_samples > 0) {
599 double tonal_pct =
600 g_received_stats.total_samples > 0 ? (100.0 * g_received_tonal_samples / g_received_stats.total_samples) : 0;
601 log_plain("BEEP/TONE ARTIFACTS:");
602 log_plain(" Beep Events: %llu (short tonal bursts < 500ms)",
603 (unsigned long long)g_received_beep_events);
604 log_plain(" Tonal Samples: %llu samples (%.1f%%) [consistent frequency content]",
605 (unsigned long long)g_received_tonal_samples, tonal_pct);
606
607 if (g_received_beep_events > 10) {
608 log_plain(" 🔴 BEEPING DETECTED: %llu short tonal bursts - likely codec artifacts or system sounds!",
609 (unsigned long long)g_received_beep_events);
610 log_plain(" Possible causes:");
611 log_plain(" - Opus codec producing tonal artifacts during silence/transitions");
612 log_plain(" - Buffer underruns creating synthetic tones");
613 log_plain(" - AEC3 suppressor resonance");
614 log_plain(" - System notification sounds bleeding through");
615 } else if (g_received_beep_events > 3) {
616 log_plain(" ⚠️ Some beep artifacts detected (%llu events)", (unsigned long long)g_received_beep_events);
617 }
618 }
619
620 log_plain("DIAGNOSTICS:");
621 if (g_sent_stats.peak_level == 0) {
622 log_plain(" No audio captured from microphone!");
623 }
624 if (g_received_stats.total_samples == 0) {
625 log_plain(" No audio received from server!");
626 } else if (g_received_stats.peak_level < 0.01f) {
627 log_plain(" ⚠️ Received audio is very quiet (peak < 0.01)");
628 }
629 if (g_sent_stats.clipping_count > 0) {
630 log_plain(" Microphone input is clipping - reduce microphone volume");
631 }
632
633 // Echo detection diagnostics
634 log_plain("ECHO DETECTION (Echo Cancellation Quality Check):");
635 if (g_echo_correlation_sample_count > 0 && g_sent_stats.total_samples > 0) {
636 uint64_t max_matches = 0;
637 int best_delay_idx = -1;
638
639 // Find which delay has the most matches (if any)
640 for (int i = 0; i < ECHO_DELAY_COUNT; i++) {
641 if (g_echo_match_count[i] > max_matches) {
642 max_matches = g_echo_match_count[i];
643 best_delay_idx = i;
644 }
645 }
646
647 double echo_threshold_pct = 5.0; // If > 5% of samples match at a delay, it's echo
648
649 if (best_delay_idx >= 0) {
650 double match_pct = (100.0 * g_echo_match_count[best_delay_idx]) / g_echo_correlation_sample_count;
651 log_plain(" Echo correlation at different delays:");
652 for (int i = 0; i < ECHO_DELAY_COUNT; i++) {
653 double pct = (100.0 * g_echo_match_count[i]) / g_echo_correlation_sample_count;
654 const char *status = pct > echo_threshold_pct ? "⚠️ ECHO DETECTED" : "✓ OK";
655 log_plain(" %3u ms delay: %.1f%% match rate %s", g_echo_delays_ms[i], pct, status);
656 }
657
658 if (match_pct > echo_threshold_pct) {
659 g_detected_echo_delay_ms = g_echo_delays_ms[best_delay_idx];
660 log_plain(" 🔴 ECHO CANCELLATION NOT WORKING: Strong echo at %u ms delay!", g_detected_echo_delay_ms);
661 log_plain(" Received audio contains %.1f%% samples matching sent audio from %u ms ago", match_pct,
662 g_detected_echo_delay_ms);
663 } else {
664 log_plain(" ✓ Echo cancellation working: No significant echo detected");
665 }
666 }
667 } else {
668 log_plain(" Insufficient data for echo detection (need both sent and received audio)");
669 }
670
671 // AEC3 metrics from WebRTC (if available)
672 if (g_aec3_metrics_available) {
673 log_plain("AEC3 METRICS (from WebRTC GetMetrics()):");
674 log_plain(" Echo Return Loss (ERL): %.2f dB (how much echo is attenuated; >10 dB is good)",
675 g_aec3_echo_return_loss);
676 log_plain(" Echo Return Loss Enhancement (ERLE): %.2f dB (residual echo suppression)",
677 g_aec3_echo_return_loss_enhancement);
678 log_plain(" Estimated Echo Delay: %d ms", g_aec3_delay_ns);
679
680 if (g_aec3_echo_return_loss > 10.0) {
681 log_plain(" ✓ Good echo attenuation (ERL > 10 dB)");
682 } else if (g_aec3_echo_return_loss > 3.0) {
683 log_plain(" ⚠️ Moderate echo attenuation (3-10 dB)");
684 } else {
685 log_plain(" 🔴 Poor echo attenuation (ERL < 3 dB)");
686 }
687 }
688
689 // Audio quality diagnostics
690 if (recv_rms < 0.005f) {
691 log_plain(" ⚠️ CRITICAL: Received audio RMS is extremely low (%.6f) - barely audible!", recv_rms);
692 } else if (recv_rms < 0.02f) {
693 log_plain(" ⚠️ WARNING: Received audio RMS is low (%.6f) - may sound quiet or muddy", recv_rms);
694 }
695
696 // Silence analysis
697 double received_silence_pct = g_received_stats.total_samples > 0
698 ? (100.0 * g_received_stats.silent_samples / g_received_stats.total_samples)
699 : 0;
700
701 if (received_silence_pct > 30.0) {
702 log_plain(" ⚠️ SCRATCHY AUDIO DETECTED: Too much silence in received audio!");
703 log_plain(" - Silence: %.1f%% of received samples (should be < 10%%)", received_silence_pct);
704 log_plain(" - Max silence burst: %llu samples", (unsigned long long)g_received_max_silence_burst);
705 log_plain(" - This creates jittery/choppy playback between audio bursts");
706 } else if (received_silence_pct > 15.0) {
707 log_plain(" ⚠️ WARNING: Moderate silence detected (%.1f%%)", received_silence_pct);
708 }
709
710 // Sharp transition analysis (clicks/pops)
711 double sent_sharp_pct =
712 g_sent_transition_samples > 0 ? (100.0 * g_sent_sharp_transitions / g_sent_transition_samples) : 0;
713 double recv_sharp_pct =
714 g_received_transition_samples > 0 ? (100.0 * g_received_sharp_transitions / g_received_transition_samples) : 0;
715
716 // Zero crossing rate analysis (spectral content)
717 // Music: 1-5%, Speech: 5-15%, Static/Noise: 15-50%
718 double sent_zero_cross_pct =
719 g_sent_stats.total_samples > 0 ? (100.0 * g_sent_zero_crossings / g_sent_stats.total_samples) : 0;
720 double recv_zero_cross_pct =
721 g_received_stats.total_samples > 0 ? (100.0 * g_received_zero_crossings / g_received_stats.total_samples) : 0;
722
723 log_plain("WAVEFORM ANALYSIS (Is it clean music or corrupted/static?):");
724 log_plain("SENT AUDIO:");
725 log_plain(" Zero crossings: %.2f%% of samples (music: 1-5%%, noise: 15-50%%)", sent_zero_cross_pct);
726 log_plain(" Sharp transitions (clicks/pops): %.2f%% of samples", sent_sharp_pct);
727 log_plain(" Clipping samples: %llu (%.3f%%)", (unsigned long long)g_sent_clipping_samples,
728 g_sent_stats.total_samples > 0 ? (100.0 * g_sent_clipping_samples / g_sent_stats.total_samples) : 0);
729
730 log_plain("RECEIVED AUDIO:");
731 log_plain(" Zero crossings: %.2f%% of samples (music: 1-5%%, noise: 15-50%%)", recv_zero_cross_pct);
732 log_plain(" Sharp transitions (clicks/pops): %.2f%% of samples", recv_sharp_pct);
733 log_plain(" Clipping samples: %llu (%.3f%%)", (unsigned long long)g_received_clipping_samples,
734 g_received_stats.total_samples > 0 ? (100.0 * g_received_clipping_samples / g_received_stats.total_samples)
735 : 0);
736 log_plain(" Zero crossing increase: %.2f%% higher than sent (indicates corruption)",
737 recv_zero_cross_pct - sent_zero_cross_pct);
738
739 // Musicality verdict
740 log_plain("SOUND QUALITY VERDICT:");
741 if (recv_zero_cross_pct > 10.0) {
742 log_plain(" ⚠️ SOUNDS LIKE STATIC/DISTORTED: Excessive zero crossings (%.2f%%) = high frequency noise",
743 recv_zero_cross_pct);
744 log_plain(" Increase from sent: %.2f%% (waveform corruption detected)",
745 recv_zero_cross_pct - sent_zero_cross_pct);
746 log_plain(" Likely causes: Opus codec artifacts, jitter buffer issues, or packet delivery gaps");
747 } else if (recv_zero_cross_pct - sent_zero_cross_pct > 3.0) {
748 log_plain(" ⚠️ SOUNDS CORRUPTED: Zero crossing rate increased by %.2f%% (should be ±0.5%%)",
749 recv_zero_cross_pct - sent_zero_cross_pct);
750 log_plain(" Indicates waveform distortion from network/processing artifacts");
751 } else if (recv_sharp_pct > 2.0) {
752 log_plain(" ⚠️ SOUNDS LIKE STATIC: High click/pop rate (%.2f%%) indicates audio artifacts", recv_sharp_pct);
753 log_plain(" Likely causes: Packet loss, jitter buffer issues, or frame discontinuities");
754 } else if (g_received_clipping_samples > (g_received_stats.total_samples / 1000)) {
755 log_plain(" ⚠️ SOUNDS DISTORTED: Significant clipping detected (%.3f%%)",
756 100.0 * g_received_clipping_samples / g_received_stats.total_samples);
757 log_plain(" Likely causes: AGC too aggressive, gain too high, or codec compression artifacts");
758 } else if (low_energy_pct > 50.0 && recv_rms < 0.05f) {
759 log_plain(" ⚠️ SOUNDS MUDDY/QUIET: Over 50%% very quiet samples + low RMS");
760 log_plain(" Audio may sound unclear or like background noise rather than music");
761 } else if (received_silence_pct > 10.0) {
762 log_plain(" ⚠️ SOUNDS SCRATCHY: Excessive silence (%.1f%%) causes dropouts", received_silence_pct);
763 } else if (recv_rms > 0.08f && recv_zero_cross_pct < 6.0 && recv_sharp_pct < 1.0 &&
764 g_received_clipping_samples == 0) {
765 log_plain(" ✓ SOUNDS LIKE MUSIC: Good RMS (%.4f), clean waveform (%.2f%% zero crossings), minimal artifacts",
766 recv_rms, recv_zero_cross_pct);
767 log_plain(" Audio quality acceptable for communication");
768 } else {
769 log_plain(" ? BORDERLINE: Check specific metrics above");
770 }
771
772 // Low energy audio analysis
773 if (low_energy_pct > 50.0) {
774 log_plain(" ⚠️ WARNING: Over 50%% of received samples are very quiet (< 0.05 amplitude)");
775 log_plain(" - This makes audio sound muddy, unclear, or hard to understand");
776 log_plain(" - Caused by: Mixing other clients' audio with your own at wrong levels");
777 }
778
779 // Stuttering/periodic gap detection using packet inter-arrival times
780 if (g_received_packet_times_count >= 5) {
781 uint32_t inter_arrival_times_ms[MAX_PACKET_SAMPLES - 1];
782 uint32_t inter_arrival_count = 0;
783 uint32_t min_interval_ms = 0xFFFFFFFF;
784 uint32_t max_interval_ms = 0;
785 uint64_t sum_intervals_ms = 0;
786 uint32_t intervals_around_50ms = 0; // Count intervals ~40-60ms
787
788 // Calculate inter-packet arrival times
789 for (uint32_t i = 1; i < g_received_packet_times_count; i++) {
790 uint64_t prev_ns = g_received_packet_times_ns[i - 1];
791 uint64_t curr_ns = g_received_packet_times_ns[i];
792
793 uint64_t gap_ns = time_elapsed_ns(prev_ns, curr_ns);
794 uint32_t gap_ms = (uint32_t)time_ns_to_ms(gap_ns);
795
796 inter_arrival_times_ms[inter_arrival_count++] = gap_ms;
797 if (gap_ms < min_interval_ms)
798 min_interval_ms = gap_ms;
799 if (gap_ms > max_interval_ms)
800 max_interval_ms = gap_ms;
801 sum_intervals_ms += gap_ms;
802
803 // Check if interval is ~50ms (within 15ms tolerance for network jitter)
804 if (gap_ms >= 35 && gap_ms <= 70) {
805 intervals_around_50ms++;
806 }
807 }
808
809 uint32_t avg_interval_ms = (uint32_t)(sum_intervals_ms / inter_arrival_count);
810 uint32_t interval_consistency = (intervals_around_50ms * 100) / inter_arrival_count;
811
812 // Calculate how much audio is in each packet
813 // Total decoded samples / number of packets = average samples per packet
814 // At 48kHz, 960 samples = 1 Opus frame = 20ms
815 double avg_samples_per_packet =
816 g_received_stats.total_samples > 0 ? (double)g_received_stats.total_samples / inter_arrival_count : 0;
817 double frames_per_packet = avg_samples_per_packet / 960.0; // 960 samples = 1 frame @ 48kHz
818 double ms_audio_per_packet = frames_per_packet * 20.0; // 20ms per frame
819
820 // Detect if stuttering is periodic (consistent ~50ms intervals)
821 if (intervals_around_50ms >= (inter_arrival_count * 2 / 3)) {
822 // More than 66% of packets are ~50ms apart - clear periodic stuttering
823 log_plain(" 🔴 PERIODIC STUTTERING DETECTED: Server sends packets every ~%u ms (should be ~20ms)!",
824 avg_interval_ms);
825 log_plain(" - Packet inter-arrival: %u-%u ms (avg: %u ms)", min_interval_ms, max_interval_ms, avg_interval_ms);
826 log_plain(" - %u/%u packets (~%u%%) are ~50ms apart (CLEAR STUTTERING PATTERN)", intervals_around_50ms,
827 inter_arrival_count, interval_consistency);
828
829 log_plain(" - PACKET ANALYSIS:");
830 log_plain(" - Total audio samples: %llu over %u packets", (unsigned long long)g_received_stats.total_samples,
831 inter_arrival_count);
832 log_plain(" - Avg samples per packet: %.0f (= %.2f Opus frames = %.1f ms)", avg_samples_per_packet,
833 frames_per_packet, ms_audio_per_packet);
834
835 if (frames_per_packet < 1.5) {
836 log_plain(" - ❌ PROBLEM: Each packet contains < 1.5 frames (should be 2-3 frames!)");
837 log_plain(" - With only %.1f frames per packet arriving every %u ms, there are gaps between chunks",
838 frames_per_packet, avg_interval_ms);
839 log_plain(" - Audio plays for ~%.0f ms, then %u ms gap, then plays again", ms_audio_per_packet,
840 avg_interval_ms - (uint32_t)ms_audio_per_packet);
841 } else if (frames_per_packet > 2.5) {
842 log_plain(" - ✓ Packets contain %.1f frames (~%.0f ms audio each)", frames_per_packet,
843 ms_audio_per_packet);
844 log_plain(" - Should play smoothly if jitter buffer is large enough");
845 log_plain(" - If still stuttering, issue is jitter buffer depth or timing precision");
846 } else {
847 log_plain(" - Packets contain %.1f frames (~%.0f ms)", frames_per_packet, ms_audio_per_packet);
848 log_plain(" - Borderline: buffer needs to hold %.0f ms to bridge %.u ms gap", ms_audio_per_packet,
849 avg_interval_ms - (uint32_t)ms_audio_per_packet);
850 }
851 } else if (avg_interval_ms > 30) {
852 log_plain(" ⚠️ AUDIO DELIVERY INCONSISTENCY: Server packets arrive every ~%u ms (expected ~20ms)",
853 avg_interval_ms);
854 log_plain(" - Interval range: %u-%u ms", min_interval_ms, max_interval_ms);
855 log_plain(" - This causes dropouts and buffering issues");
856 }
857 }
858
859 // Packet delivery gaps
860 if (g_received_stats.max_gap_ns > 40) {
861 log_plain(" ⚠️ DISTORTION DETECTED: Packet delivery gaps too large!");
862 log_plain(" - Max gap: %u ms (should be ~20ms for smooth audio)", g_received_stats.max_gap_ns);
863 if (g_received_stats.max_gap_ns > 80) {
864 log_plain(" - SEVERE: Gaps > 80ms cause severe distortion and dropouts");
865 } else if (g_received_stats.max_gap_ns > 50) {
866 log_plain(" - Gaps > 50ms cause noticeable distortion");
867 }
868 }
869 if (g_received_stats.discontinuity_count > 0) {
870 log_plain(" Packet delivery discontinuities: %llu gaps > 100ms detected",
871 (unsigned long long)g_received_stats.discontinuity_count);
872 }
873 if (g_received_stats.jitter_count > (g_received_stats.total_samples / 100)) {
874 log_plain(" High jitter detected: > 1%% of samples have rapid amplitude changes");
875 log_plain(" - May indicate buffer underruns from sparse packet delivery");
876 }
877
878 log_plain("================================================================================");
879}
#define MAX_PACKET_SAMPLES
Definition analysis.c:76
unsigned int uint32_t
Definition common.h:58
unsigned long long uint64_t
Definition common.h:59
#define log_plain(...)
Plain logging - writes to both log file and stderr without timestamps or log levels.
Definition log/log.h:618
#define NS_PER_MS_INT
Definition time.h:156
uint64_t time_elapsed_ns(uint64_t start_ns, uint64_t end_ns)
Calculate elapsed time with wraparound safety.
Definition util/time.c:150
uint64_t silent_samples
Definition analysis.h:31
uint64_t clipping_count
Definition analysis.h:30
uint64_t discontinuity_count
Definition analysis.h:39

References audio_analysis_stats_t::clipping_count, audio_analysis_stats_t::discontinuity_count, ECHO_DELAY_COUNT, audio_analysis_stats_t::jitter_count, log_plain, audio_analysis_stats_t::max_gap_ns, MAX_PACKET_SAMPLES, NS_PER_MS_INT, audio_analysis_stats_t::packets_count, audio_analysis_stats_t::peak_level, audio_analysis_stats_t::silent_samples, time_elapsed_ns(), time_get_ns(), audio_analysis_stats_t::timestamp_end_ns, audio_analysis_stats_t::timestamp_start_ns, and audio_analysis_stats_t::total_samples.

◆ audio_analysis_set_aec3_metrics()

void audio_analysis_set_aec3_metrics ( double  echo_return_loss,
double  echo_return_loss_enhancement,
uint64_t  delay_ns 
)

Set AEC3 echo cancellation metrics.

Parameters
echo_return_lossEcho return loss (dB) - how much echo is attenuated
echo_return_loss_enhancementAdditional echo suppression (dB)
delay_nsEstimated echo delay in nanoseconds

Definition at line 510 of file analysis.c.

510 {
511 // Store AEC3 metrics for reporting
512 // These come from WebRTC EchoControl::GetMetrics() call
513 g_aec3_echo_return_loss = echo_return_loss;
514 g_aec3_echo_return_loss_enhancement = echo_return_loss_enhancement;
515 g_aec3_delay_ns = delay_ns;
516 g_aec3_metrics_available = true;
517}

Referenced by client_audio_pipeline_process_duplex().

◆ audio_analysis_track_received_packet()

void audio_analysis_track_received_packet ( size_t  size)

Track received packet.

Parameters
sizePacket size in bytes

Definition at line 469 of file analysis.c.

469 {
470 (void)size; // Unused parameter - reserved for future per-packet analysis
471 if (!g_analysis_enabled)
472 return;
473
474 uint64_t now_ns = time_get_ns();
475 int64_t now_us = (int64_t)time_ns_to_us(now_ns);
476
477 // Track packet timing for stuttering detection
478 if (g_received_packet_times_count < MAX_PACKET_SAMPLES) {
479 g_received_packet_times_ns[g_received_packet_times_count++] = now_ns;
480 }
481
482 // Detect gaps between consecutive packets (discontinuity)
483 if (g_received_stats.packets_count > 0) {
484 int64_t gap_us = now_us - g_received_last_packet_time_us;
485 int32_t gap_ms = (int32_t)(gap_us / 1000);
486
487 // Expected: ~20ms per Opus frame, flag if gap > 100ms
488 if (gap_ms > 100) {
489 g_received_stats.discontinuity_count++;
490 }
491
492 // Track max gap
493 if (gap_ms > (int32_t)g_received_stats.max_gap_ns) {
494 g_received_stats.max_gap_ns = (uint32_t)gap_ms;
495 }
496 }
497
498 g_received_last_packet_time_us = now_us;
499 g_received_stats.packets_count++;
500}

References audio_analysis_stats_t::discontinuity_count, audio_analysis_stats_t::max_gap_ns, MAX_PACKET_SAMPLES, audio_analysis_stats_t::packets_count, and time_get_ns().

◆ audio_analysis_track_received_sample()

void audio_analysis_track_received_sample ( float  sample)

Track received audio sample.

Parameters
sampleAudio sample value

Definition at line 278 of file analysis.c.

278 {
279 if (!g_analysis_enabled)
280 return;
281
282 g_received_stats.total_samples++;
283
284 // Track peak level
285 float abs_sample = fabsf(sample);
286 if (abs_sample > g_received_stats.peak_level) {
287 g_received_stats.peak_level = abs_sample;
288 }
289
290 // Track clipping (samples > 1.0) - indicates distortion
291 if (abs_sample > 1.0f) {
292 g_received_stats.clipping_count++;
293 g_received_clipping_samples++;
294 }
295
296 // Detect sharp transitions (sudden amplitude jumps > 0.3) - indicates clicks/pops/artifacts
297 float amp_change = fabsf(sample - g_received_last_sample);
298 if (amp_change > 0.3f) {
299 g_received_sharp_transitions++;
300 }
301 g_received_transition_samples++;
302
303 // Accumulate for mean calculation
304 g_received_mean += sample;
305
306 // Detect zero crossings (waveform crossing zero) - indicates spectral content
307 // Use file-scope static variable for prev sample tracking
308 // (This function is called from the protocol reception thread, separate from the
309 // audio capture thread, so using distinct static variables is safe)
310 static float s_received_prev_sample_for_zero_crossing = 0.0f;
311 if ((s_received_prev_sample_for_zero_crossing > 0 && sample < 0) ||
312 (s_received_prev_sample_for_zero_crossing < 0 && sample > 0)) {
313 g_received_zero_crossings++;
314 }
315 s_received_prev_sample_for_zero_crossing = sample;
316
317 // Track silence and low-energy audio
318 if (abs_sample < 0.001f) {
319 g_received_stats.silent_samples++;
320 g_received_silence_burst++;
321 g_received_below_noise_floor++;
322
323 // Track when silence started
324 if (g_received_silence_burst == 1) {
325 g_received_silence_start_sample = g_received_stats.total_samples;
326 }
327 } else {
328 // Silence ended - track gap interval and max burst length
329 if (g_received_silence_burst > 0) {
330 // Calculate time gap between end of last silence and start of this one
331 if (g_received_last_silence_end_sample > 0) {
332 uint64_t samples_between = g_received_silence_start_sample - g_received_last_silence_end_sample;
333 uint32_t ms_between = (uint32_t)(samples_between * 1000 / 48000); // Convert samples to ms at 48kHz
334
335 // Track the gap interval if we have room
336 if (g_received_gap_count < MAX_GAP_SAMPLES) {
337 g_received_gap_intervals_ms[g_received_gap_count++] = ms_between;
338 }
339 }
340
341 g_received_last_silence_end_sample = g_received_stats.total_samples;
342
343 // Track max burst length
344 if (g_received_silence_burst > g_received_max_silence_burst) {
345 g_received_max_silence_burst = g_received_silence_burst;
346 }
347 }
348 g_received_silence_burst = 0;
349 }
350
351 // Track very quiet audio (< 0.05 amplitude) which contributes to muddy/quiet perception
352 if (abs_sample < 0.05f) {
353 g_received_low_energy_samples++;
354 }
355
356 // Detect jitter: rapid amplitude changes > 0.5 between consecutive samples
357 float delta = fabsf(sample - g_received_last_sample);
358 if (delta > 0.5f) {
359 g_received_stats.jitter_count++;
360 }
361 g_received_last_sample = sample;
362
363 // Accumulate for RMS calculation
364 g_received_rms_accumulator += sample * sample;
365 g_received_rms_sample_count++;
366
367 // Echo detection: check if received sample matches sent sample from N ms ago
368 // This detects if echo cancellation is working (it shouldn't find matches)
369 if (g_echo_correlation_sample_count < 500000) { // Limit to first ~10 seconds
370 for (int delay_idx = 0; delay_idx < ECHO_DELAY_COUNT; delay_idx++) {
371 // Calculate sample delay: delay_ms * (sample_rate / 1000)
372 uint32_t delay_samples = (g_echo_delays_ms[delay_idx] * 48000) / 1000;
373
374 // Get sent sample from that delay ago (from circular buffer)
375 uint64_t sent_pos;
376 if (g_echo_buffer_pos >= delay_samples) {
377 sent_pos = g_echo_buffer_pos - delay_samples;
378 } else {
379 sent_pos = (g_echo_buffer_pos + ECHO_BUFFER_SIZE) - delay_samples;
380 }
381
382 float sent_sample = g_echo_buffer[sent_pos];
383
384 // Check if samples match (correlation threshold = 0.1)
385 float diff = fabsf(sample - sent_sample);
386 if (diff < 0.1f && fabsf(sent_sample) > 0.01f) { // Only count if sent is not silence
387 g_echo_match_count[delay_idx]++;
388 g_echo_correlation_strength[delay_idx] += (0.1f - diff); // Accumulate strength
389 }
390 }
391 g_echo_correlation_sample_count++;
392 }
393
394 // Beep/tone artifact detection
395 // Store sample in sliding window for frequency analysis
396 g_received_beep_window[g_received_beep_window_idx] = sample;
397 g_received_beep_window_idx = (g_received_beep_window_idx + 1) % BEEP_WINDOW_SIZE;
398
399 // Analyze window every 10ms (480 samples at 48kHz)
400 if (g_received_beep_window_idx == 0 && g_received_stats.total_samples > BEEP_WINDOW_SIZE) {
401 // Calculate zero-crossing rate in this window
402 int zero_crossings = 0;
403 float min_amp = 1.0f, max_amp = 0.0f;
404 float sum_amp = 0.0f;
405 float prev = g_received_beep_window[0];
406
407 for (int i = 1; i < BEEP_WINDOW_SIZE; i++) {
408 float curr = g_received_beep_window[i];
409 float abs_curr = fabsf(curr);
410
411 // Track amplitude range
412 if (abs_curr > max_amp)
413 max_amp = abs_curr;
414 if (abs_curr < min_amp)
415 min_amp = abs_curr;
416 sum_amp += abs_curr;
417
418 // Count zero crossings
419 if ((prev > 0 && curr < 0) || (prev < 0 && curr > 0)) {
420 zero_crossings++;
421 }
422 prev = curr;
423 }
424
425 float avg_amp = sum_amp / BEEP_WINDOW_SIZE;
426 float amp_range = max_amp - min_amp;
427
428 // A beep/tone has:
429 // 1. High zero-crossing rate (>20 per 10ms = >2000Hz equivalent, or 5-20 = 500-2000Hz)
430 // 2. Consistent amplitude (range/avg < 0.5 means sine-wave like)
431 // 3. Non-trivial amplitude (avg > 0.02)
432 bool is_tonal = (zero_crossings >= 5 && zero_crossings <= 100) && // 500Hz-10kHz range
433 (avg_amp > 0.02f) && // Not silence
434 (amp_range < avg_amp * 1.5f); // Relatively consistent amplitude
435
436 if (is_tonal) {
437 g_received_tonal_samples += BEEP_WINDOW_SIZE;
438
439 if (!g_in_beep_burst) {
440 // Starting a new beep burst
441 g_in_beep_burst = true;
442 g_beep_burst_samples = BEEP_WINDOW_SIZE;
443 } else {
444 g_beep_burst_samples += BEEP_WINDOW_SIZE;
445 }
446 } else {
447 if (g_in_beep_burst) {
448 // Beep burst ended
449 // Only count as beep event if it was short (< 500ms = 24000 samples)
450 // Long tonal sounds are likely music, not artifacts
451 if (g_beep_burst_samples > 0 && g_beep_burst_samples < 24000) {
452 g_received_beep_events++;
453 g_received_stats.beep_events = g_received_beep_events;
454 }
455 g_in_beep_burst = false;
456 g_beep_burst_samples = 0;
457 }
458 }
459
460 g_received_stats.tonal_samples = g_received_tonal_samples;
461 }
462
463 // Write to WAV file if enabled
464 if (g_received_wav) {
465 wav_writer_write(g_received_wav, &sample, 1);
466 }
467}
#define MAX_GAP_SAMPLES
Definition analysis.c:69
#define ECHO_BUFFER_SIZE
Definition analysis.c:83
#define BEEP_WINDOW_SIZE
Definition analysis.c:106
int wav_writer_write(wav_writer_t *writer, const float *samples, int num_samples)
Write audio samples to WAV file.
Definition wav_writer.c:94

References audio_analysis_stats_t::beep_events, BEEP_WINDOW_SIZE, audio_analysis_stats_t::clipping_count, ECHO_BUFFER_SIZE, ECHO_DELAY_COUNT, audio_analysis_stats_t::jitter_count, MAX_GAP_SAMPLES, audio_analysis_stats_t::peak_level, audio_analysis_stats_t::silent_samples, audio_analysis_stats_t::tonal_samples, audio_analysis_stats_t::total_samples, and wav_writer_write().

Referenced by audio_process_received_samples().

◆ audio_analysis_track_sent_packet()

void audio_analysis_track_sent_packet ( size_t  size)

Track sent packet.

Parameters
sizePacket size in bytes

Definition at line 251 of file analysis.c.

251 {
252 (void)size; // Unused parameter - reserved for future per-packet analysis
253 if (!g_analysis_enabled)
254 return;
255
256 int64_t now_us = (int64_t)time_ns_to_us(time_get_ns());
257
258 // Detect gaps between consecutive packets (discontinuity)
259 if (g_sent_stats.packets_count > 0) {
260 int64_t gap_us = now_us - g_sent_last_packet_time_us;
261 int32_t gap_ms = (int32_t)(gap_us / 1000);
262
263 // Expected: ~20ms per Opus frame, flag if gap > 100ms
264 if (gap_ms > 100) {
265 g_sent_stats.discontinuity_count++;
266 }
267
268 // Track max gap
269 if (gap_ms > (int32_t)g_sent_stats.max_gap_ns) {
270 g_sent_stats.max_gap_ns = (uint32_t)gap_ms;
271 }
272 }
273
274 g_sent_last_packet_time_us = now_us;
275 g_sent_stats.packets_count++;
276}

References audio_analysis_stats_t::discontinuity_count, audio_analysis_stats_t::max_gap_ns, audio_analysis_stats_t::packets_count, and time_get_ns().

◆ audio_analysis_track_sent_sample()

void audio_analysis_track_sent_sample ( float  sample)

Track sent audio sample.

Parameters
sampleAudio sample value

Definition at line 179 of file analysis.c.

179 {
180 if (!g_analysis_enabled)
181 return;
182
183 g_sent_stats.total_samples++;
184
185 // Track peak level
186 float abs_sample = fabsf(sample);
187 if (abs_sample > g_sent_stats.peak_level) {
188 g_sent_stats.peak_level = abs_sample;
189 }
190
191 // Track clipping (samples > 1.0) - indicates distortion
192 if (abs_sample > 1.0f) {
193 g_sent_stats.clipping_count++;
194 g_sent_clipping_samples++;
195 }
196
197 // Detect sharp transitions (sudden amplitude jumps > 0.3) - indicates clicks/pops
198 float amp_change = fabsf(sample - g_sent_last_sample);
199 if (amp_change > 0.3f) {
200 g_sent_sharp_transitions++;
201 }
202 g_sent_transition_samples++;
203
204 // Accumulate for mean calculation
205 g_sent_mean += sample;
206
207 // Detect zero crossings (waveform crossing zero) - indicates spectral content
208 // Use file-scope static variable for prev sample tracking
209 // (This function is only called from the audio capture thread, but using file-scope
210 // static is clearer and avoids shadowing the existing g_sent_last_sample variable)
211 static float s_sent_prev_sample_for_zero_crossing = 0.0f;
212 if ((s_sent_prev_sample_for_zero_crossing > 0 && sample < 0) ||
213 (s_sent_prev_sample_for_zero_crossing < 0 && sample > 0)) {
214 g_sent_zero_crossings++;
215 }
216 s_sent_prev_sample_for_zero_crossing = sample;
217
218 // Track silence (very low level)
219 if (abs_sample < 0.001f) {
220 g_sent_stats.silent_samples++;
221 g_sent_silence_burst++;
222 } else {
223 // Silence ended - track max burst length
224 if (g_sent_silence_burst > g_sent_max_silence_burst) {
225 g_sent_max_silence_burst = g_sent_silence_burst;
226 }
227 g_sent_silence_burst = 0;
228 }
229
230 // Detect jitter: rapid amplitude changes > 0.5 between consecutive samples
231 float delta = fabsf(sample - g_sent_last_sample);
232 if (delta > 0.5f) {
233 g_sent_stats.jitter_count++;
234 }
235 g_sent_last_sample = sample;
236
237 // Accumulate for RMS calculation
238 g_sent_rms_accumulator += sample * sample;
239 g_sent_rms_sample_count++;
240
241 // Write to WAV file if enabled
242 if (g_sent_wav) {
243 wav_writer_write(g_sent_wav, &sample, 1);
244 }
245
246 // Store in echo detection buffer (circular)
247 g_echo_buffer[g_echo_buffer_pos] = sample;
248 g_echo_buffer_pos = (g_echo_buffer_pos + 1) % ECHO_BUFFER_SIZE;
249}

References audio_analysis_stats_t::clipping_count, ECHO_BUFFER_SIZE, audio_analysis_stats_t::jitter_count, audio_analysis_stats_t::peak_level, audio_analysis_stats_t::silent_samples, audio_analysis_stats_t::total_samples, and wav_writer_write().