ascii-chat 0.11.33
Video chat in your terminal
Loading...
Searching...
No Matches
ffmpeg_encoder.c
Go to the documentation of this file.
1
9#include <ascii-chat/log/io.h>
12#include <string.h>
13
14#include <libavformat/avformat.h>
15#include <libavcodec/avcodec.h>
16#include <libavutil/opt.h>
17#include <libswscale/swscale.h>
18#include <libswresample/swresample.h>
19
21 // Video stream
22 AVCodecContext *codec_ctx;
23 AVStream *stream;
24 AVFrame *frame;
25 AVFrame *frame_encoded; // Frame in target pixel format for encoding
26 struct SwsContext *sws_ctx;
27
28 // Audio stream
29 AVStream *audio_stream;
30 AVCodecContext *audio_codec_ctx;
31 struct SwrContext *audio_swr_ctx;
32 AVFrame *audio_frame;
33 int64_t audio_pts;
34 int audio_frame_size; // Codec's required samples per frame
35 float *audio_partial_buf; // Accumulator for partial frames
37
38 // Shared
39 AVFormatContext *fmt_ctx;
40 AVPacket *pkt;
46 int64_t video_pts; // Cumulative PTS for proper frame timestamps
47 int fps; // Frames per second (for duration calculation)
48 int is_image; // Single-frame format (PNG, JPG)
49 enum AVPixelFormat target_pix_fmt; // RGB24 for images, YUV420P for video
50 int has_audio_stream; // Whether audio stream was created
51 int is_stdout_pipe; // Whether writing to stdout (non-seekable)
52 uint64_t previous_captured_ns; // Previous frame's capture timestamp (for snapshot mode)
53 double snapshot_actual_duration; // Actual wall-clock duration in snapshot mode (seconds)
54 uint64_t snapshot_elapsed_ns; // Current elapsed time in snapshot mode for dynamic frame duration
55
56 // Frame timing from capture timestamps
57 uint64_t first_frame_captured_ns; // Wall-clock timestamp of first frame (0 = not set)
58
59 // Snapshot mode frame distribution
60 int estimated_frame_count; // Expected frame count for snapshot mode (fps * snapshot_delay)
61 int64_t estimated_frame_duration; // Duration each frame should have in stream time_base units
62};
63
64// Determine audio codec from file extension
65static void get_audio_codec_from_extension(const char *path, const char **audio_codec, enum AVSampleFormat *sample_fmt,
66 int *has_audio) {
67 const char *dot = strrchr(path, '.');
68 *has_audio = 0;
69 *audio_codec = NULL;
70 *sample_fmt = AV_SAMPLE_FMT_NONE;
71
72 if (!dot) {
73 // Use Opus for stdout (pipe/non-seekable output, part of WebM), otherwise default to AAC
74 if (path && strcmp(path, "-") == 0) {
75 *has_audio = 1;
76 *audio_codec = "libopus";
77 *sample_fmt = AV_SAMPLE_FMT_FLT;
78 } else {
79 *has_audio = 1;
80 *audio_codec = "aac";
81 *sample_fmt = AV_SAMPLE_FMT_FLTP;
82 }
83 return;
84 }
85
86 const char *ext = dot + 1;
87 // Video formats with audio support
88 // NOLINTNEXTLINE(bugprone-branch-clone) - Each format requires distinct codec configuration
89 if (strcmp(ext, "mp4") == 0 || strcmp(ext, "mov") == 0) {
90 *has_audio = 1;
91 *audio_codec = "aac";
92 *sample_fmt = AV_SAMPLE_FMT_FLTP;
93 } else if (strcmp(ext, "mkv") == 0 || strcmp(ext, "mka") == 0) {
94 *has_audio = 1;
95 *audio_codec = "aac"; // Matroska supports both AAC and Opus
96 *sample_fmt = AV_SAMPLE_FMT_FLTP;
97 } else if (strcmp(ext, "webm") == 0) {
98 *has_audio = 1;
99 *audio_codec = "libopus";
100 *sample_fmt = AV_SAMPLE_FMT_FLT;
101 } else if (strcmp(ext, "avi") == 0) {
102 *has_audio = 1;
103 *audio_codec = "pcm_s16le";
104 *sample_fmt = AV_SAMPLE_FMT_S16;
105 } else if (strcmp(ext, "flv") == 0) {
106 *has_audio = 1;
107 *audio_codec = "aac";
108 *sample_fmt = AV_SAMPLE_FMT_FLTP;
109 } else if (strcmp(ext, "ogv") == 0 || strcmp(ext, "ogg") == 0) {
110 *has_audio = 1;
111 *audio_codec = "libvorbis";
112 *sample_fmt = AV_SAMPLE_FMT_FLT;
113 } else if (strcmp(ext, "3gp") == 0) {
114 *has_audio = 1;
115 *audio_codec = "aac";
116 *sample_fmt = AV_SAMPLE_FMT_FLTP;
117 } else if (strcmp(ext, "gif") == 0 || strcmp(ext, "png") == 0 || strcmp(ext, "jpg") == 0 ||
118 strcmp(ext, "jpeg") == 0) {
119 *has_audio = 0;
120 *audio_codec = NULL;
121 *sample_fmt = AV_SAMPLE_FMT_NONE;
122 } else {
123 // Default for unknown formats
124 *has_audio = 1;
125 *audio_codec = "aac";
126 *sample_fmt = AV_SAMPLE_FMT_FLTP;
127 }
128}
129
130// Determine codec, format, and pixel format from file extension
131static void get_codec_from_extension(const char *path, const char **codec, const char **format, int *is_image,
132 enum AVPixelFormat *pix_fmt) {
133 const char *dot = strrchr(path, '.');
134 *is_image = 0;
135 *pix_fmt = AV_PIX_FMT_YUV420P; // Default
136
137 if (!dot) {
138 // Default to MP4 for any output without extension (including stdout piping)
139 *codec = "libx264";
140 *format = "mp4";
141 return;
142 }
143
144 const char *ext = dot + 1;
145 // Video formats
146 // NOLINTNEXTLINE(bugprone-branch-clone) - Each format requires distinct codec setup
147 if (strcmp(ext, "mp4") == 0 || strcmp(ext, "mov") == 0) {
148 *codec = "libx264";
149 *format = "mp4";
150 *pix_fmt = AV_PIX_FMT_YUV420P;
151 } // NOLINTNEXTLINE(bugprone-branch-clone)
152 else if (strcmp(ext, "mkv") == 0) {
153 *codec = "libx264";
154 *format = "matroska";
155 *pix_fmt = AV_PIX_FMT_YUV420P;
156 } // NOLINTNEXTLINE(bugprone-branch-clone)
157 else if (strcmp(ext, "mka") == 0) {
158 // MKA is audio-only Matroska
159 *codec = "libx264";
160 *format = "matroska";
161 *pix_fmt = AV_PIX_FMT_YUV420P;
162 } else if (strcmp(ext, "webm") == 0) {
163 *codec = "libvpx-vp9";
164 *format = "webm";
165 *pix_fmt = AV_PIX_FMT_YUV420P;
166 } else if (strcmp(ext, "avi") == 0) {
167 *codec = "mpeg4";
168 *format = "avi";
169 *pix_fmt = AV_PIX_FMT_YUV420P;
170 } else if (strcmp(ext, "flv") == 0) {
171 *codec = "libx264";
172 *format = "flv";
173 *pix_fmt = AV_PIX_FMT_YUV420P;
174 } else if (strcmp(ext, "ogv") == 0 || strcmp(ext, "ogg") == 0) {
175 *codec = "libtheora";
176 *format = "ogg";
177 *pix_fmt = AV_PIX_FMT_YUV420P;
178 } else if (strcmp(ext, "3gp") == 0) {
179 *codec = "libx264";
180 *format = "3gp";
181 *pix_fmt = AV_PIX_FMT_YUV420P;
182 } else if (strcmp(ext, "gif") == 0) {
183 *codec = "gif";
184 *format = "gif";
185 *pix_fmt = AV_PIX_FMT_RGB8;
186 } else if (strcmp(ext, "png") == 0) {
187 *codec = "png";
188 *format = "image2";
189 *is_image = 1;
190 *pix_fmt = AV_PIX_FMT_RGB24; // PNG uses RGB24
191 } else if (strcmp(ext, "jpg") == 0 || strcmp(ext, "jpeg") == 0) {
192 *codec = "mjpeg";
193 *format = "image2";
194 *is_image = 1;
195 *pix_fmt = AV_PIX_FMT_YUVJ420P; // MJPEG uses YUVJ420P
196 } else {
197 // Default for unknown formats
198 *codec = "libx264";
199 *format = "mp4";
200 *pix_fmt = AV_PIX_FMT_YUV420P;
201 }
202}
203
204// Initialize audio stream for encoder
205static asciichat_error_t encoder_init_audio_stream(ffmpeg_encoder_t *enc, const char *output_path) {
206 const char *audio_codec_name = NULL;
207 enum AVSampleFormat target_sample_fmt = AV_SAMPLE_FMT_NONE;
208 int has_audio = 0;
209
210 get_audio_codec_from_extension(output_path, &audio_codec_name, &target_sample_fmt, &has_audio);
211
212 if (!has_audio) {
213 log_debug("encoder_init_audio_stream: no audio for format");
214 enc->has_audio_stream = 0;
215 return ASCIICHAT_OK;
216 }
217
218 // Find audio encoder
219 const AVCodec *audio_codec = avcodec_find_encoder_by_name(audio_codec_name);
220 if (!audio_codec) {
221 log_warn("encoder_init_audio_stream: audio encoder '%s' not found, skipping audio", audio_codec_name);
222 enc->has_audio_stream = 0;
223 return ASCIICHAT_OK;
224 }
225
226 // Create audio stream
227 enc->audio_stream = avformat_new_stream(enc->fmt_ctx, audio_codec);
228 if (!enc->audio_stream) {
229 log_warn("encoder_init_audio_stream: avformat_new_stream for audio failed");
230 enc->has_audio_stream = 0;
231 return ASCIICHAT_OK;
232 }
233
234 // Allocate audio codec context
235 enc->audio_codec_ctx = avcodec_alloc_context3(audio_codec);
236 if (!enc->audio_codec_ctx) {
237 log_warn("encoder_init_audio_stream: avcodec_alloc_context3 for audio failed");
238 enc->has_audio_stream = 0;
239 return ASCIICHAT_OK;
240 }
241
242 // Configure audio codec context
243 enc->audio_codec_ctx->sample_rate = 48000; // 48kHz (from audio pipeline)
244 AVChannelLayout ch_layout = AV_CHANNEL_LAYOUT_MONO;
245 av_channel_layout_copy(&enc->audio_codec_ctx->ch_layout, &ch_layout);
246 enc->audio_codec_ctx->sample_fmt = target_sample_fmt;
247 enc->audio_codec_ctx->time_base = (AVRational){1, 48000};
248 enc->audio_codec_ctx->bit_rate = 128000; // 128 kbps
249
250 // Open audio codec (capture FFmpeg logs)
251 int ret = 0;
252 LOG_IO("ffmpeg", { ret = avcodec_open2(enc->audio_codec_ctx, audio_codec, NULL); });
253 if (ret < 0) {
254 log_warn("encoder_init_audio_stream: avcodec_open2 for audio failed");
255 avcodec_free_context(&enc->audio_codec_ctx);
256 enc->has_audio_stream = 0;
257 return ASCIICHAT_OK;
258 }
259
260 // Copy codec parameters to stream
261 ret = avcodec_parameters_from_context(enc->audio_stream->codecpar, enc->audio_codec_ctx);
262 if (ret < 0) {
263 log_warn("encoder_init_audio_stream: avcodec_parameters_from_context for audio failed");
264 avcodec_free_context(&enc->audio_codec_ctx);
265 enc->has_audio_stream = 0;
266 return ASCIICHAT_OK;
267 }
268
269 // Create audio frame
270 enc->audio_frame = av_frame_alloc();
271 if (!enc->audio_frame) {
272 log_warn("encoder_init_audio_stream: av_frame_alloc for audio failed");
273 avcodec_free_context(&enc->audio_codec_ctx);
274 enc->has_audio_stream = 0;
275 return ASCIICHAT_OK;
276 }
277
278 enc->audio_frame->sample_rate = 48000;
279 AVChannelLayout frame_ch_layout = AV_CHANNEL_LAYOUT_MONO;
280 av_channel_layout_copy(&enc->audio_frame->ch_layout, &frame_ch_layout);
281 enc->audio_frame->format = target_sample_fmt;
282
283 // Get frame size from codec (if specified)
284 enc->audio_frame_size = enc->audio_codec_ctx->frame_size;
285 if (enc->audio_frame_size == 0) {
286 // Default frame size for codecs that don't specify
287 enc->audio_frame_size = 1024;
288 }
289
290 // Allocate buffer for partial frames (accumulator)
291 enc->audio_partial_buf = SAFE_MALLOC(enc->audio_frame_size * sizeof(float), float *);
292 enc->audio_partial_len = 0;
293
294 // Create resampler from input format (AV_SAMPLE_FMT_FLT mono 48kHz) to target format
295 enc->audio_swr_ctx = swr_alloc();
296 if (!enc->audio_swr_ctx) {
297 log_warn("encoder_init_audio_stream: swr_alloc failed");
299 av_frame_free(&enc->audio_frame);
300 avcodec_free_context(&enc->audio_codec_ctx);
301 enc->has_audio_stream = 0;
302 return ASCIICHAT_OK;
303 }
304
305 // Set input format (mono 48kHz float32)
306 AVChannelLayout in_ch_layout = AV_CHANNEL_LAYOUT_MONO;
307 av_opt_set_chlayout(enc->audio_swr_ctx, "in_chlayout", &in_ch_layout, 0);
308 av_opt_set_int(enc->audio_swr_ctx, "in_sample_rate", 48000, 0);
309 av_opt_set_sample_fmt(enc->audio_swr_ctx, "in_sample_fmt", AV_SAMPLE_FMT_FLT, 0);
310
311 // Set output format
312 AVChannelLayout out_ch_layout = AV_CHANNEL_LAYOUT_MONO;
313 av_opt_set_chlayout(enc->audio_swr_ctx, "out_chlayout", &out_ch_layout, 0);
314 av_opt_set_int(enc->audio_swr_ctx, "out_sample_rate", 48000, 0);
315 av_opt_set_sample_fmt(enc->audio_swr_ctx, "out_sample_fmt", target_sample_fmt, 0);
316
317 // Initialize the resampler
318 ret = swr_init(enc->audio_swr_ctx);
319 if (ret < 0) {
320 log_warn("encoder_init_audio_stream: swr_init failed");
321 swr_free(&enc->audio_swr_ctx);
323 av_frame_free(&enc->audio_frame);
324 avcodec_free_context(&enc->audio_codec_ctx);
325 enc->has_audio_stream = 0;
326 return ASCIICHAT_OK;
327 }
328
329 enc->audio_pts = 0;
330 enc->has_audio_stream = 1;
331 log_debug("encoder_init_audio_stream: audio stream initialized with codec=%s, frame_size=%d", audio_codec_name,
332 enc->audio_frame_size);
333
334 return ASCIICHAT_OK;
335}
336
337asciichat_error_t ffmpeg_encoder_create(const char *output_path, int width_px, int height_px, int fps,
338 ffmpeg_encoder_t **out) {
339 log_info("[FFMPEG_ENCODER_CREATE] START: output=%s, %dx%d @ %dfps", output_path, width_px, height_px, fps);
340
341 if (!output_path || !out || width_px <= 0 || height_px <= 0)
342 return SET_ERRNO(ERROR_INVALID_PARAM, "ffmpeg_encoder_create: invalid parameters");
343
344 // Use default FPS if not specified (0 from config means "use default")
345 if (fps <= 0) {
346 fps = 60;
347 log_debug("ffmpeg_encoder_create: fps was 0, using default 60");
348 }
349
350 ffmpeg_encoder_t *enc = SAFE_CALLOC(1, sizeof(*enc), ffmpeg_encoder_t *);
351 enc->width_px = width_px;
352 enc->height_px = height_px;
353
354 // For snapshot mode, calculate output FPS based on input FPS and snapshot duration
355 // If we capture for N seconds at input_fps, we'll get ~(N * input_fps) frames
356 // Output duration should be snapshot_delay seconds
357 // So output_fps = (input_fps * snapshot_delay) / snapshot_delay = input_fps
358 // EXCEPT: if the input was probed, fps parameter is the actual input fps
359 // Then output_fps should equal input_fps to make frame durations sum correctly
360 bool snapshot_mode = GET_OPTION(snapshot_mode);
361 double snapshot_delay = GET_OPTION(snapshot_delay);
362 log_debug("ffmpeg_encoder_create: snapshot_mode=%d, snapshot_delay=%.1f, input_fps=%d", snapshot_mode, snapshot_delay,
363 fps);
364
365 if (snapshot_mode && snapshot_delay > 0) {
366 // For snapshot mode: if we capture at input_fps for snapshot_delay seconds,
367 // we get approximately (input_fps * snapshot_delay) frames
368 // To make output span exactly snapshot_delay seconds:
369 // output_fps = frame_count / snapshot_delay = (input_fps * snapshot_delay) / snapshot_delay = input_fps
370 // So we should use input_fps as output_fps!
371 // Example: 30fps input for 3s → ~90 frames → output at 30fps → 90/30 = 3 seconds ✓
372
373 log_info("ffmpeg_encoder_create: Snapshot mode snapshot_delay=%.1f, input_fps=%d", snapshot_delay, fps);
374 log_info(" → Will estimate ~%.0f frames, output FPS=%d makes durations sum to %.1f seconds", fps * snapshot_delay,
375 fps, snapshot_delay);
376 }
377
378 enc->fps = fps;
379
380 // Snapshot mode frame distribution setup
381 enc->estimated_frame_count = 0;
383 if (snapshot_mode && snapshot_delay > 0 && fps > 0) {
384 // Estimate how many frames we'll capture
385 enc->estimated_frame_count = (int)(fps * snapshot_delay + 0.5);
386 // Each frame should have uniform duration in stream time_base (1/90000 for MP4)
387 // frame_duration = total_duration / estimated_frame_count
388 // = snapshot_delay seconds * time_base.den / estimated_frame_count
389 // We'll set this to stream->time_base later after it's initialized
390 log_info("ffmpeg_encoder_create: Snapshot mode ENABLED - estimated_frame_count=%d (fps=%d * snapshot_delay=%.2f)",
391 enc->estimated_frame_count, fps, snapshot_delay);
392 } else {
393 log_info("ffmpeg_encoder_create: Snapshot mode disabled or invalid - snapshot_mode=%d, snapshot_delay=%.2f, fps=%d",
394 snapshot_mode, snapshot_delay, fps);
395 }
396 enc->is_stdout_pipe = (output_path && strcmp(output_path, "-") == 0) ? 1 : 0;
397
398 // For codec/format detection, use output_path unless it's "-" (stdout)
399 // For stdout, use "fake.mp4" to default to MP4 format
400 const char *detection_path = output_path;
401 if (output_path && strcmp(output_path, "-") == 0) {
402 detection_path = "fake.mp4"; // Fake path for format detection when piping to stdout
403 log_debug("ffmpeg_encoder_create: Using MP4 format for stdout piping");
404 }
405
406 const char *codec_name = NULL, *format_name = NULL;
407 get_codec_from_extension(detection_path, &codec_name, &format_name, &enc->is_image, &enc->target_pix_fmt);
408
409 log_debug("ffmpeg_encoder_create: %s (%s, %dx%d @ %dfps, %s)", output_path, codec_name, width_px, height_px, fps,
410 enc->is_image ? "image" : "video");
411
412 // Allocate output media context
413 int ret = avformat_alloc_output_context2(&enc->fmt_ctx, NULL, format_name, output_path);
414 if (ret < 0) {
415 SAFE_FREE(enc);
416 return SET_ERRNO(ERROR_INIT, "ffmpeg: avformat_alloc_output_context2 failed");
417 }
418
419 // Find encoder (with fallback for FFmpeg builds without libx264)
420 const AVCodec *codec = avcodec_find_encoder_by_name(codec_name);
421 if (!codec && strcmp(codec_name, "libx264") == 0) {
422 // libx264 not available - try built-in H.264 encoders
423 static const char *h264_fallbacks[] = {"libopenh264", "h264_mf", NULL};
424 for (int i = 0; h264_fallbacks[i] && !codec; i++) {
425 codec = avcodec_find_encoder_by_name(h264_fallbacks[i]);
426 }
427 if (!codec) {
428 // Last resort: ask FFmpeg for any H.264 encoder
429 codec = avcodec_find_encoder(AV_CODEC_ID_H264);
430 }
431 if (codec) {
432 log_info("ffmpeg: using fallback H.264 encoder '%s' (libx264 not available)", codec->name);
433 }
434 }
435 if (!codec) {
436 avformat_free_context(enc->fmt_ctx);
437 SAFE_FREE(enc);
438 return SET_ERRNO(ERROR_INIT, "ffmpeg: encoder '%s' not found", codec_name);
439 }
440
441 // Create new stream
442 enc->stream = avformat_new_stream(enc->fmt_ctx, codec);
443 if (!enc->stream) {
444 avformat_free_context(enc->fmt_ctx);
445 SAFE_FREE(enc);
446 return SET_ERRNO(ERROR_INIT, "ffmpeg: avformat_new_stream failed");
447 }
448
449 // Allocate codec context
450 enc->codec_ctx = avcodec_alloc_context3(codec);
451 if (!enc->codec_ctx) {
452 avformat_free_context(enc->fmt_ctx);
453 SAFE_FREE(enc);
454 return SET_ERRNO(ERROR_INIT, "ffmpeg: avcodec_alloc_context3 failed");
455 }
456
457 // Set codec parameters
458 enc->codec_ctx->width = width_px;
459 enc->codec_ctx->height = height_px;
460 enc->codec_ctx->pix_fmt = enc->target_pix_fmt;
461 enc->codec_ctx->time_base = (AVRational){1, fps};
462 enc->codec_ctx->framerate = (AVRational){fps, 1};
463
464 // Set bitrate for high quality (music-level quality)
465 // Use ~5-10 Mbps per megapixel for visually lossless quality
466 int bitrate = (width_px * height_px * 10) / 1024; // higher quality estimate in kbps
467 if (bitrate < 4000)
468 bitrate = 4000; // Minimum 4000 kbps for quality
469 if (bitrate > 50000)
470 bitrate = 50000; // Cap at 50 Mbps max
471 enc->codec_ctx->bit_rate = bitrate * 1000;
472 log_debug("ffmpeg_encoder: bitrate set to %d kbps for %dx%d", bitrate, width_px, height_px);
473
474 // Configure codec-specific options for quality
475 AVDictionary *codec_opts = NULL;
476 if (strcmp(codec_name, "libx264") == 0) {
477 av_dict_set(&codec_opts, "preset", "ultrafast", 0);
478 av_dict_set(&codec_opts, "crf", "28", 0);
479 log_debug("ffmpeg_encoder: x264 preset=ultrafast crf=28");
480 } else if (strcmp(codec_name, "libx265") == 0) {
481 av_dict_set(&codec_opts, "preset", "ultrafast", 0);
482 av_dict_set(&codec_opts, "crf", "32", 0);
483 // x265 tag must be hvc1 for MP4 container compatibility (Apple/browser playback)
484 enc->stream->codecpar->codec_tag = MKTAG('h', 'v', 'c', '1');
485 log_debug("ffmpeg_encoder: x265 preset=ultrafast crf=32");
486 }
487
488 // Open codec (capture libx264 startup logs)
489 LOG_IO("ffmpeg", { ret = avcodec_open2(enc->codec_ctx, codec, &codec_opts); });
490 av_dict_free(&codec_opts);
491 if (ret < 0) {
492 avcodec_free_context(&enc->codec_ctx);
493 avformat_free_context(enc->fmt_ctx);
494 SAFE_FREE(enc);
495 return SET_ERRNO(ERROR_INIT, "ffmpeg: avcodec_open2 failed");
496 }
497
498 // Copy codec parameters to stream
499 ret = avcodec_parameters_from_context(enc->stream->codecpar, enc->codec_ctx);
500 if (ret < 0) {
501 avcodec_free_context(&enc->codec_ctx);
502 avformat_free_context(enc->fmt_ctx);
503 SAFE_FREE(enc);
504 return SET_ERRNO(ERROR_INIT, "ffmpeg: avcodec_parameters_from_context failed");
505 }
506
507 // Muxer options for proper MP4 container structure
508 AVDictionary *opts = NULL;
509
510 // Muxer flags configuration
511 // For stdout/non-seekable output, use fragmented MP4 (frag_keyframe)
512 // For regular files, use faststart (moov at front for streaming)
513 bool is_stdout = (output_path && strcmp(output_path, "-") == 0);
514 if (is_stdout) {
515 // Use fragmented MP4 for stdout - starts a new fragment at each video keyframe
516 // This allows streaming without requiring file seeking (FFmpeg docs recommended)
517 av_dict_set(&opts, "movflags", "+frag_keyframe", 0);
518 log_debug("ffmpeg_encoder_create: Using fragmented MP4 (+frag_keyframe) for stdout output");
519 } else {
520 // Move moov atom to the front (faststart) - helps with streaming and player compatibility
521 av_dict_set(&opts, "movflags", "faststart", 0);
522 }
523
524 if (enc->is_image) {
525 // Use -update flag for single frame output
526 av_dict_set(&opts, "update", "1", 0);
527 }
528
529 // Allocate frames
530 enc->frame = av_frame_alloc();
531 enc->frame_encoded = av_frame_alloc();
532 enc->pkt = av_packet_alloc();
533 if (!enc->frame || !enc->frame_encoded || !enc->pkt) {
534 av_frame_free(&enc->frame);
535 av_frame_free(&enc->frame_encoded);
536 av_packet_free(&enc->pkt);
537 avcodec_free_context(&enc->codec_ctx);
538 avformat_free_context(enc->fmt_ctx);
539 SAFE_FREE(enc);
540 return SET_ERRNO(ERROR_INIT, "ffmpeg: frame allocation failed");
541 }
542
543 // Set frame properties for target pixel format
544 enc->frame_encoded->format = enc->target_pix_fmt;
545 enc->frame_encoded->width = width_px;
546 enc->frame_encoded->height = height_px;
547 ret = av_frame_get_buffer(enc->frame_encoded, 32);
548 if (ret < 0) {
549 av_frame_free(&enc->frame);
550 av_frame_free(&enc->frame_encoded);
551 av_packet_free(&enc->pkt);
552 avcodec_free_context(&enc->codec_ctx);
553 avformat_free_context(enc->fmt_ctx);
554 SAFE_FREE(enc);
555 return SET_ERRNO(ERROR_INIT, "ffmpeg: av_frame_get_buffer failed");
556 }
557
558 // Create SWS context for RGBA → target format conversion
559 LOG_IO("swscaler", {
560 enc->sws_ctx = sws_getContext(width_px, height_px, AV_PIX_FMT_RGBA, width_px, height_px, enc->target_pix_fmt,
561 SWS_BICUBIC, NULL, NULL, NULL);
562 });
563 if (!enc->sws_ctx) {
564 av_frame_free(&enc->frame);
565 av_frame_free(&enc->frame_encoded);
566 av_packet_free(&enc->pkt);
567 avcodec_free_context(&enc->codec_ctx);
568 avformat_free_context(enc->fmt_ctx);
569 SAFE_FREE(enc);
570 return SET_ERRNO(ERROR_INIT, "ffmpeg: sws_getContext failed");
571 }
572
573 // Initialize audio stream if supported
574 encoder_init_audio_stream(enc, output_path);
575
576 // Open output file or stdout
577 if (!(enc->fmt_ctx->oformat->flags & AVFMT_NOFILE)) {
578 // For stdout output, use "pipe:1" (FFmpeg's standard for writing to stdout)
579 const char *open_path = output_path;
580 if (output_path && strcmp(output_path, "-") == 0) {
581 open_path = "pipe:1"; // Use pipe:1 for stdout
582 log_debug("ffmpeg: Redirecting '-' to 'pipe:1' for stdout output");
583 }
584 ret = avio_open(&enc->fmt_ctx->pb, open_path, AVIO_FLAG_WRITE);
585 if (ret < 0) {
586 sws_freeContext(enc->sws_ctx);
587 av_frame_free(&enc->frame);
588 av_frame_free(&enc->frame_encoded);
589 av_packet_free(&enc->pkt);
590 avcodec_free_context(&enc->codec_ctx);
591 avformat_free_context(enc->fmt_ctx);
592 SAFE_FREE(enc);
593 return SET_ERRNO(ERROR_INIT, "ffmpeg: avio_open failed for '%s'", open_path);
594 }
595 }
596
597 // Write header (capture FFmpeg format muxer logs)
598 LOG_IO("ffmpeg", { ret = avformat_write_header(enc->fmt_ctx, &opts); });
599 av_dict_free(&opts); // Free options after use
600 if (ret < 0) {
601 avio_closep(&enc->fmt_ctx->pb);
602 sws_freeContext(enc->sws_ctx);
603 av_frame_free(&enc->frame);
604 av_frame_free(&enc->frame_encoded);
605 av_packet_free(&enc->pkt);
606 avcodec_free_context(&enc->codec_ctx);
607 avformat_free_context(enc->fmt_ctx);
608 SAFE_FREE(enc);
609 return SET_ERRNO(ERROR_INIT, "ffmpeg: avformat_write_header failed");
610 }
611
612 // Now that stream->time_base is set, calculate estimated frame duration for snapshot mode
613 bool snapshot_mode_after_header = GET_OPTION(snapshot_mode);
614 double snapshot_delay_after_header = GET_OPTION(snapshot_delay);
615 if (snapshot_mode_after_header && enc->estimated_frame_count > 0 && snapshot_delay_after_header > 0) {
616 // Each frame should span: snapshot_delay / estimated_frame_count seconds
617 // In stream time_base units: (snapshot_delay * time_base.den) / estimated_frame_count
619 (int64_t)((snapshot_delay_after_header * (double)enc->stream->time_base.den) / enc->estimated_frame_count);
620 log_debug("ffmpeg_encoder_create: Calculated estimated_frame_duration=%lld time_base units "
621 "for snapshot mode (snapshot_delay=%.2f, estimated_frames=%d, time_base=%d/%d)",
622 (long long)enc->estimated_frame_duration, snapshot_delay_after_header, enc->estimated_frame_count,
623 enc->stream->time_base.num, enc->stream->time_base.den);
624 }
625
626 enc->frame_count = 0;
627 enc->video_pts = 0;
628 enc->snapshot_actual_duration = 0.0;
629 enc->snapshot_elapsed_ns = 0;
630 enc->first_frame_captured_ns = 0; // Initialize first frame timestamp
631 *out = enc;
632
633 /* Register encoder with named registry */
634 NAMED_REGISTER_FFMPEG_ENCODER(enc, output_path, NULL);
635
636 log_info("[FFMPEG_ENCODER_CREATE] SUCCESS: encoder initialized, frame_count=0, fps=%d, dims=%dx%d", enc->fps,
637 enc->width_px, enc->height_px);
638 return ASCIICHAT_OK;
639}
640
642 uint64_t captured_ns) {
643 if (!enc || !rgb)
644 return SET_ERRNO(ERROR_INVALID_PARAM, "ffmpeg_encoder_write_frame: NULL input");
645
646 // Initialize first frame capture time on first call
647 if (enc->first_frame_captured_ns == 0) {
648 enc->first_frame_captured_ns = captured_ns;
649 log_debug("ffmpeg_encoder_write_frame: first frame captured at %llu ns", (unsigned long long)captured_ns);
650 }
651
652 log_info("[FFMPEG_ENCODER_WRITE] frame %d, pitch=%d, dims=%dx%d", enc->frame_count, pitch, enc->width_px,
653 enc->height_px);
654 log_debug("ffmpeg_encoder_write_frame: frame %d, pitch=%d, dims=%dx%d, captured_ns=%llu", enc->frame_count, pitch,
655 enc->width_px, enc->height_px, (unsigned long long)captured_ns);
656
657 // Set up RGBA frame for conversion
658 uint8_t *data[1] = {(uint8_t *)rgb};
659 int linesize[1] = {pitch};
660
661 // Ensure frame format is set before each sws_scale call.
662 // Some FFmpeg versions reset the format field, so we set it explicitly.
663 enc->frame_encoded->format = enc->target_pix_fmt;
664
665 // Convert RGBA to target format
666 sws_scale(enc->sws_ctx, (const uint8_t *const *)data, linesize, 0, enc->height_px, enc->frame_encoded->data,
667 enc->frame_encoded->linesize);
668
669 // Set frame duration based on mode
670 int64_t frame_duration;
671 bool snapshot_mode = GET_OPTION(snapshot_mode);
672
673 if (enc->frame_count == 0) {
674 log_info("ffmpeg_encoder_write_frame: frame 0, snapshot_mode=%d", snapshot_mode);
675 }
676
677 // Use FPS-based frame duration (snapshot mode adjusts total duration in destroy())
678 frame_duration = enc->codec_ctx->time_base.den / (enc->codec_ctx->time_base.num * enc->fps);
679
680 enc->frame_encoded->duration = frame_duration;
681
682 // Track this frame's capture timestamp for duration calculation of next frame
683 if (snapshot_mode) {
684 uint64_t frame_dur_ms = (enc->previous_captured_ns > 0) ? (captured_ns - enc->previous_captured_ns) / 1000000 : 0;
685 if (enc->frame_count <= 2) {
686 log_info("ffmpeg: frame %d SET duration=%lld units (%.3fms since last, codec_base=1/%d)", enc->frame_count,
687 (long long)frame_duration, (double)frame_dur_ms, enc->codec_ctx->time_base.den);
688 }
689 enc->previous_captured_ns = captured_ns;
690 }
691
692 // Calculate PTS from actual capture timestamp
693 // time_base is {1, fps}, so 1 PTS unit = 1/fps seconds
694 // Convert elapsed nanoseconds to PTS units: elapsed_ns * fps / 1_000_000_000
695 uint64_t elapsed_ns = captured_ns - enc->first_frame_captured_ns;
696 int64_t pts_from_timestamp = (int64_t)((elapsed_ns * (uint64_t)enc->fps) / 1000000000ULL);
697
698 // Normal file rendering is constant-frame-rate. Wall-clock capture timestamps
699 // include scheduler jitter and can create gaps or duplicate PTS values, so use
700 // the encoded frame index for exact N / FPS duration. File snapshots with
701 // audio use this same timeline, excluding capture initialization delays.
702 // Live sources preserve wall-clock gaps between received frames.
703 if ((!snapshot_mode || enc->has_audio_stream) && !enc->live_timing) {
704 pts_from_timestamp = enc->frame_count;
705 }
706
707 // For snapshot sources without an estimated frame count, distribute frames
708 // linearly across the actual duration once the capture duration is known.
709 if (snapshot_mode && !enc->has_audio_stream && !enc->live_timing && enc->estimated_frame_count == 0) {
713 // Distribute frames linearly based on their capture timestamp relative to total duration
714 // pts = (frame_elapsed_ns / total_elapsed_ns) * actual_duration_sec * fps
715 // This ensures frame 0 is at 0s and last frame is at actual_duration_sec
716 double actual_duration_sec = (double)g_snapshot_actual_duration_ms / 1000.0;
717 double total_elapsed_sec = (double)g_snapshot_last_capture_elapsed_ns / (double)NS_PER_SEC_INT;
718
719 if (total_elapsed_sec > 0.001) {
720 double frame_fraction = (double)elapsed_ns / (double)g_snapshot_last_capture_elapsed_ns;
721 double target_pts_sec = frame_fraction * actual_duration_sec;
722 int64_t pts_before = pts_from_timestamp;
723 // Convert target seconds to time_base units: pts = target_sec * time_base.den / time_base.num
724 pts_from_timestamp =
725 (int64_t)(target_pts_sec * (double)enc->codec_ctx->time_base.den / (double)enc->codec_ctx->time_base.num);
726
727 if (enc->frame_count < 3 || enc->frame_count % 10 == 0) {
728 log_debug("ffmpeg: snapshot frame %d: PTS distributed (before=%lld, after=%lld, sec=%lld->%lld, frac=%.3f)",
729 enc->frame_count, (long long)pts_before, (long long)pts_from_timestamp,
730 (long long)((double)pts_before * enc->codec_ctx->time_base.num / enc->codec_ctx->time_base.den),
731 (long long)target_pts_sec, frame_fraction);
732 }
733 }
734 }
735 }
736
737 // Use timestamp-based (or scaled) PTS
738 enc->frame_encoded->pts = pts_from_timestamp;
739 enc->last_video_pts = pts_from_timestamp;
740 enc->video_pts += frame_duration;
741
742 if (!snapshot_mode) {
743 log_debug("ffmpeg_encoder_write_frame: PTS calculation - elapsed_ns=%llu, pts_from_timestamp=%lld, fps=%d",
744 (unsigned long long)elapsed_ns, (long long)pts_from_timestamp, enc->fps);
745 }
746
747 // Send frame to encoder
748 int ret = avcodec_send_frame(enc->codec_ctx, enc->frame_encoded);
749 if (ret < 0) {
750 log_warn_every(5 * 1000000000LL, "ffmpeg: avcodec_send_frame failed");
751 return ASCIICHAT_OK; // Continue anyway
752 }
753
754 // Receive packets from encoder and write to file
755 static int write_frame_pkt_count = 0;
756
757 while (ret >= 0) {
758 ret = avcodec_receive_packet(enc->codec_ctx, enc->pkt);
759 if (ret == AVERROR(EAGAIN) || ret == AVERROR_EOF)
760 break;
761 if (ret < 0) {
762 log_warn_every(5 * 1000000000LL, "ffmpeg: avcodec_receive_packet failed");
763 break;
764 }
765
766 write_frame_pkt_count++;
767
768 // Set packet duration in codec time base (duration of one frame at target FPS)
769 // This ensures proper stream duration calculation
770 if (enc->pkt->duration == 0) {
771 enc->pkt->duration = enc->codec_ctx->time_base.den / (enc->codec_ctx->time_base.num * enc->fps);
772 }
773
774 av_packet_rescale_ts(enc->pkt, enc->codec_ctx->time_base, enc->stream->time_base);
775 enc->pkt->stream_index = enc->stream->index;
776
777 // In snapshot mode, apply adjusted duration if snapshot_actual_duration is set
778 // This happens when we get the actual duration from the capture thread
779 bool is_snapshot_mode = GET_OPTION(snapshot_mode);
780 if (is_snapshot_mode && !enc->has_audio_stream && enc->snapshot_actual_duration > 0 && enc->frame_count > 0) {
781 int64_t adjusted_dur = (int64_t)((enc->snapshot_actual_duration * enc->stream->time_base.den) / enc->frame_count);
782 // Only apply if it's significantly different from the default
783 if (adjusted_dur > enc->pkt->duration * 2) {
784 enc->pkt->duration = adjusted_dur;
785 log_debug("ffmpeg: write_frame packet %d duration adjusted to %lld units based on actual duration",
786 write_frame_pkt_count, (long long)enc->pkt->duration);
787 }
788 } else if (!is_snapshot_mode && write_frame_pkt_count < 3) {
789 log_debug("ffmpeg: write_frame pkt %d, duration=%lld, snapshot_mode=%d, actual_dur=%.2f, frame_count=%d",
790 write_frame_pkt_count, (long long)enc->pkt->duration, is_snapshot_mode, enc->snapshot_actual_duration,
791 enc->frame_count);
792 }
793
794 if (write_frame_pkt_count <= 2 || write_frame_pkt_count >= 16) {
795 log_debug("ffmpeg: write_frame pkt %d with final duration=%lld (%.6f sec in stream time_base), total frames=%d",
796 write_frame_pkt_count, (long long)enc->pkt->duration,
797 (double)enc->pkt->duration / enc->stream->time_base.den, enc->frame_count);
798 }
799
800 ret = av_interleaved_write_frame(enc->fmt_ctx, enc->pkt);
801 av_packet_unref(enc->pkt);
802 if (ret < 0) {
803 log_warn_every(5 * 1000000000LL, "ffmpeg: av_interleaved_write_frame failed");
804 break;
805 }
806
807 // Flush output buffer for stdout to ensure frames are written immediately
808 if (enc->is_stdout_pipe) {
809 avio_flush(enc->fmt_ctx->pb);
810 }
811 }
812
813 enc->frame_count++;
814 return ASCIICHAT_OK;
815}
816
817asciichat_error_t ffmpeg_encoder_write_audio(ffmpeg_encoder_t *enc, const float *samples, int num_samples) {
818 if (!enc || !samples || num_samples <= 0)
819 return ASCIICHAT_OK; // Silently ignore invalid calls
820
821 // No audio stream for this format
822 if (!enc->has_audio_stream)
823 return ASCIICHAT_OK;
824
825 // Accumulate samples into partial buffer
826 int samples_to_process = num_samples;
827 int samples_offset = 0;
828
829 while (samples_to_process > 0) {
830 // How many samples can we add to the partial buffer?
831 int space_in_buffer = enc->audio_frame_size - enc->audio_partial_len;
832 int samples_to_copy = (samples_to_process < space_in_buffer) ? samples_to_process : space_in_buffer;
833
834 // Copy samples to partial buffer
835 memcpy(&enc->audio_partial_buf[enc->audio_partial_len], &samples[samples_offset], samples_to_copy * sizeof(float));
836 enc->audio_partial_len += samples_to_copy;
837 samples_offset += samples_to_copy;
838 samples_to_process -= samples_to_copy;
839
840 // If buffer is full, encode and write frame
841 if (enc->audio_partial_len >= enc->audio_frame_size) {
842 // Set up audio frame with accumulated samples
843 enc->audio_frame->data[0] = (uint8_t *)enc->audio_partial_buf;
844 enc->audio_frame->linesize[0] = enc->audio_frame_size * sizeof(float);
845 enc->audio_frame->nb_samples = enc->audio_frame_size;
846 enc->audio_frame->pts = enc->audio_pts;
847
848 // Send frame to encoder
849 int ret = avcodec_send_frame(enc->audio_codec_ctx, enc->audio_frame);
850 if (ret < 0) {
851 log_warn_every(5 * 1000000000LL, "ffmpeg: avcodec_send_frame for audio failed");
852 } else {
853 // Receive and write packets
854 while (ret >= 0) {
855 ret = avcodec_receive_packet(enc->audio_codec_ctx, enc->pkt);
856 if (ret == AVERROR(EAGAIN) || ret == AVERROR_EOF)
857 break;
858 if (ret < 0) {
859 log_warn_every(5 * 1000000000LL, "ffmpeg: avcodec_receive_packet for audio failed");
860 break;
861 }
862
863 // Rescale timestamp and write
864 av_packet_rescale_ts(enc->pkt, enc->audio_codec_ctx->time_base, enc->audio_stream->time_base);
865 enc->pkt->stream_index = enc->audio_stream->index;
866 ret = av_interleaved_write_frame(enc->fmt_ctx, enc->pkt);
867 av_packet_unref(enc->pkt);
868 if (ret < 0) {
869 log_warn_every(5 * 1000000000LL, "ffmpeg: av_interleaved_write_frame for audio failed");
870 break;
871 }
872 }
873 }
874
875 // Update PTS for next frame
876 enc->audio_pts += enc->audio_frame_size;
877 enc->audio_partial_len = 0;
878 }
879 }
880
881 return ASCIICHAT_OK;
882}
883
884void ffmpeg_encoder_set_snapshot_actual_duration(ffmpeg_encoder_t *enc, double actual_duration_sec) {
885 if (!enc)
886 return;
887 enc->snapshot_actual_duration = actual_duration_sec;
888
889 // If we already have frame count, we can calculate the adjusted packet duration now
890 // This will be used for any remaining frames and during flush
891 if (enc->frame_count > 0 && actual_duration_sec > 0) {
892 // Calculate the packet duration that would make all frames fit in actual_duration_sec
893 // packet_duration_units = (actual_duration_sec * time_base.den) / frame_count
894 // This is only a starting point - the actual duration might be different if more frames are captured
895 log_debug("ffmpeg_encoder_set_snapshot_actual_duration: %.3f sec with %d frames so far", actual_duration_sec,
896 enc->frame_count);
897 } else {
898 log_info(
899 "ffmpeg_encoder_set_snapshot_actual_duration: Setting duration to %.3f seconds (already encoded %d frames)",
900 actual_duration_sec, enc->frame_count);
901 }
902}
903
905 if (!enc)
906 return ASCIICHAT_OK;
907
908 NAMED_UNREGISTER(enc);
909
910 // In snapshot mode, calculate the FPS needed to match the target duration
911 bool snapshot_mode = GET_OPTION(snapshot_mode);
912 int64_t adjusted_packet_duration = 0;
913 double snapshot_delay = GET_OPTION(snapshot_delay);
914
915 if (snapshot_mode && !enc->has_audio_stream && !enc->live_timing && enc->frame_count > 0 && snapshot_delay > 0) {
916 // Calculate output FPS that distributes frames across snapshot_delay seconds
917 // output_fps = frame_count / snapshot_delay
918 // packet_duration (in stream time_base) = stream_time_base.den / output_fps
919 // = (stream_time_base.den * snapshot_delay) / frame_count
920 adjusted_packet_duration = (int64_t)((snapshot_delay * enc->stream->time_base.den) / enc->frame_count);
921 log_info("ffmpeg_encoder_destroy: Snapshot mode - adjusting packet durations for %d frames over %.2f seconds",
922 enc->frame_count, snapshot_delay);
923 log_debug(" Adjusted packet duration: %lld units (%.6f seconds per frame)", (long long)adjusted_packet_duration,
924 (double)adjusted_packet_duration / enc->stream->time_base.den);
925 }
926
927 // Flush delayed video packets. Encoders such as VP9 may buffer every short
928 // snapshot frame until this point.
929 LOG_IO("ffmpeg", {
930 avcodec_send_frame(enc->codec_ctx, NULL);
931 while (1) {
932 int ret = avcodec_receive_packet(enc->codec_ctx, enc->pkt);
933 if (ret == AVERROR(EAGAIN) || ret == AVERROR_EOF)
934 break;
935 if (ret < 0)
936 break;
937
938 // Set packet duration in codec time base if not already set
939 if (enc->pkt->duration == 0) {
940 enc->pkt->duration = enc->codec_ctx->time_base.den / (enc->codec_ctx->time_base.num * enc->fps);
941 }
942
943 av_packet_rescale_ts(enc->pkt, enc->codec_ctx->time_base, enc->stream->time_base);
944 enc->pkt->stream_index = enc->stream->index;
945 av_interleaved_write_frame(enc->fmt_ctx, enc->pkt);
946 av_packet_unref(enc->pkt);
947 }
948 });
949
950 // Flush audio encoder if present
951 if (enc->has_audio_stream && enc->audio_codec_ctx) {
952 // Write any remaining partial audio samples as padding
953 if (enc->audio_partial_len > 0) {
954 // Zero-pad the remaining samples
955 memset(&enc->audio_partial_buf[enc->audio_partial_len], 0,
956 (enc->audio_frame_size - enc->audio_partial_len) * sizeof(float));
957 enc->audio_frame->data[0] = (uint8_t *)enc->audio_partial_buf;
958 enc->audio_frame->linesize[0] = enc->audio_frame_size * sizeof(float);
959 enc->audio_frame->nb_samples = enc->audio_frame_size;
960 enc->audio_frame->pts = enc->audio_pts;
961 avcodec_send_frame(enc->audio_codec_ctx, enc->audio_frame);
962 }
963
964 // Flush any remaining packets (capture FFmpeg audio codec logs)
965 LOG_IO("ffmpeg", {
966 avcodec_send_frame(enc->audio_codec_ctx, NULL);
967 while (1) {
968 int ret = avcodec_receive_packet(enc->audio_codec_ctx, enc->pkt);
969 if (ret == AVERROR(EAGAIN) || ret == AVERROR_EOF)
970 break;
971 if (ret < 0)
972 break;
973
974 av_packet_rescale_ts(enc->pkt, enc->audio_codec_ctx->time_base, enc->audio_stream->time_base);
975 enc->pkt->stream_index = enc->audio_stream->index;
976 av_interleaved_write_frame(enc->fmt_ctx, enc->pkt);
977 av_packet_unref(enc->pkt);
978 }
979 });
980 }
981
982 // Set stream duration for proper metadata (must be before trailer)
983 if (enc->stream && enc->frame_count > 0) {
984 int64_t duration;
985 bool snapshot_mode = GET_OPTION(snapshot_mode);
986 if (snapshot_mode && !enc->has_audio_stream && !enc->live_timing) {
988 double snapshot_delay = GET_OPTION(snapshot_delay);
989 // Use actual capture elapsed time if available, otherwise use snapshot_delay
990 // This ensures frames are distributed across the correct time window
991 double actual_capture_duration = g_snapshot_actual_duration_ms / 1000.0;
992 double effective_duration = (actual_capture_duration > 0) ? actual_capture_duration : snapshot_delay;
993
994 // In snapshot mode with wall-clock timing, the video duration should equal the actual elapsed time
995 // duration_seconds = effective_duration
996 // duration_time_base_units = effective_duration * time_base.den
997 duration = (int64_t)(effective_duration * enc->stream->time_base.den);
998
999 // Calculate what the frame durations should be to sum to effective_duration
1000 // This ensures FFmpeg plays the captured frames across the full effective_duration
1001 // effective_frame_duration = effective_duration / frame_count (seconds per frame)
1002 int64_t effective_frame_duration =
1003 (int64_t)((effective_duration / (double)enc->frame_count) * enc->stream->time_base.den);
1004
1005 log_info("ffmpeg_encoder_destroy: Snapshot mode - snapshot_delay=%.2f, frames=%d, output_duration=%.2f sec",
1006 snapshot_delay, enc->frame_count, effective_duration);
1007 log_debug(" Using effective_duration=%.2f seconds for output", effective_duration);
1008 log_debug(" stream->time_base=%d/%d, effective frame duration=%lld time_base units", enc->stream->time_base.num,
1009 enc->stream->time_base.den, (long long)effective_frame_duration);
1010 log_debug(" (each frame should span %.3f seconds to total %.2f seconds)",
1011 effective_duration / (double)enc->frame_count, effective_duration);
1012
1013 log_info("ffmpeg_encoder_destroy: Setting stream->duration=%lld (effective_duration=%.3f, time_base=%d/%d, "
1014 "equiv=%.3f sec)",
1015 (long long)duration, effective_duration, enc->stream->time_base.num, enc->stream->time_base.den,
1016 (double)duration * enc->stream->time_base.num / enc->stream->time_base.den);
1017 } else {
1018 // Normal mode: calculate duration from frame count and FPS
1019 // time_base = 1 / time_base.den seconds per unit
1020 // Each frame is 1/fps seconds, so duration = frame_count / fps seconds
1021 // In time base units: duration = (frame_count / fps) * time_base.den = frame_count * time_base.den / fps
1022 duration = (enc->live_timing ? enc->last_video_pts + 1 : (int64_t)enc->frame_count) * enc->stream->time_base.den /
1023 enc->fps;
1024 log_debug("ffmpeg_encoder_destroy: Normal mode - Set stream duration=%lld (frames=%d, fps=%d, time_base=%d/%d)",
1025 (long long)duration, enc->frame_count, enc->fps, 1, enc->stream->time_base.den);
1026 }
1027 enc->stream->duration = duration;
1028
1029 // Each frame was encoded with FPS-based duration, but snapshot mode needs them distributed
1030 // across the actual capture duration. Since we can't change past frame durations, we extend
1031 // the last frame's duration to make the sum equal the desired total.
1032 if (snapshot_mode) {
1033 // Calculate sum of all frame durations (each frame currently has the same FPS-based duration)
1034 int64_t current_total_duration = enc->video_pts; // This is the sum of all frame durations so far
1035 int64_t desired_duration = duration;
1036
1037 // The last frame needs to be extended to: desired - (sum of all other frames)
1038 // This ensures the video plays for exactly the snapshot_delay duration
1039 int64_t last_frame_extension = desired_duration - current_total_duration;
1040
1041 if (last_frame_extension > 0) {
1042 // We need to extend the last frame. But we can't modify it directly after encoding.
1043 // Instead, the muxer will use stream->duration as the authoritative source.
1044 log_debug("ffmpeg_encoder_destroy: Last frame would need extension of %lld units to reach total %lld "
1045 "(currently %lld)",
1046 (long long)last_frame_extension, (long long)desired_duration, (long long)current_total_duration);
1047 }
1048 }
1049 }
1050
1051 // Write trailer (capture FFmpeg muxer logs and final frame statistics)
1052 LOG_IO("ffmpeg", { av_write_trailer(enc->fmt_ctx); });
1053
1054 // Flush output buffer for stdout/pipes before closing
1055 if (enc->fmt_ctx && enc->fmt_ctx->pb) {
1056 avio_flush(enc->fmt_ctx->pb);
1057 log_debug("ffmpeg_encoder_destroy: Flushed output buffer");
1058 }
1059
1060 // Close output file
1061 if (enc->fmt_ctx && !(enc->fmt_ctx->oformat->flags & AVFMT_NOFILE))
1062 avio_closep(&enc->fmt_ctx->pb);
1063
1064 // Cleanup
1065 sws_freeContext(enc->sws_ctx);
1066 av_frame_free(&enc->frame);
1067 av_frame_free(&enc->frame_encoded);
1068
1069 // Cleanup audio resources (capture any remaining FFmpeg logs from codec cleanup)
1070 if (enc->has_audio_stream) {
1071 swr_free(&enc->audio_swr_ctx);
1072 av_frame_free(&enc->audio_frame);
1073 LOG_IO("ffmpeg", { avcodec_free_context(&enc->audio_codec_ctx); });
1075 }
1076
1077 av_packet_free(&enc->pkt);
1078 LOG_IO("ffmpeg", { avcodec_free_context(&enc->codec_ctx); });
1079 if (enc->fmt_ctx)
1080 avformat_free_context(enc->fmt_ctx);
1081
1082 log_debug("ffmpeg_encoder_destroy: wrote %d frames", enc->frame_count);
1083 SAFE_FREE(enc);
1084 return ASCIICHAT_OK;
1085}
1086
1088 if (enc)
1089 enc->live_timing = true;
1090}
uint64_t g_snapshot_last_capture_elapsed_ns
uint64_t g_snapshot_actual_duration_ms
Named object registry for debugging — log identifiable resource names.
void ffmpeg_encoder_set_snapshot_actual_duration(ffmpeg_encoder_t *enc, double actual_duration_sec)
void ffmpeg_encoder_set_live_timing(ffmpeg_encoder_t *enc)
asciichat_error_t ffmpeg_encoder_write_frame(ffmpeg_encoder_t *enc, const uint8_t *rgb, int pitch, uint64_t captured_ns)
asciichat_error_t ffmpeg_encoder_destroy(ffmpeg_encoder_t *enc)
asciichat_error_t ffmpeg_encoder_write_audio(ffmpeg_encoder_t *enc, const float *samples, int num_samples)
asciichat_error_t ffmpeg_encoder_create(const char *output_path, int width_px, int height_px, int fps, ffmpeg_encoder_t **out)
FFmpeg video/image file encoder — codec selected from file extension.
#define SAFE_FREE(ptr)
Definition common.h:376
#define SAFE_MALLOC(size, cast)
Definition common.h:264
#define SAFE_CALLOC(count, size, cast)
Definition common.h:274
unsigned long long uint64_t
Definition common.h:59
unsigned char uint8_t
Definition common.h:56
#define NAMED_REGISTER_FFMPEG_ENCODER(encoder, name, parent_ptr)
Register an FFmpeg encoder with automatic format specifier.
#define NAMED_UNREGISTER(ptr)
Unregister a pointer.
#define SET_ERRNO(code, context_msg,...)
Set error code with custom context message and log it, returning the error code.
asciichat_error_t
Error and exit codes - unified status values (0-255)
Definition error_codes.h:49
@ ASCIICHAT_OK
Definition error_codes.h:51
@ ERROR_INIT
Definition error_codes.h:61
@ ERROR_INVALID_PARAM
#define log_warn(...)
Log a WARN message.
Definition log/log.h:574
#define log_info(...)
Log an INFO message.
Definition log/log.h:561
#define log_debug(...)
Log a DEBUG message.
Definition log/log.h:548
#define NS_PER_SEC_INT
Definition time.h:157
#define GET_OPTION(field)
Safely get a specific option field (lock-free read)
#define EAGAIN
⚙️ Unified options parsing system for ascii-chat with builder pattern and lock-free access
Capture stdout/stderr and redirect to logging system.
#define LOG_IO(prefix, block)
Capture output from a code block and log it.
Definition io.h:81
📝 Logging API with multiple log levels and terminal output control
#define log_warn_every(interval_us, fmt,...)
Rate-limited WARN logging.
Definition log/log.h:708
Cross-platform memory allocation utilities.
struct SwsContext * sws_ctx
AVFrame * frame_encoded
uint64_t first_frame_captured_ns
AVFrame * audio_frame
enum AVPixelFormat target_pix_fmt
float * audio_partial_buf
AVStream * audio_stream
AVCodecContext * audio_codec_ctx
struct SwrContext * audio_swr_ctx
uint64_t snapshot_elapsed_ns
AVFormatContext * fmt_ctx
double snapshot_actual_duration
AVCodecContext * codec_ctx
int64_t estimated_frame_duration
uint64_t previous_captured_ns