diff --git a/README.md b/README.md index fc07611..87b5f00 100644 --- a/README.md +++ b/README.md @@ -7,7 +7,8 @@ A native Linux peer-to-peer screencast application. - Built with **C++20**, **Meson**, **PipeWire**, **FFmpeg**, **RTP/UDP**, and **SDL**. -> This project is in early development. Only the skeleton exists so far. +> This project is in early development. The H.264 codec path is implemented; +> capture, transport, and rendering are still in progress. ## Quick start @@ -16,7 +17,8 @@ Requirements: - C++20 compiler (GCC 12+, Clang 16+) - Meson >= 0.63 - Ninja -- (later phases) FFmpeg dev packages, PipeWire dev, SDL2/3 dev +- FFmpeg development packages (`libavcodec`, `libavutil`, `libswscale`) +- (later phases) PipeWire dev, SDL2/3 dev Build and run tests: @@ -40,7 +42,7 @@ See `docs/ARCHITECTURE.md` for module boundaries and design rules. Development is split into phases in `docs/PHASES.md`. -Current phase: **Phase 2 — software H.264 encode / decode**. +Current phase: **Phase 3 — PipeWire screen capture**. ## License diff --git a/docs/PHASES.md b/docs/PHASES.md index 3f9302f..29c5d8b 100644 --- a/docs/PHASES.md +++ b/docs/PHASES.md @@ -20,10 +20,10 @@ previous one is validated. **Goal**: encode raw pixel buffers to H.264 and decode them back, purely with FFmpeg software paths. -- Add `libavcodec`, `libavutil`, `libswscale` dependencies. -- Implement `EncoderFactory::create()` and `Encoder::encode()`. -- Implement `DecoderFactory::create()` and `Decoder::decode()`. -- Round-trip test: synthetic RGB frames → H.264 → decoded RGB. +- [x] Add `libavcodec`, `libavutil`, `libswscale` dependencies. +- [x] Implement `EncoderFactory::create()` and `Encoder::encode()`. +- [x] Implement `DecoderFactory::create()` and `Decoder::decode()`. +- [x] Round-trip test: synthetic RGB frames → H.264 → decoded RGB. **Validation**: unit test produces visually/structurally correct round-trip frames. @@ -89,4 +89,4 @@ where available. ## Current phase -Phase 2 — software H.264 encode/decode. +Phase 3 — PipeWire screen capture. diff --git a/include/screencast/codec/decoder.h b/include/screencast/codec/decoder.h index 821ea2e..909f015 100644 --- a/include/screencast/codec/decoder.h +++ b/include/screencast/codec/decoder.h @@ -1,11 +1,11 @@ #pragma once #include "screencast/codec/encoder.h" +#include "screencast/codec/error.h" #include #include -#include -#include +#include #include namespace sc { @@ -18,22 +18,23 @@ struct DecodedFrame { }; struct DecoderConfig { - std::string_view codec_name = "h264"; + std::string codec_name = "h264"; int width = 0; int height = 0; + std::vector extradata; // SPS/PPS for H.264 }; class Decoder { public: - virtual ~Decoder () = default; + virtual ~Decoder() = default; // Feed one encoded frame. Returns decoded frames when available. - virtual std::vector decode ( const EncodedFrame &frame ) = 0; + virtual CodecResult> decode(const EncodedFrame& frame) = 0; }; class DecoderFactory { public: - static std::unique_ptr create ( const DecoderConfig &config ); + static CodecResult> create(const DecoderConfig& config); }; -} // namespace sc +} // namespace sc diff --git a/include/screencast/codec/encoder.h b/include/screencast/codec/encoder.h index becbc21..04a317f 100644 --- a/include/screencast/codec/encoder.h +++ b/include/screencast/codec/encoder.h @@ -1,11 +1,11 @@ #pragma once #include "screencast/capture/capture.h" +#include "screencast/codec/error.h" #include #include -#include -#include +#include #include namespace sc { @@ -14,11 +14,11 @@ struct EncodedFrame { uint64_t capture_timestamp_ns = 0; uint32_t rtp_timestamp = 0; bool is_keyframe = false; - std::vector data; // Annex-B or AVCC depending on config + std::vector data; // Annex-B H.264 NAL units }; struct EncoderConfig { - std::string_view codec_name = "h264"; + std::string codec_name = "h264"; int width = 0; int height = 0; int frame_rate_num = 30; @@ -29,19 +29,25 @@ struct EncoderConfig { class Encoder { public: - virtual ~Encoder () = default; + virtual ~Encoder() = default; - // Encode one captured frame. Returns empty if the encoder emits no packet - // for this frame. - virtual std::vector encode ( const CapturedFrame &frame ) = 0; + // Encode one captured frame. Returns an empty vector if the encoder emits + // no packet for this frame. + virtual CodecResult> encode(const CapturedFrame& frame) = 0; - // Force the next output to be a keyframe. - virtual void request_keyframe () = 0; + // Flush the encoder and return any remaining packets. + virtual CodecResult> flush() = 0; + + // Force the next output frame to be a keyframe. + virtual void request_keyframe() = 0; + + // Codec-specific parameters required by a decoder (SPS/PPS for H.264). + virtual std::vector get_extradata() const = 0; }; class EncoderFactory { public: - static std::unique_ptr create ( const EncoderConfig &config ); + static CodecResult> create(const EncoderConfig& config); }; -} // namespace sc +} // namespace sc diff --git a/include/screencast/codec/error.h b/include/screencast/codec/error.h new file mode 100644 index 0000000..e3bef54 --- /dev/null +++ b/include/screencast/codec/error.h @@ -0,0 +1,36 @@ +#pragma once + +#include +#include + +namespace sc { + +struct CodecError { + std::string message; +}; + +// C++20 does not provide std::expected. Use a variant-based result type so +// fallible codec operations do not rely on exceptions. +template using CodecResult = std::variant; + +template constexpr bool is_codec_error(const CodecResult& result) noexcept { + return std::holds_alternative(result); +} + +template T& codec_value(CodecResult& result) { + return std::get(result); +} + +template const T& codec_value(const CodecResult& result) { + return std::get(result); +} + +template CodecError& codec_error(CodecResult& result) { + return std::get(result); +} + +template const CodecError& codec_error(const CodecResult& result) { + return std::get(result); +} + +} // namespace sc diff --git a/src/codec/ffmpeg_decoder.cpp b/src/codec/ffmpeg_decoder.cpp new file mode 100644 index 0000000..675db11 --- /dev/null +++ b/src/codec/ffmpeg_decoder.cpp @@ -0,0 +1,237 @@ +#include "screencast/codec/decoder.h" + +#include "ffmpeg_raii.h" + +#include +#include +#include +#include +#include +#include + +namespace sc { +namespace { + +using namespace sc::detail; + +enum class ReceiveStatus { + Frame, + Again, + Eof, + Error, +}; + +struct ReceiveResult { + ReceiveStatus status = ReceiveStatus::Error; + AvFramePtr frame; + std::string error; +}; + +} // namespace + +class FfmpegDecoder final : public Decoder { + public: + FfmpegDecoder(AvCodecContextPtr ctx, DecoderConfig config) : ctx_(std::move(ctx)), config_(std::move(config)) {} + + CodecResult> decode(const EncodedFrame& frame) override { + if (frame.data.empty()) { + return std::vector{}; + } + + AvPacketPtr packet(av_packet_alloc(), AvPacketDeleter{}); + if (packet == nullptr) { + return CodecError{"failed to allocate AVPacket"}; + } + + packet->pts = av_rescale_q( + static_cast(frame.capture_timestamp_ns), AVRational{1, 1'000'000'000}, ctx_->time_base); + + packet->data = static_cast(av_malloc(frame.data.size() + AV_INPUT_BUFFER_PADDING_SIZE)); + if (packet->data == nullptr) { + return CodecError{"failed to allocate packet data"}; + } + std::memcpy(packet->data, frame.data.data(), frame.data.size()); + packet->size = static_cast(frame.data.size()); + + std::vector out; + int send_ret = avcodec_send_packet(ctx_.get(), packet.get()); + while (send_ret == AVERROR(EAGAIN)) { + auto received = receive_one_frame(); + if (received.status == ReceiveStatus::Error) { + return CodecError{received.error}; + } + if (received.status == ReceiveStatus::Frame) { + auto frame_result = make_decoded_frame(received.frame.get(), frame.capture_timestamp_ns); + if (is_codec_error(frame_result)) { + return CodecError{codec_error(frame_result).message}; + } + out.push_back(std::move(codec_value(frame_result))); + } + if (received.status == ReceiveStatus::Again) { + return CodecError{"decoder stalled before producing output"}; + } + send_ret = avcodec_send_packet(ctx_.get(), packet.get()); + } + + if (send_ret < 0) { + return CodecError{ffmpeg_error(send_ret)}; + } + + if (auto error = drain_frames(frame.capture_timestamp_ns, out)) { + return CodecError{std::move(*error)}; + } + + return out; + } + + private: + ReceiveResult receive_one_frame() { + AvFramePtr frame(av_frame_alloc(), AvFrameDeleter{}); + if (frame == nullptr) { + return {ReceiveStatus::Error, nullptr, "failed to allocate AVFrame"}; + } + + int ret = avcodec_receive_frame(ctx_.get(), frame.get()); + if (ret == AVERROR(EAGAIN)) { + return {ReceiveStatus::Again, nullptr, {}}; + } + if (ret == AVERROR_EOF) { + return {ReceiveStatus::Eof, nullptr, {}}; + } + if (ret < 0) { + return {ReceiveStatus::Error, nullptr, ffmpeg_error(ret)}; + } + + return {ReceiveStatus::Frame, std::move(frame), {}}; + } + + std::optional drain_frames(uint64_t fallback_timestamp_ns, std::vector& out) { + while (true) { + auto received = receive_one_frame(); + if (received.status == ReceiveStatus::Again || received.status == ReceiveStatus::Eof) { + return std::nullopt; + } + if (received.status == ReceiveStatus::Error) { + return received.error; + } + + auto frame_result = make_decoded_frame(received.frame.get(), fallback_timestamp_ns); + if (is_codec_error(frame_result)) { + return codec_error(frame_result).message; + } + out.push_back(std::move(codec_value(frame_result))); + } + } + + CodecResult make_decoded_frame(const AVFrame* input, uint64_t fallback_timestamp_ns) const { + if (input->width <= 0 || input->height <= 0) { + return CodecError{"decoded frame dimensions must be positive"}; + } + + DecodedFrame frame; + frame.width = input->width; + frame.height = input->height; + frame.capture_timestamp_ns = + input->pts != AV_NOPTS_VALUE + ? static_cast(av_rescale_q(input->pts, ctx_->time_base, AVRational{1, 1'000'000'000})) + : fallback_timestamp_ns; + + const std::size_t buffer_size = + static_cast(av_image_get_buffer_size(AV_PIX_FMT_RGBA, frame.width, frame.height, 1)); + frame.rgba_pixels.resize(buffer_size); + + if (!ensure_scaler(input->width, input->height, static_cast(input->format))) { + return CodecError{"failed to create swscale context"}; + } + + std::array dst{nullptr, nullptr, nullptr, nullptr}; + std::array dst_lines{0, 0, 0, 0}; + if (av_image_fill_arrays(dst.data(), + dst_lines.data(), + as_u8(frame.rgba_pixels.data()), + AV_PIX_FMT_RGBA, + frame.width, + frame.height, + 1) < 0) { + return CodecError{"failed to fill output pixel arrays"}; + } + + if (sws_scale(scaler_.get(), input->data, input->linesize, 0, input->height, dst.data(), dst_lines.data()) <= + 0) { + return CodecError{"failed to convert decoded frame to RGBA"}; + } + + return frame; + } + + bool ensure_scaler(int width, int height, AVPixelFormat input_format) const { + if (scaler_ != nullptr && scaler_input_width_ == width && scaler_input_height_ == height && + scaler_input_format_ == input_format) { + return true; + } + + scaler_.reset(sws_getContext( + width, height, input_format, width, height, AV_PIX_FMT_RGBA, SWS_BILINEAR, nullptr, nullptr, nullptr)); + if (scaler_ == nullptr) { + return false; + } + scaler_input_width_ = width; + scaler_input_height_ = height; + scaler_input_format_ = input_format; + return true; + } + + AvCodecContextPtr ctx_; + DecoderConfig config_; + + mutable SwsContextPtr scaler_; + mutable int scaler_input_width_ = 0; + mutable int scaler_input_height_ = 0; + mutable AVPixelFormat scaler_input_format_ = AV_PIX_FMT_NONE; +}; + +CodecResult> DecoderFactory::create(const DecoderConfig& config) { + if (config.codec_name != "h264") { + return CodecError{"only h264 is supported in phase 2"}; + } + + const AVCodec* codec = avcodec_find_decoder(AV_CODEC_ID_H264); + if (codec == nullptr) { + return CodecError{"h264 decoder not found"}; + } + + AvCodecContextPtr ctx(avcodec_alloc_context3(codec), AvCodecContextDeleter{}); + if (ctx == nullptr) { + return CodecError{"failed to allocate decoder context"}; + } + + ctx->codec_type = AVMEDIA_TYPE_VIDEO; + if (config.width > 0) { + ctx->width = config.width; + } + if (config.height > 0) { + ctx->height = config.height; + } + ctx->thread_count = 1; + + if (!config.extradata.empty()) { + if (config.extradata.size() > static_cast(std::numeric_limits::max())) { + return CodecError{"decoder extradata is too large"}; + } + ctx->extradata = static_cast(av_malloc(config.extradata.size())); + if (ctx->extradata == nullptr) { + return CodecError{"failed to allocate decoder extradata"}; + } + std::memcpy(ctx->extradata, config.extradata.data(), config.extradata.size()); + ctx->extradata_size = static_cast(config.extradata.size()); + } + + int open_ret = avcodec_open2(ctx.get(), codec, nullptr); + if (open_ret < 0) { + return CodecError{std::string{"failed to open h264 decoder: "} + ffmpeg_error(open_ret)}; + } + + return std::make_unique(std::move(ctx), config); +} + +} // namespace sc diff --git a/src/codec/ffmpeg_encoder.cpp b/src/codec/ffmpeg_encoder.cpp new file mode 100644 index 0000000..0ba6d65 --- /dev/null +++ b/src/codec/ffmpeg_encoder.cpp @@ -0,0 +1,371 @@ +#include "screencast/codec/encoder.h" +#include "screencast/utils/clock.h" + +#include "ffmpeg_raii.h" + +#include +#include +#include +#include +#include +#include +#include +#include + +namespace sc { +namespace { + +using namespace sc::detail; + +AVPixelFormat to_ffmpeg_format(PixelFormat format) noexcept { + return format == PixelFormat::Rgba ? AV_PIX_FMT_RGBA : AV_PIX_FMT_YUV420P; +} + +int expected_first_plane_stride(PixelFormat format, int width) noexcept { + return format == PixelFormat::Rgba ? width * 4 : width; +} + +std::size_t expected_buffer_size(PixelFormat format, int width, int height) { + return static_cast(av_image_get_buffer_size(to_ffmpeg_format(format), width, height, 1)); +} + +enum class ReceiveStatus { + Packet, + Again, + Eof, + Error, +}; + +struct ReceiveResult { + ReceiveStatus status = ReceiveStatus::Error; + AvPacketPtr packet; + std::string error; +}; + +// Convert an H.264 elementary stream to Annex-B form. +// +// FFmpeg's libx264 packets are already Annex-B, so this is primarily a +// normalization helper for future encoders that may emit AVCC length-prefixed +// NAL units. Only 4-byte length prefixes are supported. +std::vector to_annex_b_h264(std::span in) { + if (in.empty()) { + return {}; + } + + // Try to parse the input as AVCC. If the length-prefixed NALs consume + // exactly the whole buffer, treat it as AVCC and rewrite it as Annex-B. + if (in.size() >= 4) { + std::vector out; + out.reserve(in.size()); // Annex-B is the same size or larger. + std::size_t offset = 0; + bool ok = true; + + while (offset < in.size()) { + if (offset + 4 > in.size()) { + ok = false; + break; + } + + const auto* p = in.data() + offset; + const uint32_t length = (static_cast(static_cast(p[0])) << 24) | + (static_cast(static_cast(p[1])) << 16) | + (static_cast(static_cast(p[2])) << 8) | + (static_cast(static_cast(p[3]))); + + // Guard against malformed or pathological length prefixes. + if (length == 0 || length > std::numeric_limits::max() / 2 || offset + 4 + length > in.size()) { + ok = false; + break; + } + + out.push_back(std::byte{0x00}); + out.push_back(std::byte{0x00}); + out.push_back(std::byte{0x00}); + out.push_back(std::byte{0x01}); + out.insert(out.end(), + in.begin() + static_cast(offset + 4), + in.begin() + static_cast(offset + 4 + length)); + offset += 4 + length; + } + + if (ok && offset == in.size()) { + return out; + } + } + + return std::vector(in.begin(), in.end()); +} + +} // namespace + +class FfmpegEncoder final : public Encoder { + public: + FfmpegEncoder(AvCodecContextPtr ctx, EncoderConfig config) : ctx_(std::move(ctx)), config_(std::move(config)) {} + + CodecResult> encode(const CapturedFrame& frame) override { + auto input_result = make_input_frame(frame); + if (is_codec_error(input_result)) { + return CodecError{codec_error(input_result).message}; + } + auto input = std::move(codec_value(input_result)); + + if (force_keyframe_.exchange(false)) { + input->pict_type = AV_PICTURE_TYPE_I; + } + + std::vector out; + int send_ret = avcodec_send_frame(ctx_.get(), input.get()); + while (send_ret == AVERROR(EAGAIN)) { + auto received = receive_one_packet(); + if (received.status == ReceiveStatus::Error) { + return CodecError{received.error}; + } + if (received.status == ReceiveStatus::Packet) { + out.push_back(to_encoded_frame(received.packet.get(), frame.timestamp_ns)); + } + if (received.status == ReceiveStatus::Again) { + return CodecError{"encoder stalled before producing output"}; + } + send_ret = avcodec_send_frame(ctx_.get(), input.get()); + } + + if (send_ret < 0) { + return CodecError{ffmpeg_error(send_ret)}; + } + + if (auto error = drain_packets(frame.timestamp_ns, out)) { + return CodecError{std::move(*error)}; + } + + return out; + } + + CodecResult> flush() override { + std::vector out; + int ret = avcodec_send_frame(ctx_.get(), nullptr); + while (ret == AVERROR(EAGAIN)) { + auto received = receive_one_packet(); + if (received.status == ReceiveStatus::Error) { + return CodecError{received.error}; + } + if (received.status == ReceiveStatus::Packet) { + out.push_back(to_encoded_frame(received.packet.get(), 0)); + } + if (received.status == ReceiveStatus::Again) { + return CodecError{"encoder stalled during flush"}; + } + ret = avcodec_send_frame(ctx_.get(), nullptr); + } + + if (ret < 0) { + return CodecError{ffmpeg_error(ret)}; + } + + if (auto error = drain_packets(0, out)) { + return CodecError{std::move(*error)}; + } + + return out; + } + + void request_keyframe() override { + force_keyframe_ = true; + } + + std::vector get_extradata() const override { + if (ctx_->extradata == nullptr || ctx_->extradata_size <= 0) { + return {}; + } + return std::vector(as_bytes(ctx_->extradata), as_bytes(ctx_->extradata) + ctx_->extradata_size); + } + + private: + ReceiveResult receive_one_packet() { + AvPacketPtr packet(av_packet_alloc(), AvPacketDeleter{}); + if (packet == nullptr) { + return {ReceiveStatus::Error, nullptr, "failed to allocate RTP packet"}; + } + + int ret = avcodec_receive_packet(ctx_.get(), packet.get()); + if (ret == AVERROR(EAGAIN)) { + return {ReceiveStatus::Again, nullptr, {}}; + } + if (ret == AVERROR_EOF) { + return {ReceiveStatus::Eof, nullptr, {}}; + } + if (ret < 0) { + return {ReceiveStatus::Error, nullptr, ffmpeg_error(ret)}; + } + + return {ReceiveStatus::Packet, std::move(packet), {}}; + } + + std::optional drain_packets(uint64_t fallback_timestamp_ns, std::vector& out) { + while (true) { + auto received = receive_one_packet(); + if (received.status == ReceiveStatus::Again || received.status == ReceiveStatus::Eof) { + return std::nullopt; + } + if (received.status == ReceiveStatus::Error) { + return received.error; + } + out.push_back(to_encoded_frame(received.packet.get(), fallback_timestamp_ns)); + } + } + + EncodedFrame to_encoded_frame(const AVPacket* packet, uint64_t fallback_timestamp_ns) const { + EncodedFrame frame; + if (packet->pts != AV_NOPTS_VALUE) { + frame.capture_timestamp_ns = + static_cast(av_rescale_q(packet->pts, ctx_->time_base, AVRational{1, 1'000'000'000})); + } else { + frame.capture_timestamp_ns = fallback_timestamp_ns; + } + frame.rtp_timestamp = rtp_timestamp_from_ns(frame.capture_timestamp_ns); + frame.is_keyframe = (packet->flags & AV_PKT_FLAG_KEY) != 0; + + if (packet->data != nullptr && packet->size > 0) { + frame.data = to_annex_b_h264(as_byte_span(packet->data, static_cast(packet->size))); + } + + return frame; + } + + CodecResult make_input_frame(const CapturedFrame& frame) const { + if (frame.width <= 0 || frame.height <= 0) { + return CodecError{"frame dimensions must be positive"}; + } + + const AVPixelFormat input_format = to_ffmpeg_format(frame.pixel_format); + const int expected_stride = expected_first_plane_stride(frame.pixel_format, frame.width); + if (frame.stride != 0 && frame.stride != expected_stride) { + return CodecError{"unsupported input stride"}; + } + + const std::size_t required_size = expected_buffer_size(frame.pixel_format, frame.width, frame.height); + if (frame.pixels.size() < required_size) { + return CodecError{"input pixel buffer is too small"}; + } + + AvFramePtr output(av_frame_alloc(), AvFrameDeleter{}); + if (output == nullptr) { + return CodecError{"failed to allocate AVFrame"}; + } + + output->width = frame.width; + output->height = frame.height; + output->format = AV_PIX_FMT_YUV420P; + output->time_base = ctx_->time_base; + output->pts = + av_rescale_q(static_cast(frame.timestamp_ns), AVRational{1, 1'000'000'000}, ctx_->time_base); + + if (av_frame_get_buffer(output.get(), 0) < 0) { + return CodecError{"failed to allocate AVFrame buffer"}; + } + + if (!ensure_scaler(frame.width, frame.height, input_format)) { + return CodecError{"failed to create swscale context"}; + } + + std::array src{nullptr, nullptr, nullptr, nullptr}; + std::array src_lines{0, 0, 0, 0}; + if (av_image_fill_arrays( + src.data(), src_lines.data(), as_u8(frame.pixels.data()), input_format, frame.width, frame.height, 1) < + 0) { + return CodecError{"failed to fill input pixel arrays"}; + } + + if (sws_scale(scaler_.get(), src.data(), src_lines.data(), 0, frame.height, output->data, output->linesize) <= + 0) { + return CodecError{"failed to convert input frame to YUV420P"}; + } + + return output; + } + + bool ensure_scaler(int width, int height, AVPixelFormat input_format) const { + if (scaler_ != nullptr && scaler_input_width_ == width && scaler_input_height_ == height && + scaler_input_format_ == input_format) { + return true; + } + + scaler_.reset(sws_getContext( + width, height, input_format, width, height, AV_PIX_FMT_YUV420P, SWS_BILINEAR, nullptr, nullptr, nullptr)); + if (scaler_ == nullptr) { + return false; + } + scaler_input_width_ = width; + scaler_input_height_ = height; + scaler_input_format_ = input_format; + return true; + } + + AvCodecContextPtr ctx_; + EncoderConfig config_; + std::atomic force_keyframe_{false}; + + mutable SwsContextPtr scaler_; + mutable int scaler_input_width_ = 0; + mutable int scaler_input_height_ = 0; + mutable AVPixelFormat scaler_input_format_ = AV_PIX_FMT_NONE; +}; + +CodecResult> EncoderFactory::create(const EncoderConfig& config) { + if (config.width <= 0 || config.height <= 0) { + return CodecError{"encoder width and height must be positive"}; + } + if (config.frame_rate_num <= 0 || config.frame_rate_den <= 0) { + return CodecError{"encoder frame rate must be positive"}; + } + if (config.bitrate_kbps <= 0) { + return CodecError{"encoder bitrate must be positive"}; + } + if (config.codec_name != "h264" && config.codec_name != "libx264") { + return CodecError{"only h264 is supported in phase 2"}; + } + if (config.hardware_accel) { + return CodecError{"hardware acceleration is not implemented in phase 2"}; + } + + const AVCodec* codec = avcodec_find_encoder_by_name("libx264"); + if (codec == nullptr) { + return CodecError{"libx264 encoder not found"}; + } + + AvCodecContextPtr ctx(avcodec_alloc_context3(codec), AvCodecContextDeleter{}); + if (ctx == nullptr) { + return CodecError{"failed to allocate encoder context"}; + } + + ctx->codec_type = AVMEDIA_TYPE_VIDEO; + ctx->width = config.width; + ctx->height = config.height; + ctx->time_base = AVRational{config.frame_rate_den, config.frame_rate_num}; + ctx->framerate = AVRational{config.frame_rate_num, config.frame_rate_den}; + ctx->pix_fmt = AV_PIX_FMT_YUV420P; + ctx->bit_rate = static_cast(config.bitrate_kbps) * 1000; + ctx->gop_size = config.frame_rate_num; + ctx->max_b_frames = 0; + ctx->thread_count = 1; + ctx->profile = AV_PROFILE_H264_MAIN; + ctx->flags |= AV_CODEC_FLAG_LOW_DELAY | AV_CODEC_FLAG_GLOBAL_HEADER; + + if (av_opt_set(ctx->priv_data, "preset", "ultrafast", 0) < 0) { + return CodecError{"failed to set libx264 preset"}; + } + if (av_opt_set(ctx->priv_data, "tune", "zerolatency", 0) < 0) { + return CodecError{"failed to set libx264 tune"}; + } + if (av_opt_set(ctx->priv_data, "forced-idr", "1", 0) < 0) { + return CodecError{"failed to enable forced IDR keyframes"}; + } + + int open_ret = avcodec_open2(ctx.get(), codec, nullptr); + if (open_ret < 0) { + return CodecError{std::string{"failed to open libx264 encoder: "} + ffmpeg_error(open_ret)}; + } + + return std::make_unique(std::move(ctx), config); +} + +} // namespace sc diff --git a/src/codec/ffmpeg_headers.h b/src/codec/ffmpeg_headers.h new file mode 100644 index 0000000..28ced16 --- /dev/null +++ b/src/codec/ffmpeg_headers.h @@ -0,0 +1,15 @@ +#pragma once + +#if defined(__cplusplus) +extern "C" { +#endif + +#include +#include +#include +#include +#include + +#if defined(__cplusplus) +} +#endif diff --git a/src/codec/ffmpeg_raii.h b/src/codec/ffmpeg_raii.h new file mode 100644 index 0000000..4c4fd15 --- /dev/null +++ b/src/codec/ffmpeg_raii.h @@ -0,0 +1,78 @@ +#pragma once + +#include "ffmpeg_headers.h" + +#include +#include +#include +#include + +namespace sc::detail { + +// Helpers to centralize the std::byte <-> uint8_t casts required by FFmpeg. +inline uint8_t* as_u8(std::byte* data) noexcept { + return reinterpret_cast(data); +} + +inline const uint8_t* as_u8(const std::byte* data) noexcept { + return reinterpret_cast(data); +} + +inline std::byte* as_bytes(uint8_t* data) noexcept { + return reinterpret_cast(data); +} + +inline const std::byte* as_bytes(const uint8_t* data) noexcept { + return reinterpret_cast(data); +} + +inline std::span as_byte_span(const uint8_t* data, std::size_t size) noexcept { + return std::span(as_bytes(data), size); +} + +struct AvCodecContextDeleter { + void operator()(AVCodecContext* ctx) const noexcept { + if (ctx) { + avcodec_free_context(&ctx); + } + } +}; + +struct AvFrameDeleter { + void operator()(AVFrame* frame) const noexcept { + if (frame) { + av_frame_free(&frame); + } + } +}; + +struct AvPacketDeleter { + void operator()(AVPacket* packet) const noexcept { + if (packet) { + av_packet_free(&packet); + } + } +}; + +struct SwsContextDeleter { + void operator()(SwsContext* ctx) const noexcept { + if (ctx) { + sws_freeContext(ctx); + } + } +}; + +using AvCodecContextPtr = std::unique_ptr; +using AvFramePtr = std::unique_ptr; +using AvPacketPtr = std::unique_ptr; +using SwsContextPtr = std::unique_ptr; + +inline std::string ffmpeg_error(int errnum) { + std::array buffer{}; + if (av_strerror(errnum, buffer.data(), buffer.size()) < 0) { + return "unknown FFmpeg error"; + } + return buffer.data(); +} + +} // namespace sc::detail diff --git a/src/meson.build b/src/meson.build index b392b3e..b7527ba 100644 --- a/src/meson.build +++ b/src/meson.build @@ -1,6 +1,22 @@ -# Core public API is currently header-only. Source files will be added as -# modules are implemented in later phases. - -# Declare an include-only dependency so that tests and executables can depend on -# the public headers consistently. +# Core public headers / include dependency. sc_core_inc = include_directories('../include') + +# Phase 2 codec dependencies. +dep_avcodec = dependency('libavcodec') +dep_avutil = dependency('libavutil') +dep_swscale = dependency('libswscale') + +sc_codec_sources = files( + 'codec/ffmpeg_encoder.cpp', + 'codec/ffmpeg_decoder.cpp', +) + +sc_codec = static_library('sc_codec', + sc_codec_sources, + include_directories : sc_core_inc, + dependencies : [dep_avcodec, dep_avutil, dep_swscale]) + +sc_codec_dep = declare_dependency( + link_with : sc_codec, + include_directories : sc_core_inc, + dependencies : [dep_avcodec, dep_avutil, dep_swscale]) diff --git a/tests/codec/test_roundtrip.cpp b/tests/codec/test_roundtrip.cpp new file mode 100644 index 0000000..15c2f98 --- /dev/null +++ b/tests/codec/test_roundtrip.cpp @@ -0,0 +1,146 @@ +#include "screencast/codec/decoder.h" +#include "screencast/codec/encoder.h" + +#include +#include +#include +#include +#include +#include +#include + +namespace { + +template void expect_codec_result(const char* what, const sc::CodecResult& result) { + if (sc::is_codec_error(result)) { + std::cerr << what << " failed: " << sc::codec_error(result).message << '\n'; + std::abort(); + } +} + +constexpr int kWidth = 128; +constexpr int kHeight = 128; +constexpr int kFrames = 30; +constexpr int kFrameRate = 25; +constexpr uint64_t kNsPerFrame = 1'000'000'000ULL / kFrameRate; + +sc::CapturedFrame make_frame(uint32_t index) { + sc::CapturedFrame frame; + frame.width = kWidth; + frame.height = kHeight; + frame.timestamp_ns = static_cast(index) * kNsPerFrame; + frame.pixel_format = sc::PixelFormat::Rgba; + frame.stride = kWidth * 4; + frame.pixels.resize(static_cast(kWidth) * kHeight * 4); + + auto* pixels = reinterpret_cast(frame.pixels.data()); + for (int y = 0; y < kHeight; ++y) { + for (int x = 0; x < kWidth; ++x) { + const std::size_t offset = (static_cast(y) * kWidth + x) * 4; + pixels[offset + 0] = static_cast((x + static_cast(index) * 4) & 0xFF); + pixels[offset + 1] = static_cast((y + static_cast(index) * 2) & 0xFF); + pixels[offset + 2] = static_cast(((x ^ y) + static_cast(index)) & 0xFF); + pixels[offset + 3] = 0xFF; + } + } + + return frame; +} + +int max_channel_difference(const sc::DecodedFrame& decoded, const sc::CapturedFrame& expected) { + const auto* decoded_pixels = reinterpret_cast(decoded.rgba_pixels.data()); + const auto* expected_pixels = reinterpret_cast(expected.pixels.data()); + const std::size_t count = std::min(decoded.rgba_pixels.size(), expected.pixels.size()); + + int max_diff = 0; + for (std::size_t i = 0; i < count; ++i) { + const int diff = decoded_pixels[i] >= expected_pixels[i] ? decoded_pixels[i] - expected_pixels[i] + : expected_pixels[i] - decoded_pixels[i]; + max_diff = std::max(max_diff, diff); + } + return max_diff; +} + +bool starts_with_annex_b_prefix(const sc::EncodedFrame& frame) { + if (frame.data.size() < 4) { + return false; + } + return frame.data[0] == std::byte{0x00} && frame.data[1] == std::byte{0x00} && frame.data[2] == std::byte{0x00} && + frame.data[3] == std::byte{0x01}; +} + +} // namespace + +int main() { + sc::EncoderConfig encoder_config; + encoder_config.codec_name = "h264"; + encoder_config.width = kWidth; + encoder_config.height = kHeight; + encoder_config.frame_rate_num = kFrameRate; + encoder_config.frame_rate_den = 1; + encoder_config.bitrate_kbps = 8000; + encoder_config.hardware_accel = false; + + auto encoder_result = sc::EncoderFactory::create(encoder_config); + expect_codec_result("encoder create", encoder_result); + auto encoder = std::move(sc::codec_value(encoder_result)); + + const auto extradata = encoder->get_extradata(); + assert(!extradata.empty()); + + sc::DecoderConfig decoder_config; + decoder_config.codec_name = "h264"; + decoder_config.width = kWidth; + decoder_config.height = kHeight; + decoder_config.extradata = extradata; + + auto decoder_result = sc::DecoderFactory::create(decoder_config); + expect_codec_result("decoder create", decoder_result); + auto decoder = std::move(sc::codec_value(decoder_result)); + + std::map expected_by_timestamp; + bool saw_keyframe = false; + bool saw_annex_b_prefix = false; + + auto decode_packets = [&](const std::vector& packets) { + for (const auto& packet : packets) { + saw_keyframe = saw_keyframe || packet.is_keyframe; + saw_annex_b_prefix = saw_annex_b_prefix || starts_with_annex_b_prefix(packet); + + auto decoded_result = decoder->decode(packet); + expect_codec_result("decode", decoded_result); + + for (const auto& decoded : sc::codec_value(decoded_result)) { + assert(decoded.width == kWidth); + assert(decoded.height == kHeight); + + auto it = expected_by_timestamp.find(decoded.capture_timestamp_ns); + assert(it != expected_by_timestamp.end()); + assert(max_channel_difference(decoded, it->second) <= 64); + } + } + }; + + for (uint32_t i = 0; i < kFrames; ++i) { + const auto frame = make_frame(i); + expected_by_timestamp.emplace(frame.timestamp_ns, frame); + + if (i == kFrames / 2) { + encoder->request_keyframe(); + } + + auto encoded_result = encoder->encode(frame); + expect_codec_result("encode", encoded_result); + decode_packets(sc::codec_value(encoded_result)); + } + + auto flushed_result = encoder->flush(); + expect_codec_result("flush", flushed_result); + decode_packets(sc::codec_value(flushed_result)); + + assert(!expected_by_timestamp.empty()); + assert(saw_keyframe); + assert(saw_annex_b_prefix); + + return 0; +} diff --git a/tests/meson.build b/tests/meson.build index 0b1b499..03eff97 100644 --- a/tests/meson.build +++ b/tests/meson.build @@ -3,3 +3,9 @@ test_clock = executable('test_clock', include_directories : sc_core_inc) test('clock utils', test_clock) + +test_codec_roundtrip = executable('test_codec_roundtrip', + 'codec/test_roundtrip.cpp', + dependencies : sc_codec_dep) + +test('h264 roundtrip', test_codec_roundtrip)