implement proper avcodec encode/decode loops

This commit is contained in:
David Bears 2026-01-16 02:30:14 -05:00
parent e01e1a490d
commit 921f8eb690
No known key found for this signature in database
GPG Key ID: FB975E12C69F7177
2 changed files with 178 additions and 135 deletions

View File

@ -131,7 +131,7 @@ void av_frame_free(AVFrame **frame)
#endif // MINGW
VideoProcessor::VideoProcessor()
:_encoded_frame_size(640,480) , vpMtx("VideoProcessor")
:_encoded_frame_size(640,480)
{
//_lastTimeToShowFrame = time(NULL);
_decoded_output_device = NULL ;
@ -154,12 +154,10 @@ VideoProcessor::~VideoProcessor()
{
// clear encoding queue
RS_STACK_MUTEX(vpMtx) ;
while(!_encoded_out_queue.empty())
{
_encoded_out_queue.back().clear() ;
_encoded_out_queue.pop_back() ;
_encoded_out_queue.getTail().clear();
_encoded_out_queue.pop() ;
}
}
@ -181,21 +179,12 @@ bool VideoProcessor::processImage(const QImage& img)
if(codec)
{
RsVOIPDataChunk chunk ;
if(codec->encodeData(img.scaled(_encoded_frame_size,Qt::IgnoreAspectRatio,Qt::SmoothTransformation),_target_bandwidth_out,chunk) && chunk.size > 0)
{
RS_STACK_MUTEX(vpMtx) ;
_encoded_out_queue.push_back(chunk) ;
_total_encoded_size_out += chunk.size ;
}
codec->encodeData(img.scaled(_encoded_frame_size,Qt::IgnoreAspectRatio,Qt::SmoothTransformation),_target_bandwidth_out, _encoded_out_queue);
time_t now = time(NULL) ;
if(now > _last_bw_estimate_out_TS)
{
RS_STACK_MUTEX(vpMtx) ;
_estimated_bandwidth_out = uint32_t(0.75*_estimated_bandwidth_out + 0.25 * (_total_encoded_size_out / (float)(now - _last_bw_estimate_out_TS))) ;
_total_encoded_size_out = 0 ;
@ -217,14 +206,8 @@ bool VideoProcessor::processImage(const QImage& img)
bool VideoProcessor::nextEncodedPacket(RsVOIPDataChunk& chunk)
{
RS_STACK_MUTEX(vpMtx) ;
if(_encoded_out_queue.empty())
return false ;
chunk = _encoded_out_queue.front() ;
_encoded_out_queue.pop_front() ;
return true ;
return _encoded_out_queue.pop(chunk)
&& ((_total_encoded_size_out += chunk.size), true);
}
void VideoProcessor::setInternalFrameSize(QSize s)
@ -275,7 +258,6 @@ void VideoProcessor::receiveEncodedData(const RsVOIPDataChunk& chunk)
}
{
RS_STACK_MUTEX(vpMtx) ;
_total_encoded_size_in += chunk.size ;
time_t now = time(NULL) ;
@ -368,10 +350,13 @@ bool JPEGVideo::decodeData(const RsVOIPDataChunk& chunk,QImage& image)
return true ;
}
bool JPEGVideo::encodeData(const QImage& image,uint32_t /* size_hint */,RsVOIPDataChunk& voip_chunk)
bool JPEGVideo::encodeData(const QImage& image,uint32_t /* size_hint */, NetQueue<RsVOIPDataChunk, 8>& dst)
{
// check if we make a diff image, or if we use the full frame.
if(dst.full()) return false;
RsVOIPDataChunk& voip_chunk = dst.getHead();
QImage encoded_frame ;
bool differential_frame ;
@ -426,6 +411,8 @@ bool JPEGVideo::encodeData(const QImage& image,uint32_t /* size_hint */,RsVOIPDa
voip_chunk.size = HEADER_SIZE + qb.size() ;
voip_chunk.type = RsVOIPDataChunk::RS_VOIP_DATA_TYPE_VIDEO ;
dst.push();
return true ;
}
@ -617,12 +604,81 @@ FFmpegVideo::~FFmpegVideo()
#define MAX_FFMPEG_ENCODING_BITRATE 81920
bool FFmpegVideo::encodeData(const QImage& image, uint32_t target_encoding_bitrate, RsVOIPDataChunk& voip_chunk)
static void imageToFrame(AVFrame *frame, const QImage& image) {
QImage input;
if(image.width() == frame->width && image.height() == frame->height)
input = image;
else
input = image.scaled(QSize(frame->width, frame->height),
Qt::IgnoreAspectRatio, Qt::SmoothTransformation);
for (int y = 0; y < frame->height - 1; y += 2)
for (int x = 0; x < frame->width - 1; x += 2) {
QRgb pix00 = input.pixel(QPoint(x+0,y+0));
QRgb pix10 = input.pixel(QPoint(x+1,y+0));
QRgb pix01 = input.pixel(QPoint(x+0,y+1));
QRgb pix11 = input.pixel(QPoint(x+1,y+1));
int R00 = pix00>>16 & 0xff, G00 = pix00>>8 & 0xff, B00 = pix00>>0 & 0xff;
int R10 = pix10>>16 & 0xff, G10 = pix10>>8 & 0xff, B10 = pix10>>0 & 0xff;
int R01 = pix01>>16 & 0xff, G01 = pix01>>8 & 0xff, B01 = pix01>>0 & 0xff;
int R11 = pix11>>16 & 0xff, G11 = pix11>>8 & 0xff, B11 = pix11>>0 & 0xff;
float R = 0.25*(R00+R01+R10+R11);
float G = 0.25*(G00+G01+G10+G11);
float B = 0.25*(B00+B01+B10+B11);
int Y00 = (0.257 * R00) + (0.504 * G00) + (0.098 * B00) + 16.5;
int Y01 = (0.257 * R01) + (0.504 * G01) + (0.098 * B01) + 16.5;
int Y10 = (0.257 * R10) + (0.504 * G10) + (0.098 * B10) + 16.5;
int Y11 = (0.257 * R11) + (0.504 * G11) + (0.098 * B11) + 16.5;
int U = (0.439 * R) - (0.368 * G) - (0.071 * B) + 128.5 ;
int V = -(0.148 * R) - (0.291 * G) + (0.439 * B) + 128.5 ;
frame->data[0][(y+0) * frame->linesize[0] + x+0] = Y00; // Y
frame->data[0][(y+0) * frame->linesize[0] + x+1] = Y01; // Y
frame->data[0][(y+1) * frame->linesize[0] + x+0] = Y10; // Y
frame->data[0][(y+1) * frame->linesize[0] + x+1] = Y11; // Y
frame->data[1][y/2 * frame->linesize[1] + x/2] = U; // Cr
frame->data[2][y/2 * frame->linesize[2] + x/2] = V; // Cb
}
}
static void frameToImage(QImage& image, const AVFrame *frame) {
image = QImage(QSize(frame->width,frame->height),QImage::Format_ARGB32);
#ifdef DEBUG_MPEG_VIDEO
std::cerr << "Decoded frame. Size=" << image.width() << "x" << image.height() << std::endl;
#endif
for (int y = 0; y < frame->height; y++)
for (int x = 0; x < frame->width; x++) {
int Y = frame->data[0][y * frame->linesize[0] + x];
int U = frame->data[1][(y/2) * frame->linesize[1] + x/2];
int V = frame->data[2][(y/2) * frame->linesize[2] + x/2];
int B = (int)(1.164*Y + 1.596*V - 222.421);
int G = (int)(1.164*Y - 0.391*U - 0.813*V + 136.076);
int R = (int)(1.164*Y + 2.018*U - 277.336);
B = std::max(0,std::min(255,B));
G = std::max(0,std::min(255,G));
R = std::max(0,std::min(255,R));
image.setPixel(QPoint(x,y), QRgb(0xff000000 | (R << 16) | (G << 8) | B));
}
}
bool FFmpegVideo::encodeData(const QImage& image, uint32_t target_encoding_bitrate, NetQueue<RsVOIPDataChunk, 8>& dst)
{
#if LIBAVCODEC_VERSION_MAJOR < 58
if(dst.full()) return false;
#endif
#ifdef DEBUG_MPEG_VIDEO
std::cerr << "Encoding frame of size " << image.width() << "x" << image.height() << ", resized to " << encoding_frame_buffer->width << "x" << encoding_frame_buffer->height << " : ";
#endif
QImage input ;
if(target_encoding_bitrate > MAX_FFMPEG_ENCODING_BITRATE)
{
@ -633,54 +689,12 @@ bool FFmpegVideo::encodeData(const QImage& image, uint32_t target_encoding_bitra
encoding_context->rc_max_rate = target_encoding_bitrate;
//encoding_context->bit_rate_tolerance = target_encoding_bitrate;
if(image.width() != encoding_frame_buffer->width || image.height() != encoding_frame_buffer->height)
input = image.scaled(QSize(encoding_frame_buffer->width,encoding_frame_buffer->height),Qt::IgnoreAspectRatio,Qt::SmoothTransformation) ;
else
input = image ;
/* prepare a dummy image */
/* Y */
for (int y = 0; y < encoding_context->height/2; y++)
for (int x = 0; x < encoding_context->width/2; x++)
{
QRgb pix00 = input.pixel(QPoint(2*x+0,2*y+0)) ;
QRgb pix01 = input.pixel(QPoint(2*x+0,2*y+1)) ;
QRgb pix10 = input.pixel(QPoint(2*x+1,2*y+0)) ;
QRgb pix11 = input.pixel(QPoint(2*x+1,2*y+1)) ;
int R00 = (pix00 >> 16) & 0xff ; int G00 = (pix00 >> 8) & 0xff ; int B00 = (pix00 >> 0) & 0xff ;
int R01 = (pix01 >> 16) & 0xff ; int G01 = (pix01 >> 8) & 0xff ; int B01 = (pix01 >> 0) & 0xff ;
int R10 = (pix10 >> 16) & 0xff ; int G10 = (pix10 >> 8) & 0xff ; int B10 = (pix10 >> 0) & 0xff ;
int R11 = (pix11 >> 16) & 0xff ; int G11 = (pix11 >> 8) & 0xff ; int B11 = (pix11 >> 0) & 0xff ;
int Y00 = (0.257 * R00) + (0.504 * G00) + (0.098 * B00) + 16 ;
int Y01 = (0.257 * R01) + (0.504 * G01) + (0.098 * B01) + 16 ;
int Y10 = (0.257 * R10) + (0.504 * G10) + (0.098 * B10) + 16 ;
int Y11 = (0.257 * R11) + (0.504 * G11) + (0.098 * B11) + 16 ;
float R = 0.25*(R00+R01+R10+R11) ;
float G = 0.25*(G00+G01+G10+G11) ;
float B = 0.25*(B00+B01+B10+B11) ;
int U = (0.439 * R) - (0.368 * G) - (0.071 * B) + 128 ;
int V = -(0.148 * R) - (0.291 * G) + (0.439 * B) + 128 ;
encoding_frame_buffer->data[0][(2*y+0) * encoding_frame_buffer->linesize[0] + 2*x+0] = std::min(255,std::max(0,Y00)); // Y
encoding_frame_buffer->data[0][(2*y+0) * encoding_frame_buffer->linesize[0] + 2*x+1] = std::min(255,std::max(0,Y01)); // Y
encoding_frame_buffer->data[0][(2*y+1) * encoding_frame_buffer->linesize[0] + 2*x+0] = std::min(255,std::max(0,Y10)); // Y
encoding_frame_buffer->data[0][(2*y+1) * encoding_frame_buffer->linesize[0] + 2*x+1] = std::min(255,std::max(0,Y11)); // Y
encoding_frame_buffer->data[1][y * encoding_frame_buffer->linesize[1] + x] = std::min(255,std::max(0,U));// Cr
encoding_frame_buffer->data[2][y * encoding_frame_buffer->linesize[2] + x] = std::min(255,std::max(0,V));// Cb
}
imageToFrame(encoding_frame_buffer, image);
encoding_frame_buffer->pts = encoding_frame_count++;
/* encode the image */
int got_output = 0;
AVPacket pkt ;
av_init_packet(&pkt);
#if LIBAVCODEC_VERSION_MAJOR < 54
@ -689,6 +703,7 @@ bool FFmpegVideo::encodeData(const QImage& image, uint32_t target_encoding_bitra
// do
// {
int got_output = 0;
int ret = avcodec_encode_video(encoding_context, pkt.data, pkt.size, encoding_frame_buffer) ;
if (ret > 0) {
got_output = ret;
@ -699,14 +714,11 @@ bool FFmpegVideo::encodeData(const QImage& image, uint32_t target_encoding_bitra
// do
// {
int got_output = 0;
int ret = avcodec_encode_video2(encoding_context, &pkt, encoding_frame_buffer, &got_output) ;
#else
int ret = avcodec_send_frame(encoding_context, encoding_frame_buffer);
if(!ret || ret == AVERROR(EAGAIN))
ret = avcodec_receive_packet(encoding_context, &pkt);
if(!ret)
got_output = 1;
if(ret == AVERROR(EAGAIN))
ret = 0;
#endif
@ -714,18 +726,29 @@ bool FFmpegVideo::encodeData(const QImage& image, uint32_t target_encoding_bitra
if (ret < 0)
{
std::cerr << "Error encoding frame!" << std::endl;
return false ;
goto fail;
}
// frame = NULL ; // next attempts: do not encode anything. Do this to just flush the buffer
//
// } while(got_output) ;
if(got_output)
#if LIBAVCODEC_VERSION_MAJOR >= 58
while(!dst.full() && !(ret = avcodec_receive_packet(encoding_context, &pkt)))
#else
if(!got_output) {
std::cerr << "No output produced." << std::endl;
}
else
#endif
{
RsVOIPDataChunk& voip_chunk = dst.getHead();
voip_chunk.data = rs_malloc(pkt.size + HEADER_SIZE) ;
if(!voip_chunk.data)
return false ;
if(!voip_chunk.data) {
std::cerr << "error: failed to allocate RsVOIPDataChunk data"
<< std::endl;
goto fail;
}
uint32_t flags = 0;
@ -739,32 +762,36 @@ bool FFmpegVideo::encodeData(const QImage& image, uint32_t target_encoding_bitra
voip_chunk.size = pkt.size + HEADER_SIZE;
voip_chunk.type = RsVOIPDataChunk::RS_VOIP_DATA_TYPE_VIDEO ;
dst.push();
#ifdef DEBUG_MPEG_VIDEO
std::cerr << "Output : " << pkt.size << " bytes." << std::endl;
fwrite(pkt.data,1,pkt.size,encoding_debug_file) ;
fflush(encoding_debug_file) ;
#endif
#if LIBAVCODEC_VERSION_MAJOR < 58
av_free_packet(&pkt);
#else
av_packet_unref(&pkt);
}
#if LIBAVCODEC_VERSION_MAJOR >= 58
if(ret == AVERROR(EAGAIN))
ret = 0;
else if(ret < 0) {
std::cerr << "error: failed to get packet from encoder: " << ret
<< std::endl;
}
#endif
return true ;
}
else
{
voip_chunk.data = NULL;
voip_chunk.size = 0;
voip_chunk.type = RsVOIPDataChunk::RS_VOIP_DATA_TYPE_VIDEO ;
fail:
std::cerr << "No output produced." << std::endl;
return false ;
}
#if LIBAVCODEC_VERSION_MAJOR < 58
av_free_packet(&pkt);
#else
av_packet_unref(&pkt);
#endif
return ret == 0;
}
bool FFmpegVideo::decodeData(const RsVOIPDataChunk& chunk,QImage& image)
bool FFmpegVideo::decodeData(const RsVOIPDataChunk& chunk, QImage& image)
{
#ifdef DEBUG_MPEG_VIDEO
std::cerr << "Decoding data of size " << chunk.size << std::endl;
@ -786,68 +813,67 @@ bool FFmpegVideo::decodeData(const RsVOIPDataChunk& chunk,QImage& image)
av_packet_from_data(&decoding_buffer, tmp, s);
int got_frame = 1 ;
#if LIBAVCODEC_VERSION_MAJOR >= 58
int ret = avcodec_send_packet(decoding_context, &decoding_buffer);
if(ret && ret != AVERROR(EAGAIN)) {
std::cerr << "Error decoding frame! Return=" << ret << std::endl;
got_frame = 0;
av_packet_unref(&decoding_buffer);
if(ret) {
// EAGAIN is an error here because the decoder's buffer should be empty
std::cerr << "error: failed to send packet to decoder: " << ret
<< std::endl;
}
else {
av_packet_unref(&decoding_buffer);
ret = avcodec_receive_frame(decoding_context, decoding_frame_buffer);
if(ret == AVERROR(EAGAIN)) {
std::cerr << "warning: "
<< "decoder consumed a packet but didn't generate a frame" << std::endl;
return false;
}
else if(ret) {
std::cerr << "error: failed to fetch frame from decoder: " << ret
<< std::endl;
return false ;
}
frameToImage(image, decoding_frame_buffer);
#ifdef DEBUG_MPEG_VIDEO
std::cerr << "debug(MPEG_VIDEO): decoded frame: size="
<< image.width() << "x" << image.height() << std::endl;
#endif
#if LIBAVCODEC_VERSION_MAJOR < 58
while (decoding_buffer.size > 0 || (!decoding_buffer.data && got_frame))
// clear out the internal buffer and drop the frames to avoid latency
// Video decoders typically only create one frame per packet, so this
// shouldn't happen under normal circumstances.
while(!avcodec_receive_frame(decoding_context, decoding_frame_buffer)) {
std::cerr << "warning: dropping decoded frame";
}
#else
while (got_frame)
#endif
{
#if LIBAVCODEC_VERSION_MAJOR < 58
int got_frame = 1 ;
while (decoding_buffer.size > 0 || (!decoding_buffer.data && got_frame)) {
int len = avcodec_decode_video2(decoding_context,decoding_frame_buffer,&got_frame,&decoding_buffer) ;
#else
int len = avcodec_receive_frame(decoding_context, decoding_frame_buffer);
if(len == AVERROR(EAGAIN)) break;
#endif
if (len < 0)
{
std::cerr << "Error decoding frame! Return=" << len << std::endl;
return false ;
}
#if LIBAVCODEC_VERSION_MAJOR < 58
decoding_buffer.data += len;
decoding_buffer.size -= len;
#endif
if(got_frame)
{
image = QImage(QSize(decoding_frame_buffer->width,decoding_frame_buffer->height),QImage::Format_ARGB32) ;
frameToImage(image, decoding_frame_buffer);
#ifdef DEBUG_MPEG_VIDEO
std::cerr << "Decoded frame. Size=" << image.width() << "x" << image.height() << std::endl;
#endif
for (int y = 0; y < decoding_frame_buffer->height; y++)
for (int x = 0; x < decoding_frame_buffer->width; x++)
{
int Y = decoding_frame_buffer->data[0][y * decoding_frame_buffer->linesize[0] + x] ;
int U = decoding_frame_buffer->data[1][(y/2) * decoding_frame_buffer->linesize[1] + x/2] ;
int V = decoding_frame_buffer->data[2][(y/2) * decoding_frame_buffer->linesize[2] + x/2] ;
int B = std::min(255,std::max(0,(int)(1.164*(Y - 16) + 1.596*(V - 128)))) ;
int G = std::min(255,std::max(0,(int)(1.164*(Y - 16) - 0.813*(V - 128) - 0.391*(U - 128)))) ;
int R = std::min(255,std::max(0,(int)(1.164*(Y - 16) + 2.018*(U - 128)))) ;
image.setPixel(QPoint(x,y),QRgb( 0xff000000 + (R << 16) + (G << 8) + B)) ;
}
}
}
/* flush the decoder */
#if LIBAVCODEC_VERSION_MAJOR < 58
decoding_buffer.data = NULL;
decoding_buffer.size = 0;
//avcodec_decode_video2(decoding_context,decoding_frame_buffer,&got_frame,&decoding_buffer) ;

View File

@ -30,10 +30,28 @@ extern "C" {
class QVideoOutputDevice ;
template<typename T, int N>
class NetQueue {
std::array<T, N> arr;
std::atomic<int> head = 0;
std::atomic<int> tail = 0;
public:
bool full() const { return head - tail == N; }
T& getHead() { return arr[head % N]; }
void push() { head++; }
bool push(const T& e) { return !full() && ((arr[head++ % N] = e), true); }
bool empty() const { return head == tail; }
T& getTail() { return arr[tail % N]; }
void pop() { tail++; }
bool pop(T& e) { return !empty() && ((e = arr[tail++ % N]), true); }
};
class VideoCodec
{
public:
virtual bool encodeData(const QImage& Image, uint32_t size_hint, RsVOIPDataChunk& chunk) = 0;
virtual bool encodeData(const QImage& Image, uint32_t size_hint, NetQueue<RsVOIPDataChunk, 8>& dst) = 0;
virtual bool decodeData(const RsVOIPDataChunk& chunk,QImage& image) = 0;
protected:
@ -49,8 +67,8 @@ public:
JPEGVideo() ;
protected:
virtual bool encodeData(const QImage& Image, uint32_t target_encoding_bitrate, RsVOIPDataChunk& chunk) ;
virtual bool decodeData(const RsVOIPDataChunk& chunk,QImage& image) ;
virtual bool encodeData(const QImage& Image, uint32_t target_encoding_bitrate, NetQueue<RsVOIPDataChunk, 8>& dst) ;
virtual bool decodeData(const RsVOIPDataChunk& chunk, QImage& image) ;
static const uint32_t JPEG_VIDEO_FLAGS_DIFFERENTIAL_FRAME = 0x0001 ;
private:
@ -73,7 +91,7 @@ public:
~FFmpegVideo() ;
protected:
virtual bool encodeData(const QImage& Image, uint32_t target_encoding_bitrate, RsVOIPDataChunk& chunk) ;
virtual bool encodeData(const QImage& Image, uint32_t target_encoding_bitrate, NetQueue<RsVOIPDataChunk, 8>& dst) ;
virtual bool decodeData(const RsVOIPDataChunk& chunk,QImage& image) ;
private:
@ -147,7 +165,8 @@ class VideoProcessor
uint32_t currentBandwidthOut() const { return _estimated_bandwidth_out ; }
protected:
std::list<RsVOIPDataChunk> _encoded_out_queue ;
NetQueue<RsVOIPDataChunk, 8> _encoded_out_queue ;
QSize _encoded_frame_size ;
// =====================================================================================
@ -169,6 +188,4 @@ class VideoProcessor
float _estimated_bandwidth_out ;
float _target_bandwidth_out ;
RsMutex vpMtx ;
};