x86: sbrdsp: Implement SSE2 qmf_deint_bfly

[ffmpeg] / libavcodec / proresdec.c
diff --git a/libavcodec/proresdec.c b/libavcodec/proresdec.c

index 207fd9742eb7b0e8954f2d96f58720c6267151bb..c42e44415398ba4bf5cdb73b6cf11ef33500f2c5 100644 (file)
--- a/libavcodec/proresdec.c
+++ b/libavcodec/proresdec.c
@@ -34,6 +34,8 @@
  
  #include "libavutil/intmath.h"
  #include "avcodec.h"
+#include "dsputil.h"
+#include "internal.h"
  #include "proresdata.h"
  #include "proresdsp.h"
  #include "get_bits.h"
@@ -44,14 +46,14 @@ typedef struct {
      int x_pos, y_pos;
      int slice_width;
      int prev_slice_sf;               ///< scalefactor of the previous decoded slice
-    DECLARE_ALIGNED(16, DCTELEM, blocks)[8 * 4 * 64];
+    DECLARE_ALIGNED(16, int16_t, blocks)[8 * 4 * 64];
      DECLARE_ALIGNED(16, int16_t, qmat_luma_scaled)[64];
      DECLARE_ALIGNED(16, int16_t, qmat_chroma_scaled)[64];
  } ProresThreadData;
  
  typedef struct {
      ProresDSPContext dsp;
-    AVFrame    picture;
+    AVFrame    *frame;
      ScanTable  scantable;
      int        scantable_type;           ///< -1 = uninitialized, 0 = progressive, 1/2 = interlaced
  
@@ -86,11 +88,6 @@ static av_cold int decode_init(AVCodecContext *avctx)
      avctx->bits_per_raw_sample = PRORES_BITS_PER_SAMPLE;
      ff_proresdsp_init(&ctx->dsp);
  
-    avctx->coded_frame = &ctx->picture;
-    avcodec_get_frame_defaults(&ctx->picture);
-    ctx->picture.type      = AV_PICTURE_TYPE_I;
-    ctx->picture.key_frame = 1;
-
      ctx->scantable_type = -1;   // set scantable type to uninitialized
      memset(ctx->qmat_luma, 4, 64);
      memset(ctx->qmat_chroma, 4, 64);
@@ -139,10 +136,10 @@ static int decode_frame_header(ProresContext *ctx, const uint8_t *buf,
      ctx->num_chroma_blocks = (1 << ctx->chroma_factor) >> 1;
      switch (ctx->chroma_factor) {
      case 2:
-        avctx->pix_fmt = PIX_FMT_YUV422P10;
+        avctx->pix_fmt = AV_PIX_FMT_YUV422P10;
          break;
      case 3:
-        avctx->pix_fmt = PIX_FMT_YUV444P10;
+        avctx->pix_fmt = AV_PIX_FMT_YUV444P10;
          break;
      default:
          av_log(avctx, AV_LOG_ERROR,
@@ -161,13 +158,19 @@ static int decode_frame_header(ProresContext *ctx, const uint8_t *buf,
      }
  
      if (ctx->frame_type) {      /* if interlaced */
-        ctx->picture.interlaced_frame = 1;
-        ctx->picture.top_field_first  = ctx->frame_type & 1;
+        ctx->frame->interlaced_frame = 1;
+        ctx->frame->top_field_first  = ctx->frame_type & 1;
+    } else {
+        ctx->frame->interlaced_frame = 0;
      }
  
+    avctx->color_primaries = buf[14];
+    avctx->color_trc       = buf[15];
+    avctx->colorspace      = buf[16];
+
      ctx->alpha_info = buf[17] & 0xf;
      if (ctx->alpha_info)
-        av_log_missing_feature(avctx, "alpha channel", 0);
+        avpriv_report_missing_feature(avctx, "Alpha channel");
  
      ctx->qmat_changed = 0;
      ptr   = buf + 20;
@@ -239,8 +242,8 @@ static int decode_picture_header(ProresContext *ctx, const uint8_t *buf,
  
      ctx->num_x_mbs = (avctx->width + 15) >> 4;
      ctx->num_y_mbs = (avctx->height +
-                      (1 << (4 + ctx->picture.interlaced_frame)) - 1) >>
-                     (4 + ctx->picture.interlaced_frame);
+                      (1 << (4 + ctx->frame->interlaced_frame)) - 1) >>
+                     (4 + ctx->frame->interlaced_frame);
  
      remainder    = ctx->num_x_mbs & ((1 << slice_width_factor) - 1);
      num_x_slices = (ctx->num_x_mbs >> slice_width_factor) + (remainder & 1) +
@@ -289,7 +292,7 @@ static int decode_picture_header(ProresContext *ctx, const uint8_t *buf,
  /**
   * Read an unsigned rice/exp golomb codeword.
   */
-static inline int decode_vlc_codeword(GetBitContext *gb, uint8_t codebook)
+static inline int decode_vlc_codeword(GetBitContext *gb, unsigned codebook)
  {
      unsigned int rice_order, exp_order, switch_bits;
      unsigned int buf, code;
@@ -333,10 +336,10 @@ static inline int decode_vlc_codeword(GetBitContext *gb, uint8_t codebook)
  /**
   * Decode DC coefficients for all blocks in a slice.
   */
-static inline void decode_dc_coeffs(GetBitContext *gb, DCTELEM *out,
+static inline void decode_dc_coeffs(GetBitContext *gb, int16_t *out,
                                      int nblocks)
  {
-    DCTELEM prev_dc;
+    int16_t prev_dc;
      int     i, sign;
      int16_t delta;
      unsigned int code;
@@ -361,7 +364,7 @@ static inline void decode_dc_coeffs(GetBitContext *gb, DCTELEM *out,
  /**
   * Decode AC coefficients for all blocks in a slice.
   */
-static inline void decode_ac_coeffs(GetBitContext *gb, DCTELEM *out,
+static inline void decode_ac_coeffs(GetBitContext *gb, int16_t *out,
                                      int blocks_per_slice,
                                      int plane_size_factor,
                                      const uint8_t *scan)
@@ -411,10 +414,10 @@ static void decode_slice_plane(ProresContext *ctx, ProresThreadData *td,
                                 int data_size, uint16_t *out_ptr,
                                 int linesize, int mbs_per_slice,
                                 int blocks_per_mb, int plane_size_factor,
-                               const int16_t *qmat)
+                               const int16_t *qmat, int is_chroma)
  {
      GetBitContext gb;
-    DCTELEM *block_ptr;
+    int16_t *block_ptr;
      int mb_num, blocks_per_slice;
  
      blocks_per_slice = mbs_per_slice * blocks_per_mb;
@@ -431,18 +434,33 @@ static void decode_slice_plane(ProresContext *ctx, ProresThreadData *td,
      /* inverse quantization, inverse transform and output */
      block_ptr = td->blocks;
  
-    for (mb_num = 0; mb_num < mbs_per_slice; mb_num++, out_ptr += blocks_per_mb * 4) {
-        ctx->dsp.idct_put(out_ptr,                    linesize, block_ptr, qmat);
-        block_ptr += 64;
-        if (blocks_per_mb > 2) {
-            ctx->dsp.idct_put(out_ptr + 8,            linesize, block_ptr, qmat);
+    if (!is_chroma) {
+        for (mb_num = 0; mb_num < mbs_per_slice; mb_num++, out_ptr += blocks_per_mb * 4) {
+            ctx->dsp.idct_put(out_ptr,                    linesize, block_ptr, qmat);
              block_ptr += 64;
+            if (blocks_per_mb > 2) {
+                ctx->dsp.idct_put(out_ptr + 8,            linesize, block_ptr, qmat);
+                block_ptr += 64;
+            }
+            ctx->dsp.idct_put(out_ptr + linesize * 4,     linesize, block_ptr, qmat);
+            block_ptr += 64;
+            if (blocks_per_mb > 2) {
+                ctx->dsp.idct_put(out_ptr + linesize * 4 + 8, linesize, block_ptr, qmat);
+                block_ptr += 64;
+            }
          }
-        ctx->dsp.idct_put(out_ptr + linesize * 4,     linesize, block_ptr, qmat);
-        block_ptr += 64;
-        if (blocks_per_mb > 2) {
-            ctx->dsp.idct_put(out_ptr + linesize * 4 + 8, linesize, block_ptr, qmat);
+    } else {
+        for (mb_num = 0; mb_num < mbs_per_slice; mb_num++, out_ptr += blocks_per_mb * 4) {
+            ctx->dsp.idct_put(out_ptr,                    linesize, block_ptr, qmat);
              block_ptr += 64;
+            ctx->dsp.idct_put(out_ptr + linesize * 4,     linesize, block_ptr, qmat);
+            block_ptr += 64;
+            if (blocks_per_mb > 2) {
+                ctx->dsp.idct_put(out_ptr + 8,            linesize, block_ptr, qmat);
+                block_ptr += 64;
+                ctx->dsp.idct_put(out_ptr + linesize * 4 + 8, linesize, block_ptr, qmat);
+                block_ptr += 64;
+            }
          }
      }
  }
@@ -459,7 +477,7 @@ static int decode_slice(AVCodecContext *avctx, void *tdata)
      int mbs_per_slice = td->slice_width;
      const uint8_t *buf;
      uint8_t *y_data, *u_data, *v_data;
-    AVFrame *pic = avctx->coded_frame;
+    AVFrame *pic = ctx->frame;
      int i, sf, slice_width_factor;
      int slice_data_size, hdr_size, y_data_size, u_data_size, v_data_size;
      int y_linesize, u_linesize, v_linesize;
@@ -523,7 +541,7 @@ static int decode_slice(AVCodecContext *avctx, void *tdata)
                         (uint16_t*) (y_data + (mb_y_pos << 4) * y_linesize +
                                      (mb_x_pos << 5)), y_linesize,
                         mbs_per_slice, 4, slice_width_factor + 2,
-                       td->qmat_luma_scaled);
+                       td->qmat_luma_scaled, 0);
  
      /* decode U chroma plane */
      decode_slice_plane(ctx, td, buf + hdr_size + y_data_size, u_data_size,
@@ -531,7 +549,7 @@ static int decode_slice(AVCodecContext *avctx, void *tdata)
                                      (mb_x_pos << ctx->mb_chroma_factor)),
                         u_linesize, mbs_per_slice, ctx->num_chroma_blocks,
                         slice_width_factor + ctx->chroma_factor - 1,
-                       td->qmat_chroma_scaled);
+                       td->qmat_chroma_scaled, 1);
  
      /* decode V chroma plane */
      decode_slice_plane(ctx, td, buf + hdr_size + y_data_size + u_data_size,
@@ -540,7 +558,7 @@ static int decode_slice(AVCodecContext *avctx, void *tdata)
                                      (mb_x_pos << ctx->mb_chroma_factor)),
                         v_linesize, mbs_per_slice, ctx->num_chroma_blocks,
                         slice_width_factor + ctx->chroma_factor - 1,
-                       td->qmat_chroma_scaled);
+                       td->qmat_chroma_scaled, 1);
  
      return 0;
  }
@@ -579,15 +597,18 @@ static int decode_picture(ProresContext *ctx, int pic_num,
  
  #define MOVE_DATA_PTR(nbytes) buf += (nbytes); buf_size -= (nbytes)
  
-static int decode_frame(AVCodecContext *avctx, void *data, int *data_size,
+static int decode_frame(AVCodecContext *avctx, void *data, int *got_frame,
                          AVPacket *avpkt)
  {
      ProresContext *ctx = avctx->priv_data;
-    AVFrame *picture   = avctx->coded_frame;
      const uint8_t *buf = avpkt->data;
      int buf_size       = avpkt->size;
      int frame_hdr_size, pic_num, pic_data_size;
  
+    ctx->frame            = data;
+    ctx->frame->pict_type = AV_PICTURE_TYPE_I;
+    ctx->frame->key_frame = 1;
+
      /* check frame atom container */
      if (buf_size < 28 || buf_size < AV_RB32(buf) ||
          AV_RB32(buf + 4) != FRAME_ID) {
@@ -603,14 +624,10 @@ static int decode_frame(AVCodecContext *avctx, void *data, int *data_size,
  
      MOVE_DATA_PTR(frame_hdr_size);
  
-    if (picture->data[0])
-        avctx->release_buffer(avctx, picture);
-
-    picture->reference = 0;
-    if (avctx->get_buffer(avctx, picture) < 0)
+    if (ff_get_buffer(avctx, ctx->frame, 0) < 0)
          return -1;
  
-    for (pic_num = 0; ctx->picture.interlaced_frame - pic_num + 1; pic_num++) {
+    for (pic_num = 0; ctx->frame->interlaced_frame - pic_num + 1; pic_num++) {
          pic_data_size = decode_picture_header(ctx, buf, buf_size, avctx);
          if (pic_data_size < 0)
              return AVERROR_INVALIDDATA;
@@ -621,8 +638,8 @@ static int decode_frame(AVCodecContext *avctx, void *data, int *data_size,
          MOVE_DATA_PTR(pic_data_size);
      }
  
-    *data_size       = sizeof(AVPicture);
-    *(AVFrame*) data = *avctx->coded_frame;
+    ctx->frame = NULL;
+    *got_frame = 1;
  
      return avpkt->size;
  }
@@ -632,9 +649,6 @@ static av_cold int decode_close(AVCodecContext *avctx)
  {
      ProresContext *ctx = avctx->priv_data;
  
-    if (ctx->picture.data[0])
-        avctx->release_buffer(avctx, &ctx->picture);
-
      av_freep(&ctx->slice_data);
  
      return 0;
@@ -644,7 +658,7 @@ static av_cold int decode_close(AVCodecContext *avctx)
  AVCodec ff_prores_decoder = {
      .name           = "prores",
      .type           = AVMEDIA_TYPE_VIDEO,
-    .id             = CODEC_ID_PRORES,
+    .id             = AV_CODEC_ID_PRORES,
      .priv_data_size = sizeof(ProresContext),
      .init           = decode_init,
      .close          = decode_close,