mpegaudiodec: move imdct and windowing function to mpegaudiodsp

[ffmpeg] / libavcodec / wmavoice.c
diff --git a/libavcodec/wmavoice.c b/libavcodec/wmavoice.c

index 5e7f8a6739e43f8e6c3c03d6f5acdec434405855..8854e35d936dea5aa1cf7bda1ad2c87f03436d9e 100644 (file)
--- a/libavcodec/wmavoice.c
+++ b/libavcodec/wmavoice.c
@@ -25,6 +25,8 @@
   * @author Ronald S. Bultje <rsbultje@gmail.com>
   */
  
+#define UNCHECKED_BITSTREAM_READER 1
+
  #include <math.h>
  #include "avcodec.h"
  #include "get_bits.h"
@@ -36,8 +38,9 @@
  #include "acelp_filters.h"
  #include "lsp.h"
  #include "libavutil/lzo.h"
-#include "avfft.h"
-#include "fft.h"
+#include "dct.h"
+#include "rdft.h"
+#include "sinewin.h"
  
  #define MAX_BLOCKS           8   ///< maximum number of blocks per frame
  #define MAX_LSPS             16  ///< maximum filter order
@@ -127,11 +130,10 @@ static const struct frame_type_desc {
   */
  typedef struct {
      /**
-     * @defgroup struct_global Global values
-     * Global values, specified in the stream header / extradata or used
-     * all over.
+     * @name Global values specified in the stream header / extradata or used all over.
       * @{
       */
+    AVFrame frame;
      GetBitContext gb;             ///< packet bitreader. During decoder init,
                                    ///< it contains the extradata from the
                                    ///< demuxer. During decoding, it contains
@@ -181,14 +183,15 @@ typedef struct {
  
      /**
       * @}
-     * @defgroup struct_packet Packet values
-     * Packet values, specified in the packet header or related to a packet.
+     *
+     * @name Packet values specified in the packet header or related to a packet.
+     *
       * A packet is considered to be a single unit of data provided to this
       * decoder by the demuxer.
       * @{
       */
      int spillover_nbits;          ///< number of bits of the previous packet's
-                                  ///< last superframe preceeding this
+                                  ///< last superframe preceding this
                                    ///< packet's first full superframe (useful
                                    ///< for re-synchronization also)
      int has_residual_lsps;        ///< if set, superframes contain one set of
@@ -212,7 +215,8 @@ typedef struct {
  
      /**
       * @}
-     * @defgroup struct_frame Frame and superframe values
+     *
+     * @name Frame and superframe values
       * Superframe and frame data - these can change from frame to frame,
       * although some of them do in that case serve as a cache / history for
       * the next frame or superframe.
@@ -255,7 +259,9 @@ typedef struct {
      float synth_history[MAX_LSPS]; ///< see #excitation_history
      /**
       * @}
-     * @defgroup post_filter Postfilter values
+     *
+     * @name Postfilter values
+     *
       * Variables used for postfilter implementation, mostly history for
       * smoothing and so on, and context variables for FFT/iFFT.
       * @{
@@ -274,11 +280,11 @@ typedef struct {
                                    ///< by postfilter
      float denoise_filter_cache[MAX_FRAMESIZE];
      int   denoise_filter_cache_size; ///< samples in #denoise_filter_cache
-    DECLARE_ALIGNED(16, float, tilted_lpcs_pf)[0x80];
+    DECLARE_ALIGNED(32, float, tilted_lpcs_pf)[0x80];
                                    ///< aligned buffer for LPC tilting
-    DECLARE_ALIGNED(16, float, denoise_coeffs_pf)[0x80];
+    DECLARE_ALIGNED(32, float, denoise_coeffs_pf)[0x80];
                                    ///< aligned buffer for denoise coefficients
-    DECLARE_ALIGNED(16, float, synth_filter_out_buf)[0x80 + MAX_LSPS_ALIGN16];
+    DECLARE_ALIGNED(32, float, synth_filter_out_buf)[0x80 + MAX_LSPS_ALIGN16];
                                    ///< aligned buffer for postfilter speech
                                    ///< synthesis
      /**
@@ -314,7 +320,7 @@ static av_cold int decode_vbmtree(GetBitContext *gb, int8_t vbm_tree[25])
      };
      int cntr[8], n, res;
  
-    memset(vbm_tree, 0xff, sizeof(vbm_tree));
+    memset(vbm_tree, 0xff, sizeof(vbm_tree[0]) * 25);
      memset(cntr,     0,    sizeof(cntr));
      for (n = 0; n < 17; n++) {
          res = get_bits(gb, 3);
@@ -398,6 +404,10 @@ static av_cold int wmavoice_decode_init(AVCodecContext *ctx)
      s->min_pitch_val    = ((ctx->sample_rate << 8)      /  400 + 50) >> 8;
      s->max_pitch_val    = ((ctx->sample_rate << 8) * 37 / 2000 + 50) >> 8;
      pitch_range         = s->max_pitch_val - s->min_pitch_val;
+    if (pitch_range <= 0) {
+        av_log(ctx, AV_LOG_ERROR, "Invalid pitch range; broken extradata?\n");
+        return -1;
+    }
      s->pitch_nbits      = av_ceil_log2(pitch_range);
      s->last_pitch_val   = 40;
      s->last_acb_type    = ACB_TYPE_NONE;
@@ -419,6 +429,10 @@ static av_cold int wmavoice_decode_init(AVCodecContext *ctx)
      s->block_conv_table[2]      = (pitch_range * 44) >> 6;
      s->block_conv_table[3]      = s->max_pitch_val - 1;
      s->block_delta_pitch_hrange = (pitch_range >> 3) & ~0xF;
+    if (s->block_delta_pitch_hrange <= 0) {
+        av_log(ctx, AV_LOG_ERROR, "Invalid delta pitch hrange; broken extradata?\n");
+        return -1;
+    }
      s->block_delta_pitch_nbits  = 1 + av_ceil_log2(s->block_delta_pitch_hrange);
      s->block_pitch_range        = s->block_conv_table[2] +
                                    s->block_conv_table[3] + 1 +
@@ -427,11 +441,14 @@ static av_cold int wmavoice_decode_init(AVCodecContext *ctx)
  
      ctx->sample_fmt             = AV_SAMPLE_FMT_FLT;
  
+    avcodec_get_frame_defaults(&s->frame);
+    ctx->coded_frame = &s->frame;
+
      return 0;
  }
  
  /**
- * @defgroup postfilter Postfilter functions
+ * @name Postfilter functions
   * Postfilter functions (gain control, wiener denoise filter, DC filter,
   * kalman smoothening, plus surrounding code to wrap it)
   * @{
@@ -558,7 +575,7 @@ static void calc_input_response(WMAVoiceContext *s, float *lpcs,
      int n, idx;
  
      /* Create frequency power spectrum of speech input (i.e. RDFT of LPCs) */
-    ff_rdft_calc(&s->rdft, lpcs);
+    s->rdft.rdft_calc(&s->rdft, lpcs);
  #define log_range(var, assign) do { \
          float tmp = log10f(assign);  var = tmp; \
          max       = FFMAX(max, tmp); min = FFMIN(min, tmp); \
@@ -601,8 +618,8 @@ static void calc_input_response(WMAVoiceContext *s, float *lpcs,
       * is a sinus input) by doing a phase shift (in theory, H(sin())=cos()).
       * Hilbert_Transform(RDFT(x)) = Laplace_Transform(x), which calculates the
       * "moment" of the LPCs in this filter. */
-    ff_dct_calc(&s->dct, lpcs);
-    ff_dct_calc(&s->dst, lpcs);
+    s->dct.dct_calc(&s->dct, lpcs);
+    s->dst.dct_calc(&s->dst, lpcs);
  
      /* Split out the coefficient indexes into phase/magnitude pairs */
      idx = 255 + av_clip(lpcs[64],               -255, 255);
@@ -623,7 +640,7 @@ static void calc_input_response(WMAVoiceContext *s, float *lpcs,
      coeffs[1] = last_coeff;
  
      /* move into real domain */
-    ff_rdft_calc(&s->irdft, coeffs);
+    s->irdft.rdft_calc(&s->irdft, coeffs);
  
      /* tilt correction and normalize scale */
      memset(&coeffs[remainder], 0, sizeof(coeffs[0]) * (128 - remainder));
@@ -693,8 +710,8 @@ static void wiener_denoise(WMAVoiceContext *s, int fcb_type,
          /* apply coefficients (in frequency spectrum domain), i.e. complex
           * number multiplication */
          memset(&synth_pf[size], 0, sizeof(synth_pf[0]) * (128 - size));
-        ff_rdft_calc(&s->rdft, synth_pf);
-        ff_rdft_calc(&s->rdft, coeffs);
+        s->rdft.rdft_calc(&s->rdft, synth_pf);
+        s->rdft.rdft_calc(&s->rdft, coeffs);
          synth_pf[0] *= coeffs[0];
          synth_pf[1] *= coeffs[1];
          for (n = 1; n < 64; n++) {
@@ -702,7 +719,7 @@ static void wiener_denoise(WMAVoiceContext *s, int fcb_type,
              synth_pf[n * 2]     = v1 * coeffs[n * 2] - v2 * coeffs[n * 2 + 1];
              synth_pf[n * 2 + 1] = v2 * coeffs[n * 2] + v1 * coeffs[n * 2 + 1];
          }
-        ff_rdft_calc(&s->irdft, synth_pf);
+        s->irdft.rdft_calc(&s->irdft, synth_pf);
      }
  
      /* merge filter output with the history of previous runs */
@@ -824,7 +841,7 @@ static void dequant_lsps(double *lsps, int num,
  }
  
  /**
- * @defgroup lsp_dequant LSP dequantization routines
+ * @name LSP dequantization routines
   * LSP dequantization routines, for 10/16LSPs and independent/residual coding.
   * @note we assume enough bits are available, caller should check.
   * lsp10i() consumes 24 bits; lsp10r() consumes an additional 24 bits;
@@ -968,7 +985,7 @@ static void dequant_lsp16r(GetBitContext *gb,
  
  /**
   * @}
- * @defgroup aw Pitch-adaptive window coding functions
+ * @name Pitch-adaptive window coding functions
   * The next few functions are for pitch-adaptive window coding.
   * @{
   */
@@ -1074,7 +1091,7 @@ static void aw_pulse_set2(WMAVoiceContext *s, GetBitContext *gb,
              int excl_range         = s->aw_pulse_range; // always 16 or 24
              uint16_t *use_mask_ptr = &use_mask[idx >> 4];
              int first_sh           = 16 - (idx & 15);
-            *use_mask_ptr++       &= 0xFFFF << first_sh;
+            *use_mask_ptr++       &= 0xFFFFu << first_sh;
              excl_range            -= first_sh;
              if (excl_range >= 16) {
                  *use_mask_ptr++    = 0;
@@ -1714,8 +1731,7 @@ static int check_bits_for_superframe(GetBitContext *orig_gb,
   * @return 0 on success, <0 on error or 1 if there was not enough data to
   *         fully parse the superframe
   */
-static int synth_superframe(AVCodecContext *ctx,
-                            float *samples, int *data_size)
+static int synth_superframe(AVCodecContext *ctx, int *got_frame_ptr)
  {
      WMAVoiceContext *s = ctx->priv_data;
      GetBitContext *gb = &s->gb, s_gb;
@@ -1725,6 +1741,7 @@ static int synth_superframe(AVCodecContext *ctx,
          wmavoice_mean_lsf16[s->lsp_def_mode] : wmavoice_mean_lsf10[s->lsp_def_mode];
      float excitation[MAX_SIGNAL_HISTORY + MAX_SFRAMESIZE + 12];
      float synth[MAX_LSPS + MAX_SFRAMESIZE];
+    float *samples;
  
      memcpy(synth,      s->synth_history,
             s->lsps             * sizeof(*synth));
@@ -1737,7 +1754,10 @@ static int synth_superframe(AVCodecContext *ctx,
          s->sframe_cache_size = 0;
      }
  
-    if ((res = check_bits_for_superframe(gb, s)) == 1) return 1;
+    if ((res = check_bits_for_superframe(gb, s)) == 1) {
+        *got_frame_ptr = 0;
+        return 1;
+    }
  
      /* First bit is speech/music bit, it differentiates between WMAVoice
       * speech samples (the actual codec) and WMAVoice music samples, which
@@ -1778,7 +1798,16 @@ static int synth_superframe(AVCodecContext *ctx,
              stabilize_lsps(lsps[n], s->lsps);
      }
  
-    /* Parse frames, optionally preceeded by per-frame (independent) LSPs. */
+    /* get output buffer */
+    s->frame.nb_samples = 480;
+    if ((res = ctx->get_buffer(ctx, &s->frame)) < 0) {
+        av_log(ctx, AV_LOG_ERROR, "get_buffer() failed\n");
+        return res;
+    }
+    s->frame.nb_samples = n_samples;
+    samples = (float *)s->frame.data[0];
+
+    /* Parse frames, optionally preceded by per-frame (independent) LSPs. */
      for (n = 0; n < 3; n++) {
          if (!s->has_residual_lsps) {
              int m;
@@ -1797,8 +1826,10 @@ static int synth_superframe(AVCodecContext *ctx,
                                 &samples[n * MAX_FRAMESIZE],
                                 lsps[n], n == 0 ? s->prev_lsps : lsps[n - 1],
                                 &excitation[s->history_nsamples + n * MAX_FRAMESIZE],
-                               &synth[s->lsps + n * MAX_FRAMESIZE])))
+                               &synth[s->lsps + n * MAX_FRAMESIZE]))) {
+            *got_frame_ptr = 0;
              return res;
+        }
      }
  
      /* Statistics? FIXME - we don't check for length, a slight overrun
@@ -1809,8 +1840,7 @@ static int synth_superframe(AVCodecContext *ctx,
          skip_bits(gb, 10 * (res + 1));
      }
  
-    /* Specify nr. of output samples */
-    *data_size = n_samples * sizeof(float);
+    *got_frame_ptr = 1;
  
      /* Update history */
      memcpy(s->prev_lsps,           lsps[2],
@@ -1861,7 +1891,7 @@ static int parse_packet_header(WMAVoiceContext *s)
   * @param size size of the source data, in bytes
   * @param gb bit I/O context specifying the current position in the source.
   *           data. This function might use this to align the bit position to
- *           a whole-byte boundary before calling #ff_copy_bits() on aligned
+ *           a whole-byte boundary before calling #avpriv_copy_bits() on aligned
   *           source data
   * @param nbits the amount of bits to copy from source to target
   *
@@ -1877,10 +1907,12 @@ static void copy_bits(PutBitContext *pb,
      rmn_bits = rmn_bytes = get_bits_left(gb);
      if (rmn_bits < nbits)
          return;
+    if (nbits > pb->size_in_bits - put_bits_count(pb))
+        return;
      rmn_bits &= 7; rmn_bytes >>= 3;
      if ((rmn_bits = FFMIN(rmn_bits, nbits)) > 0)
          put_bits(pb, rmn_bits, get_bits(gb, rmn_bits));
-    ff_copy_bits(pb, data + size - rmn_bytes,
+    avpriv_copy_bits(pb, data + size - rmn_bytes,
                   FFMIN(nbits - rmn_bits, rmn_bytes << 3));
  }
  
@@ -1896,28 +1928,22 @@ static void copy_bits(PutBitContext *pb,
   * For more information about frames, see #synth_superframe().
   */
  static int wmavoice_decode_packet(AVCodecContext *ctx, void *data,
-                                  int *data_size, AVPacket *avpkt)
+                                  int *got_frame_ptr, AVPacket *avpkt)
  {
      WMAVoiceContext *s = ctx->priv_data;
      GetBitContext *gb = &s->gb;
      int size, res, pos;
  
-    if (*data_size < 480 * sizeof(float)) {
-        av_log(ctx, AV_LOG_ERROR,
-               "Output buffer too small (%d given - %zu needed)\n",
-               *data_size, 480 * sizeof(float));
-        return -1;
-    }
-    *data_size = 0;
-
      /* Packets are sometimes a multiple of ctx->block_align, with a packet
-     * header at each ctx->block_align bytes. However, FFmpeg's ASF demuxer
+     * header at each ctx->block_align bytes. However, Libav's ASF demuxer
       * feeds us ASF packets, which may concatenate multiple "codec" packets
       * in a single "muxer" packet, so we artificially emulate that by
       * capping the packet size at ctx->block_align. */
      for (size = avpkt->size; size > ctx->block_align; size -= ctx->block_align);
-    if (!size)
+    if (!size) {
+        *got_frame_ptr = 0;
          return 0;
+    }
      init_get_bits(&s->gb, avpkt->data, size << 3);
  
      /* size == ctx->block_align is used to indicate whether we are dealing with
@@ -1936,10 +1962,11 @@ static int wmavoice_decode_packet(AVCodecContext *ctx, void *data,
                  copy_bits(&s->pb, avpkt->data, size, gb, s->spillover_nbits);
                  flush_put_bits(&s->pb);
                  s->sframe_cache_size += s->spillover_nbits;
-                if ((res = synth_superframe(ctx, data, data_size)) == 0 &&
-                    *data_size > 0) {
+                if ((res = synth_superframe(ctx, got_frame_ptr)) == 0 &&
+                    *got_frame_ptr) {
                      cnt += s->spillover_nbits;
                      s->skip_bits_next = cnt & 7;
+                    *(AVFrame *)data = s->frame;
                      return cnt >> 3;
                  } else
                      skip_bits_long (gb, s->spillover_nbits - cnt +
@@ -1954,11 +1981,12 @@ static int wmavoice_decode_packet(AVCodecContext *ctx, void *data,
      s->sframe_cache_size = 0;
      s->skip_bits_next = 0;
      pos = get_bits_left(gb);
-    if ((res = synth_superframe(ctx, data, data_size)) < 0) {
+    if ((res = synth_superframe(ctx, got_frame_ptr)) < 0) {
          return res;
-    } else if (*data_size > 0) {
+    } else if (*got_frame_ptr) {
          int cnt = get_bits_count(gb);
          s->skip_bits_next = cnt & 7;
+        *(AVFrame *)data = s->frame;
          return cnt >> 3;
      } else if ((s->sframe_cache_size = pos) > 0) {
          /* rewind bit reader to start of last (incomplete) superframe... */
@@ -2019,15 +2047,14 @@ static av_cold void wmavoice_flush(AVCodecContext *ctx)
  }
  
  AVCodec ff_wmavoice_decoder = {
-    "wmavoice",
-    AVMEDIA_TYPE_AUDIO,
-    CODEC_ID_WMAVOICE,
-    sizeof(WMAVoiceContext),
-    wmavoice_decode_init,
-    NULL,
-    wmavoice_decode_end,
-    wmavoice_decode_packet,
-    CODEC_CAP_SUBFRAMES,
+    .name           = "wmavoice",
+    .type           = AVMEDIA_TYPE_AUDIO,
+    .id             = CODEC_ID_WMAVOICE,
+    .priv_data_size = sizeof(WMAVoiceContext),
+    .init           = wmavoice_decode_init,
+    .close          = wmavoice_decode_end,
+    .decode         = wmavoice_decode_packet,
+    .capabilities   = CODEC_CAP_SUBFRAMES | CODEC_CAP_DR1,
      .flush     = wmavoice_flush,
      .long_name = NULL_IF_CONFIG_SMALL("Windows Media Audio Voice"),
  };