Merge remote-tracking branch 'qatar/master'

[ffmpeg] / libswscale / x86 / swscale_mmx.c
diff --git a/libswscale/x86/swscale_mmx.c b/libswscale/x86/swscale_mmx.c

index 9dffe2b203f38f275f6d78188dae65ec97c7025b..0c8732ce2a9af55f01028d18326bbf7afc39d720 100644 (file)
--- a/libswscale/x86/swscale_mmx.c
+++ b/libswscale/x86/swscale_mmx.c
@@ -27,6 +27,8 @@
  #include "libavutil/cpu.h"
  #include "libavutil/pixdesc.h"
  
+#define DITHER1XBPP
+
  DECLARE_ASM_CONST(8, uint64_t, bF8)=       0xF8F8F8F8F8F8F8F8LL;
  DECLARE_ASM_CONST(8, uint64_t, bFC)=       0xFCFCFCFCFCFCFCFCLL;
  DECLARE_ASM_CONST(8, uint64_t, w10)=       0x0010001000100010LL;
@@ -92,8 +94,8 @@ void updateMMXDitherTables(SwsContext *c, int dstY, int lumBufIndex, int chrBufI
      int16_t **alpPixBuf= c->alpPixBuf;
      const int vLumBufSize= c->vLumBufSize;
      const int vChrBufSize= c->vChrBufSize;
-    int16_t *vLumFilterPos= c->vLumFilterPos;
-    int16_t *vChrFilterPos= c->vChrFilterPos;
+    int32_t *vLumFilterPos= c->vLumFilterPos;
+    int32_t *vChrFilterPos= c->vChrFilterPos;
      int16_t *vLumFilter= c->vLumFilter;
      int16_t *vChrFilter= c->vChrFilter;
      int32_t *lumMmxFilter= c->lumMmxFilter;
@@ -116,6 +118,44 @@ void updateMMXDitherTables(SwsContext *c, int dstY, int lumBufIndex, int chrBufI
          const int16_t **chrUSrcPtr= (const int16_t **)(void*) chrUPixBuf + chrBufIndex + firstChrSrcY - lastInChrBuf + vChrBufSize;
          const int16_t **alpSrcPtr= (CONFIG_SWSCALE_ALPHA && alpPixBuf) ? (const int16_t **)(void*) alpPixBuf + lumBufIndex + firstLumSrcY - lastInLumBuf + vLumBufSize : NULL;
          int i;
+
+        if (firstLumSrcY < 0 || firstLumSrcY + vLumFilterSize > c->srcH) {
+            const int16_t **tmpY = (const int16_t **) lumPixBuf + 2 * vLumBufSize;
+            int neg = -firstLumSrcY, i, end = FFMIN(c->srcH - firstLumSrcY, vLumFilterSize);
+            for (i = 0; i < neg;            i++)
+                tmpY[i] = lumSrcPtr[neg];
+            for (     ; i < end;            i++)
+                tmpY[i] = lumSrcPtr[i];
+            for (     ; i < vLumFilterSize; i++)
+                tmpY[i] = tmpY[i-1];
+            lumSrcPtr = tmpY;
+
+            if (alpSrcPtr) {
+                const int16_t **tmpA = (const int16_t **) alpPixBuf + 2 * vLumBufSize;
+                for (i = 0; i < neg;            i++)
+                    tmpA[i] = alpSrcPtr[neg];
+                for (     ; i < end;            i++)
+                    tmpA[i] = alpSrcPtr[i];
+                for (     ; i < vLumFilterSize; i++)
+                    tmpA[i] = tmpA[i - 1];
+                alpSrcPtr = tmpA;
+            }
+        }
+        if (firstChrSrcY < 0 || firstChrSrcY + vChrFilterSize > c->chrSrcH) {
+            const int16_t **tmpU = (const int16_t **) chrUPixBuf + 2 * vChrBufSize;
+            int neg = -firstChrSrcY, i, end = FFMIN(c->chrSrcH - firstChrSrcY, vChrFilterSize);
+            for (i = 0; i < neg;            i++) {
+                tmpU[i] = chrUSrcPtr[neg];
+            }
+            for (     ; i < end;            i++) {
+                tmpU[i] = chrUSrcPtr[i];
+            }
+            for (     ; i < vChrFilterSize; i++) {
+                tmpU[i] = tmpU[i - 1];
+            }
+            chrUSrcPtr = tmpU;
+        }
+
          if (flags & SWS_ACCURATE_RND) {
              int s= APCK_SIZE / 8;
              for (i=0; i<vLumFilterSize; i+=2) {
@@ -226,7 +266,7 @@ extern void ff_hscale ## from_bpc ## to ## to_bpc ## _ ## filter_n ## _ ## opt(
                                                  SwsContext *c, int16_t *data, \
                                                  int dstW, const uint8_t *src, \
                                                  const int16_t *filter, \
-                                                const int16_t *filterPos, int filterSize)
+                                                const int32_t *filterPos, int filterSize)
  
  #define SCALE_FUNCS(filter_n, opt) \
      SCALE_FUNC(filter_n,  8, 15, opt); \
@@ -306,6 +346,10 @@ extern void ff_ ## fmt ## ToUV_ ## opt(uint8_t *dstU, uint8_t *dstV, \
      INPUT_FUNC(yuyv, opt); \
      INPUT_UV_FUNC(nv12, opt); \
      INPUT_UV_FUNC(nv21, opt); \
+    INPUT_FUNC(rgba, opt); \
+    INPUT_FUNC(bgra, opt); \
+    INPUT_FUNC(argb, opt); \
+    INPUT_FUNC(abgr, opt); \
      INPUT_FUNC(rgb24, opt); \
      INPUT_FUNC(bgr24, opt)
  
@@ -358,9 +402,9 @@ void ff_sws_init_swScale_mmx(SwsContext *c)
      }
  #define ASSIGN_VSCALEX_FUNC(vscalefn, opt, do_16_case) \
  switch(c->dstBpc){ \
-    case 16:                          /*do_16_case;*/                          break; \
-    case 10: if (!isBE(c->dstFormat)) /*vscalefn = ff_yuv2planeX_10_ ## opt;*/ break; \
-    case 9:  if (!isBE(c->dstFormat)) /*vscalefn = ff_yuv2planeX_9_  ## opt;*/ break; \
+    case 16:                          do_16_case;                          break; \
+    case 10: if (!isBE(c->dstFormat)) vscalefn = ff_yuv2planeX_10_ ## opt; break; \
+    case 9:  if (!isBE(c->dstFormat)) vscalefn = ff_yuv2planeX_9_  ## opt; break; \
      default:                          /*vscalefn = ff_yuv2planeX_8_  ## opt;*/ break; \
      }
  #define ASSIGN_VSCALE_FUNC(vscalefn, opt1, opt2, opt2chk) \
@@ -404,6 +448,10 @@ switch(c->dstBpc){ \
              break;
          case_rgb(rgb24, RGB24, mmx);
          case_rgb(bgr24, BGR24, mmx);
+        case_rgb(bgra,  BGRA,  mmx);
+        case_rgb(rgba,  RGBA,  mmx);
+        case_rgb(abgr,  ABGR,  mmx);
+        case_rgb(argb,  ARGB,  mmx);
          default:
              break;
          }
@@ -448,6 +496,10 @@ switch(c->dstBpc){ \
              break;
          case_rgb(rgb24, RGB24, sse2);
          case_rgb(bgr24, BGR24, sse2);
+        case_rgb(bgra,  BGRA,  sse2);
+        case_rgb(rgba,  RGBA,  sse2);
+        case_rgb(abgr,  ABGR,  sse2);
+        case_rgb(argb,  ARGB,  sse2);
          default:
              break;
          }
@@ -491,6 +543,10 @@ switch(c->dstBpc){ \
              break;
          case_rgb(rgb24, RGB24, avx);
          case_rgb(bgr24, BGR24, avx);
+        case_rgb(bgra,  BGRA,  avx);
+        case_rgb(rgba,  RGBA,  avx);
+        case_rgb(abgr,  ABGR,  avx);
+        case_rgb(argb,  ARGB,  avx);
          default:
              break;
          }