swscale_internal.h 45.3 KB
Newer Older
1
/*
M
Michael Niedermayer 已提交
2
 * Copyright (C) 2001-2011 Michael Niedermayer <michaelni@gmx.at>
3 4 5
 *
 * This file is part of FFmpeg.
 *
6 7 8 9
 * FFmpeg is free software; you can redistribute it and/or
 * modify it under the terms of the GNU Lesser General Public
 * License as published by the Free Software Foundation; either
 * version 2.1 of the License, or (at your option) any later version.
10 11 12
 *
 * FFmpeg is distributed in the hope that it will be useful,
 * but WITHOUT ANY WARRANTY; without even the implied warranty of
13 14
 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
 * Lesser General Public License for more details.
15
 *
16 17
 * You should have received a copy of the GNU Lesser General Public
 * License along with FFmpeg; if not, write to the Free Software
18
 * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
19
 */
20

21 22
#ifndef SWSCALE_SWSCALE_INTERNAL_H
#define SWSCALE_SWSCALE_INTERNAL_H
23

24 25
#include "config.h"

26
#if HAVE_ALTIVEC_H
27 28 29
#include <altivec.h>
#endif

30 31
#include "version.h"

32
#include "libavutil/avassert.h"
33
#include "libavutil/avutil.h"
34
#include "libavutil/common.h"
35
#include "libavutil/intreadwrite.h"
M
Mans Rullgard 已提交
36
#include "libavutil/log.h"
37
#include "libavutil/pixfmt.h"
38
#include "libavutil/pixdesc.h"
39

40
#define STR(s) AV_TOSTRING(s) // AV_STRINGIFY is too long
41

42
#define YUVRGB_TABLE_HEADROOM 256
43
#define YUVRGB_TABLE_LUMA_HEADROOM 0
44

45
#define MAX_FILTER_SIZE SWS_MAX_FILTER_SIZE
46

47 48
#define DITHER1XBPP

49
#if HAVE_BIGENDIAN
50 51 52
#define ALT32_CORR (-1)
#else
#define ALT32_CORR   1
53 54
#endif

55
#if ARCH_X86_64
56
#   define APCK_PTR2  8
57 58 59
#   define APCK_COEF 16
#   define APCK_SIZE 24
#else
60 61
#   define APCK_PTR2  4
#   define APCK_COEF  8
62
#   define APCK_SIZE 16
63 64
#endif

65 66
#define RETCODE_USE_CASCADE -12345

67 68
struct SwsContext;

M
Michael Niedermayer 已提交
69 70 71 72 73
typedef enum SwsDither {
    SWS_DITHER_NONE = 0,
    SWS_DITHER_AUTO,
    SWS_DITHER_BAYER,
    SWS_DITHER_ED,
74 75
    SWS_DITHER_A_DITHER,
    SWS_DITHER_X_DITHER,
M
Michael Niedermayer 已提交
76 77 78
    NB_SWS_DITHER,
} SwsDither;

79 80 81
typedef enum SwsAlphaBlend {
    SWS_ALPHA_BLEND_NONE  = 0,
    SWS_ALPHA_BLEND_UNIFORM,
82
    SWS_ALPHA_BLEND_CHECKERBOARD,
83 84 85
    SWS_ALPHA_BLEND_NB,
} SwsAlphaBlend;

86
typedef int (*SwsFunc)(struct SwsContext *context, const uint8_t *src[],
87
                       int srcStride[], int srcSliceY, int srcSliceH,
88
                       uint8_t *dst[], int dstStride[]);
89

90
/**
91
 * Write one line of horizontally scaled data to planar output
92 93
 * without any additional vertical scaling (or point-scaling).
 *
94
 * @param src     scaled source data, 15bit for 8-10bit output,
95
 *                19-bit for 16bit output (in int32_t)
96
 * @param dest    pointer to the output plane. For >8bit
97
 *                output, this is in uint16_t
98 99 100
 * @param dstW    width of destination in pixels
 * @param dither  ordered dither array of type int16_t and size 8
 * @param offset  Dither offset
101
 */
102 103
typedef void (*yuv2planar1_fn)(const int16_t *src, uint8_t *dest, int dstW,
                               const uint8_t *dither, int offset);
104

105
/**
K
Kieran Kunhya 已提交
106
 * Write one line of horizontally scaled data to planar output
107 108
 * with multi-point vertical scaling between input pixels.
 *
K
Kieran Kunhya 已提交
109 110
 * @param filter        vertical luma/alpha scaling coefficients, 12bit [0,4096]
 * @param src           scaled luma (Y) or alpha (A) source data, 15bit for 8-10bit output,
111
 *                      19-bit for 16bit output (in int32_t)
K
Kieran Kunhya 已提交
112 113 114 115 116 117
 * @param filterSize    number of vertical input lines to scale
 * @param dest          pointer to output plane. For >8bit
 *                      output, this is in uint16_t
 * @param dstW          width of destination pixels
 * @param offset        Dither offset
 */
118 119 120
typedef void (*yuv2planarX_fn)(const int16_t *filter, int filterSize,
                               const int16_t **src, uint8_t *dest, int dstW,
                               const uint8_t *dither, int offset);
K
Kieran Kunhya 已提交
121 122 123 124 125 126

/**
 * Write one line of horizontally scaled chroma to interleaved output
 * with multi-point vertical scaling between input pixels.
 *
 * @param c             SWS scaling context
127
 * @param chrFilter     vertical chroma scaling coefficients, 12bit [0,4096]
128 129 130 131
 * @param chrUSrc       scaled chroma (U) source data, 15bit for 8-10bit output,
 *                      19-bit for 16bit output (in int32_t)
 * @param chrVSrc       scaled chroma (V) source data, 15bit for 8-10bit output,
 *                      19-bit for 16bit output (in int32_t)
132
 * @param chrFilterSize number of vertical chroma input lines to scale
K
Kieran Kunhya 已提交
133
 * @param dest          pointer to the output plane. For >8bit
134
 *                      output, this is in uint16_t
K
Kieran Kunhya 已提交
135
 * @param dstW          width of chroma planes
136
 */
137 138 139 140 141 142
typedef void (*yuv2interleavedX_fn)(struct SwsContext *c,
                                    const int16_t *chrFilter,
                                    int chrFilterSize,
                                    const int16_t **chrUSrc,
                                    const int16_t **chrVSrc,
                                    uint8_t *dest, int dstW);
K
Kieran Kunhya 已提交
143

144 145 146 147 148 149
/**
 * Write one line of horizontally scaled Y/U/V/A to packed-pixel YUV/RGB
 * output without any additional vertical scaling (or point-scaling). Note
 * that this function may do chroma scaling, see the "uvalpha" argument.
 *
 * @param c       SWS scaling context
150 151 152 153 154 155 156 157 158 159
 * @param lumSrc  scaled luma (Y) source data, 15bit for 8-10bit output,
 *                19-bit for 16bit output (in int32_t)
 * @param chrUSrc scaled chroma (U) source data, 15bit for 8-10bit output,
 *                19-bit for 16bit output (in int32_t)
 * @param chrVSrc scaled chroma (V) source data, 15bit for 8-10bit output,
 *                19-bit for 16bit output (in int32_t)
 * @param alpSrc  scaled alpha (A) source data, 15bit for 8-10bit output,
 *                19-bit for 16bit output (in int32_t)
 * @param dest    pointer to the output plane. For 16bit output, this is
 *                uint16_t
160 161 162 163 164 165 166 167 168 169 170 171 172
 * @param dstW    width of lumSrc and alpSrc in pixels, number of pixels
 *                to write into dest[]
 * @param uvalpha chroma scaling coefficient for the second line of chroma
 *                pixels, either 2048 or 0. If 0, one chroma input is used
 *                for 2 output pixels (or if the SWS_FLAG_FULL_CHR_INT flag
 *                is set, it generates 1 output pixel). If 2048, two chroma
 *                input pixels should be averaged for 2 output pixels (this
 *                only happens if SWS_FLAG_FULL_CHR_INT is not set)
 * @param y       vertical line number for this output. This does not need
 *                to be used to calculate the offset in the destination,
 *                but can be used to generate comfort noise using dithering
 *                for some output formats.
 */
173 174 175 176 177
typedef void (*yuv2packed1_fn)(struct SwsContext *c, const int16_t *lumSrc,
                               const int16_t *chrUSrc[2],
                               const int16_t *chrVSrc[2],
                               const int16_t *alpSrc, uint8_t *dest,
                               int dstW, int uvalpha, int y);
178 179 180 181 182
/**
 * Write one line of horizontally scaled Y/U/V/A to packed-pixel YUV/RGB
 * output by doing bilinear scaling between two input lines.
 *
 * @param c       SWS scaling context
183 184 185 186 187 188 189 190 191 192
 * @param lumSrc  scaled luma (Y) source data, 15bit for 8-10bit output,
 *                19-bit for 16bit output (in int32_t)
 * @param chrUSrc scaled chroma (U) source data, 15bit for 8-10bit output,
 *                19-bit for 16bit output (in int32_t)
 * @param chrVSrc scaled chroma (V) source data, 15bit for 8-10bit output,
 *                19-bit for 16bit output (in int32_t)
 * @param alpSrc  scaled alpha (A) source data, 15bit for 8-10bit output,
 *                19-bit for 16bit output (in int32_t)
 * @param dest    pointer to the output plane. For 16bit output, this is
 *                uint16_t
193 194 195 196 197 198 199 200 201 202 203 204 205
 * @param dstW    width of lumSrc and alpSrc in pixels, number of pixels
 *                to write into dest[]
 * @param yalpha  luma/alpha scaling coefficients for the second input line.
 *                The first line's coefficients can be calculated by using
 *                4096 - yalpha
 * @param uvalpha chroma scaling coefficient for the second input line. The
 *                first line's coefficients can be calculated by using
 *                4096 - uvalpha
 * @param y       vertical line number for this output. This does not need
 *                to be used to calculate the offset in the destination,
 *                but can be used to generate comfort noise using dithering
 *                for some output formats.
 */
206 207 208 209 210 211
typedef void (*yuv2packed2_fn)(struct SwsContext *c, const int16_t *lumSrc[2],
                               const int16_t *chrUSrc[2],
                               const int16_t *chrVSrc[2],
                               const int16_t *alpSrc[2],
                               uint8_t *dest,
                               int dstW, int yalpha, int uvalpha, int y);
212 213 214 215 216 217
/**
 * Write one line of horizontally scaled Y/U/V/A to packed-pixel YUV/RGB
 * output by doing multi-point vertical scaling between input pixels.
 *
 * @param c             SWS scaling context
 * @param lumFilter     vertical luma/alpha scaling coefficients, 12bit [0,4096]
218 219
 * @param lumSrc        scaled luma (Y) source data, 15bit for 8-10bit output,
 *                      19-bit for 16bit output (in int32_t)
220 221
 * @param lumFilterSize number of vertical luma/alpha input lines to scale
 * @param chrFilter     vertical chroma scaling coefficients, 12bit [0,4096]
222 223 224 225
 * @param chrUSrc       scaled chroma (U) source data, 15bit for 8-10bit output,
 *                      19-bit for 16bit output (in int32_t)
 * @param chrVSrc       scaled chroma (V) source data, 15bit for 8-10bit output,
 *                      19-bit for 16bit output (in int32_t)
226
 * @param chrFilterSize number of vertical chroma input lines to scale
227 228 229 230
 * @param alpSrc        scaled alpha (A) source data, 15bit for 8-10bit output,
 *                      19-bit for 16bit output (in int32_t)
 * @param dest          pointer to the output plane. For 16bit output, this is
 *                      uint16_t
231 232 233 234 235 236 237
 * @param dstW          width of lumSrc and alpSrc in pixels, number of pixels
 *                      to write into dest[]
 * @param y             vertical line number for this output. This does not need
 *                      to be used to calculate the offset in the destination,
 *                      but can be used to generate comfort noise using dithering
 *                      or some output formats.
 */
238 239 240 241 242 243 244
typedef void (*yuv2packedX_fn)(struct SwsContext *c, const int16_t *lumFilter,
                               const int16_t **lumSrc, int lumFilterSize,
                               const int16_t *chrFilter,
                               const int16_t **chrUSrc,
                               const int16_t **chrVSrc, int chrFilterSize,
                               const int16_t **alpSrc, uint8_t *dest,
                               int dstW, int y);
245

M
Michael Niedermayer 已提交
246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272
/**
 * Write one line of horizontally scaled Y/U/V/A to YUV/RGB
 * output by doing multi-point vertical scaling between input pixels.
 *
 * @param c             SWS scaling context
 * @param lumFilter     vertical luma/alpha scaling coefficients, 12bit [0,4096]
 * @param lumSrc        scaled luma (Y) source data, 15bit for 8-10bit output,
 *                      19-bit for 16bit output (in int32_t)
 * @param lumFilterSize number of vertical luma/alpha input lines to scale
 * @param chrFilter     vertical chroma scaling coefficients, 12bit [0,4096]
 * @param chrUSrc       scaled chroma (U) source data, 15bit for 8-10bit output,
 *                      19-bit for 16bit output (in int32_t)
 * @param chrVSrc       scaled chroma (V) source data, 15bit for 8-10bit output,
 *                      19-bit for 16bit output (in int32_t)
 * @param chrFilterSize number of vertical chroma input lines to scale
 * @param alpSrc        scaled alpha (A) source data, 15bit for 8-10bit output,
 *                      19-bit for 16bit output (in int32_t)
 * @param dest          pointer to the output planes. For 16bit output, this is
 *                      uint16_t
 * @param dstW          width of lumSrc and alpSrc in pixels, number of pixels
 *                      to write into dest[]
 * @param y             vertical line number for this output. This does not need
 *                      to be used to calculate the offset in the destination,
 *                      but can be used to generate comfort noise using dithering
 *                      or some output formats.
 */
typedef void (*yuv2anyX_fn)(struct SwsContext *c, const int16_t *lumFilter,
273 274 275 276 277 278
                            const int16_t **lumSrc, int lumFilterSize,
                            const int16_t *chrFilter,
                            const int16_t **chrUSrc,
                            const int16_t **chrVSrc, int chrFilterSize,
                            const int16_t **alpSrc, uint8_t **dest,
                            int dstW, int y);
M
Michael Niedermayer 已提交
279

280 281 282
struct SwsSlice;
struct SwsFilterDescriptor;

283
/* This struct should be aligned on at least a 32-byte boundary. */
R
Ramiro Polla 已提交
284
typedef struct SwsContext {
285 286 287
    /**
     * info on struct for av_log
     */
288
    const AVClass *av_class;
289 290

    /**
D
Diego Biurrun 已提交
291 292
     * Note that src, dst, srcStride, dstStride will be copied in the
     * sws_scale() wrapper so they can be freely modified here.
293
     */
294
    SwsFunc swscale;
R
Ramiro Polla 已提交
295 296 297 298 299 300 301
    int srcW;                     ///< Width  of source      luma/alpha planes.
    int srcH;                     ///< Height of source      luma/alpha planes.
    int dstH;                     ///< Height of destination luma/alpha planes.
    int chrSrcW;                  ///< Width  of source      chroma     planes.
    int chrSrcH;                  ///< Height of source      chroma     planes.
    int chrDstW;                  ///< Width  of destination chroma     planes.
    int chrDstH;                  ///< Height of destination chroma     planes.
302 303
    int lumXInc, chrXInc;
    int lumYInc, chrYInc;
304 305
    enum AVPixelFormat dstFormat; ///< Destination pixel format.
    enum AVPixelFormat srcFormat; ///< Source      pixel format.
306 307
    int dstFormatBpp;             ///< Number of bits per pixel of the destination pixel format.
    int srcFormatBpp;             ///< Number of bits per pixel of the source      pixel format.
308
    int dstBpc, srcBpc;
R
Ramiro Polla 已提交
309 310 311 312 313 314
    int chrSrcHSubSample;         ///< Binary logarithm of horizontal subsampling factor between luma/alpha and chroma planes in source      image.
    int chrSrcVSubSample;         ///< Binary logarithm of vertical   subsampling factor between luma/alpha and chroma planes in source      image.
    int chrDstHSubSample;         ///< Binary logarithm of horizontal subsampling factor between luma/alpha and chroma planes in destination image.
    int chrDstVSubSample;         ///< Binary logarithm of vertical   subsampling factor between luma/alpha and chroma planes in destination image.
    int vChrDrop;                 ///< Binary logarithm of extra vertical subsampling factor in source image chroma planes specified by user.
    int sliceDir;                 ///< Direction that slices are fed to the scaler (1 = top-to-bottom, -1 = bottom-to-top).
R
Ramiro Polla 已提交
315
    double param[2];              ///< Input parameters for scaling algorithms that need them.
316

317 318 319 320
    /* The cascaded_* fields allow spliting a scaler task into multiple
     * sequential steps, this is for example used to limit the maximum
     * downscaling factor that needs to be supported in one scaler.
     */
321
    struct SwsContext *cascaded_context[3];
322 323
    int cascaded_tmpStride[4];
    uint8_t *cascaded_tmp[4];
324 325
    int cascaded1_tmpStride[4];
    uint8_t *cascaded1_tmp[4];
326
    int cascaded_mainindex;
327 328 329

    double gamma_value;
    int gamma_flag;
330
    int is_internal_gamma;
331 332
    uint16_t *gamma;
    uint16_t *inv_gamma;
333

334 335 336 337 338 339
    int numDesc;
    int descIndex[2];
    int numSlice;
    struct SwsSlice *slice;
    struct SwsFilterDescriptor *desc;

340 341 342
    uint32_t pal_yuv[256];
    uint32_t pal_rgb[256];

R
Ramiro Polla 已提交
343 344 345 346 347 348 349 350 351 352 353
    /**
     * @name Scaled horizontal lines ring buffer.
     * The horizontal scaler keeps just enough scaled lines in a ring buffer
     * so they may be passed to the vertical scaler. The pointers to the
     * allocated buffers for each line are duplicated in sequence in the ring
     * buffer to simplify indexing and avoid wrapping around between lines
     * inside the vertical scaler code. The wrapping is done before the
     * vertical scaler is called.
     */
    //@{
    int16_t **lumPixBuf;          ///< Ring buffer for scaled horizontal luma   plane lines to be fed to the vertical scaler.
354 355
    int16_t **chrUPixBuf;         ///< Ring buffer for scaled horizontal chroma plane lines to be fed to the vertical scaler.
    int16_t **chrVPixBuf;         ///< Ring buffer for scaled horizontal chroma plane lines to be fed to the vertical scaler.
R
Ramiro Polla 已提交
356
    int16_t **alpPixBuf;          ///< Ring buffer for scaled horizontal alpha  plane lines to be fed to the vertical scaler.
357 358 359 360 361 362
    int vLumBufSize;              ///< Number of vertical luma/alpha lines allocated in the ring buffer.
    int vChrBufSize;              ///< Number of vertical chroma     lines allocated in the ring buffer.
    int lastInLumBuf;             ///< Last scaled horizontal luma/alpha line from source in the ring buffer.
    int lastInChrBuf;             ///< Last scaled horizontal chroma     line from source in the ring buffer.
    int lumBufIndex;              ///< Index in ring buffer of the last scaled horizontal luma/alpha line from source.
    int chrBufIndex;              ///< Index in ring buffer of the last scaled horizontal chroma     line from source.
R
Ramiro Polla 已提交
363
    //@}
364

365
    uint8_t *formatConvBuffer;
366

R
Ramiro Polla 已提交
367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384
    /**
     * @name Horizontal and vertical filters.
     * To better understand the following fields, here is a pseudo-code of
     * their usage in filtering a horizontal line:
     * @code
     * for (i = 0; i < width; i++) {
     *     dst[i] = 0;
     *     for (j = 0; j < filterSize; j++)
     *         dst[i] += src[ filterPos[i] + j ] * filter[ filterSize * i + j ];
     *     dst[i] >>= FRAC_BITS; // The actual implementation is fixed-point.
     * }
     * @endcode
     */
    //@{
    int16_t *hLumFilter;          ///< Array of horizontal filter coefficients for luma/alpha planes.
    int16_t *hChrFilter;          ///< Array of horizontal filter coefficients for chroma     planes.
    int16_t *vLumFilter;          ///< Array of vertical   filter coefficients for luma/alpha planes.
    int16_t *vChrFilter;          ///< Array of vertical   filter coefficients for chroma     planes.
385 386 387 388
    int32_t *hLumFilterPos;       ///< Array of horizontal filter starting positions for each dst[i] for luma/alpha planes.
    int32_t *hChrFilterPos;       ///< Array of horizontal filter starting positions for each dst[i] for chroma     planes.
    int32_t *vLumFilterPos;       ///< Array of vertical   filter starting positions for each dst[i] for luma/alpha planes.
    int32_t *vChrFilterPos;       ///< Array of vertical   filter starting positions for each dst[i] for chroma     planes.
389 390 391 392
    int hLumFilterSize;           ///< Horizontal filter size for luma/alpha pixels.
    int hChrFilterSize;           ///< Horizontal filter size for chroma     pixels.
    int vLumFilterSize;           ///< Vertical   filter size for luma/alpha pixels.
    int vChrFilterSize;           ///< Vertical   filter size for chroma     pixels.
R
Ramiro Polla 已提交
393
    //@}
394

395 396 397 398
    int lumMmxextFilterCodeSize;  ///< Runtime-generated MMXEXT horizontal fast bilinear scaler code size for luma/alpha planes.
    int chrMmxextFilterCodeSize;  ///< Runtime-generated MMXEXT horizontal fast bilinear scaler code size for chroma planes.
    uint8_t *lumMmxextFilterCode; ///< Runtime-generated MMXEXT horizontal fast bilinear scaler code for luma/alpha planes.
    uint8_t *chrMmxextFilterCode; ///< Runtime-generated MMXEXT horizontal fast bilinear scaler code for chroma planes.
399

400
    int canMMXEXTBeUsed;
401

R
Ramiro Polla 已提交
402 403
    int dstY;                     ///< Last destination vertical line output from last slice.
    int flags;                    ///< Flags passed by the user to select scaler algorithm, optimizations, subsampling, etc...
404
    void *yuvTable;             // pointer to the yuv->rgb table start so it can be freed()
405 406 407
    // alignment ensures the offset can be added in a single
    // instruction on e.g. ARM
    DECLARE_ALIGNED(16, int, table_gV)[256 + 2*YUVRGB_TABLE_HEADROOM];
408 409 410
    uint8_t *table_rV[256 + 2*YUVRGB_TABLE_HEADROOM];
    uint8_t *table_gU[256 + 2*YUVRGB_TABLE_HEADROOM];
    uint8_t *table_bU[256 + 2*YUVRGB_TABLE_HEADROOM];
M
Michael Niedermayer 已提交
411
    DECLARE_ALIGNED(16, int32_t, input_rgb2yuv_table)[16+40*4]; // This table can contain both C and SIMD formatted values, the C vales are always at the XY_IDX points
412 413 414 415 416 417 418 419 420
#define RY_IDX 0
#define GY_IDX 1
#define BY_IDX 2
#define RU_IDX 3
#define GU_IDX 4
#define BU_IDX 5
#define RV_IDX 6
#define GV_IDX 7
#define BV_IDX 8
421
#define RGB2YUV_SHIFT 15
422

423 424
    int *dither_error[4];

425 426 427 428
    //Colorspace stuff
    int contrast, brightness, saturation;    // for sws_getColorspaceDetails
    int srcColorspaceTable[4];
    int dstColorspaceTable[4];
R
Ramiro Polla 已提交
429 430
    int srcRange;                 ///< 0 = MPG YUV range, 1 = JPG YUV range (source      image).
    int dstRange;                 ///< 0 = MPG YUV range, 1 = JPG YUV range (destination image).
431 432
    int src0Alpha;
    int dst0Alpha;
M
Michael Niedermayer 已提交
433 434
    int srcXYZ;
    int dstXYZ;
435 436 437 438
    int src_h_chr_pos;
    int dst_h_chr_pos;
    int src_v_chr_pos;
    int dst_v_chr_pos;
439 440 441 442 443 444
    int yuv2rgb_y_offset;
    int yuv2rgb_y_coeff;
    int yuv2rgb_v2r_coeff;
    int yuv2rgb_v2g_coeff;
    int yuv2rgb_u2g_coeff;
    int yuv2rgb_u2b_coeff;
445 446 447 448 449 450 451 452 453 454 455 456

#define RED_DITHER            "0*8"
#define GREEN_DITHER          "1*8"
#define BLUE_DITHER           "2*8"
#define Y_COEFF               "3*8"
#define VR_COEFF              "4*8"
#define UB_COEFF              "5*8"
#define VG_COEFF              "6*8"
#define UG_COEFF              "7*8"
#define Y_OFFSET              "8*8"
#define U_OFFSET              "9*8"
#define V_OFFSET              "10*8"
457
#define LUM_MMX_FILTER_OFFSET "11*8"
458
#define CHR_MMX_FILTER_OFFSET "11*8+4*4*"AV_STRINGIFY(MAX_FILTER_SIZE)
459
#define DSTW_OFFSET           "11*8+4*4*"AV_STRINGIFY(MAX_FILTER_SIZE)"*2"
460 461 462 463 464 465 466 467 468 469
#define ESP_OFFSET            "11*8+4*4*"AV_STRINGIFY(MAX_FILTER_SIZE)"*2+8"
#define VROUNDER_OFFSET       "11*8+4*4*"AV_STRINGIFY(MAX_FILTER_SIZE)"*2+16"
#define U_TEMP                "11*8+4*4*"AV_STRINGIFY(MAX_FILTER_SIZE)"*2+24"
#define V_TEMP                "11*8+4*4*"AV_STRINGIFY(MAX_FILTER_SIZE)"*2+32"
#define Y_TEMP                "11*8+4*4*"AV_STRINGIFY(MAX_FILTER_SIZE)"*2+40"
#define ALP_MMX_FILTER_OFFSET "11*8+4*4*"AV_STRINGIFY(MAX_FILTER_SIZE)"*2+48"
#define UV_OFF_PX             "11*8+4*4*"AV_STRINGIFY(MAX_FILTER_SIZE)"*3+48"
#define UV_OFF_BYTE           "11*8+4*4*"AV_STRINGIFY(MAX_FILTER_SIZE)"*3+56"
#define DITHER16              "11*8+4*4*"AV_STRINGIFY(MAX_FILTER_SIZE)"*3+64"
#define DITHER32              "11*8+4*4*"AV_STRINGIFY(MAX_FILTER_SIZE)"*3+80"
A
Andreas Cadhalpun 已提交
470
#define DITHER32_INT          (11*8+4*4*MAX_FILTER_SIZE*3+80) // value equal to above, used for checking that the struct hasn't been changed by mistake
471

472 473 474 475 476 477 478 479 480 481 482 483
    DECLARE_ALIGNED(8, uint64_t, redDither);
    DECLARE_ALIGNED(8, uint64_t, greenDither);
    DECLARE_ALIGNED(8, uint64_t, blueDither);

    DECLARE_ALIGNED(8, uint64_t, yCoeff);
    DECLARE_ALIGNED(8, uint64_t, vrCoeff);
    DECLARE_ALIGNED(8, uint64_t, ubCoeff);
    DECLARE_ALIGNED(8, uint64_t, vgCoeff);
    DECLARE_ALIGNED(8, uint64_t, ugCoeff);
    DECLARE_ALIGNED(8, uint64_t, yOffset);
    DECLARE_ALIGNED(8, uint64_t, uOffset);
    DECLARE_ALIGNED(8, uint64_t, vOffset);
484 485
    int32_t lumMmxFilter[4 * MAX_FILTER_SIZE];
    int32_t chrMmxFilter[4 * MAX_FILTER_SIZE];
R
Ramiro Polla 已提交
486
    int dstW;                     ///< Width  of destination luma/alpha planes.
487 488 489 490 491
    DECLARE_ALIGNED(8, uint64_t, esp);
    DECLARE_ALIGNED(8, uint64_t, vRounder);
    DECLARE_ALIGNED(8, uint64_t, u_temp);
    DECLARE_ALIGNED(8, uint64_t, v_temp);
    DECLARE_ALIGNED(8, uint64_t, y_temp);
R
Ronald S. Bultje 已提交
492
    int32_t alpMmxFilter[4 * MAX_FILTER_SIZE];
493 494 495
    // alignment of these values is not necessary, but merely here
    // to maintain the same offset across x8632 and x86-64. Once we
    // use proper offset macros in the asm, they can be removed.
496
    DECLARE_ALIGNED(8, ptrdiff_t, uv_off); ///< offset (in pixels) between u and v planes
497
    DECLARE_ALIGNED(8, ptrdiff_t, uv_offx2); ///< offset (in bytes) between u and v planes
498 499
    DECLARE_ALIGNED(8, uint16_t, dither16)[8];
    DECLARE_ALIGNED(8, uint32_t, dither32)[8];
M
Michael Niedermayer 已提交
500

501 502
    const uint8_t *chrDither8, *lumDither8;

503
#if HAVE_ALTIVEC
R
Ramiro Polla 已提交
504 505 506 507 508 509 510
    vector signed short   CY;
    vector signed short   CRV;
    vector signed short   CBU;
    vector signed short   CGU;
    vector signed short   CGV;
    vector signed short   OY;
    vector unsigned short CSHIFT;
511
    vector signed short  *vYCoeffsBank, *vCCoeffsBank;
M
Michael Niedermayer 已提交
512 513
#endif

514
    int use_mmx_vfilter;
515

M
Michael Niedermayer 已提交
516 517 518
/* pre defined color-spaces gamma */
#define XYZ_GAMMA (2.6f)
#define RGB_GAMMA (2.2f)
519 520
    int16_t *xyzgamma;
    int16_t *rgbgamma;
C
clook 已提交
521 522
    int16_t *xyzgammainv;
    int16_t *rgbgammainv;
M
Michael Niedermayer 已提交
523
    int16_t xyz2rgb_matrix[3][4];
C
clook 已提交
524
    int16_t rgb2xyz_matrix[3][4];
M
Michael Niedermayer 已提交
525

526
    /* function pointers for swscale() */
527 528 529
    yuv2planar1_fn yuv2plane1;
    yuv2planarX_fn yuv2planeX;
    yuv2interleavedX_fn yuv2nv12cX;
530 531 532
    yuv2packed1_fn yuv2packed1;
    yuv2packed2_fn yuv2packed2;
    yuv2packedX_fn yuv2packedX;
M
Michael Niedermayer 已提交
533
    yuv2anyX_fn yuv2anyX;
534

535
    /// Unscaled conversion of luma plane to YV12 for horizontal scaler.
536
    void (*lumToYV12)(uint8_t *dst, const uint8_t *src, const uint8_t *src2, const uint8_t *src3,
537 538
                      int width, uint32_t *pal);
    /// Unscaled conversion of alpha plane to YV12 for horizontal scaler.
539
    void (*alpToYV12)(uint8_t *dst, const uint8_t *src, const uint8_t *src2, const uint8_t *src3,
540 541
                      int width, uint32_t *pal);
    /// Unscaled conversion of chroma planes to YV12 for horizontal scaler.
542
    void (*chrToYV12)(uint8_t *dstU, uint8_t *dstV,
543
                      const uint8_t *src1, const uint8_t *src2, const uint8_t *src3,
544
                      int width, uint32_t *pal);
545 546

    /**
547
     * Functions to read planar input, such as planar RGB, and convert
548
     * internally to Y/UV/A.
549
     */
550
    /** @{ */
551
    void (*readLumPlanar)(uint8_t *dst, const uint8_t *src[4], int width, int32_t *rgb2yuv);
552
    void (*readChrPlanar)(uint8_t *dstU, uint8_t *dstV, const uint8_t *src[4],
553
                          int width, int32_t *rgb2yuv);
554
    void (*readAlpPlanar)(uint8_t *dst, const uint8_t *src[4], int width, int32_t *rgb2yuv);
555 556
    /** @} */

557 558 559 560 561 562 563 564 565 566 567 568 569 570 571 572 573 574 575
    /**
     * Scale one horizontal line of input data using a bilinear filter
     * to produce one line of output data. Compared to SwsContext->hScale(),
     * please take note of the following caveats when using these:
     * - Scaling is done using only 7bit instead of 14bit coefficients.
     * - You can use no more than 5 input pixels to produce 4 output
     *   pixels. Therefore, this filter should not be used for downscaling
     *   by more than ~20% in width (because that equals more than 5/4th
     *   downscaling and thus more than 5 pixels input per 4 pixels output).
     * - In general, bilinear filters create artifacts during downscaling
     *   (even when <20%), because one output pixel will span more than one
     *   input pixel, and thus some pixels will need edges of both neighbor
     *   pixels to interpolate the output pixel. Since you can use at most
     *   two input pixels per output pixel in bilinear scaling, this is
     *   impossible and thus downscaling by any size will create artifacts.
     * To enable this type of scaling, set SWS_FLAG_FAST_BILINEAR
     * in SwsContext->flags.
     */
    /** @{ */
576
    void (*hyscale_fast)(struct SwsContext *c,
A
Anton Khirnov 已提交
577
                         int16_t *dst, int dstWidth,
578 579
                         const uint8_t *src, int srcW, int xInc);
    void (*hcscale_fast)(struct SwsContext *c,
A
Anton Khirnov 已提交
580
                         int16_t *dst1, int16_t *dst2, int dstWidth,
581 582
                         const uint8_t *src1, const uint8_t *src2,
                         int srcW, int xInc);
583
    /** @} */
584

585 586 587 588 589
    /**
     * Scale one horizontal line of input data using a filter over the input
     * lines, to produce one (differently sized) line of output data.
     *
     * @param dst        pointer to destination buffer for horizontally scaled
590 591 592 593 594
     *                   data. If the number of bits per component of one
     *                   destination pixel (SwsContext->dstBpc) is <= 10, data
     *                   will be 15bpc in 16bits (int16_t) width. Else (i.e.
     *                   SwsContext->dstBpc == 16), data will be 19bpc in
     *                   32bits (int32_t) width.
595
     * @param dstW       width of destination image
596 597 598 599 600 601 602
     * @param src        pointer to source data to be scaled. If the number of
     *                   bits per component of a source pixel (SwsContext->srcBpc)
     *                   is 8, this is 8bpc in 8bits (uint8_t) width. Else
     *                   (i.e. SwsContext->dstBpc > 8), this is native depth
     *                   in 16bits (uint16_t) width. In other words, for 9-bit
     *                   YUV input, this is 9bpc, for 10-bit YUV input, this is
     *                   10bpc, and for 16-bit RGB or YUV, this is 16bpc.
603 604 605 606 607 608 609 610 611 612 613 614
     * @param filter     filter coefficients to be used per output pixel for
     *                   scaling. This contains 14bpp filtering coefficients.
     *                   Guaranteed to contain dstW * filterSize entries.
     * @param filterPos  position of the first input pixel to be used for
     *                   each output pixel during scaling. Guaranteed to
     *                   contain dstW entries.
     * @param filterSize the number of input coefficients to be used (and
     *                   thus the number of input pixels to be used) for
     *                   creating a single output pixel. Is aligned to 4
     *                   (and input coefficients thus padded with zeroes)
     *                   to simplify creating SIMD code.
     */
615
    /** @{ */
616 617
    void (*hyScale)(struct SwsContext *c, int16_t *dst, int dstW,
                    const uint8_t *src, const int16_t *filter,
618
                    const int32_t *filterPos, int filterSize);
619 620
    void (*hcScale)(struct SwsContext *c, int16_t *dst, int dstW,
                    const uint8_t *src, const int16_t *filter,
621
                    const int32_t *filterPos, int filterSize);
622
    /** @} */
623

624 625 626 627
    /// Color range conversion function for luma plane if needed.
    void (*lumConvertRange)(int16_t *dst, int width);
    /// Color range conversion function for chroma planes if needed.
    void (*chrConvertRange)(int16_t *dst1, int16_t *dst2, int width);
628

629
    int needs_hcscale; ///< Set if there are chroma planes to be converted.
M
Michael Niedermayer 已提交
630 631

    SwsDither dither;
632 633

    SwsAlphaBlend alphablend;
634 635 636
} SwsContext;
//FIXME check init (where 0)

637
SwsFunc ff_yuv2rgb_get_func_ptr(SwsContext *c);
638 639 640
int ff_yuv2rgb_c_init_tables(SwsContext *c, const int inv_table[4],
                             int fullRange, int brightness,
                             int contrast, int saturation);
641 642
void ff_yuv2rgb_init_tables_ppc(SwsContext *c, const int inv_table[4],
                                int brightness, int contrast, int saturation);
643

644
void ff_updateMMXDitherTables(SwsContext *c, int dstY, int lumBufIndex, int chrBufIndex,
645 646
                           int lastInLumBuf, int lastInChrBuf);

647 648
av_cold void ff_sws_init_range_convert(SwsContext *c);

649
SwsFunc ff_yuv2rgb_init_x86(SwsContext *c);
650
SwsFunc ff_yuv2rgb_init_ppc(SwsContext *c);
651

652 653 654 655
static av_always_inline int is16BPS(enum AVPixelFormat pix_fmt)
{
    const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(pix_fmt);
    av_assert0(desc);
656
    return desc->comp[0].depth == 16;
657
}
M
Michael Niedermayer 已提交
658

659 660 661 662
static av_always_inline int is9_OR_10BPS(enum AVPixelFormat pix_fmt)
{
    const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(pix_fmt);
    av_assert0(desc);
663
    return desc->comp[0].depth >= 9 && desc->comp[0].depth <= 14;
664
}
M
Michael Niedermayer 已提交
665

666 667
#define isNBPS(x) is9_OR_10BPS(x)

668 669 670 671
static av_always_inline int isBE(enum AVPixelFormat pix_fmt)
{
    const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(pix_fmt);
    av_assert0(desc);
672
    return desc->flags & AV_PIX_FMT_FLAG_BE;
673 674 675 676 677 678
}

static av_always_inline int isYUV(enum AVPixelFormat pix_fmt)
{
    const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(pix_fmt);
    av_assert0(desc);
679
    return !(desc->flags & AV_PIX_FMT_FLAG_RGB) && desc->nb_components >= 2;
680 681 682 683 684 685
}

static av_always_inline int isPlanarYUV(enum AVPixelFormat pix_fmt)
{
    const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(pix_fmt);
    av_assert0(desc);
686
    return ((desc->flags & AV_PIX_FMT_FLAG_PLANAR) && isYUV(pix_fmt));
687 688 689 690 691 692
}

static av_always_inline int isRGB(enum AVPixelFormat pix_fmt)
{
    const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(pix_fmt);
    av_assert0(desc);
693
    return (desc->flags & AV_PIX_FMT_FLAG_RGB);
694
}
695

696
#if 0 // FIXME
697
#define isGray(x) \
698
    (!(av_pix_fmt_desc_get(x)->flags & AV_PIX_FMT_FLAG_PAL) && \
699
     av_pix_fmt_desc_get(x)->nb_components <= 2)
700
#else
701
#define isGray(x)                      \
702
    ((x) == AV_PIX_FMT_GRAY8       ||  \
703
     (x) == AV_PIX_FMT_YA8         ||  \
704
     (x) == AV_PIX_FMT_GRAY16BE    ||  \
705 706 707
     (x) == AV_PIX_FMT_GRAY16LE    ||  \
     (x) == AV_PIX_FMT_YA16BE      ||  \
     (x) == AV_PIX_FMT_YA16LE)
708
#endif
709

710 711
#define isRGBinInt(x) \
    (           \
712
     (x) == AV_PIX_FMT_RGB48BE     ||  \
713 714 715 716 717 718 719 720 721 722 723 724 725
     (x) == AV_PIX_FMT_RGB48LE     ||  \
     (x) == AV_PIX_FMT_RGB32       ||  \
     (x) == AV_PIX_FMT_RGB32_1     ||  \
     (x) == AV_PIX_FMT_RGB24       ||  \
     (x) == AV_PIX_FMT_RGB565BE    ||  \
     (x) == AV_PIX_FMT_RGB565LE    ||  \
     (x) == AV_PIX_FMT_RGB555BE    ||  \
     (x) == AV_PIX_FMT_RGB555LE    ||  \
     (x) == AV_PIX_FMT_RGB444BE    ||  \
     (x) == AV_PIX_FMT_RGB444LE    ||  \
     (x) == AV_PIX_FMT_RGB8        ||  \
     (x) == AV_PIX_FMT_RGB4        ||  \
     (x) == AV_PIX_FMT_RGB4_BYTE   ||  \
J
Jean First 已提交
726 727
     (x) == AV_PIX_FMT_RGBA64BE    ||  \
     (x) == AV_PIX_FMT_RGBA64LE    ||  \
728
     (x) == AV_PIX_FMT_MONOBLACK   ||  \
729
     (x) == AV_PIX_FMT_MONOWHITE   \
730
    )
731 732
#define isBGRinInt(x) \
    (           \
733
     (x) == AV_PIX_FMT_BGR48BE     ||  \
734 735 736 737 738 739 740 741 742 743 744 745 746
     (x) == AV_PIX_FMT_BGR48LE     ||  \
     (x) == AV_PIX_FMT_BGR32       ||  \
     (x) == AV_PIX_FMT_BGR32_1     ||  \
     (x) == AV_PIX_FMT_BGR24       ||  \
     (x) == AV_PIX_FMT_BGR565BE    ||  \
     (x) == AV_PIX_FMT_BGR565LE    ||  \
     (x) == AV_PIX_FMT_BGR555BE    ||  \
     (x) == AV_PIX_FMT_BGR555LE    ||  \
     (x) == AV_PIX_FMT_BGR444BE    ||  \
     (x) == AV_PIX_FMT_BGR444LE    ||  \
     (x) == AV_PIX_FMT_BGR8        ||  \
     (x) == AV_PIX_FMT_BGR4        ||  \
     (x) == AV_PIX_FMT_BGR4_BYTE   ||  \
J
Jean First 已提交
747 748
     (x) == AV_PIX_FMT_BGRA64BE    ||  \
     (x) == AV_PIX_FMT_BGRA64LE    ||  \
749
     (x) == AV_PIX_FMT_MONOBLACK   ||  \
750
     (x) == AV_PIX_FMT_MONOWHITE   \
751
    )
752

753
#define isRGBinBytes(x) (           \
754 755 756 757 758 759 760
           (x) == AV_PIX_FMT_RGB48BE     \
        || (x) == AV_PIX_FMT_RGB48LE     \
        || (x) == AV_PIX_FMT_RGBA64BE    \
        || (x) == AV_PIX_FMT_RGBA64LE    \
        || (x) == AV_PIX_FMT_RGBA        \
        || (x) == AV_PIX_FMT_ARGB        \
        || (x) == AV_PIX_FMT_RGB24       \
761 762
    )
#define isBGRinBytes(x) (           \
763 764 765 766 767 768 769
           (x) == AV_PIX_FMT_BGR48BE     \
        || (x) == AV_PIX_FMT_BGR48LE     \
        || (x) == AV_PIX_FMT_BGRA64BE    \
        || (x) == AV_PIX_FMT_BGRA64LE    \
        || (x) == AV_PIX_FMT_BGRA        \
        || (x) == AV_PIX_FMT_ABGR        \
        || (x) == AV_PIX_FMT_BGR24       \
770
    )
771

772 773 774 775 776 777 778 779 780 781 782 783 784 785 786
#define isBayer(x) ( \
           (x)==AV_PIX_FMT_BAYER_BGGR8    \
        || (x)==AV_PIX_FMT_BAYER_BGGR16LE \
        || (x)==AV_PIX_FMT_BAYER_BGGR16BE \
        || (x)==AV_PIX_FMT_BAYER_RGGB8    \
        || (x)==AV_PIX_FMT_BAYER_RGGB16LE \
        || (x)==AV_PIX_FMT_BAYER_RGGB16BE \
        || (x)==AV_PIX_FMT_BAYER_GBRG8    \
        || (x)==AV_PIX_FMT_BAYER_GBRG16LE \
        || (x)==AV_PIX_FMT_BAYER_GBRG16BE \
        || (x)==AV_PIX_FMT_BAYER_GRBG8    \
        || (x)==AV_PIX_FMT_BAYER_GRBG16LE \
        || (x)==AV_PIX_FMT_BAYER_GRBG16BE \
    )

787 788
#define isAnyRGB(x) \
    (           \
789
          isBayer(x)          ||    \
790 791
          isRGBinInt(x)       ||    \
          isBGRinInt(x)       ||    \
792
          isRGB(x)      \
793
    )
794

795 796 797 798
static av_always_inline int isALPHA(enum AVPixelFormat pix_fmt)
{
    const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(pix_fmt);
    av_assert0(desc);
799 800
    if (pix_fmt == AV_PIX_FMT_PAL8)
        return 1;
801
    return desc->flags & AV_PIX_FMT_FLAG_ALPHA;
802
}
803

804
#if 1
805
#define isPacked(x)         (       \
806 807
           (x)==AV_PIX_FMT_PAL8        \
        || (x)==AV_PIX_FMT_YUYV422     \
808
        || (x)==AV_PIX_FMT_YVYU422     \
809
        || (x)==AV_PIX_FMT_UYVY422     \
810
        || (x)==AV_PIX_FMT_YA8       \
811 812
        || (x)==AV_PIX_FMT_YA16LE      \
        || (x)==AV_PIX_FMT_YA16BE      \
P
Paul B Mahol 已提交
813 814
        || (x)==AV_PIX_FMT_AYUV64LE    \
        || (x)==AV_PIX_FMT_AYUV64BE    \
M
Michael Niedermayer 已提交
815 816
        ||  isRGBinInt(x)           \
        ||  isBGRinInt(x)           \
817
    )
818
#else
819 820 821 822
static av_always_inline int isPacked(enum AVPixelFormat pix_fmt)
{
    const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(pix_fmt);
    av_assert0(desc);
823
    return ((desc->nb_components >= 2 && !(desc->flags & AV_PIX_FMT_FLAG_PLANAR)) ||
824 825
            pix_fmt == AV_PIX_FMT_PAL8);
}
826

827
#endif
828 829 830 831
static av_always_inline int isPlanar(enum AVPixelFormat pix_fmt)
{
    const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(pix_fmt);
    av_assert0(desc);
832
    return (desc->nb_components >= 2 && (desc->flags & AV_PIX_FMT_FLAG_PLANAR));
833 834 835 836 837 838
}

static av_always_inline int isPackedRGB(enum AVPixelFormat pix_fmt)
{
    const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(pix_fmt);
    av_assert0(desc);
839
    return ((desc->flags & (AV_PIX_FMT_FLAG_PLANAR | AV_PIX_FMT_FLAG_RGB)) == AV_PIX_FMT_FLAG_RGB);
840 841 842 843 844 845
}

static av_always_inline int isPlanarRGB(enum AVPixelFormat pix_fmt)
{
    const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(pix_fmt);
    av_assert0(desc);
846 847
    return ((desc->flags & (AV_PIX_FMT_FLAG_PLANAR | AV_PIX_FMT_FLAG_RGB)) ==
            (AV_PIX_FMT_FLAG_PLANAR | AV_PIX_FMT_FLAG_RGB));
848 849 850 851 852 853
}

static av_always_inline int usePal(enum AVPixelFormat pix_fmt)
{
    const AVPixFmtDescriptor *desc = av_pix_fmt_desc_get(pix_fmt);
    av_assert0(desc);
854
    return (desc->flags & AV_PIX_FMT_FLAG_PAL) || (desc->flags & AV_PIX_FMT_FLAG_PSEUDOPAL);
855
}
856

857 858
extern const uint64_t ff_dither4[2];
extern const uint64_t ff_dither8[2];
859

860 861 862 863 864
extern const uint8_t ff_dither_2x2_4[3][8];
extern const uint8_t ff_dither_2x2_8[3][8];
extern const uint8_t ff_dither_4x4_16[5][8];
extern const uint8_t ff_dither_8x8_32[9][8];
extern const uint8_t ff_dither_8x8_73[9][8];
865
extern const uint8_t ff_dither_8x8_128[9][8];
866
extern const uint8_t ff_dither_8x8_220[9][8];
867 868

extern const int32_t ff_yuv2rgb_coeffs[8][4];
869

870
extern const AVClass ff_sws_context_class;
871

872
/**
873
 * Set c->swscale to an unscaled converter if one exists for the specific
874 875 876
 * source and destination formats, bit depths, flags, etc.
 */
void ff_get_unscaled_swscale(SwsContext *c);
877
void ff_get_unscaled_swscale_ppc(SwsContext *c);
878
void ff_get_unscaled_swscale_arm(SwsContext *c);
879

880
/**
881
 * Return function pointer to fastest main scaler path function depending
882 883 884 885
 * on architecture and available optimizations.
 */
SwsFunc ff_getSwsFunc(SwsContext *c);

886
void ff_sws_init_input_funcs(SwsContext *c);
887 888 889 890 891 892
void ff_sws_init_output_funcs(SwsContext *c,
                              yuv2planar1_fn *yuv2plane1,
                              yuv2planarX_fn *yuv2planeX,
                              yuv2interleavedX_fn *yuv2nv12cX,
                              yuv2packed1_fn *yuv2packed1,
                              yuv2packed2_fn *yuv2packed2,
M
Michael Niedermayer 已提交
893 894
                              yuv2packedX_fn *yuv2packedX,
                              yuv2anyX_fn *yuv2anyX);
895
void ff_sws_init_swscale_ppc(SwsContext *c);
896
void ff_sws_init_swscale_x86(SwsContext *c);
897

898 899 900 901 902
void ff_hyscale_fast_c(SwsContext *c, int16_t *dst, int dstWidth,
                       const uint8_t *src, int srcW, int xInc);
void ff_hcscale_fast_c(SwsContext *c, int16_t *dst1, int16_t *dst2,
                       int dstWidth, const uint8_t *src1,
                       const uint8_t *src2, int srcW, int xInc);
903 904 905 906 907 908 909 910 911
int ff_init_hscaler_mmxext(int dstW, int xInc, uint8_t *filterCode,
                           int16_t *filter, int32_t *filterPos,
                           int numSplits);
void ff_hyscale_fast_mmxext(SwsContext *c, int16_t *dst,
                            int dstWidth, const uint8_t *src,
                            int srcW, int xInc);
void ff_hcscale_fast_mmxext(SwsContext *c, int16_t *dst1, int16_t *dst2,
                            int dstWidth, const uint8_t *src1,
                            const uint8_t *src2, int srcW, int xInc);
912

913 914 915 916 917 918 919 920 921 922 923
/**
 * Allocate and return an SwsContext.
 * This is like sws_getContext() but does not perform the init step, allowing
 * the user to set additional AVOptions.
 *
 * @see sws_getContext()
 */
struct SwsContext *sws_alloc_set_opts(int srcW, int srcH, enum AVPixelFormat srcFormat,
                                      int dstW, int dstH, enum AVPixelFormat dstFormat,
                                      int flags, const double *param);

924 925 926 927
int ff_sws_alphablendaway(SwsContext *c, const uint8_t *src[],
                          int srcStride[], int srcSliceY, int srcSliceH,
                          uint8_t *dst[], int dstStride[]);

928 929 930 931 932
static inline void fillPlane16(uint8_t *plane, int stride, int width, int height, int y,
                               int alpha, int bits, const int big_endian)
{
    int i, j;
    uint8_t *ptr = plane + stride * y;
933
    int v = alpha ? 0xFFFF>>(16-bits) : (1<<(bits-1));
934 935 936 937 938 939 940 941 942 943 944 945 946 947
    for (i = 0; i < height; i++) {
#define FILL(wfunc) \
        for (j = 0; j < width; j++) {\
            wfunc(ptr+2*j, v);\
        }
        if (big_endian) {
            FILL(AV_WB16);
        } else {
            FILL(AV_WL16);
        }
        ptr += stride;
    }
}

948 949 950 951 952 953 954 955 956 957 958 959 960 961 962 963 964 965 966 967 968 969 970 971 972 973 974 975 976 977 978 979 980 981 982 983 984 985 986 987 988 989 990 991 992 993 994 995 996 997 998 999 1000 1001 1002 1003 1004 1005 1006
#define MAX_SLICE_PLANES 4

/// Slice plane
typedef struct SwsPlane
{
    int available_lines;    ///< max number of lines that can be hold by this plane
    int sliceY;             ///< index of first line
    int sliceH;             ///< number of lines
    uint8_t **line;         ///< line buffer
    uint8_t **tmp;          ///< Tmp line buffer used by mmx code
} SwsPlane;

/**
 * Struct which defines a slice of an image to be scaled or a output for
 * a scaled slice.
 * A slice can also be used as intermediate ring buffer for scaling steps.
 */
typedef struct SwsSlice
{
    int width;              ///< Slice line width
    int h_chr_sub_sample;   ///< horizontal chroma subsampling factor
    int v_chr_sub_sample;   ///< vertical chroma subsampling factor
    int is_ring;            ///< flag to identify if this slice is a ring buffer
    int should_free_lines;  ///< flag to identify if there are dynamic allocated lines
    enum AVPixelFormat fmt; ///< planes pixel format
    SwsPlane plane[MAX_SLICE_PLANES];   ///< color planes
} SwsSlice;

/**
 * Struct which holds all necessary data for processing a slice.
 * A processing step can be a color conversion or horizontal/vertical scaling.
 */
typedef struct SwsFilterDescriptor
{
    SwsSlice *src;  ///< Source slice
    SwsSlice *dst;  ///< Output slice

    int alpha;      ///< Flag for processing alpha channel
    void *instance; ///< Filter instance data

    /// Function for processing input slice sliceH lines starting from line sliceY
    int (*process)(SwsContext *c, struct SwsFilterDescriptor *desc, int sliceY, int sliceH);
} SwsFilterDescriptor;

/// Color conversion instance data
typedef struct ColorContext
{
    uint32_t *pal;
} ColorContext;

/// Scaler instance data
typedef struct FilterContext
{
    uint16_t *filter;
    int *filter_pos;
    int filter_size;
    int xInc;
} FilterContext;

P
Pedro Arthur 已提交
1007 1008 1009 1010 1011 1012 1013 1014 1015
typedef struct VScalerContext
{
    uint16_t *filter[2];
    int32_t  *filter_pos;
    int filter_size;
    int isMMX;
    void *pfn;
} VScalerContext;

1016
// warp input lines in the form (src + width*i + j) to slice format (line[i][j])
P
Pedro Arthur 已提交
1017 1018
// relative=true means first line src[x][0] otherwise first line is src[x][lum/crh Y]
int ff_init_slice_from_src(SwsSlice * s, uint8_t *src[4], int stride[4], int srcW, int lumY, int lumH, int chrY, int chrH, int relative);
1019 1020 1021 1022 1023 1024 1025 1026 1027 1028 1029

// Initialize scaler filter descriptor chain
int ff_init_filters(SwsContext *c);

// Free all filter data
int ff_free_filters(SwsContext *c);

/*
 function for applying ring buffer logic into slice s
 It checks if the slice can hold more @lum lines, if yes
 do nothing otherwise remove @lum least used lines.
P
Pedro Arthur 已提交
1030
 It applies the same procedure for @chr lines.
1031 1032 1033
*/
int ff_rotate_slice(SwsSlice *s, int lum, int chr);

P
Pedro Arthur 已提交
1034 1035 1036
/// initializes gamma conversion descriptor
int ff_init_gamma_convert(SwsFilterDescriptor *desc, SwsSlice * src, uint16_t *table);

1037 1038 1039 1040 1041 1042
/// initializes lum pixel format conversion descriptor
int ff_init_desc_fmt_convert(SwsFilterDescriptor *desc, SwsSlice * src, SwsSlice *dst, uint32_t *pal);

/// initializes lum horizontal scaling descriptor
int ff_init_desc_hscale(SwsFilterDescriptor *desc, SwsSlice *src, SwsSlice *dst, uint16_t *filter, int * filter_pos, int filter_size, int xInc);

P
Pedro Arthur 已提交
1043
/// initializes chr pixel format conversion descriptor
1044 1045 1046 1047 1048 1049 1050
int ff_init_desc_cfmt_convert(SwsFilterDescriptor *desc, SwsSlice * src, SwsSlice *dst, uint32_t *pal);

/// initializes chr horizontal scaling descriptor
int ff_init_desc_chscale(SwsFilterDescriptor *desc, SwsSlice *src, SwsSlice *dst, uint16_t *filter, int * filter_pos, int filter_size, int xInc);

int ff_init_desc_no_chr(SwsFilterDescriptor *desc, SwsSlice * src, SwsSlice *dst);

P
Pedro Arthur 已提交
1051 1052 1053 1054 1055 1056 1057 1058
/// initializes vertical scaling descriptors
int ff_init_vscale(SwsContext *c, SwsFilterDescriptor *desc, SwsSlice *src, SwsSlice *dst);

/// setup vertical scaler functions
void ff_init_vscale_pfn(SwsContext *c, yuv2planar1_fn yuv2plane1, yuv2planarX_fn yuv2planeX,
    yuv2interleavedX_fn yuv2nv12cX, yuv2packed1_fn yuv2packed1, yuv2packed2_fn yuv2packed2,
    yuv2packedX_fn yuv2packedX, yuv2anyX_fn yuv2anyX, int use_mmx);

1059 1060 1061
//number of extra lines to process
#define MAX_LINES_AHEAD 4

P
Pedro Arthur 已提交
1062
// enable use of refactored scaler code
P
Pedro Arthur 已提交
1063
#define NEW_FILTER
P
Pedro Arthur 已提交
1064

1065
#endif /* SWSCALE_SWSCALE_INTERNAL_H */