Accelerated 32bpp scalers, Preview 1
git-svn-id: svn://svn.code.sf.net/p/sc2/code/trunk@1943 8092fc87-c524-0410-9efc-e669fe64eaf9
This commit is contained in:
+104
@@ -0,0 +1,104 @@
|
||||
/*
|
||||
* This program is free software; you can redistribute it and/or modify
|
||||
* it under the terms of the GNU General Public License as published by
|
||||
* the Free Software Foundation; either version 2 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU General Public License
|
||||
* along with this program; if not, write to the Free Software
|
||||
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||
*/
|
||||
|
||||
#ifdef GFXMODULE_SDL
|
||||
|
||||
#include "port.h"
|
||||
#include "libs/platform.h"
|
||||
|
||||
#if defined(MMX_ASM)
|
||||
|
||||
#include "libs/graphics/sdl/sdl_common.h"
|
||||
#include "types.h"
|
||||
#include "scalers.h"
|
||||
#include "scaleint.h"
|
||||
#include "2xscalers.h"
|
||||
#include "2xscalers_mmx.h"
|
||||
|
||||
// 3DNow! name for all functions
|
||||
#undef SCALE_
|
||||
#define SCALE_(name) Scale ## _3DNow_ ## name
|
||||
|
||||
// Tell them which opcodes we want to support
|
||||
#undef USE_MOVNTQ
|
||||
#define USE_PREFETCH AMD_PREFETCH
|
||||
#undef USE_PSADBW
|
||||
// Bring in inline asm functions
|
||||
#include "scalemmx.h"
|
||||
|
||||
|
||||
// Scaler function lookup table
|
||||
//
|
||||
const Scale_FuncDef_t
|
||||
Scale_3DNow_Functions[] =
|
||||
{
|
||||
{TFB_GFXFLAGS_SCALE_BILINEAR, Scale_3DNow_BilinearFilter},
|
||||
{TFB_GFXFLAGS_SCALE_BIADAPT, Scale_BiAdaptFilter},
|
||||
{TFB_GFXFLAGS_SCALE_BIADAPTADV, Scale_MMX_BiAdaptAdvFilter},
|
||||
{TFB_GFXFLAGS_SCALE_TRISCAN, Scale_MMX_TriScanFilter},
|
||||
{TFB_GFXFLAGS_SCALE_HQXX, Scale_MMX_HqFilter},
|
||||
// Default
|
||||
{0, Scale_3DNow_Nearest}
|
||||
};
|
||||
|
||||
|
||||
void
|
||||
Scale_3DNow_PrepPlatform (const SDL_PixelFormat* fmt)
|
||||
{
|
||||
Scale_MMX_PrepPlatform (fmt);
|
||||
}
|
||||
|
||||
// Nearest Neighbor scaling to 2x
|
||||
// void Scale_3DNow_Nearest (SDL_Surface *src,
|
||||
// SDL_Surface *dst, SDL_Rect *r)
|
||||
|
||||
#include "nearest2x.c"
|
||||
|
||||
|
||||
// Bilinear scaling to 2x
|
||||
// void Scale_3DNow_BilinearFilter (SDL_Surface *src,
|
||||
// SDL_Surface *dst, SDL_Rect *r)
|
||||
|
||||
#include "bilinear2x.c"
|
||||
|
||||
|
||||
#if 0 && NO_IMPROVEMENT
|
||||
|
||||
// Advanced Biadapt scaling to 2x
|
||||
// void Scale_3DNow_BiAdaptAdvFilter (SDL_Surface *src,
|
||||
// SDL_Surface *dst, SDL_Rect *r)
|
||||
|
||||
#include "biadv2x.c"
|
||||
|
||||
|
||||
// Triscan scaling to 2x
|
||||
// derivative of scale2x -- scale2x.sf.net
|
||||
// void Scale_3DNow_TriScanFilter (SDL_Surface *src,
|
||||
// SDL_Surface *dst, SDL_Rect *r)
|
||||
|
||||
#include "triscan2x.c"
|
||||
|
||||
// Hq2x scaling
|
||||
// (adapted from 'hq2x' by Maxim Stepin -- www.hiend3d.com/hq2x.html)
|
||||
// void Scale_3DNow_HqFilter (SDL_Surface *src,
|
||||
// SDL_Surface *dst, SDL_Rect *r)
|
||||
|
||||
#include "hq2x.c"
|
||||
|
||||
#endif /* NO_IMPROVEMENT */
|
||||
|
||||
#endif /* MMX_ASM */
|
||||
#endif /* GFXMODULE_SDL */
|
||||
+138
@@ -0,0 +1,138 @@
|
||||
/*
|
||||
* This program is free software; you can redistribute it and/or modify
|
||||
* it under the terms of the GNU General Public License as published by
|
||||
* the Free Software Foundation; either version 2 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU General Public License
|
||||
* along with this program; if not, write to the Free Software
|
||||
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||
*/
|
||||
|
||||
#ifdef GFXMODULE_SDL
|
||||
|
||||
#include "port.h"
|
||||
#include "libs/platform.h"
|
||||
|
||||
#if defined(MMX_ASM)
|
||||
|
||||
#include "libs/graphics/sdl/sdl_common.h"
|
||||
#include "types.h"
|
||||
#include "scalers.h"
|
||||
#include "scaleint.h"
|
||||
#include "2xscalers.h"
|
||||
#include "2xscalers_mmx.h"
|
||||
|
||||
// MMX name for all functions
|
||||
#undef SCALE_
|
||||
#define SCALE_(name) Scale ## _MMX_ ## name
|
||||
|
||||
// Tell them which opcodes we want to support
|
||||
#undef USE_MOVNTQ
|
||||
#undef USE_PREFETCH
|
||||
#undef USE_PSADBW
|
||||
// And Bring in inline asm functions
|
||||
#include "scalemmx.h"
|
||||
|
||||
|
||||
// Scaler function lookup table
|
||||
//
|
||||
const Scale_FuncDef_t
|
||||
Scale_MMX_Functions[] =
|
||||
{
|
||||
{TFB_GFXFLAGS_SCALE_BILINEAR, Scale_MMX_BilinearFilter},
|
||||
{TFB_GFXFLAGS_SCALE_BIADAPT, Scale_BiAdaptFilter},
|
||||
{TFB_GFXFLAGS_SCALE_BIADAPTADV, Scale_MMX_BiAdaptAdvFilter},
|
||||
{TFB_GFXFLAGS_SCALE_TRISCAN, Scale_MMX_TriScanFilter},
|
||||
{TFB_GFXFLAGS_SCALE_HQXX, Scale_MMX_HqFilter},
|
||||
// Default
|
||||
{0, Scale_MMX_Nearest}
|
||||
};
|
||||
|
||||
// MMX transformation multipliers
|
||||
Uint64 mmx_888to555_mult;
|
||||
Uint64 mmx_Y_mult;
|
||||
Uint64 mmx_U_mult;
|
||||
Uint64 mmx_V_mult;
|
||||
// Uint64 mmx_YUV_threshold = 0x00300706; original hq2x threshold
|
||||
//Uint64 mmx_YUV_threshold = 0x0030100e;
|
||||
Uint64 mmx_YUV_threshold = 0x0040120c;
|
||||
|
||||
void
|
||||
Scale_MMX_PrepPlatform (const SDL_PixelFormat* fmt)
|
||||
{
|
||||
// prepare the channel-shuffle multiplier
|
||||
mmx_888to555_mult = ((Uint64)0x0400) << (fmt->Rshift * 2)
|
||||
| ((Uint64)0x0020) << (fmt->Gshift * 2)
|
||||
| ((Uint64)0x0001) << (fmt->Bshift * 2);
|
||||
|
||||
// prepare the RGB->YUV multipliers
|
||||
mmx_Y_mult = ((Uint64)(uint16)YUV_matrix[YUV_XFORM_R][YUV_XFORM_Y])
|
||||
<< (fmt->Rshift * 2)
|
||||
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_G][YUV_XFORM_Y])
|
||||
<< (fmt->Gshift * 2)
|
||||
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_B][YUV_XFORM_Y])
|
||||
<< (fmt->Bshift * 2);
|
||||
|
||||
mmx_U_mult = ((Uint64)(uint16)YUV_matrix[YUV_XFORM_R][YUV_XFORM_U])
|
||||
<< (fmt->Rshift * 2)
|
||||
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_G][YUV_XFORM_U])
|
||||
<< (fmt->Gshift * 2)
|
||||
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_B][YUV_XFORM_U])
|
||||
<< (fmt->Bshift * 2);
|
||||
|
||||
mmx_V_mult = ((Uint64)(uint16)YUV_matrix[YUV_XFORM_R][YUV_XFORM_V])
|
||||
<< (fmt->Rshift * 2)
|
||||
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_G][YUV_XFORM_V])
|
||||
<< (fmt->Gshift * 2)
|
||||
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_B][YUV_XFORM_V])
|
||||
<< (fmt->Bshift * 2);
|
||||
|
||||
mmx_YUV_threshold = (SCALE_DIFFYUV_TY << 16) | (SCALE_DIFFYUV_TU << 8)
|
||||
| SCALE_DIFFYUV_TV;
|
||||
}
|
||||
|
||||
|
||||
// Nearest Neighbor scaling to 2x
|
||||
// void Scale_MMX_Nearest (SDL_Surface *src,
|
||||
// SDL_Surface *dst, SDL_Rect *r)
|
||||
|
||||
#include "nearest2x.c"
|
||||
|
||||
|
||||
// Bilinear scaling to 2x
|
||||
// void Scale_MMX_BilinearFilter (SDL_Surface *src,
|
||||
// SDL_Surface *dst, SDL_Rect *r)
|
||||
|
||||
#include "bilinear2x.c"
|
||||
|
||||
|
||||
// Advanced Biadapt scaling to 2x
|
||||
// void Scale_MMX_BiAdaptAdvFilter (SDL_Surface *src,
|
||||
// SDL_Surface *dst, SDL_Rect *r)
|
||||
|
||||
#include "biadv2x.c"
|
||||
|
||||
|
||||
// Triscan scaling to 2x
|
||||
// derivative of 'scale2x' -- scale2x.sf.net
|
||||
// void Scale_MMX_TriScanFilter (SDL_Surface *src,
|
||||
// SDL_Surface *dst, SDL_Rect *r)
|
||||
|
||||
#include "triscan2x.c"
|
||||
|
||||
// Hq2x scaling
|
||||
// (adapted from 'hq2x' by Maxim Stepin -- www.hiend3d.com/hq2x.html)
|
||||
// void Scale_MMX_HqFilter (SDL_Surface *src,
|
||||
// SDL_Surface *dst, SDL_Rect *r)
|
||||
|
||||
#include "hq2x.c"
|
||||
|
||||
|
||||
#endif /* MMX_ASM */
|
||||
#endif /* GFXMODULE_SDL */
|
||||
+56
@@ -0,0 +1,56 @@
|
||||
/*
|
||||
* This program is free software; you can redistribute it and/or modify
|
||||
* it under the terms of the GNU General Public License as published by
|
||||
* the Free Software Foundation; either version 2 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU General Public License
|
||||
* along with this program; if not, write to the Free Software
|
||||
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||
*/
|
||||
|
||||
#ifndef _2XSCALERS_MMX_H_
|
||||
#define _2XSCALERS_MMX_H_
|
||||
|
||||
// MMX versions
|
||||
void Scale_MMX_PrepPlatform (const SDL_PixelFormat* fmt);
|
||||
|
||||
void Scale_MMX_Nearest (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||
void Scale_MMX_BilinearFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||
void Scale_MMX_BiAdaptAdvFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||
void Scale_MMX_TriScanFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||
void Scale_MMX_HqFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||
|
||||
extern const Scale_FuncDef_t Scale_MMX_Functions[];
|
||||
|
||||
|
||||
// SSE (Intel)/MMX Ext (Athlon) versions
|
||||
void Scale_SSE_PrepPlatform (const SDL_PixelFormat* fmt);
|
||||
|
||||
void Scale_SSE_Nearest (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||
void Scale_SSE_BilinearFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||
void Scale_SSE_BiAdaptAdvFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||
void Scale_SSE_TriScanFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||
void Scale_SSE_HqFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||
|
||||
extern const Scale_FuncDef_t Scale_SSE_Functions[];
|
||||
|
||||
|
||||
// 3DNow (AMD K6/Athlon) versions
|
||||
void Scale_3DNow_PrepPlatform (const SDL_PixelFormat* fmt);
|
||||
|
||||
void Scale_3DNow_Nearest (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||
void Scale_3DNow_BilinearFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||
void Scale_3DNow_BiAdaptAdvFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||
void Scale_3DNow_TriScanFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||
void Scale_3DNow_HqFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||
|
||||
extern const Scale_FuncDef_t Scale_3DNow_Functions[];
|
||||
|
||||
|
||||
#endif /* _2XSCALERS_MMX_H_ */
|
||||
+102
@@ -0,0 +1,102 @@
|
||||
/*
|
||||
* This program is free software; you can redistribute it and/or modify
|
||||
* it under the terms of the GNU General Public License as published by
|
||||
* the Free Software Foundation; either version 2 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU General Public License
|
||||
* along with this program; if not, write to the Free Software
|
||||
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||
*/
|
||||
|
||||
#ifdef GFXMODULE_SDL
|
||||
|
||||
#include "port.h"
|
||||
#include "libs/platform.h"
|
||||
|
||||
#if defined(MMX_ASM)
|
||||
|
||||
#include "libs/graphics/sdl/sdl_common.h"
|
||||
#include "types.h"
|
||||
#include "scalers.h"
|
||||
#include "scaleint.h"
|
||||
#include "2xscalers.h"
|
||||
#include "2xscalers_mmx.h"
|
||||
|
||||
// SSE name for all functions
|
||||
#undef SCALE_
|
||||
#define SCALE_(name) Scale ## _SSE_ ## name
|
||||
|
||||
// Tell them which opcodes we want to support
|
||||
#define USE_MOVNTQ
|
||||
#define USE_PREFETCH INTEL_PREFETCH
|
||||
#define USE_PSADBW
|
||||
// Bring in inline asm functions
|
||||
#include "scalemmx.h"
|
||||
|
||||
|
||||
// Scaler function lookup table
|
||||
//
|
||||
const Scale_FuncDef_t
|
||||
Scale_SSE_Functions[] =
|
||||
{
|
||||
{TFB_GFXFLAGS_SCALE_BILINEAR, Scale_SSE_BilinearFilter},
|
||||
{TFB_GFXFLAGS_SCALE_BIADAPT, Scale_BiAdaptFilter},
|
||||
{TFB_GFXFLAGS_SCALE_BIADAPTADV, Scale_SSE_BiAdaptAdvFilter},
|
||||
{TFB_GFXFLAGS_SCALE_TRISCAN, Scale_SSE_TriScanFilter},
|
||||
{TFB_GFXFLAGS_SCALE_HQXX, Scale_MMX_HqFilter},
|
||||
// Default
|
||||
{0, Scale_SSE_Nearest}
|
||||
};
|
||||
|
||||
|
||||
void
|
||||
Scale_SSE_PrepPlatform (const SDL_PixelFormat* fmt)
|
||||
{
|
||||
Scale_MMX_PrepPlatform (fmt);
|
||||
}
|
||||
|
||||
// Nearest Neighbor scaling to 2x
|
||||
// void Scale_SSE_Nearest (SDL_Surface *src,
|
||||
// SDL_Surface *dst, SDL_Rect *r)
|
||||
|
||||
#include "nearest2x.c"
|
||||
|
||||
|
||||
// Bilinear scaling to 2x
|
||||
// void Scale_SSE_BilinearFilter (SDL_Surface *src,
|
||||
// SDL_Surface *dst, SDL_Rect *r)
|
||||
|
||||
#include "bilinear2x.c"
|
||||
|
||||
|
||||
// Advanced Biadapt scaling to 2x
|
||||
// void Scale_SSE_BiAdaptAdvFilter (SDL_Surface *src,
|
||||
// SDL_Surface *dst, SDL_Rect *r)
|
||||
|
||||
#include "biadv2x.c"
|
||||
|
||||
|
||||
// Triscan scaling to 2x
|
||||
// derivative of scale2x -- scale2x.sf.net
|
||||
// void Scale_SSE_TriScanFilter (SDL_Surface *src,
|
||||
// SDL_Surface *dst, SDL_Rect *r)
|
||||
|
||||
#include "triscan2x.c"
|
||||
|
||||
#if 0 && NO_IMPROVEMENT
|
||||
// Hq2x scaling
|
||||
// (adapted from 'hq2x' by Maxim Stepin -- www.hiend3d.com/hq2x.html)
|
||||
// void Scale_SSE_HqFilter (SDL_Surface *src,
|
||||
// SDL_Surface *dst, SDL_Rect *r)
|
||||
|
||||
#include "hq2x.c"
|
||||
#endif
|
||||
|
||||
#endif /* MMX_ASM */
|
||||
#endif /* GFXMODULE_SDL */
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
/*
|
||||
* This program is free software; you can redistribute it and/or modify
|
||||
* it under the terms of the GNU General Public License as published by
|
||||
* the Free Software Foundation; either version 2 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU General Public License
|
||||
* along with this program; if not, write to the Free Software
|
||||
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||
*/
|
||||
|
||||
#ifndef _2XSCALERS_SSE_H_
|
||||
#define _2XSCALERS_SSE_H_
|
||||
|
||||
|
||||
#endif /* _2XSCALERS_SSE_H_ */
|
||||
Executable
+535
@@ -0,0 +1,535 @@
|
||||
/*
|
||||
* Portions Copyright (C) 2003-2005 Alex Volkov (codepro@usa.net)
|
||||
*
|
||||
* This program is free software; you can redistribute it and/or modify
|
||||
* it under the terms of the GNU General Public License as published by
|
||||
* the Free Software Foundation; either version 2 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU General Public License
|
||||
* along with this program; if not, write to the Free Software
|
||||
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||
*/
|
||||
|
||||
// Core algorithm of the Advanced BiAdaptive screen scaler
|
||||
// Template
|
||||
// When this file is built standalone is produces a plain C version
|
||||
// Also #included by 2xscalers_mmx.c for an MMX version
|
||||
|
||||
#ifdef GFXMODULE_SDL
|
||||
|
||||
#include "libs/graphics/sdl/sdl_common.h"
|
||||
#include "types.h"
|
||||
#include "scalers.h"
|
||||
#include "scaleint.h"
|
||||
#include "2xscalers.h"
|
||||
|
||||
|
||||
// Advanced biadapt scaling to 2x
|
||||
// The name expands to either
|
||||
// Scale_BiAdaptAdvFilter (for plain C) or
|
||||
// Scale_MMX_BiAdaptAdvFilter (for MMX)
|
||||
// [others when platforms are added]
|
||||
void
|
||||
SCALE_(BiAdaptAdvFilter) (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r)
|
||||
{
|
||||
int x, y;
|
||||
const int w = src->w, h = src->h;
|
||||
int xend, yend;
|
||||
int dsrc, ddst;
|
||||
SDL_Rect *region = r;
|
||||
SDL_Rect limits;
|
||||
SDL_PixelFormat *fmt = dst->format;
|
||||
const int sp = src->pitch, dp = dst->pitch;
|
||||
const int bpp = fmt->BytesPerPixel;
|
||||
const int slen = sp / bpp, dlen = dp / bpp;
|
||||
// for clarity purposes, the 'pixels' array here is transposed
|
||||
Uint32 pixels[4][4];
|
||||
static int resolve_coord[][2] =
|
||||
{
|
||||
{0, -1}, {1, -1}, { 2, 0}, { 2, 1},
|
||||
{1, 2}, {0, 2}, {-1, 1}, {-1, 0},
|
||||
{100, 100} // term
|
||||
};
|
||||
Uint32 *src_p = (Uint32 *)src->pixels;
|
||||
Uint32 *dst_p = (Uint32 *)dst->pixels;
|
||||
|
||||
// these macros are for clarity; they make the current pixel (0,0)
|
||||
// and allow to access pixels in all directions
|
||||
#define PIX(x, y) (pixels[1 + (x)][1 + (y)])
|
||||
#define SRC(x, y) (src_p + (x) + ((y) * slen))
|
||||
// commonly used operations, for clarity also
|
||||
// others are defined at their respective bpp levels
|
||||
#define BIADAPT_RGBHIGH 8000
|
||||
#define BIADAPT_YUVLOW 30
|
||||
#define BIADAPT_YUVMED 70
|
||||
#define BIADAPT_YUVHIGH 130
|
||||
|
||||
// high tolerance pixel comparison
|
||||
#define BIADAPT_CMPRGB_HIGH(p1, p2) \
|
||||
(p1 == p2 || SCALE_CMPRGB (p1, p2) <= BIADAPT_RGBHIGH)
|
||||
|
||||
// low tolerance pixel comparison
|
||||
#define BIADAPT_CMPYUV_LOW(p1, p2) \
|
||||
(p1 == p2 || SCALE_CMPYUV (p1, p2, BIADAPT_YUVLOW))
|
||||
// medium tolerance pixel comparison
|
||||
#define BIADAPT_CMPYUV_MED(p1, p2) \
|
||||
(p1 == p2 || SCALE_CMPYUV (p1, p2, BIADAPT_YUVMED))
|
||||
// high tolerance pixel comparison
|
||||
#define BIADAPT_CMPYUV_HIGH(p1, p2) \
|
||||
(p1 == p2 || SCALE_CMPYUV (p1, p2, BIADAPT_YUVHIGH))
|
||||
|
||||
SCALE_(PlatInit) ();
|
||||
|
||||
// expand updated region if necessary
|
||||
// pixels neighbooring the updated region may
|
||||
// change as a result of updates
|
||||
limits.x = 0;
|
||||
limits.y = 0;
|
||||
limits.w = src->w;
|
||||
limits.h = src->h;
|
||||
Scale_ExpandRect (region, 2, &limits);
|
||||
|
||||
xend = region->x + region->w;
|
||||
yend = region->y + region->h;
|
||||
dsrc = slen - region->w;
|
||||
ddst = (dlen - region->w) * 2;
|
||||
|
||||
#define SCALE_GETPIX(p) ( *(Uint32 *)(p) )
|
||||
#define SCALE_SETPIX(p, c) ( *(Uint32 *)(p) = (c) )
|
||||
|
||||
// move ptrs to the first updated pixel
|
||||
src_p += slen * region->y + region->x;
|
||||
dst_p += (dlen * region->y + region->x) * 2;
|
||||
|
||||
for (y = region->y; y < yend; ++y, dst_p += ddst, src_p += dsrc)
|
||||
{
|
||||
for (x = region->x; x < xend; ++x, ++src_p, ++dst_p)
|
||||
{
|
||||
// pixel equality counter
|
||||
int cmatch;
|
||||
|
||||
// most pixels will fall into 'all 4 equal'
|
||||
// pattern, so we check it first
|
||||
cmatch = 0;
|
||||
|
||||
PIX (0, 0) = SCALE_GETPIX (SRC (0, 0));
|
||||
|
||||
SCALE_SETPIX (dst_p, PIX (0, 0));
|
||||
|
||||
if (y + 1 < h)
|
||||
{
|
||||
// check pixel below the current one
|
||||
PIX (0, 1) = SCALE_GETPIX (SRC (0, 1));
|
||||
|
||||
if (PIX (0, 0) == PIX (0, 1))
|
||||
{
|
||||
SCALE_SETPIX (dst_p + dlen, PIX (0, 0));
|
||||
cmatch |= 1;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// last pixel in column - propagate
|
||||
PIX (0, 1) = PIX (0, 0);
|
||||
SCALE_SETPIX (dst_p + dlen, PIX (0, 0));
|
||||
cmatch |= 1;
|
||||
|
||||
}
|
||||
|
||||
if (x + 1 < w)
|
||||
{
|
||||
// check pixel to the right from the current one
|
||||
PIX (1, 0) = SCALE_GETPIX (SRC (1, 0));
|
||||
|
||||
if (PIX (0, 0) == PIX (1, 0))
|
||||
{
|
||||
SCALE_SETPIX (dst_p + 1, PIX (0, 0));
|
||||
cmatch |= 2;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// last pixel in row - propagate
|
||||
PIX (1, 0) = PIX (0, 0);
|
||||
SCALE_SETPIX (dst_p + 1, PIX (0, 0));
|
||||
cmatch |= 2;
|
||||
}
|
||||
|
||||
if (cmatch == 3)
|
||||
{
|
||||
if (y + 1 >= h || x + 1 >= w)
|
||||
{
|
||||
// last pixel in row/column and nearest
|
||||
// neighboor is identical
|
||||
dst_p++;
|
||||
SCALE_SETPIX (dst_p + dlen, PIX (0, 0));
|
||||
continue;
|
||||
}
|
||||
|
||||
// check pixel to the bottom-right
|
||||
PIX (1, 1) = SCALE_GETPIX (SRC (1, 1));
|
||||
|
||||
if (PIX (0, 0) == PIX (1, 1))
|
||||
{
|
||||
// all 4 are equal - propagate
|
||||
dst_p++;
|
||||
SCALE_SETPIX (dst_p + dlen, PIX (0, 0));
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
// some neighboors are different, lets check them
|
||||
|
||||
if (x > 0)
|
||||
PIX (-1, 0) = SCALE_GETPIX (SRC (-1, 0));
|
||||
else
|
||||
PIX (-1, 0) = PIX (0, 0);
|
||||
|
||||
if (x + 2 < w)
|
||||
PIX (2, 0) = SCALE_GETPIX (SRC (2, 0));
|
||||
else
|
||||
PIX (2, 0) = PIX (1, 0);
|
||||
|
||||
if (y + 1 < h)
|
||||
{
|
||||
if (x > 0)
|
||||
PIX (-1, 1) = SCALE_GETPIX (SRC (-1, 1));
|
||||
else
|
||||
PIX (-1, 1) = PIX (0, 1);
|
||||
|
||||
if (x + 2 < w)
|
||||
{
|
||||
PIX (1, 1) = SCALE_GETPIX (SRC (1, 1));
|
||||
PIX (2, 1) = SCALE_GETPIX (SRC (2, 1));
|
||||
}
|
||||
else if (x + 1 < w)
|
||||
{
|
||||
PIX (1, 1) = SCALE_GETPIX (SRC (1, 1));
|
||||
PIX (2, 1) = PIX (1, 1);
|
||||
}
|
||||
else
|
||||
{
|
||||
PIX (1, 1) = PIX (0, 1);
|
||||
PIX (2, 1) = PIX (0, 1);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// last pixel in column
|
||||
PIX (-1, 1) = PIX (-1, 0);
|
||||
PIX (1, 1) = PIX (1, 0);
|
||||
PIX (2, 1) = PIX (2, 0);
|
||||
}
|
||||
|
||||
if (y + 2 < h)
|
||||
{
|
||||
PIX (0, 2) = SCALE_GETPIX (SRC (0, 2));
|
||||
|
||||
if (x > 0)
|
||||
PIX (-1, 2) = SCALE_GETPIX (SRC (-1, 2));
|
||||
else
|
||||
PIX (-1, 2) = PIX (0, 2);
|
||||
|
||||
if (x + 2 < w)
|
||||
{
|
||||
PIX (1, 2) = SCALE_GETPIX (SRC (1, 2));
|
||||
PIX (2, 2) = SCALE_GETPIX (SRC (2, 2));
|
||||
}
|
||||
else if (x + 1 < w)
|
||||
{
|
||||
PIX (1, 2) = SCALE_GETPIX (SRC (1, 2));
|
||||
PIX (2, 2) = PIX (1, 2);
|
||||
}
|
||||
else
|
||||
{
|
||||
PIX (1, 2) = PIX (0, 2);
|
||||
PIX (2, 2) = PIX (0, 2);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
// last pixel in column
|
||||
PIX (-1, 2) = PIX (-1, 1);
|
||||
PIX (0, 2) = PIX (0, 1);
|
||||
PIX (1, 2) = PIX (1, 1);
|
||||
PIX (2, 2) = PIX (2, 1);
|
||||
}
|
||||
|
||||
if (y > 0)
|
||||
{
|
||||
PIX (0, -1) = SCALE_GETPIX (SRC (0, -1));
|
||||
|
||||
if (x > 0)
|
||||
PIX (-1, -1) = SCALE_GETPIX (SRC (-1, -1));
|
||||
else
|
||||
PIX (-1, -1) = PIX (0, -1);
|
||||
|
||||
if (x + 2 < w)
|
||||
{
|
||||
PIX (1, -1) = SCALE_GETPIX (SRC (1, -1));
|
||||
PIX (2, -1) = SCALE_GETPIX (SRC (2, -1));
|
||||
}
|
||||
else if (x + 1 < w)
|
||||
{
|
||||
PIX (1, -1) = SCALE_GETPIX (SRC (1, -1));
|
||||
PIX (2, -1) = PIX (1, -1);
|
||||
}
|
||||
else
|
||||
{
|
||||
PIX (1, -1) = PIX (0, -1);
|
||||
PIX (2, -1) = PIX (0, -1);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
PIX (-1, -1) = PIX (-1, 0);
|
||||
PIX (0, -1) = PIX (0, 0);
|
||||
PIX (1, -1) = PIX (1, 0);
|
||||
PIX (2, -1) = PIX (2, 0);
|
||||
}
|
||||
|
||||
// check pixel below the current one
|
||||
if (!(cmatch & 1))
|
||||
{
|
||||
if (SCALE_CMPYUV (PIX (0, 0), PIX (0, 1), BIADAPT_YUVLOW))
|
||||
{
|
||||
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
|
||||
PIX (0, 0), PIX (0, 1))
|
||||
);
|
||||
cmatch |= 1;
|
||||
}
|
||||
// detect a 2:1 line going across the current pixel
|
||||
else if ( (PIX (0, 0) == PIX (-1, 0)
|
||||
&& PIX (0, 0) == PIX (1, 1)
|
||||
&& PIX (0, 0) == PIX (2, 1) &&
|
||||
|
||||
((!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, -1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, -1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, 0))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, 0))) ||
|
||||
(!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, 2))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, 2))))) ||
|
||||
|
||||
(PIX (0, 0) == PIX (1, 0)
|
||||
&& PIX (0, 0) == PIX (-1, 1)
|
||||
&& PIX (0, 0) == PIX (2, -1) &&
|
||||
|
||||
((!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 0))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, -1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, -1))) ||
|
||||
(!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 2))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, 1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, 0))))) )
|
||||
{
|
||||
SCALE_SETPIX (dst_p + dlen, PIX (0, 0));
|
||||
}
|
||||
// detect a 2:1 line going across the pixel below current
|
||||
else if ( (PIX (0, 1) == PIX (-1, 0)
|
||||
&& PIX (0, 1) == PIX (1, 1)
|
||||
&& PIX (0, 1) == PIX (2, 2) &&
|
||||
|
||||
((!BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (-1, -1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (1, 0))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (2, 1))) ||
|
||||
(!BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (-1, 1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (0, 2))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (1, 2))))) ||
|
||||
|
||||
(PIX (0, 1) == PIX (1, 0)
|
||||
&& PIX (0, 1) == PIX (-1, 1)
|
||||
&& PIX (0, 1) == PIX (2, 0) &&
|
||||
|
||||
((!BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (-1, 0))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (1, -1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (2, -1))) ||
|
||||
(!BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (-1, 2))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (0, 2))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (1, 1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (2, 1))))) )
|
||||
{
|
||||
SCALE_SETPIX (dst_p + dlen, PIX (0, 1));
|
||||
}
|
||||
else
|
||||
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
|
||||
PIX (0, 0), PIX (0, 1))
|
||||
);
|
||||
}
|
||||
|
||||
dst_p++;
|
||||
|
||||
// check pixel to the right from the current one
|
||||
if (!(cmatch & 2))
|
||||
{
|
||||
if (SCALE_CMPYUV (PIX (0, 0), PIX (1, 0), BIADAPT_YUVLOW))
|
||||
{
|
||||
SCALE_SETPIX (dst_p, Scale_Blend_11 (
|
||||
PIX (0, 0), PIX (1, 0))
|
||||
);
|
||||
cmatch |= 2;
|
||||
}
|
||||
// detect a 1:2 line going across the current pixel
|
||||
else if ( (PIX (0, 0) == PIX (1, -1)
|
||||
&& PIX (0, 0) == PIX (0, 1)
|
||||
&& PIX (0, 0) == PIX (-1, 2) &&
|
||||
|
||||
((!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, -1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 0))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 1))) ||
|
||||
(!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, -1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, 1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, 2))))) ||
|
||||
|
||||
(PIX (0, 0) == PIX (0, -1)
|
||||
&& PIX (0, 0) == PIX (1, 1)
|
||||
&& PIX (0, 0) == PIX (1, 2) &&
|
||||
|
||||
((!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, -1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 0))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, 1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, 2))) ||
|
||||
(!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, -1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, 1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, 2))))) )
|
||||
|
||||
{
|
||||
SCALE_SETPIX (dst_p, PIX (0, 0));
|
||||
}
|
||||
// detect a 1:2 line going across the pixel to the right
|
||||
else if ( (PIX (1, 0) == PIX (1, -1)
|
||||
&& PIX (1, 0) == PIX (0, 1)
|
||||
&& PIX (1, 0) == PIX (0, 2) &&
|
||||
|
||||
((!BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (0, -1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (-1, 1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (-1, 2))) ||
|
||||
(!BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (2, -1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (2, 0))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (1, 1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (1, 2))))) ||
|
||||
|
||||
(PIX (1, 0) == PIX (0, -1)
|
||||
&& PIX (1, 0) == PIX (1, 1)
|
||||
&& PIX (1, 0) == PIX (2, 2) &&
|
||||
|
||||
((!BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (-1, -1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (0, 1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (1, 2))) ||
|
||||
(!BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (1, -1))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (2, 0))
|
||||
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (2, 1))))) )
|
||||
{
|
||||
SCALE_SETPIX (dst_p, PIX (1, 0));
|
||||
}
|
||||
else
|
||||
SCALE_SETPIX (dst_p, Scale_Blend_11 (
|
||||
PIX (0, 0), PIX (1, 0))
|
||||
);
|
||||
}
|
||||
|
||||
if (PIX (0, 0) == PIX (1, 1) && PIX (1, 0) == PIX (0, 1))
|
||||
{
|
||||
// diagonals are equal
|
||||
int *coord;
|
||||
int cl, cr;
|
||||
Uint32 clr;
|
||||
|
||||
// both pairs are equal, have to resolve the pixel
|
||||
// race; we try detecting which color is
|
||||
// the background by looking for a line or an edge
|
||||
// examine 8 pixels surrounding the current quad
|
||||
|
||||
cl = cr = 2;
|
||||
for (coord = resolve_coord[0]; *coord < 100; coord += 2)
|
||||
{
|
||||
clr = PIX (coord[0], coord[1]);
|
||||
|
||||
if (BIADAPT_CMPYUV_MED (clr, PIX (0, 0)))
|
||||
cl++;
|
||||
else if (BIADAPT_CMPYUV_MED (clr, PIX (1, 0)))
|
||||
cr++;
|
||||
}
|
||||
|
||||
// least count wins
|
||||
if (cl > cr)
|
||||
clr = PIX (1, 0);
|
||||
else if (cr > cl)
|
||||
clr = PIX (0, 0);
|
||||
else
|
||||
clr = Scale_Blend_11 (PIX (0, 0), PIX (1, 0));
|
||||
|
||||
SCALE_SETPIX (dst_p + dlen, clr);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (cmatch == 3
|
||||
|| (BIADAPT_CMPYUV_LOW (PIX (1, 0), PIX (0, 1))
|
||||
&& BIADAPT_CMPYUV_LOW (PIX (1, 0), PIX (1, 1))))
|
||||
{
|
||||
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
|
||||
PIX (0, 1), PIX (1, 0))
|
||||
);
|
||||
continue;
|
||||
}
|
||||
else if (cmatch && BIADAPT_CMPYUV_LOW (PIX (0, 0), PIX (1, 1)))
|
||||
{
|
||||
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
|
||||
PIX (0, 0), PIX (1, 1))
|
||||
);
|
||||
continue;
|
||||
}
|
||||
|
||||
// check pixel to the bottom-right
|
||||
if (BIADAPT_CMPYUV_HIGH (PIX (0, 0), PIX (1, 1))
|
||||
&& BIADAPT_CMPYUV_HIGH (PIX (1, 0), PIX (0, 1)))
|
||||
{
|
||||
if (SCALE_GETY (PIX (0, 0)) > SCALE_GETY (PIX (1, 0)))
|
||||
{
|
||||
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
|
||||
PIX (0, 0), PIX (1, 1))
|
||||
);
|
||||
}
|
||||
else
|
||||
{
|
||||
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
|
||||
PIX (1, 0), PIX (0, 1))
|
||||
);
|
||||
}
|
||||
}
|
||||
else if (BIADAPT_CMPYUV_HIGH (PIX (0, 0), PIX (1, 1)))
|
||||
{
|
||||
// main diagonal is same color
|
||||
// use its value
|
||||
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
|
||||
PIX (0, 0), PIX (1, 1))
|
||||
);
|
||||
}
|
||||
else if (BIADAPT_CMPYUV_HIGH (PIX (1, 0), PIX (0, 1)))
|
||||
{
|
||||
// 2nd diagonal is same color
|
||||
// use its value
|
||||
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
|
||||
PIX (1, 0), PIX (0, 1))
|
||||
);
|
||||
}
|
||||
else
|
||||
{
|
||||
// blend all 4
|
||||
SCALE_SETPIX (dst_p + dlen, Scale_Blend_1111 (
|
||||
PIX (0, 0), PIX (0, 1),
|
||||
PIX (1, 0), PIX (1, 1)
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SCALE_(PlatDone) ();
|
||||
}
|
||||
|
||||
#endif /* GFXMODULE_SDL */
|
||||
+115
@@ -0,0 +1,115 @@
|
||||
/*
|
||||
* This program is free software; you can redistribute it and/or modify
|
||||
* it under the terms of the GNU General Public License as published by
|
||||
* the Free Software Foundation; either version 2 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU General Public License
|
||||
* along with this program; if not, write to the Free Software
|
||||
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||
*/
|
||||
|
||||
// Core algorithm of the BiLinear screen scaler
|
||||
// Template
|
||||
// When this file is built standalone is produces a plain C version
|
||||
// Also #included by 2xscalers_mmx.c for an MMX version
|
||||
|
||||
#ifdef GFXMODULE_SDL
|
||||
|
||||
#include "libs/graphics/sdl/sdl_common.h"
|
||||
#include "types.h"
|
||||
#include "scalers.h"
|
||||
#include "scaleint.h"
|
||||
#include "2xscalers.h"
|
||||
|
||||
|
||||
// Bilinear scaling to 2x
|
||||
// The name expands to either
|
||||
// Scale_BilinearFilter (for plain C) or
|
||||
// Scale_MMX_BilinearFilter (for MMX)
|
||||
// Scale_SSE_BilinearFilter (for SSE)
|
||||
// [others when platforms are added]
|
||||
void
|
||||
SCALE_(BilinearFilter) (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r)
|
||||
{
|
||||
int x, y;
|
||||
const int w = src->w, h = src->h;
|
||||
int xend, yend;
|
||||
int dsrc, ddst;
|
||||
SDL_Rect *region = r;
|
||||
SDL_Rect limits;
|
||||
SDL_PixelFormat *fmt = dst->format;
|
||||
const int pitch = src->pitch, dp = dst->pitch;
|
||||
const int bpp = fmt->BytesPerPixel;
|
||||
const int len = pitch / bpp, dlen = dp / bpp;
|
||||
Uint32 p[4]; // influential pixels array
|
||||
Uint32 *srow0 = (Uint32 *) src->pixels;
|
||||
Uint32 *dst_p = (Uint32 *) dst->pixels;
|
||||
|
||||
SCALE_(PlatInit) ();
|
||||
|
||||
// expand updated region if necessary
|
||||
// pixels neighbooring the updated region may
|
||||
// change as a result of updates
|
||||
limits.x = 0;
|
||||
limits.y = 0;
|
||||
limits.w = w;
|
||||
limits.h = h;
|
||||
Scale_ExpandRect (region, 1, &limits);
|
||||
|
||||
xend = region->x + region->w;
|
||||
yend = region->y + region->h;
|
||||
dsrc = len - region->w;
|
||||
ddst = (dlen - region->w) * 2;
|
||||
|
||||
// move ptrs to the first updated pixel
|
||||
srow0 += len * region->y + region->x;
|
||||
dst_p += (dlen * region->y + region->x) * 2;
|
||||
|
||||
for (y = region->y; y < yend; ++y, dst_p += ddst, srow0 += dsrc)
|
||||
{
|
||||
Uint32 *srow1;
|
||||
|
||||
SCALE_(Prefetch) (srow0 + 16);
|
||||
SCALE_(Prefetch) (srow0 + 32);
|
||||
|
||||
if (y < h - 1)
|
||||
srow1 = srow0 + len;
|
||||
else
|
||||
srow1 = srow0;
|
||||
|
||||
SCALE_(Prefetch) (srow1 + 16);
|
||||
SCALE_(Prefetch) (srow1 + 32);
|
||||
|
||||
for (x = region->x; x < xend; ++x, ++srow0, ++srow1, dst_p += 2)
|
||||
{
|
||||
if (x < w - 1)
|
||||
{ // can blend directly from pixels
|
||||
SCALE_BILINEAR_BLEND4 (srow0, srow1, dst_p, dlen);
|
||||
}
|
||||
else
|
||||
{ // need to make temp pixel rows
|
||||
p[0] = srow0[0];
|
||||
p[1] = p[0];
|
||||
p[2] = srow1[0];
|
||||
p[3] = p[2];
|
||||
|
||||
SCALE_BILINEAR_BLEND4 (&p[0], &p[2], dst_p, dlen);
|
||||
}
|
||||
}
|
||||
|
||||
SCALE_(Prefetch) (srow0 + dsrc);
|
||||
SCALE_(Prefetch) (srow0 + dsrc + 16);
|
||||
SCALE_(Prefetch) (srow1 + dsrc);
|
||||
SCALE_(Prefetch) (srow1 + dsrc + 16);
|
||||
}
|
||||
|
||||
SCALE_(PlatDone) ();
|
||||
}
|
||||
|
||||
#endif /* GFXMODULE_SDL */
|
||||
+208
@@ -0,0 +1,208 @@
|
||||
/*
|
||||
* This program is free software; you can redistribute it and/or modify
|
||||
* it under the terms of the GNU General Public License as published by
|
||||
* the Free Software Foundation; either version 2 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU General Public License
|
||||
* along with this program; if not, write to the Free Software
|
||||
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||
*/
|
||||
|
||||
// Core algorithm of the BiLinear screen scaler
|
||||
// Template
|
||||
// When this file is built standalone is produces a plain C version
|
||||
// Also #included by 2xscalers_mmx.c for an MMX version
|
||||
|
||||
#ifdef GFXMODULE_SDL
|
||||
|
||||
#include "libs/graphics/sdl/sdl_common.h"
|
||||
#include "types.h"
|
||||
#include "scalers.h"
|
||||
#include "scaleint.h"
|
||||
#include "2xscalers.h"
|
||||
|
||||
// Nearest Neighbor scaling to 2x
|
||||
// The name expands to
|
||||
// Scale_Nearest (for plain C)
|
||||
// Scale_MMX_Nearest (for MMX)
|
||||
// Scale_SSE_Nearest (for SSE)
|
||||
// [others when platforms are added]
|
||||
void
|
||||
SCALE_(Nearest) (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r)
|
||||
{
|
||||
int y;
|
||||
const int rw = r->w, rh = r->h;
|
||||
const int sp = src->pitch, dp = dst->pitch;
|
||||
const int bpp = dst->format->BytesPerPixel;
|
||||
const int slen = sp / bpp, dlen = dp / bpp;
|
||||
const int dsrc = slen-rw, ddst = (dlen-rw) * 2;
|
||||
|
||||
Uint32 *src_p = (Uint32 *)src->pixels;
|
||||
Uint32 *dst_p = (Uint32 *)dst->pixels;
|
||||
|
||||
// guard asm code against such atrocities
|
||||
if (rw == 0 || rh == 0)
|
||||
return;
|
||||
|
||||
SCALE_(PlatInit) ();
|
||||
|
||||
// move ptrs to the first updated pixel
|
||||
src_p += slen * r->y + r->x;
|
||||
dst_p += (dlen * r->y + r->x) * 2;
|
||||
|
||||
#if defined(MMX_ASM) && defined(MSVC_ASM)
|
||||
// Just about everything has to be done in asm for MSVC
|
||||
// to actually take advantage of asm here
|
||||
// MSVC does not support beautiful GCC-like asm templates
|
||||
|
||||
y = rh;
|
||||
__asm
|
||||
{
|
||||
// setup vars
|
||||
mov esi, src_p
|
||||
mov edi, dst_p
|
||||
|
||||
PREFETCH (esi + 0x40)
|
||||
PREFETCH (esi + 0x80)
|
||||
PREFETCH (esi + 0xc0)
|
||||
|
||||
mov edx, dlen
|
||||
lea edx, [edx * 4]
|
||||
mov eax, dsrc
|
||||
lea eax, [eax * 4]
|
||||
mov ebx, ddst
|
||||
lea ebx, [ebx * 4]
|
||||
|
||||
mov ecx, rw
|
||||
loop_y:
|
||||
test ecx, 1
|
||||
jz even_x
|
||||
|
||||
// one-pixel transfer
|
||||
movd mm1, [esi]
|
||||
punpckldq mm1, mm1 // pix1 | pix1 -> mm1
|
||||
add esi, 4
|
||||
MOVNTQ (edi, mm1)
|
||||
add edi, 8
|
||||
MOVNTQ (edi - 8 + edx, mm1)
|
||||
|
||||
even_x:
|
||||
shr ecx, 1 // x = rw / 2
|
||||
|
||||
loop_x:
|
||||
// two-pixel transfer
|
||||
movq mm1, [esi]
|
||||
movq mm2, mm1
|
||||
PREFETCH (esi + 0x100)
|
||||
punpckldq mm1, mm1 // pix1 | pix1 -> mm1
|
||||
add esi, 8
|
||||
MOVNTQ (edi, mm1)
|
||||
punpckhdq mm2, mm2 // pix2 | pix2 -> mm2
|
||||
MOVNTQ (edi + edx, mm1)
|
||||
add edi, 16
|
||||
MOVNTQ (edi - 8, mm2)
|
||||
MOVNTQ (edi - 8 + edx, mm2)
|
||||
|
||||
dec ecx
|
||||
jnz loop_x
|
||||
|
||||
// try to prefetch as early as possible to have it on time
|
||||
PREFETCH (esi + eax)
|
||||
|
||||
mov ecx, rw
|
||||
add esi, eax
|
||||
|
||||
PREFETCH (esi + 0x40)
|
||||
PREFETCH (esi + 0x80)
|
||||
PREFETCH (esi + 0xc0)
|
||||
|
||||
add edi, ebx
|
||||
|
||||
dec y
|
||||
jnz loop_y
|
||||
}
|
||||
|
||||
#elif defined(MMX_ASM) && defined(GCC_ASM)
|
||||
|
||||
SCALE_(Prefetch) (src_p + 16);
|
||||
SCALE_(Prefetch) (src_p + 32);
|
||||
SCALE_(Prefetch) (src_p + 48);
|
||||
|
||||
for (y = rh; y; --y)
|
||||
{
|
||||
int x = rw;
|
||||
|
||||
if (x & 1)
|
||||
{ // one-pixel transfer
|
||||
__asm__ (
|
||||
"movd (%0), %%mm1 \n\t"
|
||||
"punpckldq %%mm1, %%mm1 \n\t"
|
||||
MOVNTQ (%%mm1, (%1)) "\n\t"
|
||||
MOVNTQ (%%mm1, (%1,%2)) "\n\t"
|
||||
|
||||
: /* nothing */
|
||||
: /*0*/"r" (src_p), /*1*/"r" (dst_p), /*2*/"r" (dlen*sizeof(Uint32))
|
||||
);
|
||||
|
||||
++src_p;
|
||||
dst_p += 2;
|
||||
--x;
|
||||
}
|
||||
|
||||
for (x >>= 1; x; --x, src_p += 2, dst_p += 4)
|
||||
{ // two-pixel transfer
|
||||
__asm__ (
|
||||
"movq (%0), %%mm1 \n\t"
|
||||
"movq %%mm1, %%mm2 \n\t"
|
||||
PREFETCH (0x100(%0)) "\n\t"
|
||||
"punpckldq %%mm1, %%mm1 \n\t"
|
||||
MOVNTQ (%%mm1, (%1)) "\n\t"
|
||||
MOVNTQ (%%mm1, (%1,%2)) "\n\t"
|
||||
"punpckhdq %%mm2, %%mm2 \n\t"
|
||||
MOVNTQ (%%mm2, 8(%1)) "\n\t"
|
||||
MOVNTQ (%%mm2, 8(%1,%2)) "\n\t"
|
||||
|
||||
: /* nothing */
|
||||
: /*0*/"r" (src_p), /*1*/"r" (dst_p), /*2*/"r" (dlen*sizeof(Uint32))
|
||||
);
|
||||
}
|
||||
|
||||
src_p += dsrc;
|
||||
// try to prefetch as early as possible to have it on time
|
||||
SCALE_(Prefetch) (src_p);
|
||||
|
||||
dst_p += ddst;
|
||||
|
||||
SCALE_(Prefetch) (src_p + 16);
|
||||
SCALE_(Prefetch) (src_p + 32);
|
||||
SCALE_(Prefetch) (src_p + 48);
|
||||
}
|
||||
|
||||
#else
|
||||
// Plain C version
|
||||
for (y = 0; y < rh; ++y)
|
||||
{
|
||||
int x;
|
||||
for (x = 0; x < rw; ++x, ++src_p, dst_p += 2)
|
||||
{
|
||||
Uint32 pix = *src_p;
|
||||
dst_p[0] = pix;
|
||||
dst_p[1] = pix;
|
||||
dst_p[dlen] = pix;
|
||||
dst_p[dlen + 1] = pix;
|
||||
}
|
||||
dst_p += ddst;
|
||||
src_p += dsrc;
|
||||
}
|
||||
#endif
|
||||
|
||||
SCALE_(PlatDone) ();
|
||||
}
|
||||
|
||||
#endif /* GFXMODULE_SDL */
|
||||
Executable
+433
@@ -0,0 +1,433 @@
|
||||
/*
|
||||
* Copyright (C) 2005 Alex Volkov (codepro@usa.net)
|
||||
*
|
||||
* This program is free software; you can redistribute it and/or modify
|
||||
* it under the terms of the GNU General Public License as published by
|
||||
* the Free Software Foundation; either version 2 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU General Public License
|
||||
* along with this program; if not, write to the Free Software
|
||||
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||
*/
|
||||
|
||||
// Scalers Internals
|
||||
|
||||
#ifndef SCALEINT_H_
|
||||
#define SCALEINT_H_
|
||||
|
||||
#include "libs/graphics/sdl/sdl_common.h"
|
||||
#include "types.h"
|
||||
|
||||
|
||||
// Plain C names
|
||||
#define SCALE_(name) Scale ## _ ## name
|
||||
|
||||
// These are defaults
|
||||
#define SCALE_GETPIX(p) ( *(Uint32 *)(p) )
|
||||
#define SCALE_SETPIX(p, c) ( *(Uint32 *)(p) = (c) )
|
||||
|
||||
// Plain C defaults
|
||||
#define SCALE_CMPRGB(p1, p2) \
|
||||
SCALE_(GetRGBDelta) (fmt, p1, p2)
|
||||
|
||||
#define SCALE_TOYUV(p) \
|
||||
SCALE_(RGBtoYUV) (fmt, p)
|
||||
|
||||
#define SCALE_CMPYUV(p1, p2, toler) \
|
||||
SCALE_(CmpYUV) (fmt, p1, p2, toler)
|
||||
|
||||
#define SCALE_DIFFYUV(p1, p2) \
|
||||
SCALE_(DiffYUV) (p1, p2)
|
||||
#define SCALE_DIFFYUV_TY 0x40
|
||||
#define SCALE_DIFFYUV_TU 0x12
|
||||
#define SCALE_DIFFYUV_TV 0x0c
|
||||
|
||||
#define SCALE_GETY(p) \
|
||||
SCALE_(GetPixY) (fmt, p)
|
||||
|
||||
#define SCALE_BILINEAR_BLEND4(r0, r1, dst, dlen) \
|
||||
SCALE_(Blend_bilinear) (r0, r1, dst, dlen)
|
||||
|
||||
#define NO_PREFETCH 0
|
||||
#define INTEL_PREFETCH 1
|
||||
#define AMD_PREFETCH 2
|
||||
|
||||
typedef enum
|
||||
{
|
||||
YUV_XFORM_R = 0,
|
||||
YUV_XFORM_G = 1,
|
||||
YUV_XFORM_B = 2,
|
||||
YUV_XFORM_Y = 0,
|
||||
YUV_XFORM_U = 1,
|
||||
YUV_XFORM_V = 2
|
||||
} RGB_YUV_INDEX;
|
||||
|
||||
extern const int YUV_matrix[3][3];
|
||||
|
||||
// pre-computed transformations for 8 bits per channel
|
||||
extern int RGB_to_YUV[/*RGB*/ 3][/*YUV*/ 3][ /*mult-res*/ 256];
|
||||
extern sint16 dRGB_to_dYUV[/*RGB*/ 3][/*YUV*/ 3][ /*mult-res*/ 512];
|
||||
|
||||
typedef Uint32 YUV_VECTOR;
|
||||
// pre-computed transformations for RGB555
|
||||
extern YUV_VECTOR RGB15_to_YUV[0x8000];
|
||||
|
||||
|
||||
// Platform+Scaler function lookups
|
||||
//
|
||||
typedef struct
|
||||
{
|
||||
int flag;
|
||||
TFB_ScaleFunc func;
|
||||
} Scale_FuncDef_t;
|
||||
|
||||
|
||||
// expands the given rectangle in all directions by 'expansion'
|
||||
// guarded by 'limits'
|
||||
extern void Scale_ExpandRect (SDL_Rect* rect, int expansion,
|
||||
const SDL_Rect* limits);
|
||||
|
||||
|
||||
// Standard plain C versions of support functions
|
||||
|
||||
// Initialize various platform-specific features
|
||||
static inline void
|
||||
SCALE_(PlatInit) (void)
|
||||
{
|
||||
}
|
||||
|
||||
// Finish with various platform-specific features
|
||||
static inline void
|
||||
SCALE_(PlatDone) (void)
|
||||
{
|
||||
}
|
||||
|
||||
#if 0
|
||||
static inline void
|
||||
SCALE_(Prefetch) (const void* p)
|
||||
{
|
||||
/* no-op in pure C */
|
||||
(void)p;
|
||||
}
|
||||
#else
|
||||
# define Scale_Prefetch(p)
|
||||
#endif
|
||||
|
||||
// compute the RGB distance squared between 2 pixels
|
||||
// Plain C version
|
||||
static inline int
|
||||
SCALE_(GetRGBDelta) (const SDL_PixelFormat* fmt, Uint32 pix1, Uint32 pix2)
|
||||
{
|
||||
int c;
|
||||
int delta;
|
||||
|
||||
c = ((pix1 >> fmt->Rshift) & 0xff) - ((pix2 >> fmt->Rshift) & 0xff);
|
||||
delta = c * c;
|
||||
|
||||
c = ((pix1 >> fmt->Gshift) & 0xff) - ((pix2 >> fmt->Gshift) & 0xff);
|
||||
delta += c * c;
|
||||
|
||||
c = ((pix1 >> fmt->Bshift) & 0xff) - ((pix2 >> fmt->Bshift) & 0xff);
|
||||
delta += c * c;
|
||||
|
||||
return delta;
|
||||
}
|
||||
|
||||
// retrieve the Y (intensity) component of pixel's YUV
|
||||
// Plain C version
|
||||
static inline int
|
||||
SCALE_(GetPixY) (const SDL_PixelFormat* fmt, Uint32 pix)
|
||||
{
|
||||
Uint32 r, g, b;
|
||||
|
||||
r = (pix >> fmt->Rshift) & 0xff;
|
||||
g = (pix >> fmt->Gshift) & 0xff;
|
||||
b = (pix >> fmt->Bshift) & 0xff;
|
||||
|
||||
return RGB_to_YUV [YUV_XFORM_R][YUV_XFORM_Y][r]
|
||||
+ RGB_to_YUV [YUV_XFORM_G][YUV_XFORM_Y][g]
|
||||
+ RGB_to_YUV [YUV_XFORM_B][YUV_XFORM_Y][b];
|
||||
}
|
||||
|
||||
static inline YUV_VECTOR
|
||||
SCALE_(RGBtoYUV) (const SDL_PixelFormat* fmt, Uint32 pix)
|
||||
{
|
||||
return RGB15_to_YUV[
|
||||
(((pix >> (fmt->Rshift + 3)) & 0x1f) << 10) |
|
||||
(((pix >> (fmt->Gshift + 3)) & 0x1f) << 5) |
|
||||
(((pix >> (fmt->Bshift + 3)) & 0x1f) )
|
||||
];
|
||||
}
|
||||
|
||||
// compare 2 pixels with respect to their YUV representations
|
||||
// tolerance set by toler arg
|
||||
// returns true: close; false: distant (-gt toler)
|
||||
// Plain C version
|
||||
static inline bool
|
||||
SCALE_(CmpYUV) (const SDL_PixelFormat* fmt, Uint32 pix1, Uint32 pix2, int toler)
|
||||
#if 1
|
||||
{
|
||||
int dr, dg, db;
|
||||
int delta;
|
||||
|
||||
dr = ((pix1 >> fmt->Rshift) & 0xff) - ((pix2 >> fmt->Rshift) & 0xff) + 255;
|
||||
dg = ((pix1 >> fmt->Gshift) & 0xff) - ((pix2 >> fmt->Gshift) & 0xff) + 255;
|
||||
db = ((pix1 >> fmt->Bshift) & 0xff) - ((pix2 >> fmt->Bshift) & 0xff) + 255;
|
||||
|
||||
// compute Y delta
|
||||
delta = abs (dRGB_to_dYUV [YUV_XFORM_R][YUV_XFORM_Y][dr]
|
||||
+ dRGB_to_dYUV [YUV_XFORM_G][YUV_XFORM_Y][dg]
|
||||
+ dRGB_to_dYUV [YUV_XFORM_B][YUV_XFORM_Y][db]);
|
||||
if (delta > toler)
|
||||
return false;
|
||||
|
||||
// compute U delta
|
||||
delta += abs (dRGB_to_dYUV [YUV_XFORM_R][YUV_XFORM_U][dr]
|
||||
+ dRGB_to_dYUV [YUV_XFORM_G][YUV_XFORM_U][dg]
|
||||
+ dRGB_to_dYUV [YUV_XFORM_B][YUV_XFORM_U][db]);
|
||||
if (delta > toler)
|
||||
return false;
|
||||
|
||||
// compute V delta
|
||||
delta += abs (dRGB_to_dYUV [YUV_XFORM_R][YUV_XFORM_V][dr]
|
||||
+ dRGB_to_dYUV [YUV_XFORM_G][YUV_XFORM_V][dg]
|
||||
+ dRGB_to_dYUV [YUV_XFORM_B][YUV_XFORM_V][db]);
|
||||
|
||||
return delta <= toler;
|
||||
}
|
||||
#else
|
||||
{
|
||||
int delta;
|
||||
Uint32 yuv1, yuv2;
|
||||
|
||||
yuv1 = RGB15_to_YUV[
|
||||
(((pix1 >> (fmt->Rshift + 3)) & 0x1f) << 10) |
|
||||
(((pix1 >> (fmt->Gshift + 3)) & 0x1f) << 5) |
|
||||
(((pix1 >> (fmt->Bshift + 3)) & 0x1f) )
|
||||
];
|
||||
|
||||
yuv2 = RGB15_to_YUV[
|
||||
(((pix2 >> (fmt->Rshift + 3)) & 0x1f) << 10) |
|
||||
(((pix2 >> (fmt->Gshift + 3)) & 0x1f) << 5) |
|
||||
(((pix2 >> (fmt->Bshift + 3)) & 0x1f) )
|
||||
];
|
||||
|
||||
// compute Y delta
|
||||
delta = abs ((yuv1 & 0xff0000) - (yuv2 & 0xff0000)) >> 16;
|
||||
if (delta > toler)
|
||||
return false;
|
||||
|
||||
// compute U delta
|
||||
delta += abs ((yuv1 & 0x00ff00) - (yuv2 & 0x00ff00)) >> 8;
|
||||
if (delta > toler)
|
||||
return false;
|
||||
|
||||
// compute V delta
|
||||
delta += abs ((yuv1 & 0x0000ff) - (yuv2 & 0x0000ff));
|
||||
|
||||
return delta <= toler;
|
||||
}
|
||||
#endif
|
||||
|
||||
// Check if 2 pixels are different with respect to their
|
||||
// YUV representations
|
||||
// returns 0: close; ~0: distant
|
||||
static inline int
|
||||
SCALE_(DiffYUV) (Uint32 yuv1, Uint32 yuv2)
|
||||
{
|
||||
// non-branching version -- assumes 2's complement integers
|
||||
// delta math only needs 25 bits and we have 32 available;
|
||||
// only interested in the sign bits after subtraction
|
||||
sint32 delta, ret;
|
||||
|
||||
if (yuv1 == yuv2)
|
||||
return 0;
|
||||
|
||||
// compute Y delta
|
||||
delta = abs ((yuv1 & 0xff0000) - (yuv2 & 0xff0000));
|
||||
ret = (SCALE_DIFFYUV_TY << 16) - delta; // save sign bit
|
||||
|
||||
// compute U delta
|
||||
delta = abs ((yuv1 & 0x00ff00) - (yuv2 & 0x00ff00));
|
||||
ret |= (SCALE_DIFFYUV_TU << 8) - delta; // save sign bit
|
||||
|
||||
// compute V delta
|
||||
delta = abs ((yuv1 & 0x0000ff) - (yuv2 & 0x0000ff));
|
||||
ret |= SCALE_DIFFYUV_TV - delta; // save sign bit
|
||||
|
||||
return (ret >> 31);
|
||||
}
|
||||
|
||||
// blends two pixels with 1:1 ratio
|
||||
static inline Uint32
|
||||
SCALE_(Blend_11) (Uint32 pix1, Uint32 pix2)
|
||||
{
|
||||
/* (pix1 + pix2) >> 1 */
|
||||
return
|
||||
/* lower bits can be safely ignored - the error is minimal
|
||||
expression that calcs them is left for posterity
|
||||
(pix1 & pix2 & low_mask) +
|
||||
*/
|
||||
((pix1 & 0xfefefefe) >> 1) + ((pix2 & 0xfefefefe) >> 1);
|
||||
}
|
||||
|
||||
// blends four pixels with 1:1:1:1 ratio
|
||||
static inline Uint32
|
||||
SCALE_(Blend_1111) (Uint32 pix1, Uint32 pix2,
|
||||
Uint32 pix3, Uint32 pix4)
|
||||
{
|
||||
/* (pix1 + pix2 + pix3 + pix4) >> 2 */
|
||||
return
|
||||
/* lower bits can be safely ignored - the error is minimal
|
||||
expression that calcs them is left for posterity
|
||||
((((pix1 & low_mask) + (pix2 & low_mask) +
|
||||
(pix3 & low_mask) + (pix4 & low_mask)
|
||||
) >> 2) & low_mask) +
|
||||
*/
|
||||
((pix1 & 0xfcfcfcfc) >> 2) + ((pix2 & 0xfcfcfcfc) >> 2) +
|
||||
((pix3 & 0xfcfcfcfc) >> 2) + ((pix4 & 0xfcfcfcfc) >> 2);
|
||||
}
|
||||
|
||||
// blends pixels with 3:1 ratio
|
||||
static inline Uint32
|
||||
Scale_Blend_31 (Uint32 pix1, Uint32 pix2)
|
||||
{
|
||||
/* (pix1 * 3 + pix2) / 4 */
|
||||
/* lower bits can be safely ignored - the error is minimal */
|
||||
return ((pix1 & 0xfefefefe) >> 1) + ((pix1 & 0xfcfcfcfc) >> 2) +
|
||||
((pix2 & 0xfcfcfcfc) >> 2);
|
||||
}
|
||||
|
||||
// blends pixels with 2:1:1 ratio
|
||||
static inline Uint32
|
||||
Scale_Blend_211 (Uint32 pix1, Uint32 pix2, Uint32 pix3)
|
||||
{
|
||||
/* (pix1 * 2 + pix2 + pix3) / 4 */
|
||||
/* lower bits can be safely ignored - the error is minimal */
|
||||
return ((pix1 & 0xfefefefe) >> 1) +
|
||||
((pix2 & 0xfcfcfcfc) >> 2) +
|
||||
((pix3 & 0xfcfcfcfc) >> 2);
|
||||
}
|
||||
|
||||
// blends pixels with 5:2:1 ratio
|
||||
static inline Uint32
|
||||
Scale_Blend_521 (Uint32 pix1, Uint32 pix2, Uint32 pix3)
|
||||
{
|
||||
/* (pix1 * 5 + pix2 * 2 + pix3) / 8 */
|
||||
/* lower bits can be safely ignored - the error is minimal */
|
||||
return ((pix1 & 0xfefefefe) >> 1) + ((pix1 & 0xf8f8f8f8) >> 3) +
|
||||
((pix2 & 0xfcfcfcfc) >> 2) +
|
||||
((pix3 & 0xf8f8f8f8) >> 3) +
|
||||
0x02020202 /* half-error */;
|
||||
}
|
||||
|
||||
// blends pixels with 6:1:1 ratio
|
||||
static inline Uint32
|
||||
Scale_Blend_611 (Uint32 pix1, Uint32 pix2, Uint32 pix3)
|
||||
{
|
||||
/* (pix1 * 6 + pix2 + pix3) / 8 */
|
||||
/* lower bits can be safely ignored - the error is minimal */
|
||||
return ((pix1 & 0xfefefefe) >> 1) + ((pix1 & 0xfcfcfcfc) >> 2) +
|
||||
((pix2 & 0xf8f8f8f8) >> 3) +
|
||||
((pix3 & 0xf8f8f8f8) >> 3) +
|
||||
0x02020202 /* half-error */;
|
||||
}
|
||||
|
||||
// blends pixels with 2:3:3 ratio
|
||||
static inline Uint32
|
||||
Scale_Blend_233 (Uint32 pix1, Uint32 pix2, Uint32 pix3)
|
||||
{
|
||||
/* (pix1 * 2 + pix2 * 3 + pix3 * 3) / 8 */
|
||||
/* lower bits can be safely ignored - the error is minimal */
|
||||
return ((pix1 & 0xfcfcfcfc) >> 2) +
|
||||
((pix2 & 0xfcfcfcfc) >> 2) + ((pix2 & 0xf8f8f8f8) >> 3) +
|
||||
((pix3 & 0xfcfcfcfc) >> 2) + ((pix3 & 0xf8f8f8f8) >> 3) +
|
||||
0x02020202 /* half-error */;
|
||||
}
|
||||
|
||||
// blends pixels with 14:1:1 ratio
|
||||
static inline Uint32
|
||||
Scale_Blend_e11 (Uint32 pix1, Uint32 pix2, Uint32 pix3)
|
||||
{
|
||||
/* (pix1 * 14 + pix2 + pix3) >> 4 */
|
||||
/* lower bits can be safely ignored - the error is minimal */
|
||||
return ((pix1 & 0xfefefefe) >> 1) + ((pix1 & 0xfcfcfcfc) >> 2) +
|
||||
((pix1 & 0xf8f8f8f8) >> 3) +
|
||||
((pix2 & 0xf0f0f0f0) >> 4) +
|
||||
((pix3 & 0xf0f0f0f0) >> 4) +
|
||||
0x03030303 /* half-error */;
|
||||
}
|
||||
|
||||
// Halfs the pixel's intensity
|
||||
static inline Uint32
|
||||
SCALE_(HalfPixel) (Uint32 pix)
|
||||
{
|
||||
return ((pix & 0xfefefefe) >> 1);
|
||||
}
|
||||
|
||||
|
||||
// Bilinear weighted blend of four pixels
|
||||
// Function produces 4 blended pixels and writes them
|
||||
// out to the surface (in 2x2 matrix)
|
||||
// Pixels are computed using expanded weight matrix like so:
|
||||
// ('sp' - source pixel, 'dp' - destination pixel)
|
||||
// dp[0] = (9*sp[0] + 3*sp[1] + 3*sp[2] + 1*sp[3]) / 16
|
||||
// dp[1] = (3*sp[0] + 9*sp[1] + 1*sp[2] + 3*sp[3]) / 16
|
||||
// dp[2] = (3*sp[0] + 1*sp[1] + 9*sp[2] + 3*sp[3]) / 16
|
||||
// dp[3] = (1*sp[0] + 3*sp[1] + 3*sp[2] + 9*sp[3]) / 16
|
||||
static inline void
|
||||
SCALE_(Blend_bilinear) (const Uint32* row0, const Uint32* row1,
|
||||
Uint32* dst_p, Uint32 dlen)
|
||||
{
|
||||
// We loose some lower bits here and try to compensate for
|
||||
// that by adding half-error values.
|
||||
// In general, the error is minimal (+-7)
|
||||
// The >>4 reduction is achieved gradually
|
||||
# define BL_PACKED_HALF(p) \
|
||||
(((p) & 0xfefefefe) >> 1)
|
||||
# define BL_SUM(p1, p2) \
|
||||
(BL_PACKED_HALF(p1) + BL_PACKED_HALF(p2))
|
||||
# define BL_HALF_ERR 0x01010101
|
||||
# define BL_SUM_WERR(p1, p2) \
|
||||
(BL_PACKED_HALF(p1) + BL_PACKED_HALF(p2) + BL_HALF_ERR)
|
||||
|
||||
Uint32 sum1111, sum1331, sum3113;
|
||||
|
||||
// cache p[0] + 3*(p[1] + p[2]) + p[3] in sum1331
|
||||
// cache p[1] + 3*(p[0] + p[3]) + p[2] in sum3113
|
||||
sum1331 = BL_SUM (row0[1], row1[0]);
|
||||
sum3113 = BL_SUM (row0[0], row1[1]);
|
||||
|
||||
// cache p[0] + p[1] + p[2] + p[3] in sum1111
|
||||
sum1111 = BL_SUM_WERR (sum1331, sum3113);
|
||||
|
||||
sum1331 = BL_SUM_WERR (sum1331, sum1111);
|
||||
sum1331 = BL_PACKED_HALF (sum1331);
|
||||
sum3113 = BL_SUM_WERR (sum3113, sum1111);
|
||||
sum3113 = BL_PACKED_HALF (sum3113);
|
||||
|
||||
// pixel 0 math -- (9*p[0] + 3*(p[1] + p[2]) + p[3]) / 16
|
||||
dst_p[0] = BL_PACKED_HALF (row0[0]) + sum1331;
|
||||
|
||||
// pixel 1 math -- (9*p[1] + 3*(p[0] + p[3]) + p[2]) / 16
|
||||
dst_p[1] = BL_PACKED_HALF (row0[1]) + sum3113;
|
||||
|
||||
// pixel 2 math -- (9*p[2] + 3*(p[0] + p[3]) + p[1]) / 16
|
||||
dst_p[dlen] = BL_PACKED_HALF (row1[0]) + sum3113;
|
||||
|
||||
// pixel 3 math -- (9*p[3] + 3*(p[1] + p[2]) + p[0]) / 16
|
||||
dst_p[dlen + 1] = BL_PACKED_HALF (row1[1]) + sum1331;
|
||||
|
||||
# undef BL_PACKED_HALF
|
||||
# undef BL_SUM
|
||||
# undef BL_HALF_ERR
|
||||
# undef BL_SUM_WERR
|
||||
}
|
||||
|
||||
#endif /* SCALEINT_H_ */
|
||||
Executable
+790
@@ -0,0 +1,790 @@
|
||||
/*
|
||||
* Copyright (C) 2005 Alex Volkov (codepro@usa.net)
|
||||
*
|
||||
* This program is free software; you can redistribute it and/or modify
|
||||
* it under the terms of the GNU General Public License as published by
|
||||
* the Free Software Foundation; either version 2 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU General Public License
|
||||
* along with this program; if not, write to the Free Software
|
||||
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||
*/
|
||||
|
||||
#ifndef SCALEMMX_H_
|
||||
#define SCALEMMX_H_
|
||||
|
||||
#if !defined(SCALE_)
|
||||
# error Please define SCALE_(name) before including scalemmx.h
|
||||
#endif
|
||||
|
||||
#if !defined(MSVC_ASM) && !defined(GCC_ASM)
|
||||
# error Please define target assembler (MSVC_ASM, GCC_ASM) before including scalemmx.h
|
||||
#endif
|
||||
|
||||
// MMX defaults (no Format param)
|
||||
#undef SCALE_CMPRGB
|
||||
#define SCALE_CMPRGB(p1, p2) \
|
||||
SCALE_(GetRGBDelta) (p1, p2)
|
||||
|
||||
#undef SCALE_TOYUV
|
||||
#define SCALE_TOYUV(p) \
|
||||
SCALE_(RGBtoYUV) (p)
|
||||
|
||||
#undef SCALE_CMPYUV
|
||||
#define SCALE_CMPYUV(p1, p2, toler) \
|
||||
SCALE_(CmpYUV) (p1, p2, toler)
|
||||
|
||||
#undef SCALE_GETY
|
||||
#define SCALE_GETY(p) \
|
||||
SCALE_(GetPixY) (p)
|
||||
|
||||
// MMX transformation multipliers
|
||||
extern Uint64 mmx_888to555_mult;
|
||||
extern Uint64 mmx_Y_mult;
|
||||
extern Uint64 mmx_U_mult;
|
||||
extern Uint64 mmx_V_mult;
|
||||
extern Uint64 mmx_YUV_threshold;
|
||||
|
||||
#define USE_YUV_LOOKUP
|
||||
|
||||
#if defined(MSVC_ASM)
|
||||
// MSVC inline assembly versions
|
||||
|
||||
#if defined(USE_MOVNTQ)
|
||||
# define MOVNTQ(addr, val) movntq [addr], val
|
||||
#else
|
||||
# define MOVNTQ(addr, val) movq [addr], val
|
||||
#endif
|
||||
|
||||
#if USE_PREFETCH == INTEL_PREFETCH
|
||||
// using Intel SSE non-temporal prefetch
|
||||
# define PREFETCH(addr) prefetchnta [addr]
|
||||
# define HAVE_PREFETCH
|
||||
#elif USE_PREFETCH == AMD_PREFETCH
|
||||
// using AMD 3DNOW! prefetch
|
||||
# define PREFETCH(addr) prefetch [addr]
|
||||
# define HAVE_PREFETCH
|
||||
#else
|
||||
// no prefetch -- too bad for poor MMX-only souls
|
||||
# define PREFETCH(addr)
|
||||
# undef HAVE_PREFETCH
|
||||
#endif
|
||||
|
||||
|
||||
static inline void
|
||||
SCALE_(PlatInit) (void)
|
||||
{
|
||||
__asm
|
||||
{
|
||||
// mm0 will be kept == 0 throughout
|
||||
// 0 is needed for bytes->words unpack instructions
|
||||
pxor mm0, mm0
|
||||
|
||||
// mm5-mm7 contains RGB->YUV mults
|
||||
movq mm7, mmx_Y_mult
|
||||
movq mm6, mmx_U_mult
|
||||
movq mm5, mmx_V_mult
|
||||
#ifdef USE_YUV_LOOKUP
|
||||
// mm6 contains RGB888->555 shuffle mult
|
||||
movq mm6, mmx_888to555_mult
|
||||
// mm5 contains DiffYUV threshold
|
||||
movq mm5, mmx_YUV_threshold
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
static inline void
|
||||
SCALE_(PlatDone) (void)
|
||||
{
|
||||
// finish with MMX registers and yield them to FPU
|
||||
__asm
|
||||
{
|
||||
emms
|
||||
}
|
||||
}
|
||||
|
||||
#if defined(HAVE_PREFETCH)
|
||||
static inline void
|
||||
SCALE_(Prefetch) (const void* p)
|
||||
{
|
||||
__asm
|
||||
{
|
||||
mov eax, p
|
||||
PREFETCH (eax)
|
||||
}
|
||||
}
|
||||
|
||||
#else /* Not HAVE_PREFETCH */
|
||||
|
||||
static inline void
|
||||
SCALE_(Prefetch) (const void* p) { /* no-op */ }
|
||||
|
||||
#endif /* HAVE_PREFETCH */
|
||||
|
||||
// compute the RGB distance squared between 2 pixels
|
||||
static inline int
|
||||
SCALE_(GetRGBDelta) (Uint32 pix1, Uint32 pix2)
|
||||
{
|
||||
__asm
|
||||
{
|
||||
// load pixels
|
||||
movd mm1, pix1
|
||||
punpcklbw mm1, mm0
|
||||
movd mm2, pix2
|
||||
punpcklbw mm2, mm0
|
||||
// get the difference between RGBA components
|
||||
psubw mm1, mm2
|
||||
// squared and sumed
|
||||
pmaddwd mm1, mm1
|
||||
// finish suming the squares
|
||||
movq mm2, mm1
|
||||
punpckhdq mm2, mm0
|
||||
paddd mm1, mm2
|
||||
// store result
|
||||
movd eax, mm1
|
||||
}
|
||||
}
|
||||
|
||||
// retrieve the Y (intensity) component of pixel's YUV
|
||||
static inline int
|
||||
SCALE_(GetPixY) (Uint32 pix)
|
||||
{
|
||||
__asm
|
||||
{
|
||||
// load pixel
|
||||
movd mm1, pix
|
||||
punpcklbw mm1, mm0
|
||||
// process
|
||||
pmaddwd mm1, mmx_Y_mult // RGB * Yvec
|
||||
movq mm2, mm1 // finish suming
|
||||
punpckhdq mm2, mm0 // ditto
|
||||
paddd mm1, mm2 // ditto
|
||||
// store result
|
||||
movd eax, mm1
|
||||
shr eax, 14
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef USE_YUV_LOOKUP
|
||||
|
||||
// convert pixel RGB vector into YUV representation vector
|
||||
static inline YUV_VECTOR
|
||||
SCALE_(RGBtoYUV) (Uint32 pix)
|
||||
{
|
||||
__asm
|
||||
{
|
||||
// convert RGB888 to 555
|
||||
movd mm1, pix
|
||||
punpcklbw mm1, mm0
|
||||
psrlw mm1, 3 // 8->5 bit
|
||||
pmaddwd mm1, mmx_888to555_mult // shuffle into the right channel order
|
||||
movq mm2, mm1 // finish shuffling
|
||||
punpckhdq mm2, mm0 // ditto
|
||||
por mm1, mm2 // ditto
|
||||
|
||||
// lookup the YUV vector
|
||||
movd eax, mm1
|
||||
mov eax, [RGB15_to_YUV + eax * 4]
|
||||
}
|
||||
}
|
||||
|
||||
// compare 2 pixels with respect to their YUV representations
|
||||
// tolerance set by toler arg
|
||||
// returns true: close; false: distant (-gt toler)
|
||||
static inline bool
|
||||
SCALE_(CmpYUV) (Uint32 pix1, Uint32 pix2, int toler)
|
||||
{
|
||||
__asm
|
||||
{
|
||||
// convert RGB888 to 555
|
||||
movd mm1, pix1
|
||||
punpcklbw mm1, mm0
|
||||
psrlw mm1, 3 // 8->5 bit
|
||||
movd mm3, pix2
|
||||
punpcklbw mm3, mm0
|
||||
psrlw mm3, 3 // 8->5 bit
|
||||
pmaddwd mm1, mmx_888to555_mult // shuffle into the right channel order
|
||||
movq mm2, mm1 // finish shuffling
|
||||
pmaddwd mm3, mmx_888to555_mult // shuffle into the right channel order
|
||||
movq mm4, mm3 // finish shuffling
|
||||
punpckhdq mm2, mm0 // ditto
|
||||
por mm1, mm2 // ditto
|
||||
punpckhdq mm4, mm0 // ditto
|
||||
por mm3, mm4 // ditto
|
||||
|
||||
// lookup the YUV vector
|
||||
movd eax, mm1
|
||||
movd edx, mm3
|
||||
movd mm1, [RGB15_to_YUV + eax * 4]
|
||||
movq mm4, mm1
|
||||
movd mm2, [RGB15_to_YUV + edx * 4]
|
||||
|
||||
// get abs difference between YUV components
|
||||
#ifdef USE_PSADBW
|
||||
// we can use PSADBW and save us some grief
|
||||
psadbw mm1, mm2
|
||||
movd edx, mm1
|
||||
#else
|
||||
// no PSADBW -- have to do it the hard way
|
||||
psubusb mm1, mm2
|
||||
psubusb mm2, mm4
|
||||
por mm1, mm2
|
||||
|
||||
// sum the differences
|
||||
// XXX: technically, this produces a MAX diff of 510
|
||||
// but we do not need anything bigger, currently
|
||||
movq mm2, mm1
|
||||
psrlq mm2, 8
|
||||
paddusb mm1, mm2
|
||||
psrlq mm2, 8
|
||||
paddusb mm1, mm2
|
||||
movd edx, mm1
|
||||
and edx, 0xff
|
||||
#endif /* USE_PSADBW */
|
||||
xor eax, eax
|
||||
shl edx, 1
|
||||
cmp edx, toler
|
||||
// store result
|
||||
setle al
|
||||
}
|
||||
}
|
||||
|
||||
#else /* Not USE_YUV_LOOKUP */
|
||||
|
||||
// convert pixel RGB vector into YUV representation vector
|
||||
static inline YUV_VECTOR
|
||||
SCALE_(RGBtoYUV) (Uint32 pix)
|
||||
{
|
||||
__asm
|
||||
{
|
||||
movd mm1, pix
|
||||
punpcklbw mm1, mm0
|
||||
|
||||
movq mm2, mm1
|
||||
|
||||
// Y vector multiply
|
||||
pmaddwd mm1, mmx_Y_mult
|
||||
movq mm4, mm1
|
||||
punpckhdq mm4, mm0
|
||||
punpckldq mm1, mm0 // clear out the high dword
|
||||
paddd mm1, mm4
|
||||
psrad mm1, 15
|
||||
|
||||
movq mm3, mm2
|
||||
|
||||
// U vector multiply
|
||||
pmaddwd mm2, mmx_U_mult
|
||||
psrad mm2, 10
|
||||
|
||||
// V vector multiply
|
||||
pmaddwd mm3, mmx_V_mult
|
||||
psrad mm3, 10
|
||||
|
||||
// load (1|1|1|1) into mm4
|
||||
pcmpeqw mm4, mm4
|
||||
psrlw mm4, 15
|
||||
|
||||
packssdw mm3, mm2
|
||||
pmaddwd mm3, mm4
|
||||
psrad mm3, 5
|
||||
|
||||
// load (64|64) into mm4
|
||||
punpcklwd mm4, mm0
|
||||
pslld mm4, 6
|
||||
paddd mm3, mm4
|
||||
|
||||
packssdw mm3, mm1
|
||||
packuswb mm3, mm0
|
||||
|
||||
movd eax, mm3
|
||||
}
|
||||
}
|
||||
|
||||
// compare 2 pixels with respect to their YUV representations
|
||||
// tolerance set by toler arg
|
||||
// returns true: close; false: distant (-gt toler)
|
||||
static inline bool
|
||||
SCALE_(CmpYUV) (Uint32 pix1, Uint32 pix2, int toler)
|
||||
{
|
||||
__asm
|
||||
{
|
||||
movd mm1, pix1
|
||||
punpcklbw mm1, mm0
|
||||
movd mm2, pix2
|
||||
punpcklbw mm2, mm0
|
||||
|
||||
psubw mm1, mm2
|
||||
movq mm2, mm1
|
||||
|
||||
// Y vector multiply
|
||||
pmaddwd mm1, mmx_Y_mult
|
||||
movq mm4, mm1
|
||||
punpckhdq mm4, mm0
|
||||
paddd mm1, mm4
|
||||
// abs()
|
||||
movq mm4, mm1
|
||||
psrad mm4, 31
|
||||
pxor mm4, mm1
|
||||
psubd mm1, mm4
|
||||
|
||||
movq mm3, mm2
|
||||
|
||||
// U vector multiply
|
||||
pmaddwd mm2, mmx_U_mult
|
||||
movq mm4, mm2
|
||||
punpckhdq mm4, mm0
|
||||
paddd mm2, mm4
|
||||
// abs()
|
||||
movq mm4, mm2
|
||||
psrad mm4, 31
|
||||
pxor mm4, mm2
|
||||
psubd mm2, mm4
|
||||
|
||||
paddd mm1, mm2
|
||||
|
||||
// V vector multiply
|
||||
pmaddwd mm3, mmx_V_mult
|
||||
movq mm4, mm3
|
||||
punpckhdq mm3, mm0
|
||||
paddd mm3, mm4
|
||||
// abs()
|
||||
movq mm4, mm3
|
||||
psrad mm4, 31
|
||||
pxor mm4, mm3
|
||||
psubd mm3, mm4
|
||||
|
||||
paddd mm1, mm3
|
||||
|
||||
movd edx, mm1
|
||||
xor eax, eax
|
||||
shr edx, 14
|
||||
cmp edx, toler
|
||||
// store result
|
||||
setle al
|
||||
}
|
||||
}
|
||||
|
||||
#endif /* USE_YUV_LOOKUP */
|
||||
|
||||
// Check if 2 pixels are different with respect to their
|
||||
// YUV representations
|
||||
// returns 0: close; ~0: distant
|
||||
static inline int
|
||||
SCALE_(DiffYUV) (Uint32 yuv1, Uint32 yuv2)
|
||||
{
|
||||
__asm
|
||||
{
|
||||
// load YUV pixels
|
||||
movd mm1, yuv1
|
||||
movq mm4, mm1
|
||||
movd mm2, yuv2
|
||||
// abs difference between channels
|
||||
psubusb mm1, mm2
|
||||
psubusb mm2, mm4
|
||||
por mm1, mm2
|
||||
// compare to threshold
|
||||
psubusb mm1, mmx_YUV_threshold
|
||||
|
||||
movd edx, mm1
|
||||
// transform eax to 0 or ~0
|
||||
xor eax, eax
|
||||
or edx, edx
|
||||
setz al
|
||||
dec eax
|
||||
}
|
||||
}
|
||||
|
||||
// bilinear weighted blend of four pixels
|
||||
// MSVC asm version
|
||||
static inline void
|
||||
SCALE_(Blend_bilinear) (const Uint32* row0, const Uint32* row1,
|
||||
Uint32* dst_p, Uint32 dlen)
|
||||
{
|
||||
__asm
|
||||
{
|
||||
// EL0: setup vars
|
||||
mov ebx, row0 // EL0
|
||||
|
||||
// EL0: load pixels
|
||||
movq mm1, [ebx] // EL0
|
||||
movq mm2, mm1 // EL0: p[1] -> mm2
|
||||
PREFETCH (ebx + 0x80)
|
||||
punpckhbw mm2, mm0 // EL0: p[1] -> mm2
|
||||
mov ebx, row1
|
||||
punpcklbw mm1, mm0 // EL0: p[0] -> mm1
|
||||
movq mm3, [ebx]
|
||||
movq mm4, mm3 // EL0: p[3] -> mm4
|
||||
movq mm6, mm2 // EL1.1: p[1] -> mm6
|
||||
PREFETCH (ebx + 0x80)
|
||||
punpcklbw mm3, mm0 // EL0: p[2] -> mm3
|
||||
movq mm5, mm1 // EL1.1: p[0] -> mm5
|
||||
punpckhbw mm4, mm0 // EL0: p[3] -> mm4
|
||||
|
||||
mov edi, dst_p // EL0
|
||||
|
||||
// EL1: cache p[0] + 3*(p[1] + p[2]) + p[3] in mm6
|
||||
paddw mm6, mm3 // EL1.2: p[1] + p[2] -> mm6
|
||||
// EL1: cache p[0] + p[1] + p[2] + p[3] in mm7
|
||||
movq mm7, mm6 // EL1.3: p[1] + p[2] -> mm7
|
||||
// EL1: cache p[1] + 3*(p[0] + p[3]) + p[2] in mm5
|
||||
paddw mm5, mm4 // EL1.2: p[0] + p[3] -> mm5
|
||||
psllw mm6, 1 // EL1.4: 2*(p[1] + p[2]) -> mm6
|
||||
paddw mm7, mm5 // EL1.4: sum(p[]) -> mm7
|
||||
psllw mm5, 1 // EL1.5: 2*(p[0] + p[3]) -> mm5
|
||||
paddw mm6, mm7 // EL1.5: p[0] + 3*(p[1] + p[2]) + p[3] -> mm6
|
||||
paddw mm5, mm7 // EL1.6: p[1] + 3*(p[0] + p[3]) + p[2] -> mm5
|
||||
|
||||
// EL2: pixel 0 math -- (9*p[0] + 3*(p[1] + p[2]) + p[3]) / 16
|
||||
psllw mm1, 3 // EL2.1: 8*p[0] -> mm1
|
||||
paddw mm1, mm6 // EL2.2: 9*p[0] + 3*(p[1] + p[2]) + p[3] -> mm1
|
||||
psrlw mm1, 4 // EL2.3: sum[0]/16 -> mm1
|
||||
|
||||
mov edx, dlen // EL0
|
||||
|
||||
// EL3: pixel 1 math -- (9*p[1] + 3*(p[0] + p[3]) + p[2]) / 16
|
||||
psllw mm2, 3 // EL3.1: 8*p[1] -> mm2
|
||||
paddw mm2, mm5 // EL3.2: 9*p[1] + 3*(p[0] + p[3]) + p[2] -> mm2
|
||||
psrlw mm2, 4 // EL3.3: sum[1]/16 -> mm5
|
||||
|
||||
// EL2/3: store pixels 0 & 1
|
||||
packuswb mm1, mm2 // EL2/3: pack into bytes
|
||||
MOVNTQ (edi, mm1) // EL2/3: store 2 pixels
|
||||
|
||||
// EL4: pixel 2 math -- (9*p[2] + 3*(p[0] + p[3]) + p[1]) / 16
|
||||
psllw mm3, 3 // EL4.1: 8*p[2] -> mm3
|
||||
paddw mm3, mm5 // EL4.2: 9*p[2] + 3*(p[0] + p[3]) + p[1] -> mm3
|
||||
psrlw mm3, 4 // EL4.3: sum[2]/16 -> mm3
|
||||
|
||||
// EL5: pixel 3 math -- (9*p[3] + 3*(p[1] + p[2]) + p[0]) / 16
|
||||
psllw mm4, 3 // EL5.1: 8*p[3] -> mm4
|
||||
paddw mm4, mm6 // EL5.2: 9*p[3] + 3*(p[1] + p[2]) + p[0] -> mm4
|
||||
psrlw mm4, 4 // EL5.3: sum[3]/16 -> mm4
|
||||
|
||||
// EL4/5: store pixels 2 & 3
|
||||
packuswb mm3, mm4 // EL4/5: pack into bytes
|
||||
MOVNTQ (edi + edx*4, mm3) // EL4/5: store 2 pixels
|
||||
}
|
||||
}
|
||||
// End MSVC_ASM
|
||||
|
||||
#elif defined(GCC_ASM)
|
||||
// GCC inline assembly versions
|
||||
|
||||
#if defined(USE_MOVNTQ)
|
||||
# define MOVNTQ(val, addr) "movntq " #val "," #addr
|
||||
#else
|
||||
# define MOVNTQ(val, addr) "movq " #val "," #addr
|
||||
#endif
|
||||
|
||||
#if USE_PREFETCH == INTEL_PREFETCH
|
||||
// using Intel SSE non-temporal prefetch
|
||||
# define PREFETCH(addr) "prefetchnta " #addr
|
||||
#elif USE_PREFETCH == AMD_PREFETCH
|
||||
// using AMD 3DNOW! prefetch
|
||||
# define PREFETCH(addr) "prefetch " #addr
|
||||
#else
|
||||
// no prefetch -- too bad for poor MMX-only souls
|
||||
# define PREFETCH(addr)
|
||||
#endif
|
||||
|
||||
static inline void
|
||||
SCALE_(PlatInit) (void)
|
||||
{
|
||||
__asm__ (
|
||||
// mm0 will be kept == 0 throughout
|
||||
// 0 is needed for bytes->words unpack instructions
|
||||
"pxor %%mm0, %%mm0 \n\t"
|
||||
|
||||
// mm5-mm7 contains RGB->YUV mults
|
||||
"movq %0, %%mm7 \n\t"
|
||||
"movq %1, %%mm6 \n\t"
|
||||
"movq %2, %%mm5 \n\t"
|
||||
#ifdef USE_YUV_LOOKUP
|
||||
// mm6 contains RGB888->555 shuffle mult
|
||||
"movq %3, %%mm6 \n\t"
|
||||
// mm5 contains DiffYUV threshold
|
||||
"movq %4, %%mm5 \n\t"
|
||||
#endif
|
||||
|
||||
: /* nothing */
|
||||
: /*0*/"m" (mmx_Y_mult), /*1*/"m" (mmx_U_mult), /*2*/"m" (mmx_V_mult)
|
||||
, /*3*/"m" (mmx_888to555_mult), /*4*/"m" (mmx_YUV_threshold)
|
||||
);
|
||||
}
|
||||
|
||||
static inline void
|
||||
SCALE_(PlatDone) (void)
|
||||
{
|
||||
// finish with MMX registers and yield them to FPU
|
||||
__asm__ (
|
||||
"emms \n\t"
|
||||
: /* nothing */ : /* nothing */
|
||||
);
|
||||
}
|
||||
|
||||
static inline void
|
||||
SCALE_(Prefetch) (const void* p)
|
||||
{
|
||||
__asm__ __volatile__ ("" PREFETCH (%0) : /*nothing*/ : "m" (p) );
|
||||
}
|
||||
|
||||
// compute the RGB distance squared between 2 pixels
|
||||
static inline int
|
||||
SCALE_(GetRGBDelta) (Uint32 pix1, Uint32 pix2)
|
||||
{
|
||||
int res;
|
||||
|
||||
__asm__ (
|
||||
// load pixels
|
||||
"movd %1, %%mm1 \n\t"
|
||||
"punpcklbw %%mm0, %%mm1 \n\t"
|
||||
"movd %2, %%mm2 \n\t"
|
||||
"punpcklbw %%mm0, %%mm2 \n\t"
|
||||
// get the difference between RGBA components
|
||||
"psubw %%mm2, %%mm1 \n\t"
|
||||
// squared and sumed
|
||||
"pmaddwd %%mm1, %%mm1 \n\t"
|
||||
// finish suming the squares
|
||||
"movq %%mm1, %%mm2 \n\t"
|
||||
"punpckhdq %%mm0, %%mm2 \n\t"
|
||||
"paddd %%mm2, %%mm1 \n\t"
|
||||
// store result
|
||||
"movd %%mm1, %0 \n\t"
|
||||
|
||||
: /*0*/"=r" (res)
|
||||
: /*1*/"rm" (pix1), /*2*/"rm" (pix2)
|
||||
);
|
||||
|
||||
return res;
|
||||
}
|
||||
|
||||
// retrieve the Y (intensity) component of pixel's YUV
|
||||
static inline int
|
||||
SCALE_(GetPixY) (Uint32 pix)
|
||||
{
|
||||
int ret;
|
||||
|
||||
__asm__ (
|
||||
// load pixel
|
||||
"movd %1, %%mm1 \n\t"
|
||||
"punpcklbw %%mm0, %%mm1 \n\t"
|
||||
// process
|
||||
"pmaddwd %2, %%mm1 \n\t" // R,G,B * Yvec
|
||||
"movq %%mm1, %%mm2 \n\t" // finish suming
|
||||
"punpckhdq %%mm0, %%mm2 \n\t" // ditto
|
||||
"paddd %%mm2, %%mm1 \n\t" // ditto
|
||||
// store index
|
||||
"movd %%mm1, %0 \n\t"
|
||||
|
||||
: /*0*/"=r" (ret)
|
||||
: /*1*/"rm" (pix), /*2*/"m" (mmx_Y_mult)
|
||||
);
|
||||
return ret >> 14;
|
||||
}
|
||||
|
||||
#ifdef USE_YUV_LOOKUP
|
||||
|
||||
// convert pixel RGB vector into YUV representation vector
|
||||
static inline YUV_VECTOR
|
||||
SCALE_(RGBtoYUV) (Uint32 pix)
|
||||
{
|
||||
int i;
|
||||
|
||||
__asm__ (
|
||||
// convert RGB888 to 555
|
||||
"movd %1, %%mm1 \n\t"
|
||||
"punpcklbw %%mm0, %%mm1 \n\t"
|
||||
"psrlw $3, %%mm1 \n\t" // 8->5 bit
|
||||
"pmaddwd %2, %%mm1 \n\t" // shuffle into the right channel order
|
||||
"movq %%mm1, %%mm2 \n\t" // finish shuffling
|
||||
"punpckhdq %%mm0, %%mm2 \n\t" // ditto
|
||||
"por %%mm2, %%mm1 \n\t" // ditto
|
||||
"movd %%mm1, %0 \n\t"
|
||||
|
||||
: /*0*/"=r" (i)
|
||||
: /*1*/"rm" (pix), /*2*/"m" (mmx_888to555_mult)
|
||||
);
|
||||
return RGB15_to_YUV[i];
|
||||
}
|
||||
|
||||
// compare 2 pixels with respect to their YUV representations
|
||||
// tolerance set by toler arg
|
||||
// returns true: close; false: distant (-gt toler)
|
||||
static inline bool
|
||||
SCALE_(CmpYUV) (Uint32 pix1, Uint32 pix2, int toler)
|
||||
{
|
||||
int delta;
|
||||
|
||||
__asm__ (
|
||||
"movd %1, %%mm1 \n\t"
|
||||
"movd %2, %%mm3 \n\t"
|
||||
|
||||
// convert RGB888 to 555
|
||||
// this is somewhat parallelized
|
||||
"punpcklbw %%mm0, %%mm1 \n\t"
|
||||
"psrlw $3, %%mm1 \n\t" // 8->5 bit
|
||||
"punpcklbw %%mm0, %%mm3 \n\t"
|
||||
"psrlw $3, %%mm3 \n\t" // 8->5 bit
|
||||
"pmaddwd %4, %%mm1 \n\t" // shuffle into the right channel order
|
||||
"movq %%mm1, %%mm2 \n\t" // finish shuffling
|
||||
"pmaddwd %4, %%mm3 \n\t" // shuffle into the right channel order
|
||||
"movq %%mm3, %%mm4 \n\t" // finish shuffling
|
||||
"punpckhdq %%mm0, %%mm2 \n\t" // ditto
|
||||
"por %%mm2, %%mm1 \n\t" // ditto
|
||||
"punpckhdq %%mm0, %%mm4 \n\t" // ditto
|
||||
"por %%mm4, %%mm3 \n\t" // ditto
|
||||
|
||||
// lookup the YUV vector
|
||||
"movd %%mm1, %%eax \n\t"
|
||||
"movd %%mm3, %%edx \n\t"
|
||||
"movd %3(,%%eax,4), %%mm1 \n\t"
|
||||
"movq %%mm1, %%mm4 \n\t"
|
||||
"movd %3(,%%edx,4), %%mm2 \n\t"
|
||||
|
||||
// get abs difference between YUV components
|
||||
#ifdef USE_PSADBW
|
||||
// we can use PSADBW and save us some grief
|
||||
"psadbw %%mm2, %%mm1 \n\t"
|
||||
"movd %%mm1, %0 \n\t"
|
||||
#else
|
||||
// no PSADBW -- have to do it the hard way
|
||||
"psubusb %%mm2, %%mm1 \n\t"
|
||||
"psubusb %%mm4, %%mm2 \n\t"
|
||||
"por %%mm2, %%mm1 \n\t"
|
||||
|
||||
// sum the differences
|
||||
// technically, this produces a MAX diff of 510
|
||||
// but we do not need anything bigger, currently
|
||||
"movq %%mm1, %%mm2 \n\t"
|
||||
"psrlq $8, %%mm2 \n\t"
|
||||
"paddusb %%mm2, %%mm1 \n\t"
|
||||
"psrlq $8, %%mm2 \n\t"
|
||||
"paddusb %%mm2, %%mm1 \n\t"
|
||||
// store intermediate delta
|
||||
"movd %%mm1, %0 \n\t"
|
||||
"andl $0xff, %0 \n\t"
|
||||
#endif /* USE_PSADBW */
|
||||
: /*0*/"=r" (delta)
|
||||
: /*1*/"rm" (pix1), /*2*/"rm" (pix2),
|
||||
/*3*/"m" (*RGB15_to_YUV), /*4*/"m" (mmx_888to555_mult)
|
||||
: "%eax", "%edx"
|
||||
);
|
||||
|
||||
return (delta << 1) <= toler;
|
||||
}
|
||||
|
||||
#endif /* USE_YUV_LOOKUP */
|
||||
|
||||
// Check if 2 pixels are different with respect to their
|
||||
// YUV representations
|
||||
// returns 0: close; ~0: distant
|
||||
static inline int
|
||||
SCALE_(DiffYUV) (Uint32 yuv1, Uint32 yuv2)
|
||||
{
|
||||
sint32 ret;
|
||||
|
||||
__asm__ (
|
||||
// load YUV pixels
|
||||
"movd %1, %%mm1 \n\t"
|
||||
"movq %%mm1, %%mm4 \n\t"
|
||||
"movd %2, %%mm2 \n\t"
|
||||
// abs difference between channels
|
||||
"psubusb %%mm2, %%mm1 \n\t"
|
||||
"psubusb %%mm4, %%mm2 \n\t"
|
||||
"por %%mm2, %%mm1 \n\t"
|
||||
// compare to threshold
|
||||
"psubusb %3, %%mm1 \n\t"
|
||||
|
||||
"movd %%mm1, %%edx \n\t"
|
||||
// transform eax to 0 or ~0
|
||||
"xor %%eax, %%eax \n\t"
|
||||
"or %%edx, %%edx \n\t"
|
||||
"setz %%al \n\t"
|
||||
"dec %%eax \n\t"
|
||||
|
||||
: /*0*/"=a" (ret)
|
||||
: /*1*/"rm" (yuv1), /*2*/"rm" (yuv2),
|
||||
/*3*/"m" (mmx_YUV_threshold)
|
||||
: "%edx"
|
||||
);
|
||||
return ret;
|
||||
}
|
||||
|
||||
// Bilinear weighted blend of four pixels
|
||||
// Function produces 4 blended pixels (in 2x2 matrix) and writes them
|
||||
// out to the surface
|
||||
// Last version
|
||||
static inline void
|
||||
SCALE_(Blend_bilinear) (const Uint32* row0, const Uint32* row1,
|
||||
Uint32* dst_p, Uint32 dlen)
|
||||
{
|
||||
__asm__ (
|
||||
// EL0: load pixels
|
||||
"movq %0, %%mm1 \n\t" // EL0
|
||||
"movq %%mm1, %%mm2 \n\t" // EL0: p[1] -> mm2
|
||||
PREFETCH (0x80%0) "\n\t"
|
||||
"punpckhbw %%mm0, %%mm2 \n\t" // EL0: p[1] -> mm2
|
||||
"punpcklbw %%mm0, %%mm1 \n\t" // EL0: p[0] -> mm1
|
||||
"movq %1, %%mm3 \n\t"
|
||||
"movq %%mm3, %%mm4 \n\t" // EL0: p[3] -> mm4
|
||||
"movq %%mm2, %%mm6 \n\t" // EL1.1: p[1] -> mm6
|
||||
PREFETCH (0x80%1) "\n\t"
|
||||
"punpcklbw %%mm0, %%mm3 \n\t" // EL0: p[2] -> mm3
|
||||
"movq %%mm1, %%mm5 \n\t" // EL1.1: p[0] -> mm5
|
||||
"punpckhbw %%mm0, %%mm4 \n\t" // EL0: p[3] -> mm4
|
||||
|
||||
// EL1: cache p[0] + 3*(p[1] + p[2]) + p[3] in mm6
|
||||
"paddw %%mm3, %%mm6 \n\t" // EL1.2: p[1] + p[2] -> mm6
|
||||
// EL1: cache p[0] + p[1] + p[2] + p[3] in mm7
|
||||
"movq %%mm6, %%mm7 \n\t" // EL1.3: p[1] + p[2] -> mm7
|
||||
// EL1: cache p[1] + 3*(p[0] + p[3]) + p[2] in mm5
|
||||
"paddw %%mm4, %%mm5 \n\t" // EL1.2: p[0] + p[3] -> mm5
|
||||
"psllw $1, %%mm6 \n\t" // EL1.4: 2*(p[1] + p[2]) -> mm6
|
||||
"paddw %%mm5, %%mm7 \n\t" // EL1.4: sum(p[]) -> mm7
|
||||
"psllw $1, %%mm5 \n\t" // EL1.5: 2*(p[0] + p[3]) -> mm5
|
||||
"paddw %%mm7, %%mm6 \n\t" // EL1.5: p[0] + 3*(p[1] + p[2]) + p[3] -> mm6
|
||||
"paddw %%mm7, %%mm5 \n\t" // EL1.6: p[1] + 3*(p[0] + p[3]) + p[2] -> mm5
|
||||
|
||||
// EL2: pixel 0 math -- (9*p[0] + 3*(p[1] + p[2]) + p[3]) / 16
|
||||
"psllw $3, %%mm1 \n\t" // EL2.1: 8*p[0] -> mm1
|
||||
"paddw %%mm6, %%mm1 \n\t" // EL2.2: 9*p[0] + 3*(p[1] + p[2]) + p[3] -> mm1
|
||||
"psrlw $4, %%mm1 \n\t" // EL2.3: sum[0]/16 -> mm1
|
||||
|
||||
// EL3: pixel 1 math -- (9*p[1] + 3*(p[0] + p[3]) + p[2]) / 16
|
||||
"psllw $3, %%mm2 \n\t" // EL3.1: 8*p[1] -> mm2
|
||||
"paddw %%mm5, %%mm2 \n\t" // EL3.2: 9*p[1] + 3*(p[0] + p[3]) + p[2] -> mm5
|
||||
"psrlw $4, %%mm2 \n\t" // EL3.3: sum[1]/16 -> mm5
|
||||
|
||||
// EL2/4: store pixels 0 & 1
|
||||
"packuswb %%mm2, %%mm1 \n\t" // EL2/4: pack into bytes
|
||||
MOVNTQ (%%mm1, (%2)) "\n\t" // EL2/4: store 2 pixels
|
||||
|
||||
// EL4: pixel 2 math -- (9*p[2] + 3*(p[0] + p[3]) + p[1]) / 16
|
||||
"psllw $3, %%mm3 \n\t" // EL4.1: 8*p[2] -> mm3
|
||||
"paddw %%mm5, %%mm3 \n\t" // EL4.2: 9*p[2] + 3*(p[0] + p[3]) + p[1] -> mm3
|
||||
"psrlw $4, %%mm3 \n\t" // EL4.3: sum[2]/16 -> mm3
|
||||
|
||||
// EL5: pixel 3 math -- (9*p[3] + 3*(p[1] + p[2]) + p[0]) / 16
|
||||
"psllw $3, %%mm4 \n\t" // EL5.1: 8*p[3] -> mm4
|
||||
"paddw %%mm6, %%mm4 \n\t" // EL5.2: 9*p[3] + 3*(p[1] + p[2]) + p[0] -> mm4
|
||||
"psrlw $4, %%mm4 \n\t" // EL5.3: sum[3]/16 -> mm4
|
||||
|
||||
// EL4/5: store pixels 2 & 3
|
||||
"packuswb %%mm4, %%mm3 \n\t" // EL4/5: pack into bytes
|
||||
MOVNTQ (%%mm3, (%2,%3,4)) "\n\t" // EL4/5: store 2 pixels
|
||||
|
||||
: /* nothing */
|
||||
: /*0*/"m" (*row0), /*1*/"m" (*row1), /*2*/"r" (dst_p), /*3*/"r" (dlen)
|
||||
: "memory"
|
||||
);
|
||||
}
|
||||
|
||||
#endif // GCC_ASM
|
||||
|
||||
#endif /* SCALEMMX_H_ */
|
||||
Executable
+286
@@ -0,0 +1,286 @@
|
||||
/*
|
||||
* This program is free software; you can redistribute it and/or modify
|
||||
* it under the terms of the GNU General Public License as published by
|
||||
* the Free Software Foundation; either version 2 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU General Public License
|
||||
* along with this program; if not, write to the Free Software
|
||||
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||
*/
|
||||
|
||||
#ifdef GFXMODULE_SDL
|
||||
|
||||
#include "types.h"
|
||||
#include "libs/graphics/sdl/sdl_common.h"
|
||||
#include SDL_INCLUDE(sdl_cpuinfo.h)
|
||||
#include "libs/platform.h"
|
||||
#include "scalers.h"
|
||||
#include "scaleint.h"
|
||||
#include "2xscalers.h"
|
||||
#ifdef MMX_ASM
|
||||
# include "2xscalers_mmx.h"
|
||||
#endif
|
||||
|
||||
|
||||
typedef enum
|
||||
{
|
||||
SCALEPLAT_NULL = PLATFORM_NULL,
|
||||
SCALEPLAT_C = PLATFORM_C,
|
||||
SCALEPLAT_MMX = PLATFORM_MMX,
|
||||
SCALEPLAT_SSE = PLATFORM_SSE,
|
||||
SCALEPLAT_3DNOW = PLATFORM_3DNOW,
|
||||
SCALEPLAT_ALTIVEC = PLATFORM_ALTIVEC,
|
||||
|
||||
SCALEPLAT_C_RGBA,
|
||||
SCALEPLAT_C_BGRA,
|
||||
SCALEPLAT_C_ARGB,
|
||||
SCALEPLAT_C_ABGR,
|
||||
|
||||
} Scale_PlatType_t;
|
||||
|
||||
|
||||
// RGB -> YUV transformation
|
||||
// the RGB vector is multiplied by the transformation matrix
|
||||
// to get the YUV vector
|
||||
#if 0
|
||||
// original table -- not used
|
||||
const int YUV_matrix[3][3] =
|
||||
{
|
||||
/* Y U V */
|
||||
/* R */ {0.2989, -0.1687, 0.5000},
|
||||
/* G */ {0.5867, -0.3312, -0.4183},
|
||||
/* B */ {0.1144, 0.5000, -0.0816}
|
||||
};
|
||||
#else
|
||||
// scaled up by a 2^14 factor, with Y doubled
|
||||
const int YUV_matrix[3][3] =
|
||||
{
|
||||
/* Y U V */
|
||||
/* R */ { 9794, -2764, 8192},
|
||||
/* G */ {19224, -5428, -6853},
|
||||
/* B */ { 3749, 8192, -1339}
|
||||
};
|
||||
#endif
|
||||
|
||||
// pre-computed transformations for 8 bits per channel
|
||||
int RGB_to_YUV[/*RGB*/ 3][/*YUV*/ 3][ /*mult-res*/ 256];
|
||||
sint16 dRGB_to_dYUV[/*RGB*/ 3][/*YUV*/ 3][ /*mult-res*/ 512];
|
||||
|
||||
// pre-computed transformations for RGB555
|
||||
YUV_VECTOR RGB15_to_YUV[0x8000];
|
||||
|
||||
PLATFORM_TYPE force_platform = PLATFORM_NULL;
|
||||
Scale_PlatType_t Scale_Platform = SCALEPLAT_NULL;
|
||||
|
||||
|
||||
// pre-compute the RGB->YUV transformations
|
||||
void
|
||||
Scale_Init (void)
|
||||
{
|
||||
int i1, i2, i3;
|
||||
|
||||
for (i1 = 0; i1 < 3; i1++) // enum R,G,B
|
||||
for (i2 = 0; i2 < 3; i2++) // enum Y,U,V
|
||||
for (i3 = 0; i3 < 256; i3++) // enum possible channel vals
|
||||
{
|
||||
RGB_to_YUV[i1][i2][i3] =
|
||||
(YUV_matrix[i1][i2] * i3) >> 14;
|
||||
}
|
||||
|
||||
for (i1 = 0; i1 < 3; i1++) // enum R,G,B
|
||||
for (i2 = 0; i2 < 3; i2++) // enum Y,U,V
|
||||
for (i3 = -255; i3 < 256; i3++) // enum possible channel delta vals
|
||||
{
|
||||
dRGB_to_dYUV[i1][i2][i3 + 255] =
|
||||
(YUV_matrix[i1][i2] * i3) >> 14;
|
||||
}
|
||||
|
||||
for (i1 = 0; i1 < 32; ++i1)
|
||||
for (i2 = 0; i2 < 32; ++i2)
|
||||
for (i3 = 0; i3 < 32; ++i3)
|
||||
{
|
||||
int y, u, v;
|
||||
// adding upper bits halved for error correction
|
||||
int r = (i1 << 3) | (i1 >> 3);
|
||||
int g = (i2 << 3) | (i2 >> 3);
|
||||
int b = (i3 << 3) | (i3 >> 3);
|
||||
|
||||
y = ( r * YUV_matrix[YUV_XFORM_R][YUV_XFORM_Y]
|
||||
+ g * YUV_matrix[YUV_XFORM_G][YUV_XFORM_Y]
|
||||
+ b * YUV_matrix[YUV_XFORM_B][YUV_XFORM_Y]
|
||||
) >> 15; // we dont need Y doubled, need Y to fit 8 bits
|
||||
|
||||
// U and V are half the importance of Y
|
||||
u = 64+(( r * YUV_matrix[YUV_XFORM_R][YUV_XFORM_U]
|
||||
+ g * YUV_matrix[YUV_XFORM_G][YUV_XFORM_U]
|
||||
+ b * YUV_matrix[YUV_XFORM_B][YUV_XFORM_U]
|
||||
) >> 15); // halved
|
||||
|
||||
v = 64+(( r * YUV_matrix[YUV_XFORM_R][YUV_XFORM_V]
|
||||
+ g * YUV_matrix[YUV_XFORM_G][YUV_XFORM_V]
|
||||
+ b * YUV_matrix[YUV_XFORM_B][YUV_XFORM_V]
|
||||
) >> 15); // halved
|
||||
|
||||
RGB15_to_YUV[(i1 << 10) | (i2 << 5) | i3] = (y << 16) | (u << 8) | v;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// expands the given rectangle in all directions by 'expansion'
|
||||
// guarded by 'limits'
|
||||
void
|
||||
Scale_ExpandRect (SDL_Rect* rect, int expansion, const SDL_Rect* limits)
|
||||
{
|
||||
if (rect->x - expansion >= limits->x)
|
||||
{
|
||||
rect->w += expansion;
|
||||
rect->x -= expansion;
|
||||
}
|
||||
else
|
||||
{
|
||||
rect->w += rect->x - limits->x;
|
||||
rect->x = limits->x;
|
||||
}
|
||||
|
||||
if (rect->y - expansion >= limits->y)
|
||||
{
|
||||
rect->h += expansion;
|
||||
rect->y -= expansion;
|
||||
}
|
||||
else
|
||||
{
|
||||
rect->h += rect->y - limits->y;
|
||||
rect->y = limits->y;
|
||||
}
|
||||
|
||||
if (rect->x + rect->w + expansion <= limits->w)
|
||||
rect->w += expansion;
|
||||
else
|
||||
rect->w = limits->w - rect->x;
|
||||
|
||||
if (rect->y + rect->h + expansion <= limits->h)
|
||||
rect->h += expansion;
|
||||
else
|
||||
rect->h = limits->h - rect->y;
|
||||
}
|
||||
|
||||
|
||||
// Platform+Scaler function lookups
|
||||
|
||||
typedef struct
|
||||
{
|
||||
Scale_PlatType_t platform;
|
||||
const Scale_FuncDef_t* funcdefs;
|
||||
} Scale_PlatDef_t;
|
||||
|
||||
|
||||
const static Scale_PlatDef_t
|
||||
Scale_PlatDefs[] =
|
||||
{
|
||||
#if defined(MMX_ASM)
|
||||
{SCALEPLAT_SSE, Scale_SSE_Functions},
|
||||
{SCALEPLAT_3DNOW, Scale_3DNow_Functions},
|
||||
{SCALEPLAT_MMX, Scale_MMX_Functions},
|
||||
#endif /* MMX_ASM */
|
||||
// Default
|
||||
{SCALEPLAT_NULL, Scale_C_Functions}
|
||||
};
|
||||
|
||||
|
||||
TFB_ScaleFunc
|
||||
Scale_PrepPlatform (int flags, const SDL_PixelFormat* fmt)
|
||||
{
|
||||
const Scale_PlatDef_t* pdef;
|
||||
const Scale_FuncDef_t* fdef;
|
||||
|
||||
(void)flags;
|
||||
|
||||
Scale_Platform = SCALEPLAT_NULL;
|
||||
|
||||
// XXX: Hack to test some code
|
||||
Scale_MMX_PrepPlatform (fmt);
|
||||
|
||||
// first match wins
|
||||
// add better platform techs to the top
|
||||
#ifdef MMX_ASM
|
||||
if ( (!force_platform && (SDL_HasSSE () || SDL_HasMMXExt ()))
|
||||
|| force_platform == SCALEPLAT_SSE)
|
||||
{
|
||||
fprintf (stderr, "Screen scalers are using SSE/MMX-Ext/MMX code\n");
|
||||
Scale_Platform = SCALEPLAT_SSE;
|
||||
|
||||
Scale_SSE_PrepPlatform (fmt);
|
||||
}
|
||||
else
|
||||
if ( (!force_platform && SDL_HasAltiVec ())
|
||||
|| force_platform == SCALEPLAT_ALTIVEC)
|
||||
{
|
||||
fprintf (stderr, "Screen scalers would use AltiVec code "
|
||||
"if someone actually wrote it\n");
|
||||
//Scale_Platform = SCALEPLAT_ALTIVEC;
|
||||
}
|
||||
else
|
||||
if ( (!force_platform && SDL_Has3DNow ())
|
||||
|| force_platform == SCALEPLAT_3DNOW)
|
||||
{
|
||||
fprintf (stderr, "Screen scalers are using 3DNow/MMX code\n");
|
||||
Scale_Platform = SCALEPLAT_3DNOW;
|
||||
|
||||
Scale_3DNow_PrepPlatform (fmt);
|
||||
}
|
||||
else
|
||||
if ( (!force_platform && SDL_HasMMX ())
|
||||
|| force_platform == SCALEPLAT_MMX)
|
||||
{
|
||||
fprintf (stderr, "Screen scalers are using MMX code\n");
|
||||
Scale_Platform = SCALEPLAT_MMX;
|
||||
|
||||
Scale_MMX_PrepPlatform (fmt);
|
||||
}
|
||||
#endif
|
||||
|
||||
if (Scale_Platform == SCALEPLAT_NULL)
|
||||
{ // Plain C versions
|
||||
if (fmt->Rmask == 0xff000000)
|
||||
Scale_Platform = SCALEPLAT_C_RGBA;
|
||||
else if (fmt->Rmask == 0x00ff0000)
|
||||
Scale_Platform = SCALEPLAT_C_ARGB;
|
||||
else if (fmt->Rmask == 0x0000ff00)
|
||||
Scale_Platform = SCALEPLAT_C_BGRA;
|
||||
else if (fmt->Rmask == 0x000000ff)
|
||||
Scale_Platform = SCALEPLAT_C_ABGR;
|
||||
else
|
||||
{ // use slowest default
|
||||
fprintf (stderr, "Scale_PrepPlatform(): "
|
||||
"unknown Red mask (0x%08x)\n", fmt->Rmask);
|
||||
Scale_Platform = SCALEPLAT_C;
|
||||
}
|
||||
|
||||
if (Scale_Platform == SCALEPLAT_C)
|
||||
fprintf (stderr, "Screen scalers are using slow generic C code\n");
|
||||
else
|
||||
fprintf (stderr, "Screen scalers are using optimized C code\n");
|
||||
}
|
||||
|
||||
// Lookup the scaling function
|
||||
// First find the right platform
|
||||
for (pdef = Scale_PlatDefs;
|
||||
pdef->platform != Scale_Platform && pdef->platform != SCALEPLAT_NULL;
|
||||
++pdef)
|
||||
;
|
||||
// Next find the right function
|
||||
for (fdef = pdef->funcdefs;
|
||||
(flags & fdef->flag) != fdef->flag;
|
||||
++fdef)
|
||||
;
|
||||
|
||||
return fdef->func;
|
||||
}
|
||||
|
||||
#endif
|
||||
Executable
+27
@@ -0,0 +1,27 @@
|
||||
/*
|
||||
* This program is free software; you can redistribute it and/or modify
|
||||
* it under the terms of the GNU General Public License as published by
|
||||
* the Free Software Foundation; either version 2 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU General Public License
|
||||
* along with this program; if not, write to the Free Software
|
||||
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||
*/
|
||||
|
||||
#ifndef SCALERS_H_
|
||||
#define SCALERS_H_
|
||||
|
||||
void Scale_Init (void);
|
||||
|
||||
typedef void (* TFB_ScaleFunc) (SDL_Surface *src, SDL_Surface *dst,
|
||||
SDL_Rect *r);
|
||||
|
||||
TFB_ScaleFunc Scale_PrepPlatform (int flags, const SDL_PixelFormat* fmt);
|
||||
|
||||
#endif /* SCALERS_H_ */
|
||||
+158
@@ -0,0 +1,158 @@
|
||||
/*
|
||||
* This program is free software; you can redistribute it and/or modify
|
||||
* it under the terms of the GNU General Public License as published by
|
||||
* the Free Software Foundation; either version 2 of the License, or
|
||||
* (at your option) any later version.
|
||||
*
|
||||
* This program is distributed in the hope that it will be useful,
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
* GNU General Public License for more details.
|
||||
*
|
||||
* You should have received a copy of the GNU General Public License
|
||||
* along with this program; if not, write to the Free Software
|
||||
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||
*/
|
||||
|
||||
// Core algorithm of the Triscan screen scaler (based on Scale2x)
|
||||
// (for scale2x please see http://scale2x.sf.net)
|
||||
// Template
|
||||
// When this file is built standalone is produces a plain C version
|
||||
// Also #included by 2xscalers_mmx.c for an MMX version
|
||||
|
||||
#ifdef GFXMODULE_SDL
|
||||
|
||||
#include "libs/graphics/sdl/sdl_common.h"
|
||||
#include "types.h"
|
||||
#include "scalers.h"
|
||||
#include "scaleint.h"
|
||||
#include "2xscalers.h"
|
||||
|
||||
|
||||
// Triscan scaling to 2x
|
||||
// derivative of scale2x -- scale2x.sf.net
|
||||
// The name expands to either
|
||||
// Scale_TriScanFilter (for plain C) or
|
||||
// Scale_MMX_TriScanFilter (for MMX)
|
||||
// [others when platforms are added]
|
||||
void
|
||||
SCALE_(TriScanFilter) (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r)
|
||||
{
|
||||
int x, y;
|
||||
const int w = src->w, h = src->h;
|
||||
int xend, yend;
|
||||
int dsrc, ddst;
|
||||
SDL_Rect *region = r;
|
||||
SDL_Rect limits;
|
||||
SDL_PixelFormat *fmt = dst->format;
|
||||
const int sp = src->pitch, dp = dst->pitch;
|
||||
const int bpp = fmt->BytesPerPixel;
|
||||
const int slen = sp / bpp, dlen = dp / bpp;
|
||||
// for clarity purposes, the 'pixels' array here is transposed
|
||||
Uint32 pixels[3][3];
|
||||
Uint32 *src_p = (Uint32 *)src->pixels;
|
||||
Uint32 *dst_p = (Uint32 *)dst->pixels;
|
||||
|
||||
int prevline, nextline;
|
||||
|
||||
// these macros are for clarity; they make the current pixel (0,0)
|
||||
// and allow to access pixels in all directions
|
||||
#define PIX(x, y) (pixels[1 + (x)][1 + (y)])
|
||||
|
||||
#define TRISCAN_YUV_MED 100
|
||||
// medium tolerance pixel comparison
|
||||
#define TRISCAN_CMPYUV(p1, p2) \
|
||||
(PIX p1 == PIX p2 || SCALE_CMPYUV (PIX p1, PIX p2, TRISCAN_YUV_MED))
|
||||
|
||||
|
||||
SCALE_(PlatInit) ();
|
||||
|
||||
// expand updated region if necessary
|
||||
// pixels neighbooring the updated region may
|
||||
// change as a result of updates
|
||||
limits.x = 0;
|
||||
limits.y = 0;
|
||||
limits.w = src->w;
|
||||
limits.h = src->h;
|
||||
Scale_ExpandRect (region, 1, &limits);
|
||||
|
||||
xend = region->x + region->w;
|
||||
yend = region->y + region->h;
|
||||
dsrc = slen - region->w;
|
||||
ddst = (dlen - region->w) * 2;
|
||||
|
||||
// move ptrs to the first updated pixel
|
||||
src_p += slen * region->y + region->x;
|
||||
dst_p += (dlen * region->y + region->x) * 2;
|
||||
|
||||
for (y = region->y; y < yend; ++y, dst_p += ddst, src_p += dsrc)
|
||||
{
|
||||
if (y > 0)
|
||||
prevline = -slen;
|
||||
else
|
||||
prevline = 0;
|
||||
|
||||
if (y < h - 1)
|
||||
nextline = slen;
|
||||
else
|
||||
nextline = 0;
|
||||
|
||||
// prime the (tiny) sliding-window pixel arrays
|
||||
PIX( 1, 0) = src_p[0];
|
||||
|
||||
if (region->x > 0)
|
||||
PIX( 0, 0) = src_p[-1];
|
||||
else
|
||||
PIX( 0, 0) = PIX( 1, 0);
|
||||
|
||||
for (x = region->x; x < xend; ++x, ++src_p, dst_p += 2)
|
||||
{
|
||||
// slide the window
|
||||
PIX(-1, 0) = PIX( 0, 0);
|
||||
|
||||
PIX( 0, -1) = src_p[prevline];
|
||||
PIX( 0, 0) = PIX( 1, 0);
|
||||
PIX( 0, 1) = src_p[nextline];
|
||||
|
||||
if (x < w - 1)
|
||||
PIX( 1, 0) = src_p[1];
|
||||
else
|
||||
PIX( 1, 0) = PIX( 0, 0);
|
||||
|
||||
if (!TRISCAN_CMPYUV (( 0, -1), ( 0, 1)) &&
|
||||
!TRISCAN_CMPYUV ((-1, 0), ( 1, 0)))
|
||||
{
|
||||
if (TRISCAN_CMPYUV ((-1, 0), ( 0, -1)))
|
||||
dst_p[0] = Scale_Blend_11 (PIX(-1, 0), PIX(0, -1));
|
||||
else
|
||||
dst_p[0] = PIX(0, 0);
|
||||
|
||||
if (TRISCAN_CMPYUV (( 1, 0), ( 0, -1)))
|
||||
dst_p[1] = Scale_Blend_11 (PIX(1, 0), PIX(0, -1));
|
||||
else
|
||||
dst_p[1] = PIX(0, 0);
|
||||
|
||||
if (TRISCAN_CMPYUV ((-1, 0), ( 0, 1)))
|
||||
dst_p[dlen] = Scale_Blend_11 (PIX(-1, 0), PIX(0, 1));
|
||||
else
|
||||
dst_p[dlen] = PIX(0, 0);
|
||||
|
||||
if (TRISCAN_CMPYUV (( 1, 0), ( 0, 1)))
|
||||
dst_p[dlen+1] = Scale_Blend_11 (PIX(1, 0), PIX(0, 1));
|
||||
else
|
||||
dst_p[dlen+1] = PIX(0, 0);
|
||||
}
|
||||
else
|
||||
{
|
||||
dst_p[0] = PIX(0, 0);
|
||||
dst_p[1] = PIX(0, 0);
|
||||
dst_p[dlen] = PIX(0, 0);
|
||||
dst_p[dlen+1] = PIX(0, 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
SCALE_(PlatDone) ();
|
||||
}
|
||||
|
||||
#endif /* GFXMODULE_SDL */
|
||||
Reference in New Issue
Block a user