Accelerated 32bpp scalers, Preview 1

git-svn-id: svn://svn.code.sf.net/p/sc2/code/trunk@1943 8092fc87-c524-0410-9efc-e669fe64eaf9
This commit is contained in:
avolkov
2005-11-03 01:40:38 +00:00
parent ab348cc7f2
commit 75f9492a91
13 changed files with 2973 additions and 0 deletions
+104
View File
@@ -0,0 +1,104 @@
/*
* This program is free software; you can redistribute it and/or modify
* it under the terms of the GNU General Public License as published by
* the Free Software Foundation; either version 2 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU General Public License for more details.
*
* You should have received a copy of the GNU General Public License
* along with this program; if not, write to the Free Software
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
*/
#ifdef GFXMODULE_SDL
#include "port.h"
#include "libs/platform.h"
#if defined(MMX_ASM)
#include "libs/graphics/sdl/sdl_common.h"
#include "types.h"
#include "scalers.h"
#include "scaleint.h"
#include "2xscalers.h"
#include "2xscalers_mmx.h"
// 3DNow! name for all functions
#undef SCALE_
#define SCALE_(name) Scale ## _3DNow_ ## name
// Tell them which opcodes we want to support
#undef USE_MOVNTQ
#define USE_PREFETCH AMD_PREFETCH
#undef USE_PSADBW
// Bring in inline asm functions
#include "scalemmx.h"
// Scaler function lookup table
//
const Scale_FuncDef_t
Scale_3DNow_Functions[] =
{
{TFB_GFXFLAGS_SCALE_BILINEAR, Scale_3DNow_BilinearFilter},
{TFB_GFXFLAGS_SCALE_BIADAPT, Scale_BiAdaptFilter},
{TFB_GFXFLAGS_SCALE_BIADAPTADV, Scale_MMX_BiAdaptAdvFilter},
{TFB_GFXFLAGS_SCALE_TRISCAN, Scale_MMX_TriScanFilter},
{TFB_GFXFLAGS_SCALE_HQXX, Scale_MMX_HqFilter},
// Default
{0, Scale_3DNow_Nearest}
};
void
Scale_3DNow_PrepPlatform (const SDL_PixelFormat* fmt)
{
Scale_MMX_PrepPlatform (fmt);
}
// Nearest Neighbor scaling to 2x
// void Scale_3DNow_Nearest (SDL_Surface *src,
// SDL_Surface *dst, SDL_Rect *r)
#include "nearest2x.c"
// Bilinear scaling to 2x
// void Scale_3DNow_BilinearFilter (SDL_Surface *src,
// SDL_Surface *dst, SDL_Rect *r)
#include "bilinear2x.c"
#if 0 && NO_IMPROVEMENT
// Advanced Biadapt scaling to 2x
// void Scale_3DNow_BiAdaptAdvFilter (SDL_Surface *src,
// SDL_Surface *dst, SDL_Rect *r)
#include "biadv2x.c"
// Triscan scaling to 2x
// derivative of scale2x -- scale2x.sf.net
// void Scale_3DNow_TriScanFilter (SDL_Surface *src,
// SDL_Surface *dst, SDL_Rect *r)
#include "triscan2x.c"
// Hq2x scaling
// (adapted from 'hq2x' by Maxim Stepin -- www.hiend3d.com/hq2x.html)
// void Scale_3DNow_HqFilter (SDL_Surface *src,
// SDL_Surface *dst, SDL_Rect *r)
#include "hq2x.c"
#endif /* NO_IMPROVEMENT */
#endif /* MMX_ASM */
#endif /* GFXMODULE_SDL */
+138
View File
@@ -0,0 +1,138 @@
/*
* This program is free software; you can redistribute it and/or modify
* it under the terms of the GNU General Public License as published by
* the Free Software Foundation; either version 2 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU General Public License for more details.
*
* You should have received a copy of the GNU General Public License
* along with this program; if not, write to the Free Software
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
*/
#ifdef GFXMODULE_SDL
#include "port.h"
#include "libs/platform.h"
#if defined(MMX_ASM)
#include "libs/graphics/sdl/sdl_common.h"
#include "types.h"
#include "scalers.h"
#include "scaleint.h"
#include "2xscalers.h"
#include "2xscalers_mmx.h"
// MMX name for all functions
#undef SCALE_
#define SCALE_(name) Scale ## _MMX_ ## name
// Tell them which opcodes we want to support
#undef USE_MOVNTQ
#undef USE_PREFETCH
#undef USE_PSADBW
// And Bring in inline asm functions
#include "scalemmx.h"
// Scaler function lookup table
//
const Scale_FuncDef_t
Scale_MMX_Functions[] =
{
{TFB_GFXFLAGS_SCALE_BILINEAR, Scale_MMX_BilinearFilter},
{TFB_GFXFLAGS_SCALE_BIADAPT, Scale_BiAdaptFilter},
{TFB_GFXFLAGS_SCALE_BIADAPTADV, Scale_MMX_BiAdaptAdvFilter},
{TFB_GFXFLAGS_SCALE_TRISCAN, Scale_MMX_TriScanFilter},
{TFB_GFXFLAGS_SCALE_HQXX, Scale_MMX_HqFilter},
// Default
{0, Scale_MMX_Nearest}
};
// MMX transformation multipliers
Uint64 mmx_888to555_mult;
Uint64 mmx_Y_mult;
Uint64 mmx_U_mult;
Uint64 mmx_V_mult;
// Uint64 mmx_YUV_threshold = 0x00300706; original hq2x threshold
//Uint64 mmx_YUV_threshold = 0x0030100e;
Uint64 mmx_YUV_threshold = 0x0040120c;
void
Scale_MMX_PrepPlatform (const SDL_PixelFormat* fmt)
{
// prepare the channel-shuffle multiplier
mmx_888to555_mult = ((Uint64)0x0400) << (fmt->Rshift * 2)
| ((Uint64)0x0020) << (fmt->Gshift * 2)
| ((Uint64)0x0001) << (fmt->Bshift * 2);
// prepare the RGB->YUV multipliers
mmx_Y_mult = ((Uint64)(uint16)YUV_matrix[YUV_XFORM_R][YUV_XFORM_Y])
<< (fmt->Rshift * 2)
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_G][YUV_XFORM_Y])
<< (fmt->Gshift * 2)
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_B][YUV_XFORM_Y])
<< (fmt->Bshift * 2);
mmx_U_mult = ((Uint64)(uint16)YUV_matrix[YUV_XFORM_R][YUV_XFORM_U])
<< (fmt->Rshift * 2)
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_G][YUV_XFORM_U])
<< (fmt->Gshift * 2)
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_B][YUV_XFORM_U])
<< (fmt->Bshift * 2);
mmx_V_mult = ((Uint64)(uint16)YUV_matrix[YUV_XFORM_R][YUV_XFORM_V])
<< (fmt->Rshift * 2)
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_G][YUV_XFORM_V])
<< (fmt->Gshift * 2)
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_B][YUV_XFORM_V])
<< (fmt->Bshift * 2);
mmx_YUV_threshold = (SCALE_DIFFYUV_TY << 16) | (SCALE_DIFFYUV_TU << 8)
| SCALE_DIFFYUV_TV;
}
// Nearest Neighbor scaling to 2x
// void Scale_MMX_Nearest (SDL_Surface *src,
// SDL_Surface *dst, SDL_Rect *r)
#include "nearest2x.c"
// Bilinear scaling to 2x
// void Scale_MMX_BilinearFilter (SDL_Surface *src,
// SDL_Surface *dst, SDL_Rect *r)
#include "bilinear2x.c"
// Advanced Biadapt scaling to 2x
// void Scale_MMX_BiAdaptAdvFilter (SDL_Surface *src,
// SDL_Surface *dst, SDL_Rect *r)
#include "biadv2x.c"
// Triscan scaling to 2x
// derivative of 'scale2x' -- scale2x.sf.net
// void Scale_MMX_TriScanFilter (SDL_Surface *src,
// SDL_Surface *dst, SDL_Rect *r)
#include "triscan2x.c"
// Hq2x scaling
// (adapted from 'hq2x' by Maxim Stepin -- www.hiend3d.com/hq2x.html)
// void Scale_MMX_HqFilter (SDL_Surface *src,
// SDL_Surface *dst, SDL_Rect *r)
#include "hq2x.c"
#endif /* MMX_ASM */
#endif /* GFXMODULE_SDL */
+56
View File
@@ -0,0 +1,56 @@
/*
* This program is free software; you can redistribute it and/or modify
* it under the terms of the GNU General Public License as published by
* the Free Software Foundation; either version 2 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU General Public License for more details.
*
* You should have received a copy of the GNU General Public License
* along with this program; if not, write to the Free Software
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
*/
#ifndef _2XSCALERS_MMX_H_
#define _2XSCALERS_MMX_H_
// MMX versions
void Scale_MMX_PrepPlatform (const SDL_PixelFormat* fmt);
void Scale_MMX_Nearest (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
void Scale_MMX_BilinearFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
void Scale_MMX_BiAdaptAdvFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
void Scale_MMX_TriScanFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
void Scale_MMX_HqFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
extern const Scale_FuncDef_t Scale_MMX_Functions[];
// SSE (Intel)/MMX Ext (Athlon) versions
void Scale_SSE_PrepPlatform (const SDL_PixelFormat* fmt);
void Scale_SSE_Nearest (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
void Scale_SSE_BilinearFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
void Scale_SSE_BiAdaptAdvFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
void Scale_SSE_TriScanFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
void Scale_SSE_HqFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
extern const Scale_FuncDef_t Scale_SSE_Functions[];
// 3DNow (AMD K6/Athlon) versions
void Scale_3DNow_PrepPlatform (const SDL_PixelFormat* fmt);
void Scale_3DNow_Nearest (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
void Scale_3DNow_BilinearFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
void Scale_3DNow_BiAdaptAdvFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
void Scale_3DNow_TriScanFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
void Scale_3DNow_HqFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
extern const Scale_FuncDef_t Scale_3DNow_Functions[];
#endif /* _2XSCALERS_MMX_H_ */
+102
View File
@@ -0,0 +1,102 @@
/*
* This program is free software; you can redistribute it and/or modify
* it under the terms of the GNU General Public License as published by
* the Free Software Foundation; either version 2 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU General Public License for more details.
*
* You should have received a copy of the GNU General Public License
* along with this program; if not, write to the Free Software
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
*/
#ifdef GFXMODULE_SDL
#include "port.h"
#include "libs/platform.h"
#if defined(MMX_ASM)
#include "libs/graphics/sdl/sdl_common.h"
#include "types.h"
#include "scalers.h"
#include "scaleint.h"
#include "2xscalers.h"
#include "2xscalers_mmx.h"
// SSE name for all functions
#undef SCALE_
#define SCALE_(name) Scale ## _SSE_ ## name
// Tell them which opcodes we want to support
#define USE_MOVNTQ
#define USE_PREFETCH INTEL_PREFETCH
#define USE_PSADBW
// Bring in inline asm functions
#include "scalemmx.h"
// Scaler function lookup table
//
const Scale_FuncDef_t
Scale_SSE_Functions[] =
{
{TFB_GFXFLAGS_SCALE_BILINEAR, Scale_SSE_BilinearFilter},
{TFB_GFXFLAGS_SCALE_BIADAPT, Scale_BiAdaptFilter},
{TFB_GFXFLAGS_SCALE_BIADAPTADV, Scale_SSE_BiAdaptAdvFilter},
{TFB_GFXFLAGS_SCALE_TRISCAN, Scale_SSE_TriScanFilter},
{TFB_GFXFLAGS_SCALE_HQXX, Scale_MMX_HqFilter},
// Default
{0, Scale_SSE_Nearest}
};
void
Scale_SSE_PrepPlatform (const SDL_PixelFormat* fmt)
{
Scale_MMX_PrepPlatform (fmt);
}
// Nearest Neighbor scaling to 2x
// void Scale_SSE_Nearest (SDL_Surface *src,
// SDL_Surface *dst, SDL_Rect *r)
#include "nearest2x.c"
// Bilinear scaling to 2x
// void Scale_SSE_BilinearFilter (SDL_Surface *src,
// SDL_Surface *dst, SDL_Rect *r)
#include "bilinear2x.c"
// Advanced Biadapt scaling to 2x
// void Scale_SSE_BiAdaptAdvFilter (SDL_Surface *src,
// SDL_Surface *dst, SDL_Rect *r)
#include "biadv2x.c"
// Triscan scaling to 2x
// derivative of scale2x -- scale2x.sf.net
// void Scale_SSE_TriScanFilter (SDL_Surface *src,
// SDL_Surface *dst, SDL_Rect *r)
#include "triscan2x.c"
#if 0 && NO_IMPROVEMENT
// Hq2x scaling
// (adapted from 'hq2x' by Maxim Stepin -- www.hiend3d.com/hq2x.html)
// void Scale_SSE_HqFilter (SDL_Surface *src,
// SDL_Surface *dst, SDL_Rect *r)
#include "hq2x.c"
#endif
#endif /* MMX_ASM */
#endif /* GFXMODULE_SDL */
+21
View File
@@ -0,0 +1,21 @@
/*
* This program is free software; you can redistribute it and/or modify
* it under the terms of the GNU General Public License as published by
* the Free Software Foundation; either version 2 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU General Public License for more details.
*
* You should have received a copy of the GNU General Public License
* along with this program; if not, write to the Free Software
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
*/
#ifndef _2XSCALERS_SSE_H_
#define _2XSCALERS_SSE_H_
#endif /* _2XSCALERS_SSE_H_ */
+535
View File
@@ -0,0 +1,535 @@
/*
* Portions Copyright (C) 2003-2005 Alex Volkov (codepro@usa.net)
*
* This program is free software; you can redistribute it and/or modify
* it under the terms of the GNU General Public License as published by
* the Free Software Foundation; either version 2 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU General Public License for more details.
*
* You should have received a copy of the GNU General Public License
* along with this program; if not, write to the Free Software
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
*/
// Core algorithm of the Advanced BiAdaptive screen scaler
// Template
// When this file is built standalone is produces a plain C version
// Also #included by 2xscalers_mmx.c for an MMX version
#ifdef GFXMODULE_SDL
#include "libs/graphics/sdl/sdl_common.h"
#include "types.h"
#include "scalers.h"
#include "scaleint.h"
#include "2xscalers.h"
// Advanced biadapt scaling to 2x
// The name expands to either
// Scale_BiAdaptAdvFilter (for plain C) or
// Scale_MMX_BiAdaptAdvFilter (for MMX)
// [others when platforms are added]
void
SCALE_(BiAdaptAdvFilter) (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r)
{
int x, y;
const int w = src->w, h = src->h;
int xend, yend;
int dsrc, ddst;
SDL_Rect *region = r;
SDL_Rect limits;
SDL_PixelFormat *fmt = dst->format;
const int sp = src->pitch, dp = dst->pitch;
const int bpp = fmt->BytesPerPixel;
const int slen = sp / bpp, dlen = dp / bpp;
// for clarity purposes, the 'pixels' array here is transposed
Uint32 pixels[4][4];
static int resolve_coord[][2] =
{
{0, -1}, {1, -1}, { 2, 0}, { 2, 1},
{1, 2}, {0, 2}, {-1, 1}, {-1, 0},
{100, 100} // term
};
Uint32 *src_p = (Uint32 *)src->pixels;
Uint32 *dst_p = (Uint32 *)dst->pixels;
// these macros are for clarity; they make the current pixel (0,0)
// and allow to access pixels in all directions
#define PIX(x, y) (pixels[1 + (x)][1 + (y)])
#define SRC(x, y) (src_p + (x) + ((y) * slen))
// commonly used operations, for clarity also
// others are defined at their respective bpp levels
#define BIADAPT_RGBHIGH 8000
#define BIADAPT_YUVLOW 30
#define BIADAPT_YUVMED 70
#define BIADAPT_YUVHIGH 130
// high tolerance pixel comparison
#define BIADAPT_CMPRGB_HIGH(p1, p2) \
(p1 == p2 || SCALE_CMPRGB (p1, p2) <= BIADAPT_RGBHIGH)
// low tolerance pixel comparison
#define BIADAPT_CMPYUV_LOW(p1, p2) \
(p1 == p2 || SCALE_CMPYUV (p1, p2, BIADAPT_YUVLOW))
// medium tolerance pixel comparison
#define BIADAPT_CMPYUV_MED(p1, p2) \
(p1 == p2 || SCALE_CMPYUV (p1, p2, BIADAPT_YUVMED))
// high tolerance pixel comparison
#define BIADAPT_CMPYUV_HIGH(p1, p2) \
(p1 == p2 || SCALE_CMPYUV (p1, p2, BIADAPT_YUVHIGH))
SCALE_(PlatInit) ();
// expand updated region if necessary
// pixels neighbooring the updated region may
// change as a result of updates
limits.x = 0;
limits.y = 0;
limits.w = src->w;
limits.h = src->h;
Scale_ExpandRect (region, 2, &limits);
xend = region->x + region->w;
yend = region->y + region->h;
dsrc = slen - region->w;
ddst = (dlen - region->w) * 2;
#define SCALE_GETPIX(p) ( *(Uint32 *)(p) )
#define SCALE_SETPIX(p, c) ( *(Uint32 *)(p) = (c) )
// move ptrs to the first updated pixel
src_p += slen * region->y + region->x;
dst_p += (dlen * region->y + region->x) * 2;
for (y = region->y; y < yend; ++y, dst_p += ddst, src_p += dsrc)
{
for (x = region->x; x < xend; ++x, ++src_p, ++dst_p)
{
// pixel equality counter
int cmatch;
// most pixels will fall into 'all 4 equal'
// pattern, so we check it first
cmatch = 0;
PIX (0, 0) = SCALE_GETPIX (SRC (0, 0));
SCALE_SETPIX (dst_p, PIX (0, 0));
if (y + 1 < h)
{
// check pixel below the current one
PIX (0, 1) = SCALE_GETPIX (SRC (0, 1));
if (PIX (0, 0) == PIX (0, 1))
{
SCALE_SETPIX (dst_p + dlen, PIX (0, 0));
cmatch |= 1;
}
}
else
{
// last pixel in column - propagate
PIX (0, 1) = PIX (0, 0);
SCALE_SETPIX (dst_p + dlen, PIX (0, 0));
cmatch |= 1;
}
if (x + 1 < w)
{
// check pixel to the right from the current one
PIX (1, 0) = SCALE_GETPIX (SRC (1, 0));
if (PIX (0, 0) == PIX (1, 0))
{
SCALE_SETPIX (dst_p + 1, PIX (0, 0));
cmatch |= 2;
}
}
else
{
// last pixel in row - propagate
PIX (1, 0) = PIX (0, 0);
SCALE_SETPIX (dst_p + 1, PIX (0, 0));
cmatch |= 2;
}
if (cmatch == 3)
{
if (y + 1 >= h || x + 1 >= w)
{
// last pixel in row/column and nearest
// neighboor is identical
dst_p++;
SCALE_SETPIX (dst_p + dlen, PIX (0, 0));
continue;
}
// check pixel to the bottom-right
PIX (1, 1) = SCALE_GETPIX (SRC (1, 1));
if (PIX (0, 0) == PIX (1, 1))
{
// all 4 are equal - propagate
dst_p++;
SCALE_SETPIX (dst_p + dlen, PIX (0, 0));
continue;
}
}
// some neighboors are different, lets check them
if (x > 0)
PIX (-1, 0) = SCALE_GETPIX (SRC (-1, 0));
else
PIX (-1, 0) = PIX (0, 0);
if (x + 2 < w)
PIX (2, 0) = SCALE_GETPIX (SRC (2, 0));
else
PIX (2, 0) = PIX (1, 0);
if (y + 1 < h)
{
if (x > 0)
PIX (-1, 1) = SCALE_GETPIX (SRC (-1, 1));
else
PIX (-1, 1) = PIX (0, 1);
if (x + 2 < w)
{
PIX (1, 1) = SCALE_GETPIX (SRC (1, 1));
PIX (2, 1) = SCALE_GETPIX (SRC (2, 1));
}
else if (x + 1 < w)
{
PIX (1, 1) = SCALE_GETPIX (SRC (1, 1));
PIX (2, 1) = PIX (1, 1);
}
else
{
PIX (1, 1) = PIX (0, 1);
PIX (2, 1) = PIX (0, 1);
}
}
else
{
// last pixel in column
PIX (-1, 1) = PIX (-1, 0);
PIX (1, 1) = PIX (1, 0);
PIX (2, 1) = PIX (2, 0);
}
if (y + 2 < h)
{
PIX (0, 2) = SCALE_GETPIX (SRC (0, 2));
if (x > 0)
PIX (-1, 2) = SCALE_GETPIX (SRC (-1, 2));
else
PIX (-1, 2) = PIX (0, 2);
if (x + 2 < w)
{
PIX (1, 2) = SCALE_GETPIX (SRC (1, 2));
PIX (2, 2) = SCALE_GETPIX (SRC (2, 2));
}
else if (x + 1 < w)
{
PIX (1, 2) = SCALE_GETPIX (SRC (1, 2));
PIX (2, 2) = PIX (1, 2);
}
else
{
PIX (1, 2) = PIX (0, 2);
PIX (2, 2) = PIX (0, 2);
}
}
else
{
// last pixel in column
PIX (-1, 2) = PIX (-1, 1);
PIX (0, 2) = PIX (0, 1);
PIX (1, 2) = PIX (1, 1);
PIX (2, 2) = PIX (2, 1);
}
if (y > 0)
{
PIX (0, -1) = SCALE_GETPIX (SRC (0, -1));
if (x > 0)
PIX (-1, -1) = SCALE_GETPIX (SRC (-1, -1));
else
PIX (-1, -1) = PIX (0, -1);
if (x + 2 < w)
{
PIX (1, -1) = SCALE_GETPIX (SRC (1, -1));
PIX (2, -1) = SCALE_GETPIX (SRC (2, -1));
}
else if (x + 1 < w)
{
PIX (1, -1) = SCALE_GETPIX (SRC (1, -1));
PIX (2, -1) = PIX (1, -1);
}
else
{
PIX (1, -1) = PIX (0, -1);
PIX (2, -1) = PIX (0, -1);
}
}
else
{
PIX (-1, -1) = PIX (-1, 0);
PIX (0, -1) = PIX (0, 0);
PIX (1, -1) = PIX (1, 0);
PIX (2, -1) = PIX (2, 0);
}
// check pixel below the current one
if (!(cmatch & 1))
{
if (SCALE_CMPYUV (PIX (0, 0), PIX (0, 1), BIADAPT_YUVLOW))
{
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
PIX (0, 0), PIX (0, 1))
);
cmatch |= 1;
}
// detect a 2:1 line going across the current pixel
else if ( (PIX (0, 0) == PIX (-1, 0)
&& PIX (0, 0) == PIX (1, 1)
&& PIX (0, 0) == PIX (2, 1) &&
((!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, -1))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, -1))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, 0))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, 0))) ||
(!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 1))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, 2))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, 2))))) ||
(PIX (0, 0) == PIX (1, 0)
&& PIX (0, 0) == PIX (-1, 1)
&& PIX (0, 0) == PIX (2, -1) &&
((!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 0))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, -1))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, -1))) ||
(!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 2))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, 1))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, 0))))) )
{
SCALE_SETPIX (dst_p + dlen, PIX (0, 0));
}
// detect a 2:1 line going across the pixel below current
else if ( (PIX (0, 1) == PIX (-1, 0)
&& PIX (0, 1) == PIX (1, 1)
&& PIX (0, 1) == PIX (2, 2) &&
((!BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (-1, -1))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (1, 0))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (2, 1))) ||
(!BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (-1, 1))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (0, 2))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (1, 2))))) ||
(PIX (0, 1) == PIX (1, 0)
&& PIX (0, 1) == PIX (-1, 1)
&& PIX (0, 1) == PIX (2, 0) &&
((!BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (-1, 0))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (1, -1))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (2, -1))) ||
(!BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (-1, 2))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (0, 2))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (1, 1))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (2, 1))))) )
{
SCALE_SETPIX (dst_p + dlen, PIX (0, 1));
}
else
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
PIX (0, 0), PIX (0, 1))
);
}
dst_p++;
// check pixel to the right from the current one
if (!(cmatch & 2))
{
if (SCALE_CMPYUV (PIX (0, 0), PIX (1, 0), BIADAPT_YUVLOW))
{
SCALE_SETPIX (dst_p, Scale_Blend_11 (
PIX (0, 0), PIX (1, 0))
);
cmatch |= 2;
}
// detect a 1:2 line going across the current pixel
else if ( (PIX (0, 0) == PIX (1, -1)
&& PIX (0, 0) == PIX (0, 1)
&& PIX (0, 0) == PIX (-1, 2) &&
((!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, -1))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 0))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 1))) ||
(!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, -1))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, 1))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, 2))))) ||
(PIX (0, 0) == PIX (0, -1)
&& PIX (0, 0) == PIX (1, 1)
&& PIX (0, 0) == PIX (1, 2) &&
((!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, -1))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 0))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, 1))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, 2))) ||
(!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, -1))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, 1))
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, 2))))) )
{
SCALE_SETPIX (dst_p, PIX (0, 0));
}
// detect a 1:2 line going across the pixel to the right
else if ( (PIX (1, 0) == PIX (1, -1)
&& PIX (1, 0) == PIX (0, 1)
&& PIX (1, 0) == PIX (0, 2) &&
((!BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (0, -1))
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (-1, 1))
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (-1, 2))) ||
(!BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (2, -1))
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (2, 0))
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (1, 1))
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (1, 2))))) ||
(PIX (1, 0) == PIX (0, -1)
&& PIX (1, 0) == PIX (1, 1)
&& PIX (1, 0) == PIX (2, 2) &&
((!BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (-1, -1))
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (0, 1))
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (1, 2))) ||
(!BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (1, -1))
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (2, 0))
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (2, 1))))) )
{
SCALE_SETPIX (dst_p, PIX (1, 0));
}
else
SCALE_SETPIX (dst_p, Scale_Blend_11 (
PIX (0, 0), PIX (1, 0))
);
}
if (PIX (0, 0) == PIX (1, 1) && PIX (1, 0) == PIX (0, 1))
{
// diagonals are equal
int *coord;
int cl, cr;
Uint32 clr;
// both pairs are equal, have to resolve the pixel
// race; we try detecting which color is
// the background by looking for a line or an edge
// examine 8 pixels surrounding the current quad
cl = cr = 2;
for (coord = resolve_coord[0]; *coord < 100; coord += 2)
{
clr = PIX (coord[0], coord[1]);
if (BIADAPT_CMPYUV_MED (clr, PIX (0, 0)))
cl++;
else if (BIADAPT_CMPYUV_MED (clr, PIX (1, 0)))
cr++;
}
// least count wins
if (cl > cr)
clr = PIX (1, 0);
else if (cr > cl)
clr = PIX (0, 0);
else
clr = Scale_Blend_11 (PIX (0, 0), PIX (1, 0));
SCALE_SETPIX (dst_p + dlen, clr);
continue;
}
if (cmatch == 3
|| (BIADAPT_CMPYUV_LOW (PIX (1, 0), PIX (0, 1))
&& BIADAPT_CMPYUV_LOW (PIX (1, 0), PIX (1, 1))))
{
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
PIX (0, 1), PIX (1, 0))
);
continue;
}
else if (cmatch && BIADAPT_CMPYUV_LOW (PIX (0, 0), PIX (1, 1)))
{
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
PIX (0, 0), PIX (1, 1))
);
continue;
}
// check pixel to the bottom-right
if (BIADAPT_CMPYUV_HIGH (PIX (0, 0), PIX (1, 1))
&& BIADAPT_CMPYUV_HIGH (PIX (1, 0), PIX (0, 1)))
{
if (SCALE_GETY (PIX (0, 0)) > SCALE_GETY (PIX (1, 0)))
{
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
PIX (0, 0), PIX (1, 1))
);
}
else
{
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
PIX (1, 0), PIX (0, 1))
);
}
}
else if (BIADAPT_CMPYUV_HIGH (PIX (0, 0), PIX (1, 1)))
{
// main diagonal is same color
// use its value
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
PIX (0, 0), PIX (1, 1))
);
}
else if (BIADAPT_CMPYUV_HIGH (PIX (1, 0), PIX (0, 1)))
{
// 2nd diagonal is same color
// use its value
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
PIX (1, 0), PIX (0, 1))
);
}
else
{
// blend all 4
SCALE_SETPIX (dst_p + dlen, Scale_Blend_1111 (
PIX (0, 0), PIX (0, 1),
PIX (1, 0), PIX (1, 1)
));
}
}
}
SCALE_(PlatDone) ();
}
#endif /* GFXMODULE_SDL */
+115
View File
@@ -0,0 +1,115 @@
/*
* This program is free software; you can redistribute it and/or modify
* it under the terms of the GNU General Public License as published by
* the Free Software Foundation; either version 2 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU General Public License for more details.
*
* You should have received a copy of the GNU General Public License
* along with this program; if not, write to the Free Software
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
*/
// Core algorithm of the BiLinear screen scaler
// Template
// When this file is built standalone is produces a plain C version
// Also #included by 2xscalers_mmx.c for an MMX version
#ifdef GFXMODULE_SDL
#include "libs/graphics/sdl/sdl_common.h"
#include "types.h"
#include "scalers.h"
#include "scaleint.h"
#include "2xscalers.h"
// Bilinear scaling to 2x
// The name expands to either
// Scale_BilinearFilter (for plain C) or
// Scale_MMX_BilinearFilter (for MMX)
// Scale_SSE_BilinearFilter (for SSE)
// [others when platforms are added]
void
SCALE_(BilinearFilter) (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r)
{
int x, y;
const int w = src->w, h = src->h;
int xend, yend;
int dsrc, ddst;
SDL_Rect *region = r;
SDL_Rect limits;
SDL_PixelFormat *fmt = dst->format;
const int pitch = src->pitch, dp = dst->pitch;
const int bpp = fmt->BytesPerPixel;
const int len = pitch / bpp, dlen = dp / bpp;
Uint32 p[4]; // influential pixels array
Uint32 *srow0 = (Uint32 *) src->pixels;
Uint32 *dst_p = (Uint32 *) dst->pixels;
SCALE_(PlatInit) ();
// expand updated region if necessary
// pixels neighbooring the updated region may
// change as a result of updates
limits.x = 0;
limits.y = 0;
limits.w = w;
limits.h = h;
Scale_ExpandRect (region, 1, &limits);
xend = region->x + region->w;
yend = region->y + region->h;
dsrc = len - region->w;
ddst = (dlen - region->w) * 2;
// move ptrs to the first updated pixel
srow0 += len * region->y + region->x;
dst_p += (dlen * region->y + region->x) * 2;
for (y = region->y; y < yend; ++y, dst_p += ddst, srow0 += dsrc)
{
Uint32 *srow1;
SCALE_(Prefetch) (srow0 + 16);
SCALE_(Prefetch) (srow0 + 32);
if (y < h - 1)
srow1 = srow0 + len;
else
srow1 = srow0;
SCALE_(Prefetch) (srow1 + 16);
SCALE_(Prefetch) (srow1 + 32);
for (x = region->x; x < xend; ++x, ++srow0, ++srow1, dst_p += 2)
{
if (x < w - 1)
{ // can blend directly from pixels
SCALE_BILINEAR_BLEND4 (srow0, srow1, dst_p, dlen);
}
else
{ // need to make temp pixel rows
p[0] = srow0[0];
p[1] = p[0];
p[2] = srow1[0];
p[3] = p[2];
SCALE_BILINEAR_BLEND4 (&p[0], &p[2], dst_p, dlen);
}
}
SCALE_(Prefetch) (srow0 + dsrc);
SCALE_(Prefetch) (srow0 + dsrc + 16);
SCALE_(Prefetch) (srow1 + dsrc);
SCALE_(Prefetch) (srow1 + dsrc + 16);
}
SCALE_(PlatDone) ();
}
#endif /* GFXMODULE_SDL */
+208
View File
@@ -0,0 +1,208 @@
/*
* This program is free software; you can redistribute it and/or modify
* it under the terms of the GNU General Public License as published by
* the Free Software Foundation; either version 2 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU General Public License for more details.
*
* You should have received a copy of the GNU General Public License
* along with this program; if not, write to the Free Software
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
*/
// Core algorithm of the BiLinear screen scaler
// Template
// When this file is built standalone is produces a plain C version
// Also #included by 2xscalers_mmx.c for an MMX version
#ifdef GFXMODULE_SDL
#include "libs/graphics/sdl/sdl_common.h"
#include "types.h"
#include "scalers.h"
#include "scaleint.h"
#include "2xscalers.h"
// Nearest Neighbor scaling to 2x
// The name expands to
// Scale_Nearest (for plain C)
// Scale_MMX_Nearest (for MMX)
// Scale_SSE_Nearest (for SSE)
// [others when platforms are added]
void
SCALE_(Nearest) (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r)
{
int y;
const int rw = r->w, rh = r->h;
const int sp = src->pitch, dp = dst->pitch;
const int bpp = dst->format->BytesPerPixel;
const int slen = sp / bpp, dlen = dp / bpp;
const int dsrc = slen-rw, ddst = (dlen-rw) * 2;
Uint32 *src_p = (Uint32 *)src->pixels;
Uint32 *dst_p = (Uint32 *)dst->pixels;
// guard asm code against such atrocities
if (rw == 0 || rh == 0)
return;
SCALE_(PlatInit) ();
// move ptrs to the first updated pixel
src_p += slen * r->y + r->x;
dst_p += (dlen * r->y + r->x) * 2;
#if defined(MMX_ASM) && defined(MSVC_ASM)
// Just about everything has to be done in asm for MSVC
// to actually take advantage of asm here
// MSVC does not support beautiful GCC-like asm templates
y = rh;
__asm
{
// setup vars
mov esi, src_p
mov edi, dst_p
PREFETCH (esi + 0x40)
PREFETCH (esi + 0x80)
PREFETCH (esi + 0xc0)
mov edx, dlen
lea edx, [edx * 4]
mov eax, dsrc
lea eax, [eax * 4]
mov ebx, ddst
lea ebx, [ebx * 4]
mov ecx, rw
loop_y:
test ecx, 1
jz even_x
// one-pixel transfer
movd mm1, [esi]
punpckldq mm1, mm1 // pix1 | pix1 -> mm1
add esi, 4
MOVNTQ (edi, mm1)
add edi, 8
MOVNTQ (edi - 8 + edx, mm1)
even_x:
shr ecx, 1 // x = rw / 2
loop_x:
// two-pixel transfer
movq mm1, [esi]
movq mm2, mm1
PREFETCH (esi + 0x100)
punpckldq mm1, mm1 // pix1 | pix1 -> mm1
add esi, 8
MOVNTQ (edi, mm1)
punpckhdq mm2, mm2 // pix2 | pix2 -> mm2
MOVNTQ (edi + edx, mm1)
add edi, 16
MOVNTQ (edi - 8, mm2)
MOVNTQ (edi - 8 + edx, mm2)
dec ecx
jnz loop_x
// try to prefetch as early as possible to have it on time
PREFETCH (esi + eax)
mov ecx, rw
add esi, eax
PREFETCH (esi + 0x40)
PREFETCH (esi + 0x80)
PREFETCH (esi + 0xc0)
add edi, ebx
dec y
jnz loop_y
}
#elif defined(MMX_ASM) && defined(GCC_ASM)
SCALE_(Prefetch) (src_p + 16);
SCALE_(Prefetch) (src_p + 32);
SCALE_(Prefetch) (src_p + 48);
for (y = rh; y; --y)
{
int x = rw;
if (x & 1)
{ // one-pixel transfer
__asm__ (
"movd (%0), %%mm1 \n\t"
"punpckldq %%mm1, %%mm1 \n\t"
MOVNTQ (%%mm1, (%1)) "\n\t"
MOVNTQ (%%mm1, (%1,%2)) "\n\t"
: /* nothing */
: /*0*/"r" (src_p), /*1*/"r" (dst_p), /*2*/"r" (dlen*sizeof(Uint32))
);
++src_p;
dst_p += 2;
--x;
}
for (x >>= 1; x; --x, src_p += 2, dst_p += 4)
{ // two-pixel transfer
__asm__ (
"movq (%0), %%mm1 \n\t"
"movq %%mm1, %%mm2 \n\t"
PREFETCH (0x100(%0)) "\n\t"
"punpckldq %%mm1, %%mm1 \n\t"
MOVNTQ (%%mm1, (%1)) "\n\t"
MOVNTQ (%%mm1, (%1,%2)) "\n\t"
"punpckhdq %%mm2, %%mm2 \n\t"
MOVNTQ (%%mm2, 8(%1)) "\n\t"
MOVNTQ (%%mm2, 8(%1,%2)) "\n\t"
: /* nothing */
: /*0*/"r" (src_p), /*1*/"r" (dst_p), /*2*/"r" (dlen*sizeof(Uint32))
);
}
src_p += dsrc;
// try to prefetch as early as possible to have it on time
SCALE_(Prefetch) (src_p);
dst_p += ddst;
SCALE_(Prefetch) (src_p + 16);
SCALE_(Prefetch) (src_p + 32);
SCALE_(Prefetch) (src_p + 48);
}
#else
// Plain C version
for (y = 0; y < rh; ++y)
{
int x;
for (x = 0; x < rw; ++x, ++src_p, dst_p += 2)
{
Uint32 pix = *src_p;
dst_p[0] = pix;
dst_p[1] = pix;
dst_p[dlen] = pix;
dst_p[dlen + 1] = pix;
}
dst_p += ddst;
src_p += dsrc;
}
#endif
SCALE_(PlatDone) ();
}
#endif /* GFXMODULE_SDL */
+433
View File
@@ -0,0 +1,433 @@
/*
* Copyright (C) 2005 Alex Volkov (codepro@usa.net)
*
* This program is free software; you can redistribute it and/or modify
* it under the terms of the GNU General Public License as published by
* the Free Software Foundation; either version 2 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU General Public License for more details.
*
* You should have received a copy of the GNU General Public License
* along with this program; if not, write to the Free Software
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
*/
// Scalers Internals
#ifndef SCALEINT_H_
#define SCALEINT_H_
#include "libs/graphics/sdl/sdl_common.h"
#include "types.h"
// Plain C names
#define SCALE_(name) Scale ## _ ## name
// These are defaults
#define SCALE_GETPIX(p) ( *(Uint32 *)(p) )
#define SCALE_SETPIX(p, c) ( *(Uint32 *)(p) = (c) )
// Plain C defaults
#define SCALE_CMPRGB(p1, p2) \
SCALE_(GetRGBDelta) (fmt, p1, p2)
#define SCALE_TOYUV(p) \
SCALE_(RGBtoYUV) (fmt, p)
#define SCALE_CMPYUV(p1, p2, toler) \
SCALE_(CmpYUV) (fmt, p1, p2, toler)
#define SCALE_DIFFYUV(p1, p2) \
SCALE_(DiffYUV) (p1, p2)
#define SCALE_DIFFYUV_TY 0x40
#define SCALE_DIFFYUV_TU 0x12
#define SCALE_DIFFYUV_TV 0x0c
#define SCALE_GETY(p) \
SCALE_(GetPixY) (fmt, p)
#define SCALE_BILINEAR_BLEND4(r0, r1, dst, dlen) \
SCALE_(Blend_bilinear) (r0, r1, dst, dlen)
#define NO_PREFETCH 0
#define INTEL_PREFETCH 1
#define AMD_PREFETCH 2
typedef enum
{
YUV_XFORM_R = 0,
YUV_XFORM_G = 1,
YUV_XFORM_B = 2,
YUV_XFORM_Y = 0,
YUV_XFORM_U = 1,
YUV_XFORM_V = 2
} RGB_YUV_INDEX;
extern const int YUV_matrix[3][3];
// pre-computed transformations for 8 bits per channel
extern int RGB_to_YUV[/*RGB*/ 3][/*YUV*/ 3][ /*mult-res*/ 256];
extern sint16 dRGB_to_dYUV[/*RGB*/ 3][/*YUV*/ 3][ /*mult-res*/ 512];
typedef Uint32 YUV_VECTOR;
// pre-computed transformations for RGB555
extern YUV_VECTOR RGB15_to_YUV[0x8000];
// Platform+Scaler function lookups
//
typedef struct
{
int flag;
TFB_ScaleFunc func;
} Scale_FuncDef_t;
// expands the given rectangle in all directions by 'expansion'
// guarded by 'limits'
extern void Scale_ExpandRect (SDL_Rect* rect, int expansion,
const SDL_Rect* limits);
// Standard plain C versions of support functions
// Initialize various platform-specific features
static inline void
SCALE_(PlatInit) (void)
{
}
// Finish with various platform-specific features
static inline void
SCALE_(PlatDone) (void)
{
}
#if 0
static inline void
SCALE_(Prefetch) (const void* p)
{
/* no-op in pure C */
(void)p;
}
#else
# define Scale_Prefetch(p)
#endif
// compute the RGB distance squared between 2 pixels
// Plain C version
static inline int
SCALE_(GetRGBDelta) (const SDL_PixelFormat* fmt, Uint32 pix1, Uint32 pix2)
{
int c;
int delta;
c = ((pix1 >> fmt->Rshift) & 0xff) - ((pix2 >> fmt->Rshift) & 0xff);
delta = c * c;
c = ((pix1 >> fmt->Gshift) & 0xff) - ((pix2 >> fmt->Gshift) & 0xff);
delta += c * c;
c = ((pix1 >> fmt->Bshift) & 0xff) - ((pix2 >> fmt->Bshift) & 0xff);
delta += c * c;
return delta;
}
// retrieve the Y (intensity) component of pixel's YUV
// Plain C version
static inline int
SCALE_(GetPixY) (const SDL_PixelFormat* fmt, Uint32 pix)
{
Uint32 r, g, b;
r = (pix >> fmt->Rshift) & 0xff;
g = (pix >> fmt->Gshift) & 0xff;
b = (pix >> fmt->Bshift) & 0xff;
return RGB_to_YUV [YUV_XFORM_R][YUV_XFORM_Y][r]
+ RGB_to_YUV [YUV_XFORM_G][YUV_XFORM_Y][g]
+ RGB_to_YUV [YUV_XFORM_B][YUV_XFORM_Y][b];
}
static inline YUV_VECTOR
SCALE_(RGBtoYUV) (const SDL_PixelFormat* fmt, Uint32 pix)
{
return RGB15_to_YUV[
(((pix >> (fmt->Rshift + 3)) & 0x1f) << 10) |
(((pix >> (fmt->Gshift + 3)) & 0x1f) << 5) |
(((pix >> (fmt->Bshift + 3)) & 0x1f) )
];
}
// compare 2 pixels with respect to their YUV representations
// tolerance set by toler arg
// returns true: close; false: distant (-gt toler)
// Plain C version
static inline bool
SCALE_(CmpYUV) (const SDL_PixelFormat* fmt, Uint32 pix1, Uint32 pix2, int toler)
#if 1
{
int dr, dg, db;
int delta;
dr = ((pix1 >> fmt->Rshift) & 0xff) - ((pix2 >> fmt->Rshift) & 0xff) + 255;
dg = ((pix1 >> fmt->Gshift) & 0xff) - ((pix2 >> fmt->Gshift) & 0xff) + 255;
db = ((pix1 >> fmt->Bshift) & 0xff) - ((pix2 >> fmt->Bshift) & 0xff) + 255;
// compute Y delta
delta = abs (dRGB_to_dYUV [YUV_XFORM_R][YUV_XFORM_Y][dr]
+ dRGB_to_dYUV [YUV_XFORM_G][YUV_XFORM_Y][dg]
+ dRGB_to_dYUV [YUV_XFORM_B][YUV_XFORM_Y][db]);
if (delta > toler)
return false;
// compute U delta
delta += abs (dRGB_to_dYUV [YUV_XFORM_R][YUV_XFORM_U][dr]
+ dRGB_to_dYUV [YUV_XFORM_G][YUV_XFORM_U][dg]
+ dRGB_to_dYUV [YUV_XFORM_B][YUV_XFORM_U][db]);
if (delta > toler)
return false;
// compute V delta
delta += abs (dRGB_to_dYUV [YUV_XFORM_R][YUV_XFORM_V][dr]
+ dRGB_to_dYUV [YUV_XFORM_G][YUV_XFORM_V][dg]
+ dRGB_to_dYUV [YUV_XFORM_B][YUV_XFORM_V][db]);
return delta <= toler;
}
#else
{
int delta;
Uint32 yuv1, yuv2;
yuv1 = RGB15_to_YUV[
(((pix1 >> (fmt->Rshift + 3)) & 0x1f) << 10) |
(((pix1 >> (fmt->Gshift + 3)) & 0x1f) << 5) |
(((pix1 >> (fmt->Bshift + 3)) & 0x1f) )
];
yuv2 = RGB15_to_YUV[
(((pix2 >> (fmt->Rshift + 3)) & 0x1f) << 10) |
(((pix2 >> (fmt->Gshift + 3)) & 0x1f) << 5) |
(((pix2 >> (fmt->Bshift + 3)) & 0x1f) )
];
// compute Y delta
delta = abs ((yuv1 & 0xff0000) - (yuv2 & 0xff0000)) >> 16;
if (delta > toler)
return false;
// compute U delta
delta += abs ((yuv1 & 0x00ff00) - (yuv2 & 0x00ff00)) >> 8;
if (delta > toler)
return false;
// compute V delta
delta += abs ((yuv1 & 0x0000ff) - (yuv2 & 0x0000ff));
return delta <= toler;
}
#endif
// Check if 2 pixels are different with respect to their
// YUV representations
// returns 0: close; ~0: distant
static inline int
SCALE_(DiffYUV) (Uint32 yuv1, Uint32 yuv2)
{
// non-branching version -- assumes 2's complement integers
// delta math only needs 25 bits and we have 32 available;
// only interested in the sign bits after subtraction
sint32 delta, ret;
if (yuv1 == yuv2)
return 0;
// compute Y delta
delta = abs ((yuv1 & 0xff0000) - (yuv2 & 0xff0000));
ret = (SCALE_DIFFYUV_TY << 16) - delta; // save sign bit
// compute U delta
delta = abs ((yuv1 & 0x00ff00) - (yuv2 & 0x00ff00));
ret |= (SCALE_DIFFYUV_TU << 8) - delta; // save sign bit
// compute V delta
delta = abs ((yuv1 & 0x0000ff) - (yuv2 & 0x0000ff));
ret |= SCALE_DIFFYUV_TV - delta; // save sign bit
return (ret >> 31);
}
// blends two pixels with 1:1 ratio
static inline Uint32
SCALE_(Blend_11) (Uint32 pix1, Uint32 pix2)
{
/* (pix1 + pix2) >> 1 */
return
/* lower bits can be safely ignored - the error is minimal
expression that calcs them is left for posterity
(pix1 & pix2 & low_mask) +
*/
((pix1 & 0xfefefefe) >> 1) + ((pix2 & 0xfefefefe) >> 1);
}
// blends four pixels with 1:1:1:1 ratio
static inline Uint32
SCALE_(Blend_1111) (Uint32 pix1, Uint32 pix2,
Uint32 pix3, Uint32 pix4)
{
/* (pix1 + pix2 + pix3 + pix4) >> 2 */
return
/* lower bits can be safely ignored - the error is minimal
expression that calcs them is left for posterity
((((pix1 & low_mask) + (pix2 & low_mask) +
(pix3 & low_mask) + (pix4 & low_mask)
) >> 2) & low_mask) +
*/
((pix1 & 0xfcfcfcfc) >> 2) + ((pix2 & 0xfcfcfcfc) >> 2) +
((pix3 & 0xfcfcfcfc) >> 2) + ((pix4 & 0xfcfcfcfc) >> 2);
}
// blends pixels with 3:1 ratio
static inline Uint32
Scale_Blend_31 (Uint32 pix1, Uint32 pix2)
{
/* (pix1 * 3 + pix2) / 4 */
/* lower bits can be safely ignored - the error is minimal */
return ((pix1 & 0xfefefefe) >> 1) + ((pix1 & 0xfcfcfcfc) >> 2) +
((pix2 & 0xfcfcfcfc) >> 2);
}
// blends pixels with 2:1:1 ratio
static inline Uint32
Scale_Blend_211 (Uint32 pix1, Uint32 pix2, Uint32 pix3)
{
/* (pix1 * 2 + pix2 + pix3) / 4 */
/* lower bits can be safely ignored - the error is minimal */
return ((pix1 & 0xfefefefe) >> 1) +
((pix2 & 0xfcfcfcfc) >> 2) +
((pix3 & 0xfcfcfcfc) >> 2);
}
// blends pixels with 5:2:1 ratio
static inline Uint32
Scale_Blend_521 (Uint32 pix1, Uint32 pix2, Uint32 pix3)
{
/* (pix1 * 5 + pix2 * 2 + pix3) / 8 */
/* lower bits can be safely ignored - the error is minimal */
return ((pix1 & 0xfefefefe) >> 1) + ((pix1 & 0xf8f8f8f8) >> 3) +
((pix2 & 0xfcfcfcfc) >> 2) +
((pix3 & 0xf8f8f8f8) >> 3) +
0x02020202 /* half-error */;
}
// blends pixels with 6:1:1 ratio
static inline Uint32
Scale_Blend_611 (Uint32 pix1, Uint32 pix2, Uint32 pix3)
{
/* (pix1 * 6 + pix2 + pix3) / 8 */
/* lower bits can be safely ignored - the error is minimal */
return ((pix1 & 0xfefefefe) >> 1) + ((pix1 & 0xfcfcfcfc) >> 2) +
((pix2 & 0xf8f8f8f8) >> 3) +
((pix3 & 0xf8f8f8f8) >> 3) +
0x02020202 /* half-error */;
}
// blends pixels with 2:3:3 ratio
static inline Uint32
Scale_Blend_233 (Uint32 pix1, Uint32 pix2, Uint32 pix3)
{
/* (pix1 * 2 + pix2 * 3 + pix3 * 3) / 8 */
/* lower bits can be safely ignored - the error is minimal */
return ((pix1 & 0xfcfcfcfc) >> 2) +
((pix2 & 0xfcfcfcfc) >> 2) + ((pix2 & 0xf8f8f8f8) >> 3) +
((pix3 & 0xfcfcfcfc) >> 2) + ((pix3 & 0xf8f8f8f8) >> 3) +
0x02020202 /* half-error */;
}
// blends pixels with 14:1:1 ratio
static inline Uint32
Scale_Blend_e11 (Uint32 pix1, Uint32 pix2, Uint32 pix3)
{
/* (pix1 * 14 + pix2 + pix3) >> 4 */
/* lower bits can be safely ignored - the error is minimal */
return ((pix1 & 0xfefefefe) >> 1) + ((pix1 & 0xfcfcfcfc) >> 2) +
((pix1 & 0xf8f8f8f8) >> 3) +
((pix2 & 0xf0f0f0f0) >> 4) +
((pix3 & 0xf0f0f0f0) >> 4) +
0x03030303 /* half-error */;
}
// Halfs the pixel's intensity
static inline Uint32
SCALE_(HalfPixel) (Uint32 pix)
{
return ((pix & 0xfefefefe) >> 1);
}
// Bilinear weighted blend of four pixels
// Function produces 4 blended pixels and writes them
// out to the surface (in 2x2 matrix)
// Pixels are computed using expanded weight matrix like so:
// ('sp' - source pixel, 'dp' - destination pixel)
// dp[0] = (9*sp[0] + 3*sp[1] + 3*sp[2] + 1*sp[3]) / 16
// dp[1] = (3*sp[0] + 9*sp[1] + 1*sp[2] + 3*sp[3]) / 16
// dp[2] = (3*sp[0] + 1*sp[1] + 9*sp[2] + 3*sp[3]) / 16
// dp[3] = (1*sp[0] + 3*sp[1] + 3*sp[2] + 9*sp[3]) / 16
static inline void
SCALE_(Blend_bilinear) (const Uint32* row0, const Uint32* row1,
Uint32* dst_p, Uint32 dlen)
{
// We loose some lower bits here and try to compensate for
// that by adding half-error values.
// In general, the error is minimal (+-7)
// The >>4 reduction is achieved gradually
# define BL_PACKED_HALF(p) \
(((p) & 0xfefefefe) >> 1)
# define BL_SUM(p1, p2) \
(BL_PACKED_HALF(p1) + BL_PACKED_HALF(p2))
# define BL_HALF_ERR 0x01010101
# define BL_SUM_WERR(p1, p2) \
(BL_PACKED_HALF(p1) + BL_PACKED_HALF(p2) + BL_HALF_ERR)
Uint32 sum1111, sum1331, sum3113;
// cache p[0] + 3*(p[1] + p[2]) + p[3] in sum1331
// cache p[1] + 3*(p[0] + p[3]) + p[2] in sum3113
sum1331 = BL_SUM (row0[1], row1[0]);
sum3113 = BL_SUM (row0[0], row1[1]);
// cache p[0] + p[1] + p[2] + p[3] in sum1111
sum1111 = BL_SUM_WERR (sum1331, sum3113);
sum1331 = BL_SUM_WERR (sum1331, sum1111);
sum1331 = BL_PACKED_HALF (sum1331);
sum3113 = BL_SUM_WERR (sum3113, sum1111);
sum3113 = BL_PACKED_HALF (sum3113);
// pixel 0 math -- (9*p[0] + 3*(p[1] + p[2]) + p[3]) / 16
dst_p[0] = BL_PACKED_HALF (row0[0]) + sum1331;
// pixel 1 math -- (9*p[1] + 3*(p[0] + p[3]) + p[2]) / 16
dst_p[1] = BL_PACKED_HALF (row0[1]) + sum3113;
// pixel 2 math -- (9*p[2] + 3*(p[0] + p[3]) + p[1]) / 16
dst_p[dlen] = BL_PACKED_HALF (row1[0]) + sum3113;
// pixel 3 math -- (9*p[3] + 3*(p[1] + p[2]) + p[0]) / 16
dst_p[dlen + 1] = BL_PACKED_HALF (row1[1]) + sum1331;
# undef BL_PACKED_HALF
# undef BL_SUM
# undef BL_HALF_ERR
# undef BL_SUM_WERR
}
#endif /* SCALEINT_H_ */
+790
View File
@@ -0,0 +1,790 @@
/*
* Copyright (C) 2005 Alex Volkov (codepro@usa.net)
*
* This program is free software; you can redistribute it and/or modify
* it under the terms of the GNU General Public License as published by
* the Free Software Foundation; either version 2 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU General Public License for more details.
*
* You should have received a copy of the GNU General Public License
* along with this program; if not, write to the Free Software
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
*/
#ifndef SCALEMMX_H_
#define SCALEMMX_H_
#if !defined(SCALE_)
# error Please define SCALE_(name) before including scalemmx.h
#endif
#if !defined(MSVC_ASM) && !defined(GCC_ASM)
# error Please define target assembler (MSVC_ASM, GCC_ASM) before including scalemmx.h
#endif
// MMX defaults (no Format param)
#undef SCALE_CMPRGB
#define SCALE_CMPRGB(p1, p2) \
SCALE_(GetRGBDelta) (p1, p2)
#undef SCALE_TOYUV
#define SCALE_TOYUV(p) \
SCALE_(RGBtoYUV) (p)
#undef SCALE_CMPYUV
#define SCALE_CMPYUV(p1, p2, toler) \
SCALE_(CmpYUV) (p1, p2, toler)
#undef SCALE_GETY
#define SCALE_GETY(p) \
SCALE_(GetPixY) (p)
// MMX transformation multipliers
extern Uint64 mmx_888to555_mult;
extern Uint64 mmx_Y_mult;
extern Uint64 mmx_U_mult;
extern Uint64 mmx_V_mult;
extern Uint64 mmx_YUV_threshold;
#define USE_YUV_LOOKUP
#if defined(MSVC_ASM)
// MSVC inline assembly versions
#if defined(USE_MOVNTQ)
# define MOVNTQ(addr, val) movntq [addr], val
#else
# define MOVNTQ(addr, val) movq [addr], val
#endif
#if USE_PREFETCH == INTEL_PREFETCH
// using Intel SSE non-temporal prefetch
# define PREFETCH(addr) prefetchnta [addr]
# define HAVE_PREFETCH
#elif USE_PREFETCH == AMD_PREFETCH
// using AMD 3DNOW! prefetch
# define PREFETCH(addr) prefetch [addr]
# define HAVE_PREFETCH
#else
// no prefetch -- too bad for poor MMX-only souls
# define PREFETCH(addr)
# undef HAVE_PREFETCH
#endif
static inline void
SCALE_(PlatInit) (void)
{
__asm
{
// mm0 will be kept == 0 throughout
// 0 is needed for bytes->words unpack instructions
pxor mm0, mm0
// mm5-mm7 contains RGB->YUV mults
movq mm7, mmx_Y_mult
movq mm6, mmx_U_mult
movq mm5, mmx_V_mult
#ifdef USE_YUV_LOOKUP
// mm6 contains RGB888->555 shuffle mult
movq mm6, mmx_888to555_mult
// mm5 contains DiffYUV threshold
movq mm5, mmx_YUV_threshold
#endif
}
}
static inline void
SCALE_(PlatDone) (void)
{
// finish with MMX registers and yield them to FPU
__asm
{
emms
}
}
#if defined(HAVE_PREFETCH)
static inline void
SCALE_(Prefetch) (const void* p)
{
__asm
{
mov eax, p
PREFETCH (eax)
}
}
#else /* Not HAVE_PREFETCH */
static inline void
SCALE_(Prefetch) (const void* p) { /* no-op */ }
#endif /* HAVE_PREFETCH */
// compute the RGB distance squared between 2 pixels
static inline int
SCALE_(GetRGBDelta) (Uint32 pix1, Uint32 pix2)
{
__asm
{
// load pixels
movd mm1, pix1
punpcklbw mm1, mm0
movd mm2, pix2
punpcklbw mm2, mm0
// get the difference between RGBA components
psubw mm1, mm2
// squared and sumed
pmaddwd mm1, mm1
// finish suming the squares
movq mm2, mm1
punpckhdq mm2, mm0
paddd mm1, mm2
// store result
movd eax, mm1
}
}
// retrieve the Y (intensity) component of pixel's YUV
static inline int
SCALE_(GetPixY) (Uint32 pix)
{
__asm
{
// load pixel
movd mm1, pix
punpcklbw mm1, mm0
// process
pmaddwd mm1, mmx_Y_mult // RGB * Yvec
movq mm2, mm1 // finish suming
punpckhdq mm2, mm0 // ditto
paddd mm1, mm2 // ditto
// store result
movd eax, mm1
shr eax, 14
}
}
#ifdef USE_YUV_LOOKUP
// convert pixel RGB vector into YUV representation vector
static inline YUV_VECTOR
SCALE_(RGBtoYUV) (Uint32 pix)
{
__asm
{
// convert RGB888 to 555
movd mm1, pix
punpcklbw mm1, mm0
psrlw mm1, 3 // 8->5 bit
pmaddwd mm1, mmx_888to555_mult // shuffle into the right channel order
movq mm2, mm1 // finish shuffling
punpckhdq mm2, mm0 // ditto
por mm1, mm2 // ditto
// lookup the YUV vector
movd eax, mm1
mov eax, [RGB15_to_YUV + eax * 4]
}
}
// compare 2 pixels with respect to their YUV representations
// tolerance set by toler arg
// returns true: close; false: distant (-gt toler)
static inline bool
SCALE_(CmpYUV) (Uint32 pix1, Uint32 pix2, int toler)
{
__asm
{
// convert RGB888 to 555
movd mm1, pix1
punpcklbw mm1, mm0
psrlw mm1, 3 // 8->5 bit
movd mm3, pix2
punpcklbw mm3, mm0
psrlw mm3, 3 // 8->5 bit
pmaddwd mm1, mmx_888to555_mult // shuffle into the right channel order
movq mm2, mm1 // finish shuffling
pmaddwd mm3, mmx_888to555_mult // shuffle into the right channel order
movq mm4, mm3 // finish shuffling
punpckhdq mm2, mm0 // ditto
por mm1, mm2 // ditto
punpckhdq mm4, mm0 // ditto
por mm3, mm4 // ditto
// lookup the YUV vector
movd eax, mm1
movd edx, mm3
movd mm1, [RGB15_to_YUV + eax * 4]
movq mm4, mm1
movd mm2, [RGB15_to_YUV + edx * 4]
// get abs difference between YUV components
#ifdef USE_PSADBW
// we can use PSADBW and save us some grief
psadbw mm1, mm2
movd edx, mm1
#else
// no PSADBW -- have to do it the hard way
psubusb mm1, mm2
psubusb mm2, mm4
por mm1, mm2
// sum the differences
// XXX: technically, this produces a MAX diff of 510
// but we do not need anything bigger, currently
movq mm2, mm1
psrlq mm2, 8
paddusb mm1, mm2
psrlq mm2, 8
paddusb mm1, mm2
movd edx, mm1
and edx, 0xff
#endif /* USE_PSADBW */
xor eax, eax
shl edx, 1
cmp edx, toler
// store result
setle al
}
}
#else /* Not USE_YUV_LOOKUP */
// convert pixel RGB vector into YUV representation vector
static inline YUV_VECTOR
SCALE_(RGBtoYUV) (Uint32 pix)
{
__asm
{
movd mm1, pix
punpcklbw mm1, mm0
movq mm2, mm1
// Y vector multiply
pmaddwd mm1, mmx_Y_mult
movq mm4, mm1
punpckhdq mm4, mm0
punpckldq mm1, mm0 // clear out the high dword
paddd mm1, mm4
psrad mm1, 15
movq mm3, mm2
// U vector multiply
pmaddwd mm2, mmx_U_mult
psrad mm2, 10
// V vector multiply
pmaddwd mm3, mmx_V_mult
psrad mm3, 10
// load (1|1|1|1) into mm4
pcmpeqw mm4, mm4
psrlw mm4, 15
packssdw mm3, mm2
pmaddwd mm3, mm4
psrad mm3, 5
// load (64|64) into mm4
punpcklwd mm4, mm0
pslld mm4, 6
paddd mm3, mm4
packssdw mm3, mm1
packuswb mm3, mm0
movd eax, mm3
}
}
// compare 2 pixels with respect to their YUV representations
// tolerance set by toler arg
// returns true: close; false: distant (-gt toler)
static inline bool
SCALE_(CmpYUV) (Uint32 pix1, Uint32 pix2, int toler)
{
__asm
{
movd mm1, pix1
punpcklbw mm1, mm0
movd mm2, pix2
punpcklbw mm2, mm0
psubw mm1, mm2
movq mm2, mm1
// Y vector multiply
pmaddwd mm1, mmx_Y_mult
movq mm4, mm1
punpckhdq mm4, mm0
paddd mm1, mm4
// abs()
movq mm4, mm1
psrad mm4, 31
pxor mm4, mm1
psubd mm1, mm4
movq mm3, mm2
// U vector multiply
pmaddwd mm2, mmx_U_mult
movq mm4, mm2
punpckhdq mm4, mm0
paddd mm2, mm4
// abs()
movq mm4, mm2
psrad mm4, 31
pxor mm4, mm2
psubd mm2, mm4
paddd mm1, mm2
// V vector multiply
pmaddwd mm3, mmx_V_mult
movq mm4, mm3
punpckhdq mm3, mm0
paddd mm3, mm4
// abs()
movq mm4, mm3
psrad mm4, 31
pxor mm4, mm3
psubd mm3, mm4
paddd mm1, mm3
movd edx, mm1
xor eax, eax
shr edx, 14
cmp edx, toler
// store result
setle al
}
}
#endif /* USE_YUV_LOOKUP */
// Check if 2 pixels are different with respect to their
// YUV representations
// returns 0: close; ~0: distant
static inline int
SCALE_(DiffYUV) (Uint32 yuv1, Uint32 yuv2)
{
__asm
{
// load YUV pixels
movd mm1, yuv1
movq mm4, mm1
movd mm2, yuv2
// abs difference between channels
psubusb mm1, mm2
psubusb mm2, mm4
por mm1, mm2
// compare to threshold
psubusb mm1, mmx_YUV_threshold
movd edx, mm1
// transform eax to 0 or ~0
xor eax, eax
or edx, edx
setz al
dec eax
}
}
// bilinear weighted blend of four pixels
// MSVC asm version
static inline void
SCALE_(Blend_bilinear) (const Uint32* row0, const Uint32* row1,
Uint32* dst_p, Uint32 dlen)
{
__asm
{
// EL0: setup vars
mov ebx, row0 // EL0
// EL0: load pixels
movq mm1, [ebx] // EL0
movq mm2, mm1 // EL0: p[1] -> mm2
PREFETCH (ebx + 0x80)
punpckhbw mm2, mm0 // EL0: p[1] -> mm2
mov ebx, row1
punpcklbw mm1, mm0 // EL0: p[0] -> mm1
movq mm3, [ebx]
movq mm4, mm3 // EL0: p[3] -> mm4
movq mm6, mm2 // EL1.1: p[1] -> mm6
PREFETCH (ebx + 0x80)
punpcklbw mm3, mm0 // EL0: p[2] -> mm3
movq mm5, mm1 // EL1.1: p[0] -> mm5
punpckhbw mm4, mm0 // EL0: p[3] -> mm4
mov edi, dst_p // EL0
// EL1: cache p[0] + 3*(p[1] + p[2]) + p[3] in mm6
paddw mm6, mm3 // EL1.2: p[1] + p[2] -> mm6
// EL1: cache p[0] + p[1] + p[2] + p[3] in mm7
movq mm7, mm6 // EL1.3: p[1] + p[2] -> mm7
// EL1: cache p[1] + 3*(p[0] + p[3]) + p[2] in mm5
paddw mm5, mm4 // EL1.2: p[0] + p[3] -> mm5
psllw mm6, 1 // EL1.4: 2*(p[1] + p[2]) -> mm6
paddw mm7, mm5 // EL1.4: sum(p[]) -> mm7
psllw mm5, 1 // EL1.5: 2*(p[0] + p[3]) -> mm5
paddw mm6, mm7 // EL1.5: p[0] + 3*(p[1] + p[2]) + p[3] -> mm6
paddw mm5, mm7 // EL1.6: p[1] + 3*(p[0] + p[3]) + p[2] -> mm5
// EL2: pixel 0 math -- (9*p[0] + 3*(p[1] + p[2]) + p[3]) / 16
psllw mm1, 3 // EL2.1: 8*p[0] -> mm1
paddw mm1, mm6 // EL2.2: 9*p[0] + 3*(p[1] + p[2]) + p[3] -> mm1
psrlw mm1, 4 // EL2.3: sum[0]/16 -> mm1
mov edx, dlen // EL0
// EL3: pixel 1 math -- (9*p[1] + 3*(p[0] + p[3]) + p[2]) / 16
psllw mm2, 3 // EL3.1: 8*p[1] -> mm2
paddw mm2, mm5 // EL3.2: 9*p[1] + 3*(p[0] + p[3]) + p[2] -> mm2
psrlw mm2, 4 // EL3.3: sum[1]/16 -> mm5
// EL2/3: store pixels 0 & 1
packuswb mm1, mm2 // EL2/3: pack into bytes
MOVNTQ (edi, mm1) // EL2/3: store 2 pixels
// EL4: pixel 2 math -- (9*p[2] + 3*(p[0] + p[3]) + p[1]) / 16
psllw mm3, 3 // EL4.1: 8*p[2] -> mm3
paddw mm3, mm5 // EL4.2: 9*p[2] + 3*(p[0] + p[3]) + p[1] -> mm3
psrlw mm3, 4 // EL4.3: sum[2]/16 -> mm3
// EL5: pixel 3 math -- (9*p[3] + 3*(p[1] + p[2]) + p[0]) / 16
psllw mm4, 3 // EL5.1: 8*p[3] -> mm4
paddw mm4, mm6 // EL5.2: 9*p[3] + 3*(p[1] + p[2]) + p[0] -> mm4
psrlw mm4, 4 // EL5.3: sum[3]/16 -> mm4
// EL4/5: store pixels 2 & 3
packuswb mm3, mm4 // EL4/5: pack into bytes
MOVNTQ (edi + edx*4, mm3) // EL4/5: store 2 pixels
}
}
// End MSVC_ASM
#elif defined(GCC_ASM)
// GCC inline assembly versions
#if defined(USE_MOVNTQ)
# define MOVNTQ(val, addr) "movntq " #val "," #addr
#else
# define MOVNTQ(val, addr) "movq " #val "," #addr
#endif
#if USE_PREFETCH == INTEL_PREFETCH
// using Intel SSE non-temporal prefetch
# define PREFETCH(addr) "prefetchnta " #addr
#elif USE_PREFETCH == AMD_PREFETCH
// using AMD 3DNOW! prefetch
# define PREFETCH(addr) "prefetch " #addr
#else
// no prefetch -- too bad for poor MMX-only souls
# define PREFETCH(addr)
#endif
static inline void
SCALE_(PlatInit) (void)
{
__asm__ (
// mm0 will be kept == 0 throughout
// 0 is needed for bytes->words unpack instructions
"pxor %%mm0, %%mm0 \n\t"
// mm5-mm7 contains RGB->YUV mults
"movq %0, %%mm7 \n\t"
"movq %1, %%mm6 \n\t"
"movq %2, %%mm5 \n\t"
#ifdef USE_YUV_LOOKUP
// mm6 contains RGB888->555 shuffle mult
"movq %3, %%mm6 \n\t"
// mm5 contains DiffYUV threshold
"movq %4, %%mm5 \n\t"
#endif
: /* nothing */
: /*0*/"m" (mmx_Y_mult), /*1*/"m" (mmx_U_mult), /*2*/"m" (mmx_V_mult)
, /*3*/"m" (mmx_888to555_mult), /*4*/"m" (mmx_YUV_threshold)
);
}
static inline void
SCALE_(PlatDone) (void)
{
// finish with MMX registers and yield them to FPU
__asm__ (
"emms \n\t"
: /* nothing */ : /* nothing */
);
}
static inline void
SCALE_(Prefetch) (const void* p)
{
__asm__ __volatile__ ("" PREFETCH (%0) : /*nothing*/ : "m" (p) );
}
// compute the RGB distance squared between 2 pixels
static inline int
SCALE_(GetRGBDelta) (Uint32 pix1, Uint32 pix2)
{
int res;
__asm__ (
// load pixels
"movd %1, %%mm1 \n\t"
"punpcklbw %%mm0, %%mm1 \n\t"
"movd %2, %%mm2 \n\t"
"punpcklbw %%mm0, %%mm2 \n\t"
// get the difference between RGBA components
"psubw %%mm2, %%mm1 \n\t"
// squared and sumed
"pmaddwd %%mm1, %%mm1 \n\t"
// finish suming the squares
"movq %%mm1, %%mm2 \n\t"
"punpckhdq %%mm0, %%mm2 \n\t"
"paddd %%mm2, %%mm1 \n\t"
// store result
"movd %%mm1, %0 \n\t"
: /*0*/"=r" (res)
: /*1*/"rm" (pix1), /*2*/"rm" (pix2)
);
return res;
}
// retrieve the Y (intensity) component of pixel's YUV
static inline int
SCALE_(GetPixY) (Uint32 pix)
{
int ret;
__asm__ (
// load pixel
"movd %1, %%mm1 \n\t"
"punpcklbw %%mm0, %%mm1 \n\t"
// process
"pmaddwd %2, %%mm1 \n\t" // R,G,B * Yvec
"movq %%mm1, %%mm2 \n\t" // finish suming
"punpckhdq %%mm0, %%mm2 \n\t" // ditto
"paddd %%mm2, %%mm1 \n\t" // ditto
// store index
"movd %%mm1, %0 \n\t"
: /*0*/"=r" (ret)
: /*1*/"rm" (pix), /*2*/"m" (mmx_Y_mult)
);
return ret >> 14;
}
#ifdef USE_YUV_LOOKUP
// convert pixel RGB vector into YUV representation vector
static inline YUV_VECTOR
SCALE_(RGBtoYUV) (Uint32 pix)
{
int i;
__asm__ (
// convert RGB888 to 555
"movd %1, %%mm1 \n\t"
"punpcklbw %%mm0, %%mm1 \n\t"
"psrlw $3, %%mm1 \n\t" // 8->5 bit
"pmaddwd %2, %%mm1 \n\t" // shuffle into the right channel order
"movq %%mm1, %%mm2 \n\t" // finish shuffling
"punpckhdq %%mm0, %%mm2 \n\t" // ditto
"por %%mm2, %%mm1 \n\t" // ditto
"movd %%mm1, %0 \n\t"
: /*0*/"=r" (i)
: /*1*/"rm" (pix), /*2*/"m" (mmx_888to555_mult)
);
return RGB15_to_YUV[i];
}
// compare 2 pixels with respect to their YUV representations
// tolerance set by toler arg
// returns true: close; false: distant (-gt toler)
static inline bool
SCALE_(CmpYUV) (Uint32 pix1, Uint32 pix2, int toler)
{
int delta;
__asm__ (
"movd %1, %%mm1 \n\t"
"movd %2, %%mm3 \n\t"
// convert RGB888 to 555
// this is somewhat parallelized
"punpcklbw %%mm0, %%mm1 \n\t"
"psrlw $3, %%mm1 \n\t" // 8->5 bit
"punpcklbw %%mm0, %%mm3 \n\t"
"psrlw $3, %%mm3 \n\t" // 8->5 bit
"pmaddwd %4, %%mm1 \n\t" // shuffle into the right channel order
"movq %%mm1, %%mm2 \n\t" // finish shuffling
"pmaddwd %4, %%mm3 \n\t" // shuffle into the right channel order
"movq %%mm3, %%mm4 \n\t" // finish shuffling
"punpckhdq %%mm0, %%mm2 \n\t" // ditto
"por %%mm2, %%mm1 \n\t" // ditto
"punpckhdq %%mm0, %%mm4 \n\t" // ditto
"por %%mm4, %%mm3 \n\t" // ditto
// lookup the YUV vector
"movd %%mm1, %%eax \n\t"
"movd %%mm3, %%edx \n\t"
"movd %3(,%%eax,4), %%mm1 \n\t"
"movq %%mm1, %%mm4 \n\t"
"movd %3(,%%edx,4), %%mm2 \n\t"
// get abs difference between YUV components
#ifdef USE_PSADBW
// we can use PSADBW and save us some grief
"psadbw %%mm2, %%mm1 \n\t"
"movd %%mm1, %0 \n\t"
#else
// no PSADBW -- have to do it the hard way
"psubusb %%mm2, %%mm1 \n\t"
"psubusb %%mm4, %%mm2 \n\t"
"por %%mm2, %%mm1 \n\t"
// sum the differences
// technically, this produces a MAX diff of 510
// but we do not need anything bigger, currently
"movq %%mm1, %%mm2 \n\t"
"psrlq $8, %%mm2 \n\t"
"paddusb %%mm2, %%mm1 \n\t"
"psrlq $8, %%mm2 \n\t"
"paddusb %%mm2, %%mm1 \n\t"
// store intermediate delta
"movd %%mm1, %0 \n\t"
"andl $0xff, %0 \n\t"
#endif /* USE_PSADBW */
: /*0*/"=r" (delta)
: /*1*/"rm" (pix1), /*2*/"rm" (pix2),
/*3*/"m" (*RGB15_to_YUV), /*4*/"m" (mmx_888to555_mult)
: "%eax", "%edx"
);
return (delta << 1) <= toler;
}
#endif /* USE_YUV_LOOKUP */
// Check if 2 pixels are different with respect to their
// YUV representations
// returns 0: close; ~0: distant
static inline int
SCALE_(DiffYUV) (Uint32 yuv1, Uint32 yuv2)
{
sint32 ret;
__asm__ (
// load YUV pixels
"movd %1, %%mm1 \n\t"
"movq %%mm1, %%mm4 \n\t"
"movd %2, %%mm2 \n\t"
// abs difference between channels
"psubusb %%mm2, %%mm1 \n\t"
"psubusb %%mm4, %%mm2 \n\t"
"por %%mm2, %%mm1 \n\t"
// compare to threshold
"psubusb %3, %%mm1 \n\t"
"movd %%mm1, %%edx \n\t"
// transform eax to 0 or ~0
"xor %%eax, %%eax \n\t"
"or %%edx, %%edx \n\t"
"setz %%al \n\t"
"dec %%eax \n\t"
: /*0*/"=a" (ret)
: /*1*/"rm" (yuv1), /*2*/"rm" (yuv2),
/*3*/"m" (mmx_YUV_threshold)
: "%edx"
);
return ret;
}
// Bilinear weighted blend of four pixels
// Function produces 4 blended pixels (in 2x2 matrix) and writes them
// out to the surface
// Last version
static inline void
SCALE_(Blend_bilinear) (const Uint32* row0, const Uint32* row1,
Uint32* dst_p, Uint32 dlen)
{
__asm__ (
// EL0: load pixels
"movq %0, %%mm1 \n\t" // EL0
"movq %%mm1, %%mm2 \n\t" // EL0: p[1] -> mm2
PREFETCH (0x80%0) "\n\t"
"punpckhbw %%mm0, %%mm2 \n\t" // EL0: p[1] -> mm2
"punpcklbw %%mm0, %%mm1 \n\t" // EL0: p[0] -> mm1
"movq %1, %%mm3 \n\t"
"movq %%mm3, %%mm4 \n\t" // EL0: p[3] -> mm4
"movq %%mm2, %%mm6 \n\t" // EL1.1: p[1] -> mm6
PREFETCH (0x80%1) "\n\t"
"punpcklbw %%mm0, %%mm3 \n\t" // EL0: p[2] -> mm3
"movq %%mm1, %%mm5 \n\t" // EL1.1: p[0] -> mm5
"punpckhbw %%mm0, %%mm4 \n\t" // EL0: p[3] -> mm4
// EL1: cache p[0] + 3*(p[1] + p[2]) + p[3] in mm6
"paddw %%mm3, %%mm6 \n\t" // EL1.2: p[1] + p[2] -> mm6
// EL1: cache p[0] + p[1] + p[2] + p[3] in mm7
"movq %%mm6, %%mm7 \n\t" // EL1.3: p[1] + p[2] -> mm7
// EL1: cache p[1] + 3*(p[0] + p[3]) + p[2] in mm5
"paddw %%mm4, %%mm5 \n\t" // EL1.2: p[0] + p[3] -> mm5
"psllw $1, %%mm6 \n\t" // EL1.4: 2*(p[1] + p[2]) -> mm6
"paddw %%mm5, %%mm7 \n\t" // EL1.4: sum(p[]) -> mm7
"psllw $1, %%mm5 \n\t" // EL1.5: 2*(p[0] + p[3]) -> mm5
"paddw %%mm7, %%mm6 \n\t" // EL1.5: p[0] + 3*(p[1] + p[2]) + p[3] -> mm6
"paddw %%mm7, %%mm5 \n\t" // EL1.6: p[1] + 3*(p[0] + p[3]) + p[2] -> mm5
// EL2: pixel 0 math -- (9*p[0] + 3*(p[1] + p[2]) + p[3]) / 16
"psllw $3, %%mm1 \n\t" // EL2.1: 8*p[0] -> mm1
"paddw %%mm6, %%mm1 \n\t" // EL2.2: 9*p[0] + 3*(p[1] + p[2]) + p[3] -> mm1
"psrlw $4, %%mm1 \n\t" // EL2.3: sum[0]/16 -> mm1
// EL3: pixel 1 math -- (9*p[1] + 3*(p[0] + p[3]) + p[2]) / 16
"psllw $3, %%mm2 \n\t" // EL3.1: 8*p[1] -> mm2
"paddw %%mm5, %%mm2 \n\t" // EL3.2: 9*p[1] + 3*(p[0] + p[3]) + p[2] -> mm5
"psrlw $4, %%mm2 \n\t" // EL3.3: sum[1]/16 -> mm5
// EL2/4: store pixels 0 & 1
"packuswb %%mm2, %%mm1 \n\t" // EL2/4: pack into bytes
MOVNTQ (%%mm1, (%2)) "\n\t" // EL2/4: store 2 pixels
// EL4: pixel 2 math -- (9*p[2] + 3*(p[0] + p[3]) + p[1]) / 16
"psllw $3, %%mm3 \n\t" // EL4.1: 8*p[2] -> mm3
"paddw %%mm5, %%mm3 \n\t" // EL4.2: 9*p[2] + 3*(p[0] + p[3]) + p[1] -> mm3
"psrlw $4, %%mm3 \n\t" // EL4.3: sum[2]/16 -> mm3
// EL5: pixel 3 math -- (9*p[3] + 3*(p[1] + p[2]) + p[0]) / 16
"psllw $3, %%mm4 \n\t" // EL5.1: 8*p[3] -> mm4
"paddw %%mm6, %%mm4 \n\t" // EL5.2: 9*p[3] + 3*(p[1] + p[2]) + p[0] -> mm4
"psrlw $4, %%mm4 \n\t" // EL5.3: sum[3]/16 -> mm4
// EL4/5: store pixels 2 & 3
"packuswb %%mm4, %%mm3 \n\t" // EL4/5: pack into bytes
MOVNTQ (%%mm3, (%2,%3,4)) "\n\t" // EL4/5: store 2 pixels
: /* nothing */
: /*0*/"m" (*row0), /*1*/"m" (*row1), /*2*/"r" (dst_p), /*3*/"r" (dlen)
: "memory"
);
}
#endif // GCC_ASM
#endif /* SCALEMMX_H_ */
+286
View File
@@ -0,0 +1,286 @@
/*
* This program is free software; you can redistribute it and/or modify
* it under the terms of the GNU General Public License as published by
* the Free Software Foundation; either version 2 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU General Public License for more details.
*
* You should have received a copy of the GNU General Public License
* along with this program; if not, write to the Free Software
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
*/
#ifdef GFXMODULE_SDL
#include "types.h"
#include "libs/graphics/sdl/sdl_common.h"
#include SDL_INCLUDE(sdl_cpuinfo.h)
#include "libs/platform.h"
#include "scalers.h"
#include "scaleint.h"
#include "2xscalers.h"
#ifdef MMX_ASM
# include "2xscalers_mmx.h"
#endif
typedef enum
{
SCALEPLAT_NULL = PLATFORM_NULL,
SCALEPLAT_C = PLATFORM_C,
SCALEPLAT_MMX = PLATFORM_MMX,
SCALEPLAT_SSE = PLATFORM_SSE,
SCALEPLAT_3DNOW = PLATFORM_3DNOW,
SCALEPLAT_ALTIVEC = PLATFORM_ALTIVEC,
SCALEPLAT_C_RGBA,
SCALEPLAT_C_BGRA,
SCALEPLAT_C_ARGB,
SCALEPLAT_C_ABGR,
} Scale_PlatType_t;
// RGB -> YUV transformation
// the RGB vector is multiplied by the transformation matrix
// to get the YUV vector
#if 0
// original table -- not used
const int YUV_matrix[3][3] =
{
/* Y U V */
/* R */ {0.2989, -0.1687, 0.5000},
/* G */ {0.5867, -0.3312, -0.4183},
/* B */ {0.1144, 0.5000, -0.0816}
};
#else
// scaled up by a 2^14 factor, with Y doubled
const int YUV_matrix[3][3] =
{
/* Y U V */
/* R */ { 9794, -2764, 8192},
/* G */ {19224, -5428, -6853},
/* B */ { 3749, 8192, -1339}
};
#endif
// pre-computed transformations for 8 bits per channel
int RGB_to_YUV[/*RGB*/ 3][/*YUV*/ 3][ /*mult-res*/ 256];
sint16 dRGB_to_dYUV[/*RGB*/ 3][/*YUV*/ 3][ /*mult-res*/ 512];
// pre-computed transformations for RGB555
YUV_VECTOR RGB15_to_YUV[0x8000];
PLATFORM_TYPE force_platform = PLATFORM_NULL;
Scale_PlatType_t Scale_Platform = SCALEPLAT_NULL;
// pre-compute the RGB->YUV transformations
void
Scale_Init (void)
{
int i1, i2, i3;
for (i1 = 0; i1 < 3; i1++) // enum R,G,B
for (i2 = 0; i2 < 3; i2++) // enum Y,U,V
for (i3 = 0; i3 < 256; i3++) // enum possible channel vals
{
RGB_to_YUV[i1][i2][i3] =
(YUV_matrix[i1][i2] * i3) >> 14;
}
for (i1 = 0; i1 < 3; i1++) // enum R,G,B
for (i2 = 0; i2 < 3; i2++) // enum Y,U,V
for (i3 = -255; i3 < 256; i3++) // enum possible channel delta vals
{
dRGB_to_dYUV[i1][i2][i3 + 255] =
(YUV_matrix[i1][i2] * i3) >> 14;
}
for (i1 = 0; i1 < 32; ++i1)
for (i2 = 0; i2 < 32; ++i2)
for (i3 = 0; i3 < 32; ++i3)
{
int y, u, v;
// adding upper bits halved for error correction
int r = (i1 << 3) | (i1 >> 3);
int g = (i2 << 3) | (i2 >> 3);
int b = (i3 << 3) | (i3 >> 3);
y = ( r * YUV_matrix[YUV_XFORM_R][YUV_XFORM_Y]
+ g * YUV_matrix[YUV_XFORM_G][YUV_XFORM_Y]
+ b * YUV_matrix[YUV_XFORM_B][YUV_XFORM_Y]
) >> 15; // we dont need Y doubled, need Y to fit 8 bits
// U and V are half the importance of Y
u = 64+(( r * YUV_matrix[YUV_XFORM_R][YUV_XFORM_U]
+ g * YUV_matrix[YUV_XFORM_G][YUV_XFORM_U]
+ b * YUV_matrix[YUV_XFORM_B][YUV_XFORM_U]
) >> 15); // halved
v = 64+(( r * YUV_matrix[YUV_XFORM_R][YUV_XFORM_V]
+ g * YUV_matrix[YUV_XFORM_G][YUV_XFORM_V]
+ b * YUV_matrix[YUV_XFORM_B][YUV_XFORM_V]
) >> 15); // halved
RGB15_to_YUV[(i1 << 10) | (i2 << 5) | i3] = (y << 16) | (u << 8) | v;
}
}
// expands the given rectangle in all directions by 'expansion'
// guarded by 'limits'
void
Scale_ExpandRect (SDL_Rect* rect, int expansion, const SDL_Rect* limits)
{
if (rect->x - expansion >= limits->x)
{
rect->w += expansion;
rect->x -= expansion;
}
else
{
rect->w += rect->x - limits->x;
rect->x = limits->x;
}
if (rect->y - expansion >= limits->y)
{
rect->h += expansion;
rect->y -= expansion;
}
else
{
rect->h += rect->y - limits->y;
rect->y = limits->y;
}
if (rect->x + rect->w + expansion <= limits->w)
rect->w += expansion;
else
rect->w = limits->w - rect->x;
if (rect->y + rect->h + expansion <= limits->h)
rect->h += expansion;
else
rect->h = limits->h - rect->y;
}
// Platform+Scaler function lookups
typedef struct
{
Scale_PlatType_t platform;
const Scale_FuncDef_t* funcdefs;
} Scale_PlatDef_t;
const static Scale_PlatDef_t
Scale_PlatDefs[] =
{
#if defined(MMX_ASM)
{SCALEPLAT_SSE, Scale_SSE_Functions},
{SCALEPLAT_3DNOW, Scale_3DNow_Functions},
{SCALEPLAT_MMX, Scale_MMX_Functions},
#endif /* MMX_ASM */
// Default
{SCALEPLAT_NULL, Scale_C_Functions}
};
TFB_ScaleFunc
Scale_PrepPlatform (int flags, const SDL_PixelFormat* fmt)
{
const Scale_PlatDef_t* pdef;
const Scale_FuncDef_t* fdef;
(void)flags;
Scale_Platform = SCALEPLAT_NULL;
// XXX: Hack to test some code
Scale_MMX_PrepPlatform (fmt);
// first match wins
// add better platform techs to the top
#ifdef MMX_ASM
if ( (!force_platform && (SDL_HasSSE () || SDL_HasMMXExt ()))
|| force_platform == SCALEPLAT_SSE)
{
fprintf (stderr, "Screen scalers are using SSE/MMX-Ext/MMX code\n");
Scale_Platform = SCALEPLAT_SSE;
Scale_SSE_PrepPlatform (fmt);
}
else
if ( (!force_platform && SDL_HasAltiVec ())
|| force_platform == SCALEPLAT_ALTIVEC)
{
fprintf (stderr, "Screen scalers would use AltiVec code "
"if someone actually wrote it\n");
//Scale_Platform = SCALEPLAT_ALTIVEC;
}
else
if ( (!force_platform && SDL_Has3DNow ())
|| force_platform == SCALEPLAT_3DNOW)
{
fprintf (stderr, "Screen scalers are using 3DNow/MMX code\n");
Scale_Platform = SCALEPLAT_3DNOW;
Scale_3DNow_PrepPlatform (fmt);
}
else
if ( (!force_platform && SDL_HasMMX ())
|| force_platform == SCALEPLAT_MMX)
{
fprintf (stderr, "Screen scalers are using MMX code\n");
Scale_Platform = SCALEPLAT_MMX;
Scale_MMX_PrepPlatform (fmt);
}
#endif
if (Scale_Platform == SCALEPLAT_NULL)
{ // Plain C versions
if (fmt->Rmask == 0xff000000)
Scale_Platform = SCALEPLAT_C_RGBA;
else if (fmt->Rmask == 0x00ff0000)
Scale_Platform = SCALEPLAT_C_ARGB;
else if (fmt->Rmask == 0x0000ff00)
Scale_Platform = SCALEPLAT_C_BGRA;
else if (fmt->Rmask == 0x000000ff)
Scale_Platform = SCALEPLAT_C_ABGR;
else
{ // use slowest default
fprintf (stderr, "Scale_PrepPlatform(): "
"unknown Red mask (0x%08x)\n", fmt->Rmask);
Scale_Platform = SCALEPLAT_C;
}
if (Scale_Platform == SCALEPLAT_C)
fprintf (stderr, "Screen scalers are using slow generic C code\n");
else
fprintf (stderr, "Screen scalers are using optimized C code\n");
}
// Lookup the scaling function
// First find the right platform
for (pdef = Scale_PlatDefs;
pdef->platform != Scale_Platform && pdef->platform != SCALEPLAT_NULL;
++pdef)
;
// Next find the right function
for (fdef = pdef->funcdefs;
(flags & fdef->flag) != fdef->flag;
++fdef)
;
return fdef->func;
}
#endif
+27
View File
@@ -0,0 +1,27 @@
/*
* This program is free software; you can redistribute it and/or modify
* it under the terms of the GNU General Public License as published by
* the Free Software Foundation; either version 2 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU General Public License for more details.
*
* You should have received a copy of the GNU General Public License
* along with this program; if not, write to the Free Software
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
*/
#ifndef SCALERS_H_
#define SCALERS_H_
void Scale_Init (void);
typedef void (* TFB_ScaleFunc) (SDL_Surface *src, SDL_Surface *dst,
SDL_Rect *r);
TFB_ScaleFunc Scale_PrepPlatform (int flags, const SDL_PixelFormat* fmt);
#endif /* SCALERS_H_ */
+158
View File
@@ -0,0 +1,158 @@
/*
* This program is free software; you can redistribute it and/or modify
* it under the terms of the GNU General Public License as published by
* the Free Software Foundation; either version 2 of the License, or
* (at your option) any later version.
*
* This program is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
* GNU General Public License for more details.
*
* You should have received a copy of the GNU General Public License
* along with this program; if not, write to the Free Software
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
*/
// Core algorithm of the Triscan screen scaler (based on Scale2x)
// (for scale2x please see http://scale2x.sf.net)
// Template
// When this file is built standalone is produces a plain C version
// Also #included by 2xscalers_mmx.c for an MMX version
#ifdef GFXMODULE_SDL
#include "libs/graphics/sdl/sdl_common.h"
#include "types.h"
#include "scalers.h"
#include "scaleint.h"
#include "2xscalers.h"
// Triscan scaling to 2x
// derivative of scale2x -- scale2x.sf.net
// The name expands to either
// Scale_TriScanFilter (for plain C) or
// Scale_MMX_TriScanFilter (for MMX)
// [others when platforms are added]
void
SCALE_(TriScanFilter) (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r)
{
int x, y;
const int w = src->w, h = src->h;
int xend, yend;
int dsrc, ddst;
SDL_Rect *region = r;
SDL_Rect limits;
SDL_PixelFormat *fmt = dst->format;
const int sp = src->pitch, dp = dst->pitch;
const int bpp = fmt->BytesPerPixel;
const int slen = sp / bpp, dlen = dp / bpp;
// for clarity purposes, the 'pixels' array here is transposed
Uint32 pixels[3][3];
Uint32 *src_p = (Uint32 *)src->pixels;
Uint32 *dst_p = (Uint32 *)dst->pixels;
int prevline, nextline;
// these macros are for clarity; they make the current pixel (0,0)
// and allow to access pixels in all directions
#define PIX(x, y) (pixels[1 + (x)][1 + (y)])
#define TRISCAN_YUV_MED 100
// medium tolerance pixel comparison
#define TRISCAN_CMPYUV(p1, p2) \
(PIX p1 == PIX p2 || SCALE_CMPYUV (PIX p1, PIX p2, TRISCAN_YUV_MED))
SCALE_(PlatInit) ();
// expand updated region if necessary
// pixels neighbooring the updated region may
// change as a result of updates
limits.x = 0;
limits.y = 0;
limits.w = src->w;
limits.h = src->h;
Scale_ExpandRect (region, 1, &limits);
xend = region->x + region->w;
yend = region->y + region->h;
dsrc = slen - region->w;
ddst = (dlen - region->w) * 2;
// move ptrs to the first updated pixel
src_p += slen * region->y + region->x;
dst_p += (dlen * region->y + region->x) * 2;
for (y = region->y; y < yend; ++y, dst_p += ddst, src_p += dsrc)
{
if (y > 0)
prevline = -slen;
else
prevline = 0;
if (y < h - 1)
nextline = slen;
else
nextline = 0;
// prime the (tiny) sliding-window pixel arrays
PIX( 1, 0) = src_p[0];
if (region->x > 0)
PIX( 0, 0) = src_p[-1];
else
PIX( 0, 0) = PIX( 1, 0);
for (x = region->x; x < xend; ++x, ++src_p, dst_p += 2)
{
// slide the window
PIX(-1, 0) = PIX( 0, 0);
PIX( 0, -1) = src_p[prevline];
PIX( 0, 0) = PIX( 1, 0);
PIX( 0, 1) = src_p[nextline];
if (x < w - 1)
PIX( 1, 0) = src_p[1];
else
PIX( 1, 0) = PIX( 0, 0);
if (!TRISCAN_CMPYUV (( 0, -1), ( 0, 1)) &&
!TRISCAN_CMPYUV ((-1, 0), ( 1, 0)))
{
if (TRISCAN_CMPYUV ((-1, 0), ( 0, -1)))
dst_p[0] = Scale_Blend_11 (PIX(-1, 0), PIX(0, -1));
else
dst_p[0] = PIX(0, 0);
if (TRISCAN_CMPYUV (( 1, 0), ( 0, -1)))
dst_p[1] = Scale_Blend_11 (PIX(1, 0), PIX(0, -1));
else
dst_p[1] = PIX(0, 0);
if (TRISCAN_CMPYUV ((-1, 0), ( 0, 1)))
dst_p[dlen] = Scale_Blend_11 (PIX(-1, 0), PIX(0, 1));
else
dst_p[dlen] = PIX(0, 0);
if (TRISCAN_CMPYUV (( 1, 0), ( 0, 1)))
dst_p[dlen+1] = Scale_Blend_11 (PIX(1, 0), PIX(0, 1));
else
dst_p[dlen+1] = PIX(0, 0);
}
else
{
dst_p[0] = PIX(0, 0);
dst_p[1] = PIX(0, 0);
dst_p[dlen] = PIX(0, 0);
dst_p[dlen+1] = PIX(0, 0);
}
}
}
SCALE_(PlatDone) ();
}
#endif /* GFXMODULE_SDL */