Accelerated 32bpp scalers, Preview 1
git-svn-id: svn://svn.code.sf.net/p/sc2/code/trunk@1943 8092fc87-c524-0410-9efc-e669fe64eaf9
This commit is contained in:
+104
@@ -0,0 +1,104 @@
|
|||||||
|
/*
|
||||||
|
* This program is free software; you can redistribute it and/or modify
|
||||||
|
* it under the terms of the GNU General Public License as published by
|
||||||
|
* the Free Software Foundation; either version 2 of the License, or
|
||||||
|
* (at your option) any later version.
|
||||||
|
*
|
||||||
|
* This program is distributed in the hope that it will be useful,
|
||||||
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
* GNU General Public License for more details.
|
||||||
|
*
|
||||||
|
* You should have received a copy of the GNU General Public License
|
||||||
|
* along with this program; if not, write to the Free Software
|
||||||
|
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||||
|
*/
|
||||||
|
|
||||||
|
#ifdef GFXMODULE_SDL
|
||||||
|
|
||||||
|
#include "port.h"
|
||||||
|
#include "libs/platform.h"
|
||||||
|
|
||||||
|
#if defined(MMX_ASM)
|
||||||
|
|
||||||
|
#include "libs/graphics/sdl/sdl_common.h"
|
||||||
|
#include "types.h"
|
||||||
|
#include "scalers.h"
|
||||||
|
#include "scaleint.h"
|
||||||
|
#include "2xscalers.h"
|
||||||
|
#include "2xscalers_mmx.h"
|
||||||
|
|
||||||
|
// 3DNow! name for all functions
|
||||||
|
#undef SCALE_
|
||||||
|
#define SCALE_(name) Scale ## _3DNow_ ## name
|
||||||
|
|
||||||
|
// Tell them which opcodes we want to support
|
||||||
|
#undef USE_MOVNTQ
|
||||||
|
#define USE_PREFETCH AMD_PREFETCH
|
||||||
|
#undef USE_PSADBW
|
||||||
|
// Bring in inline asm functions
|
||||||
|
#include "scalemmx.h"
|
||||||
|
|
||||||
|
|
||||||
|
// Scaler function lookup table
|
||||||
|
//
|
||||||
|
const Scale_FuncDef_t
|
||||||
|
Scale_3DNow_Functions[] =
|
||||||
|
{
|
||||||
|
{TFB_GFXFLAGS_SCALE_BILINEAR, Scale_3DNow_BilinearFilter},
|
||||||
|
{TFB_GFXFLAGS_SCALE_BIADAPT, Scale_BiAdaptFilter},
|
||||||
|
{TFB_GFXFLAGS_SCALE_BIADAPTADV, Scale_MMX_BiAdaptAdvFilter},
|
||||||
|
{TFB_GFXFLAGS_SCALE_TRISCAN, Scale_MMX_TriScanFilter},
|
||||||
|
{TFB_GFXFLAGS_SCALE_HQXX, Scale_MMX_HqFilter},
|
||||||
|
// Default
|
||||||
|
{0, Scale_3DNow_Nearest}
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
void
|
||||||
|
Scale_3DNow_PrepPlatform (const SDL_PixelFormat* fmt)
|
||||||
|
{
|
||||||
|
Scale_MMX_PrepPlatform (fmt);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Nearest Neighbor scaling to 2x
|
||||||
|
// void Scale_3DNow_Nearest (SDL_Surface *src,
|
||||||
|
// SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
|
||||||
|
#include "nearest2x.c"
|
||||||
|
|
||||||
|
|
||||||
|
// Bilinear scaling to 2x
|
||||||
|
// void Scale_3DNow_BilinearFilter (SDL_Surface *src,
|
||||||
|
// SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
|
||||||
|
#include "bilinear2x.c"
|
||||||
|
|
||||||
|
|
||||||
|
#if 0 && NO_IMPROVEMENT
|
||||||
|
|
||||||
|
// Advanced Biadapt scaling to 2x
|
||||||
|
// void Scale_3DNow_BiAdaptAdvFilter (SDL_Surface *src,
|
||||||
|
// SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
|
||||||
|
#include "biadv2x.c"
|
||||||
|
|
||||||
|
|
||||||
|
// Triscan scaling to 2x
|
||||||
|
// derivative of scale2x -- scale2x.sf.net
|
||||||
|
// void Scale_3DNow_TriScanFilter (SDL_Surface *src,
|
||||||
|
// SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
|
||||||
|
#include "triscan2x.c"
|
||||||
|
|
||||||
|
// Hq2x scaling
|
||||||
|
// (adapted from 'hq2x' by Maxim Stepin -- www.hiend3d.com/hq2x.html)
|
||||||
|
// void Scale_3DNow_HqFilter (SDL_Surface *src,
|
||||||
|
// SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
|
||||||
|
#include "hq2x.c"
|
||||||
|
|
||||||
|
#endif /* NO_IMPROVEMENT */
|
||||||
|
|
||||||
|
#endif /* MMX_ASM */
|
||||||
|
#endif /* GFXMODULE_SDL */
|
||||||
+138
@@ -0,0 +1,138 @@
|
|||||||
|
/*
|
||||||
|
* This program is free software; you can redistribute it and/or modify
|
||||||
|
* it under the terms of the GNU General Public License as published by
|
||||||
|
* the Free Software Foundation; either version 2 of the License, or
|
||||||
|
* (at your option) any later version.
|
||||||
|
*
|
||||||
|
* This program is distributed in the hope that it will be useful,
|
||||||
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
* GNU General Public License for more details.
|
||||||
|
*
|
||||||
|
* You should have received a copy of the GNU General Public License
|
||||||
|
* along with this program; if not, write to the Free Software
|
||||||
|
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||||
|
*/
|
||||||
|
|
||||||
|
#ifdef GFXMODULE_SDL
|
||||||
|
|
||||||
|
#include "port.h"
|
||||||
|
#include "libs/platform.h"
|
||||||
|
|
||||||
|
#if defined(MMX_ASM)
|
||||||
|
|
||||||
|
#include "libs/graphics/sdl/sdl_common.h"
|
||||||
|
#include "types.h"
|
||||||
|
#include "scalers.h"
|
||||||
|
#include "scaleint.h"
|
||||||
|
#include "2xscalers.h"
|
||||||
|
#include "2xscalers_mmx.h"
|
||||||
|
|
||||||
|
// MMX name for all functions
|
||||||
|
#undef SCALE_
|
||||||
|
#define SCALE_(name) Scale ## _MMX_ ## name
|
||||||
|
|
||||||
|
// Tell them which opcodes we want to support
|
||||||
|
#undef USE_MOVNTQ
|
||||||
|
#undef USE_PREFETCH
|
||||||
|
#undef USE_PSADBW
|
||||||
|
// And Bring in inline asm functions
|
||||||
|
#include "scalemmx.h"
|
||||||
|
|
||||||
|
|
||||||
|
// Scaler function lookup table
|
||||||
|
//
|
||||||
|
const Scale_FuncDef_t
|
||||||
|
Scale_MMX_Functions[] =
|
||||||
|
{
|
||||||
|
{TFB_GFXFLAGS_SCALE_BILINEAR, Scale_MMX_BilinearFilter},
|
||||||
|
{TFB_GFXFLAGS_SCALE_BIADAPT, Scale_BiAdaptFilter},
|
||||||
|
{TFB_GFXFLAGS_SCALE_BIADAPTADV, Scale_MMX_BiAdaptAdvFilter},
|
||||||
|
{TFB_GFXFLAGS_SCALE_TRISCAN, Scale_MMX_TriScanFilter},
|
||||||
|
{TFB_GFXFLAGS_SCALE_HQXX, Scale_MMX_HqFilter},
|
||||||
|
// Default
|
||||||
|
{0, Scale_MMX_Nearest}
|
||||||
|
};
|
||||||
|
|
||||||
|
// MMX transformation multipliers
|
||||||
|
Uint64 mmx_888to555_mult;
|
||||||
|
Uint64 mmx_Y_mult;
|
||||||
|
Uint64 mmx_U_mult;
|
||||||
|
Uint64 mmx_V_mult;
|
||||||
|
// Uint64 mmx_YUV_threshold = 0x00300706; original hq2x threshold
|
||||||
|
//Uint64 mmx_YUV_threshold = 0x0030100e;
|
||||||
|
Uint64 mmx_YUV_threshold = 0x0040120c;
|
||||||
|
|
||||||
|
void
|
||||||
|
Scale_MMX_PrepPlatform (const SDL_PixelFormat* fmt)
|
||||||
|
{
|
||||||
|
// prepare the channel-shuffle multiplier
|
||||||
|
mmx_888to555_mult = ((Uint64)0x0400) << (fmt->Rshift * 2)
|
||||||
|
| ((Uint64)0x0020) << (fmt->Gshift * 2)
|
||||||
|
| ((Uint64)0x0001) << (fmt->Bshift * 2);
|
||||||
|
|
||||||
|
// prepare the RGB->YUV multipliers
|
||||||
|
mmx_Y_mult = ((Uint64)(uint16)YUV_matrix[YUV_XFORM_R][YUV_XFORM_Y])
|
||||||
|
<< (fmt->Rshift * 2)
|
||||||
|
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_G][YUV_XFORM_Y])
|
||||||
|
<< (fmt->Gshift * 2)
|
||||||
|
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_B][YUV_XFORM_Y])
|
||||||
|
<< (fmt->Bshift * 2);
|
||||||
|
|
||||||
|
mmx_U_mult = ((Uint64)(uint16)YUV_matrix[YUV_XFORM_R][YUV_XFORM_U])
|
||||||
|
<< (fmt->Rshift * 2)
|
||||||
|
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_G][YUV_XFORM_U])
|
||||||
|
<< (fmt->Gshift * 2)
|
||||||
|
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_B][YUV_XFORM_U])
|
||||||
|
<< (fmt->Bshift * 2);
|
||||||
|
|
||||||
|
mmx_V_mult = ((Uint64)(uint16)YUV_matrix[YUV_XFORM_R][YUV_XFORM_V])
|
||||||
|
<< (fmt->Rshift * 2)
|
||||||
|
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_G][YUV_XFORM_V])
|
||||||
|
<< (fmt->Gshift * 2)
|
||||||
|
| ((Uint64)(uint16)YUV_matrix[YUV_XFORM_B][YUV_XFORM_V])
|
||||||
|
<< (fmt->Bshift * 2);
|
||||||
|
|
||||||
|
mmx_YUV_threshold = (SCALE_DIFFYUV_TY << 16) | (SCALE_DIFFYUV_TU << 8)
|
||||||
|
| SCALE_DIFFYUV_TV;
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
// Nearest Neighbor scaling to 2x
|
||||||
|
// void Scale_MMX_Nearest (SDL_Surface *src,
|
||||||
|
// SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
|
||||||
|
#include "nearest2x.c"
|
||||||
|
|
||||||
|
|
||||||
|
// Bilinear scaling to 2x
|
||||||
|
// void Scale_MMX_BilinearFilter (SDL_Surface *src,
|
||||||
|
// SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
|
||||||
|
#include "bilinear2x.c"
|
||||||
|
|
||||||
|
|
||||||
|
// Advanced Biadapt scaling to 2x
|
||||||
|
// void Scale_MMX_BiAdaptAdvFilter (SDL_Surface *src,
|
||||||
|
// SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
|
||||||
|
#include "biadv2x.c"
|
||||||
|
|
||||||
|
|
||||||
|
// Triscan scaling to 2x
|
||||||
|
// derivative of 'scale2x' -- scale2x.sf.net
|
||||||
|
// void Scale_MMX_TriScanFilter (SDL_Surface *src,
|
||||||
|
// SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
|
||||||
|
#include "triscan2x.c"
|
||||||
|
|
||||||
|
// Hq2x scaling
|
||||||
|
// (adapted from 'hq2x' by Maxim Stepin -- www.hiend3d.com/hq2x.html)
|
||||||
|
// void Scale_MMX_HqFilter (SDL_Surface *src,
|
||||||
|
// SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
|
||||||
|
#include "hq2x.c"
|
||||||
|
|
||||||
|
|
||||||
|
#endif /* MMX_ASM */
|
||||||
|
#endif /* GFXMODULE_SDL */
|
||||||
+56
@@ -0,0 +1,56 @@
|
|||||||
|
/*
|
||||||
|
* This program is free software; you can redistribute it and/or modify
|
||||||
|
* it under the terms of the GNU General Public License as published by
|
||||||
|
* the Free Software Foundation; either version 2 of the License, or
|
||||||
|
* (at your option) any later version.
|
||||||
|
*
|
||||||
|
* This program is distributed in the hope that it will be useful,
|
||||||
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
* GNU General Public License for more details.
|
||||||
|
*
|
||||||
|
* You should have received a copy of the GNU General Public License
|
||||||
|
* along with this program; if not, write to the Free Software
|
||||||
|
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||||
|
*/
|
||||||
|
|
||||||
|
#ifndef _2XSCALERS_MMX_H_
|
||||||
|
#define _2XSCALERS_MMX_H_
|
||||||
|
|
||||||
|
// MMX versions
|
||||||
|
void Scale_MMX_PrepPlatform (const SDL_PixelFormat* fmt);
|
||||||
|
|
||||||
|
void Scale_MMX_Nearest (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||||
|
void Scale_MMX_BilinearFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||||
|
void Scale_MMX_BiAdaptAdvFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||||
|
void Scale_MMX_TriScanFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||||
|
void Scale_MMX_HqFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||||
|
|
||||||
|
extern const Scale_FuncDef_t Scale_MMX_Functions[];
|
||||||
|
|
||||||
|
|
||||||
|
// SSE (Intel)/MMX Ext (Athlon) versions
|
||||||
|
void Scale_SSE_PrepPlatform (const SDL_PixelFormat* fmt);
|
||||||
|
|
||||||
|
void Scale_SSE_Nearest (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||||
|
void Scale_SSE_BilinearFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||||
|
void Scale_SSE_BiAdaptAdvFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||||
|
void Scale_SSE_TriScanFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||||
|
void Scale_SSE_HqFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||||
|
|
||||||
|
extern const Scale_FuncDef_t Scale_SSE_Functions[];
|
||||||
|
|
||||||
|
|
||||||
|
// 3DNow (AMD K6/Athlon) versions
|
||||||
|
void Scale_3DNow_PrepPlatform (const SDL_PixelFormat* fmt);
|
||||||
|
|
||||||
|
void Scale_3DNow_Nearest (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||||
|
void Scale_3DNow_BilinearFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||||
|
void Scale_3DNow_BiAdaptAdvFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||||
|
void Scale_3DNow_TriScanFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||||
|
void Scale_3DNow_HqFilter (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r);
|
||||||
|
|
||||||
|
extern const Scale_FuncDef_t Scale_3DNow_Functions[];
|
||||||
|
|
||||||
|
|
||||||
|
#endif /* _2XSCALERS_MMX_H_ */
|
||||||
+102
@@ -0,0 +1,102 @@
|
|||||||
|
/*
|
||||||
|
* This program is free software; you can redistribute it and/or modify
|
||||||
|
* it under the terms of the GNU General Public License as published by
|
||||||
|
* the Free Software Foundation; either version 2 of the License, or
|
||||||
|
* (at your option) any later version.
|
||||||
|
*
|
||||||
|
* This program is distributed in the hope that it will be useful,
|
||||||
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
* GNU General Public License for more details.
|
||||||
|
*
|
||||||
|
* You should have received a copy of the GNU General Public License
|
||||||
|
* along with this program; if not, write to the Free Software
|
||||||
|
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||||
|
*/
|
||||||
|
|
||||||
|
#ifdef GFXMODULE_SDL
|
||||||
|
|
||||||
|
#include "port.h"
|
||||||
|
#include "libs/platform.h"
|
||||||
|
|
||||||
|
#if defined(MMX_ASM)
|
||||||
|
|
||||||
|
#include "libs/graphics/sdl/sdl_common.h"
|
||||||
|
#include "types.h"
|
||||||
|
#include "scalers.h"
|
||||||
|
#include "scaleint.h"
|
||||||
|
#include "2xscalers.h"
|
||||||
|
#include "2xscalers_mmx.h"
|
||||||
|
|
||||||
|
// SSE name for all functions
|
||||||
|
#undef SCALE_
|
||||||
|
#define SCALE_(name) Scale ## _SSE_ ## name
|
||||||
|
|
||||||
|
// Tell them which opcodes we want to support
|
||||||
|
#define USE_MOVNTQ
|
||||||
|
#define USE_PREFETCH INTEL_PREFETCH
|
||||||
|
#define USE_PSADBW
|
||||||
|
// Bring in inline asm functions
|
||||||
|
#include "scalemmx.h"
|
||||||
|
|
||||||
|
|
||||||
|
// Scaler function lookup table
|
||||||
|
//
|
||||||
|
const Scale_FuncDef_t
|
||||||
|
Scale_SSE_Functions[] =
|
||||||
|
{
|
||||||
|
{TFB_GFXFLAGS_SCALE_BILINEAR, Scale_SSE_BilinearFilter},
|
||||||
|
{TFB_GFXFLAGS_SCALE_BIADAPT, Scale_BiAdaptFilter},
|
||||||
|
{TFB_GFXFLAGS_SCALE_BIADAPTADV, Scale_SSE_BiAdaptAdvFilter},
|
||||||
|
{TFB_GFXFLAGS_SCALE_TRISCAN, Scale_SSE_TriScanFilter},
|
||||||
|
{TFB_GFXFLAGS_SCALE_HQXX, Scale_MMX_HqFilter},
|
||||||
|
// Default
|
||||||
|
{0, Scale_SSE_Nearest}
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
void
|
||||||
|
Scale_SSE_PrepPlatform (const SDL_PixelFormat* fmt)
|
||||||
|
{
|
||||||
|
Scale_MMX_PrepPlatform (fmt);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Nearest Neighbor scaling to 2x
|
||||||
|
// void Scale_SSE_Nearest (SDL_Surface *src,
|
||||||
|
// SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
|
||||||
|
#include "nearest2x.c"
|
||||||
|
|
||||||
|
|
||||||
|
// Bilinear scaling to 2x
|
||||||
|
// void Scale_SSE_BilinearFilter (SDL_Surface *src,
|
||||||
|
// SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
|
||||||
|
#include "bilinear2x.c"
|
||||||
|
|
||||||
|
|
||||||
|
// Advanced Biadapt scaling to 2x
|
||||||
|
// void Scale_SSE_BiAdaptAdvFilter (SDL_Surface *src,
|
||||||
|
// SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
|
||||||
|
#include "biadv2x.c"
|
||||||
|
|
||||||
|
|
||||||
|
// Triscan scaling to 2x
|
||||||
|
// derivative of scale2x -- scale2x.sf.net
|
||||||
|
// void Scale_SSE_TriScanFilter (SDL_Surface *src,
|
||||||
|
// SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
|
||||||
|
#include "triscan2x.c"
|
||||||
|
|
||||||
|
#if 0 && NO_IMPROVEMENT
|
||||||
|
// Hq2x scaling
|
||||||
|
// (adapted from 'hq2x' by Maxim Stepin -- www.hiend3d.com/hq2x.html)
|
||||||
|
// void Scale_SSE_HqFilter (SDL_Surface *src,
|
||||||
|
// SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
|
||||||
|
#include "hq2x.c"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#endif /* MMX_ASM */
|
||||||
|
#endif /* GFXMODULE_SDL */
|
||||||
+21
@@ -0,0 +1,21 @@
|
|||||||
|
/*
|
||||||
|
* This program is free software; you can redistribute it and/or modify
|
||||||
|
* it under the terms of the GNU General Public License as published by
|
||||||
|
* the Free Software Foundation; either version 2 of the License, or
|
||||||
|
* (at your option) any later version.
|
||||||
|
*
|
||||||
|
* This program is distributed in the hope that it will be useful,
|
||||||
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
* GNU General Public License for more details.
|
||||||
|
*
|
||||||
|
* You should have received a copy of the GNU General Public License
|
||||||
|
* along with this program; if not, write to the Free Software
|
||||||
|
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||||
|
*/
|
||||||
|
|
||||||
|
#ifndef _2XSCALERS_SSE_H_
|
||||||
|
#define _2XSCALERS_SSE_H_
|
||||||
|
|
||||||
|
|
||||||
|
#endif /* _2XSCALERS_SSE_H_ */
|
||||||
Executable
+535
@@ -0,0 +1,535 @@
|
|||||||
|
/*
|
||||||
|
* Portions Copyright (C) 2003-2005 Alex Volkov (codepro@usa.net)
|
||||||
|
*
|
||||||
|
* This program is free software; you can redistribute it and/or modify
|
||||||
|
* it under the terms of the GNU General Public License as published by
|
||||||
|
* the Free Software Foundation; either version 2 of the License, or
|
||||||
|
* (at your option) any later version.
|
||||||
|
*
|
||||||
|
* This program is distributed in the hope that it will be useful,
|
||||||
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
* GNU General Public License for more details.
|
||||||
|
*
|
||||||
|
* You should have received a copy of the GNU General Public License
|
||||||
|
* along with this program; if not, write to the Free Software
|
||||||
|
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||||
|
*/
|
||||||
|
|
||||||
|
// Core algorithm of the Advanced BiAdaptive screen scaler
|
||||||
|
// Template
|
||||||
|
// When this file is built standalone is produces a plain C version
|
||||||
|
// Also #included by 2xscalers_mmx.c for an MMX version
|
||||||
|
|
||||||
|
#ifdef GFXMODULE_SDL
|
||||||
|
|
||||||
|
#include "libs/graphics/sdl/sdl_common.h"
|
||||||
|
#include "types.h"
|
||||||
|
#include "scalers.h"
|
||||||
|
#include "scaleint.h"
|
||||||
|
#include "2xscalers.h"
|
||||||
|
|
||||||
|
|
||||||
|
// Advanced biadapt scaling to 2x
|
||||||
|
// The name expands to either
|
||||||
|
// Scale_BiAdaptAdvFilter (for plain C) or
|
||||||
|
// Scale_MMX_BiAdaptAdvFilter (for MMX)
|
||||||
|
// [others when platforms are added]
|
||||||
|
void
|
||||||
|
SCALE_(BiAdaptAdvFilter) (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
{
|
||||||
|
int x, y;
|
||||||
|
const int w = src->w, h = src->h;
|
||||||
|
int xend, yend;
|
||||||
|
int dsrc, ddst;
|
||||||
|
SDL_Rect *region = r;
|
||||||
|
SDL_Rect limits;
|
||||||
|
SDL_PixelFormat *fmt = dst->format;
|
||||||
|
const int sp = src->pitch, dp = dst->pitch;
|
||||||
|
const int bpp = fmt->BytesPerPixel;
|
||||||
|
const int slen = sp / bpp, dlen = dp / bpp;
|
||||||
|
// for clarity purposes, the 'pixels' array here is transposed
|
||||||
|
Uint32 pixels[4][4];
|
||||||
|
static int resolve_coord[][2] =
|
||||||
|
{
|
||||||
|
{0, -1}, {1, -1}, { 2, 0}, { 2, 1},
|
||||||
|
{1, 2}, {0, 2}, {-1, 1}, {-1, 0},
|
||||||
|
{100, 100} // term
|
||||||
|
};
|
||||||
|
Uint32 *src_p = (Uint32 *)src->pixels;
|
||||||
|
Uint32 *dst_p = (Uint32 *)dst->pixels;
|
||||||
|
|
||||||
|
// these macros are for clarity; they make the current pixel (0,0)
|
||||||
|
// and allow to access pixels in all directions
|
||||||
|
#define PIX(x, y) (pixels[1 + (x)][1 + (y)])
|
||||||
|
#define SRC(x, y) (src_p + (x) + ((y) * slen))
|
||||||
|
// commonly used operations, for clarity also
|
||||||
|
// others are defined at their respective bpp levels
|
||||||
|
#define BIADAPT_RGBHIGH 8000
|
||||||
|
#define BIADAPT_YUVLOW 30
|
||||||
|
#define BIADAPT_YUVMED 70
|
||||||
|
#define BIADAPT_YUVHIGH 130
|
||||||
|
|
||||||
|
// high tolerance pixel comparison
|
||||||
|
#define BIADAPT_CMPRGB_HIGH(p1, p2) \
|
||||||
|
(p1 == p2 || SCALE_CMPRGB (p1, p2) <= BIADAPT_RGBHIGH)
|
||||||
|
|
||||||
|
// low tolerance pixel comparison
|
||||||
|
#define BIADAPT_CMPYUV_LOW(p1, p2) \
|
||||||
|
(p1 == p2 || SCALE_CMPYUV (p1, p2, BIADAPT_YUVLOW))
|
||||||
|
// medium tolerance pixel comparison
|
||||||
|
#define BIADAPT_CMPYUV_MED(p1, p2) \
|
||||||
|
(p1 == p2 || SCALE_CMPYUV (p1, p2, BIADAPT_YUVMED))
|
||||||
|
// high tolerance pixel comparison
|
||||||
|
#define BIADAPT_CMPYUV_HIGH(p1, p2) \
|
||||||
|
(p1 == p2 || SCALE_CMPYUV (p1, p2, BIADAPT_YUVHIGH))
|
||||||
|
|
||||||
|
SCALE_(PlatInit) ();
|
||||||
|
|
||||||
|
// expand updated region if necessary
|
||||||
|
// pixels neighbooring the updated region may
|
||||||
|
// change as a result of updates
|
||||||
|
limits.x = 0;
|
||||||
|
limits.y = 0;
|
||||||
|
limits.w = src->w;
|
||||||
|
limits.h = src->h;
|
||||||
|
Scale_ExpandRect (region, 2, &limits);
|
||||||
|
|
||||||
|
xend = region->x + region->w;
|
||||||
|
yend = region->y + region->h;
|
||||||
|
dsrc = slen - region->w;
|
||||||
|
ddst = (dlen - region->w) * 2;
|
||||||
|
|
||||||
|
#define SCALE_GETPIX(p) ( *(Uint32 *)(p) )
|
||||||
|
#define SCALE_SETPIX(p, c) ( *(Uint32 *)(p) = (c) )
|
||||||
|
|
||||||
|
// move ptrs to the first updated pixel
|
||||||
|
src_p += slen * region->y + region->x;
|
||||||
|
dst_p += (dlen * region->y + region->x) * 2;
|
||||||
|
|
||||||
|
for (y = region->y; y < yend; ++y, dst_p += ddst, src_p += dsrc)
|
||||||
|
{
|
||||||
|
for (x = region->x; x < xend; ++x, ++src_p, ++dst_p)
|
||||||
|
{
|
||||||
|
// pixel equality counter
|
||||||
|
int cmatch;
|
||||||
|
|
||||||
|
// most pixels will fall into 'all 4 equal'
|
||||||
|
// pattern, so we check it first
|
||||||
|
cmatch = 0;
|
||||||
|
|
||||||
|
PIX (0, 0) = SCALE_GETPIX (SRC (0, 0));
|
||||||
|
|
||||||
|
SCALE_SETPIX (dst_p, PIX (0, 0));
|
||||||
|
|
||||||
|
if (y + 1 < h)
|
||||||
|
{
|
||||||
|
// check pixel below the current one
|
||||||
|
PIX (0, 1) = SCALE_GETPIX (SRC (0, 1));
|
||||||
|
|
||||||
|
if (PIX (0, 0) == PIX (0, 1))
|
||||||
|
{
|
||||||
|
SCALE_SETPIX (dst_p + dlen, PIX (0, 0));
|
||||||
|
cmatch |= 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
// last pixel in column - propagate
|
||||||
|
PIX (0, 1) = PIX (0, 0);
|
||||||
|
SCALE_SETPIX (dst_p + dlen, PIX (0, 0));
|
||||||
|
cmatch |= 1;
|
||||||
|
|
||||||
|
}
|
||||||
|
|
||||||
|
if (x + 1 < w)
|
||||||
|
{
|
||||||
|
// check pixel to the right from the current one
|
||||||
|
PIX (1, 0) = SCALE_GETPIX (SRC (1, 0));
|
||||||
|
|
||||||
|
if (PIX (0, 0) == PIX (1, 0))
|
||||||
|
{
|
||||||
|
SCALE_SETPIX (dst_p + 1, PIX (0, 0));
|
||||||
|
cmatch |= 2;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
// last pixel in row - propagate
|
||||||
|
PIX (1, 0) = PIX (0, 0);
|
||||||
|
SCALE_SETPIX (dst_p + 1, PIX (0, 0));
|
||||||
|
cmatch |= 2;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (cmatch == 3)
|
||||||
|
{
|
||||||
|
if (y + 1 >= h || x + 1 >= w)
|
||||||
|
{
|
||||||
|
// last pixel in row/column and nearest
|
||||||
|
// neighboor is identical
|
||||||
|
dst_p++;
|
||||||
|
SCALE_SETPIX (dst_p + dlen, PIX (0, 0));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
// check pixel to the bottom-right
|
||||||
|
PIX (1, 1) = SCALE_GETPIX (SRC (1, 1));
|
||||||
|
|
||||||
|
if (PIX (0, 0) == PIX (1, 1))
|
||||||
|
{
|
||||||
|
// all 4 are equal - propagate
|
||||||
|
dst_p++;
|
||||||
|
SCALE_SETPIX (dst_p + dlen, PIX (0, 0));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// some neighboors are different, lets check them
|
||||||
|
|
||||||
|
if (x > 0)
|
||||||
|
PIX (-1, 0) = SCALE_GETPIX (SRC (-1, 0));
|
||||||
|
else
|
||||||
|
PIX (-1, 0) = PIX (0, 0);
|
||||||
|
|
||||||
|
if (x + 2 < w)
|
||||||
|
PIX (2, 0) = SCALE_GETPIX (SRC (2, 0));
|
||||||
|
else
|
||||||
|
PIX (2, 0) = PIX (1, 0);
|
||||||
|
|
||||||
|
if (y + 1 < h)
|
||||||
|
{
|
||||||
|
if (x > 0)
|
||||||
|
PIX (-1, 1) = SCALE_GETPIX (SRC (-1, 1));
|
||||||
|
else
|
||||||
|
PIX (-1, 1) = PIX (0, 1);
|
||||||
|
|
||||||
|
if (x + 2 < w)
|
||||||
|
{
|
||||||
|
PIX (1, 1) = SCALE_GETPIX (SRC (1, 1));
|
||||||
|
PIX (2, 1) = SCALE_GETPIX (SRC (2, 1));
|
||||||
|
}
|
||||||
|
else if (x + 1 < w)
|
||||||
|
{
|
||||||
|
PIX (1, 1) = SCALE_GETPIX (SRC (1, 1));
|
||||||
|
PIX (2, 1) = PIX (1, 1);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
PIX (1, 1) = PIX (0, 1);
|
||||||
|
PIX (2, 1) = PIX (0, 1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
// last pixel in column
|
||||||
|
PIX (-1, 1) = PIX (-1, 0);
|
||||||
|
PIX (1, 1) = PIX (1, 0);
|
||||||
|
PIX (2, 1) = PIX (2, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (y + 2 < h)
|
||||||
|
{
|
||||||
|
PIX (0, 2) = SCALE_GETPIX (SRC (0, 2));
|
||||||
|
|
||||||
|
if (x > 0)
|
||||||
|
PIX (-1, 2) = SCALE_GETPIX (SRC (-1, 2));
|
||||||
|
else
|
||||||
|
PIX (-1, 2) = PIX (0, 2);
|
||||||
|
|
||||||
|
if (x + 2 < w)
|
||||||
|
{
|
||||||
|
PIX (1, 2) = SCALE_GETPIX (SRC (1, 2));
|
||||||
|
PIX (2, 2) = SCALE_GETPIX (SRC (2, 2));
|
||||||
|
}
|
||||||
|
else if (x + 1 < w)
|
||||||
|
{
|
||||||
|
PIX (1, 2) = SCALE_GETPIX (SRC (1, 2));
|
||||||
|
PIX (2, 2) = PIX (1, 2);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
PIX (1, 2) = PIX (0, 2);
|
||||||
|
PIX (2, 2) = PIX (0, 2);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
// last pixel in column
|
||||||
|
PIX (-1, 2) = PIX (-1, 1);
|
||||||
|
PIX (0, 2) = PIX (0, 1);
|
||||||
|
PIX (1, 2) = PIX (1, 1);
|
||||||
|
PIX (2, 2) = PIX (2, 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (y > 0)
|
||||||
|
{
|
||||||
|
PIX (0, -1) = SCALE_GETPIX (SRC (0, -1));
|
||||||
|
|
||||||
|
if (x > 0)
|
||||||
|
PIX (-1, -1) = SCALE_GETPIX (SRC (-1, -1));
|
||||||
|
else
|
||||||
|
PIX (-1, -1) = PIX (0, -1);
|
||||||
|
|
||||||
|
if (x + 2 < w)
|
||||||
|
{
|
||||||
|
PIX (1, -1) = SCALE_GETPIX (SRC (1, -1));
|
||||||
|
PIX (2, -1) = SCALE_GETPIX (SRC (2, -1));
|
||||||
|
}
|
||||||
|
else if (x + 1 < w)
|
||||||
|
{
|
||||||
|
PIX (1, -1) = SCALE_GETPIX (SRC (1, -1));
|
||||||
|
PIX (2, -1) = PIX (1, -1);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
PIX (1, -1) = PIX (0, -1);
|
||||||
|
PIX (2, -1) = PIX (0, -1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
PIX (-1, -1) = PIX (-1, 0);
|
||||||
|
PIX (0, -1) = PIX (0, 0);
|
||||||
|
PIX (1, -1) = PIX (1, 0);
|
||||||
|
PIX (2, -1) = PIX (2, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
// check pixel below the current one
|
||||||
|
if (!(cmatch & 1))
|
||||||
|
{
|
||||||
|
if (SCALE_CMPYUV (PIX (0, 0), PIX (0, 1), BIADAPT_YUVLOW))
|
||||||
|
{
|
||||||
|
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
|
||||||
|
PIX (0, 0), PIX (0, 1))
|
||||||
|
);
|
||||||
|
cmatch |= 1;
|
||||||
|
}
|
||||||
|
// detect a 2:1 line going across the current pixel
|
||||||
|
else if ( (PIX (0, 0) == PIX (-1, 0)
|
||||||
|
&& PIX (0, 0) == PIX (1, 1)
|
||||||
|
&& PIX (0, 0) == PIX (2, 1) &&
|
||||||
|
|
||||||
|
((!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, -1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, -1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, 0))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, 0))) ||
|
||||||
|
(!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, 2))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, 2))))) ||
|
||||||
|
|
||||||
|
(PIX (0, 0) == PIX (1, 0)
|
||||||
|
&& PIX (0, 0) == PIX (-1, 1)
|
||||||
|
&& PIX (0, 0) == PIX (2, -1) &&
|
||||||
|
|
||||||
|
((!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 0))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, -1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, -1))) ||
|
||||||
|
(!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 2))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, 1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, 0))))) )
|
||||||
|
{
|
||||||
|
SCALE_SETPIX (dst_p + dlen, PIX (0, 0));
|
||||||
|
}
|
||||||
|
// detect a 2:1 line going across the pixel below current
|
||||||
|
else if ( (PIX (0, 1) == PIX (-1, 0)
|
||||||
|
&& PIX (0, 1) == PIX (1, 1)
|
||||||
|
&& PIX (0, 1) == PIX (2, 2) &&
|
||||||
|
|
||||||
|
((!BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (-1, -1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (1, 0))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (2, 1))) ||
|
||||||
|
(!BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (-1, 1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (0, 2))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (1, 2))))) ||
|
||||||
|
|
||||||
|
(PIX (0, 1) == PIX (1, 0)
|
||||||
|
&& PIX (0, 1) == PIX (-1, 1)
|
||||||
|
&& PIX (0, 1) == PIX (2, 0) &&
|
||||||
|
|
||||||
|
((!BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (-1, 0))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (1, -1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (2, -1))) ||
|
||||||
|
(!BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (-1, 2))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (0, 2))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (1, 1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 1), PIX (2, 1))))) )
|
||||||
|
{
|
||||||
|
SCALE_SETPIX (dst_p + dlen, PIX (0, 1));
|
||||||
|
}
|
||||||
|
else
|
||||||
|
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
|
||||||
|
PIX (0, 0), PIX (0, 1))
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
dst_p++;
|
||||||
|
|
||||||
|
// check pixel to the right from the current one
|
||||||
|
if (!(cmatch & 2))
|
||||||
|
{
|
||||||
|
if (SCALE_CMPYUV (PIX (0, 0), PIX (1, 0), BIADAPT_YUVLOW))
|
||||||
|
{
|
||||||
|
SCALE_SETPIX (dst_p, Scale_Blend_11 (
|
||||||
|
PIX (0, 0), PIX (1, 0))
|
||||||
|
);
|
||||||
|
cmatch |= 2;
|
||||||
|
}
|
||||||
|
// detect a 1:2 line going across the current pixel
|
||||||
|
else if ( (PIX (0, 0) == PIX (1, -1)
|
||||||
|
&& PIX (0, 0) == PIX (0, 1)
|
||||||
|
&& PIX (0, 0) == PIX (-1, 2) &&
|
||||||
|
|
||||||
|
((!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, -1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 0))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 1))) ||
|
||||||
|
(!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, -1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, 1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, 2))))) ||
|
||||||
|
|
||||||
|
(PIX (0, 0) == PIX (0, -1)
|
||||||
|
&& PIX (0, 0) == PIX (1, 1)
|
||||||
|
&& PIX (0, 0) == PIX (1, 2) &&
|
||||||
|
|
||||||
|
((!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, -1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (-1, 0))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, 1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (0, 2))) ||
|
||||||
|
(!BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (1, -1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, 1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (0, 0), PIX (2, 2))))) )
|
||||||
|
|
||||||
|
{
|
||||||
|
SCALE_SETPIX (dst_p, PIX (0, 0));
|
||||||
|
}
|
||||||
|
// detect a 1:2 line going across the pixel to the right
|
||||||
|
else if ( (PIX (1, 0) == PIX (1, -1)
|
||||||
|
&& PIX (1, 0) == PIX (0, 1)
|
||||||
|
&& PIX (1, 0) == PIX (0, 2) &&
|
||||||
|
|
||||||
|
((!BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (0, -1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (-1, 1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (-1, 2))) ||
|
||||||
|
(!BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (2, -1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (2, 0))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (1, 1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (1, 2))))) ||
|
||||||
|
|
||||||
|
(PIX (1, 0) == PIX (0, -1)
|
||||||
|
&& PIX (1, 0) == PIX (1, 1)
|
||||||
|
&& PIX (1, 0) == PIX (2, 2) &&
|
||||||
|
|
||||||
|
((!BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (-1, -1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (0, 1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (1, 2))) ||
|
||||||
|
(!BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (1, -1))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (2, 0))
|
||||||
|
&& !BIADAPT_CMPRGB_HIGH (PIX (1, 0), PIX (2, 1))))) )
|
||||||
|
{
|
||||||
|
SCALE_SETPIX (dst_p, PIX (1, 0));
|
||||||
|
}
|
||||||
|
else
|
||||||
|
SCALE_SETPIX (dst_p, Scale_Blend_11 (
|
||||||
|
PIX (0, 0), PIX (1, 0))
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (PIX (0, 0) == PIX (1, 1) && PIX (1, 0) == PIX (0, 1))
|
||||||
|
{
|
||||||
|
// diagonals are equal
|
||||||
|
int *coord;
|
||||||
|
int cl, cr;
|
||||||
|
Uint32 clr;
|
||||||
|
|
||||||
|
// both pairs are equal, have to resolve the pixel
|
||||||
|
// race; we try detecting which color is
|
||||||
|
// the background by looking for a line or an edge
|
||||||
|
// examine 8 pixels surrounding the current quad
|
||||||
|
|
||||||
|
cl = cr = 2;
|
||||||
|
for (coord = resolve_coord[0]; *coord < 100; coord += 2)
|
||||||
|
{
|
||||||
|
clr = PIX (coord[0], coord[1]);
|
||||||
|
|
||||||
|
if (BIADAPT_CMPYUV_MED (clr, PIX (0, 0)))
|
||||||
|
cl++;
|
||||||
|
else if (BIADAPT_CMPYUV_MED (clr, PIX (1, 0)))
|
||||||
|
cr++;
|
||||||
|
}
|
||||||
|
|
||||||
|
// least count wins
|
||||||
|
if (cl > cr)
|
||||||
|
clr = PIX (1, 0);
|
||||||
|
else if (cr > cl)
|
||||||
|
clr = PIX (0, 0);
|
||||||
|
else
|
||||||
|
clr = Scale_Blend_11 (PIX (0, 0), PIX (1, 0));
|
||||||
|
|
||||||
|
SCALE_SETPIX (dst_p + dlen, clr);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (cmatch == 3
|
||||||
|
|| (BIADAPT_CMPYUV_LOW (PIX (1, 0), PIX (0, 1))
|
||||||
|
&& BIADAPT_CMPYUV_LOW (PIX (1, 0), PIX (1, 1))))
|
||||||
|
{
|
||||||
|
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
|
||||||
|
PIX (0, 1), PIX (1, 0))
|
||||||
|
);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
else if (cmatch && BIADAPT_CMPYUV_LOW (PIX (0, 0), PIX (1, 1)))
|
||||||
|
{
|
||||||
|
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
|
||||||
|
PIX (0, 0), PIX (1, 1))
|
||||||
|
);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
// check pixel to the bottom-right
|
||||||
|
if (BIADAPT_CMPYUV_HIGH (PIX (0, 0), PIX (1, 1))
|
||||||
|
&& BIADAPT_CMPYUV_HIGH (PIX (1, 0), PIX (0, 1)))
|
||||||
|
{
|
||||||
|
if (SCALE_GETY (PIX (0, 0)) > SCALE_GETY (PIX (1, 0)))
|
||||||
|
{
|
||||||
|
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
|
||||||
|
PIX (0, 0), PIX (1, 1))
|
||||||
|
);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
|
||||||
|
PIX (1, 0), PIX (0, 1))
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
else if (BIADAPT_CMPYUV_HIGH (PIX (0, 0), PIX (1, 1)))
|
||||||
|
{
|
||||||
|
// main diagonal is same color
|
||||||
|
// use its value
|
||||||
|
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
|
||||||
|
PIX (0, 0), PIX (1, 1))
|
||||||
|
);
|
||||||
|
}
|
||||||
|
else if (BIADAPT_CMPYUV_HIGH (PIX (1, 0), PIX (0, 1)))
|
||||||
|
{
|
||||||
|
// 2nd diagonal is same color
|
||||||
|
// use its value
|
||||||
|
SCALE_SETPIX (dst_p + dlen, Scale_Blend_11 (
|
||||||
|
PIX (1, 0), PIX (0, 1))
|
||||||
|
);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
// blend all 4
|
||||||
|
SCALE_SETPIX (dst_p + dlen, Scale_Blend_1111 (
|
||||||
|
PIX (0, 0), PIX (0, 1),
|
||||||
|
PIX (1, 0), PIX (1, 1)
|
||||||
|
));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SCALE_(PlatDone) ();
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif /* GFXMODULE_SDL */
|
||||||
+115
@@ -0,0 +1,115 @@
|
|||||||
|
/*
|
||||||
|
* This program is free software; you can redistribute it and/or modify
|
||||||
|
* it under the terms of the GNU General Public License as published by
|
||||||
|
* the Free Software Foundation; either version 2 of the License, or
|
||||||
|
* (at your option) any later version.
|
||||||
|
*
|
||||||
|
* This program is distributed in the hope that it will be useful,
|
||||||
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
* GNU General Public License for more details.
|
||||||
|
*
|
||||||
|
* You should have received a copy of the GNU General Public License
|
||||||
|
* along with this program; if not, write to the Free Software
|
||||||
|
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||||
|
*/
|
||||||
|
|
||||||
|
// Core algorithm of the BiLinear screen scaler
|
||||||
|
// Template
|
||||||
|
// When this file is built standalone is produces a plain C version
|
||||||
|
// Also #included by 2xscalers_mmx.c for an MMX version
|
||||||
|
|
||||||
|
#ifdef GFXMODULE_SDL
|
||||||
|
|
||||||
|
#include "libs/graphics/sdl/sdl_common.h"
|
||||||
|
#include "types.h"
|
||||||
|
#include "scalers.h"
|
||||||
|
#include "scaleint.h"
|
||||||
|
#include "2xscalers.h"
|
||||||
|
|
||||||
|
|
||||||
|
// Bilinear scaling to 2x
|
||||||
|
// The name expands to either
|
||||||
|
// Scale_BilinearFilter (for plain C) or
|
||||||
|
// Scale_MMX_BilinearFilter (for MMX)
|
||||||
|
// Scale_SSE_BilinearFilter (for SSE)
|
||||||
|
// [others when platforms are added]
|
||||||
|
void
|
||||||
|
SCALE_(BilinearFilter) (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
{
|
||||||
|
int x, y;
|
||||||
|
const int w = src->w, h = src->h;
|
||||||
|
int xend, yend;
|
||||||
|
int dsrc, ddst;
|
||||||
|
SDL_Rect *region = r;
|
||||||
|
SDL_Rect limits;
|
||||||
|
SDL_PixelFormat *fmt = dst->format;
|
||||||
|
const int pitch = src->pitch, dp = dst->pitch;
|
||||||
|
const int bpp = fmt->BytesPerPixel;
|
||||||
|
const int len = pitch / bpp, dlen = dp / bpp;
|
||||||
|
Uint32 p[4]; // influential pixels array
|
||||||
|
Uint32 *srow0 = (Uint32 *) src->pixels;
|
||||||
|
Uint32 *dst_p = (Uint32 *) dst->pixels;
|
||||||
|
|
||||||
|
SCALE_(PlatInit) ();
|
||||||
|
|
||||||
|
// expand updated region if necessary
|
||||||
|
// pixels neighbooring the updated region may
|
||||||
|
// change as a result of updates
|
||||||
|
limits.x = 0;
|
||||||
|
limits.y = 0;
|
||||||
|
limits.w = w;
|
||||||
|
limits.h = h;
|
||||||
|
Scale_ExpandRect (region, 1, &limits);
|
||||||
|
|
||||||
|
xend = region->x + region->w;
|
||||||
|
yend = region->y + region->h;
|
||||||
|
dsrc = len - region->w;
|
||||||
|
ddst = (dlen - region->w) * 2;
|
||||||
|
|
||||||
|
// move ptrs to the first updated pixel
|
||||||
|
srow0 += len * region->y + region->x;
|
||||||
|
dst_p += (dlen * region->y + region->x) * 2;
|
||||||
|
|
||||||
|
for (y = region->y; y < yend; ++y, dst_p += ddst, srow0 += dsrc)
|
||||||
|
{
|
||||||
|
Uint32 *srow1;
|
||||||
|
|
||||||
|
SCALE_(Prefetch) (srow0 + 16);
|
||||||
|
SCALE_(Prefetch) (srow0 + 32);
|
||||||
|
|
||||||
|
if (y < h - 1)
|
||||||
|
srow1 = srow0 + len;
|
||||||
|
else
|
||||||
|
srow1 = srow0;
|
||||||
|
|
||||||
|
SCALE_(Prefetch) (srow1 + 16);
|
||||||
|
SCALE_(Prefetch) (srow1 + 32);
|
||||||
|
|
||||||
|
for (x = region->x; x < xend; ++x, ++srow0, ++srow1, dst_p += 2)
|
||||||
|
{
|
||||||
|
if (x < w - 1)
|
||||||
|
{ // can blend directly from pixels
|
||||||
|
SCALE_BILINEAR_BLEND4 (srow0, srow1, dst_p, dlen);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{ // need to make temp pixel rows
|
||||||
|
p[0] = srow0[0];
|
||||||
|
p[1] = p[0];
|
||||||
|
p[2] = srow1[0];
|
||||||
|
p[3] = p[2];
|
||||||
|
|
||||||
|
SCALE_BILINEAR_BLEND4 (&p[0], &p[2], dst_p, dlen);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SCALE_(Prefetch) (srow0 + dsrc);
|
||||||
|
SCALE_(Prefetch) (srow0 + dsrc + 16);
|
||||||
|
SCALE_(Prefetch) (srow1 + dsrc);
|
||||||
|
SCALE_(Prefetch) (srow1 + dsrc + 16);
|
||||||
|
}
|
||||||
|
|
||||||
|
SCALE_(PlatDone) ();
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif /* GFXMODULE_SDL */
|
||||||
+208
@@ -0,0 +1,208 @@
|
|||||||
|
/*
|
||||||
|
* This program is free software; you can redistribute it and/or modify
|
||||||
|
* it under the terms of the GNU General Public License as published by
|
||||||
|
* the Free Software Foundation; either version 2 of the License, or
|
||||||
|
* (at your option) any later version.
|
||||||
|
*
|
||||||
|
* This program is distributed in the hope that it will be useful,
|
||||||
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
* GNU General Public License for more details.
|
||||||
|
*
|
||||||
|
* You should have received a copy of the GNU General Public License
|
||||||
|
* along with this program; if not, write to the Free Software
|
||||||
|
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||||
|
*/
|
||||||
|
|
||||||
|
// Core algorithm of the BiLinear screen scaler
|
||||||
|
// Template
|
||||||
|
// When this file is built standalone is produces a plain C version
|
||||||
|
// Also #included by 2xscalers_mmx.c for an MMX version
|
||||||
|
|
||||||
|
#ifdef GFXMODULE_SDL
|
||||||
|
|
||||||
|
#include "libs/graphics/sdl/sdl_common.h"
|
||||||
|
#include "types.h"
|
||||||
|
#include "scalers.h"
|
||||||
|
#include "scaleint.h"
|
||||||
|
#include "2xscalers.h"
|
||||||
|
|
||||||
|
// Nearest Neighbor scaling to 2x
|
||||||
|
// The name expands to
|
||||||
|
// Scale_Nearest (for plain C)
|
||||||
|
// Scale_MMX_Nearest (for MMX)
|
||||||
|
// Scale_SSE_Nearest (for SSE)
|
||||||
|
// [others when platforms are added]
|
||||||
|
void
|
||||||
|
SCALE_(Nearest) (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
{
|
||||||
|
int y;
|
||||||
|
const int rw = r->w, rh = r->h;
|
||||||
|
const int sp = src->pitch, dp = dst->pitch;
|
||||||
|
const int bpp = dst->format->BytesPerPixel;
|
||||||
|
const int slen = sp / bpp, dlen = dp / bpp;
|
||||||
|
const int dsrc = slen-rw, ddst = (dlen-rw) * 2;
|
||||||
|
|
||||||
|
Uint32 *src_p = (Uint32 *)src->pixels;
|
||||||
|
Uint32 *dst_p = (Uint32 *)dst->pixels;
|
||||||
|
|
||||||
|
// guard asm code against such atrocities
|
||||||
|
if (rw == 0 || rh == 0)
|
||||||
|
return;
|
||||||
|
|
||||||
|
SCALE_(PlatInit) ();
|
||||||
|
|
||||||
|
// move ptrs to the first updated pixel
|
||||||
|
src_p += slen * r->y + r->x;
|
||||||
|
dst_p += (dlen * r->y + r->x) * 2;
|
||||||
|
|
||||||
|
#if defined(MMX_ASM) && defined(MSVC_ASM)
|
||||||
|
// Just about everything has to be done in asm for MSVC
|
||||||
|
// to actually take advantage of asm here
|
||||||
|
// MSVC does not support beautiful GCC-like asm templates
|
||||||
|
|
||||||
|
y = rh;
|
||||||
|
__asm
|
||||||
|
{
|
||||||
|
// setup vars
|
||||||
|
mov esi, src_p
|
||||||
|
mov edi, dst_p
|
||||||
|
|
||||||
|
PREFETCH (esi + 0x40)
|
||||||
|
PREFETCH (esi + 0x80)
|
||||||
|
PREFETCH (esi + 0xc0)
|
||||||
|
|
||||||
|
mov edx, dlen
|
||||||
|
lea edx, [edx * 4]
|
||||||
|
mov eax, dsrc
|
||||||
|
lea eax, [eax * 4]
|
||||||
|
mov ebx, ddst
|
||||||
|
lea ebx, [ebx * 4]
|
||||||
|
|
||||||
|
mov ecx, rw
|
||||||
|
loop_y:
|
||||||
|
test ecx, 1
|
||||||
|
jz even_x
|
||||||
|
|
||||||
|
// one-pixel transfer
|
||||||
|
movd mm1, [esi]
|
||||||
|
punpckldq mm1, mm1 // pix1 | pix1 -> mm1
|
||||||
|
add esi, 4
|
||||||
|
MOVNTQ (edi, mm1)
|
||||||
|
add edi, 8
|
||||||
|
MOVNTQ (edi - 8 + edx, mm1)
|
||||||
|
|
||||||
|
even_x:
|
||||||
|
shr ecx, 1 // x = rw / 2
|
||||||
|
|
||||||
|
loop_x:
|
||||||
|
// two-pixel transfer
|
||||||
|
movq mm1, [esi]
|
||||||
|
movq mm2, mm1
|
||||||
|
PREFETCH (esi + 0x100)
|
||||||
|
punpckldq mm1, mm1 // pix1 | pix1 -> mm1
|
||||||
|
add esi, 8
|
||||||
|
MOVNTQ (edi, mm1)
|
||||||
|
punpckhdq mm2, mm2 // pix2 | pix2 -> mm2
|
||||||
|
MOVNTQ (edi + edx, mm1)
|
||||||
|
add edi, 16
|
||||||
|
MOVNTQ (edi - 8, mm2)
|
||||||
|
MOVNTQ (edi - 8 + edx, mm2)
|
||||||
|
|
||||||
|
dec ecx
|
||||||
|
jnz loop_x
|
||||||
|
|
||||||
|
// try to prefetch as early as possible to have it on time
|
||||||
|
PREFETCH (esi + eax)
|
||||||
|
|
||||||
|
mov ecx, rw
|
||||||
|
add esi, eax
|
||||||
|
|
||||||
|
PREFETCH (esi + 0x40)
|
||||||
|
PREFETCH (esi + 0x80)
|
||||||
|
PREFETCH (esi + 0xc0)
|
||||||
|
|
||||||
|
add edi, ebx
|
||||||
|
|
||||||
|
dec y
|
||||||
|
jnz loop_y
|
||||||
|
}
|
||||||
|
|
||||||
|
#elif defined(MMX_ASM) && defined(GCC_ASM)
|
||||||
|
|
||||||
|
SCALE_(Prefetch) (src_p + 16);
|
||||||
|
SCALE_(Prefetch) (src_p + 32);
|
||||||
|
SCALE_(Prefetch) (src_p + 48);
|
||||||
|
|
||||||
|
for (y = rh; y; --y)
|
||||||
|
{
|
||||||
|
int x = rw;
|
||||||
|
|
||||||
|
if (x & 1)
|
||||||
|
{ // one-pixel transfer
|
||||||
|
__asm__ (
|
||||||
|
"movd (%0), %%mm1 \n\t"
|
||||||
|
"punpckldq %%mm1, %%mm1 \n\t"
|
||||||
|
MOVNTQ (%%mm1, (%1)) "\n\t"
|
||||||
|
MOVNTQ (%%mm1, (%1,%2)) "\n\t"
|
||||||
|
|
||||||
|
: /* nothing */
|
||||||
|
: /*0*/"r" (src_p), /*1*/"r" (dst_p), /*2*/"r" (dlen*sizeof(Uint32))
|
||||||
|
);
|
||||||
|
|
||||||
|
++src_p;
|
||||||
|
dst_p += 2;
|
||||||
|
--x;
|
||||||
|
}
|
||||||
|
|
||||||
|
for (x >>= 1; x; --x, src_p += 2, dst_p += 4)
|
||||||
|
{ // two-pixel transfer
|
||||||
|
__asm__ (
|
||||||
|
"movq (%0), %%mm1 \n\t"
|
||||||
|
"movq %%mm1, %%mm2 \n\t"
|
||||||
|
PREFETCH (0x100(%0)) "\n\t"
|
||||||
|
"punpckldq %%mm1, %%mm1 \n\t"
|
||||||
|
MOVNTQ (%%mm1, (%1)) "\n\t"
|
||||||
|
MOVNTQ (%%mm1, (%1,%2)) "\n\t"
|
||||||
|
"punpckhdq %%mm2, %%mm2 \n\t"
|
||||||
|
MOVNTQ (%%mm2, 8(%1)) "\n\t"
|
||||||
|
MOVNTQ (%%mm2, 8(%1,%2)) "\n\t"
|
||||||
|
|
||||||
|
: /* nothing */
|
||||||
|
: /*0*/"r" (src_p), /*1*/"r" (dst_p), /*2*/"r" (dlen*sizeof(Uint32))
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
src_p += dsrc;
|
||||||
|
// try to prefetch as early as possible to have it on time
|
||||||
|
SCALE_(Prefetch) (src_p);
|
||||||
|
|
||||||
|
dst_p += ddst;
|
||||||
|
|
||||||
|
SCALE_(Prefetch) (src_p + 16);
|
||||||
|
SCALE_(Prefetch) (src_p + 32);
|
||||||
|
SCALE_(Prefetch) (src_p + 48);
|
||||||
|
}
|
||||||
|
|
||||||
|
#else
|
||||||
|
// Plain C version
|
||||||
|
for (y = 0; y < rh; ++y)
|
||||||
|
{
|
||||||
|
int x;
|
||||||
|
for (x = 0; x < rw; ++x, ++src_p, dst_p += 2)
|
||||||
|
{
|
||||||
|
Uint32 pix = *src_p;
|
||||||
|
dst_p[0] = pix;
|
||||||
|
dst_p[1] = pix;
|
||||||
|
dst_p[dlen] = pix;
|
||||||
|
dst_p[dlen + 1] = pix;
|
||||||
|
}
|
||||||
|
dst_p += ddst;
|
||||||
|
src_p += dsrc;
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
SCALE_(PlatDone) ();
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif /* GFXMODULE_SDL */
|
||||||
Executable
+433
@@ -0,0 +1,433 @@
|
|||||||
|
/*
|
||||||
|
* Copyright (C) 2005 Alex Volkov (codepro@usa.net)
|
||||||
|
*
|
||||||
|
* This program is free software; you can redistribute it and/or modify
|
||||||
|
* it under the terms of the GNU General Public License as published by
|
||||||
|
* the Free Software Foundation; either version 2 of the License, or
|
||||||
|
* (at your option) any later version.
|
||||||
|
*
|
||||||
|
* This program is distributed in the hope that it will be useful,
|
||||||
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
* GNU General Public License for more details.
|
||||||
|
*
|
||||||
|
* You should have received a copy of the GNU General Public License
|
||||||
|
* along with this program; if not, write to the Free Software
|
||||||
|
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||||
|
*/
|
||||||
|
|
||||||
|
// Scalers Internals
|
||||||
|
|
||||||
|
#ifndef SCALEINT_H_
|
||||||
|
#define SCALEINT_H_
|
||||||
|
|
||||||
|
#include "libs/graphics/sdl/sdl_common.h"
|
||||||
|
#include "types.h"
|
||||||
|
|
||||||
|
|
||||||
|
// Plain C names
|
||||||
|
#define SCALE_(name) Scale ## _ ## name
|
||||||
|
|
||||||
|
// These are defaults
|
||||||
|
#define SCALE_GETPIX(p) ( *(Uint32 *)(p) )
|
||||||
|
#define SCALE_SETPIX(p, c) ( *(Uint32 *)(p) = (c) )
|
||||||
|
|
||||||
|
// Plain C defaults
|
||||||
|
#define SCALE_CMPRGB(p1, p2) \
|
||||||
|
SCALE_(GetRGBDelta) (fmt, p1, p2)
|
||||||
|
|
||||||
|
#define SCALE_TOYUV(p) \
|
||||||
|
SCALE_(RGBtoYUV) (fmt, p)
|
||||||
|
|
||||||
|
#define SCALE_CMPYUV(p1, p2, toler) \
|
||||||
|
SCALE_(CmpYUV) (fmt, p1, p2, toler)
|
||||||
|
|
||||||
|
#define SCALE_DIFFYUV(p1, p2) \
|
||||||
|
SCALE_(DiffYUV) (p1, p2)
|
||||||
|
#define SCALE_DIFFYUV_TY 0x40
|
||||||
|
#define SCALE_DIFFYUV_TU 0x12
|
||||||
|
#define SCALE_DIFFYUV_TV 0x0c
|
||||||
|
|
||||||
|
#define SCALE_GETY(p) \
|
||||||
|
SCALE_(GetPixY) (fmt, p)
|
||||||
|
|
||||||
|
#define SCALE_BILINEAR_BLEND4(r0, r1, dst, dlen) \
|
||||||
|
SCALE_(Blend_bilinear) (r0, r1, dst, dlen)
|
||||||
|
|
||||||
|
#define NO_PREFETCH 0
|
||||||
|
#define INTEL_PREFETCH 1
|
||||||
|
#define AMD_PREFETCH 2
|
||||||
|
|
||||||
|
typedef enum
|
||||||
|
{
|
||||||
|
YUV_XFORM_R = 0,
|
||||||
|
YUV_XFORM_G = 1,
|
||||||
|
YUV_XFORM_B = 2,
|
||||||
|
YUV_XFORM_Y = 0,
|
||||||
|
YUV_XFORM_U = 1,
|
||||||
|
YUV_XFORM_V = 2
|
||||||
|
} RGB_YUV_INDEX;
|
||||||
|
|
||||||
|
extern const int YUV_matrix[3][3];
|
||||||
|
|
||||||
|
// pre-computed transformations for 8 bits per channel
|
||||||
|
extern int RGB_to_YUV[/*RGB*/ 3][/*YUV*/ 3][ /*mult-res*/ 256];
|
||||||
|
extern sint16 dRGB_to_dYUV[/*RGB*/ 3][/*YUV*/ 3][ /*mult-res*/ 512];
|
||||||
|
|
||||||
|
typedef Uint32 YUV_VECTOR;
|
||||||
|
// pre-computed transformations for RGB555
|
||||||
|
extern YUV_VECTOR RGB15_to_YUV[0x8000];
|
||||||
|
|
||||||
|
|
||||||
|
// Platform+Scaler function lookups
|
||||||
|
//
|
||||||
|
typedef struct
|
||||||
|
{
|
||||||
|
int flag;
|
||||||
|
TFB_ScaleFunc func;
|
||||||
|
} Scale_FuncDef_t;
|
||||||
|
|
||||||
|
|
||||||
|
// expands the given rectangle in all directions by 'expansion'
|
||||||
|
// guarded by 'limits'
|
||||||
|
extern void Scale_ExpandRect (SDL_Rect* rect, int expansion,
|
||||||
|
const SDL_Rect* limits);
|
||||||
|
|
||||||
|
|
||||||
|
// Standard plain C versions of support functions
|
||||||
|
|
||||||
|
// Initialize various platform-specific features
|
||||||
|
static inline void
|
||||||
|
SCALE_(PlatInit) (void)
|
||||||
|
{
|
||||||
|
}
|
||||||
|
|
||||||
|
// Finish with various platform-specific features
|
||||||
|
static inline void
|
||||||
|
SCALE_(PlatDone) (void)
|
||||||
|
{
|
||||||
|
}
|
||||||
|
|
||||||
|
#if 0
|
||||||
|
static inline void
|
||||||
|
SCALE_(Prefetch) (const void* p)
|
||||||
|
{
|
||||||
|
/* no-op in pure C */
|
||||||
|
(void)p;
|
||||||
|
}
|
||||||
|
#else
|
||||||
|
# define Scale_Prefetch(p)
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// compute the RGB distance squared between 2 pixels
|
||||||
|
// Plain C version
|
||||||
|
static inline int
|
||||||
|
SCALE_(GetRGBDelta) (const SDL_PixelFormat* fmt, Uint32 pix1, Uint32 pix2)
|
||||||
|
{
|
||||||
|
int c;
|
||||||
|
int delta;
|
||||||
|
|
||||||
|
c = ((pix1 >> fmt->Rshift) & 0xff) - ((pix2 >> fmt->Rshift) & 0xff);
|
||||||
|
delta = c * c;
|
||||||
|
|
||||||
|
c = ((pix1 >> fmt->Gshift) & 0xff) - ((pix2 >> fmt->Gshift) & 0xff);
|
||||||
|
delta += c * c;
|
||||||
|
|
||||||
|
c = ((pix1 >> fmt->Bshift) & 0xff) - ((pix2 >> fmt->Bshift) & 0xff);
|
||||||
|
delta += c * c;
|
||||||
|
|
||||||
|
return delta;
|
||||||
|
}
|
||||||
|
|
||||||
|
// retrieve the Y (intensity) component of pixel's YUV
|
||||||
|
// Plain C version
|
||||||
|
static inline int
|
||||||
|
SCALE_(GetPixY) (const SDL_PixelFormat* fmt, Uint32 pix)
|
||||||
|
{
|
||||||
|
Uint32 r, g, b;
|
||||||
|
|
||||||
|
r = (pix >> fmt->Rshift) & 0xff;
|
||||||
|
g = (pix >> fmt->Gshift) & 0xff;
|
||||||
|
b = (pix >> fmt->Bshift) & 0xff;
|
||||||
|
|
||||||
|
return RGB_to_YUV [YUV_XFORM_R][YUV_XFORM_Y][r]
|
||||||
|
+ RGB_to_YUV [YUV_XFORM_G][YUV_XFORM_Y][g]
|
||||||
|
+ RGB_to_YUV [YUV_XFORM_B][YUV_XFORM_Y][b];
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline YUV_VECTOR
|
||||||
|
SCALE_(RGBtoYUV) (const SDL_PixelFormat* fmt, Uint32 pix)
|
||||||
|
{
|
||||||
|
return RGB15_to_YUV[
|
||||||
|
(((pix >> (fmt->Rshift + 3)) & 0x1f) << 10) |
|
||||||
|
(((pix >> (fmt->Gshift + 3)) & 0x1f) << 5) |
|
||||||
|
(((pix >> (fmt->Bshift + 3)) & 0x1f) )
|
||||||
|
];
|
||||||
|
}
|
||||||
|
|
||||||
|
// compare 2 pixels with respect to their YUV representations
|
||||||
|
// tolerance set by toler arg
|
||||||
|
// returns true: close; false: distant (-gt toler)
|
||||||
|
// Plain C version
|
||||||
|
static inline bool
|
||||||
|
SCALE_(CmpYUV) (const SDL_PixelFormat* fmt, Uint32 pix1, Uint32 pix2, int toler)
|
||||||
|
#if 1
|
||||||
|
{
|
||||||
|
int dr, dg, db;
|
||||||
|
int delta;
|
||||||
|
|
||||||
|
dr = ((pix1 >> fmt->Rshift) & 0xff) - ((pix2 >> fmt->Rshift) & 0xff) + 255;
|
||||||
|
dg = ((pix1 >> fmt->Gshift) & 0xff) - ((pix2 >> fmt->Gshift) & 0xff) + 255;
|
||||||
|
db = ((pix1 >> fmt->Bshift) & 0xff) - ((pix2 >> fmt->Bshift) & 0xff) + 255;
|
||||||
|
|
||||||
|
// compute Y delta
|
||||||
|
delta = abs (dRGB_to_dYUV [YUV_XFORM_R][YUV_XFORM_Y][dr]
|
||||||
|
+ dRGB_to_dYUV [YUV_XFORM_G][YUV_XFORM_Y][dg]
|
||||||
|
+ dRGB_to_dYUV [YUV_XFORM_B][YUV_XFORM_Y][db]);
|
||||||
|
if (delta > toler)
|
||||||
|
return false;
|
||||||
|
|
||||||
|
// compute U delta
|
||||||
|
delta += abs (dRGB_to_dYUV [YUV_XFORM_R][YUV_XFORM_U][dr]
|
||||||
|
+ dRGB_to_dYUV [YUV_XFORM_G][YUV_XFORM_U][dg]
|
||||||
|
+ dRGB_to_dYUV [YUV_XFORM_B][YUV_XFORM_U][db]);
|
||||||
|
if (delta > toler)
|
||||||
|
return false;
|
||||||
|
|
||||||
|
// compute V delta
|
||||||
|
delta += abs (dRGB_to_dYUV [YUV_XFORM_R][YUV_XFORM_V][dr]
|
||||||
|
+ dRGB_to_dYUV [YUV_XFORM_G][YUV_XFORM_V][dg]
|
||||||
|
+ dRGB_to_dYUV [YUV_XFORM_B][YUV_XFORM_V][db]);
|
||||||
|
|
||||||
|
return delta <= toler;
|
||||||
|
}
|
||||||
|
#else
|
||||||
|
{
|
||||||
|
int delta;
|
||||||
|
Uint32 yuv1, yuv2;
|
||||||
|
|
||||||
|
yuv1 = RGB15_to_YUV[
|
||||||
|
(((pix1 >> (fmt->Rshift + 3)) & 0x1f) << 10) |
|
||||||
|
(((pix1 >> (fmt->Gshift + 3)) & 0x1f) << 5) |
|
||||||
|
(((pix1 >> (fmt->Bshift + 3)) & 0x1f) )
|
||||||
|
];
|
||||||
|
|
||||||
|
yuv2 = RGB15_to_YUV[
|
||||||
|
(((pix2 >> (fmt->Rshift + 3)) & 0x1f) << 10) |
|
||||||
|
(((pix2 >> (fmt->Gshift + 3)) & 0x1f) << 5) |
|
||||||
|
(((pix2 >> (fmt->Bshift + 3)) & 0x1f) )
|
||||||
|
];
|
||||||
|
|
||||||
|
// compute Y delta
|
||||||
|
delta = abs ((yuv1 & 0xff0000) - (yuv2 & 0xff0000)) >> 16;
|
||||||
|
if (delta > toler)
|
||||||
|
return false;
|
||||||
|
|
||||||
|
// compute U delta
|
||||||
|
delta += abs ((yuv1 & 0x00ff00) - (yuv2 & 0x00ff00)) >> 8;
|
||||||
|
if (delta > toler)
|
||||||
|
return false;
|
||||||
|
|
||||||
|
// compute V delta
|
||||||
|
delta += abs ((yuv1 & 0x0000ff) - (yuv2 & 0x0000ff));
|
||||||
|
|
||||||
|
return delta <= toler;
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// Check if 2 pixels are different with respect to their
|
||||||
|
// YUV representations
|
||||||
|
// returns 0: close; ~0: distant
|
||||||
|
static inline int
|
||||||
|
SCALE_(DiffYUV) (Uint32 yuv1, Uint32 yuv2)
|
||||||
|
{
|
||||||
|
// non-branching version -- assumes 2's complement integers
|
||||||
|
// delta math only needs 25 bits and we have 32 available;
|
||||||
|
// only interested in the sign bits after subtraction
|
||||||
|
sint32 delta, ret;
|
||||||
|
|
||||||
|
if (yuv1 == yuv2)
|
||||||
|
return 0;
|
||||||
|
|
||||||
|
// compute Y delta
|
||||||
|
delta = abs ((yuv1 & 0xff0000) - (yuv2 & 0xff0000));
|
||||||
|
ret = (SCALE_DIFFYUV_TY << 16) - delta; // save sign bit
|
||||||
|
|
||||||
|
// compute U delta
|
||||||
|
delta = abs ((yuv1 & 0x00ff00) - (yuv2 & 0x00ff00));
|
||||||
|
ret |= (SCALE_DIFFYUV_TU << 8) - delta; // save sign bit
|
||||||
|
|
||||||
|
// compute V delta
|
||||||
|
delta = abs ((yuv1 & 0x0000ff) - (yuv2 & 0x0000ff));
|
||||||
|
ret |= SCALE_DIFFYUV_TV - delta; // save sign bit
|
||||||
|
|
||||||
|
return (ret >> 31);
|
||||||
|
}
|
||||||
|
|
||||||
|
// blends two pixels with 1:1 ratio
|
||||||
|
static inline Uint32
|
||||||
|
SCALE_(Blend_11) (Uint32 pix1, Uint32 pix2)
|
||||||
|
{
|
||||||
|
/* (pix1 + pix2) >> 1 */
|
||||||
|
return
|
||||||
|
/* lower bits can be safely ignored - the error is minimal
|
||||||
|
expression that calcs them is left for posterity
|
||||||
|
(pix1 & pix2 & low_mask) +
|
||||||
|
*/
|
||||||
|
((pix1 & 0xfefefefe) >> 1) + ((pix2 & 0xfefefefe) >> 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
// blends four pixels with 1:1:1:1 ratio
|
||||||
|
static inline Uint32
|
||||||
|
SCALE_(Blend_1111) (Uint32 pix1, Uint32 pix2,
|
||||||
|
Uint32 pix3, Uint32 pix4)
|
||||||
|
{
|
||||||
|
/* (pix1 + pix2 + pix3 + pix4) >> 2 */
|
||||||
|
return
|
||||||
|
/* lower bits can be safely ignored - the error is minimal
|
||||||
|
expression that calcs them is left for posterity
|
||||||
|
((((pix1 & low_mask) + (pix2 & low_mask) +
|
||||||
|
(pix3 & low_mask) + (pix4 & low_mask)
|
||||||
|
) >> 2) & low_mask) +
|
||||||
|
*/
|
||||||
|
((pix1 & 0xfcfcfcfc) >> 2) + ((pix2 & 0xfcfcfcfc) >> 2) +
|
||||||
|
((pix3 & 0xfcfcfcfc) >> 2) + ((pix4 & 0xfcfcfcfc) >> 2);
|
||||||
|
}
|
||||||
|
|
||||||
|
// blends pixels with 3:1 ratio
|
||||||
|
static inline Uint32
|
||||||
|
Scale_Blend_31 (Uint32 pix1, Uint32 pix2)
|
||||||
|
{
|
||||||
|
/* (pix1 * 3 + pix2) / 4 */
|
||||||
|
/* lower bits can be safely ignored - the error is minimal */
|
||||||
|
return ((pix1 & 0xfefefefe) >> 1) + ((pix1 & 0xfcfcfcfc) >> 2) +
|
||||||
|
((pix2 & 0xfcfcfcfc) >> 2);
|
||||||
|
}
|
||||||
|
|
||||||
|
// blends pixels with 2:1:1 ratio
|
||||||
|
static inline Uint32
|
||||||
|
Scale_Blend_211 (Uint32 pix1, Uint32 pix2, Uint32 pix3)
|
||||||
|
{
|
||||||
|
/* (pix1 * 2 + pix2 + pix3) / 4 */
|
||||||
|
/* lower bits can be safely ignored - the error is minimal */
|
||||||
|
return ((pix1 & 0xfefefefe) >> 1) +
|
||||||
|
((pix2 & 0xfcfcfcfc) >> 2) +
|
||||||
|
((pix3 & 0xfcfcfcfc) >> 2);
|
||||||
|
}
|
||||||
|
|
||||||
|
// blends pixels with 5:2:1 ratio
|
||||||
|
static inline Uint32
|
||||||
|
Scale_Blend_521 (Uint32 pix1, Uint32 pix2, Uint32 pix3)
|
||||||
|
{
|
||||||
|
/* (pix1 * 5 + pix2 * 2 + pix3) / 8 */
|
||||||
|
/* lower bits can be safely ignored - the error is minimal */
|
||||||
|
return ((pix1 & 0xfefefefe) >> 1) + ((pix1 & 0xf8f8f8f8) >> 3) +
|
||||||
|
((pix2 & 0xfcfcfcfc) >> 2) +
|
||||||
|
((pix3 & 0xf8f8f8f8) >> 3) +
|
||||||
|
0x02020202 /* half-error */;
|
||||||
|
}
|
||||||
|
|
||||||
|
// blends pixels with 6:1:1 ratio
|
||||||
|
static inline Uint32
|
||||||
|
Scale_Blend_611 (Uint32 pix1, Uint32 pix2, Uint32 pix3)
|
||||||
|
{
|
||||||
|
/* (pix1 * 6 + pix2 + pix3) / 8 */
|
||||||
|
/* lower bits can be safely ignored - the error is minimal */
|
||||||
|
return ((pix1 & 0xfefefefe) >> 1) + ((pix1 & 0xfcfcfcfc) >> 2) +
|
||||||
|
((pix2 & 0xf8f8f8f8) >> 3) +
|
||||||
|
((pix3 & 0xf8f8f8f8) >> 3) +
|
||||||
|
0x02020202 /* half-error */;
|
||||||
|
}
|
||||||
|
|
||||||
|
// blends pixels with 2:3:3 ratio
|
||||||
|
static inline Uint32
|
||||||
|
Scale_Blend_233 (Uint32 pix1, Uint32 pix2, Uint32 pix3)
|
||||||
|
{
|
||||||
|
/* (pix1 * 2 + pix2 * 3 + pix3 * 3) / 8 */
|
||||||
|
/* lower bits can be safely ignored - the error is minimal */
|
||||||
|
return ((pix1 & 0xfcfcfcfc) >> 2) +
|
||||||
|
((pix2 & 0xfcfcfcfc) >> 2) + ((pix2 & 0xf8f8f8f8) >> 3) +
|
||||||
|
((pix3 & 0xfcfcfcfc) >> 2) + ((pix3 & 0xf8f8f8f8) >> 3) +
|
||||||
|
0x02020202 /* half-error */;
|
||||||
|
}
|
||||||
|
|
||||||
|
// blends pixels with 14:1:1 ratio
|
||||||
|
static inline Uint32
|
||||||
|
Scale_Blend_e11 (Uint32 pix1, Uint32 pix2, Uint32 pix3)
|
||||||
|
{
|
||||||
|
/* (pix1 * 14 + pix2 + pix3) >> 4 */
|
||||||
|
/* lower bits can be safely ignored - the error is minimal */
|
||||||
|
return ((pix1 & 0xfefefefe) >> 1) + ((pix1 & 0xfcfcfcfc) >> 2) +
|
||||||
|
((pix1 & 0xf8f8f8f8) >> 3) +
|
||||||
|
((pix2 & 0xf0f0f0f0) >> 4) +
|
||||||
|
((pix3 & 0xf0f0f0f0) >> 4) +
|
||||||
|
0x03030303 /* half-error */;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Halfs the pixel's intensity
|
||||||
|
static inline Uint32
|
||||||
|
SCALE_(HalfPixel) (Uint32 pix)
|
||||||
|
{
|
||||||
|
return ((pix & 0xfefefefe) >> 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
// Bilinear weighted blend of four pixels
|
||||||
|
// Function produces 4 blended pixels and writes them
|
||||||
|
// out to the surface (in 2x2 matrix)
|
||||||
|
// Pixels are computed using expanded weight matrix like so:
|
||||||
|
// ('sp' - source pixel, 'dp' - destination pixel)
|
||||||
|
// dp[0] = (9*sp[0] + 3*sp[1] + 3*sp[2] + 1*sp[3]) / 16
|
||||||
|
// dp[1] = (3*sp[0] + 9*sp[1] + 1*sp[2] + 3*sp[3]) / 16
|
||||||
|
// dp[2] = (3*sp[0] + 1*sp[1] + 9*sp[2] + 3*sp[3]) / 16
|
||||||
|
// dp[3] = (1*sp[0] + 3*sp[1] + 3*sp[2] + 9*sp[3]) / 16
|
||||||
|
static inline void
|
||||||
|
SCALE_(Blend_bilinear) (const Uint32* row0, const Uint32* row1,
|
||||||
|
Uint32* dst_p, Uint32 dlen)
|
||||||
|
{
|
||||||
|
// We loose some lower bits here and try to compensate for
|
||||||
|
// that by adding half-error values.
|
||||||
|
// In general, the error is minimal (+-7)
|
||||||
|
// The >>4 reduction is achieved gradually
|
||||||
|
# define BL_PACKED_HALF(p) \
|
||||||
|
(((p) & 0xfefefefe) >> 1)
|
||||||
|
# define BL_SUM(p1, p2) \
|
||||||
|
(BL_PACKED_HALF(p1) + BL_PACKED_HALF(p2))
|
||||||
|
# define BL_HALF_ERR 0x01010101
|
||||||
|
# define BL_SUM_WERR(p1, p2) \
|
||||||
|
(BL_PACKED_HALF(p1) + BL_PACKED_HALF(p2) + BL_HALF_ERR)
|
||||||
|
|
||||||
|
Uint32 sum1111, sum1331, sum3113;
|
||||||
|
|
||||||
|
// cache p[0] + 3*(p[1] + p[2]) + p[3] in sum1331
|
||||||
|
// cache p[1] + 3*(p[0] + p[3]) + p[2] in sum3113
|
||||||
|
sum1331 = BL_SUM (row0[1], row1[0]);
|
||||||
|
sum3113 = BL_SUM (row0[0], row1[1]);
|
||||||
|
|
||||||
|
// cache p[0] + p[1] + p[2] + p[3] in sum1111
|
||||||
|
sum1111 = BL_SUM_WERR (sum1331, sum3113);
|
||||||
|
|
||||||
|
sum1331 = BL_SUM_WERR (sum1331, sum1111);
|
||||||
|
sum1331 = BL_PACKED_HALF (sum1331);
|
||||||
|
sum3113 = BL_SUM_WERR (sum3113, sum1111);
|
||||||
|
sum3113 = BL_PACKED_HALF (sum3113);
|
||||||
|
|
||||||
|
// pixel 0 math -- (9*p[0] + 3*(p[1] + p[2]) + p[3]) / 16
|
||||||
|
dst_p[0] = BL_PACKED_HALF (row0[0]) + sum1331;
|
||||||
|
|
||||||
|
// pixel 1 math -- (9*p[1] + 3*(p[0] + p[3]) + p[2]) / 16
|
||||||
|
dst_p[1] = BL_PACKED_HALF (row0[1]) + sum3113;
|
||||||
|
|
||||||
|
// pixel 2 math -- (9*p[2] + 3*(p[0] + p[3]) + p[1]) / 16
|
||||||
|
dst_p[dlen] = BL_PACKED_HALF (row1[0]) + sum3113;
|
||||||
|
|
||||||
|
// pixel 3 math -- (9*p[3] + 3*(p[1] + p[2]) + p[0]) / 16
|
||||||
|
dst_p[dlen + 1] = BL_PACKED_HALF (row1[1]) + sum1331;
|
||||||
|
|
||||||
|
# undef BL_PACKED_HALF
|
||||||
|
# undef BL_SUM
|
||||||
|
# undef BL_HALF_ERR
|
||||||
|
# undef BL_SUM_WERR
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif /* SCALEINT_H_ */
|
||||||
Executable
+790
@@ -0,0 +1,790 @@
|
|||||||
|
/*
|
||||||
|
* Copyright (C) 2005 Alex Volkov (codepro@usa.net)
|
||||||
|
*
|
||||||
|
* This program is free software; you can redistribute it and/or modify
|
||||||
|
* it under the terms of the GNU General Public License as published by
|
||||||
|
* the Free Software Foundation; either version 2 of the License, or
|
||||||
|
* (at your option) any later version.
|
||||||
|
*
|
||||||
|
* This program is distributed in the hope that it will be useful,
|
||||||
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
* GNU General Public License for more details.
|
||||||
|
*
|
||||||
|
* You should have received a copy of the GNU General Public License
|
||||||
|
* along with this program; if not, write to the Free Software
|
||||||
|
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||||
|
*/
|
||||||
|
|
||||||
|
#ifndef SCALEMMX_H_
|
||||||
|
#define SCALEMMX_H_
|
||||||
|
|
||||||
|
#if !defined(SCALE_)
|
||||||
|
# error Please define SCALE_(name) before including scalemmx.h
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#if !defined(MSVC_ASM) && !defined(GCC_ASM)
|
||||||
|
# error Please define target assembler (MSVC_ASM, GCC_ASM) before including scalemmx.h
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// MMX defaults (no Format param)
|
||||||
|
#undef SCALE_CMPRGB
|
||||||
|
#define SCALE_CMPRGB(p1, p2) \
|
||||||
|
SCALE_(GetRGBDelta) (p1, p2)
|
||||||
|
|
||||||
|
#undef SCALE_TOYUV
|
||||||
|
#define SCALE_TOYUV(p) \
|
||||||
|
SCALE_(RGBtoYUV) (p)
|
||||||
|
|
||||||
|
#undef SCALE_CMPYUV
|
||||||
|
#define SCALE_CMPYUV(p1, p2, toler) \
|
||||||
|
SCALE_(CmpYUV) (p1, p2, toler)
|
||||||
|
|
||||||
|
#undef SCALE_GETY
|
||||||
|
#define SCALE_GETY(p) \
|
||||||
|
SCALE_(GetPixY) (p)
|
||||||
|
|
||||||
|
// MMX transformation multipliers
|
||||||
|
extern Uint64 mmx_888to555_mult;
|
||||||
|
extern Uint64 mmx_Y_mult;
|
||||||
|
extern Uint64 mmx_U_mult;
|
||||||
|
extern Uint64 mmx_V_mult;
|
||||||
|
extern Uint64 mmx_YUV_threshold;
|
||||||
|
|
||||||
|
#define USE_YUV_LOOKUP
|
||||||
|
|
||||||
|
#if defined(MSVC_ASM)
|
||||||
|
// MSVC inline assembly versions
|
||||||
|
|
||||||
|
#if defined(USE_MOVNTQ)
|
||||||
|
# define MOVNTQ(addr, val) movntq [addr], val
|
||||||
|
#else
|
||||||
|
# define MOVNTQ(addr, val) movq [addr], val
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#if USE_PREFETCH == INTEL_PREFETCH
|
||||||
|
// using Intel SSE non-temporal prefetch
|
||||||
|
# define PREFETCH(addr) prefetchnta [addr]
|
||||||
|
# define HAVE_PREFETCH
|
||||||
|
#elif USE_PREFETCH == AMD_PREFETCH
|
||||||
|
// using AMD 3DNOW! prefetch
|
||||||
|
# define PREFETCH(addr) prefetch [addr]
|
||||||
|
# define HAVE_PREFETCH
|
||||||
|
#else
|
||||||
|
// no prefetch -- too bad for poor MMX-only souls
|
||||||
|
# define PREFETCH(addr)
|
||||||
|
# undef HAVE_PREFETCH
|
||||||
|
#endif
|
||||||
|
|
||||||
|
|
||||||
|
static inline void
|
||||||
|
SCALE_(PlatInit) (void)
|
||||||
|
{
|
||||||
|
__asm
|
||||||
|
{
|
||||||
|
// mm0 will be kept == 0 throughout
|
||||||
|
// 0 is needed for bytes->words unpack instructions
|
||||||
|
pxor mm0, mm0
|
||||||
|
|
||||||
|
// mm5-mm7 contains RGB->YUV mults
|
||||||
|
movq mm7, mmx_Y_mult
|
||||||
|
movq mm6, mmx_U_mult
|
||||||
|
movq mm5, mmx_V_mult
|
||||||
|
#ifdef USE_YUV_LOOKUP
|
||||||
|
// mm6 contains RGB888->555 shuffle mult
|
||||||
|
movq mm6, mmx_888to555_mult
|
||||||
|
// mm5 contains DiffYUV threshold
|
||||||
|
movq mm5, mmx_YUV_threshold
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline void
|
||||||
|
SCALE_(PlatDone) (void)
|
||||||
|
{
|
||||||
|
// finish with MMX registers and yield them to FPU
|
||||||
|
__asm
|
||||||
|
{
|
||||||
|
emms
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#if defined(HAVE_PREFETCH)
|
||||||
|
static inline void
|
||||||
|
SCALE_(Prefetch) (const void* p)
|
||||||
|
{
|
||||||
|
__asm
|
||||||
|
{
|
||||||
|
mov eax, p
|
||||||
|
PREFETCH (eax)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#else /* Not HAVE_PREFETCH */
|
||||||
|
|
||||||
|
static inline void
|
||||||
|
SCALE_(Prefetch) (const void* p) { /* no-op */ }
|
||||||
|
|
||||||
|
#endif /* HAVE_PREFETCH */
|
||||||
|
|
||||||
|
// compute the RGB distance squared between 2 pixels
|
||||||
|
static inline int
|
||||||
|
SCALE_(GetRGBDelta) (Uint32 pix1, Uint32 pix2)
|
||||||
|
{
|
||||||
|
__asm
|
||||||
|
{
|
||||||
|
// load pixels
|
||||||
|
movd mm1, pix1
|
||||||
|
punpcklbw mm1, mm0
|
||||||
|
movd mm2, pix2
|
||||||
|
punpcklbw mm2, mm0
|
||||||
|
// get the difference between RGBA components
|
||||||
|
psubw mm1, mm2
|
||||||
|
// squared and sumed
|
||||||
|
pmaddwd mm1, mm1
|
||||||
|
// finish suming the squares
|
||||||
|
movq mm2, mm1
|
||||||
|
punpckhdq mm2, mm0
|
||||||
|
paddd mm1, mm2
|
||||||
|
// store result
|
||||||
|
movd eax, mm1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// retrieve the Y (intensity) component of pixel's YUV
|
||||||
|
static inline int
|
||||||
|
SCALE_(GetPixY) (Uint32 pix)
|
||||||
|
{
|
||||||
|
__asm
|
||||||
|
{
|
||||||
|
// load pixel
|
||||||
|
movd mm1, pix
|
||||||
|
punpcklbw mm1, mm0
|
||||||
|
// process
|
||||||
|
pmaddwd mm1, mmx_Y_mult // RGB * Yvec
|
||||||
|
movq mm2, mm1 // finish suming
|
||||||
|
punpckhdq mm2, mm0 // ditto
|
||||||
|
paddd mm1, mm2 // ditto
|
||||||
|
// store result
|
||||||
|
movd eax, mm1
|
||||||
|
shr eax, 14
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#ifdef USE_YUV_LOOKUP
|
||||||
|
|
||||||
|
// convert pixel RGB vector into YUV representation vector
|
||||||
|
static inline YUV_VECTOR
|
||||||
|
SCALE_(RGBtoYUV) (Uint32 pix)
|
||||||
|
{
|
||||||
|
__asm
|
||||||
|
{
|
||||||
|
// convert RGB888 to 555
|
||||||
|
movd mm1, pix
|
||||||
|
punpcklbw mm1, mm0
|
||||||
|
psrlw mm1, 3 // 8->5 bit
|
||||||
|
pmaddwd mm1, mmx_888to555_mult // shuffle into the right channel order
|
||||||
|
movq mm2, mm1 // finish shuffling
|
||||||
|
punpckhdq mm2, mm0 // ditto
|
||||||
|
por mm1, mm2 // ditto
|
||||||
|
|
||||||
|
// lookup the YUV vector
|
||||||
|
movd eax, mm1
|
||||||
|
mov eax, [RGB15_to_YUV + eax * 4]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// compare 2 pixels with respect to their YUV representations
|
||||||
|
// tolerance set by toler arg
|
||||||
|
// returns true: close; false: distant (-gt toler)
|
||||||
|
static inline bool
|
||||||
|
SCALE_(CmpYUV) (Uint32 pix1, Uint32 pix2, int toler)
|
||||||
|
{
|
||||||
|
__asm
|
||||||
|
{
|
||||||
|
// convert RGB888 to 555
|
||||||
|
movd mm1, pix1
|
||||||
|
punpcklbw mm1, mm0
|
||||||
|
psrlw mm1, 3 // 8->5 bit
|
||||||
|
movd mm3, pix2
|
||||||
|
punpcklbw mm3, mm0
|
||||||
|
psrlw mm3, 3 // 8->5 bit
|
||||||
|
pmaddwd mm1, mmx_888to555_mult // shuffle into the right channel order
|
||||||
|
movq mm2, mm1 // finish shuffling
|
||||||
|
pmaddwd mm3, mmx_888to555_mult // shuffle into the right channel order
|
||||||
|
movq mm4, mm3 // finish shuffling
|
||||||
|
punpckhdq mm2, mm0 // ditto
|
||||||
|
por mm1, mm2 // ditto
|
||||||
|
punpckhdq mm4, mm0 // ditto
|
||||||
|
por mm3, mm4 // ditto
|
||||||
|
|
||||||
|
// lookup the YUV vector
|
||||||
|
movd eax, mm1
|
||||||
|
movd edx, mm3
|
||||||
|
movd mm1, [RGB15_to_YUV + eax * 4]
|
||||||
|
movq mm4, mm1
|
||||||
|
movd mm2, [RGB15_to_YUV + edx * 4]
|
||||||
|
|
||||||
|
// get abs difference between YUV components
|
||||||
|
#ifdef USE_PSADBW
|
||||||
|
// we can use PSADBW and save us some grief
|
||||||
|
psadbw mm1, mm2
|
||||||
|
movd edx, mm1
|
||||||
|
#else
|
||||||
|
// no PSADBW -- have to do it the hard way
|
||||||
|
psubusb mm1, mm2
|
||||||
|
psubusb mm2, mm4
|
||||||
|
por mm1, mm2
|
||||||
|
|
||||||
|
// sum the differences
|
||||||
|
// XXX: technically, this produces a MAX diff of 510
|
||||||
|
// but we do not need anything bigger, currently
|
||||||
|
movq mm2, mm1
|
||||||
|
psrlq mm2, 8
|
||||||
|
paddusb mm1, mm2
|
||||||
|
psrlq mm2, 8
|
||||||
|
paddusb mm1, mm2
|
||||||
|
movd edx, mm1
|
||||||
|
and edx, 0xff
|
||||||
|
#endif /* USE_PSADBW */
|
||||||
|
xor eax, eax
|
||||||
|
shl edx, 1
|
||||||
|
cmp edx, toler
|
||||||
|
// store result
|
||||||
|
setle al
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#else /* Not USE_YUV_LOOKUP */
|
||||||
|
|
||||||
|
// convert pixel RGB vector into YUV representation vector
|
||||||
|
static inline YUV_VECTOR
|
||||||
|
SCALE_(RGBtoYUV) (Uint32 pix)
|
||||||
|
{
|
||||||
|
__asm
|
||||||
|
{
|
||||||
|
movd mm1, pix
|
||||||
|
punpcklbw mm1, mm0
|
||||||
|
|
||||||
|
movq mm2, mm1
|
||||||
|
|
||||||
|
// Y vector multiply
|
||||||
|
pmaddwd mm1, mmx_Y_mult
|
||||||
|
movq mm4, mm1
|
||||||
|
punpckhdq mm4, mm0
|
||||||
|
punpckldq mm1, mm0 // clear out the high dword
|
||||||
|
paddd mm1, mm4
|
||||||
|
psrad mm1, 15
|
||||||
|
|
||||||
|
movq mm3, mm2
|
||||||
|
|
||||||
|
// U vector multiply
|
||||||
|
pmaddwd mm2, mmx_U_mult
|
||||||
|
psrad mm2, 10
|
||||||
|
|
||||||
|
// V vector multiply
|
||||||
|
pmaddwd mm3, mmx_V_mult
|
||||||
|
psrad mm3, 10
|
||||||
|
|
||||||
|
// load (1|1|1|1) into mm4
|
||||||
|
pcmpeqw mm4, mm4
|
||||||
|
psrlw mm4, 15
|
||||||
|
|
||||||
|
packssdw mm3, mm2
|
||||||
|
pmaddwd mm3, mm4
|
||||||
|
psrad mm3, 5
|
||||||
|
|
||||||
|
// load (64|64) into mm4
|
||||||
|
punpcklwd mm4, mm0
|
||||||
|
pslld mm4, 6
|
||||||
|
paddd mm3, mm4
|
||||||
|
|
||||||
|
packssdw mm3, mm1
|
||||||
|
packuswb mm3, mm0
|
||||||
|
|
||||||
|
movd eax, mm3
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// compare 2 pixels with respect to their YUV representations
|
||||||
|
// tolerance set by toler arg
|
||||||
|
// returns true: close; false: distant (-gt toler)
|
||||||
|
static inline bool
|
||||||
|
SCALE_(CmpYUV) (Uint32 pix1, Uint32 pix2, int toler)
|
||||||
|
{
|
||||||
|
__asm
|
||||||
|
{
|
||||||
|
movd mm1, pix1
|
||||||
|
punpcklbw mm1, mm0
|
||||||
|
movd mm2, pix2
|
||||||
|
punpcklbw mm2, mm0
|
||||||
|
|
||||||
|
psubw mm1, mm2
|
||||||
|
movq mm2, mm1
|
||||||
|
|
||||||
|
// Y vector multiply
|
||||||
|
pmaddwd mm1, mmx_Y_mult
|
||||||
|
movq mm4, mm1
|
||||||
|
punpckhdq mm4, mm0
|
||||||
|
paddd mm1, mm4
|
||||||
|
// abs()
|
||||||
|
movq mm4, mm1
|
||||||
|
psrad mm4, 31
|
||||||
|
pxor mm4, mm1
|
||||||
|
psubd mm1, mm4
|
||||||
|
|
||||||
|
movq mm3, mm2
|
||||||
|
|
||||||
|
// U vector multiply
|
||||||
|
pmaddwd mm2, mmx_U_mult
|
||||||
|
movq mm4, mm2
|
||||||
|
punpckhdq mm4, mm0
|
||||||
|
paddd mm2, mm4
|
||||||
|
// abs()
|
||||||
|
movq mm4, mm2
|
||||||
|
psrad mm4, 31
|
||||||
|
pxor mm4, mm2
|
||||||
|
psubd mm2, mm4
|
||||||
|
|
||||||
|
paddd mm1, mm2
|
||||||
|
|
||||||
|
// V vector multiply
|
||||||
|
pmaddwd mm3, mmx_V_mult
|
||||||
|
movq mm4, mm3
|
||||||
|
punpckhdq mm3, mm0
|
||||||
|
paddd mm3, mm4
|
||||||
|
// abs()
|
||||||
|
movq mm4, mm3
|
||||||
|
psrad mm4, 31
|
||||||
|
pxor mm4, mm3
|
||||||
|
psubd mm3, mm4
|
||||||
|
|
||||||
|
paddd mm1, mm3
|
||||||
|
|
||||||
|
movd edx, mm1
|
||||||
|
xor eax, eax
|
||||||
|
shr edx, 14
|
||||||
|
cmp edx, toler
|
||||||
|
// store result
|
||||||
|
setle al
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif /* USE_YUV_LOOKUP */
|
||||||
|
|
||||||
|
// Check if 2 pixels are different with respect to their
|
||||||
|
// YUV representations
|
||||||
|
// returns 0: close; ~0: distant
|
||||||
|
static inline int
|
||||||
|
SCALE_(DiffYUV) (Uint32 yuv1, Uint32 yuv2)
|
||||||
|
{
|
||||||
|
__asm
|
||||||
|
{
|
||||||
|
// load YUV pixels
|
||||||
|
movd mm1, yuv1
|
||||||
|
movq mm4, mm1
|
||||||
|
movd mm2, yuv2
|
||||||
|
// abs difference between channels
|
||||||
|
psubusb mm1, mm2
|
||||||
|
psubusb mm2, mm4
|
||||||
|
por mm1, mm2
|
||||||
|
// compare to threshold
|
||||||
|
psubusb mm1, mmx_YUV_threshold
|
||||||
|
|
||||||
|
movd edx, mm1
|
||||||
|
// transform eax to 0 or ~0
|
||||||
|
xor eax, eax
|
||||||
|
or edx, edx
|
||||||
|
setz al
|
||||||
|
dec eax
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// bilinear weighted blend of four pixels
|
||||||
|
// MSVC asm version
|
||||||
|
static inline void
|
||||||
|
SCALE_(Blend_bilinear) (const Uint32* row0, const Uint32* row1,
|
||||||
|
Uint32* dst_p, Uint32 dlen)
|
||||||
|
{
|
||||||
|
__asm
|
||||||
|
{
|
||||||
|
// EL0: setup vars
|
||||||
|
mov ebx, row0 // EL0
|
||||||
|
|
||||||
|
// EL0: load pixels
|
||||||
|
movq mm1, [ebx] // EL0
|
||||||
|
movq mm2, mm1 // EL0: p[1] -> mm2
|
||||||
|
PREFETCH (ebx + 0x80)
|
||||||
|
punpckhbw mm2, mm0 // EL0: p[1] -> mm2
|
||||||
|
mov ebx, row1
|
||||||
|
punpcklbw mm1, mm0 // EL0: p[0] -> mm1
|
||||||
|
movq mm3, [ebx]
|
||||||
|
movq mm4, mm3 // EL0: p[3] -> mm4
|
||||||
|
movq mm6, mm2 // EL1.1: p[1] -> mm6
|
||||||
|
PREFETCH (ebx + 0x80)
|
||||||
|
punpcklbw mm3, mm0 // EL0: p[2] -> mm3
|
||||||
|
movq mm5, mm1 // EL1.1: p[0] -> mm5
|
||||||
|
punpckhbw mm4, mm0 // EL0: p[3] -> mm4
|
||||||
|
|
||||||
|
mov edi, dst_p // EL0
|
||||||
|
|
||||||
|
// EL1: cache p[0] + 3*(p[1] + p[2]) + p[3] in mm6
|
||||||
|
paddw mm6, mm3 // EL1.2: p[1] + p[2] -> mm6
|
||||||
|
// EL1: cache p[0] + p[1] + p[2] + p[3] in mm7
|
||||||
|
movq mm7, mm6 // EL1.3: p[1] + p[2] -> mm7
|
||||||
|
// EL1: cache p[1] + 3*(p[0] + p[3]) + p[2] in mm5
|
||||||
|
paddw mm5, mm4 // EL1.2: p[0] + p[3] -> mm5
|
||||||
|
psllw mm6, 1 // EL1.4: 2*(p[1] + p[2]) -> mm6
|
||||||
|
paddw mm7, mm5 // EL1.4: sum(p[]) -> mm7
|
||||||
|
psllw mm5, 1 // EL1.5: 2*(p[0] + p[3]) -> mm5
|
||||||
|
paddw mm6, mm7 // EL1.5: p[0] + 3*(p[1] + p[2]) + p[3] -> mm6
|
||||||
|
paddw mm5, mm7 // EL1.6: p[1] + 3*(p[0] + p[3]) + p[2] -> mm5
|
||||||
|
|
||||||
|
// EL2: pixel 0 math -- (9*p[0] + 3*(p[1] + p[2]) + p[3]) / 16
|
||||||
|
psllw mm1, 3 // EL2.1: 8*p[0] -> mm1
|
||||||
|
paddw mm1, mm6 // EL2.2: 9*p[0] + 3*(p[1] + p[2]) + p[3] -> mm1
|
||||||
|
psrlw mm1, 4 // EL2.3: sum[0]/16 -> mm1
|
||||||
|
|
||||||
|
mov edx, dlen // EL0
|
||||||
|
|
||||||
|
// EL3: pixel 1 math -- (9*p[1] + 3*(p[0] + p[3]) + p[2]) / 16
|
||||||
|
psllw mm2, 3 // EL3.1: 8*p[1] -> mm2
|
||||||
|
paddw mm2, mm5 // EL3.2: 9*p[1] + 3*(p[0] + p[3]) + p[2] -> mm2
|
||||||
|
psrlw mm2, 4 // EL3.3: sum[1]/16 -> mm5
|
||||||
|
|
||||||
|
// EL2/3: store pixels 0 & 1
|
||||||
|
packuswb mm1, mm2 // EL2/3: pack into bytes
|
||||||
|
MOVNTQ (edi, mm1) // EL2/3: store 2 pixels
|
||||||
|
|
||||||
|
// EL4: pixel 2 math -- (9*p[2] + 3*(p[0] + p[3]) + p[1]) / 16
|
||||||
|
psllw mm3, 3 // EL4.1: 8*p[2] -> mm3
|
||||||
|
paddw mm3, mm5 // EL4.2: 9*p[2] + 3*(p[0] + p[3]) + p[1] -> mm3
|
||||||
|
psrlw mm3, 4 // EL4.3: sum[2]/16 -> mm3
|
||||||
|
|
||||||
|
// EL5: pixel 3 math -- (9*p[3] + 3*(p[1] + p[2]) + p[0]) / 16
|
||||||
|
psllw mm4, 3 // EL5.1: 8*p[3] -> mm4
|
||||||
|
paddw mm4, mm6 // EL5.2: 9*p[3] + 3*(p[1] + p[2]) + p[0] -> mm4
|
||||||
|
psrlw mm4, 4 // EL5.3: sum[3]/16 -> mm4
|
||||||
|
|
||||||
|
// EL4/5: store pixels 2 & 3
|
||||||
|
packuswb mm3, mm4 // EL4/5: pack into bytes
|
||||||
|
MOVNTQ (edi + edx*4, mm3) // EL4/5: store 2 pixels
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// End MSVC_ASM
|
||||||
|
|
||||||
|
#elif defined(GCC_ASM)
|
||||||
|
// GCC inline assembly versions
|
||||||
|
|
||||||
|
#if defined(USE_MOVNTQ)
|
||||||
|
# define MOVNTQ(val, addr) "movntq " #val "," #addr
|
||||||
|
#else
|
||||||
|
# define MOVNTQ(val, addr) "movq " #val "," #addr
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#if USE_PREFETCH == INTEL_PREFETCH
|
||||||
|
// using Intel SSE non-temporal prefetch
|
||||||
|
# define PREFETCH(addr) "prefetchnta " #addr
|
||||||
|
#elif USE_PREFETCH == AMD_PREFETCH
|
||||||
|
// using AMD 3DNOW! prefetch
|
||||||
|
# define PREFETCH(addr) "prefetch " #addr
|
||||||
|
#else
|
||||||
|
// no prefetch -- too bad for poor MMX-only souls
|
||||||
|
# define PREFETCH(addr)
|
||||||
|
#endif
|
||||||
|
|
||||||
|
static inline void
|
||||||
|
SCALE_(PlatInit) (void)
|
||||||
|
{
|
||||||
|
__asm__ (
|
||||||
|
// mm0 will be kept == 0 throughout
|
||||||
|
// 0 is needed for bytes->words unpack instructions
|
||||||
|
"pxor %%mm0, %%mm0 \n\t"
|
||||||
|
|
||||||
|
// mm5-mm7 contains RGB->YUV mults
|
||||||
|
"movq %0, %%mm7 \n\t"
|
||||||
|
"movq %1, %%mm6 \n\t"
|
||||||
|
"movq %2, %%mm5 \n\t"
|
||||||
|
#ifdef USE_YUV_LOOKUP
|
||||||
|
// mm6 contains RGB888->555 shuffle mult
|
||||||
|
"movq %3, %%mm6 \n\t"
|
||||||
|
// mm5 contains DiffYUV threshold
|
||||||
|
"movq %4, %%mm5 \n\t"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
: /* nothing */
|
||||||
|
: /*0*/"m" (mmx_Y_mult), /*1*/"m" (mmx_U_mult), /*2*/"m" (mmx_V_mult)
|
||||||
|
, /*3*/"m" (mmx_888to555_mult), /*4*/"m" (mmx_YUV_threshold)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline void
|
||||||
|
SCALE_(PlatDone) (void)
|
||||||
|
{
|
||||||
|
// finish with MMX registers and yield them to FPU
|
||||||
|
__asm__ (
|
||||||
|
"emms \n\t"
|
||||||
|
: /* nothing */ : /* nothing */
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline void
|
||||||
|
SCALE_(Prefetch) (const void* p)
|
||||||
|
{
|
||||||
|
__asm__ __volatile__ ("" PREFETCH (%0) : /*nothing*/ : "m" (p) );
|
||||||
|
}
|
||||||
|
|
||||||
|
// compute the RGB distance squared between 2 pixels
|
||||||
|
static inline int
|
||||||
|
SCALE_(GetRGBDelta) (Uint32 pix1, Uint32 pix2)
|
||||||
|
{
|
||||||
|
int res;
|
||||||
|
|
||||||
|
__asm__ (
|
||||||
|
// load pixels
|
||||||
|
"movd %1, %%mm1 \n\t"
|
||||||
|
"punpcklbw %%mm0, %%mm1 \n\t"
|
||||||
|
"movd %2, %%mm2 \n\t"
|
||||||
|
"punpcklbw %%mm0, %%mm2 \n\t"
|
||||||
|
// get the difference between RGBA components
|
||||||
|
"psubw %%mm2, %%mm1 \n\t"
|
||||||
|
// squared and sumed
|
||||||
|
"pmaddwd %%mm1, %%mm1 \n\t"
|
||||||
|
// finish suming the squares
|
||||||
|
"movq %%mm1, %%mm2 \n\t"
|
||||||
|
"punpckhdq %%mm0, %%mm2 \n\t"
|
||||||
|
"paddd %%mm2, %%mm1 \n\t"
|
||||||
|
// store result
|
||||||
|
"movd %%mm1, %0 \n\t"
|
||||||
|
|
||||||
|
: /*0*/"=r" (res)
|
||||||
|
: /*1*/"rm" (pix1), /*2*/"rm" (pix2)
|
||||||
|
);
|
||||||
|
|
||||||
|
return res;
|
||||||
|
}
|
||||||
|
|
||||||
|
// retrieve the Y (intensity) component of pixel's YUV
|
||||||
|
static inline int
|
||||||
|
SCALE_(GetPixY) (Uint32 pix)
|
||||||
|
{
|
||||||
|
int ret;
|
||||||
|
|
||||||
|
__asm__ (
|
||||||
|
// load pixel
|
||||||
|
"movd %1, %%mm1 \n\t"
|
||||||
|
"punpcklbw %%mm0, %%mm1 \n\t"
|
||||||
|
// process
|
||||||
|
"pmaddwd %2, %%mm1 \n\t" // R,G,B * Yvec
|
||||||
|
"movq %%mm1, %%mm2 \n\t" // finish suming
|
||||||
|
"punpckhdq %%mm0, %%mm2 \n\t" // ditto
|
||||||
|
"paddd %%mm2, %%mm1 \n\t" // ditto
|
||||||
|
// store index
|
||||||
|
"movd %%mm1, %0 \n\t"
|
||||||
|
|
||||||
|
: /*0*/"=r" (ret)
|
||||||
|
: /*1*/"rm" (pix), /*2*/"m" (mmx_Y_mult)
|
||||||
|
);
|
||||||
|
return ret >> 14;
|
||||||
|
}
|
||||||
|
|
||||||
|
#ifdef USE_YUV_LOOKUP
|
||||||
|
|
||||||
|
// convert pixel RGB vector into YUV representation vector
|
||||||
|
static inline YUV_VECTOR
|
||||||
|
SCALE_(RGBtoYUV) (Uint32 pix)
|
||||||
|
{
|
||||||
|
int i;
|
||||||
|
|
||||||
|
__asm__ (
|
||||||
|
// convert RGB888 to 555
|
||||||
|
"movd %1, %%mm1 \n\t"
|
||||||
|
"punpcklbw %%mm0, %%mm1 \n\t"
|
||||||
|
"psrlw $3, %%mm1 \n\t" // 8->5 bit
|
||||||
|
"pmaddwd %2, %%mm1 \n\t" // shuffle into the right channel order
|
||||||
|
"movq %%mm1, %%mm2 \n\t" // finish shuffling
|
||||||
|
"punpckhdq %%mm0, %%mm2 \n\t" // ditto
|
||||||
|
"por %%mm2, %%mm1 \n\t" // ditto
|
||||||
|
"movd %%mm1, %0 \n\t"
|
||||||
|
|
||||||
|
: /*0*/"=r" (i)
|
||||||
|
: /*1*/"rm" (pix), /*2*/"m" (mmx_888to555_mult)
|
||||||
|
);
|
||||||
|
return RGB15_to_YUV[i];
|
||||||
|
}
|
||||||
|
|
||||||
|
// compare 2 pixels with respect to their YUV representations
|
||||||
|
// tolerance set by toler arg
|
||||||
|
// returns true: close; false: distant (-gt toler)
|
||||||
|
static inline bool
|
||||||
|
SCALE_(CmpYUV) (Uint32 pix1, Uint32 pix2, int toler)
|
||||||
|
{
|
||||||
|
int delta;
|
||||||
|
|
||||||
|
__asm__ (
|
||||||
|
"movd %1, %%mm1 \n\t"
|
||||||
|
"movd %2, %%mm3 \n\t"
|
||||||
|
|
||||||
|
// convert RGB888 to 555
|
||||||
|
// this is somewhat parallelized
|
||||||
|
"punpcklbw %%mm0, %%mm1 \n\t"
|
||||||
|
"psrlw $3, %%mm1 \n\t" // 8->5 bit
|
||||||
|
"punpcklbw %%mm0, %%mm3 \n\t"
|
||||||
|
"psrlw $3, %%mm3 \n\t" // 8->5 bit
|
||||||
|
"pmaddwd %4, %%mm1 \n\t" // shuffle into the right channel order
|
||||||
|
"movq %%mm1, %%mm2 \n\t" // finish shuffling
|
||||||
|
"pmaddwd %4, %%mm3 \n\t" // shuffle into the right channel order
|
||||||
|
"movq %%mm3, %%mm4 \n\t" // finish shuffling
|
||||||
|
"punpckhdq %%mm0, %%mm2 \n\t" // ditto
|
||||||
|
"por %%mm2, %%mm1 \n\t" // ditto
|
||||||
|
"punpckhdq %%mm0, %%mm4 \n\t" // ditto
|
||||||
|
"por %%mm4, %%mm3 \n\t" // ditto
|
||||||
|
|
||||||
|
// lookup the YUV vector
|
||||||
|
"movd %%mm1, %%eax \n\t"
|
||||||
|
"movd %%mm3, %%edx \n\t"
|
||||||
|
"movd %3(,%%eax,4), %%mm1 \n\t"
|
||||||
|
"movq %%mm1, %%mm4 \n\t"
|
||||||
|
"movd %3(,%%edx,4), %%mm2 \n\t"
|
||||||
|
|
||||||
|
// get abs difference between YUV components
|
||||||
|
#ifdef USE_PSADBW
|
||||||
|
// we can use PSADBW and save us some grief
|
||||||
|
"psadbw %%mm2, %%mm1 \n\t"
|
||||||
|
"movd %%mm1, %0 \n\t"
|
||||||
|
#else
|
||||||
|
// no PSADBW -- have to do it the hard way
|
||||||
|
"psubusb %%mm2, %%mm1 \n\t"
|
||||||
|
"psubusb %%mm4, %%mm2 \n\t"
|
||||||
|
"por %%mm2, %%mm1 \n\t"
|
||||||
|
|
||||||
|
// sum the differences
|
||||||
|
// technically, this produces a MAX diff of 510
|
||||||
|
// but we do not need anything bigger, currently
|
||||||
|
"movq %%mm1, %%mm2 \n\t"
|
||||||
|
"psrlq $8, %%mm2 \n\t"
|
||||||
|
"paddusb %%mm2, %%mm1 \n\t"
|
||||||
|
"psrlq $8, %%mm2 \n\t"
|
||||||
|
"paddusb %%mm2, %%mm1 \n\t"
|
||||||
|
// store intermediate delta
|
||||||
|
"movd %%mm1, %0 \n\t"
|
||||||
|
"andl $0xff, %0 \n\t"
|
||||||
|
#endif /* USE_PSADBW */
|
||||||
|
: /*0*/"=r" (delta)
|
||||||
|
: /*1*/"rm" (pix1), /*2*/"rm" (pix2),
|
||||||
|
/*3*/"m" (*RGB15_to_YUV), /*4*/"m" (mmx_888to555_mult)
|
||||||
|
: "%eax", "%edx"
|
||||||
|
);
|
||||||
|
|
||||||
|
return (delta << 1) <= toler;
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif /* USE_YUV_LOOKUP */
|
||||||
|
|
||||||
|
// Check if 2 pixels are different with respect to their
|
||||||
|
// YUV representations
|
||||||
|
// returns 0: close; ~0: distant
|
||||||
|
static inline int
|
||||||
|
SCALE_(DiffYUV) (Uint32 yuv1, Uint32 yuv2)
|
||||||
|
{
|
||||||
|
sint32 ret;
|
||||||
|
|
||||||
|
__asm__ (
|
||||||
|
// load YUV pixels
|
||||||
|
"movd %1, %%mm1 \n\t"
|
||||||
|
"movq %%mm1, %%mm4 \n\t"
|
||||||
|
"movd %2, %%mm2 \n\t"
|
||||||
|
// abs difference between channels
|
||||||
|
"psubusb %%mm2, %%mm1 \n\t"
|
||||||
|
"psubusb %%mm4, %%mm2 \n\t"
|
||||||
|
"por %%mm2, %%mm1 \n\t"
|
||||||
|
// compare to threshold
|
||||||
|
"psubusb %3, %%mm1 \n\t"
|
||||||
|
|
||||||
|
"movd %%mm1, %%edx \n\t"
|
||||||
|
// transform eax to 0 or ~0
|
||||||
|
"xor %%eax, %%eax \n\t"
|
||||||
|
"or %%edx, %%edx \n\t"
|
||||||
|
"setz %%al \n\t"
|
||||||
|
"dec %%eax \n\t"
|
||||||
|
|
||||||
|
: /*0*/"=a" (ret)
|
||||||
|
: /*1*/"rm" (yuv1), /*2*/"rm" (yuv2),
|
||||||
|
/*3*/"m" (mmx_YUV_threshold)
|
||||||
|
: "%edx"
|
||||||
|
);
|
||||||
|
return ret;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Bilinear weighted blend of four pixels
|
||||||
|
// Function produces 4 blended pixels (in 2x2 matrix) and writes them
|
||||||
|
// out to the surface
|
||||||
|
// Last version
|
||||||
|
static inline void
|
||||||
|
SCALE_(Blend_bilinear) (const Uint32* row0, const Uint32* row1,
|
||||||
|
Uint32* dst_p, Uint32 dlen)
|
||||||
|
{
|
||||||
|
__asm__ (
|
||||||
|
// EL0: load pixels
|
||||||
|
"movq %0, %%mm1 \n\t" // EL0
|
||||||
|
"movq %%mm1, %%mm2 \n\t" // EL0: p[1] -> mm2
|
||||||
|
PREFETCH (0x80%0) "\n\t"
|
||||||
|
"punpckhbw %%mm0, %%mm2 \n\t" // EL0: p[1] -> mm2
|
||||||
|
"punpcklbw %%mm0, %%mm1 \n\t" // EL0: p[0] -> mm1
|
||||||
|
"movq %1, %%mm3 \n\t"
|
||||||
|
"movq %%mm3, %%mm4 \n\t" // EL0: p[3] -> mm4
|
||||||
|
"movq %%mm2, %%mm6 \n\t" // EL1.1: p[1] -> mm6
|
||||||
|
PREFETCH (0x80%1) "\n\t"
|
||||||
|
"punpcklbw %%mm0, %%mm3 \n\t" // EL0: p[2] -> mm3
|
||||||
|
"movq %%mm1, %%mm5 \n\t" // EL1.1: p[0] -> mm5
|
||||||
|
"punpckhbw %%mm0, %%mm4 \n\t" // EL0: p[3] -> mm4
|
||||||
|
|
||||||
|
// EL1: cache p[0] + 3*(p[1] + p[2]) + p[3] in mm6
|
||||||
|
"paddw %%mm3, %%mm6 \n\t" // EL1.2: p[1] + p[2] -> mm6
|
||||||
|
// EL1: cache p[0] + p[1] + p[2] + p[3] in mm7
|
||||||
|
"movq %%mm6, %%mm7 \n\t" // EL1.3: p[1] + p[2] -> mm7
|
||||||
|
// EL1: cache p[1] + 3*(p[0] + p[3]) + p[2] in mm5
|
||||||
|
"paddw %%mm4, %%mm5 \n\t" // EL1.2: p[0] + p[3] -> mm5
|
||||||
|
"psllw $1, %%mm6 \n\t" // EL1.4: 2*(p[1] + p[2]) -> mm6
|
||||||
|
"paddw %%mm5, %%mm7 \n\t" // EL1.4: sum(p[]) -> mm7
|
||||||
|
"psllw $1, %%mm5 \n\t" // EL1.5: 2*(p[0] + p[3]) -> mm5
|
||||||
|
"paddw %%mm7, %%mm6 \n\t" // EL1.5: p[0] + 3*(p[1] + p[2]) + p[3] -> mm6
|
||||||
|
"paddw %%mm7, %%mm5 \n\t" // EL1.6: p[1] + 3*(p[0] + p[3]) + p[2] -> mm5
|
||||||
|
|
||||||
|
// EL2: pixel 0 math -- (9*p[0] + 3*(p[1] + p[2]) + p[3]) / 16
|
||||||
|
"psllw $3, %%mm1 \n\t" // EL2.1: 8*p[0] -> mm1
|
||||||
|
"paddw %%mm6, %%mm1 \n\t" // EL2.2: 9*p[0] + 3*(p[1] + p[2]) + p[3] -> mm1
|
||||||
|
"psrlw $4, %%mm1 \n\t" // EL2.3: sum[0]/16 -> mm1
|
||||||
|
|
||||||
|
// EL3: pixel 1 math -- (9*p[1] + 3*(p[0] + p[3]) + p[2]) / 16
|
||||||
|
"psllw $3, %%mm2 \n\t" // EL3.1: 8*p[1] -> mm2
|
||||||
|
"paddw %%mm5, %%mm2 \n\t" // EL3.2: 9*p[1] + 3*(p[0] + p[3]) + p[2] -> mm5
|
||||||
|
"psrlw $4, %%mm2 \n\t" // EL3.3: sum[1]/16 -> mm5
|
||||||
|
|
||||||
|
// EL2/4: store pixels 0 & 1
|
||||||
|
"packuswb %%mm2, %%mm1 \n\t" // EL2/4: pack into bytes
|
||||||
|
MOVNTQ (%%mm1, (%2)) "\n\t" // EL2/4: store 2 pixels
|
||||||
|
|
||||||
|
// EL4: pixel 2 math -- (9*p[2] + 3*(p[0] + p[3]) + p[1]) / 16
|
||||||
|
"psllw $3, %%mm3 \n\t" // EL4.1: 8*p[2] -> mm3
|
||||||
|
"paddw %%mm5, %%mm3 \n\t" // EL4.2: 9*p[2] + 3*(p[0] + p[3]) + p[1] -> mm3
|
||||||
|
"psrlw $4, %%mm3 \n\t" // EL4.3: sum[2]/16 -> mm3
|
||||||
|
|
||||||
|
// EL5: pixel 3 math -- (9*p[3] + 3*(p[1] + p[2]) + p[0]) / 16
|
||||||
|
"psllw $3, %%mm4 \n\t" // EL5.1: 8*p[3] -> mm4
|
||||||
|
"paddw %%mm6, %%mm4 \n\t" // EL5.2: 9*p[3] + 3*(p[1] + p[2]) + p[0] -> mm4
|
||||||
|
"psrlw $4, %%mm4 \n\t" // EL5.3: sum[3]/16 -> mm4
|
||||||
|
|
||||||
|
// EL4/5: store pixels 2 & 3
|
||||||
|
"packuswb %%mm4, %%mm3 \n\t" // EL4/5: pack into bytes
|
||||||
|
MOVNTQ (%%mm3, (%2,%3,4)) "\n\t" // EL4/5: store 2 pixels
|
||||||
|
|
||||||
|
: /* nothing */
|
||||||
|
: /*0*/"m" (*row0), /*1*/"m" (*row1), /*2*/"r" (dst_p), /*3*/"r" (dlen)
|
||||||
|
: "memory"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif // GCC_ASM
|
||||||
|
|
||||||
|
#endif /* SCALEMMX_H_ */
|
||||||
Executable
+286
@@ -0,0 +1,286 @@
|
|||||||
|
/*
|
||||||
|
* This program is free software; you can redistribute it and/or modify
|
||||||
|
* it under the terms of the GNU General Public License as published by
|
||||||
|
* the Free Software Foundation; either version 2 of the License, or
|
||||||
|
* (at your option) any later version.
|
||||||
|
*
|
||||||
|
* This program is distributed in the hope that it will be useful,
|
||||||
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
* GNU General Public License for more details.
|
||||||
|
*
|
||||||
|
* You should have received a copy of the GNU General Public License
|
||||||
|
* along with this program; if not, write to the Free Software
|
||||||
|
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||||
|
*/
|
||||||
|
|
||||||
|
#ifdef GFXMODULE_SDL
|
||||||
|
|
||||||
|
#include "types.h"
|
||||||
|
#include "libs/graphics/sdl/sdl_common.h"
|
||||||
|
#include SDL_INCLUDE(sdl_cpuinfo.h)
|
||||||
|
#include "libs/platform.h"
|
||||||
|
#include "scalers.h"
|
||||||
|
#include "scaleint.h"
|
||||||
|
#include "2xscalers.h"
|
||||||
|
#ifdef MMX_ASM
|
||||||
|
# include "2xscalers_mmx.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
|
||||||
|
typedef enum
|
||||||
|
{
|
||||||
|
SCALEPLAT_NULL = PLATFORM_NULL,
|
||||||
|
SCALEPLAT_C = PLATFORM_C,
|
||||||
|
SCALEPLAT_MMX = PLATFORM_MMX,
|
||||||
|
SCALEPLAT_SSE = PLATFORM_SSE,
|
||||||
|
SCALEPLAT_3DNOW = PLATFORM_3DNOW,
|
||||||
|
SCALEPLAT_ALTIVEC = PLATFORM_ALTIVEC,
|
||||||
|
|
||||||
|
SCALEPLAT_C_RGBA,
|
||||||
|
SCALEPLAT_C_BGRA,
|
||||||
|
SCALEPLAT_C_ARGB,
|
||||||
|
SCALEPLAT_C_ABGR,
|
||||||
|
|
||||||
|
} Scale_PlatType_t;
|
||||||
|
|
||||||
|
|
||||||
|
// RGB -> YUV transformation
|
||||||
|
// the RGB vector is multiplied by the transformation matrix
|
||||||
|
// to get the YUV vector
|
||||||
|
#if 0
|
||||||
|
// original table -- not used
|
||||||
|
const int YUV_matrix[3][3] =
|
||||||
|
{
|
||||||
|
/* Y U V */
|
||||||
|
/* R */ {0.2989, -0.1687, 0.5000},
|
||||||
|
/* G */ {0.5867, -0.3312, -0.4183},
|
||||||
|
/* B */ {0.1144, 0.5000, -0.0816}
|
||||||
|
};
|
||||||
|
#else
|
||||||
|
// scaled up by a 2^14 factor, with Y doubled
|
||||||
|
const int YUV_matrix[3][3] =
|
||||||
|
{
|
||||||
|
/* Y U V */
|
||||||
|
/* R */ { 9794, -2764, 8192},
|
||||||
|
/* G */ {19224, -5428, -6853},
|
||||||
|
/* B */ { 3749, 8192, -1339}
|
||||||
|
};
|
||||||
|
#endif
|
||||||
|
|
||||||
|
// pre-computed transformations for 8 bits per channel
|
||||||
|
int RGB_to_YUV[/*RGB*/ 3][/*YUV*/ 3][ /*mult-res*/ 256];
|
||||||
|
sint16 dRGB_to_dYUV[/*RGB*/ 3][/*YUV*/ 3][ /*mult-res*/ 512];
|
||||||
|
|
||||||
|
// pre-computed transformations for RGB555
|
||||||
|
YUV_VECTOR RGB15_to_YUV[0x8000];
|
||||||
|
|
||||||
|
PLATFORM_TYPE force_platform = PLATFORM_NULL;
|
||||||
|
Scale_PlatType_t Scale_Platform = SCALEPLAT_NULL;
|
||||||
|
|
||||||
|
|
||||||
|
// pre-compute the RGB->YUV transformations
|
||||||
|
void
|
||||||
|
Scale_Init (void)
|
||||||
|
{
|
||||||
|
int i1, i2, i3;
|
||||||
|
|
||||||
|
for (i1 = 0; i1 < 3; i1++) // enum R,G,B
|
||||||
|
for (i2 = 0; i2 < 3; i2++) // enum Y,U,V
|
||||||
|
for (i3 = 0; i3 < 256; i3++) // enum possible channel vals
|
||||||
|
{
|
||||||
|
RGB_to_YUV[i1][i2][i3] =
|
||||||
|
(YUV_matrix[i1][i2] * i3) >> 14;
|
||||||
|
}
|
||||||
|
|
||||||
|
for (i1 = 0; i1 < 3; i1++) // enum R,G,B
|
||||||
|
for (i2 = 0; i2 < 3; i2++) // enum Y,U,V
|
||||||
|
for (i3 = -255; i3 < 256; i3++) // enum possible channel delta vals
|
||||||
|
{
|
||||||
|
dRGB_to_dYUV[i1][i2][i3 + 255] =
|
||||||
|
(YUV_matrix[i1][i2] * i3) >> 14;
|
||||||
|
}
|
||||||
|
|
||||||
|
for (i1 = 0; i1 < 32; ++i1)
|
||||||
|
for (i2 = 0; i2 < 32; ++i2)
|
||||||
|
for (i3 = 0; i3 < 32; ++i3)
|
||||||
|
{
|
||||||
|
int y, u, v;
|
||||||
|
// adding upper bits halved for error correction
|
||||||
|
int r = (i1 << 3) | (i1 >> 3);
|
||||||
|
int g = (i2 << 3) | (i2 >> 3);
|
||||||
|
int b = (i3 << 3) | (i3 >> 3);
|
||||||
|
|
||||||
|
y = ( r * YUV_matrix[YUV_XFORM_R][YUV_XFORM_Y]
|
||||||
|
+ g * YUV_matrix[YUV_XFORM_G][YUV_XFORM_Y]
|
||||||
|
+ b * YUV_matrix[YUV_XFORM_B][YUV_XFORM_Y]
|
||||||
|
) >> 15; // we dont need Y doubled, need Y to fit 8 bits
|
||||||
|
|
||||||
|
// U and V are half the importance of Y
|
||||||
|
u = 64+(( r * YUV_matrix[YUV_XFORM_R][YUV_XFORM_U]
|
||||||
|
+ g * YUV_matrix[YUV_XFORM_G][YUV_XFORM_U]
|
||||||
|
+ b * YUV_matrix[YUV_XFORM_B][YUV_XFORM_U]
|
||||||
|
) >> 15); // halved
|
||||||
|
|
||||||
|
v = 64+(( r * YUV_matrix[YUV_XFORM_R][YUV_XFORM_V]
|
||||||
|
+ g * YUV_matrix[YUV_XFORM_G][YUV_XFORM_V]
|
||||||
|
+ b * YUV_matrix[YUV_XFORM_B][YUV_XFORM_V]
|
||||||
|
) >> 15); // halved
|
||||||
|
|
||||||
|
RGB15_to_YUV[(i1 << 10) | (i2 << 5) | i3] = (y << 16) | (u << 8) | v;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
// expands the given rectangle in all directions by 'expansion'
|
||||||
|
// guarded by 'limits'
|
||||||
|
void
|
||||||
|
Scale_ExpandRect (SDL_Rect* rect, int expansion, const SDL_Rect* limits)
|
||||||
|
{
|
||||||
|
if (rect->x - expansion >= limits->x)
|
||||||
|
{
|
||||||
|
rect->w += expansion;
|
||||||
|
rect->x -= expansion;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
rect->w += rect->x - limits->x;
|
||||||
|
rect->x = limits->x;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (rect->y - expansion >= limits->y)
|
||||||
|
{
|
||||||
|
rect->h += expansion;
|
||||||
|
rect->y -= expansion;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
rect->h += rect->y - limits->y;
|
||||||
|
rect->y = limits->y;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (rect->x + rect->w + expansion <= limits->w)
|
||||||
|
rect->w += expansion;
|
||||||
|
else
|
||||||
|
rect->w = limits->w - rect->x;
|
||||||
|
|
||||||
|
if (rect->y + rect->h + expansion <= limits->h)
|
||||||
|
rect->h += expansion;
|
||||||
|
else
|
||||||
|
rect->h = limits->h - rect->y;
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
// Platform+Scaler function lookups
|
||||||
|
|
||||||
|
typedef struct
|
||||||
|
{
|
||||||
|
Scale_PlatType_t platform;
|
||||||
|
const Scale_FuncDef_t* funcdefs;
|
||||||
|
} Scale_PlatDef_t;
|
||||||
|
|
||||||
|
|
||||||
|
const static Scale_PlatDef_t
|
||||||
|
Scale_PlatDefs[] =
|
||||||
|
{
|
||||||
|
#if defined(MMX_ASM)
|
||||||
|
{SCALEPLAT_SSE, Scale_SSE_Functions},
|
||||||
|
{SCALEPLAT_3DNOW, Scale_3DNow_Functions},
|
||||||
|
{SCALEPLAT_MMX, Scale_MMX_Functions},
|
||||||
|
#endif /* MMX_ASM */
|
||||||
|
// Default
|
||||||
|
{SCALEPLAT_NULL, Scale_C_Functions}
|
||||||
|
};
|
||||||
|
|
||||||
|
|
||||||
|
TFB_ScaleFunc
|
||||||
|
Scale_PrepPlatform (int flags, const SDL_PixelFormat* fmt)
|
||||||
|
{
|
||||||
|
const Scale_PlatDef_t* pdef;
|
||||||
|
const Scale_FuncDef_t* fdef;
|
||||||
|
|
||||||
|
(void)flags;
|
||||||
|
|
||||||
|
Scale_Platform = SCALEPLAT_NULL;
|
||||||
|
|
||||||
|
// XXX: Hack to test some code
|
||||||
|
Scale_MMX_PrepPlatform (fmt);
|
||||||
|
|
||||||
|
// first match wins
|
||||||
|
// add better platform techs to the top
|
||||||
|
#ifdef MMX_ASM
|
||||||
|
if ( (!force_platform && (SDL_HasSSE () || SDL_HasMMXExt ()))
|
||||||
|
|| force_platform == SCALEPLAT_SSE)
|
||||||
|
{
|
||||||
|
fprintf (stderr, "Screen scalers are using SSE/MMX-Ext/MMX code\n");
|
||||||
|
Scale_Platform = SCALEPLAT_SSE;
|
||||||
|
|
||||||
|
Scale_SSE_PrepPlatform (fmt);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
if ( (!force_platform && SDL_HasAltiVec ())
|
||||||
|
|| force_platform == SCALEPLAT_ALTIVEC)
|
||||||
|
{
|
||||||
|
fprintf (stderr, "Screen scalers would use AltiVec code "
|
||||||
|
"if someone actually wrote it\n");
|
||||||
|
//Scale_Platform = SCALEPLAT_ALTIVEC;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
if ( (!force_platform && SDL_Has3DNow ())
|
||||||
|
|| force_platform == SCALEPLAT_3DNOW)
|
||||||
|
{
|
||||||
|
fprintf (stderr, "Screen scalers are using 3DNow/MMX code\n");
|
||||||
|
Scale_Platform = SCALEPLAT_3DNOW;
|
||||||
|
|
||||||
|
Scale_3DNow_PrepPlatform (fmt);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
if ( (!force_platform && SDL_HasMMX ())
|
||||||
|
|| force_platform == SCALEPLAT_MMX)
|
||||||
|
{
|
||||||
|
fprintf (stderr, "Screen scalers are using MMX code\n");
|
||||||
|
Scale_Platform = SCALEPLAT_MMX;
|
||||||
|
|
||||||
|
Scale_MMX_PrepPlatform (fmt);
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
if (Scale_Platform == SCALEPLAT_NULL)
|
||||||
|
{ // Plain C versions
|
||||||
|
if (fmt->Rmask == 0xff000000)
|
||||||
|
Scale_Platform = SCALEPLAT_C_RGBA;
|
||||||
|
else if (fmt->Rmask == 0x00ff0000)
|
||||||
|
Scale_Platform = SCALEPLAT_C_ARGB;
|
||||||
|
else if (fmt->Rmask == 0x0000ff00)
|
||||||
|
Scale_Platform = SCALEPLAT_C_BGRA;
|
||||||
|
else if (fmt->Rmask == 0x000000ff)
|
||||||
|
Scale_Platform = SCALEPLAT_C_ABGR;
|
||||||
|
else
|
||||||
|
{ // use slowest default
|
||||||
|
fprintf (stderr, "Scale_PrepPlatform(): "
|
||||||
|
"unknown Red mask (0x%08x)\n", fmt->Rmask);
|
||||||
|
Scale_Platform = SCALEPLAT_C;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (Scale_Platform == SCALEPLAT_C)
|
||||||
|
fprintf (stderr, "Screen scalers are using slow generic C code\n");
|
||||||
|
else
|
||||||
|
fprintf (stderr, "Screen scalers are using optimized C code\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
// Lookup the scaling function
|
||||||
|
// First find the right platform
|
||||||
|
for (pdef = Scale_PlatDefs;
|
||||||
|
pdef->platform != Scale_Platform && pdef->platform != SCALEPLAT_NULL;
|
||||||
|
++pdef)
|
||||||
|
;
|
||||||
|
// Next find the right function
|
||||||
|
for (fdef = pdef->funcdefs;
|
||||||
|
(flags & fdef->flag) != fdef->flag;
|
||||||
|
++fdef)
|
||||||
|
;
|
||||||
|
|
||||||
|
return fdef->func;
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif
|
||||||
Executable
+27
@@ -0,0 +1,27 @@
|
|||||||
|
/*
|
||||||
|
* This program is free software; you can redistribute it and/or modify
|
||||||
|
* it under the terms of the GNU General Public License as published by
|
||||||
|
* the Free Software Foundation; either version 2 of the License, or
|
||||||
|
* (at your option) any later version.
|
||||||
|
*
|
||||||
|
* This program is distributed in the hope that it will be useful,
|
||||||
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
* GNU General Public License for more details.
|
||||||
|
*
|
||||||
|
* You should have received a copy of the GNU General Public License
|
||||||
|
* along with this program; if not, write to the Free Software
|
||||||
|
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||||
|
*/
|
||||||
|
|
||||||
|
#ifndef SCALERS_H_
|
||||||
|
#define SCALERS_H_
|
||||||
|
|
||||||
|
void Scale_Init (void);
|
||||||
|
|
||||||
|
typedef void (* TFB_ScaleFunc) (SDL_Surface *src, SDL_Surface *dst,
|
||||||
|
SDL_Rect *r);
|
||||||
|
|
||||||
|
TFB_ScaleFunc Scale_PrepPlatform (int flags, const SDL_PixelFormat* fmt);
|
||||||
|
|
||||||
|
#endif /* SCALERS_H_ */
|
||||||
+158
@@ -0,0 +1,158 @@
|
|||||||
|
/*
|
||||||
|
* This program is free software; you can redistribute it and/or modify
|
||||||
|
* it under the terms of the GNU General Public License as published by
|
||||||
|
* the Free Software Foundation; either version 2 of the License, or
|
||||||
|
* (at your option) any later version.
|
||||||
|
*
|
||||||
|
* This program is distributed in the hope that it will be useful,
|
||||||
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
* GNU General Public License for more details.
|
||||||
|
*
|
||||||
|
* You should have received a copy of the GNU General Public License
|
||||||
|
* along with this program; if not, write to the Free Software
|
||||||
|
* Foundation, Inc., 59 Temple Place - Suite 330, Boston, MA 02111-1307, USA.
|
||||||
|
*/
|
||||||
|
|
||||||
|
// Core algorithm of the Triscan screen scaler (based on Scale2x)
|
||||||
|
// (for scale2x please see http://scale2x.sf.net)
|
||||||
|
// Template
|
||||||
|
// When this file is built standalone is produces a plain C version
|
||||||
|
// Also #included by 2xscalers_mmx.c for an MMX version
|
||||||
|
|
||||||
|
#ifdef GFXMODULE_SDL
|
||||||
|
|
||||||
|
#include "libs/graphics/sdl/sdl_common.h"
|
||||||
|
#include "types.h"
|
||||||
|
#include "scalers.h"
|
||||||
|
#include "scaleint.h"
|
||||||
|
#include "2xscalers.h"
|
||||||
|
|
||||||
|
|
||||||
|
// Triscan scaling to 2x
|
||||||
|
// derivative of scale2x -- scale2x.sf.net
|
||||||
|
// The name expands to either
|
||||||
|
// Scale_TriScanFilter (for plain C) or
|
||||||
|
// Scale_MMX_TriScanFilter (for MMX)
|
||||||
|
// [others when platforms are added]
|
||||||
|
void
|
||||||
|
SCALE_(TriScanFilter) (SDL_Surface *src, SDL_Surface *dst, SDL_Rect *r)
|
||||||
|
{
|
||||||
|
int x, y;
|
||||||
|
const int w = src->w, h = src->h;
|
||||||
|
int xend, yend;
|
||||||
|
int dsrc, ddst;
|
||||||
|
SDL_Rect *region = r;
|
||||||
|
SDL_Rect limits;
|
||||||
|
SDL_PixelFormat *fmt = dst->format;
|
||||||
|
const int sp = src->pitch, dp = dst->pitch;
|
||||||
|
const int bpp = fmt->BytesPerPixel;
|
||||||
|
const int slen = sp / bpp, dlen = dp / bpp;
|
||||||
|
// for clarity purposes, the 'pixels' array here is transposed
|
||||||
|
Uint32 pixels[3][3];
|
||||||
|
Uint32 *src_p = (Uint32 *)src->pixels;
|
||||||
|
Uint32 *dst_p = (Uint32 *)dst->pixels;
|
||||||
|
|
||||||
|
int prevline, nextline;
|
||||||
|
|
||||||
|
// these macros are for clarity; they make the current pixel (0,0)
|
||||||
|
// and allow to access pixels in all directions
|
||||||
|
#define PIX(x, y) (pixels[1 + (x)][1 + (y)])
|
||||||
|
|
||||||
|
#define TRISCAN_YUV_MED 100
|
||||||
|
// medium tolerance pixel comparison
|
||||||
|
#define TRISCAN_CMPYUV(p1, p2) \
|
||||||
|
(PIX p1 == PIX p2 || SCALE_CMPYUV (PIX p1, PIX p2, TRISCAN_YUV_MED))
|
||||||
|
|
||||||
|
|
||||||
|
SCALE_(PlatInit) ();
|
||||||
|
|
||||||
|
// expand updated region if necessary
|
||||||
|
// pixels neighbooring the updated region may
|
||||||
|
// change as a result of updates
|
||||||
|
limits.x = 0;
|
||||||
|
limits.y = 0;
|
||||||
|
limits.w = src->w;
|
||||||
|
limits.h = src->h;
|
||||||
|
Scale_ExpandRect (region, 1, &limits);
|
||||||
|
|
||||||
|
xend = region->x + region->w;
|
||||||
|
yend = region->y + region->h;
|
||||||
|
dsrc = slen - region->w;
|
||||||
|
ddst = (dlen - region->w) * 2;
|
||||||
|
|
||||||
|
// move ptrs to the first updated pixel
|
||||||
|
src_p += slen * region->y + region->x;
|
||||||
|
dst_p += (dlen * region->y + region->x) * 2;
|
||||||
|
|
||||||
|
for (y = region->y; y < yend; ++y, dst_p += ddst, src_p += dsrc)
|
||||||
|
{
|
||||||
|
if (y > 0)
|
||||||
|
prevline = -slen;
|
||||||
|
else
|
||||||
|
prevline = 0;
|
||||||
|
|
||||||
|
if (y < h - 1)
|
||||||
|
nextline = slen;
|
||||||
|
else
|
||||||
|
nextline = 0;
|
||||||
|
|
||||||
|
// prime the (tiny) sliding-window pixel arrays
|
||||||
|
PIX( 1, 0) = src_p[0];
|
||||||
|
|
||||||
|
if (region->x > 0)
|
||||||
|
PIX( 0, 0) = src_p[-1];
|
||||||
|
else
|
||||||
|
PIX( 0, 0) = PIX( 1, 0);
|
||||||
|
|
||||||
|
for (x = region->x; x < xend; ++x, ++src_p, dst_p += 2)
|
||||||
|
{
|
||||||
|
// slide the window
|
||||||
|
PIX(-1, 0) = PIX( 0, 0);
|
||||||
|
|
||||||
|
PIX( 0, -1) = src_p[prevline];
|
||||||
|
PIX( 0, 0) = PIX( 1, 0);
|
||||||
|
PIX( 0, 1) = src_p[nextline];
|
||||||
|
|
||||||
|
if (x < w - 1)
|
||||||
|
PIX( 1, 0) = src_p[1];
|
||||||
|
else
|
||||||
|
PIX( 1, 0) = PIX( 0, 0);
|
||||||
|
|
||||||
|
if (!TRISCAN_CMPYUV (( 0, -1), ( 0, 1)) &&
|
||||||
|
!TRISCAN_CMPYUV ((-1, 0), ( 1, 0)))
|
||||||
|
{
|
||||||
|
if (TRISCAN_CMPYUV ((-1, 0), ( 0, -1)))
|
||||||
|
dst_p[0] = Scale_Blend_11 (PIX(-1, 0), PIX(0, -1));
|
||||||
|
else
|
||||||
|
dst_p[0] = PIX(0, 0);
|
||||||
|
|
||||||
|
if (TRISCAN_CMPYUV (( 1, 0), ( 0, -1)))
|
||||||
|
dst_p[1] = Scale_Blend_11 (PIX(1, 0), PIX(0, -1));
|
||||||
|
else
|
||||||
|
dst_p[1] = PIX(0, 0);
|
||||||
|
|
||||||
|
if (TRISCAN_CMPYUV ((-1, 0), ( 0, 1)))
|
||||||
|
dst_p[dlen] = Scale_Blend_11 (PIX(-1, 0), PIX(0, 1));
|
||||||
|
else
|
||||||
|
dst_p[dlen] = PIX(0, 0);
|
||||||
|
|
||||||
|
if (TRISCAN_CMPYUV (( 1, 0), ( 0, 1)))
|
||||||
|
dst_p[dlen+1] = Scale_Blend_11 (PIX(1, 0), PIX(0, 1));
|
||||||
|
else
|
||||||
|
dst_p[dlen+1] = PIX(0, 0);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
dst_p[0] = PIX(0, 0);
|
||||||
|
dst_p[1] = PIX(0, 0);
|
||||||
|
dst_p[dlen] = PIX(0, 0);
|
||||||
|
dst_p[dlen+1] = PIX(0, 0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
SCALE_(PlatDone) ();
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif /* GFXMODULE_SDL */
|
||||||
Reference in New Issue
Block a user