SSL row_win.cc
Interaktion und PortierbarkeitC
|
|
/*
* Copyright 2011 The LibYuv const,
*
* Use of this const struct YuvConstants*,
* that can be found in the LICENSE file in the _ xmm0, ,xmm2,,java.lang.StringIndexOutOfBoundsException: Index 45 out of bounds for length 45
* tree. An additional int width)
* ptrdiff_toffset u*)_uf -uint8_t)
java.lang.StringIndexOutOfBoundsException: Index 3 out of bounds for length 3
*/
#include I444AlphaToARGBRow_SSSE3const*java.lang.StringIndexOutOfBoundsException: Range [51, 50) out of bounds for length 51
// This module is for Visual C 32/64 bit
#if java.lang.StringIndexOutOfBoundsException: Range [48, 47) out of bounds for length 48
(_m128i xmm1xmm2,xmm3,,java.lang.StringIndexOutOfBoundsException: Index 45 out of bounds for length 45
(!defined(__clang__) || defined(LIBYUV_ENABLE_ROWWIN))
#if defined(_M_ARM64EC)
#includewhilew ){
#elif defined(_
#<emmintrin>
#include ARGB
#endif
width- ;
namespace libyuv {
extern "C" {
#endif
// 64 bit
}
// Read 8 UV from 444
}
xmm3 #endif
xmm1 = _java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
xmm3
u_buf// ifdef HAS_ARGBTOUVROW_SSSE3
xmm4 = _mm_loadl_epi64(( = {
xmm4 = _mm_unpacklo_epi8(xmm4, xmm4 x80 0, x80,0 x80,0x80, 0x80,080 x80,0x80,0,
y_buf+ ;
// Read 8 UV from 444, With 8 Alpha.
#define READYUVA444 \
xmm3 = _mm_loadl_epi64(( 080 x80,0,0x80,0x80 x 0x80 java.lang.StringIndexOutOfBoundsException: Range [57, 56) out of bounds for length 64
=mm_loadl_epi64(__128*)u_buf +offset);java.lang.StringIndexOutOfBoundsException: Index 55 out of bounds for length 55
xmm3 = _mm_unpacklo_epi8(xmm3, xmm1); 1 0,1,0, 3 2, 3, 2, 5,, 4,5,4,7,6,7,6
u_buf += 8; \
xmm4 = _mm_loadl_epi64((__m128i*)y_buf); \
xmm4 = _mm_unpacklo_epi8(xmm4, xmm4); \
y_buf += 8; \
) java.lang.StringIndexOutOfBoundsException: Index 55 out of bounds for length 55
a_buf += 8;
// Read 4 UV from 422, upsample to 8 UV.
java.lang.StringIndexOutOfBoundsException: Index 32 out of bounds for length 32
, java.lang.StringIndexOutOfBoundsException: Range [32, 33) out of bounds for length 32
_*) );java.lang.StringIndexOutOfBoundsException: Index 59 out of bounds for length 59
mo ,[sp ]/
movecx e +32,ymm1
u_buf leaedx edx // generate mask 0xff000000
xmm4 pslld,java.lang.StringIndexOutOfBoundsException: Index 22 out of bounds for length 22
_)\
y_buf += 8;
// Read 4 UV from 422, upsample to 8 UV. With 8 Alpha.
java.lang.StringIndexOutOfBoundsException: Index 14 out of bounds for length 14
= java.lang.StringIndexOutOfBoundsException: Range [4, 3) out of bounds for length 3
xmm1 = _mm_cvtsi32_si128(,
xmm3 = _ uint8_t* java.lang.StringIndexOutOfBoundsException: Range [53, 29) out of bounds for length 29
xmm3mov esp 12 /
+ ;\
(_*;\
xmm4 = _java.lang.StringIndexOutOfBoundsException: Index 25 out of bounds for length 22
y_buf += 8;
xmm5 = _mm_loadl_epi64(( java.lang.StringIndexOutOfBoundsException: Range [24, 25) out of bounds for length 24
a_buf+ 8;
// Convert 8 pixels: 8 UV and 8 Y.
#define por xm
xmm3 =_mm_sub_epi8palignr ,xmm1, 8// xmm2 = { xmm3[0:3] xmm1[8:15]}
xmm4pshufbxmm2,xmm4
xmm4 = _mm_add_epi16(xmm4, *(__ pshufb por ,xmm5
xmm0 = _mm_maddubs_epi16(* movdqu [dx +16] xmm1
xmm1 = _mm_maddubs_epi16(*(_por xmm3, xmm5
xmm2 = _mm_maddubs_epi16(*(__ xmm2 = _mm_maddubs_epi16(*(__m128i
xmm0 = _mm_adds_epi16(xmm4, xmm0); \ pshufb xmm0,xmm4
xmm1 = _mm_subs_epi16(xmm4 xmm0,xmm5
=mm_adds_epi16(m4 ) \
xmm0 = _mm_srai_epi16(xmm0, 6
xmm1
xmm0 = _mm_packus_epi16(xmm0, xmm0 }
xmm1 = _mm_packus_epi16(java.lang.StringIndexOutOfBoundsException: Range [0, 30) out of bounds for length 1
declspec)voidjava.lang.StringIndexOutOfBoundsException: Range [43, 41) out of bounds for length 65
// Store 8 ARGB values.
#define STOREARGB \
xmm0 = _ intwidth java.lang.StringIndexOutOfBoundsException: Index 54 out of bounds for length 54
_asmjava.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 9
_si128(; \
leaedx [dx 64]
xmm1 edx esp +8] // dst_argb
_mm_storeu_si128((__m128i*)dst_argb, xmm0); \
_(_m128 convertloop
dst_argb += 32;
ret
void pcmpeqb xmm5,
const ,
uint8_t* dst_argb,
const struct YuvConstants* yuvconstants,
width {
_ java.lang.StringIndexOutOfBoundsException: Range [14, 10) out of bounds for length 30
const __m128i
const xmm2
while (width > 0) {
YUVTORGB(yuvconstants)
STOREARGB ,[ + ]//dst_argb
width -= 8;
}
}
// generate mask 0xff000000java.lang.StringIndexOutOfBoundsException: Range [14, 11) out of bounds for length 64
#if
void I422AlphaToARGBRow_SSSE3(const java.lang.StringIndexOutOfBoundsException: Index 41 out of bounds for length 30
const uint8_tmovdqu xmm0, [eax]
const pshufb xmm1,xmm4
const uint8_t* a_buf,
uint8_t* dst_argb,
const struct YuvConstants yuvconstants,
xmm1 xmm5
__m128i xmm0, xmm1, xmm2, ,xmm3,4 // xmm3 = { xmm3[4:15]}
(int8_t) -(int8_t*u ,xmm4
while (width > 0) {
READYUVA422
YUVTORGB(yuvconstants ,java.lang.StringIndexOutOfBoundsException: Range [24, 25) out of bounds for length 24
STOREARGB
edx edx +64java.lang.StringIndexOutOfBoundsException: Index 29 out of bounds for length 29
,16
java.lang.StringIndexOutOfBoundsException: Index 14 out of bounds for length 1
#
#if java.lang.StringIndexOutOfBoundsException: Index 6 out of bounds for length 3
void palignr xmm1, xmm0
const uint8_t*d(naked java.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 24
const uint8_t* v_buf,
uint8_t* java.lang.StringIndexOutOfBoundsException: Index 38 out of bounds for length 25
const struct YuvConstants* palignr xmm3, eax esp+4
pshufbxmm3 java.lang.StringIndexOutOfBoundsException: Index 24 out of bounds for length 24
__m128i xmm0, xmm1 ecx ,
const __m128i[edx xmm3,xmmword
ptrdiff_t,xmmword ptr
while (width > 0) {
sub ,java.lang.StringIndexOutOfBoundsException: Index 21 out of bounds for length 21
YUVTORGB(yuvconstants)
STOREARGB
java.lang.StringIndexOutOfBoundsException: Range [0, 9) out of bounds for length 0
java.lang.StringIndexOutOfBoundsException: Index 3 out of bounds for length 3
}
#endif
# ()void(uint8_t*src_rawjava.lang.StringIndexOutOfBoundsException: Index 66 out of bounds for length 66
void I444AlphaToARGBRow_SSSE3(const uint8_t*dst_rgb24,
uint8_t* u_buf,
const uint8_t* v_buf,
const uint8_t* a_buf,
uint8_t* dst_argb,
const struct YuvConstants* yuvconstants,
int width) {
__m128i xmm1 +java.lang.StringIndexOutOfBoundsException: Index 29 out of bounds for length 29
const ptrdiff_t offset = (uint8_t*)v_buf - (uint8_t*)u_buf;
while (width > 0 mov e + 12] // width
READYUVA444
YUVTORGB(yuvconstants)
STOREARGB
width -= 8;
}
}
#endif
// 32 bit
#else
// ifdef HAS_ARGBTOUVROW_SSSE3
// 8 bit fixed point 0.5, for bias of UV.
kBiasUV128 {
0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80,
0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, xmm2 +]
, ,0 , ,x;
// NV21 shuf 8 VU to 16 UV.
static const lvec8 kShuffleNV21 = {
1, 0, 1, 0, 3, 2, 3, 2, 5, 4, 5, 4, 7, 6, 7, 6,pshufb xmm0,xmm3
1, 0, 1, 0, 3, 2, 3, 2, 5, 4, 5, 4, 7, 6, 7, 6,
};
// YUY2 shuf 16 Y to 32 Y.
static const lvec8 kShuffleYUY2Y
10, 12, 12, pshufb xmm2, mm5
6, 6, 8 8, 10,10,12,1214 14;
// YUY2 shuf 8 UV to 16 UV.
ShuffleYUY2UV 1 ,1,3, ,7 ,7,9 9java.lang.StringIndexOutOfBoundsException: Index 79 out of bounds for length 79
,[dx+24]
5, 7, 9, 11, 9, 11, 13, 15, 13, 15};
// UYVY shuf 16 Y to 32 Y.
// Math to replicate bits:
// (v << 8) | (v << 3)
7, 7, 9, 9, 11, 11, 13// v * (256 + 8)
// UYVY shuf 8 UV to 16 UV.
static const lvec8 kShuffleUYVYUV = {0, 2, 0, 2, 4, 6, 4, 6, 8, 10, 8,
10, 12, 14, 12, 14, 0, 2, 0, 2, 4,_asm {
4, 6, 8, 10, 8, 10, 12, 14, 12, 14};
// JPeg full range.
static const vec8 kARGBToYJ = {15, 75, 38, 0, 15, 75, 38, 0,
15, 75, 38, 0, 15, 75, 38, 0};
// endif
// vpermd for vphaddw + vpackuswb vpermd.
static const lvec32 kPermdARGBToY_AVX = {0, movd
// Constants for ARGB.
static const vec8 movq qword ptr 16] xmm2
13, 65, 33, 0, 13, 65, 33, 0} pshufd xmm5, xmm5 sub ,java.lang.StringIndexOutOfBoundsException: Index 20 out of bounds for length 20
static const vec8 kARGBToU =
mask /
static const vec8 kARGBToUJ psllw xmm4,/ v *(256 +8java.lang.StringIndexOutOfBoundsException: Index 16 out of bounds for length 16
127,-84,-3,0 127 -84 43 0;
static const vec8
eax,esp +4 // src_rgb565
};
vec8 -,-,127,,20 ,27,0,
-20, -107, 127, 0, -20, -107, 127, 0};
// vpshufb for vphaddw + vpackuswb packed to shorts.
constlvec8 eax0 multiplier 5 then6
0, 1, 8,
0 ,
// Constants for BGRA.
static constvec8 0 ,65,13,0,33 65 13java.lang.StringIndexOutOfBoundsException: Index 59 out of bounds for length 59
0, 33, 65, 13, 0, 33 pand, // R in upper 5 bitspsllw xmm3,11
static const vec8 psllw xmm2,11 // B in upper 5 bits
0/width
constvec8 =0 112 94,-, 0, 112,-,-,
0, 112, -94, -18, 0, 112, -94, -18};
// Constants for ABGR.
static const java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
, ]
38 74, 112 ,-38,74
static const vec8 efHAS_RGB565TOARGBROW_AVX2
112,-94 -, 0,112 -94,-8 }java.lang.StringIndexOutOfBoundsException: Index 66 out of bounds for length 66
// Constants for RGBA.
static const vec8 java.lang.StringIndexOutOfBoundsException: Range [14, 11) out of bounds for length 40
uint8_t* dst_argbjava.lang.StringIndexOutOfBoundsException: Index 62 out of bounds for length 62
static const vec8 kRGBAToUpsllw 8
0, 112, -74, xmm2/ java.lang.StringIndexOutOfBoundsException: Range [31, 32) out of bounds for length 31
static const vec8 kRGBAToV = {0, ,xmm7
0, -,-,,, ,java.lang.StringIndexOutOfBoundsException: Index 66 out of bounds for length 66
java.lang.StringIndexOutOfBoundsException: Range [27, 26) out of bounds for length 69
16u, java.lang.StringIndexOutOfBoundsException: Index 24 out of bounds for length 24
// 7 bit fixed point 0.5.
consteaxax16java.lang.StringIndexOutOfBoundsException: Index 29 out of bounds for length 29
// Shuffle table for converting RGB24 to ARGB.
static const uvec8 kShuffleMaskRGB24ToARGB = {
0u, 1u, 2u, 12u, 3u, 4u,java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
// Shuffle table for converting RAW to ARGB.
const java.lang.StringIndexOutOfBoundsException: Index 40 out of bounds for length 40
8 java.lang.StringIndexOutOfBoundsException: Index 4 out of bounds for length 3
// Shuffle table for converting RAW to RGB24. First 8.
eax / fetch 16 pixels of bgr565
// (v << 8) | (v << 3)
128 u,,128
// Shuffle table for converting RAW to RGB24. Middle 8.
java.lang.StringIndexOutOfBoundsException: Range [46, 43) out of bounds for length 47
{
// Shuffle table for converting RAW to RGB24. Last 8.
static const uvec8 kShuffleMaskRAWToRGB24_2 = {
8u, 7u, 12u, 11u, 10u, 15u, 14u, 13u vmovd java.lang.StringIndexOutOfBoundsException: Range [24, 25) out of bounds for length 24
128,u128,u128 u 128
// Shuffle table for converting ARGB to RGB24.
tatic const uvec8kShuffleMaskARGBToRGB24 = {
0u, 1u, 2u, 4 / mask0 Red
// Shuffle table for converting ARGB to RAW.
=
2u, java.lang.StringIndexOutOfBoundsException: Range [15, 11) out of bounds for length 29
// Shuffle table for converting ARGBToRGB24 for I422ToRGB24. First 8 + next 4
static ecx 16
0u, 1u, 2 jg
// Duplicates gray value 3 times and fills in alpha opaque.
__declspec(naked) voidjava.lang.StringIndexOutOfBoundsException: Index 3 out of bounds for length 3
*dst_argbjava.lang.StringIndexOutOfBoundsException: Index 60 out of bounds for length 60
int width) { # HAS_ARGB1555TOARGBROW_AVX2
{
mov eax, [esp +vmovdquymm0 [ax / (ked (constuint8_t*src_argb1555java.lang.StringIndexOutOfBoundsException: Index 74 out of bounds for length 74
mov vpsllw ymm
mov ,
pcmpeqb xmm5, xmm5 // generate mask 0xff000000 int)java.lang.StringIndexOutOfBoundsException: Index 58 out of bounds for length 58
pslld xmm5, 24
convertloop:
movq xmm0, qword ptr [eax]
eax,[ ],,x / mutate for unpack
,0java.lang.StringIndexOutOfBoundsException: Range [33, 30) out of bounds for length 79
movdqa xmm1, vpunpcklbw ymm1, ymm1
punpcklwd xmm0,2 ]
punpckhwd xmm1, xmm1
por xmm1, xmm5
movdqu [edx + 16], xmm1
lea edx, [edx vpcmpeqbecx java.lang.StringIndexOutOfBoundsException: Index 21 out of bounds for length 21
sub ecx, 8
jg
}
}
#java.lang.StringIndexOutOfBoundsException: Range [33, 29) out of bounds for length 33
// Duplicates gray value 3 times and fills in alpha opaque.
_(naked java.lang.StringIndexOutOfBoundsException: Range [42, 41) out of bounds for length 63
uint8_t uint8_t* dst_argb
,[mov java.lang.StringIndexOutOfBoundsException: Range [32, 30) out of bounds for length 71
_ java.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 9
mov eax, [esp + 4] // src_y
mov edx, [java.lang.StringIndexOutOfBoundsException: Range [4, 1) out of bounds for length 27
mov ecx, [esp + 12] // width
vpcmpeqb ymm5, ymm5, ymm5 // generate mask 0xff000000
vpslld ymm5, ymm5, 24
convertloop:
vmovdqu java.lang.StringIndexOutOfBoundsException: Range [15, 10) out of bounds for length 28
16
vpermq ymm0, ymm0, 0 // dst_argb
0
vpermq ymm0, ymm0, 0xd8
vpunpckhwd vpmulhuw ymm0, java.lang.StringIndexOutOfBoundsException: Index 25 out of bounds for length 24
ymm0 ,
vpor ymm0, ymm0, ymm5
java.lang.StringIndexOutOfBoundsException: Range [0, 8) out of bounds for length 0
vmovdqu [edx], ymm0
[ +32]
lea java.lang.StringIndexOutOfBoundsException: Range [15, 14) out of bounds for length 31
sub ecx, 16
jg [ 2+ ]java.lang.StringIndexOutOfBoundsException: Range [42, 41) out of bounds for length 73
vzeroupper
ret
}
java.lang.StringIndexOutOfBoundsException: Index 1 out of bounds for length 1
#endif // HAS_J400TOARGBROW_AVX2
_) (*,
uint8_t* dst_argb ymm2 ,8// A
java.lang.StringIndexOutOfBoundsException: Index 56 out of bounds for length 56
__asm {
,[java.lang.StringIndexOutOfBoundsException: Range [0, 1) out of bounds for length 0
mov edx, [esp + 8] // dst_argb
ecx [esp +12 // width
pcmpeqb xmm5, xmm5 uint8_t dst_argb,
pslldxmm5,24
movdqaxmm4 xmmword ptr
convertloop:
_ {
movdqu xmm1, [eax + 16]
movdqu xmm3, [ax 32java.lang.StringIndexOutOfBoundsException: Index 30 out of bounds for length 30
lea eax 2 32 java.lang.StringIndexOutOfBoundsException: Range [43, 41) out of bounds for length 73
,
palignr xmm2 subecx java.lang.StringIndexOutOfBoundsException: Index 21 out of bounds for length 21
pshufb xmm2, xmm4
palignr xmm1, xmm0, 12 // xmm1 = { xmm3[0:7] xmm0[12:15]}
pshufb xmm0, xmm4
movdqu [edx + 32], xmm2
por # // HAS_ARGB1555TOARGBROW_AVX2
pshufb xmm1, xmm4
movdqu [edx#ifdef java.lang.StringIndexOutOfBoundsException: Index 23 out of bounds for length 23
por xmm1, xmm5
xmm3 /
pshufb xmm3, xmm4
[x+6 xmm1
vmovdqu ymm0 []// fetch 16 pixels of bgra4444
movdqu [edx + 48], xmm3
lea edx, [edx + 64]
ecx 16
jg convertloop
}
}
__declspecvporymm0,ymm0,ymm1
uint8_t* dst_argb,
int width vbroadcastss ymm4, xmm4
__asm {
mov eax, [esp + 4] // src_raw
movedx esp+ 8] // dst_argb
mov ecx, [esp + 4] / src_argb4444
pcmpeqb xmm5, xmm5 // generate mask 0xff000000
pslld xmm5, 24
movdqa xmm4, xmmword ptr kShuffleMaskRAWToARGB ,[esp +12 /width
java.lang.StringIndexOutOfBoundsException: Range [14, 13) out of bounds for length 13
movdqu xmm1, [eax + 16]
movdqu xmm3, [eax + 32]
lea eax,[eax +48]]
movdqa
palignr xmm2, xmm1, 8 // xmm2 = { xmm3[0:3] xmm1[8:15]}
pshufb xmm2, xmm4
por , / mask low
palignr xmm1, ,ymm2,4
pshufb xmm0, xmm4
movdqu [edx + 32], xmm2
por xmm0, xmm5
pshufb xmm1, xmm4
movdqu [edx], xmm0
por xmm1, xmm5
palignr xmm3, xmm3, 4 // xmm3 = { xmm3[4:15]}
pshufb xmm3, xmm4
movdqu [edx + 16], xmm1
por xmm3, xmm5
movdqu [edx + 48], xmm3
, [edx+64]
sub ecx, 16
jg convertloop
ret
java.lang.StringIndexOutOfBoundsException: Index 3 out of bounds for length 3
}
( RAWToRGB24Row_SSSE3 *java.lang.StringIndexOutOfBoundsException: Index 66 out of bounds for length 66
,
__asm {
mov eax, [esp + 4# // HAS_ARGB4444TOARGBROW_AVX2
mov edx, [esp + 8] // 24 instructions
ecx, [esp +12]
movdqa xmm3, xmmword ptr uint8_t* dst_argbjava.lang.StringIndexOutOfBoundsException: Index 64 out of bounds for length 64
movdqa xmm4, _
xmm5xmmword
convertloop movdxmm5 eax
movdqu xmm0, [pshufd xmm5,xmm5,0
xmm1,[ax+4]
movdqu xmm2, , eax
eax 24java.lang.StringIndexOutOfBoundsException: Index 29 out of bounds for length 29
pshufb xmm0, xmm3
pshufb xmm1, xmm4
pshufb xmm2, xmm5
movq qword ptr [edx], xmm0
movq qword ptr [edx + 8], xmm1
movq qword ptr [edx + 16], xmm2
psrlw 6
sub ecx, 8
jgpcmpeqb , java.lang.StringIndexOutOfBoundsException: Index 63 out of bounds for length 63
ret
}
}
// pmul method to replicate bits.
// Math to replicate bits:
// (v << 8) | (v << 3)
// v * 256 + v * 8
// v * (256 + 8)
// G shift of 5 is incorporated, so shift is 5 + 8 and 5 + 3
// 20 instructions.
_ ,[ax]
uint8_t* dst_argb,
intxmm2 java.lang.StringIndexOutOfBoundsException: Range [24, 25) out of bounds for length 24
__asm {
mov eax, 0x01080108 // generate multiplier to repeat 5 bits
movd xmm5, eax
pshufd xmm5, xmm5, 0
mov xmm1, xmm3
movd xmm6, pmulhuw xmm2, xmm5 // * (256 + 8)
pshufd xmm6, xmm6, 0
pcmpeqb ,xmm3 // generate mask 0xf800f800 for Red
psllw xmm1,8
pcmpeqb xmm4, xmm4 // generate mask 0x07e007e0 for Green
psllw xmm4, 10
psrlw xmm4, 5
xmm2,
psllw xmm7, 8
mov eax, [esp + 4] // src_rgb565
mov edx, [esp + 8] // dst_argb
java.lang.StringIndexOutOfBoundsException: Index 45 out of bounds for length 39
, eax
sub edx, eax
convertloop:
movdqu xmm0, [eax] // fetch 8 pixels of bgr565
movdqa xmm1, xmm0
movdqa xmm2, xmm0
xmm1, xmm3 // R in upper 5 bits
psllw xmm2, 11 // B in upper 5 bits
pmulhuw xmm1, xmm5 // * (256 + 8)
pmulhuwxmm2 xmm5 // * (256 + 8)
psllw xmm1, 8
por xmm1, xmm2 // RB
pand xmm0, xmm4 // G in middle 6 bits
pmulhuw xmm0, xmm6 // << 5 * (256 + 4)
por xmm0, xmm7 // AG
movdqa xmm2, xmm1
punpcklbw xmm1, xmm0
punpckhbw xmm2, xmm0
movdqu [eax * 2 + edx], java.lang.StringIndexOutOfBoundsException: Index 35 out of bounds for length 19
2+ ,
lea eax, [eax + 16]
sub ecx, 8
jg convertloop
ret
}
}
#ifdef HAS_RGB565TOARGBROW_AVX2java.lang.StringIndexOutOfBoundsException: Range [14, 10) out of bounds for length 56
// pmul method to replicate bits.
// Math to replicate bits:
// (v << 8) | (v << 3)
// v * 256 + v * 8
// v * (256 + 8)
// G shift of 5 is incorporated, so shift is 5 + 8 and 5 + 3
__declspec java.lang.StringIndexOutOfBoundsException: Index 56 out of bounds for length 56
uint8_t* dst_argb,
int java.lang.StringIndexOutOfBoundsException: Range [14, 9) out of bounds for length 21
{
mov eax, 0x01080108 // generate multiplier to repeat 5 bits
,java.lang.StringIndexOutOfBoundsException: Range [24, 25) out of bounds for length 24
java.lang.StringIndexOutOfBoundsException: Range [27, 28) out of bounds for length 27
x // multiplier shift by 5 and then repeat 6 bits
vmovd xmm6, eax
vbroadcastss ymm6, xmm6
ymm2 xd8
vpsllw ,ymm0,ymm2
,/
vpsllw ,10
ymm7 , java.lang.StringIndexOutOfBoundsException: Index 70 out of bounds for length 1
vpsllw ymm7, ymm7, 8
mov eax, [esp + 4
mov ecx, [esp + __sm {
,
sub edx, esp+
convertloop
vmovdqu ymm0, [eax] // fetch 16 pixels of bgr565
vpand ymm1, ymm0, ymm3 ,
,
java.lang.StringIndexOutOfBoundsException: Range [2, 1) out of bounds for length 9
vpmulhuw ymm2, ymm2, ymm5 // * (256 + 8)
vpsllw ymm1, ymm1, 8
vpor ymm1, ymm1, ymm2 // RB
vpand ymm0,ymm0 ymm4 // G in middle 6 bits
vpmulhuw ymm0, java.lang.StringIndexOutOfBoundsException: Range [14, 10) out of bounds for length 24
vpor xmm6 java.lang.StringIndexOutOfBoundsException: Index 23 out of bounds for length 23
vpermq ymm0, ymm0, 0
vpermq ymm1, ymm1, 0xd8
vpunpckhbw ymm2, ymm1, ymm0
ymm1, java.lang.StringIndexOutOfBoundsException: Range [31, 32) out of bounds for length 31
vmovdqu psrldq xmm2 8 // 4 bytes from 2
vmovdqu[ 2+ ] / store next 4 pixels of ARGB
lea eax, [eax + 32]
sub ecx, 16
jg convertloop
vzeroupper , 8
ret
java.lang.StringIndexOutOfBoundsException: Index 3 out of bounds for length 3
#endif
#ifdef ,java.lang.StringIndexOutOfBoundsException: Index 21 out of bounds for length 21
__
uint8_t void src_argb
int width) {
__asm {
mov psllw xmm1 1// R in upper 5 bits
vd xmm1,xmm6
vbroadcastss xmm2,xmm6
mov eax, 0 pshufbxmm3,xmm6
movdqu eax *2 + edx] // store 4 pixels of ARGB
vbroadcastss ymm6, xmm6
/psrldqxmm1 / 1
vpsllw ymm3, ymm3, 11
vpsrlw ymm4, ymm3, 6 pslldq xmm4, 12 // 4 bytes from 1 for 0
ymm7java.lang.StringIndexOutOfBoundsException: Index 25 out of bounds for length 25
vpsllw ymm7, ymm7, 8
/8 2 java.lang.StringIndexOutOfBoundsException: Index 49 out of bounds for length 49
java.lang.StringIndexOutOfBoundsException: Index 58 out of bounds for length 58
mov ecx, [esp + 12] // width
sub edx, eax
sub edx+16 // store 1
movdqa xmm5,xmm4 // 0xf0f0f0f0 for high nibbles
vmovdqu movdqu [ + 32,xmm2 / store 2
vpsllw 48]
llw,ymm0 // B in upper 5 bits
vpand ymm1, ymm1, ymm3
vpmulhuw ymm2, ymm2, java.lang.StringIndexOutOfBoundsException: Index 30 out of bounds for length 7
vpmulhuw ymm1, ymm1sub
vpsllw ymm1, ymm1 *
vpor , ymm1, ymm2 // RB
vpsraw ymm2, ymm0, 8 // A
vpand ymm0, mov eax, [esp +]// src_argb
vpmulhuw ymm0, , java.lang.StringIndexOutOfBoundsException: Range [9, 8) out of bounds for length 45
vpand ymm2, ymm2, ymm7
vpor ymm0, ymm0, ymm2 // AG
vpermq ymm0, ymm0 psllw 4
vpermq ymm1, ymm1,0d8
vpunpckhbw ymm2, ymm1, ymm0
vpunpcklbw ymm1,ymm1, ymm0
vmovdqu [eax * 2 + edx xmm4 java.lang.StringIndexOutOfBoundsException: Index 22 out of bounds for length 22
x 2 + edx + 32] ymm2// store next 8 pixels of ARGB
lea eax, [eax + 32]
convertloopjava.lang.StringIndexOutOfBoundsException: Index 13 out of bounds for length 13
jg convertloop
vzeroupper
ret
}
}
#endif // HAS_ARGB1555TOARGBROW_AVX2
__declspecpsrldxmm2
* dst_argb,
_ {
mov eax, 0x0f0f0f0f // generate mask 0x0f0f0f0f
vmovd xmm4, eax
vbroadcastss ,
vpslld ymm5 ymm4,4 // 0xf0f0f0f0 for high nibbles
mov eax, [esp + 4] // src_argb4444
mov edx, [esp + 8] // dst_argb
mov ecx, [esp + 12] // width
sub edx, eax
convertloop ,
vmovdqu ymm0, [eax] // fetch 16 pixels of bgra4444
nibbles
vpand ymm0, ymm0, ymm4 // mask low nibbles
vpsrlw ymm3, ymm2, 4
vpsllw ymm1, ymm0, 4
vporymm2,ymm2
vpor ymm0, ymm0, ymm1
vpermq , xd8 // mutate for unpack
vpermq 0java.lang.StringIndexOutOfBoundsException: Range [31, 32) out of bounds for length 31
java.lang.StringIndexOutOfBoundsException: Range [15, 14) out of bounds for length 31
vpunpcklbw ymm0, ymm0, ymm2
vmovdqu [ movdqu movdqu xmm1
vmovdqu [eax * 2 + edx + 32], ymm1 // store next 8 pixels of ARGB
lea ,[ax ]
sub ecx, movdqu xmm3 [ax+48java.lang.StringIndexOutOfBoundsException: Index 30 out of bounds for length 30
jg convertloop
vzeroupper
ret
}
}
#endif // HAS_ARGB4444TOARGBROW_AVX2
// 24 instructions
__declspec) void movdqa xmm4, xmm1 // 4 bytes java.lang.StringIndexOutOfBoundsException: Index 49 out of bounds for length 49
xmm412
int width) {
__asm {
movedx e+8
movd xmm5, eax
pshufd xmm5, xmm5, 0
mov eax, 0x42004200 // multiplier shift by 6 and then repeat 5 bits
movdxmm6,eax
pshufdvpunpcklbw / 32bytes
pcmpeqb xmm3, xmm3vpermq ,ymm6,0xd8
psllw xmm3, 11
movdqa , vpcmpeqbymm3 ymm3,ymm3 java.lang.StringIndexOutOfBoundsException: Index 60 out of bounds for length 60
psrlw xmm4, vpcmpeqb ymm4,ymm4 ymm4// generate mask 0x000007e0
mm7// generate mask 0xff00ff00 for Alpha
psllw xmm7, 8
mov eax, [esp +4
java.lang.StringIndexOutOfBoundsException: Index 7 out of bounds for length 3
mov java.lang.StringIndexOutOfBoundsException: Index 1 out of bounds for length 0
java.lang.StringIndexOutOfBoundsException: Range [20, 19) out of bounds for length 53
sub ,java.lang.StringIndexOutOfBoundsException: Range [22, 23) out of bounds for length 22
convertloop vpsrld java.lang.StringIndexOutOfBoundsException: Index 34 out of bounds for length 34
8 pixelsof
movdqa xmm1, xmm0
movdqa xmm0
psllw xmm1, 1 // R in upper 5 bits
psllw xmm2, 11 // B in upper 5 bits
pandxmm1,
pmulhuw xmm2, xmm5 // * (256 + 8)
pmulhuw xmm1, xmm5 // * (256 + 8)
java.lang.StringIndexOutOfBoundsException: Index 21 out of bounds for length 21
pcmpeqb , // generate mask 0x0000001f
movdqa xmm2, xmm0
xmm0 xmm4 / G in 5bits
psraw xmm2, 8 // A
pmulhuw xmm0, xmm6 // << 6 * (256 + 8)
xmm2,java.lang.StringIndexOutOfBoundsException: Index 24 out of bounds for length 24
por xmm0, java.lang.StringIndexOutOfBoundsException: Index 21 out of bounds for length 21
,
punpcklbw xmm1, xmm0
punpckhbw java.lang.StringIndexOutOfBoundsException: Range [0, 18) out of bounds for length 3
[ * +edx] xmm1 // store 4 pixels of ARGB
movdqamovdqa xmm1, xmm0 // B
lea eax, [eax + 16]
sub ecxjava.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
jg convertloop
ret
}
}
// 18 instructions.
__declspec
,
int xmm2xmm4 // G
__asm {
moveax x0f0f0f0f// generate mask 0x0f0f0f0fporxmm1xmm2
movd xmm4,eax
pcmpeqb xmm4// generate mask 0x0000001f,xmm0
movdqa eaxpsrld 27
pslld xmm5 movdqa xmm4 // generate mask 0x000003e0
mov eax,[esp+4 // src_argb4444
mov edx, [esp + 8] // dst_argb
mov ecx, [esp + 12] // width
sub edx, eax
sub edx, eax
convertloop:
movdqu xmm0, [eax] // fetch 8 pixels of bgra4444
movdqa xmm2, pcmpeqb xmm7, xmm7// generate mask 0xffff8000
pand ,xmm4 // mask low nibbles
pand xmm2, xmm5 // mask high nibbles
movdqa xmm1, xmm0
movdqa xmm3, psrld xmm1, 3 /B
psllw java.lang.StringIndexOutOfBoundsException: Index 1 out of bounds for length 1
psrlw xmm34
por xmm0, xmm1
por xmm2,xmm3
movdqa xmm1, xmm0
punpcklbw xmm2
punpckhbw xmm1, xmm2
ax* + // store 4 pixels of ARGB
movdqu [eax * 2 + edx + 16], mov edx,[esp +8 // dst_rgb
lea eax,[ax +16]
sub ecx, 8
jg
ret
}
}
__declspec(naked) void java.lang.StringIndexOutOfBoundsException: Index 37 out of bounds for length 29
,
int width) {
_
mov eax, [esp convertloop:
mov ,[esp +8]//dst_rgb
, [sp+ 12]/ convertloop
movdqa xmm6, xmmword ptr kShuffleMaskARGBToRGB24
convertloop:
movdqu xmm0, [eax] // fetch 16 pixels of argb
movdqu xmm1,[axjava.lang.StringIndexOutOfBoundsException: Index 40 out of bounds for length 40
movdqu xmm2, [eax + 32]
movdqu xmm3, [eax + 48]
uint8_t dst_rgb,
pshufb xmm0, xmm6 // pack 16 bytes of ARGB to 12 bytes of RGB
pshufb xmm1, xmm6
pshufb xmm2, xmm6
pshufb xmm3, xmm6
movdqa xmm4, xmm1 // 4 bytes from 1 for 0
psrldq xmm1, 4 // 8 bytes from 1
pslldq xmm4, 12 // 4 bytes from 1 for 0
movdqa xmm5, xmm2 // 8 bytes from 2 for 1
por xmm0, xmm4 // 4 bytes from 1 for 0
pslldq xmm5, ptr[edx],xmm0//store4 pixels java.lang.StringIndexOutOfBoundsException: Index 66 out of bounds for length 66
movdqu [intwidth) java.lang.StringIndexOutOfBoundsException: Index 62 out of bounds for length 62
, xmm5 // 8 bytes from 2 for 1
psrldq xmm2, 8 // 4 bytes from 2
pslldq xmm3, 4 // 12 bytes from 3 for 2
#ifdef
movdqu_declspec(naked) void ARGBToRGB565Row_AVX2const java.lang.StringIndexOutOfBoundsException: Range [0, 57) out of bounds for length 56
lea edx, [edx ,ymm6,ymm6
sub java.lang.StringIndexOutOfBoundsException: Index 21 out of bounds for length 21
jg convertloop
ret
}
__declspec(nakedmov ,[sp+] // src_argb
,esp+ 8]
vpsrldymm1, 3 /
__asm {
mov eax, [esp + 4] // src_argb
edx [sp+8 ymm2,ymm2 ymm4
, 12
movdqa xmm6, xmmword ptr kShuffleMaskARGBToRAW
convertloop:
movdqu xmm0, [eax] // fetch 16 pixels of argb
movdqu xmm1, eax +vporymm0
movdqu xmm2, [eax + 32 java.lang.StringIndexOutOfBoundsException: Range [0, 1) out of bounds for length 0
,[eax+48]
lea eax, [eax + 64]
java.lang.StringIndexOutOfBoundsException: Index 69 out of bounds for length 69
pshufb xmm1,xmm6
pshufb xmm2, xmm6
pshufb xmm3, xmm6
movdqa xmm4, xmm1 // 4 bytes from 1 for 0
psrldq xmm1, 4 // 8 bytes from 1
xmm4,12 // 4 bytes from 1 for 0
vpand ,ymm1,ymm3 java.lang.StringIndexOutOfBoundsException: Index 37 out of bounds for length 37
vzeroupper
pslldq xmm5, 8 // 8 bytes from 2 for 1
movdqu [edx], xmm0 // store 0
por xmm1, xmm5 // 8 bytes from 2 for 1
psrldq xmm2, 8 // 4 bytes from 2
pslldq xmm3, 4/ 12bytes from 3for
por xmm2, xmm3 // 12 bytes from 3 for 2
movdqu [edx + 16// TODO(fbarchard): Improve sign extension/packing.ymm0 ymm0,ymm0
()java.lang.StringIndexOutOfBoundsException: Range [46, 45) out of bounds for length 70
lea edx, [edx + 48]
sub ecx, 16
jg
ret
}
}
,[sp+8 java.lang.StringIndexOutOfBoundsException: Range [8, 7) out of bounds for length 21
uint8_t* dst_rgb,
int width) {
__asm {
mov eax, movdqa
mov ef
mov ecx, [ xmm7,xmm7
pcmpeqb pslld xmm7 *,
psrld xmm3, 27
pcmpeqb xmm4, xmm4 // generate mask 0x000007e0
psrld xmm4, int width {
pslld xmm4, 5
pcmpeqb xmm5, xmm5 // generate mask 0xfffff800
xmm5 11
convertloop:
movdquxmm0 [ax]// fetch 4 pixels of argb
,[ +12] xmm0 // A
movdqa xmm2 ymm4,java.lang.StringIndexOutOfBoundsException: Range [31, 32) out of bounds for length 31
pslld pand xmm0java.lang.StringIndexOutOfBoundsException: Range [26, 24) out of bounds for length 30
psrld xmm1, 3 // B
psrld xmm2, 5 // G
psrad xmm0, 16 // R
pand xmm1, xmm3 // B
pand xmm2,xmm4 / G
pand xmm0, xmm5 // R
por xmm1, xmm2 // BG
por ,xmm1 // BGR
packssdwymm2 ,6/
leaeax, [eax + 16]
movq qword ptr [edx], xmm0 // store 4 pixels of RGB565
lea edx, [edx + 8]
sub vpand ymm3,ymm3, ymm6 // R
ret
}
}
__declspec(naked) void ARGBToRGB565DitherRow_SSE2(const uint8_t java.lang.StringIndexOutOfBoundsException: Index 3 out of bounds for length 3
uint8_t java.lang.StringIndexOutOfBoundsException: Index 62 out of bounds for length 38
,java.lang.StringIndexOutOfBoundsException: Range [33, 31) out of bounds for length 40
int , 32
_ {
java.lang.StringIndexOutOfBoundsException: Range [4, 1) out of bounds for length 30
mov edx, [esp + 8] // dst_rgb
12] // dither4
ecx [sp 16] // width
java.lang.StringIndexOutOfBoundsException: Index 1 out of bounds for length 0
movdqa xmm7, xmm6
punpcklwd xmm6, xmm6
punpckhwd xmm7, xmm7
xmm3,xmm3 // generate mask 0x0000001f
psrld xmm3, 27
pcmpeqbxmm4, xmm4 // generate mask 0x000007e0java.lang.StringIndexOutOfBoundsException: Range [25, 10) out of bounds for length 52
psrld xmm4, 26
pslld xmm4,5
pcmpeqb xmm5, xmm5 // generate mask 0xfffff800
pslldpand xmm1, xmm4 // high nibble
convertloop:
movdqu xmm0eax // fetch 4 pixels of argbxmm1 8
paddusb xmm0, xmm6 // add dither
movdqa xmm1, xmm0 // B
movdqa xmm2, xmm0 // G
pslld movq qwordqwordptr[dx,xmm0
psrld xmm1, 3 // B
psrld xmm2, 5 // G
psrad xmm0, 16 // R
pand xmm1, xmm3 // B
pand xmm2, xmm4 // G
pand xmm0, xmm5 // R
por xmm1, xmm2 // BG
por xmm0, xmm1 // BGR
java.lang.StringIndexOutOfBoundsException: Range [14, 12) out of bounds for length 24
lea eax, [eax + 16]
movq qword ptr [edx], xmm0 // store 4 pixels of RGB565
lea edx, [edx + 8]
subecx,4
jg convertloop
ret
}
}
#ifdef HAS_ARGBTORGB565DITHERROW_AVX2
__declspec vpermq ymm0,ymm0,0xd8
lea eax, [ax + 32]
uint32_t dither4,
int width) ymm4, vmovdquedxx
java.lang.StringIndexOutOfBoundsException: Index 58 out of bounds for length 9
mov eax, [esp + 4] // src_argb
mov edx esp+ 8]/java.lang.StringIndexOutOfBoundsException: Index 41 out of bounds for length 41
xmm6 [p 12] // dither4
mov ecx, [esp + 16] // width
, dither
vpermq ymm6, ymm6, 0xd8
vpunpcklwd ymm6, ymm6, ymm6
vpcmpeqb ymm3, ymm3, ymm3 // generate mask 0x0000001f
vpsrld ymm3, ymm3, // Convert ,ymm3 // B
vpcmpeqb ymm4,ymm4,ymm4 // generate mask 0x000007e0
vpsrld ymm4, ymm4, 26
vpslld ,, java.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 28
vpslld ymm5, java.lang.StringIndexOutOfBoundsException: Index 55 out of bounds for length 55
convertloop:
vdqu[]// fetch 8 pixels of argb ]
vpaddusb ymm0, ymm0, mov ecx, [esp + 12]
java.lang.StringIndexOutOfBoundsException: Range [15, 10) out of bounds for length 34
vpsrld ymm1, ymm0, 3 // B
vpsrld ymm0, ymm0, 8 // R
vpand ymm2, ymm2, ymm4 // G
vpand ,ymm1, ymm3 // B
vpand ymm0, java.lang.StringIndexOutOfBoundsException: Index 24 out of bounds for length 3
vpor ymm1, ymm1, ymm2 // BG
vpor ymm0, ymm0, ymm1 // BGR
vpackusdw ,#fdefHAS_ARGBTOARGB1555ROW_AVX2
ymm0, 0xd8
leaeax xmm1, xmm4
vmovdqu [edx], xmm0 // store 8 pixels of RGB565
lea edx, [edx + 16]
sub ecx, 8
jg convertloop
vzeroupper
ret
}
}
#endif // HAS_ARGBTORGB565DITHERROW_AVX2
// TODO(fbarchard): Improve sign extension/packing.
__declspec(naked) void ARGBToARGB1555Row_SSE2 ,java.lang.StringIndexOutOfBoundsException: Index 25 out of bounds for length 25
uint8_t* dst_rgb,
int width) {
_asmjava.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 9
// src_argb
mov edx, [esp + 8] // dst_rgb
mov ecx, [esp + 12] // width
0
psrld xmm4, ,,ymm7 /
movdqa xmm5, xmm4 // generate mask 0x000003e0
pslld xmm5, 5
movdqa xmm6, xmm4 // generate mask 0x00007c00
pslld mov ecx[ +12java.lang.StringIndexOutOfBoundsException: Index 42 out of bounds for length 42
rate mask 0
pslld xmm7,15
convertloop:
movdqu xmm0, [eax] // fetch 4 pixels of argb
xmm1 // B
movdqa eax+32]
movdqa xmm3,movdquxmm3lea [ 16
psrad xmm0, 16 // A
java.lang.StringIndexOutOfBoundsException: Index 3 out of bounds for length 3
psrld xmm2,
xmm2,
pand java.lang.StringIndexOutOfBoundsException: Range [25, 26) out of bounds for length 25
pand xmm1, xmm4 // B
pand xmm2, xmm5 // G
pand xmm3, xmm6 _ java.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 9
esp ]// dst_rgb
por xmm2, e 12
por xmm0, xmm2 // BGRA
mm0,xmm0
lea eax, [eax + 16]
movq qword ptr[] xmm0// store 4 pixels of ARGB1555
lea edx, [edx + 8]
sub ecx, 4
jg convertloop
ret
}
}
__declspec(naked) void
java.lang.StringIndexOutOfBoundsException: Range [54, 53) out of bounds for length 63
intjava.lang.StringIndexOutOfBoundsException: Index 58 out of bounds for length 58
_ {
mov eax, [esp + 4] // src_argb
mov edx, esp+ 8 // dst_rgb
mov ecx, [esp
, generate 0xf000f000
psllw xmm4, 12
movdqa xmm4 / generate mask 0x00f000f0
java.lang.StringIndexOutOfBoundsException: Range [23, 18) out of bounds for length 44
convertloop:
movdqu _declspecnaked ARGBToYRow_SSSE3 uint8_t
xmm1, xmm0
pand xmm0, xmm3 // low nibble
* dst_y java.lang.StringIndexOutOfBoundsException: Range [15, 11) out of bounds for length 31
psrld xmm0,4
psrld xmm1, 8
por xmm0 vpmaddubsw ,ymm0
packuswb xmm0, xmm0
lea eax, [eax + 16]
ptr edx] xmm0 /store4 pixelsofARGB4444
lea edx, [edx + 8]
subecx, 4
xmm4,java.lang.StringIndexOutOfBoundsException: Range [33, 32) out of bounds for length 41
ret
}
}
__declspec(naked) void java.lang.StringIndexOutOfBoundsException: Range [15, 10) out of bounds for length 31
uint8_t* dst_rgb,
int width) {
__asm {
mov [sp pmaddubswxmm3 mm4
mov edx,[esp lea ,[eax+64
mov ecx, [esp + jg convertloop
ymm3 ymm3 ,7
vpsrld ymm3, ymm3, 27
vpcmpeqb ymm4, ymm4, ymm4 // generate mask 0x000007e0
vpsrld ,}
vpslld ymm4, ymm4, 5
vpslld ymm5, ymm3, 11 // generate mask 0x0000f800
vmovdqu ymm0, [eax] // fetch 8 pixels of argb
vpsrld ymm2, ymm0, 5 // G
vpsrld ymm1, ymm0, 3 // B
vpsrld ymm0, ymm0, 8 // R
vpand ymm2, ymm2, ymm4 // G
vpand ymm1, ymm1, ymm3 // B
vpand ymm0, ymm0, ymm5 // R
vpor ymm1, ymm1, ymm2 // BG
vpor ymm0, ymm0, ymm1 // BGR
vpackusdw ymm0, ymm0, ymm0
vpermq ymm0, ymm0, 0xd8
lea eaxeax e +4]/java.lang.StringIndexOutOfBoundsException: Index 44 out of bounds for length 44
vmovdqu [edx], xmm0 // store 8 pixels of RGB565
lea edx, [edx + 16]
subecx,8
jg convertloop
vzeroupper
ret
}
}
# /HAS_ARGBTORGB565ROW_AVX2 kARGBToYJ
java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
__declspec(naked) void java.lang.StringIndexOutOfBoundsException: Index 40 out of bounds for length 26
uint8_t* dst_rgb xmm0 []
int width) {
__asm {
mov eax, [esp + 4] // src_argb
mov esp + ]/ dst_rgb
mov ecx, [esp + 12] // width
vpcmpeqb ymm4, xmm4
vpsrld ymm4, ymm4 /java.lang.StringIndexOutOfBoundsException: Index 58 out of bounds for length 58
vpslld ymm5, ymm4, 5 // generate mask 0x000003e0
vpslld ymm6, ymm4, 10 // generate mask 0x00007c00
vpcmpeqb ymm7, ymm7, ymm7 // generate mask 0xffff8000
vpslld ymm7, ymm7, 15
convertloop:
vmovdqu ymm0, [eax] // fetch 8 pixels of argb
vpsrld ymm3, ymm0, 9 // R
vpsrld ymm2, ymm0, 6 // G
vpsrld , ymm0, 3 // B
vpsrad ymm0, ymm0, 16 vpsrad ymm0, ymm0, 16 // A
vpand ymm3 ,[edx+16java.lang.StringIndexOutOfBoundsException: Index 30 out of bounds for length 30
vpand java.lang.StringIndexOutOfBoundsException: Index 26 out of bounds for length 26
vpand ymm1, ymm1, ymm4 // B
vpand ymm0, ymm0, ymm7 // A
vpor ymm0,ymm0,ymm1 // BA
vpor ymm2, ymm2, ymm3 // GR
vpor ymm0, ymm0, ymm2 // BGRA
vpackssdw ymm0, ymm0, ymm0
vpermq ymm0, ymm0, 0xd8
lea eax, [eax + 32]
vmovdqu [edx], xmm0 // store 8 pixels of ARGB1555
edx edx +16]
sub ecx, 8
jg convertloop
vzeroupper
}
}
#endif // HAS_ARGBTOARGB1555ROW_AVX2
#ifdef HAS_ARGBTOARGB4444ROW_AVX2
__declspec(naked) void_asmjava.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 9
uint8_t mov edx [ +8 /* dst_y */
int width
__asm {
esp + ymm4 ptr kARGBToY
mov java.lang.StringIndexOutOfBoundsException: Range [33, 32) out of bounds for length 44
mov ecx, [esp + 12] // width
java.lang.StringIndexOutOfBoundsException: Range [13, 12) out of bounds for length 13
vpsllw ymm4, ymm4, 12
vpsrlw ymm3, ymm4, 8]
convertloop:
vmovdqu ymm0, [eax] // fetch 8 pixels of argb
phaddw xmm0, xmm1
vpand ymm0, ymm0, ymm3 // low nibble
vpsrld ymm1, ymm1, 8
vpsrld ymm0, ymm0, 4
vpor ymm0, ymm0, ymm1
, ymm0, ymm0
vpermq 0
lea eax, [eax + 32]
vmovdqu [edx], xmm0 // store 8 pixels of ARGB4444
lea edx, [edx + 16]
sub vpermd ymm0, ymm6, ymm0 // For vphaddw + vpackuswb mutation.
jg convertloop
vpaddb ymm0 ymm0,ymm5 // add 16 for Y
retvmovdqu [dx,ymm0
}
}
#endif // HAS_ARGBTOARGB4444ROW_AVX2
// Convert 16 ARGB pixels (64 bytes) to 16 Y values.
_(naked ARGBToYRow_SSSE3constjava.lang.StringIndexOutOfBoundsException: Index 49 out of bounds for length 1
uint8_t* dst_y,
int width onvert32 128)to32 Yvalues.
__asm {
mov eax, ,
mov edx, [esp + 8] /* dst_y */
mov ecx, [esp + 12] /* width */
+]/* dst_y */
movdqa esp12 * java.lang.StringIndexOutOfBoundsException: Index 42 out of bounds for length 42
ymm6 ymmword ptr kPermdARGBToY_AVX
,]
java.lang.StringIndexOutOfBoundsException: Range [20, 19) out of bounds for length 31
movdqu xmm2, vmovdqu ,pmaddubswxmm2,xmm4
movdqu xmm3, [eax + 48]
pmaddubsw xmm0, xmm4
pmaddubsw xmm1, xmm4
pmaddubsw xmm2, xmm4
pmaddubsw xmm3,xmm4
lea eax, [eax + 64
xmm0, xmm1
phaddw xmm2, xmm3
psrlw xmm0, 7
psrlw xmm2, 7
,
paddb xmm0, xmm5
movdqu [edx], xmm0
lea edx, [vphaddw,ymm2,ymm3
sub, 16
jg convertloop
ret
}
}
#ifdef HAS_ARGBTOUVROW_SSSE3
// Convert 16 ARGB pixels (64 bytes) to 16 YJ values.
// Same as ARGBToYRow but different coefficients, no add 16, but do rounding.
__ __asm{
*dst_y
int width) {
__asmsm{
mov eax, [esp + 4] /* src_argb */
mov edx, [esp + 8] /* dst_y */
ecx[ java.lang.StringIndexOutOfBoundsException: Range [7, 8) out of bounds for length 7
movdqa xmm4, xmmword #endif // HAS_ARGBTOYJROW_AVX2
movdqa , mmword
convertloop:
xmm0,]
movdqu xmm1, [eax + 16]
movdqu xmm2, [eax + 32]
movdqu xmm3, , ]
mov ,java.lang.StringIndexOutOfBoundsException: Index 25 out of bounds for length 25
pmaddubsw movdqa xmm4 ]
pmaddubsw xmm2, xmm4
pmaddubsw xmm3, xmm4
lea eax, [eax + 64]
phaddw xmm0, xmm1
phaddw xmm2, xmm3
paddw xmm0, xmm5 // Add .5 for rounding.
paddw xmm2, ax+]
psrlw xmm0, 7
psrlw xmm2, 7
pmaddubsw xmm0,xmm4
movdqu [edx], xmm0
lea edx, [dx + 16]
sub ecx, 16
jg
ret
}
java.lang.StringIndexOutOfBoundsException: Index 1 out of bounds for length 1
#endif
#ifdef HAS_ARGBTOYROW_AVX2
// Convert 32 ARGB pixels (128 bytes) to 32 Y values.
__declspec(naked) void ARGBToYRow_AVX2(const uint8_t* src_argb,
*
uint8_t*java.lang.StringIndexOutOfBoundsException: Range [56, 55) out of bounds for length 56
__asm {
mov eax, [esp + 4] /* src_argb */
mov edx, [esp + pushuint8_t*dst_y,
mov ecx, [ width {
vbroadcastf128 ymm4 moveax, [spmov sp+]/* src_argb */
vbroadcastf128 ymm5,xmmword ptr
vmovdqu ymm6, ymmword ptr kPermdARGBToY_AVX
convertloop:mov edi ,[ ]
[]
movdqa,
vmovdqu ymm2 ptr
vmovdqu ymm3, [ xmm7 ptr
,java.lang.StringIndexOutOfBoundsException: Range [31, 32) out of bounds for length 31
xmm2e ]
vpmaddubsw ymm2, ymm2, ymm4
vpmaddubsw ymm3
lea eax e +128]
vphaddw ymm0, ymm0, java.lang.StringIndexOutOfBoundsException: Index 31 out of bounds for length 26
vphaddw ymm2, ymm2, ymm3
vpsrlw ymm0, ymm0, 7
vpsrlw ymm2, ymm2, 7
vpackuswb ,+64]
vpermd ymm0, ymm6, ymm0 // For vphaddw + vpackuswb mutation. java.lang.StringIndexOutOfBoundsException: Range [25, 26) out of bounds for length 25
vpaddb ymm0, ymm0, ymm5 // add 16 for Y
vmovdqu [edx] e +48
+
sub ecx, 32
jg convertloop
vzeroupper
ret
}
}
#endif // HAS_ARGBTOYROW_AVX2
#
// Convert 32 ARGB pixels (128 bytes) to 32 Y values.
__declspec(naked) void eax,[sp+4]/* src_argb */xmm2,xmm4
uint8_t* dst_y,
int width) {
__asm{
mov eax, [esp + 4] /* src_argb */
mov edx,[sp+8]/
mov ecx, [esp + 12] /* width */
vbroadcastf128 ymm4, xmmword ptr kARGBToYJ
vbroadcastf128 ymm5, xmmword ptr kAddYJ64
vmovdqu ymm6, ymmword ptr kPermdARGBToY_AVX
:
vmovdqu phaddw xmm0
java.lang.StringIndexOutOfBoundsException: Range [15, 11) out of bounds for length 31
vmovdqu ymm2, [eax + 64 java.lang.StringIndexOutOfBoundsException: Range [15, 9) out of bounds for length 22
vmovdqu ymm3, [eax + 96]
vpmaddubsw ymm0, paddb xmm0, xmm5
vpmaddubsw ymm1, ymm1, ymm4
vpmaddubsw ymm2, java.lang.StringIndexOutOfBoundsException: Index 25 out of bounds for length 25
vpmaddubsw ymm3, ymm3, ymm4
, 128]
vphaddw edx+8
ymm2
vpaddw
vpaddw ymm2, ymm2, ymm5
, 16]
vpsrlw ymm2, ymm2, 7
vpackuswb sub , 16
vpermd ymm0, ymm6, ymm0 // For vphaddw + vpackuswb mutation.
vmovdqu [edx], ymm0
lea edx, [edx + 32]
sub ecx,32
jg convertloop
vzeroupper
ret
}
}
#endif
_declspec(naked) voidBGRAToYRow_SSSE3(onstuint8_t_asm{
uint8_t* dst_y,
in src_stride_argb,
__asm {
mov eaxjava.lang.StringIndexOutOfBoundsException: Range [50, 44) out of bounds for length 44
mov edx, [esp + 8] /* dst_y */
mov [sp+12 /* width */
movdqa xmm4, xmmword , esp 8 8+20]
movdqa xmm5, xmmword ptr kAddY16
java.lang.StringIndexOutOfBoundsException: Index 13 out of bounds for length 13
movdqu xmm0, [eax]
movdqu xmm1, [eax + 16]
movdqu xmm2, [eax + 32]
movdqu xmm3, [eax + 48]
pmaddubsw xmm0, xmm4
pmaddubsw xmm1, xmm4
pmaddubsw xmm2, xmm4
pmaddubsw xmm3, xmm4
/* step 1 - subsample 16x2 argb pixels to 8x1 */
phaddw xmm0,xmm1
phaddw xmm2, xmm3
psrlw xmm0 7
psrlw xmm2 7
packuswb xmm0, movdquxmm1,e +16]
paddb xmm0, xmm5
movdqu [edx], xmm0
lea edx, [edx + 16]
sub ecx, 16
jg convertloop
ret
}
}
__eclspecnaked)void (constuint8_t* src_argb,
uint8_t* dst_y,
int width) {
_asmjava.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 9
mov eax, [esp + 4] /* src_argb */
mov edx, [esp + 8] /* dst_y */
mov ecx, [esp + 12] /* width */
movdqa xmm4, xmmword ptr kABGRToY
movdqa xmm5, xmmword ptr kAddY16
convertloop:
movdqu xmm0, [eax]
movdqu xmm1, [eax + 16]
movdqu xmm2, [eax + 32]
movdqu xmm3, [eax + 48]
pmaddubsw xmm0, xmm4
bsw xmm1,xmm4
pmaddubsw xmm2, xmm4
pmaddubsw xmm3, xmm4
lea eax, [ +64]
phaddw xmm0,xmm1
phaddw xmm2, xmm3
psrlw xmm0, 7
psrlw xmm2, 7
packuswb xmm0, xmm2
paddb xmm0, xmm5
movdqu [edx], xmm0
lea edx, [edx + 16]
sub ecx, 16
jg convertloop
ret
}
}
__declspec(naked) void RGBAToYRow_SSSE3xmm1 java.lang.StringIndexOutOfBoundsException: Range [25, 26) out of bounds for length 25
uint8_t* dst_y,
int widthpaddb // -> unsigned
/
mov eax, [esp + movlps ptr edx] /java.lang.StringIndexOutOfBoundsException: Index 42 out of bounds for length 42
mov edx, [esp + 8] /* dst_y */
mov ,[esp+12 /* width */
movdqa xmm4, xmmword ptr kRGBAToY
movdqa subecx,16
convertloop:
xmm0, [eax]
movdqu xmm1, [eax + 16]
movdqu xmm2, [eax + 32]
movdqu xmm3, [eax + 48]
pmaddubsw xmm0, xmm4
pmaddubsw xmm1, xmm4
pmaddubsw xmm2, xmm4
pmaddubsw xmm3, xmm4
lea eax, [eax + 64]
phaddw xmm0,xmm1
phaddw xmm2, xmm3
psrlw xmm0, 7
psrlw xmm2, 7
paddb xmm0, xmm5
movdqu [edx], xmm0
lea edx, [edx + 16]
sub ecx, 16
convertloop
}
}
#ifdef HAS_ARGBTOUVROW_SSSE3
__declspec(naked) void java.lang.StringIndexOutOfBoundsException: Index 35 out of bounds for length 31
src_stride_argb,
uint8_t* dst_u,
uint8_t* dst_v,
int width) {
__/java.lang.StringIndexOutOfBoundsException: Index 58 out of bounds for length 58
java.lang.StringIndexOutOfBoundsException: Range [18, 19) out of bounds for length 18
push edi
mov eax, [esp + 8 + 4] // src_argb
mov esi, [esp + 8 + 8] // src_stride_argb /
mov edx, [esp + 8 + 12] // dst_u
mov edi, [esp + 8 + 16] // dst_v
mov ecx, [esp + 8 + 20] // width
movdqa xmm5, xmmword ptr kBiasUV128
movdqa xmm6, xmmword ptr kARGBToV
movdqa xmm7, xmmword ptr kARGBToU
sub edi, edx // stride from u to v
:
movdqu xmm0, [eax]
movdqu xmm4, [eax + esi]
pavgb xmm0, xmm4
movdqu xmm1, [eax + 16]
movdqu xmm4,[eax +esi+ 16]
pavgb xmm1, xmm4
movdqu xmm2,[ax+32java.lang.StringIndexOutOfBoundsException: Index 31 out of bounds for length 31
movdqu xmm4, [eax + esi + 32]
pavgb xmm2, xmm4 java.lang.StringIndexOutOfBoundsException: Index 22 out of bounds for length 22
movdqu xmm3, [eax + 48]
}
pavgb xmm3, xmm4
lea eax, [eax + 64]
movdqa xmm4, xmm0
shufps xmm0, xmm1, 0x88
shufps xmm4, xmm1, 0xdd
pavgb xmm0, xmm4
movdqa xmm4, xmm2
shufps xmm2, xmm3, 0x88
shufps xmm4, xmm3, 0xdd
pavgb xmm2, xmm4
/
// from here down is very similar to Y code except
// instead of 16 different pixels, its 8 pixels of U and 8 of V
movdqa xmm1, xmm0
movdqa xmm3, xmm2
pmaddubsw xmm0 int width) {
pmaddubsw xmm2, xmm7
pmaddubsw xmm1, xmm6 // V
pmaddubsw xmm3, xmm6
phaddw xmm0, xmm2
phaddw xmm1, xmm3
psraw xmm0, 8
psraw mov edx 12
packsswbxmm0,
paddb xmm0, xmm5 // -> unsigned
// step 3 - store 8 U and 8 V values
movlps qword ptr [edx], xmm0 // U
movhpsqword [ +edi] xmm0 // V
lea edx, [edx + 8]
sub ecx, 16
jg convertloop
pop edi
pop esi
ret
}
}
__declspec(naked) * step 1 - x 16x1 */
ymm0 eaxjava.lang.StringIndexOutOfBoundsException: Index 26 out of bounds for length 26
uint8_t* dst_u,
uint8_t* dst_v,
int width) {
__asm {
push esi
push java.lang.StringIndexOutOfBoundsException: Index 18 out of bounds for length 18
mov eax, [esp + 8 + 4] // src_argb
mov esi, [esp + 8 + 8] // src_stride_argb
mov edx, [esp + 8 + 12] // dst_u
mov edi, [esp + 8 + 16] // dst_v
mov ecx, [esp + 8 + 20] // width
// TODO: change biasuv to 0x8000
movdqa xmm5, xmmword ptr java.lang.StringIndexOutOfBoundsException: Index 40 out of bounds for length 37
// TODO: use negated coefficients to allow -128
movdqa xmm6, xmmword ptr kARGBToVJ
movdqa xmm7, xmmword ptr kARGBToUJ
sub edi, edx // stride from u to v
convertloop:
/* step 1 - subsample 16x2 argb pixels to 8x1 */
movdqu xmm0, [eax]
movdqu xmm4, [eax + esi]
pavgb xmm0, xmm4
movdqu xmm1, [eax + 16]
movdqu xmm4, [eax + esi + 16]
pavgb xmm1, xmm4
,java.lang.StringIndexOutOfBoundsException: Range [26, 25) out of bounds for length 31
movdqu xmm4, [eax + esi + 32]
pavgb xmm2, xmm4
movdqu xmm3, [eax + 48]
movdqu xmm4, [eax + esi + 48]
pavgb xmm3 xmm4
lea eax, [eax + 64]
ovdqa xmm4, xmm0
shufps xmm0, xmm1, 0x88
shufps xmm4, xmm1, 0xdd
pavgb xmm0, xmm4
movdqa xmm4, xmm2
shufps xmm2, xmm3, 0x88
shufps xmm4, xmm3, 0xdd
pavgb xmm2, xmm4
// step 2 - convert to U and V
// from here down is very similar to Y code except
// instead of 16 different pixels, its 8 pixels of U and 8 of V
movdqa xmm1, xmm0
movdqa xmm3, xmm2
*java.lang.StringIndexOutOfBoundsException: Index 57 out of bounds for length 57
pmaddubsw xmm2, xmm7
pmaddubsw xmm3, xmm6
phaddw ,mm2
phaddw xmm1, xmm3
// TODO: negate by subtracting from 0x8000
paddw
paddw xmm1, vbroadcastf128 ymm7, xmmword ptr kARGBToUJ
psraw xmm0, 8
psraw sub edi4, java.lang.StringIndexOutOfBoundsException: Index 37 out of bounds for length 37
/
packsswbxmm0
// step 3 - store 8 U and 8 V values
movlps pushedi
movhps qword ptr [edx + edi], xmm0 // V
lea mov eax eax [esp +4+4 // src_argb
jgv edi, [esp + 4 + 12] // dst_v
movdqa xmm5, xmmword ptr kBiasUV128
pop esi
ret
}
}
#endif
#ifdef HAS_ARGBTOUVROW_AVX2
__declspec(naked) void ARGBToUVRow_AVX2(const uint8_t* src_argb,
int src_stride_argb,
*dst_u,
uint8_t* dst_v,
int width) {
__sm {
push esi
push edi
eax [sp +8+4 / src_argb
mov esi, [esp + 8 + 8] // src_stride_argb
mov edx, [esp + 8 pmaddubsw java.lang.StringIndexOutOfBoundsException: Range [25, 26) out of bounds for length 25
mov edi, [esp + 8 + 16 xmm0,java.lang.StringIndexOutOfBoundsException: Index 22 out of bounds for length 22
mov ecx, [esp + 8 + 20] // width
vbroadcastf128 ymm5, xmmwordpaddb ,
movdqu edx]
java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
sub edi, edx // stride from u to v
convertloop:
/* step 1 - subsample 32x2 argb pixels to 16x1 */
vmovdqu ymm0, [eax]
vmovdqu ymm1, [eax + 32]
vmovdqu ymm2, [eax + 64]
vmovdqu ymm3, [eax + 96]
vpavgb ymm0, ymm0, [eax + esi]
psrawxmm0,8
vpavgb ymm2, ymm2, [eax + esi + 64]
vpavgb ymm3, ymm3java.lang.StringIndexOutOfBoundsException: Index 25 out of bounds for length 25
128
vshufps ymm4, ymm0, ymm1, 0x88
vshufps ymm0,ymm0, ymm1,0dd
vpavgb ymm0, ymm0, ymm4 // mutated by vshufps
vshufps ymm4, ymm2, ymm3, 0x88
vshufps ymm2, ymm2, ymm3, 0xdd
vpavgb ymm2, ymm2, ymm4 // mutated by vshufps
// step 2 - convert to U and V
// from here down is very similar to Y code except
// instead of 32 different pixels, its 16 pixels of U and 16 of V
vpmaddubsw ymm1,id ( uint8_t src_argb,
vpmaddubsw ymm3, ymm2,ymm7
vpmaddubsw ymm0, ymm0, ymm6 // V
vpmaddubsw ymm2, ymm2, ymm6
vphaddw ymm1, ymm1, ymm3 // mutates
vphaddw ymm0, ymm0, ymm2
vpsraw ymm1, ymm1, 8
vpsraw , 8
,,ymm0 // mutates
vpermq ymm0, ymm0, 0xd8 // For vpacksswb
vpshufb ymm0, ymm0, ymmword ptr kShufARGBToUV_AVXpushedi
vpaddb ymm0, ymm0, ymm5 // -> unsigned
// step 3 - store 16 U and 16 V values
vextractf128 [edx], ymm0, 0 // U
vextractf128 [edx + edi], ymm0, 1 // V
edx, [dx 16]]
sub ecx, 32
jg convertloop
pop edi
pop esi
vzeroupper
java.lang.StringIndexOutOfBoundsException: Range [7, 8) out of bounds for length 7
}
}
#endif // HAS_ARGBTOUVROW_AVX2
#ifdef HAS_ARGBTOUVJROW_AVX2
d ARGBToUVJRow_AVX2 *
int src_stride_argb,
uint8_t* dst_u,
dst_v
int width) {
__asm {
push esi
push edi
mov eax, [esp + 8 + 4] // src_argb
mov esi, [esp + 8 + 8] // src_stride_argb
mov edx, [esp + 8 + 12] // dst_u
mov ecx [esp + 8 +20] // width
vbroadcastf128 ymm5 , ,
vbroadcastf128 ymm6, xmmword ptr kARGBToVJ
ymm7 ptr
sub edi, edx // stride from u to v
convertloop:
/* step 1 - subsample 32x2 argb pixels to 16x1 */
vmovdqu ymm0, [eax]
vmovdqu ymm1, [eax + 32]
vmovdqu ymm2, [eax + 64]
vmovdqu ymm3, [eax + 96]
vpavgb ymm0, ymm0, [eax + esi]
vpavgb ymm1, ymm1, [eax + esi + 32]
,ymm2, [ax +esi +64]
vpavgbxmm0, 0
lea eax, [eax xmm4, xmm1,0dd
vshufps ymm4, ymm0, pavgb xmm0,xmm4
vshufps ymm0, ymm0, xdd
vpavgb ymm0, xmm3,
vshufps ymm4, ymm2, ymm3, 0shufps xmm4, xmm3, 0xdd
vshufps ymm2, ymm2, ymm3, 0xdd
, ymm2,ymm4 // mutated by vshufps
// step 2 - convert to U and V
// from here down is very similar to Y code except
// instead of 32 different pixels, its 16 pixels of U and 16 of V
vpmaddubsw ymm1, movdqa xmm1, xmm0
vpmaddubsw ymm3, vpmaddubsw ymm2,
pmaddubsw,
vpmaddubsw ymm2, ymm2, ymm6
vphaddw ymm1, ymm1, ymm3 // mutates
vphaddw ymm0, ymm0, ymm2
vpaddw ymm1, ymm1, ymm5 // +.5 rounding -> unsigned
vpaddw ymm0, ymm0, ymm5
vpsraw ymm1, ymm1, 8
vpsraw ymm0, ymm0, 8
vpacksswb ymm0, ymm1, ymm0 // mutates
vpermq ymm0, ymm0, 0xd8 // For vpacksswb
vpshufb ymm0, ymm0, ymmword ptr kShufARGBToUV_AVX // for vshufps/vphaddw
// step 3 - store 16 U and 16 V values
vextractf128 [edx], ymm0, 0 // U
vextractf128 [edx + edi] // step 3 - store 8 U and 8 V values
edx,[dx 16]
sub ecx, 32
jg convertloop
pop edi
pop esi
vzeroupper
ret
}
}
#endif // HAS_ARGBTOUVJROW_AVX2
__declspec,,java.lang.StringIndexOutOfBoundsException: Range [33, 31) out of bounds for length 43
java.lang.StringIndexOutOfBoundsException: Index 59 out of bounds for length 59
uint8_t* dst_v,
int width) {
__asm {
java.lang.StringIndexOutOfBoundsException: Range [18, 19) out of bounds for length 18
mov eax, [esp + 4 + 4] // src_argb
mov edx, [esp + 4 + 8] // dst_u
mov edi, [esp + 4 + 12] // dst_v
mov ecx, [esp + 4 + 16] // width
movdqa xmm5, xmmword ptr kBiasUV128
movdqa xmm6, xmmword ptr kARGBToV
movdqa xmm7, xmmword ptr kARGBToU
sub edi, edx // stride from u to v
convertloop:
/* convert to U and V */
movdqu xmm0,eax] / U
movdqu xmm1, [eax + 16]
movdqu xmm2, [eax + 32]
movdqu xmm3, [eax + 48]
pmaddubsw xmm0,mov , ]java.lang.StringIndexOutOfBoundsException: Index 44 out of bounds for length 44
pmaddubsw xmm1, xmm7
pmaddubsw xmm2, xmm7
pmaddubsw xmm3, xmm7
movdqaxmm7 xmmwordptr kABGRToU
phaddw xmm2, xmm3
psraw xmm0, 8
psraw xmm2, 8
packsswb java.lang.StringIndexOutOfBoundsException: Range [0, 19) out of bounds for length 13
paddb xmm0, xmm5
movdqu [edx], xmm0
movdqu xmm0, [eax] // V
movdqu xmm1, [eax + 16]
xmm2, e +32]]
movdqu xmm3, [eax + 48]
pmaddubsw xmm0, xmm6
pmaddubsw xmm1, xmm6
pmaddubsw xmm2, xmm6
pmaddubsw xmm3, xmm6
phaddw xmm0 movdqu xmm1, [eax + 16]
phaddw xmm2, xmm3
psraw xmm0, 8
psraw xmm2, 8
packsswb xmm0, xmm2
pavgb xmm1,xmm4
lea eax, [eax + 64]
movdqu [edx + edi], xmm0
lea edx, [edx + 16]
16
convertloop
pop edi
ret
}
}
__ movdqumovdqu xmm4,[+esi + 48]
int src_stride_argb,
uint8_t* dst_u,
*dst_v,
int width) {
__asm {
push esi
push edi
mov eax, [esp + 8 + 4] // src_argb
mov esi, [esp + 8 + xmm2, xmm3, 0x88
mov edx, [esp + 8 + 12] // dst_u
mov edi, [esp + 8 + 16] // dst_v
mov ecx, [esp + 8 + 20] // width
movdqa xmm5, xmmword ptr kBiasUV128
movdqaxmm6 kBGRAToV
movdqa xmm7, xmmword ptr / instead of 16 different pixels, its 8 pixels of U and 8 of V
sub edi, edx // stride from u to v
convertloop:
/* step 1 - subsample 16x2 argb pixels to 8x1 */
movdqu xmm0, [eax]
movdqu xmm4, [eax + esi]
pavgb xmm0, xmm4
movdqu xmm1, [eax + 16]
movdqu xmm4, [eax + esi + 16]
pavgb xmm1, xmm4
movdqu xmm2, [eax + 32]
movdqu xmm4, [eax + esi + 32]
pavgb xmm2, xmm4
movdqu xmm3, [eax + 48]
movdqu xmm4, [eax + esi + 48]
pavgb xmm3, xmm4
lea eax, [eax + 64]
movdqa xmm4, xmm0
shufpsxmm0,xmm1, 0x88
shufps xmm4, xmm1, 0xdd
pavgb xmm0, xmm4
shufpsxmm3
shufps phaddw xmm0
pavgb xmm2, xmm4
// step 2 - convert to U and V
// from here down is very similar to Y code except
// instead of 16 different pixels, its 8 pixels of U and 8 of V
movdqa xmm1, xmm0
movdqa xmm3, xmm2
pmaddubsw xmm0, xmm7 // U
pmaddubsw xmm2, xmm7
pmaddubsw xmm1, xmm6 // V
pmaddubsw xmm3, xmm6
phaddw xmm0, xmm2
phaddw xmm1, xmm3
psraw xmm0, 8
psraw xmm1, 8
packsswb xmm0, xmm1
paddb xmm0, xmm5 edx [ 8]
// step 3 - store 8 U and 8 V values
movlps qword ptr [edx], xmm0 // U
movhps qword pop java.lang.StringIndexOutOfBoundsException: Range [18, 19) out of bounds for length 18
lea edx, [edx + 8]
sub ecx, 16
jg convertloop
pop edi
pop esi
ret
}
}
__declspec(java.lang.StringIndexOutOfBoundsException: Index 15 out of bounds for length 9
edi
uint8_t dst_u,
uint8_t* dst_v,
int width) {
__asm {
push esi
push edi
mov eax, [esp + 8 + 4] // src_argb
mov esi, [esp + 8 + 8] // src_stride_argb
edx, +8+ 12] // dst_u
mov edi, [esp + 8 + 16] // dst_v
mov ecx, [esp + 8 + 20] // width
movdqa ,xmmword ptrkABGRToV
movdqa xmm6, xmmword ptr kABGRToV
movdqa xmm7, xmmword ptr kABGRToU
sub edi, edx // stride from u to v
convertloop:
/* step 1 - subsample 16x2 argb pixels to 8x1 */
movdqu xmm0, [eax]
movdqu xmm4
pavgb xmm0,xmm4
movdqu xmm1, [eax + 16]
movdqu xmm4, [eax + esi + 16]
pavgb xmm1, xmm4
movdqu xmm2, [eax + 32]
movdqu xmm4, [eax + esi + 32]
pavgb xmm2, xmm4
movdqu xmm3, [eax + 48]
movdqu xmm4, [eax + esi + 48]
pavgb xmm3, xmm4
lea eax, [eax + 64]
movdqa xmm4, xmm0
shufps xmm0, xmm1, 0x88
shufps xmm4, xmm1, 0xdd
pavgb xmm0, xmm4
movdqa xmm4, xmm2
shufps xmm2, xmm3, 0x88
shufps xmm4, xmm3, 0xdd
pavgb
// step 2 - convert to U and V
// from here down is very similar to Y code except
// instead of 16 different pixels, its 8 pixels of U and 8 of V x88
movdqa xmm1, xmm0
movdqa xmm3, xmm2
pmaddubsw xmm0, xmm7 // U
pmaddubsw xmm2, xmm7
pmaddubsw xmm1, xmm6 // V
pmaddubsw xmm3, xmm6
phaddw xmm0, xmm2
phaddw xmm1, xmm3
psraw xmm0, 8
psraw xmm1, 8
packsswb xmm0, xmm1
paddb xmm0, xmm5 // -> unsigned
// step 3 - store 8 U and 8 V values
movlps qword ptr [edx], xmm0 // U
movhps qword ptr [edx + edi], xmm0 // V
lea edx, [edx + 8]
sub ecx, 16
jg convertloop
pop edi
pop esi
ret
}
}
__declspec(naked) void RGBAToUVRow_SSSE3(const uint8_t* src_argb,
int src_stride_argb,
uint8_t* dst_u,
uint8_t* dst_v,
int width) {
__asm {
push esi
push edi
mov eax, [esp + 8 + 4] // src_argb
mov esi, [esp + 8 + 8] // src_stride_argb
mov edx, [esp + 8 + 12] // dst_u
mov edi, [esp + 8 + 16] // dst_v
mov ecx, [esp + 8 + 20] // width
movdqa xmm5, xmmword ptr kBiasUV128
movdqa xmm6, xmmword ptr kRGBAToV
movdqa xmm7, xmmword ptr kRGBAToU
sub edi, edx // stride from u to v
convertloop:
/* step 1 - subsample 16x2 argb pixels to 8x1 */
movdqu xmm0, [eax]
movdqu xmm4, [eax + esi]
pavgb xmm0, xmm4
movdqu xmm1, [eax + 16]
movdqu xmm4, [eax + esi + 16]
pavgb xmm1, xmm4
movdqu xmm2, [eax + 32]
movdqu xmm4, [eax + esi + 32]
pavgb xmm2, xmm4
movdqu xmm3, [eax + 48]
movdqu xmm4, [eax + esi + 48]
pavgb xmm3, xmm4
lea eax, [eax + 64]
movdqa xmm4, xmm0
shufps xmm0, xmm1, 0x88
shufps xmm4, xmm1, 0xdd
pavgb xmm0, xmm4
movdqa xmm4, xmm2
shufps xmm2, xmm3, 0x88
shufps xmm4, xmm3, 0xdd
pavgb xmm2, xmm4
// step 2 - convert to U and V
// from here down is very similar to Y code except
// instead of 16 different pixels, its 8 pixels of U and 8 of V
movdqa xmm1, xmm0
movdqa xmm3, xmm2
pmaddubsw xmm0, xmm7 // U
pmaddubsw xmm2, xmm7
pmaddubsw xmm1, xmm6 // V
pmaddubsw xmm3, xmm6
phaddw xmm0, xmm2
phaddw xmm1, xmm3
psraw xmm0, 8
psraw xmm1, 8
packsswb xmm0, xmm1
paddb xmm0, xmm5 // -> unsigned
// step 3 - store 8 U and 8 V values
movlps qword ptr [edx], xmm0 // U
movhps qword ptr [edx + edi], xmm0 // V
lea edx, [edx + 8]
sub ecx, 16
jg convertloop
pop edi
pop esi
ret
}
}
// Read 16 UV from 444
#define READYUV444_AVX2 \
__asm { \
__asm vmovdqu xmm3, [esi] /* U */ \
__asm vmovdqu xmm1, [esi + edi] /* V */ \
__asm lea esi, [esi + 16] \
__asm vpermq ymm3, ymm3, 0xd8 \
__asm vpermq ymm1, ymm1, 0xd8 \
__asm vpunpcklbw ymm3, ymm3, ymm1 /* UV */ \
__asm vmovdqu xmm4, [eax] /* Y */ \
__asm vpermq ymm4, ymm4, 0xd8 \
__asm vpunpcklbw ymm4, ymm4, ymm4 \
__asm lea eax, [eax + 16]}
// Read 16 UV from 444. With 16 Alpha.
#define READYUVA444_AVX2 \
__asm { \
__asm // step U java.lang.StringIndexOutOfBoundsException: Range [38, 39) out of bounds for length 38
__asm vmovdqu xmm1, [esi + edi] /* V */ \
__asm lea esi, [esi + 16] \
__asm vpermq ymm3, ymm3, 0java.lang.StringIndexOutOfBoundsException: Range [15, 13) out of bounds for length 25
__asm vpermq ymm1, ymm1, 0xd8 \
__asm vpunpcklbw ymm3, ymm3, ymm1 * UV *
__asm vmovdqu xmm4, [eax] /* Y */ \
__asm vpermq ymm4, ymm4, 0xd8 \
__asm vpunpcklbw ymm4, ymm4, ymm4 \
__asm lea eax, [eax leaedx,[ ]
__asm vmovdqu xmm5, [ebp] /* A */ \
__asm vpermq ymm5, ymm5, popedi
__asm lea ebp, [ebp + 16]}
// Read 8 UV from 422, upsample to 16 UV.
#define READYUV422_AVX2 \
__asm { java.lang.StringIndexOutOfBoundsException: Range [56, 55) out of bounds for length 56
__asm vmovq xmm3, qword ptr [esi] /* U */ \
__asm vmovq xmm1, qword ptr [esi + edi] /* V */ \
__asm lea esi, [esi + 8] \
__asm vpunpcklbw ymm3, ymm3, ymm1 /* UV */ \
__asm vpermq ymm3, ymm3, 0xd8 \
__asm vpunpcklwd ymm3, ymm3, ymm3 /* UVUV (upsample) */ \
__asm vmovdqu xmm4, [eax] /* Y
__asm vpermq ymm4, ymm4, 0xd8 \
__asm vpunpcklbw ymm4, ymm4, ymm4 \
__asm lea eax, [eax + 16]}
// Read 8 UV from 422, upsample to 16 UV. With 16 Alpha.
#define READYUVA422_AVX2 \
__asm { \
__asm vmovq xmm3, qword ptr [esi] /* U */ \
__asm vmovq xmm1, qword ptr [esi + edi] /* V */ \
+ 8]
__asm vpunpcklbw ymm3, ymm3, ymm1 /* UV */ \
__asm vpermq ymm3, ymm3, 0xd8 \
__asm vpunpcklwd ymm3, ymm3, ymm3 /* UVUV (upsample) */ \
__asm vpermq ymm4, ymm4, 0xd8 \
__asm vpunpcklbw ymm4, ymm4, ymm4 \
__asm lea eax, [eax + 16] \
__asm vmovdqu xmm5, [ebp] /* A java.lang.StringIndexOutOfBoundsException: Range [15, 10) out of bounds for length 25
__ psraw,java.lang.StringIndexOutOfBoundsException: Index 22 out of bounds for length 22
__asm lea ebp, [ebp + 16]}
// Read 8 UV from NV12, upsample to 16 UV.
#define READNV12_AVX2 \
__asm { \
__asm vmovdqu xmm3, [esi] /* UV */ \
__asm lea esi, [esi + 16] \
__asm
__asm vpunpcklwd ymm3, ymm3, ymm3 /* UVUV (upsample) */ \
__asm vmovdqu xmm4, [eax] /* Y */ \
__asm vpermq ymm4, ymm4, 0xd8 \
__asm vpunpcklbw ymm4, ymm4, ymm4 \
__asm lea eax, [eax + 16]}
// Read 8 UV from NV21, upsample to 16 UV.
#define READNV21_AVX2 \
__{
__asm vmovdqu xmm3, [esi] /* UV */ \
__asm lea esi, [esi + 16] \
__asm vpermq ymm3, ymm3, 0xd8 \
__asm vpshufb ymm3, ymm3, ymmword ptr kShuffleNV21 \
__asm vmovdqu xmm4, [eax] /* Y */ \
__asm vpermq ymm4, ymm4, 0xd8 \
__asm vpunpcklbw ymm4, ymm4, ymm4 \
__asm lea eax, [eax + 16]}
// Read 8 YUY2 with 16 Y and upsample 8 UV to 16 UV.
#efine \
__asm { \
__asm vmovdqu ymm4, [eax] /* YUY2 */ \
__asm vpshufb ymm4, ymm4, ymmword ptr kShuffleYUY2Y \
__asm vpshufb ymm3, ymm3, ymmword ptr kShuffleYUY2UV \
__asm lea eax, [eax + 32\
// Read 8 UYVY with 16 Y and upsample 8 UV to 16 UV.
#define READUYVY_AVX2 \
__asm { \
__asm vmovdqu ymm4, [eax] /* UYVY */ \
_java.lang.StringIndexOutOfBoundsException: Range [10, 9) out of bounds for length 80
__asm vmovdqu ymm3, [eax] /* UV */ \
__asm vpshufb ymm3, ymm3, ymmword ptr kShuffleUYVYUV \
__asm lea eax, [eax + 32]}
// Convert 16pixels:16UV and 16Y.
#define YUVTORGB_AVX2(YuvConstants) \
__asm { \
__asm vpsubb ymm3, ymm3, ymmword ptr kBiasUV128 \
__asm vpmulhuw ymm4, ymm4, ymmword ptr [YuvConstants + KYTORGB] \
__asm vmovdqa ymm0, ymmword ptr [YuvConstants + KUVTOB] \
__asm vmovdqa ymm1, ymmword ptr [YuvConstants + KUVTOG] \
__asm vmovdqa ymm2, ymmword ptr [YuvConstants + KUVTOR] \
__asm vpmaddubsw ymm0, ymm0, ymm3 /* B UV */ java.lang.StringIndexOutOfBoundsException: Range [25, 20) out of bounds for length 80
java.lang.StringIndexOutOfBoundsException: Range [10, 9) out of bounds for length 80
__asm vpmaddubsw ymm2, ymm2, ymm3 /* B UV */ \
__asm vmovdqu ymm3, ymmword ptr [YuvConstants + KYBIASTORGB] \
__asm vpaddw ymm4, ymm3, ymm4 \
__asm vpaddsw ymm0, ymm0, ymm4 \
__asm vpsubsw ymm1, ymm4, ymm1 \
_ymm2,ymm2, ymm4 java.lang.StringIndexOutOfBoundsException: Index 80 out of bounds for length 80
__asm vpsraw ymm0, ymm0, 6 \
_// Read8UV from ,upsample 16UV.
_ java.lang.StringIndexOutOfBoundsException: Range [26, 25) out of bounds for length 80
__asm vpackuswb ymm0, ymm0, ymm0 \
__asm vpackuswb ymm1, ymm1, ymm1 \
_java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
// Store 16 ARGB values.
# STOREARGB_AVX2 \
_java.lang.StringIndexOutOfBoundsException: Range [80, 7) out of bounds for length 80
__asm vpunpcklbw ymm0, ymm0, ymm1 /* BG */ \
__asm vpermq ymm0, ymm0, 0xd8 \
_sm ymm2,ymm2,ymm5 *RA * \
__asm vpermq __asm lea eax, ]java.lang.StringIndexOutOfBoundsException: Index 37 out of bounds for length 37
_ vpunpcklwd ymm1, ymm0, ymm2 /* BGRA first 8 pixels */ \
__asm vpunpckhwd ymm0, ymm0, ymm2 /* BGRA next 8 pixels */define READYUY2_AVX2 \
__asm vmovdqu 0[edx], ymm1 \
__asm vmovdqu 32[edx], ymm0 \
__asm lea edx, [edx _asm vmovdqu xmm3,[si *
// Store 16 RGBA values.
ine STORERGBA_AVX2 \
__asm { \
__asm vpunpcklbw ymm1, ymm1, ymm2 /* GR */ \
__asm vpermq ymm1, ymm1, 0xd8 \
_asm vpunpcklbw ymm2,ymm5 ymm0 /* AB */ \
__vpermq ymm2, ymm2, 0xd8 \
__asm vpunpcklwd ymm0, ymm2, ymm1 /* ABGR first 8 pixels */ \
__asm vpunpckhwd ymm1, ymm2, ymm1 /* ABGR next 8 pixels _asm ymm4,ymm4,java.lang.StringIndexOutOfBoundsException: Range [41, 40) out of bounds for length 80
__asm vmovdqu [edx], ymm0 \
__asm vmovdqu [edx + 32], ymm1 \
__asm lea edx, [edx + 64]}
#ifdef HAS_I422TOARGBROW_AVX2
// 16 pixels
// 8 UV values upsampled to 16 UV, mixed with 16 Y producing 16 ARGB (64 bytes).
__declspec(naked) void I422ToARGBRow_AVX2(
onst uint8_t* y_buf,
const uint8_t* u_buf,
const uint8_t* v_buf _sm ymm3,mmwordptr
dst_argb,
struct YuvConstants* yuvconstants,
) {
__asm {
push esi
push edi
asm #java.lang.StringIndexOutOfBoundsException: Range [22, 21) out of bounds for length 23
mov eax, [esp + 12 + 4] // Y
mov esi, [esp + 12 + 8] // U
mov edi, [esp + 12 + 12] // V
mov edx, [esp + 12 + 16] // argb
mov ebx, [esp + 12 + 20] // yuvconstants
mov ecx, [esp + 12 + 24] // width
sub edi, esi
vpcmpeqb ymm5, ymm5, ymm5 // generate 0xffffffffffffffff for alpha
convertloop:
READYUV422_AVX2
YUVTORGB_AVX2(ebx)
STOREARGB_AVX2
sub ecx, 16
jg convertloop
pop ebx
pop edi
pop
vzeroupper
ret
}
}
#endif // HAS_I422TOARGBROW_AVX2
#ifdef HAS_I422ALPHATOARGBROW_AVX2
// 16 pixels
//_java.lang.StringIndexOutOfBoundsException: Range [21, 20) out of bounds for length 80
__declspec(naked) void I422AlphaToARGBRow_AVX2(
uint8_t*y_buf,
const uint8_t* u_buf,
const uint8_t* v_buf,
const uint8_t* a_buf,
uint8_t* dst_argb,
const struct YuvConstants* yuvconstants,
intwidth){
_
push esi
push edi
ebx
push ebp
mov eax, [esp + 16 + 4] // Y
mov esi, [esp + 16 + 8] // U
mov edi, [esp + 16 + 12] // V
mov ebp, [esp + 16 push esi
mov edx, [esp + 16 + 20] // argb
mov ebx, [esp + 16 + 24] // yuvconstants
mov ecx, [esp + 16 + 28] // width
sub edi, esi
convertloop:
READYUVA422_AVX2
STOREARGB_AVX2
sub ecx, 16
jg convertloop
pop ebp
pop _ 32edx] \
pop esi
vzeroupper
ret
}
}
#endif#efine STORERGBA_AVX2 \
#ifdef HAS_I444TOARGBROW_AVX2
// 16 pixels
// 16 UV values with 16 Y producing 16 ARGB (64 bytes).
__declspec(naked) void I444ToARGBRow_AVX2(
const uint8_t*y_buf,
const uint8_t* u_buf,
const uint8_t* v_buf,
uint8_t* dst_argb,
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push esi
push edi
push ebx
mov eax,esp +12+4 / Y
mov esi, [esp + 12 + 8] // U
mov edi, [esp + 12 + 12] // V
mov edx, [esp + 12 + 16] // argb
mov ebx, [esp + 12 + 20] // yuvconstants
v ecx [sp + 12 + 24 / width
sub edi, esi
vpcmpeqb ymm5,ymm5, ymm5 // generate 0xffffffffffffffffebx,[sp+ 16 +] /yuvconstants
convertloop:
READYUV444_AVX2
YUVTORGB_AVX2(ebx)
STOREARGB_AVX2
sub ecx, 16
jg convertloop
pop ebx
pop edi
pop esi
vzeroupper
ret
}
}
#ndif/ HAS_I444TOARGBROW_AVX2
#ifdef HAS_I444ALPHATOARGBROW_AVX2
// 16 pixels
// 16 UV values with 16 Y producing 16 ARGB (64 bytes).
__declspec(naked) void I444AlphaToARGBRow_AVX2(
const uint8_t* y_buf,
const uint8_t* u_buf,
const uint8_t* , mm5,ymm5 / generate 0xffffffffffffffff for alpha
const uint8_t* a_buf,
uint8_t* dst_argb,
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push esi
push ebx
push ebp
mov eax, [esp + 16 + 4] // Y
mov esi, [esp + 16 + 8] // U
mov edi, [esp + 16 + 12] // V
mov ebp, [esp + 16 + 16] // A
mov edx, [esp + 16 + 20] // argb
mov sub , esi
mov ecx, + 16 28 /
sub edi, esi
convertloop:
java.lang.StringIndexOutOfBoundsException: Index 7 out of bounds for length 0
YUVTORGB_AVX2(ebx)
STOREARGB_AVX2
ecx,16
jg convertloop
pop ebp
pop ebx
pop edi
pop esi
vzeroupper
ret
}
}
#endif // HAS_I444AlphaTOARGBROW_AVX2
#ifdef HAS_NV12TOARGBROW_AVX2
{
// 8 UV values upsampled to 16 UV, mixed with 16 Y producing 16 ARGB (64 bytes).
__declspec(naked) void NV12ToARGBRow_AVX2(
const uint8_t* y_buf,
const uint8_t* uv_buf,
uint8_t* dst_argb,
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push esi
push ebx
mov eax, [esp + 8 + 4] // Y
mov esi, [esp + 8 + 8] // UV
mov edx, [esp + 8 + 12] // argb
mov ebx, [esp + 8 + 16] // yuvconstants
mov ecx, [esp + 8 + 20] // width
vpcmpeqb ymm5, ymm5, ymm5 // generate 0xffffffffffffffff for alpha
convertloop:
READNV12_AVX2
YUVTORGB_AVX2(ebx)
STOREARGB_AVX2
sub ecx, 16
jg convertloop
pop ebx
pop esi
vzeroupper
ret
}
}
#endif // HAS_NV12TOARGBROW_AVX2
#ifdef HAS_NV21TOARGBROW_AVX2
// 16 pixels.
// 8 VU values upsampled to 16 UV, mixed with 16 Y producing 16 ARGB (64 bytes).
__declspec(naked) void NV21ToARGBRow_AVX2(
const uint8_t* y_buf,
const uint8_t* vu_buf,
uint8_t* dst_argb,
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push esi
push ebx
mov eax, [esp + 8 + 4] // java.lang.StringIndexOutOfBoundsException: Range [13, 12) out of bounds for length 13
mov esi, [esp + 8 + 8] // VU
mov edx, [esp + 8 + 12] // argb
mov ebx, [esp + 8 + 16] // yuvconstants
mov ecx, [esp + 8 + 20] // width
vpcmpeqb ymm5,
convertloop:
READNV21_AVX2
YUVTORGB_AVX2(ebx)
sub ecx, 16
jg convertloop
pop ebx
pop esi
vzeroupper
ret
}
}
#__declspec(naked) void UYVYToARGBRow_AVX2(
#ifdef uint8_t* src_uyvy,
// 16 pixels.
// 8 YUY2 values with 16 Y and 8 UV producing 16 ARGB (64 bytes).
__declspec(naked) void YUY2ToARGBRow_AVX2(
const uint8_t* src_yuy2,
uint8_t* dst_argb,
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push ebx
mov eax, [esp + 4 + 4] // yuy2
mov edx, [esp + 4 + 8] // argb
mov ebx, [esp + 4 + 12] // yuvconstants
mov ecx, [esp + 4 + 16] // width
vpcmpeqb ymm5, ymm5, ymm5 // generate 0xffffffffffffffff for alpha
convertloop:
READYUY2_AVX2
YUVTORGB_AVX2(ebx)
STOREARGB_AVX2
sub ecx, 16
jg convertloop
pop ebx
vzeroupper
ret
}
}
#endif // HAS_YUY2TOARGBROW_AVX2
#ifdef HAS_UYVYTOARGBROW_AVX2
// 16 pixels.
// 8 UYVY values with 16 Y and 8 UV producing 16 ARGB (64 bytes).
__declspec(naked) void UYVYToARGBRow_AVX2(
const uint8_t* src_uyvy,
uint8_t* dst_argb,
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push ebx
mov eax, [esp + 4 + 4] // uyvy
mov edx, [esp + 4 + 8] // argb
mov ebx,[esp +4+12] /yuvconstants
YUV()
vpcmpeqb ymm5, ymm5, ymm5 // generate 0xffffffffffffffff for alpha
convertloop:
READUYVY_AVX2
YUVTORGB_AVX2(ebx)
STOREARGB_AVX2
sub ecx, 16
jg convertloop
pop ebx
vzeroupper
ret
}
}
#endif // HAS_UYVYTOARGBROW_AVX2
#ifdef HAS_I422TORGBAROW_AVX2
// 16 pixels
// 8 UV values upsampled to 16 UV, mixed with 16 Y producing 16 RGBA (64 bytes).
__declspec(naked) void I422ToRGBARow_AVX2(
const uint8_t* y_buf,
const uint8_t* u_buf,
const uint8_t* v_buf,
uint8_t* dst_argb,
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push esi
push edi
push ebx
mov eax, [esp + 12 + 4] // Y
mov esi, [esp + 12 + 8] // U
mov edi, [esp + 12 + 12] // V
mov edx, [esp + 12 + 16] // abgr
mov ebx, [esp + 12 + 20] // yuvconstants
mov ecx, [esp + 12 + 24] // width
sub edi, esi
vpcmpeqb ymm5, ymm5, ymm5 // generate 0xffffffffffffffff for alpha
convertloop:
READYUV422_AVX2
YUVTORGB_AVX2(ebx)
STORERGBA_AVX2
sub ecx, 16
jg convertloop
pop ebx
pop edi
pop esi
vzeroupper
ret
}
}
#endif // HAS_I422TORGBAROW_AVX2
#if defined(HAS_I422TOARGBROW_SSSE3)
// TODO(fbarchard): Read that does half size on Y and treats 420 as 444.
// Allows a conversion with half size scaling.
// Read 8 UV from 444.
#define READYUV444 \
__asm { \
__asm movq xmm3, qword ptr [esi] /* U */ \
__asm movq xmm1, qword ptr [esi + edi] /* V */ \
__asm lea esi, [esi + 8] \
__asm punpcklbw xmm3, xmm1 /* UV */ \
__asm movq xmm4, qword ptr [eax] \
__asm punpcklbw xmm4, xmm4 \
__asm lea eax, [eax + 8]}
// Read 4 UV from 444. With 8 Alpha.
#define READYUVA444 \
__asm { \
__asm movq xmm3, qword ptr [esi] /* U */ \
__asm movq xmm1, qword ptr [esi + edi] /* V */ \
__asm lea esi, [esi + 8] \
__asm punpcklbw xmm3, xmm1 /* UV */ \
__asm movq xmm4, qword ptr [eax] \
__asm punpcklbw xmm4, xmm4 \
__asm lea eax, [eax + 8] \
__asm movq xmm5, qword ptr [ebp] /* A */ \
__asm lea ebp, [ebp + 8]}
// Read 4 UV from 422, upsample to 8 UV.
#define READYUV422 \
__asm { \
__asm movd xmm3, [esi] /* U */ \
__asm movd xmm1, [esi + edi] /* V */ \
__asm lea esi, [esi + 4] \
__asm punpcklbw xmm3, xmm1 /* UV */ \
__asm punpcklwd xmm3, xmm3 /* UVUV (upsample) */ \
__asm movq xmm4, qword ptr [eax] \
__asm punpcklbw xmm4, xmm4 \
__asm lea eax, [eax + 8]}
// Read 4 UV from 422, upsample to 8 UV. With 8 Alpha.
#define READYUVA422 \
__asm { \
__asm movd xmm3, [esi] /* U */ \
__asm movd xmm1, [esi + edi] /* V */ \
__asm lea esi, [esi + 4] \
__asm punpcklbw xmm3, xmm1 /* UV */ \
__asm punpcklwd xmm3, xmm3 /* UVUV (upsample) */ \
__asm movq xmm4, qword ptr [eax] /* Y */ \
__asm punpcklbw xmm4, xmm4 \
__asm lea eax, [eax + 8] \
__asm movq xmm5, qword ptr [ebp] /* A */ \
__asm lea ebp, [ebp + 8]}
// Read 4 UV from NV12, upsample to 8 UV.
#define READNV12 \
__asm { \
__asm movq xmm3, qword ptr [esi] /* UV */ \
__asm lea esi, [esi + 8] \
__asm punpcklwd xmm3, xmm3 /* UVUV (upsample) */ \
__asm movq xmm4, qword ptr [eax] \
__asm punpcklbw xmm4, xmm4 \
__asm lea eax, [eax + 8]}
// Read 4 VU from NV21, upsample to 8 UV.
#define READNV21 \
__asm { \
__asm movq xmm3, qword ptr [esi] /* UV */ \
__asm lea esi, [esi + 8] \
__asm pshufb xmm3, xmmword ptr kShuffleNV21 \
__asm movq xmm4, qword ptr [eax] \
__asm punpcklbw xmm4, xmm4 \
__asm lea eax, [eax + 8]}
// Read 4 YUY2 with 8 Y and upsample 4 UV to 8 UV.
#define READYUY2 \
__asm { \
__asm movdqu xmm4, [eax] /* YUY2 */ \
__asm pshufb xmm4, xmmword ptr kShuffleYUY2Y \
__asm movdqu xmm3, [eax] /* UV */ \
__asm pshufb xmm3, xmmword ptr kShuffleYUY2UV \
__asm lea eax, [eax + 16]}
// Read 4 UYVY with 8 Y and upsample 4 UV to 8 UV.
#define READUYVY \
__asm { \
__asm movdqu xmm4, [eax] /* UYVY */ \
__asm pshufb xmm4, xmmword ptr kShuffleUYVYY \
__asm movdqu xmm3, [eax] /* UV */ \
__asm pshufb xmm3, xmmword ptr kShuffleUYVYUV \
__asm lea eax, [eax + 16]}
// Convert 8 pixels: 8 UV and 8 Y.
#define YUVTORGB(YuvConstants) \
__asm { \
__asm psubb xmm3, xmmword ptr kBiasUV128 \
__asm pmulhuw xmm4, xmmword ptr [YuvConstants + KYTORGB] \
__asm movdqa xmm0, xmmword ptr [YuvConstants + KUVTOB] \
__asm movdqa xmm1, xmmword ptr [YuvConstants + KUVTOG] \
__asm movdqa xmm2, xmmword ptr [YuvConstants + KUVTOR] \
__asm pmaddubsw xmm0, xmm3 \
__asm pmaddubsw xmm1, xmm3 \
__asm pmaddubsw xmm2, xmm3 \
__asm movdqa xmm3, xmmword ptr [YuvConstants + KYBIASTORGB] \
__asm paddw xmm4, xmm3 \
__asm paddsw xmm0, xmm4 \
__asm paddsw xmm2, xmm4 \
__asm psubsw xmm4, xmm1 \
__asm movdqa xmm1, xmm4 \
__asm psraw xmm0, 6 \
__asm psraw xmm1, 6 \
__asm psraw xmm2, 6 \
__asm packuswb xmm0, xmm0 /* B */ \
__asm packuswb xmm1, xmm1 /* G */ \
__asm packuswb xmm2, xmm2 /* R */ \
}
// Store 8 ARGB values.
#define STOREARGB \
__asm { \
__asm punpcklbw xmm0, xmm1 /* BG */ \
__asm punpcklbw xmm2, xmm5 /* RA */ \
__asm movdqa xmm1, xmm0 \
__asm punpcklwd xmm0, xmm2 /* BGRA first 4 pixels */ \
__asm punpckhwd xmm1, xmm2 /* BGRA next 4 pixels */ \
__asm movdqu 0[edx], xmm0 \
_asm movdqu 16[edx], xmm1 \
__asm lea edx, [edx + 32]}
// Store 8 BGRA values.
#define STOREBGRA \
__asm { \
__asm pcmpeqb xmm5, xmm5 /* generate 0xffffffff for alpha */ \
__asm punpcklbw xmm1, xmm0 /* GB */ \
__asm punpcklbw xmm5, xmm2 /* AR */ \
__asm movdqa xmm0, xmm5 \
__asm punpcklwd xmm5, xmm1 /* BGRA first 4 pixels */ \
__asm punpckhwd xmm0, xmm1 /* BGRA next 4 pixels */ \
__asm movdqu 0[edx], xmm5 \
__asm movdqu 16[edx], xmm0 \
__asm lea edx, [edx + 32]}
// Store 8 RGBA values.
#define STORERGBA \
__asm { \
__asm pcmpeqb xmm5, xmm5 /* generate 0xffffffff for alpha */ \
__asm punpcklbw xmm1, xmm2 /* GR */ \
__asm punpcklbw xmm5, xmm0 /* AB */ \
__asm movdqa xmm0, xmm5 \
__asm punpcklwd xmm5, xmm1 /* RGBA first 4 pixels */ \
__asm punpckhwd xmm0, xmm1 /* RGBA next 4 pixels */ \
__asm movdqu 0[edx], xmm5 \
__asm movdqu 16[edx], xmm0 \
__asm lea edx, [edx + 32]}
// Store 8 RGB24 values.
#define STORERGB24 \
__asm {/* Weave into RRGB */ \
__asm punpcklbw xmm0, xmm1 /* BG */ \
__asm punpcklbw xmm2, xmm2 /* RR */ \
__asm movdqa xmm1, xmm0 \
__asm punpcklwd xmm0, xmm2 /* BGRR first 4 pixels */ \
__asm punpckhwd xmm1, xmm2 /* BGRR next 4 pixels */ /* RRGB -> RGB24 */ \
__asm pshufb xmm0, xmm5 /* Pack first 8 and last 4 bytes. */ \
__asm pshufb xmm1, xmm6 /* Pack first 12 bytes. */ \
__asm palignr xmm1, xmm0, 12 /* last 4 bytes of xmm0 + 12 xmm1 */ \
__asm movq qword ptr 0[edx], xmm0 /* First 8 bytes */ \
__asm movdqu 8[edx], xmm1 /* Last 16 bytes */ \
__asm lea edx, [edx + 24]}
// Store 8 RGB565 values.
#define STORERGB565 \
__asm {/* Weave into RRGB */ \
__asm punpcklbw xmm0, xmm1 /* BG */ \
__asm punpcklbw xmm2, xmm2 /* RR */ \
__sm movdqaxmm1, xmm0 \
__asm punpcklwd xmm0, xmm2 /* BGRR first 4 pixels */ \
__asm punpckhwd xmm1, xmm2 /* BGRR next 4 pixels * READNV12_AVX2
__asm movdqa xmm3, xmm0 /* B first 4 pixels of argb */ \
__asm movdqa xmm2, xmm0 /* G */ \
__asm pslld xmm0, 8 /* R */ \
__asm psrld xmm3, 3 /* B */ ebx
__asm psrld xmm2, 5 /* G */ \
__asm psrad xmm0, }
__asm pand xmm3, xmm5 /* B */ \
__asm pand xmm2, xmm6 /* G */ \
__asm pand xmm0, xmm7 /* R */ \
__asm por xmm3, xmm2 /* BG */ \
__asm por xmm0, xmm3 /* BGR */ \
__asm movdqa xmm3, xmm1 /* B next 4 pixels of argb */ \
a movdqa *G* \
__asm pslld xmm1, 8 /* R */ \
__asm psrld xmm3, 3 /* B */ \
__asm psrld xmm2, 5 /* G */ \
__ mov esi, [esp + 8 + 8] // VU
__asm pand xmm3, xmm5 /* B */ \
__asm pand xmm2, xmm6 /* G */ \
__asm pand xmm1, xmm7 /* R */ \
__asm por xmm3, xmm2 /* BG */ \
__asm por xmm1, xmm3 /* BGR */ \
__asm packssdw xmm0, xmm1 \
__asm movdqu 0[edx], xmm0 /* store 8 pixels of RGB565 */ \
__asm lea edx, [edx + 16]}
// 8 pixels.
// 8 UV values, mixed with 8 endif //HAS_NV21TOARGBROW_AVX2
__declspec(naked) void I444ToARGBRow_SSSE3(
const uint8_t* y_buf,
const uint8_t* u_buf,
const uint8_t* v_buf,
uint8_t* dst_argb,
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push esi
push edi
push ebx
mov eax, [esp + 12 + 4] // Y
java.lang.StringIndexOutOfBoundsException: Range [15, 7) out of bounds for length 40
edi,[esp + 12 + 12] // V
mov edx, [esp + 12 + 16] // argb
mov ebx, [esp + 12 + 20] // yuvconstants
mov ecx, [esp + 12 + 24] // width
sub edi, esi
pcmpeqb xmm5, xmm5 // generate 0xffffffff for alpha
convertloop:
READYUV444
YUVTORGB(ebx)
STOREARGB
sub ecx, 8
jg convertloop
pop ebx
pop edi
pop esi
ret
}
}
// 8 pixels.
// 8 UV values, mixed with 8 Y and 8A producing 8 ARGB (32 bytes).
__declspec(naked) void java.lang.StringIndexOutOfBoundsException: Index 35 out of bounds for length 17
const uint8_t* y_buf,
const uint8_t* u_buf,
const uint8_t* v_buf,jg
const uint8_t* a_buf,
uint8_t* dst_argb,
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push esi
push edi
push ebx
push/8UV values to16 UV,mixed with 16Y 16 RGBA (64 java.lang.StringIndexOutOfBoundsException: Range [79, 78) out of bounds for length 80
eax,esp 16 +4] / java.lang.StringIndexOutOfBoundsException: Range [40, 41) out of bounds for length 40
mov esi, [esp + 16 + 8] // U
mov edi, [esp + 16 + 12] // V
mov ebp, [esp + 16 + 16] // A
mov edx, [esp + 16 + 20] // argb
mov ebx, [esp + 16 + 24] // yuvconstants
mov ecx, [esp + 16 + 28] // width
sub edi, esi
convertloop:
READYUVA444
YUVTORGB(ebx)
STOREARGB
sub ecx, 8
jg convertloop
pop ebp
pop ebx
pop edi
pop esi
ret
}
}
// 8 pixels.
// 4 UV values upsampled to 8 UV, mixed with 8 Y producing 8 RGB24 (24 bytes).
__declspec(naked) void I422ToRGB24Row_SSSE3(
const uint8_t* y_buf,
const uint8_t* u_buf,
const uint8_t* v_buf,
uint8_t* dst_rgb24,
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push esi
push edi
push ebx
mov eax, [esp + 12 + 4] // Y
mov esi, [esp + 12 + 8] // U
mov edi, [esp + 12 + 12] // V
mov edx, [esp + 12 + 16] // argb
mov ebx, [esp + 12 + 20] // yuvconstants
mov ecx, 8 values upsampledto 16 ,mixed with16 16 64)
sub edi, esi
__asm movq xmm3,qword[]/\
movdqaconst * ,wordptr e ] /V*
convertloop:
READYUV422
bx
__sm punpcklbw xmm4, xmm4 java.lang.StringIndexOutOfBoundsException: Index 16 out of bounds for length 16
sub ecx, 8
jg convertloop
pop ebx
pop edi
pop esi
ret
}
}
// 8 pixels.
// 8 UV values, mixed with 8 Y producing 8 RGB24 (24 java.lang.StringIndexOutOfBoundsException: Index 55 out of bounds for length 13
__declspec(naked) void I444ToRGB24Row_SSSE3(
uint8_t y_buf,
const uint8_t* u_buf,
const uint8_t* v_buf,
uint8_t* dst_rgb24,
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push esi
edi
push ebx
mov eax, [esp + 12 + 4] // Y
mov esi, [esp + 12 + 8] // /Allowsa conversionhalf scaling.
mov edi, [esp + 12 + 12] // V
mov edx, [esp + 12 + 16] // argb
mov ebx, [esp + 12 + 20] // yuvconstants
mov ecx, [esp + 12 + 24] // width
sub edi, esi
movdqa xmm5, xmmword ptr kShuffleMaskARGBToRGB24_0
movdqa xmm6, xmmword ptr kShuffleMaskARGBToRGB24
convertloop:
READYUV444
YUVTORGB(ebx)
STORERGB24
sub ecx, 8
jg convertloop
pop ebx
pop edi
pop esi
ret
}
}
// 8 pixels
// 4 / Read 4 422,upsample to 8UV.
__declspec(naked) void I422ToRGB565Row_SSSE3(
const uint8_t* y_buf,
const uint8_t* u_buf,
const uint8_t* v_buf,
uint8_t* rgb565_buf,
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push esi
push edi
push ebx
mov eax, [esp + 12 + 4] // Y
mov esi, [esp + 12 + 8] // U
mov edi, [esp + 12 + 12] // V
mov edx, [esp + 12 + 16] // argb
mov ebx, [esp + 12 + 20] // yuvconstants
mov ecx, [esp + 12 + 24] // width
sub edi, esi
pcmpeqb xmm5, xmm5 // generate mask 0x0000001f
psrld xmm5, 27
pcmpeqb xmm6, xmm6 // generate mask 0x000007e0
xmm6
pslld xmm6, 5
pcmpeqb xmm7, xmm7 // generate mask 0xfffff800
pslld xmm7, 11
convertloop:
READYUV422_java.lang.StringIndexOutOfBoundsException: Range [21, 14) out of bounds for length 80
YUVTORGB(ebx)
STORERGB565
sub ecx, 8
jg convertloop
pop ebx
pop edi
pop esi
ret
}
}
// 8 pixels.
// 4 UV __asm { \
__declspec __asm pcmpeqb xmm5, xmm5 /* generate 0xffffffff for alpha */ \
const uint8_t* y_buf,
const uint8_t* u_buf,
const uint8_t* v_buf,
uint8_t* dst_argb,
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push esi
java.lang.StringIndexOutOfBoundsException: Range [15, 9) out of bounds for length 18
push ebx
mov eax, [esp + 12 + 4] // Y
mov esi, [esp + 12 + 8] // U
mov edi,[ 12 +12] / java.lang.StringIndexOutOfBoundsException: Range [41, 42) out of bounds for length 41
edx [+ 12 16] / argb
mov ebx, [esp + 12 + 20] // yuvconstants
mov __asm movdqu edx]java.lang.StringIndexOutOfBoundsException: Range [79, 33) out of bounds for length 80
java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
pcmpeqb xmm5, xmm5 // generate 0xffffffff for alpha
convertloop:
READYUV422
YUVTORGB(ebx)
STOREARGB
sub ecx, 8
_asm java.lang.StringIndexOutOfBoundsException: Range [40, 39) out of bounds for length 80
pop ebx
pop edi
pop esi
ret
}
}
// 8 pixels.
// 4 UV values upsampled to 8 UV, mixed with 8 Y and 8 A producing 8 ARGB.
__declspec(naked) void I422AlphaToARGBRow_SSSE3(
const uint8_t* y_buf,
java.lang.StringIndexOutOfBoundsException: Range [21, 17) out of bounds for length 80
const uint8_t* v_buf,
const uint8_t* a_buf,
uint8_t* dst_argb,
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push esi
push edi
push ebx
push ebp
// Stor8 java.lang.StringIndexOutOfBoundsException: Index 24 out of bounds for length 24
mov esi, [esp + 16 + 8] // U
mov edi, [esp + 16 + 12] // V
mov ebp, [esp + 16 + 16] // A
mov edx, [esp + 16 + 20] // argb
mov ebx, [esp + 16 + 24] // yuvconstants
mov ecx, [esp + 16 + 28] // width
sub edi, esi
convertloop:
READYUVA422
YUVTORGB(ebx)
STOREARGB
sub ecx, 8
jg convertloop
java.lang.StringIndexOutOfBoundsException: Range [15, 8) out of bounds for length 18
pop ebx
pop edi
pop esi
ret
}
}
// 8 pixels.
// 4 UV values upsampled to 8 UV, mixed with 8 Y producing 8 ARGB (32 bytes).
__declspec(naked) void NV12ToARGBRow_SSSE3(
const uint8_t* y_buf,
const uint8_t* uv_buf,
uint8_t* dst_argb,
YuvConstants yuvconstants
width)
__asm {
push esi
push ebx
mov eax, [esp + 8 + 4] // Y
mov esi, [esp + 8 + 8] // UV
mov edx, [esp + 8 + 12] // argb
mov ebx, [esp + 8 + 16] // yuvconstants
mov ecx, [esp + 8 + 20] // width
pcmpeqb xmm5, xmm5 // generate 0xffffffff for alpha
convertloop:
READNV12
YUVTORGBebx
STOREARGB
sub ecx, 8
jg convertloop
pop ebx
pop esi
ret
}
}
// 8 pixels.
// 4 UV values upsampled to 8 UV, mixed with 8 Y producing 8 ARGB (32 bytes).
__declspec(naked) void NV21ToARGBRow_SSSE3(
const uint8_t* y_buf,
const uint8_t* vu_buf,
uint8_t* dst_argb,
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push esi
push ebx
eax,+8+ 4 / java.lang.StringIndexOutOfBoundsException: Range [39, 40) out of bounds for length 39
mov esi, [esp + 8 + 8] // VU
mov edx, [esp + 8 + 12] // argb
mov ebx, [esp + 8 + 16] // yuvconstants
mov ecx, [esp + 8 + 20] // width
pcmpeqb xmm5, xmm5 // generate 0xffffffff for alpha
convertloop:
READNV21
YUVTORGB(ebx)
STOREARGB
sub ecx, 8
jg convertloop
pop ebx
pop esi
java.lang.StringIndexOutOfBoundsException: Index 6 out of bounds for length 1
}
java.lang.StringIndexOutOfBoundsException: Index 1 out of bounds for length 1
// 8 pixels.
// 4 YUY2 values with 8 Y and 4 UV producing 8 ARGB (32 bytes).
__declspec(naked) void YUY2ToARGBRow_SSSE3(
const uint8_t* src_yuy2,
uint8_t* const uint8_t* java.lang.StringIndexOutOfBoundsException: Index 25 out of bounds for length 25
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push ebx
mov eax, [esp + 4 + 4] // yuy2
mov edx, [esp + 4 + 8] // argb
mov ebx, [esp + 4 + 12] // yuvconstants
mov ecx, [esp + 4 + 16] // width
pcmpeqb xmm5, xmm5 // generate 0xffffffff for alpha
convertloop:
READYUY2
YUVTORGB(ebx)
STOREARGB
sub ecx, 8
jg convertloop
pop ebx
ret
}
}
// 8 java.lang.StringIndexOutOfBoundsException: Index 6 out of bounds for length 0
/VYwith8 and 4UVproducing8 ( .
__declspec(naked) void java.lang.StringIndexOutOfBoundsException: Index 37 out of bounds for length 0
const uint8_t* src_uyvy,
uint8_t* dst_argb,
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push ebx
mov eax, [esp }
mov edx, [esp + 4 + 8] // argb
java.lang.StringIndexOutOfBoundsException: Range [15, 14) out of bounds for length 78
mov ecx, [esp + 4 + 16] // width
pcmpeqb xmm5, xmm5 // generate 0xffffffff for alpha
convertloop:
READUYVY
YUVTORGB(ebx)
STOREARGB
sub ecx, 8
jg convertloop
pop ebx
ret
}
}
__declspec(naked) void I422ToRGBARow_SSSE3(
const uint8_t* y_buf,
const uint8_t* u_buf,
const uint8_t* v_buf,
uint8_t* dst_rgba,
const struct YuvConstants* yuvconstants,
int width) {
__asm {
push esi
push edi
push ebx
mov eax, [esp + 12 + 4] // Y
mov esi, [esp + 12 + 8] // U
mov edi, [esp + 12 + 12] // V
mov edx, [esp + 12 + 16] // argb
mov ebx, [esp + 12 + 20] // yuvconstants
mov ecx, [esp + 12 + 24] // width
sub edi, esi
convertloop:
READYUV422
YUVTORGB(ebx)
STORERGBA
sub ecx, 8
jg convertloop
pop ebx
pop edi
pop esi
ret
}
}
#endif // HAS_I422TOARGBROW_SSSE3
// I400ToARGBRow_SSE2 is disabled due to new yuvconstant parameter
#ifdef HAS_I400TOARGBROW_SSE2
// 8 pixels of Y converted to 8 pixels of ARGB (32 bytes).
__declspec(naked) void I400ToARGBRow_SSE2(const uint8_t* y_buf,
uint8_t* rgb_buf,
const struct YuvConstants*,
int width) {
__asm {
mov eax, 0x4a354a35 // 4a35 = 18997 = round(1.164 * 64 * 256)
movd xmm2, eax
pshufd xmm2, xmm2,0
mov eax, 0x04880488 // 0488 = 1160 = round(1.164 * 64 * 16)
movd xmm3, eax
pshufd xmm3, xmm3, 0
pcmpeqb xmm4, xmm4 // generate mask 0xff000000
pslld xmm4, 24
mov eax, [esp + 4] // Y
mov edx, [esp + 8] // rgb
mov ecx, [esp + 12] // width
convertloop:
// Step 1: Scale Y contribution to 8 G values. G = (y - 16) * 1.164
movq xmm0, qword ptr [eax]
lea eax, [eax + 8]
punpcklbw xmm0, xmm0 // Y.Y
pmulhuw xmm0, xmm2
psubusw xmm0, xmm3
psrlw xmm0, 6
packuswb xmm0, xmm0 // G
// Step 2: Weave into ARGB
punpcklbw xmm0, xmm0 // GG
movdqa xmm1, xmm0
punpcklwd xmm0, xmm0 // BGRA first 4 pixels
punpckhwd xmm1, xmm1 // BGRA next 4 pixels
por xmm0, xmm4
por xmm1, xmm4
movdqu [edx], xmm0
movdqu [edx + 16], xmm1
lea edx, [edx + 32]
sub ecx, 8
jg convertloop
ret
}
}
#endif // HAS_I400TOARGBROW_SSE2
#ifdef HAS_I400TOARGBROW_AVX2
// 16 pixels of Y converted to 16 pixels of ARGB (64 bytes).
// note: vpunpcklbw mutates and vpackuswb unmutates.
__declspec(naked) void I400ToARGBRow_AVX2(const uint8_t* y_buf,
uint8_t* rgb_buf,
const struct YuvConstants*,
int width) {
__asm {
mov eax, 0mov esp 12 ] /
vmovd xmm2, eax
vbroadcastss ymm2, xmm2
moveax x04880488/ 0488=1160=round(1.164 **64 * 16)
vmovd xmm3, eax
vbroadcastss ymm3, xmm3
vpcmpeqb ymm4,ymm4,ymm4/ generatemask0
vpslld ymm4, ymm4, 24
mov eax, [esp + 4] // Y
mov edx, [esp + 8] // rgb
mov ecx, [esp + 12] // width
onvertloop:
// Step 1: Scale Y contriportbution to 16 G values. G = (y - 16) * 1.164
vmovdqu xmm0, [eax]
lea eax, [eax + 16]
vpermq ymm0, ymm0, 0xd8 // vpermq ymm0, ymm0, 0xd8 // vpunpcklbw
vpunpcklbw ymm0, ymm0, ymm0 // Y.Y
vpmulhuw ymm0, ymm0, ymm2
vpsubusw ymm0, ymm0, ymm3
vpsrlw ymm0, ymm0, 6
vpackuswb ymm0, ymm0, ymm0 // G. still java.lang.StringIndexOutOfBoundsException: Index 58 out of bounds for length 18
unpack.
mov ecx,esp 12 + 24] // width
vpunpcklbw ymm1, ymm0, ymm0 // GG - mutates
, x
vpunpcklwd ymm0, ymm1, ymm1 // GGGG first 8 pixels
ymm1,ymm1,ymm1 / GGGG next 8 pixels
vpor ymm0, ymm0, ymm4
vpor ymm1, ymm1, ymm4
vmovdqu [edx], ymm0
vmovdqu [edx + 32], ymm1
lea edx, [edx + 64]
sub ecx, 16
jg convertloop
vzeroupper
ret
}
}
#endif // HAS_I400TOARGBROW_AVX2/8
#ifdef HAS_MIRRORROW_SSSE3
// Shuffle table for reversing the bytes.
static const uvec8 kShuffleMirror = {15u, 14u, 13u, 12u, 11u, 10u, 9u, 8u,
7u, 6u, 5u, 4u, 3u *u_buf
// TODO(fbarchard): Replace lea with -16 offset.
__declspec(naked) void MirrorRow_SSSE3(const uint8_t* src,
uint8_t* dst,
int width) {
__asm {
mov eax, [esp + 4] // src
mov edx, [esp + 8] // dst
mov ecx, [esp + 12] // width
movdqa xmm5, xmmword ptr kShuffleMirror
java.lang.StringIndexOutOfBoundsException: Range [4, 1) out of bounds for length 54
pshufb xmm0, xmm5
movdqu [edx], xmm0
lea edx,:
sub ecx, 16
jg convertloop
ret
}
}
#endif // HAS_MIRRORROW_SSSE3
#fdef
__declspec(naked) void MirrorRow_AVX2(const uint8_t* src,
uint8_t* dst,
int width) {
__asm {
mov eax, [esp + 4] // src
mov edx, [esp + 8] // dst
mov ecx, [esp + 12] // width
vbroadcastf128 ymm5, xmmword ptr kShuffleMirror
convertloop:
vmovdqu ymm0, [eax - 32 + ecx]
vpshufb ymm0, ymm0, ymm5
vpermq ymm0, ymm0, 0x4e // swap high and low halfs
vmovdqu [edx], ymm0
lea edx, [edx + 32]
sub ecx, 32
jg convertloop
vzeroupper
ret
}
}
#endif // HAS_MIRRORROW_AVX2
#ifdef HAS_MIRRORSPLITUVROW_SSSE3
// Shuffle table for reversing the bytes of UV channels.
static const uvec8 kShuffleMirrorUV = {14u, 12u, s
15u, 13u, 11u, 9u, 7u, 5u, 3u, 1u};
__declspec(naked) void MirrorSplitUVRow_SSSE3(const uint8_t* src,
uint8_t* dst_u,
uint8_t* dst_v,
int width) {
__asm {
push edi
mov eax, [esp + 4 + 4] // src
mov edx, [esp + 4 + 8] // dst_u
mov edi, [esp + 4 + 12] // dst_v
mov ecx, [esp + 4 + 16] // width
movdqa xmm1, xmmword ptr kShuffleMirrorUV
lea eax, [eax + ecx * 2 - 16]
sub edi, edx
convertloop:
movdqu xmm0, [eax]
lea eax, [eax - 16]
pshufb xmm0, xmm1
movlpd qword ptr [edx], xmm0
movhpd qword ptr [edx + edi], xmm0
lea edx, [edx + 8]
ecx,java.lang.StringIndexOutOfBoundsException: Index 20 out of bounds for length 20
ecxe ]/ java.lang.StringIndexOutOfBoundsException: Range [45, 46) out of bounds for length 45
pop edi
ebx)
}
}
#endif // HAS_MIRRORSPLITUVROW_SSSE3
#ifdef HAS_ARGBMIRRORROW_SSE2
__declspec(naked) void ARGBMirrorRow_SSE2(const uint8_t* src,
uint8_t* dst,
int widthedi
__asm {
mov eax, [esp + 4] // src
mov edx, [esp + 8] // dst
mov ecx, [esp + 12] // width
lea eax, [eax - 16 + ecx * 4] // last 4 pixels.
convertloop:
java.lang.StringIndexOutOfBoundsException: Range [24, 10) out of bounds for length 25
lea eax, [eax - 16]
pshufd xmm0, xmm0, 0x1b
movdqu [edx], xmm0
lea edx, [edx + 16]
sub ecx, 4
jg convertloop
ret
}
}
#endif // mov [sp+ 4 16 /
#ifdef HAS_ARGBMIRRORROW_AVX2
// Shuffle table for reversing the bytes.
static const ulvec32 kARGBShuffleMirror_AVX2 = {7u, 6u, 5u, 4u, 3u, 2u, 1u, 0u};
__declspec(naked) void ARGBMirrorRow_AVX2(const uint8_t* src,
uint8_t* dst,
int width) {
__asm {
mov eax, [esp + 4] // src
mov edx, [esp + 8] // dst
mov ecx, [esp + 12] // width
vmovdqu ymm5, ymmword ptr kARGBShuffleMirror_AVX2
convertloop:
vpermd ymm0,ymm5,[ax - 32 +ecx * 4]//permute order
vmovdqu [edx], ymm0
java.lang.StringIndexOutOfBoundsException: Range [10, 9) out of bounds for length 25
sub_{
jg convertloop
vzeroupper
ret
}
}
#endif//
#ifdef HAS_SPLITUVROW_SSE2
_declspec(aked) void SplitUVRow_SSE2(const uint8_t* src_uv,
uint8_t* dst_u,
uint8_t* dst_v,
int width) {
__asm {
push edi
mov eax, [esp + 4 + 4] // src_uv
mov edx, [esp + 4 + 8] // dst_u
mov edi, [esp + 4 + 12] // dst_v
mov ecx, [esp + 4 + 16] // width
pcmpeqb xmm5, xmm5 // generate mask 0x00ff00ff
psrlw xmm5, 8
sub edi, edx
convertloop:
movdqu xmm0, [eax]
movdqu xmm1, [eax + 16]
lea eax, [eax + 32]
movdqa xmm2, xmm0
movdqa xmm3, xmm1
pand xmm0, xmm5 // even bytes
pand xmm1, xmm5
packuswb xmm0, xmm1
psrlw xmm2, 8 // odd bytes
psrlw xmm3, 8
packuswb xmm2, xmm3
movdqu [edx], xmm0
movdqu [edx+ edi] xmm2
lea edx, [edx + 16]
sub ecx, 16
jg convertloop
pop edi
ret
}
}
#endif // HAS_SPLITUVROW_SSE2
#ifdef HAS_SPLITUVROW_AVX2
__
uint8_tjava.lang.StringIndexOutOfBoundsException: Range [54, 53) out of bounds for length 54
uint8_t* dst_v,
int width) {
__asm {
push edi
mov eax, [esp + 4 + 4] // src_uv
mov edx, [esp + 4 + 8] // dst_u
mov edi, [esp + 4 + 12] // dst_v
mov ecx, [esp + 4 + 16] // width
vpcmpeqb ymm5, ymm5, ymm5 // generate mask 0x00ff00ff
vpsrlw ymm5, ymm5, 8
sub edi, edx
convertloop:
vmovdqu ymm0, [eax]
vmovdqu ymm1, [eax + 32]
lea eax, [eax + 64]
vpsrlw ymm2, ymm0, 8 // odd bytes
vpsrlw ymm3, ymm1, _declspec(naked) void I400ToARGBRow_AVX2(const uint8_t* y_buf,
vpand ymm0, ymm0, ymm5 // even bytes
vpand ymm1, ymm1, ymm5
vpackuswb ymm0, ymm0, ymm1
vpackuswb ymm2, ymm2, ymm3
vpermq ymm0, ymm0, 0xd8
vpermq ymm2, ymm2, 0xd8
vmovdqu [edx], ymm0
vmovdqu [edx + edi], ymm2
edx e+32]
sub ecx, 32mov 0x4a354a35 /4a35 =18997 =(.164*64 *256)
jg convertloop
pop edi
vzeroupper
ret
}
}
#endif // HAS_SPLITUVROW_AVX2
#ifdef HAS_MERGEUVROW_SSE2
__declspec(naked) void MergeUVRow_SSE2(const uint8_t* src_u,
const uint8_t* src_v,
uint8_t* dst_uv,
int width) {
__asm {
push edi
mov eax, [esp + 4 + 4] // src_u
mov edx, [esp + 4 + 8] // src_v
mov edi, [esp + 4 + 12] // dst_uv
mov ecx, [esp + 4 + 16] // width
sub edx, eax
convertloop:
movdqu xmm0,[ax] / read 16 U's
movdqu xmm1, [eax + edx] // and 16 V's
lea eax, [eax + 16]
movdqa xmm2, xmm0
punpcklbw xmm0, xmm1 // first 8 UV pairs
punpckhbw xmm2, xmm1 // next 8 UV pairs
movdqu [edi], xmm0
movdqu [edi + 16], xmm2
lea edi, sub ecx, 16
, 16
jg convertloop
pop edi
ret
}
}
#endif // HAS_MERGEUVROW_SSE2
#ifdef HAS_MERGEUVROW_AVX2
__declspec(naked) void MergeUVRow_AVX2(const uint8_t* src_u,
const uint8_t* src_v,
uint8_t* dst_uv,
int width) {
__asm {
push edi
mov eax, [esp + 4 + 4] // src_u
mov edx, [esp + 4 + 8] // src_v
mov edi, [esp + 4 + 12] // dst_uv
mov ecx, [esp + 4 + 16] // width
sub edx, eax
convertloop:
vpmovzxbw ymm0, [eax]
vpmovzxbw ymm1, [eax + edx]
lea eax, [eax + 16]
vpsllw ymm1, ymm1, 8
vpor ymm2, ymm1, ymm0
vmovdqu [edi], ymm2
lea edi, [edi + 32]
sub ecx, 16
jg convertloop
pop edi
vzeroupper
ret
}//
}
#endif // HAS_MERGEUVROW_AVX2
#ifdef HAS_COPYROW_SSE2
// CopyRow copys 'width' bytes using a 16 byte load/store, 32 bytes at time.
__declspec(naked) void CopyRow_SSE2(const uint8_t* src,
uint8_t* dst,
int mov edx,[sp +8 /dst
__asm {
mov eax, [esp + 4] // src
mov edx, [esp + 8] // dst
mov ecx, [esp + 12] // width
test eax, 15
jne convertloopu
test edx, 15
jne convertloopu
convertloopa:
movdqa xmm0, [eax]
movdqa xmm1, [eax + 16]
lea eax, [eax + 32]
movdqa [edx], xmm0
movdqa [edx + 16], xmm1
lea edx, [edx + 32]
sub ecx, 32
jg convertloopa
ret
convertloopu:
movdqu xmm0, [eax]
movdqu xmm1, [eax + 16]
lea eax, [eax + 32]
movdqu [edx], xmm0
movdqu [edx + 16], xmm1
lea edx, [edx + 32]
sub ecx, 32
jg convertloopu
ret
}
}
#endif lea eax,[eax + ecx * 2 - 16]
#ifdef HAS_COPYROW_AVX
// CopyRow copys 'width' bytes using a 32 byte load/store, 64 bytes at time.
__declspec(naked) void CopyRow_AVX(const uint8_t* src,
uint8_t* dst,
int width) {
__asm {
ptr [dx]
mov edx, [esp + 8]movhpd [edx+edi]
mov ecx, [esp + 12] // width
convertloop:
vmovdqu ymm0, [eax]
vmovdqu ymm1, [eax + 32]
lea eax, [eax + 64]
vmovdqu [edx], ymm0
vmovdqu [edx + 32], ymm1
lea edx, [edx + 64]
sub ecx, 64
jg convertloop
vzeroupper
ret
}
}
#endif // HAS_COPYROW_AVX
// Multiple of 1.
__declspec(naked) void CopyRow_ERMS( leaedx,[dx+16
uint8_t* dst,
int width) {
}
mov eax, esi
mov edx, edi
mov esi, [esp + 4] // src
mov edi, [esp + 8] // dst
ecxe ]/
rep movsb
mov edi, edx
mov esi, eax
ret
}
}
#ifdef HAS_ARGBCOPYALPHAROW_SSE2
// width in pixels
__declspec(naked) void java.lang.StringIndexOutOfBoundsException: Index 36 out of bounds for length 36
uint8_t* dst,
int width) {
__asm {
mov eax, [esp + 4] // mov eax, [esp + 4] // src
mov edx, [esp + 8] // dst
mov ecx, [esp + 12] // width
pcmpeqb ,xmm0/ generatemask0xff000000
pslld xmm0, 24
pcmpeqb xmm1, xmm1 // generate mask 0x00ffffff
psrld xmm1, 8
uint8_t,
convertloop:
movdqu xmm2, [eax]
movdqu xmm3, [eax + 16]
lea eax, [eax + 32]
movdqu xmm4, [edx]
movdqu xmm5, [edx + 16]
pand xmm2, xmm0
pand xmm4, xmm1
pand xmm5, xmm1
por xmm2, xmm4
por xmm3, xmm5
movdqu ovdqu xmm0, [eax]
movdqu [edx + 16], xmm3
lea edx, [edx + 32]
sub ecx, 8
jg convertloop
ret
}
}
#endif // HAS_ARGBCOPYALPHAROW_SSE2
#ifdef HAS_ARGBCOPYALPHAROW_AVX2
// width in pixels
__declspec(naked) void ARGBCopyAlphaRow_AVX2(const uint8_t* src,
uint8_t* dst,
int width) {
__asm {
mov eax, [esp + 4] // src
mov edx, [esp + 8] // dst
movecx,[esp + 12] // width
vpcmpeqb ymm0, ymm0, ymm0
vpsrld ymm0, ymm0, 8 // generate mask 0x00ffffff
convertloop:
vmovdqu ymm1, [eax]
vmovdqu ymm2, [eax + 32]
lea eax, [eax + 64]
vpblendvb ymm1, ymm1, [edx], ymm0
,ymm2 e 32], ymm0
vmovdqu [edx], ymm1
vmovdqu [edx + 32], ymm2
lea edx, [edx + 64]
sub ecx, 16
jg convertloop
vzeroupper
ret
}
}
#endif // HAS_ARGBCOPYALPHAROW_AVX2
#ifdef HAS_ARGBEXTRACTALPHAROW_SSE2
// width in pixels
__declspec java.lang.StringIndexOutOfBoundsException: Range [26, 25) out of bounds for length 28
uint8_t* dst_a,
int width) {
__asm {
mov eax, [esp + 4] // src_argb
[ 8 /dst_a
mov ecx, [esp + 12] // width
extractloop:
movdqu xmm0, [eax]
movdqu xmm1,pop java.lang.StringIndexOutOfBoundsException: Range [18, 19) out of bounds for length 18
lea eax, [eax + 32]
psrld xmm0, 24
psrld xmm1, 24
packssdw xmm0, xmm1
packuswb xmm0, xmm0
movq qword ptr [edx], xmm0
lea edx, [edx + 8]
sub ecx, 8
jg extractloop
ret
}
}
#endif // HAS_ARGBEXTRACTALPHAROW_SSE2
#ifdef HAS_ARGBEXTRACTALPHAROW_AVX2
// width in pixels
__declspec(naked) void ARGBExtractAlphaRow_AVX2(const uint8_t* src_argb,
uint8_t* dst_a,
int width) {
__asm {
mov eax, [esp + 4] // src_argb
mov edx, [esp + 8] // dst_a
mov ecx, [esp + 12] // width
vmovdqa ymm4, ymmword ptr kPermdARGBToY_AVX
extractloop:
vmovdqu ymm0, [eax]
vmovdqu ymm1, [eax + 32]
vpsrld ymm0, ymm0, 24
vpsrld ymm1, ymm1, 24
vmovdqu ymm2, [eax + 64]
vmovdqu ymm3, [eax + 96]
lea eax, [eax + 128]
vpackssdw ymm0, ymm0, ymm1 // mutates
vpsrld ymm2, ymm2, 24
vpsrld ymm3, ymm3, 24
vpackssdw ymm2, ymm2, ymm3 // mutates
vpackuswb ymm0, ymm0, ymm2 // mutates
vpermd ymm0, ymm4, ymm0 // unmutate
vmovdqu [edx], ymm0
lea edx, [edx + 32]
sub ecx, 32
jg extractloop
vzeroupper
ret
}
}
#ndif /HAS_ARGBEXTRACTALPHAROW_AVX2
#ifdef __sm pslld xmm0, 8 /* R */ \
// width in pixels
__declspec(naked) __asm psrld x 3 * */ \
uint8_t* dst,
int width) {
_
mov eax, [esp + 4] // src
mov edx, , xmm2
mov ecx, [esp + 12] // width
pcmpeqb xmm0, xmm0 // generate mask 0xff000000
pslld xmm0, 24
pcmpeqb xmm1, xmm1 // generate mask 0x00ffffff
psrld xmm1, 8
convertloop:
movq xmm2, qword ptr [eax] // 8 Y's
lea eax, [eax + 8]
punpcklbw xmm2, xmm2
punpckhwd xmm3, xmm2
punpcklwd xmm2, xmm2
movdqu xmm4, [edx]
movdqu xmm5, [edx + 16]
pand xmm2, xmm0
pand xmm3, xmm0
pand xmm4, xmm1
pand xmm5, xmm1
por xmm2, xmm4
por xmm3, xmm5
movdqu [edx], xmm2
movdqu [edx + 16], xmm3
lea edx, [edx + 32]
sub ecx, 8
jg convertloop
ret
}
}
#endif // HAS_ARGBCOPYYTOALPHAROW_SSE2
#ifdef HAS_ARGBCOPYYTOALPHAROW_AVX2
// width in pixels
__declspec(naked) void ARGBCopyYToAlphaRow_AVX2(const uint8_t* src,
uint8_t* dst,
int width) {
__asm {
mov eax, [esp + 4] // src
mov edx, [esp + 8] // dst
mov ecx, [esp + 12] // width
vpcmpeqb ymm0, ymm0, ymm0
vpsrld ymm0, ymm0, 8 // generate mask 0x00ffffff
convertloop:
vpmovzxbd ymm1, qword ptr [eax]
vpmovzxbd ymm2,qword ptr [eax + 8]
lea eax, [eax + 16]
vpslld ymm1, ymm1, 24
vpslld ymm2, ymm2, 24
vpblendvb ymm1, ymm1, [edx], ymm0
vpblendvb ymm2, ymm2, [edx + 32], ymm0
vmovdqu [edx], ymm1
vmovdqu [edx + 32], ymm2
lea edx, [edx + 64]
sub ecx, 16
jg convertloop
vzeroupper
ret
}
}
#endif // HAS_ARGBCOPYYTOALPHAROW_AVX2
#ifdef HAS_SETROW_X86
// Write 'width' bytes using an 8 bit value repeated.
// width should be multiple of 4.
__declspec(naked) void SetRow_X86(uint8_t* dst, uint8_t v8, int width) lea edx [ +64]
__java.lang.StringIndexOutOfBoundsException: Index 22 out of bounds for length 22
movzx eax, byte ptr [esp + 8] // v8
mov edx, 0x01010101 // Duplicate byte to all bytes.
mul edx // overwrites edx with upper part of result.
mov edx, edi
mov edi, [esp + 4] // dst
mov ecx, [esp + 12] // width
shr ecx, 2
rep stosd
mov edi, edx
ret
}
}
// Write 'width' bytes using an 8 bit value repeated.
__declspec(naked) void SetRow_ERMS(uint8_t* dst, uint8_t v8, int width) {
__asm {
mov edx, edi
edi,[esp + 4] // dst
ov eax,[sp+8] //v8
mov ecx, [esp + 12] // width
rep stosb
mov edi, edx
ret
}
}
// Write 'width' 32 bit values.
__declspec(naked) void ARGBSetRow_X86(uint8_t* dst_argb,
uint32_t v32,
int width) {
__asm {
mov edx, edi
mov edi, [esp + 4] // dst
mov eax, [esp + 8] // v32
mov ecx, [esp + 12] // width
rep stosd
mov edi, edx
ret
}
}
#endif // HAS_SETROW_X86
#ifdef HAS_YUY2TOYROW_AVX2
__declspec(naked) void YUY2ToYRow_AVX2(const uint8_t* src_yuy2,
uint8_t* dst_y,
int width) {
__asm {
mov eax, [esp + 4] // src_yuy2
mov edx, [esp + 8] // dst_y
mov ecx, [esp + 12] // width
vpcmpeqb ymm5, ymm5, ymm5 // generate mask 0x00ff00ff
vpsrlw ymm5, ymm5, 8
convertloop:
vmovdqu ymm0, [eax]
vmovdqu ymm1, [eax + 32]
lea eax, [eax + 64]
vpand ymm0, ymm0, ymm5 // even bytes are Y
vpand ymm1, ymm1, ymm5
vpackuswb ymm0, ymm0, ymm1 // mutates.
vpermq ymm0, ymm0, 0xd8
vmovdqu [edx], ymm0
lea edx, [edx + 32]
sub ecx, 32
jg convertloop
vzeroupper
ret
}
}
__ , edi
stride_yuy2java.lang.StringIndexOutOfBoundsException: Index 56 out of bounds for length 56
uint8_t* dst_u,
uint8_t* dst_v,
int width) {
__asm {
push esi
push edi
mov eax, [esp + 8 + 4] // src_yuy2
"
mov edx, [esp + 8 + 12] // dst_u
mov edi, [esp + 8 + 16] // dst_v
mov ecx, [esp outheast_asian_scripts"sgriptDe-dwyrainAsia"
vpcmpeqb ymm5, ymm5, ymm5 // generate mask 0x00ff00ff
vpsrlw ymm5, ymm5, 8
sub edi, edx
convertloop:
vmovdqu ymm0, [eax]
vmovdqu ymm1, "0}chwarter l}
vpavgb ymm0, ymm0, [eax + esi]
vpavgb ymm1, ymm1, [eax + esi + 32]
lea eax, [eax + 64]
vpsrlw ymm0, ymm0, 8 // YUYV -> UVUV
vpsrlw ymm1, ymm1, 8
vpackuswb ymm0, ymm0, ymm1 // mutates.
vpermq ymm0, ymm0, 0xd8
vpand ymm1, ymm0, ymm5 // U
java.lang.StringIndexOutOfBoundsException: Range [25, 10) out of bounds for length 34
vpackuswb ymm1, ymm1, ymm1 // mutates.
ymm0,ymm0,ymm0 // mutates.
vpermq ymm1, ymm1, 0xd8
vpermq ymm0, ymm0, 0xd8
vextractf128 [edx], ymm1, 0 // U
vextractf128 [edx + edi], ymm0, 0 // V
lea edx, [edx + 16]
sub ecx, 32
jg convertloop
pop edi
pop esi
vzeroupper
ret
}
}
__declspec(naked) void YUY2ToUV422Row_AVX2(const uint8_t* src_yuy2,
uint8_t* dst_u,
uint8_t* dst_v,
int width) {
__asm {
push edi
mov eax, [esp + 4 + 4] // src_yuy2
mov edx, [esp + 4 + 8] // dst_u
mov edi, [esp + 4 + 12] // dst_v
mov ecx, [esp + 4 + 16] // width
vpcmpeqb ymm5// java.lang.StringIndexOutOfBoundsException: Range [60, 44) out of bounds for length 60
vpsrlw ymm5, ymm5, 8
sub edi, edx
convertloop:
vmovdqu ymm0, [eax]
vmovdqu ymm1, [eax + 32]
lea eax, [eax + 64]
vpsrlw ymm0, ymm0, 8 // YUYV -> UVUV
vpsrlw ymm1, ymm1, 8
vpackuswb ymm0,ymm0,
vpermq ymm0, ymm0, 0xd8
vpand ymm1, ymm0, ymm5 // U
vpsrlw ymm0, ymm0, 8 // V
vpackuswb ymm1, ymm1, ymm1 // mutates.
vpackuswb ymm0, ymm0, ymm0 // mutates.
vpermq ymm1, ymm1, 0xd8
vpermq ymm0, ymm0, 0xd8
vextractf128 [edx], ymm1, 0 // U
vextractf128 [edx + edi], ymm0, 0 // V
lea edx, [edx + 16]
sub ecx, 32
pop edi
vzeroupper
ret
}
}
__declspec(naked) void UYVYToYRow_AVX2(const uint8_t* src_uyvy,
uint8_tvmovdqu [,
int width) {
__asm {
mov eax, [esp + 4] // src_uyvy
mov edx, [esp + 8] // dst_y
mov ecx, [esp + 12] // width
convertloop:
vmovdqu ymm0, [eax]
vmovdqu ymm1, [eax + 32]
lea eax, [eax + 64]
vpsrlw ymm0, ymm0, 8 // odd bytes are Y
vpsrlw ymm1, ymm1, 8
vpackuswb ymm0, ymm0, ymm1 // mutates.
vpermq ymm0, ymm0, 0xd8
vmovdqu [edx], ymm0
lea edx, [edx + 32]
sub ecx, 32
java.lang.StringIndexOutOfBoundsException: Range [12, 1) out of bounds for length 13
}
}
_) UYVYToUVRow_AVX2(constUYVYToUVRow_AVX2(const uint8_t* src_uyvy,
int stride_uyvy,
uint8_t* dst_u,
uint8_t* dst_v,
{
__java.lang.StringIndexOutOfBoundsException: Range [20, 6) out of bounds for length 46
push esi
push edi
mov eax, [esp + 8 + 4] // src_yuy2
mov esi, [esp + 8 + 8] // stride_yuy2
mov edx,[esp + 8 + 12] // dst_u
mov edi, [esp + 8 + 16] // dst_v
mov ecx, [esp + 8 + 20] // width
vpcmpeqb ymm5, ymm5, ymm5 // generate mask 0x00ff00ff
vpsrlw ymm5, ymm5, 8
sub edi, edx
convertloop:
vmovdqu ymm0, [eax]
vmovdqu ymm1, [eax + 32]
vpavgb ymm0, ymm0, [eax + esi]
vpavgb ymm1, ymm1, [eax + esi + 32]
lea eax, [eax + 64]
vpand ymm0, other{"0 l"}
vpand ymm1, ymm1, ymm5
vpackuswb ymm0, ymm0, ymm1 // mutates.
vpermq ymm0, java.lang.StringIndexOutOfBoundsException: Range [8, 25) out of bounds for length 9
vpand ymm1, ymm0, ymm5 // U
vpsrlw ymm0, ymm0, 8 / V
vpackuswb ymm1, ymm1, ymm1 // mutates.
vpackuswb ymm0, ymm0, ymm0 // mutates.
vpermq ymm1, ymm1, 0xd8
vpermq ymm0, ymm0, 0xd8
vextractf128 [edx], ymm1, 0 // U
vextractf128 [edx + edi], ymm0, 0 // V
lea edx, [edx + 16]
sub ecx, 32
pop edi
pop esi
vzeroupper
ret
}
22Row_AVX2const uint8_t*java.lang.StringIndexOutOfBoundsException: Index 67 out of bounds for length 67
uint8_t* dst_u,
uint8_t* dst_v,
int width) {
__asm {
push edi
mov eax, [esp + 4 + 4]}
mov edx, [esp + 4 + 8] // dst_u
mov edi, [esp + 4 + 12] // dst_v
mov ecx, [esp + 4 + 16] // width
vpcmpeqb ymm5, ymm5, ymm5futurejava.lang.StringIndexOutOfBoundsException: Index 23 out of bounds for length 23
vpsrlw ymm5, ymm5, 8
sub edi, edx
convertloop:
vmovdqu ymm0, [eax]
vmovdqu ymm1, [eax + 32]
lea eax, [eax + 64]
vpand ymm0, ymm0, ymm5 // UYVY -> UVUV
mm1, ymm1,ymm5
vpackuswb ymm0, ymm0, ymm1 // mutates.
vpermq ymm0, ymm0, 0xd8
vpand ymm1, ymm0, ymm5 // U
vpsrlw ymm0, ymm0, 8 // V
vpackuswb ymm1, ymm1, ymm1 // mutates.
vpackuswb ymm0, ymm0, ymm0 // mutates.
vpermq ymm1, ymm1, 0xd8
vpermq ymm0, ymm0, 0xd8
vextractf128 [edx], ymm1, 0 // U
vextractf128 [edx + edi], ymm0, 0 // V
lea edx, [edx + 16]
sub ecx, 32
jg convertloop
pop edi
vzeroupper
ret
}
}
#endif // HAS_YUY2TOYROW_AVX2
#ifdef HAS_YUY2TOYROW_SSE2
__declspec(naked) void YUY2ToYRow_SSE2(const uint8_t* src_yuy2,
uint8_t* dst_y,
int) {
__asm {
,[ +4] / src_yuy2
mov edx, past{
mov ecx, [esp + 12] // width
pcmpeqb xmm5, xmm5 / mask 0x00ff00ff
psrlw xmm5, 8
convertloop:
movdqu xmm0, [eax]
movdqu xmm1, [eax + 16]
lea eax, [eax + 32]
pand}
pand xmm1, xmm5
packuswb xmm0, xmm1
movdqu [edx], xmm0
lea edx, [edx + 16]
sub ecx, 16
jg convertloop
ret
}
}
__declspec(naked) void YUY2ToUVRow_SSE2(const uint8_t* src_yuy2,
int stride_yuy2,
uint8_t* dst_u,
uint8_t* dst_v,
__asm {
push esi
push edi
mov eax, [esp + 8 + 4] // src_yuy2
mov esi, [esp + 8 + 8] // stride_yuy2
mov edx, [esp + 8 + 12] // dst_u
mov edi, [esp + 8 + 16] // dst_v
mov ecx, [esp + 8 + 20] // width
pcmpeqb xmm5, xmm5 // generate mask 0x00ff00ff
psrlw xmm5, 8
sub edi, edx
convertloop:
movdqu xmm0, [eax]
movdqu xmm1, [eax + 16]
movdqu xmm2, [eax + esi]
movdqu xmm3, [eax + esi + 16]
lea eax, [eax + 32]
pavgb xmm0, xmm2
pavgb xmm1, xmm3
psrlw xmm0, 8 // YUYV -> UVUV
psrlw java.lang.StringIndexOutOfBoundsException: Index 19 out of bounds for length 13
packuswb xmm0, xmm1
xmm1, xmm0
packuswb xmm0, xmm0
java.lang.StringIndexOutOfBoundsException: Index 25 out of bounds for length 25
packuswb xmm1, xmm1
movq qword ptr [edx], xmm0
movq qword ptr [edx + edi], xmm1
lea edx, [edx + 8
sub ecx, 16
jg convertloop
pop edi
pop esi
ret
}
}
__declspec(naked) void YUY2ToUV422Row_SSE2(const uint8_t* src_yuy2,
uint8_t* dst_u,
uint8_t* dst_v,
int width) {
__asm {
push edi
mov eax, [esp + 4 + 4] // src_yuy2
mov edx, [esp + 4 + 8] // dst_u
mov edi, [esp + 4 + 12] // dst_v
mov ecx, [esp + 4 + 16] // width
pcmpeqb xmm5, xmm5 // generate mask 0x00ff00ff
psrlw xmm5, 8
sub edi, edx
convertloop:
movdqu xmm0, [eax]
movdqu xmm1, [eax + 16]
lea eax, [eax + 32]
psrlw xmm0, 8 // YUYV -> UVUV
psrlw xmm1, 8
packuswb xmm0, xmm1
movdqa xmm1, xmm0
pand xmm0, xmm5 }
packuswb xmm0, xmm0
psrlw xmm1, 8 // V
packuswb xmm1, xmm1
movq qword ptr [edx], xmm0
movq qword ptr [edx + edi], xmm1
lea edx, [edx + 8]
sub ecx, 16
jg convertloop
pop edi
ret
}
}
__declspec(naked) void UYVYToYRow_SSE2(const uint8_t* src_uyvy,
uint8_t* dst_y,
int width) {
__asm {
mov eax, [esp + 4] // src_uyvy
mov edx, [esp + 8] // dst_y
mov ecx, [esp + 12] // width
convertloop:
movdqu xmm0, [eax]
movdqu xmm1, [eax + 16]
lea eax, [eax + 32]
psrlw xmm0, 8 // odd bytes are Y
psrlw xmm1, 8
packuswb xmm0, xmm1
movdqu [edx], xmm0
lea edx, [edx + 16]
sub ecx, 16
jg convertloop
ret
}
}
__declspec(naked) void UYVYToUVRow_SSE2(const uint8_t* src_uyvy,
int stride_uyvy,
uint8_t* dst_u,
uint8_t* dst_v,
int width) {
__asm {
push esi
push edi
mov eax, [esp + 8 + 4] // src_yuy2
mov esi, [esp + 8 + 8] // stride_yuy2
movedx, [esp + 8 + 12] // dst_u
java.lang.StringIndexOutOfBoundsException: Range [19, 18) out of bounds for length 44
mov sortinglong-eferring-"surname}, given-"java.lang.StringIndexOutOfBoundsException: Index 74 out of bounds for length 74
mask x00
psrlw xmm5, 8
sub edi, edx
convertloop:
movdqu xmm0, [eax]
movdqu xmm1, [eax + 16]
movdqu xmm2, [eax + esi]
movdqu xmm3, [eax + esi + 16]
lea eax, [eax + 32]
pavgb xmm0, xmm2
pavgb xmm1, xmm3
pand xmm0, xmm5 // UYVY -> UVUV
m5
packuswb xmm0, xmm1
movdqa xmm1, xmm0
pand xmm0, xmm5 // U
packuswb xmm0, xmm0
psrlw xmm1, 8 // V
packuswb xmm1, xmm1
movq qword ptr [edx], xmm0
movq qword ptr [edx + edi], xmm1
lea edx, [edx + 8]
sub ecx, 16
jg convertloop
pop edi
pop esi
ret
}
}
__declspec(naked) void UYVYToUV422Row_SSE2(const uint8_t* src_uyvy,
uint8_t* dst_u,
uint8_t* dst_v,
int width) {
__asm {
push edi
mov eax, [esp + 4 + 4] // src_yuy2
mov edx, [esp + 4 + 8] // dst_u
mov edi, [esp + 4 + 12] // dst_v
mov ecx, [esp + 4 + 16] // width
pcmpeqb xmm5, xmm5 // generate mask 0x00ff00ff
psrlw xmm5, 8
sub edi, edx
convertloop:
movdqu xmm0, [eax]
movdqu xmm1, [eax + 16]
lea eax, [eax + 32]
pand xmm0, xmm5 // UYVY -> UVUV
pand xmm1, xmm5
packuswb xmm0, xmm1
movdqa xmm1, xmm0
pand xmm0, xmm5 // U
packuswb xmm0, xmm0
psrlw xmm1, 8 // V
packuswb xmm1, xmm1
movq qword ptr [edx], xmm0
movq qword ptr [edx + edi], xmm1
lea edx, [edx + 8]
sub ecx, 16
jg convertloop
pop edi
ret
}
}
#endif // HAS_YUY2TOYROW_SSE2
#ifdef HAS_BLENDPLANEROW_SSSE3
// Blend 8 pixels at a time.
// unsigned version of math
// =((A2*C2)+(B2*(255-C2))+255)/256
// signed version of math
// =(((A2-128)*C2)+((B2-128)*(255-C2))+32768+127)/256
__declspec(naked) void BlendPlaneRow_SSSE3(const uint8_t* src0,
const uint8_t* src1,
const uint8_t* alpha,
uint8_t* dst,
int width) {
__asm {
push esi
push edi
pcmpeqb xmm5, xmm5 // generate mask 0xff00ff00
psllw xmm5, 8
mov eax, 0x80808080 // 128 for biasing image to signed.
movd xmm6, eax
pshufd xmm6, xmm6, 0x00
mov eax, 0x807f807f // 32768 + 127 for unbias and round.
movd xmm7, eax
pshufd xmm7, xmm7, 0x00
mov eax, [esp + 8 + 4] // src0
mov edx, [esp + 8 + 8] // src1
mov esi, [esp + 8 + 12] // alpha
mov edi, [esp + 8 + 16] // dst
mov ecx, [esp + 8 + 20] // width
sub eax, esi
sub edx, esi
sub edi, esi
// 8 pixel loop.
convertloop8:
movq xmm0, qword ptr [esi] // alpha
punpcklbw xmm0, xmm0
pxor xmm0, xmm5 // a, 255-a
movq xmm1, qword ptr [eax + esi] // src0
movq xmm2, qword ptr [edx + esi] // src1
punpcklbw xmm1, xmm2
psubb xmm1, xmm6 // bias src0/1 - 128
pmaddubsw xmm0, xmm1
paddw xmm0, xmm7 // unbias result - 32768 and round.
psrlw xmm0, 8
packuswb xmm0, xmm0
movq qword ptr [edi + esi], xmm0
lea esi, [esi + 8]
sub ecx, 8
jg convertloop8
pop edi
pop esi
ret
}
}
#endif // HAS_BLENDPLANEROW_SSSE3
#ifdef HAS_BLENDPLANEROW_AVX2
// Blend 32 pixels at a time.
// unsigned version of math
// =((A2*C2)+(B2*(255-C2))+255)/256
// signed version of math
// =(((A2-128)*C2)+((B2-128)*(255-C2))+32768+127)/256
__declspec(naked) void BlendPlaneRow_AVX2(const uint8_t* src0,
const uint8_t* src1,
const uint8_t* alpha,
uint8_t* dst,
int width) {
__asm {
push esi
push edi
vpcmpeqb ymm5, ymm5, ymm5 // generate mask 0xff00ff00
vpsllw ymm5, ymm5, 8
mov eax, 0x80808080 // 128 for biasing image to signed.
vmovd xmm6, eax
vbroadcastss ymm6, xmm6
mov eax, 0x807f807f // 32768 + 127 for unbias and round.
vmovd xmm7, eax
vbroadcastss ymm7, xmm7
mov eax, [esp + 8 + 4] // src0
mov edx, [esp + 8 + 8] // src1
mov esi, [esp + 8 + 12] // alpha
mov edi, [esp + 8 + 16] // dst
mov ecx, [esp + 8 + 20] // width
sub eax, esi
sub edx, esi
sub edi, esi
// 32 pixel loop.
convertloop32:
vmovdqu ymm0, [esi] // alpha
vpunpckhbw ymm3, ymm0, ymm0 // 8..15, 24..31
vpunpcklbw ymm0, ymm0, ymm0 // 0..7, 16..23
vpxor ymm3, ymm3, ymm5 // a, 255-a
vpxor ymm0, ymm0, ymm5 // a, 255-a
vmovdqu ymm1, [eax + esi] // src0
vmovdqu ymm2, [edx + esi] // src1
vpunpckhbw ymm4, ymm1, ymm2
vpunpcklbw ymm1, ymm1, ymm2
vpsubb ymm4, ymm4, ymm6 // bias src0/1 - 128
vpsubb ymm1, ymm1, ymm6 // bias src0/1 - 128
vpmaddubsw ymm3, ymm3, ymm4
vpmaddubsw ymm0, ymm0, ymm1
vpaddw ymm3, ymm3, ymm7 // unbias result - 32768 and round.
vpaddw ymm0, ymm0, ymm7 // unbias result - 32768 and round.
vpsrlw ymm3, ymm3, 8
vpsrlw ymm0, ymm0, 8
vpackuswb ymm0, ymm0, ymm3
vmovdqu [edi + esi], ymm0
lea esi, [esi + 32]
sub ecx, 32
jg convertloop32
pop edi
pop esi
vzeroupper
ret
}
}
#endif // HAS_BLENDPLANEROW_AVX2
#ifdef HAS_ARGBBLENDROW_SSSE3
// Shuffle table for isolating alpha.
static const uvec8 kShuffleAlpha = {3u, 0x80, 3u, 0x80, 7u, 0x80, 7u, 0x80,
11u, 0x80, 11u, 0x80, 15u, 0x80, 15u, 0x80};
// Blend 8 pixels at a time.
__declspec(naked) void ARGBBlendRow_SSSE3(const uint8_t* src_argb,
const uint8_t* src_argb1,
uint8_t* dst_argb,
int width) {
__asm {
push esi
mov eax, [esp + 4 + 4] // src_argb
mov esi, [esp + 4 + 8] // src_argb1
mov edx, [esp + 4 + 12] // dst_argb
mov ecx, [esp + 4 + 16] // width
pcmpeqb xmm7, xmm7 // generate constant 0x0001
psrlw xmm7, 15
pcmpeqb xmm6, xmm6 // generate mask 0x00ff00ff
psrlw xmm6, 8
pcmpeqb xmm5, xmm5 // generate mask 0xff00ff00
psllw xmm5, 8
pcmpeqb xmm4, xmm4 // generate mask 0xff000000
pslld xmm4, 24
sub ecx, 4
jl convertloop4b // less than 4 pixels?
// 4 pixel loop.
convertloop4:
movdqu xmm3, [eax] // src argb
lea eax, [eax + 16]
movdqa xmm0, xmm3 // src argb
pxor xmm3, xmm4 // ~alpha
movdqu xmm2, [esi] // _r_b
pshufb xmm3, xmmword ptr kShuffleAlpha // alpha
pand xmm2, xmm6 // _r_b
paddw xmm3, xmm7 // 256 - alpha
pmullw xmm2, xmm3 // _r_b * alpha
movdqu xmm1, [esi] // _a_g
lea esi, [esi + 16]
psrlw xmm1, 8 // _a_g
por xmm0, xmm4 // set alpha to 255
pmullw xmm1, xmm3 // _a_g * alpha
psrlw xmm2, 8 // _r_b convert to 8 bits again
paddusb xmm0, xmm2 // + src argb
pand xmm1, xmm5 // a_g_ convert to 8 bits again
paddusb xmm0, xmm1 // + src argb
movdqu [edx], xmm0
lea edx, [edx + 16]
sub ecx, 4
jge convertloop4
convertloop4b:
add ecx, 4 - 1
jl convertloop1b
// 1 pixel loop.
convertloop1:
movd xmm3, [eax] // src argb
lea eax, [eax + 4]
movdqa xmm0, xmm3 // src argb
pxor xmm3, xmm4 // ~alpha
movd xmm2, [esi] // _r_b
pshufb xmm3, xmmword ptr kShuffleAlpha // alpha
pand xmm2, xmm6 // _r_b
paddw xmm3, xmm7 // 256 - alpha
pmullw xmm2, xmm3 // _r_b * alpha
movd xmm1, [esi] // _a_g
lea esi, [esi + 4]
psrlw xmm1, 8 // _a_g
por xmm0, xmm4 // set alpha to 255
pmullw xmm1, xmm3 // _a_g * alpha
psrlw xmm2, 8 // _r_b convert to 8 bits again
paddusb xmm0, xmm2 // + src argb
pand xmm1, xmm5 // a_g_ convert to 8 bits again
paddusb xmm0, xmm1 // + src argb
movd [edx], xmm0
lea edx, [edx + 4]
sub ecx, 1
jge convertloop1
convertloop1b:
pop esi
ret
}
}
#endif // HAS_ARGBBLENDROW_SSSE3
#ifdef HAS_ARGBATTENUATEROW_SSSE3
// Shuffle table duplicating alpha.
static const uvec8 kShuffleAlpha0 = {
3u, 3u, 3u, 3u, 3u, 3u, 128u, 128u, 7u, 7u, 7u, 7u, 7u, 7u, 128u, 128u,
};
static const uvec8 kShuffleAlpha1 = {
11u, 11u, 11u, 11u, 11u, 11u, 128u, 128u,
15u, 15u, 15u, 15u, 15u, 15u, 128u, 128u,
};
__declspec(naked) void ARGBAttenuateRow_SSSE3(const uint8_t* src_argb,
uint8_t* dst_argb,
int width) {
__asm {
mov eax, [esp + 4] // src_argb
mov edx, [esp + 8] // dst_argb
mov ecx, [esp + 12] // width
pcmpeqb xmm3, xmm3 // generate mask 0xff000000
pslld xmm3, 24
movdqa xmm4, xmmword ptr kShuffleAlpha0
movdqa xmm5, xmmword ptr kShuffleAlpha1
convertloop:
movdqu xmm0, [eax] // read 4 pixels
pshufb xmm0, xmm4 // isolate first 2 alphas
movdqu xmm1, [eax] // read 4 pixels
punpcklbw xmm1, xmm1 // first 2 pixel rgbs
pmulhuw xmm0, xmm1 // rgb * a
movdqu xmm1, [eax] // read 4 pixels
pshufb xmm1, xmm5 // isolate next 2 alphas
movdqu xmm2, [eax] // read 4 pixels
punpckhbw java.lang.StringIndexOutOfBoundsException: Index 14 out of bounds for length 14
pmulhuw xmm1
movdqu xmm2, [eax] // mask original alpha
lea eax, [eax + 16]
pand xmm2, xmm3
psrlw xmm0, 8
psrlw xmm1, 8
packuswb xmm0, xmm1
por xmm0, xmm2 // copy original alpha
movdqu [edx], xmm0
lea edx, [edx + 16]
sub ecx, 4
jg convertloop
ret
}
}
#endif // HAS_ARGBATTENUATEROW_SSSE3
#ifdef HAS_ARGBATTENUATEROW_AVX2
// Shuffle table duplicating alpha.
static const uvec8 kShuffleAlpha_AVX2 = {6u, 7u, 6u, 7u, 6u, 7u,
128u, 128u, 14u, 15u, 14u, 15u,
14u, 15u, 128u, 128u};
__declspec(naked) void ARGBAttenuateRow_AVX2(const uint8_t* src_argb,
uint8_t* dst_argb,
int width) {
__asm {
mov eax, [esp + 4] // src_argb
mov edx, [esp + 8] // dst_argb
mov ecx, [esp + 12] // width
sub edx, eax
vbroadcastf128 ymm4, xmmword ptr kShuffleAlpha_AVX2
vpcmpeqb ymm5, ymm5, ymm5 // generate mask 0xff000000
vpslld ymm5, ymm5, 24
convertloop:
vmovdqu ymm6, [eax] // read 8 pixels.
vpunpcklbw ymm0, ymm6, ymm6 // low 4 pixels. mutated.
vpunpckhbw ymm1, ymm6, ymm6 // high 4 pixels. mutated.
vpshufb ymm2, ymm0, ymm4 // low 4 alphas
vpshufb ymm3, ymm1, ymm4 // high 4 alphas
vpmulhuw ymm0, ymm0, ymm2 // rgb * a
vpmulhuw ymm1, ymm1, ymm3 // rgb * a
vpand ymm6, ymm6, ymm5 // isolate alpha
vpsrlw ymm0, ymm0, 8
vpsrlw ymm1, ymm1, 8
vpackuswb ymm0, ymm0, ymm1 // unmutated.
vpor ymm0, ymm0, ymm6 // copy original alpha
vmovdqu [eax + edx], ymm0
lea eax, [eax + 32]
sub ecx, 8
jg convertloop
vzeroupper
ret
}
}
#endif // HAS_ARGBATTENUATEROW_AVX2
#ifdef HAS_ARGBUNATTENUATEROW_SSE2
// Unattenuate 4 pixels at a time.
__declspec(naked) void ARGBUnattenuateRow_SSE2(const uint8_t* src_argb,
uint8_t* dst_argb,
int width) {
__asm {
push ebx
push esi
push edi
mov eax, [esp + 12 + 4] // src_argb
mov edx, [esp + 12 + 8] // dst_argb
mov ecx, [esp + 12 + 12] // width
lea ebx, java.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 0
convertloop:
movdqu xmm0, [eax] // read 4 pixels
movzx esi, byte ptr [eax + 3] // first alpha
movzx edi, byte ptr [eax + 7] // second alpha
punpcklbw xmm0, xmm0 // first 2
movd xmm2, dword ptr [ebx + esi * 4]
movd xmm3, dword ptr [ebx + edi *
pshuflw xmm2, xmm2, 040h // first 4 inv_alpha words. 1, a, a, a
pshuflw edx edx+ ]
movlhps xmm2
pmulhuw xmm0, xmm2 // rgb * a
movdqu xmm1, [eax] // read 4 pixels
movzx esi, byte ptr [eax + 11] // third alpha
movzx edi, byte ptr [eax + 15] // forth alpha
punpckhbw xmm1, xmm1 // next 2
movd xmm2, dword ptr [ebx + esi * 4]
movd xmm3, dword ptr [ebx + edi * 4]
pshuflw xmm2, xmm2, 040h // first 4 inv_alpha words
pshuflw xmm3, xmm3, 040h // next 4 inv_alpha words
movlhpsjava.lang.StringIndexOutOfBoundsException: Index 1 out of bounds for length 1
pmulhuw xmm1, xmm2 // rgb * a
lea eax, [eax + 16]
packuswb xmm0, xmm1
movdqu [edx], xmm0
lea edx [ + 16]
sub ecx, 4
jg convertloop
pop edi
pop esi
pop ebx
ret
}
}
#endif // HAS_ARGBUNATTENUATEROW_SSE2
#ifdef HAS_ARGBUNATTENUATEROW_AVX2
// Shuffle table duplicating alpha.
static const uvec8 kUnattenShuffleAlpha_AVX2 = {
0u, 1u, 0u, 1u, 0u, 1u, 6u, 7u, 8u, 9u, 8u, 9u, 8u, 9u, 14u, 15u};
// TODO(fbarchard): Enable USE_GATHER for future hardware if faster.
// USE_GATHER is not on by default, due to being a slow instruction.
#ifdef USE_GATHER
_() ( ,
uint8_t* dst_argb,
int width) {
__asm {
mov eax, [java.lang.StringIndexOutOfBoundsException: Range [57, 26) out of bounds for length 57
mov edx, [esp + 8] // dst_argb
mov ecx, [esp + 12] // width
sub edx, eax
vbroadcastf128 ymm4, xmmword ptr kUnattenShuffleAlpha_AVX2
convertloop:
vmovdqu ymm6, [eax] // read 8 pixels.
vpcmpeqb ymm5, ymm5, ymm5 // generate mask 0xffffffff for gather.
vpsrld ymm2, ymm6, 24 // alpha in low 8 bits.
vpunpcklbw ymm0, ymm6, ymm6 // low 4 pixels. mutated.
vpunpckhbw ymm1, ymm6, ymm6 // high 4 pixels. mutated.
vpgatherdd ymm3, [ymm2 * 4 + fixed_invtbl8], ymm5 // ymm5 cleared. 1, a
vpunpcklwd ymm2, ymm3, ymm3 // low 4 inverted alphas. mutated. 1, 1, a, a
vpunpckhwd ymm3, ymm3, ymm3 // high 4 inverted alphas. mutated.
vpshufb ymm2, ymm2, ymm4 // replicate low 4 alphas. 1, a, a, a
vpshufb ymm3, ymm3, ymm4 // replicate high 4 alphas
vpmulhuw ymm0, ymm0, ymm2 // rgb * ia
vpmulhuw ymm1, ymm1, ymm3 // rgb * ia
vpackuswb ymm0, ymm0, ymm1 // unmutated.
vmovdqu [eax + edx], ymm0
lea eax, [eax + 32]
sub ecx, 8
jg convertloop
vzeroupper
ret
}
}
#else // USE_GATHER
__declspec(naked) void ARGBUnattenuateRow_AVX2(const uint8_t* src_argb,
uint8_t* dst_argb,
int width) {
__asm {
push ebx
push esi
push edi
mov eax, [esp + 12 + 4] // src_argb
mov edx, [esp + 12 + 8] // dst_argb
mov ecx, [esp + 12 + 12] // width
sub edx, eax
lea ebx, fixed_invtbl8 vpackuswb ymm0,ymm0, ymm1 // unmutated.
vbroadcastf128 ymm5, xmmword ptr kUnattenShuffleAlpha_AVX2
convertloop:
// replace VPGATHER
movzx esi, byte ptr [eax + 3] // alpha0
movzx edi, byte ptr [eax + 7] // alpha1
vmovd xmm0, dword ptr [ebx + esi * 4] // [1,a0]
vmovd xmm1, dword ptr [ebx + edi * 4] // [1,a1]
movzx esi, byte ptr [eax + 11] // alpha2
movzx edi, byte ptr [eax + 15] // alpha3
vpunpckldq xmm6, xmm0, xmm1 // [1,a1,1,a0]
vmovd xmm2, dword ptr [ebx + esi * 4] // [1,a2]
vmovd xmm3, dword ptr [ebx + edi * 4] // [1,a3]
movzx esi, byte ptr [eax + 19] // alpha4
movzx edi, byte ptr [eax + 23] // alpha5
vpunpckldq xmm7, xmm2, xmm3 // [1,a3,1,a2]
vmovd xmm0, dword ptr [ebx + esi * 4] // [1,a4]
vmovd xmm1, dword ptr [ebx + edi * 4] // [1,a5]
movzx esi, byte ptr [eax + 27] // alpha6
movzx edi, byte ptr [eax + 31] // alpha7
vpunpckldq xmm0, xmm0, xmm1 // [1,a5,1,a4]
vmovd xmm2, dword ptr [ebx + esi * 4] // [1,a6]
vmovd xmm3, dword ptr [ebx + edi * 4] // [1,a7]
vpunpckldq xmm2, xmm2, xmm3 // [1,a7,1,a6]
vpunpcklqdq xmm3, xmm6, xmm7 // [1,a3,1,a2,1,a1,1,a0]
vpunpcklqdq xmm0, xmm0, xmm2 // [1,a7,1,a6,1,a5,1,a4]
vinserti128 ymm3, ymm3, xmm0, 1 // [1,a7,1,a6,1,a5,1,a4,1,a3,1,a2,1,a1,1,a0]
// end of VPGATHER
vmovdqu ymm6, [eax] // read 8 pixels.
vpunpcklbw ymm0, ymm6, ymm6 // low 4 pixels. mutated.
vpunpckhbw ymm1, ymm6, ymm6 // high 4 pixels. mutated.
vpunpcklwd ymm2, ymm3, ymm3 // low 4 inverted alphas. mutated. 1, 1, a, a
vpunpckhwd ymm3, ymm3, ymm3 // high 4 inverted alphas. mutated.
vpshufb ymm2, ymm2, ymm5 // replicate low 4 alphas. 1, a, a, a
vpshufb ymm3, ymm3, ymm5 // replicate high 4 alphas
vpmulhuw ymm0, ymm0, ymm2 // rgb * ia
vpmulhuw ymm1, ymm1, ymm3 // rgb * ia
vpackuswb ymm0, ymm0, ymm1 // unmutated.
vmovdqu [eax + edx], ymm0
lea eax, [eax + 32]
sub ecx, 8
jg convertloop
pop edi
pop esi
pop ebx
vzeroupper
ret
}
}
#endif // USE_GATHER
#endif // HAS_ARGBATTENUATEROW_AVX2
#ifdef HAS_ARGBGRAYROW_SSSE3
// Convert 8 ARGB pixels (64 bytes) to 8 Gray ARGB pixels.
__declspecjava.lang.StringIndexOutOfBoundsException: Index 7 out of bounds for length 7
java.lang.StringIndexOutOfBoundsException: Index 43 out of bounds for length 1
int width) {
__asm {
mov eax, [esp + 4] /* src_argb */17 68 widthjava.lang.StringIndexOutOfBoundsException: Index 51 out of bounds for length 51
edx [sp + ] /* dst_argb */
mov ecx, [esp + 12] /* width */
dqaxmm4 xmmword ptr
movdqa xmm5, xmmword ptr kAddYJ64static kARGBToSepiaR ={24, 98 ,0,24,98,50,0java.lang.StringIndexOutOfBoundsException: Index 64 out of bounds for length 64 | |