Mercurial > sdl-ios-xcode
comparison src/video/SDL_yuv_mmx.c @ 887:b4b64bb88f2f
Date: Mon, 10 May 2004 10:17:46 -0400
From: Mike Frysinger
Subject: Re: [SDL] gcc-3.4.0 / PIC fix
here's a combined patch (yours and the one i mentioned earlier) that i tested
with gcc-3.4.0 and gcc-3.3.3
author | Sam Lantinga <slouken@libsdl.org> |
---|---|
date | Sun, 16 May 2004 17:19:48 +0000 |
parents | b8d311d90021 |
children | 8520712f8ef0 |
comparison
equal
deleted
inserted
replaced
886:05c551e5bc64 | 887:b4b64bb88f2f |
---|---|
118 "movd (%%ebx), %%mm1\n" // 0 0 0 0 v3 v2 v1 v0 | 118 "movd (%%ebx), %%mm1\n" // 0 0 0 0 v3 v2 v1 v0 |
119 "pxor %%mm7,%%mm7\n" // 00 00 00 00 00 00 00 00 | 119 "pxor %%mm7,%%mm7\n" // 00 00 00 00 00 00 00 00 |
120 "movd (%2), %%mm2\n" // 0 0 0 0 l3 l2 l1 l0 | 120 "movd (%2), %%mm2\n" // 0 0 0 0 l3 l2 l1 l0 |
121 "punpcklbw %%mm7,%%mm1\n" // 0 v3 0 v2 00 v1 00 v0 | 121 "punpcklbw %%mm7,%%mm1\n" // 0 v3 0 v2 00 v1 00 v0 |
122 "punpckldq %%mm1,%%mm1\n" // 00 v1 00 v0 00 v1 00 v0 | 122 "punpckldq %%mm1,%%mm1\n" // 00 v1 00 v0 00 v1 00 v0 |
123 "psubw _MMX_0080w,%%mm1\n" // mm1-128:r1 r1 r0 r0 r1 r1 r0 r0 | 123 "psubw %[_MMX_0080w],%%mm1\n" // mm1-128:r1 r1 r0 r0 r1 r1 r0 r0 |
124 | 124 |
125 // create Cr_g (result in mm0) | 125 // create Cr_g (result in mm0) |
126 "movq %%mm1,%%mm0\n" // r1 r1 r0 r0 r1 r1 r0 r0 | 126 "movq %%mm1,%%mm0\n" // r1 r1 r0 r0 r1 r1 r0 r0 |
127 "pmullw _MMX_VgrnRGB,%%mm0\n"// red*-46dec=0.7136*64 | 127 "pmullw %[_MMX_VgrnRGB],%%mm0\n"// red*-46dec=0.7136*64 |
128 "pmullw _MMX_VredRGB,%%mm1\n"// red*89dec=1.4013*64 | 128 "pmullw %[_MMX_VredRGB],%%mm1\n"// red*89dec=1.4013*64 |
129 "psraw $6, %%mm0\n" // red=red/64 | 129 "psraw $6, %%mm0\n" // red=red/64 |
130 "psraw $6, %%mm1\n" // red=red/64 | 130 "psraw $6, %%mm1\n" // red=red/64 |
131 | 131 |
132 // create L1 L2 (result in mm2,mm4) | 132 // create L1 L2 (result in mm2,mm4) |
133 // L2=lum+cols | 133 // L2=lum+cols |
134 "movq (%2,%4),%%mm3\n" // 0 0 0 0 L3 L2 L1 L0 | 134 "movq (%2,%4),%%mm3\n" // 0 0 0 0 L3 L2 L1 L0 |
135 "punpckldq %%mm3,%%mm2\n" // L3 L2 L1 L0 l3 l2 l1 l0 | 135 "punpckldq %%mm3,%%mm2\n" // L3 L2 L1 L0 l3 l2 l1 l0 |
136 "movq %%mm2,%%mm4\n" // L3 L2 L1 L0 l3 l2 l1 l0 | 136 "movq %%mm2,%%mm4\n" // L3 L2 L1 L0 l3 l2 l1 l0 |
137 "pand _MMX_FF00w,%%mm2\n" // L3 0 L1 0 l3 0 l1 0 | 137 "pand %[_MMX_FF00w],%%mm2\n" // L3 0 L1 0 l3 0 l1 0 |
138 "pand _MMX_00FFw,%%mm4\n" // 0 L2 0 L0 0 l2 0 l0 | 138 "pand %[_MMX_00FFw],%%mm4\n" // 0 L2 0 L0 0 l2 0 l0 |
139 "psrlw $8,%%mm2\n" // 0 L3 0 L1 0 l3 0 l1 | 139 "psrlw $8,%%mm2\n" // 0 L3 0 L1 0 l3 0 l1 |
140 | 140 |
141 // create R (result in mm6) | 141 // create R (result in mm6) |
142 "movq %%mm2,%%mm5\n" // 0 L3 0 L1 0 l3 0 l1 | 142 "movq %%mm2,%%mm5\n" // 0 L3 0 L1 0 l3 0 l1 |
143 "movq %%mm4,%%mm6\n" // 0 L2 0 L0 0 l2 0 l0 | 143 "movq %%mm4,%%mm6\n" // 0 L2 0 L0 0 l2 0 l0 |
150 | 150 |
151 // create Cb (result in mm1) | 151 // create Cb (result in mm1) |
152 "movd (%1), %%mm1\n" // 0 0 0 0 u3 u2 u1 u0 | 152 "movd (%1), %%mm1\n" // 0 0 0 0 u3 u2 u1 u0 |
153 "punpcklbw %%mm7,%%mm1\n" // 0 u3 0 u2 00 u1 00 u0 | 153 "punpcklbw %%mm7,%%mm1\n" // 0 u3 0 u2 00 u1 00 u0 |
154 "punpckldq %%mm1,%%mm1\n" // 00 u1 00 u0 00 u1 00 u0 | 154 "punpckldq %%mm1,%%mm1\n" // 00 u1 00 u0 00 u1 00 u0 |
155 "psubw _MMX_0080w,%%mm1\n" // mm1-128:u1 u1 u0 u0 u1 u1 u0 u0 | 155 "psubw %[_MMX_0080w],%%mm1\n" // mm1-128:u1 u1 u0 u0 u1 u1 u0 u0 |
156 // create Cb_g (result in mm5) | 156 // create Cb_g (result in mm5) |
157 "movq %%mm1,%%mm5\n" // u1 u1 u0 u0 u1 u1 u0 u0 | 157 "movq %%mm1,%%mm5\n" // u1 u1 u0 u0 u1 u1 u0 u0 |
158 "pmullw _MMX_UgrnRGB,%%mm5\n" // blue*-109dec=1.7129*64 | 158 "pmullw %[_MMX_UgrnRGB],%%mm5\n" // blue*-109dec=1.7129*64 |
159 "pmullw _MMX_UbluRGB,%%mm1\n" // blue*114dec=1.78125*64 | 159 "pmullw %[_MMX_UbluRGB],%%mm1\n" // blue*114dec=1.78125*64 |
160 "psraw $6, %%mm5\n" // blue=red/64 | 160 "psraw $6, %%mm5\n" // blue=red/64 |
161 "psraw $6, %%mm1\n" // blue=blue/64 | 161 "psraw $6, %%mm1\n" // blue=blue/64 |
162 | 162 |
163 // create G (result in mm7) | 163 // create G (result in mm7) |
164 "movq %%mm2,%%mm3\n" // 0 L3 0 L1 0 l3 0 l1 | 164 "movq %%mm2,%%mm3\n" // 0 L3 0 L1 0 l3 0 l1 |
236 "jl 1b\n" | 236 "jl 1b\n" |
237 "emms\n" | 237 "emms\n" |
238 "popl %%ebx\n" | 238 "popl %%ebx\n" |
239 : | 239 : |
240 : "m" (cr), "r"(cb),"r"(lum), | 240 : "m" (cr), "r"(cb),"r"(lum), |
241 "r"(row1),"r"(cols),"r"(row2),"m"(x),"m"(y),"m"(mod) | 241 "r"(row1),"r"(cols),"r"(row2),"m"(x),"m"(y),"m"(mod), |
242 : "%ebx" | 242 [_MMX_0080w] "m" (*_MMX_0080w), |
243 [_MMX_00FFw] "m" (*_MMX_00FFw), | |
244 [_MMX_FF00w] "m" (*_MMX_FF00w), | |
245 [_MMX_VgrnRGB] "m" (*_MMX_VgrnRGB), | |
246 [_MMX_VredRGB] "m" (*_MMX_VredRGB), | |
247 [_MMX_UgrnRGB] "m" (*_MMX_UgrnRGB), | |
248 [_MMX_UbluRGB] "m" (*_MMX_UbluRGB) | |
243 ); | 249 ); |
244 } | 250 } |
245 | 251 |
246 void Color565DitherYV12MMX1X( int *colortab, Uint32 *rgb_2_pix, | 252 void Color565DitherYV12MMX1X( int *colortab, Uint32 *rgb_2_pix, |
247 unsigned char *lum, unsigned char *cr, | 253 unsigned char *lum, unsigned char *cr, |
411 "jl 1b\n" | 417 "jl 1b\n" |
412 "emms\n" | 418 "emms\n" |
413 "popl %%ebx\n" | 419 "popl %%ebx\n" |
414 : | 420 : |
415 :"m" (cr), "r"(cb),"r"(lum), | 421 :"m" (cr), "r"(cb),"r"(lum), |
416 "r"(row1),"r"(cols),"r"(row2),"m"(x),"m"(y),"m"(mod) | 422 "r"(row1),"r"(cols),"r"(row2),"m"(x),"m"(y),"m"(mod), |
417 : "%ebx" | 423 [_MMX_0080w] "m" (*_MMX_0080w), |
424 [_MMX_Ugrn565] "m" (*_MMX_Ugrn565), | |
425 [_MMX_Ublu5x5] "m" (*_MMX_Ublu5x5), | |
426 [_MMX_00FFw] "m" (*_MMX_00FFw), | |
427 [_MMX_Vgrn565] "m" (*_MMX_Vgrn565), | |
428 [_MMX_Vred5x5] "m" (*_MMX_Vred5x5), | |
429 [_MMX_Ycoeff] "m" (*_MMX_Ycoeff), | |
430 [_MMX_red565] "m" (*_MMX_red565), | |
431 [_MMX_grn565] "m" (*_MMX_grn565) | |
418 ); | 432 ); |
419 } | 433 } |
420 | 434 |
421 #endif /* GCC i386 inline assembly */ | 435 #endif /* GCC i386 inline assembly */ |