mirror of the now-defunct rocklinux.org
You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

255 lines
7.3 KiB

  1. --- ./src/audio/SDL_mixer_MMX.c 9 Nov 2002 06:13:28 -0000
  2. +++ ./src/audio/SDL_mixer_MMX.c 3 May 2004 19:51:58 -0000
  3. @@ -15,13 +15,11 @@
  4. {
  5. __asm__ __volatile__ (
  6. -" movl %0,%%edi\n" // edi = dst
  7. -" movl %1,%%esi\n" // esi = src
  8. " movl %3,%%eax\n" // eax = volume
  9. -" movl %2,%%ebx\n" // ebx = size
  10. +" movl %2,%%edx\n" // edx = size
  11. -" shrl $4,%%ebx\n" // process 16 bytes per iteration = 8 samples
  12. +" shrl $4,%%edx\n" // process 16 bytes per iteration = 8 samples
  13. " jz .endS16\n"
  14. @@ -39,14 +37,14 @@
  15. ".align 16\n"
  16. " .mixloopS16:\n"
  17. -" movq (%%esi),%%mm1\n" // mm1 = a|b|c|d
  18. +" movq (%1),%%mm1\n" // mm1 = a|b|c|d
  19. " movq %%mm1,%%mm2\n" // mm2 = a|b|c|d
  20. -" movq 8(%%esi),%%mm4\n" // mm4 = e|f|g|h
  21. +" movq 8(%1),%%mm4\n" // mm4 = e|f|g|h
  22. // pr� charger le buffer dst dans mm7
  23. -" movq (%%edi),%%mm7\n" // mm7 = dst[0]"
  24. +" movq (%0),%%mm7\n" // mm7 = dst[0]"
  25. // multiplier par le volume
  26. " pmullw %%mm0,%%mm1\n" // mm1 = l(a*v)|l(b*v)|l(c*v)|l(d*v)
  27. @@ -69,11 +67,11 @@
  28. " punpcklwd %%mm5,%%mm6\n" // mm6 = g*v|h*v
  29. // pr� charger le buffer dst dans mm5
  30. -" movq 8(%%edi),%%mm5\n" // mm5 = dst[1]
  31. +" movq 8(%0),%%mm5\n" // mm5 = dst[1]
  32. // diviser par 128
  33. " psrad $7,%%mm1\n" // mm1 = a*v/128|b*v/128 , 128 = SDL_MIX_MAXVOLUME
  34. -" addl $16,%%esi\n"
  35. +" add $16,%1\n"
  36. " psrad $7,%%mm3\n" // mm3 = c*v/128|d*v/128
  37. @@ -87,15 +85,15 @@
  38. // mm4 = le sample avec le volume modifi�
  39. " packssdw %%mm4,%%mm6\n" // mm6 = s(e*v|f*v|g*v|h*v)
  40. -" movq %%mm3,(%%edi)\n"
  41. +" movq %%mm3,(%0)\n"
  42. " paddsw %%mm5,%%mm6\n" // mm6 = adjust_volume(src)+dst
  43. -" movq %%mm6,8(%%edi)\n"
  44. +" movq %%mm6,8(%0)\n"
  45. -" addl $16,%%edi\n"
  46. +" add $16,%0\n"
  47. -" dec %%ebx\n"
  48. +" dec %%edx\n"
  49. " jnz .mixloopS16\n"
  50. @@ -103,9 +101,9 @@
  51. ".endS16:\n"
  52. :
  53. - : "m" (dst), "m"(src),"m"(size),
  54. + : "r" (dst), "r"(src),"m"(size),
  55. "m"(volume)
  56. - : "eax","ebx", "esi", "edi","memory"
  57. + : "eax","edx","memory"
  58. );
  59. }
  60. @@ -119,11 +117,9 @@
  61. {
  62. __asm__ __volatile__ (
  63. -" movl %0,%%edi\n" // edi = dst
  64. -" movl %1,%%esi\n" // esi = src
  65. " movl %3,%%eax\n" // eax = volume
  66. -" movd %%ebx,%%mm0\n"
  67. +" movd %%edx,%%mm0\n"
  68. " movq %%mm0,%%mm1\n"
  69. " psllq $16,%%mm0\n"
  70. " por %%mm1,%%mm0\n"
  71. @@ -132,17 +128,17 @@
  72. " psllq $16,%%mm0\n"
  73. " por %%mm1,%%mm0\n"
  74. -" movl %2,%%ebx\n" // ebx = size
  75. -" shr $3,%%ebx\n" // process 8 bytes per iteration = 8 samples
  76. +" movl %2,%%edx\n" // edx = size
  77. +" shr $3,%%edx\n" // process 8 bytes per iteration = 8 samples
  78. -" cmp $0,%%ebx\n"
  79. +" cmp $0,%%edx\n"
  80. " je .endS8\n"
  81. ".align 16\n"
  82. " .mixloopS8:\n"
  83. " pxor %%mm2,%%mm2\n" // mm2 = 0
  84. -" movq (%%esi),%%mm1\n" // mm1 = a|b|c|d|e|f|g|h
  85. +" movq (%1),%%mm1\n" // mm1 = a|b|c|d|e|f|g|h
  86. " movq %%mm1,%%mm3\n" // mm3 = a|b|c|d|e|f|g|h
  87. @@ -152,10 +148,10 @@
  88. " punpckhbw %%mm2,%%mm1\n" // mm1 = 0|a|0|b|0|c|0|d
  89. " punpcklbw %%mm2,%%mm3\n" // mm3 = 0|e|0|f|0|g|0|h
  90. -" movq (%%edi),%%mm2\n" // mm2 = destination
  91. +" movq (%0),%%mm2\n" // mm2 = destination
  92. " pmullw %%mm0,%%mm1\n" // mm1 = v*a|v*b|v*c|v*d
  93. -" addl $8,%%esi\n"
  94. +" add $8,%1\n"
  95. " pmullw %%mm0,%%mm3\n" // mm3 = v*e|v*f|v*g|v*h
  96. " psraw $7,%%mm1\n" // mm1 = v*a/128|v*b/128|v*c/128|v*d/128
  97. @@ -166,19 +162,19 @@
  98. " paddsb %%mm2,%%mm3\n" // add to destination buffer
  99. -" movq %%mm3,(%%edi)\n" // store back to ram
  100. -" addl $8,%%edi\n"
  101. +" movq %%mm3,(%0)\n" // store back to ram
  102. +" add $8,%0\n"
  103. -" dec %%ebx\n"
  104. +" dec %%edx\n"
  105. " jnz .mixloopS8\n"
  106. ".endS8:\n"
  107. " emms\n"
  108. :
  109. - : "m" (dst), "m"(src),"m"(size),
  110. + : "r" (dst), "r"(src),"m"(size),
  111. "m"(volume)
  112. - : "eax","ebx", "esi", "edi","memory"
  113. + : "eax","edx","memory"
  114. );
  115. }
  116. #endif
  117. --- ./src/cpuinfo/SDL_cpuinfo.c 11 Apr 2004 19:49:34 -0000
  118. +++ ./src/cpuinfo/SDL_cpuinfo.c 3 May 2004 19:52:08 -0000
  119. @@ -118,7 +118,7 @@
  120. " movl %%edi,%%ebx\n"
  121. : "=m" (features)
  122. :
  123. - : "%eax", "%ebx", "%ecx", "%edx", "%edi"
  124. + : "%eax", "%ecx", "%edx", "%edi"
  125. );
  126. #elif defined(_MSC_VER)
  127. __asm {
  128. @@ -153,7 +153,7 @@
  129. " movl %%edi,%%ebx\n"
  130. : "=m" (features)
  131. :
  132. - : "%eax", "%ebx", "%ecx", "%edx", "%edi"
  133. + : "%eax", "%ecx", "%edx", "%edi"
  134. );
  135. #elif defined(_MSC_VER)
  136. __asm {
  137. --- ./src/video/SDL_yuv_mmx.c 4 Jan 2004 16:49:22 -0000
  138. +++ ./src/video/SDL_yuv_mmx.c 3 May 2004 19:52:24 -0000
  139. @@ -120,12 +120,12 @@
  140. "movd (%2), %%mm2\n" // 0 0 0 0 l3 l2 l1 l0
  141. "punpcklbw %%mm7,%%mm1\n" // 0 v3 0 v2 00 v1 00 v0
  142. "punpckldq %%mm1,%%mm1\n" // 00 v1 00 v0 00 v1 00 v0
  143. - "psubw _MMX_0080w,%%mm1\n" // mm1-128:r1 r1 r0 r0 r1 r1 r0 r0
  144. + "psubw %[_MMX_0080w],%%mm1\n" // mm1-128:r1 r1 r0 r0 r1 r1 r0 r0
  145. // create Cr_g (result in mm0)
  146. "movq %%mm1,%%mm0\n" // r1 r1 r0 r0 r1 r1 r0 r0
  147. - "pmullw _MMX_VgrnRGB,%%mm0\n"// red*-46dec=0.7136*64
  148. - "pmullw _MMX_VredRGB,%%mm1\n"// red*89dec=1.4013*64
  149. + "pmullw %[_MMX_VgrnRGB],%%mm0\n"// red*-46dec=0.7136*64
  150. + "pmullw %[_MMX_VredRGB],%%mm1\n"// red*89dec=1.4013*64
  151. "psraw $6, %%mm0\n" // red=red/64
  152. "psraw $6, %%mm1\n" // red=red/64
  153. @@ -134,8 +134,8 @@
  154. "movq (%2,%4),%%mm3\n" // 0 0 0 0 L3 L2 L1 L0
  155. "punpckldq %%mm3,%%mm2\n" // L3 L2 L1 L0 l3 l2 l1 l0
  156. "movq %%mm2,%%mm4\n" // L3 L2 L1 L0 l3 l2 l1 l0
  157. - "pand _MMX_FF00w,%%mm2\n" // L3 0 L1 0 l3 0 l1 0
  158. - "pand _MMX_00FFw,%%mm4\n" // 0 L2 0 L0 0 l2 0 l0
  159. + "pand %[_MMX_FF00w],%%mm2\n" // L3 0 L1 0 l3 0 l1 0
  160. + "pand %[_MMX_00FFw],%%mm4\n" // 0 L2 0 L0 0 l2 0 l0
  161. "psrlw $8,%%mm2\n" // 0 L3 0 L1 0 l3 0 l1
  162. // create R (result in mm6)
  163. @@ -152,11 +152,11 @@
  164. "movd (%1), %%mm1\n" // 0 0 0 0 u3 u2 u1 u0
  165. "punpcklbw %%mm7,%%mm1\n" // 0 u3 0 u2 00 u1 00 u0
  166. "punpckldq %%mm1,%%mm1\n" // 00 u1 00 u0 00 u1 00 u0
  167. - "psubw _MMX_0080w,%%mm1\n" // mm1-128:u1 u1 u0 u0 u1 u1 u0 u0
  168. + "psubw %[_MMX_0080w],%%mm1\n" // mm1-128:u1 u1 u0 u0 u1 u1 u0 u0
  169. // create Cb_g (result in mm5)
  170. "movq %%mm1,%%mm5\n" // u1 u1 u0 u0 u1 u1 u0 u0
  171. - "pmullw _MMX_UgrnRGB,%%mm5\n" // blue*-109dec=1.7129*64
  172. - "pmullw _MMX_UbluRGB,%%mm1\n" // blue*114dec=1.78125*64
  173. + "pmullw %[_MMX_UgrnRGB],%%mm5\n" // blue*-109dec=1.7129*64
  174. + "pmullw %[_MMX_UbluRGB],%%mm1\n" // blue*114dec=1.78125*64
  175. "psraw $6, %%mm5\n" // blue=red/64
  176. "psraw $6, %%mm1\n" // blue=blue/64
  177. @@ -238,8 +238,14 @@
  178. "popl %%ebx\n"
  179. :
  180. : "m" (cr), "r"(cb),"r"(lum),
  181. - "r"(row1),"r"(cols),"r"(row2),"m"(x),"m"(y),"m"(mod)
  182. - : "%ebx"
  183. + "r"(row1),"r"(cols),"r"(row2),"m"(x),"m"(y),"m"(mod),
  184. + [_MMX_0080w] "m" (*_MMX_0080w),
  185. + [_MMX_00FFw] "m" (*_MMX_00FFw),
  186. + [_MMX_FF00w] "m" (*_MMX_FF00w),
  187. + [_MMX_VgrnRGB] "m" (*_MMX_VgrnRGB),
  188. + [_MMX_VredRGB] "m" (*_MMX_VredRGB),
  189. + [_MMX_UgrnRGB] "m" (*_MMX_UgrnRGB),
  190. + [_MMX_UbluRGB] "m" (*_MMX_UbluRGB)
  191. );
  192. }
  193. @@ -413,8 +419,16 @@
  194. "popl %%ebx\n"
  195. :
  196. :"m" (cr), "r"(cb),"r"(lum),
  197. - "r"(row1),"r"(cols),"r"(row2),"m"(x),"m"(y),"m"(mod)
  198. - : "%ebx"
  199. + "r"(row1),"r"(cols),"r"(row2),"m"(x),"m"(y),"m"(mod),
  200. + [_MMX_0080w] "m" (*_MMX_0080w),
  201. + [_MMX_Ugrn565] "m" (*_MMX_Ugrn565),
  202. + [_MMX_Ublu5x5] "m" (*_MMX_Ublu5x5),
  203. + [_MMX_00FFw] "m" (*_MMX_00FFw),
  204. + [_MMX_Vgrn565] "m" (*_MMX_Vgrn565),
  205. + [_MMX_Vred5x5] "m" (*_MMX_Vred5x5),
  206. + [_MMX_Ycoeff] "m" (*_MMX_Ycoeff),
  207. + [_MMX_red565] "m" (*_MMX_red565),
  208. + [_MMX_grn565] "m" (*_MMX_grn565)
  209. );
  210. }