You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

227 lines
5.5KB

  1. ;******************************************************************************
  2. ;* MMX optimized DSP utils
  3. ;* Copyright (c) 2008 Loren Merritt
  4. ;*
  5. ;* This file is part of FFmpeg.
  6. ;*
  7. ;* FFmpeg is free software; you can redistribute it and/or
  8. ;* modify it under the terms of the GNU Lesser General Public
  9. ;* License as published by the Free Software Foundation; either
  10. ;* version 2.1 of the License, or (at your option) any later version.
  11. ;*
  12. ;* FFmpeg is distributed in the hope that it will be useful,
  13. ;* but WITHOUT ANY WARRANTY; without even the implied warranty of
  14. ;* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
  15. ;* Lesser General Public License for more details.
  16. ;*
  17. ;* You should have received a copy of the GNU Lesser General Public
  18. ;* License along with FFmpeg; if not, write to the Free Software
  19. ;* 51, Inc., Foundation Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
  20. ;******************************************************************************
  21. %include "x86inc.asm"
  22. SECTION_RODATA
  23. pb_f: times 16 db 15
  24. pb_zzzzzzzz77777777: times 8 db -1
  25. pb_7: times 8 db 7
  26. pb_zzzz3333zzzzbbbb: db -1,-1,-1,-1,3,3,3,3,-1,-1,-1,-1,11,11,11,11
  27. pb_zz11zz55zz99zzdd: db -1,-1,1,1,-1,-1,5,5,-1,-1,9,9,-1,-1,13,13
  28. section .text align=16
  29. %macro PSWAPD_SSE 2
  30. pshufw %1, %2, 0x4e
  31. %endmacro
  32. %macro PSWAPD_3DN1 2
  33. movq %1, %2
  34. psrlq %1, 32
  35. punpckldq %1, %2
  36. %endmacro
  37. %macro FLOAT_TO_INT16_INTERLEAVE6 1
  38. ; void ff_float_to_int16_interleave6_sse(int16_t *dst, const float **src, int len)
  39. cglobal float_to_int16_interleave6_%1, 2,7,0, dst, src, src1, src2, src3, src4, src5
  40. %ifdef ARCH_X86_64
  41. %define lend r10d
  42. mov lend, r2d
  43. %else
  44. %define lend dword r2m
  45. %endif
  46. mov src1q, [srcq+1*gprsize]
  47. mov src2q, [srcq+2*gprsize]
  48. mov src3q, [srcq+3*gprsize]
  49. mov src4q, [srcq+4*gprsize]
  50. mov src5q, [srcq+5*gprsize]
  51. mov srcq, [srcq]
  52. sub src1q, srcq
  53. sub src2q, srcq
  54. sub src3q, srcq
  55. sub src4q, srcq
  56. sub src5q, srcq
  57. .loop:
  58. cvtps2pi mm0, [srcq]
  59. cvtps2pi mm1, [srcq+src1q]
  60. cvtps2pi mm2, [srcq+src2q]
  61. cvtps2pi mm3, [srcq+src3q]
  62. cvtps2pi mm4, [srcq+src4q]
  63. cvtps2pi mm5, [srcq+src5q]
  64. packssdw mm0, mm3
  65. packssdw mm1, mm4
  66. packssdw mm2, mm5
  67. pswapd mm3, mm0
  68. punpcklwd mm0, mm1
  69. punpckhwd mm1, mm2
  70. punpcklwd mm2, mm3
  71. pswapd mm3, mm0
  72. punpckldq mm0, mm2
  73. punpckhdq mm2, mm1
  74. punpckldq mm1, mm3
  75. movq [dstq ], mm0
  76. movq [dstq+16], mm2
  77. movq [dstq+ 8], mm1
  78. add srcq, 8
  79. add dstq, 24
  80. sub lend, 2
  81. jg .loop
  82. emms
  83. RET
  84. %endmacro ; FLOAT_TO_INT16_INTERLEAVE6
  85. %define pswapd PSWAPD_SSE
  86. FLOAT_TO_INT16_INTERLEAVE6 sse
  87. %define cvtps2pi pf2id
  88. %define pswapd PSWAPD_3DN1
  89. FLOAT_TO_INT16_INTERLEAVE6 3dnow
  90. %undef pswapd
  91. FLOAT_TO_INT16_INTERLEAVE6 3dn2
  92. %undef cvtps2pi
  93. ; void ff_add_hfyu_median_prediction_mmx2(uint8_t *dst, const uint8_t *top, const uint8_t *diff, int w, int *left, int *left_top)
  94. cglobal add_hfyu_median_prediction_mmx2, 6,6,0, dst, top, diff, w, left, left_top
  95. movq mm0, [topq]
  96. movq mm2, mm0
  97. movd mm4, [left_topq]
  98. psllq mm2, 8
  99. movq mm1, mm0
  100. por mm4, mm2
  101. movd mm3, [leftq]
  102. psubb mm0, mm4 ; t-tl
  103. add dstq, wq
  104. add topq, wq
  105. add diffq, wq
  106. neg wq
  107. jmp .skip
  108. .loop:
  109. movq mm4, [topq+wq]
  110. movq mm0, mm4
  111. psllq mm4, 8
  112. por mm4, mm1
  113. movq mm1, mm0 ; t
  114. psubb mm0, mm4 ; t-tl
  115. .skip:
  116. movq mm2, [diffq+wq]
  117. %assign i 0
  118. %rep 8
  119. movq mm4, mm0
  120. paddb mm4, mm3 ; t-tl+l
  121. movq mm5, mm3
  122. pmaxub mm3, mm1
  123. pminub mm5, mm1
  124. pminub mm3, mm4
  125. pmaxub mm3, mm5 ; median
  126. paddb mm3, mm2 ; +residual
  127. %if i==0
  128. movq mm7, mm3
  129. psllq mm7, 56
  130. %else
  131. movq mm6, mm3
  132. psrlq mm7, 8
  133. psllq mm6, 56
  134. por mm7, mm6
  135. %endif
  136. %if i<7
  137. psrlq mm0, 8
  138. psrlq mm1, 8
  139. psrlq mm2, 8
  140. %endif
  141. %assign i i+1
  142. %endrep
  143. movq [dstq+wq], mm7
  144. add wq, 8
  145. jl .loop
  146. movzx r2d, byte [dstq-1]
  147. mov [leftq], r2d
  148. movzx r2d, byte [topq-1]
  149. mov [left_topq], r2d
  150. RET
  151. %macro ADD_HFYU_LEFT_LOOP 1 ; %1 = is_aligned
  152. add srcq, wq
  153. add dstq, wq
  154. neg wq
  155. %%.loop:
  156. mova m1, [srcq+wq]
  157. mova m2, m1
  158. psllw m1, 8
  159. paddb m1, m2
  160. mova m2, m1
  161. pshufb m1, m3
  162. paddb m1, m2
  163. pshufb m0, m5
  164. mova m2, m1
  165. pshufb m1, m4
  166. paddb m1, m2
  167. %if mmsize == 16
  168. mova m2, m1
  169. pshufb m1, m6
  170. paddb m1, m2
  171. %endif
  172. paddb m0, m1
  173. %if %1
  174. mova [dstq+wq], m0
  175. %else
  176. movq [dstq+wq], m0
  177. movhps [dstq+wq+8], m0
  178. %endif
  179. add wq, mmsize
  180. jl %%.loop
  181. mov eax, mmsize-1
  182. sub eax, wd
  183. movd m1, eax
  184. pshufb m0, m1
  185. movd eax, m0
  186. RET
  187. %endmacro
  188. ; int ff_add_hfyu_left_prediction(uint8_t *dst, const uint8_t *src, int w, int left)
  189. INIT_MMX
  190. cglobal add_hfyu_left_prediction_ssse3, 3,3,7, dst, src, w, left
  191. .skip_prologue:
  192. mova m5, [pb_7 GLOBAL]
  193. mova m4, [pb_zzzz3333zzzzbbbb GLOBAL]
  194. mova m3, [pb_zz11zz55zz99zzdd GLOBAL]
  195. movd m0, leftm
  196. psllq m0, 56
  197. ADD_HFYU_LEFT_LOOP 1
  198. INIT_XMM
  199. cglobal add_hfyu_left_prediction_sse4, 3,3,7, dst, src, w, left
  200. mova m5, [pb_f GLOBAL]
  201. mova m6, [pb_zzzzzzzz77777777 GLOBAL]
  202. mova m4, [pb_zzzz3333zzzzbbbb GLOBAL]
  203. mova m3, [pb_zz11zz55zz99zzdd GLOBAL]
  204. movd m0, leftm
  205. pslldq m0, 15
  206. test srcq, 15
  207. jnz add_hfyu_left_prediction_ssse3.skip_prologue
  208. test dstq, 15
  209. jnz .unaligned
  210. ADD_HFYU_LEFT_LOOP 1
  211. .unaligned:
  212. ADD_HFYU_LEFT_LOOP 0