Blame view

ffmpeg-4.2.2/libavcodec/x86/hpeldsp_vp3.asm 3.13 KB
aac5773f   hucm   功能基本完成,接口待打磨
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
  ;******************************************************************************
  ;* SIMD-optimized halfpel functions for VP3
  ;*
  ;* This file is part of FFmpeg.
  ;*
  ;* FFmpeg is free software; you can redistribute it and/or
  ;* modify it under the terms of the GNU Lesser General Public
  ;* License as published by the Free Software Foundation; either
  ;* version 2.1 of the License, or (at your option) any later version.
  ;*
  ;* FFmpeg is distributed in the hope that it will be useful,
  ;* but WITHOUT ANY WARRANTY; without even the implied warranty of
  ;* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
  ;* Lesser General Public License for more details.
  ;*
  ;* You should have received a copy of the GNU Lesser General Public
  ;* License along with FFmpeg; if not, write to the Free Software
  ;* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
  ;******************************************************************************
  
  %include "libavutil/x86/x86util.asm"
  
  SECTION .text
  
  ; void ff_put_no_rnd_pixels8_x2_exact(uint8_t *block, const uint8_t *pixels, ptrdiff_t line_size, int h)
  %macro PUT_NO_RND_PIXELS8_X2_EXACT 0
  cglobal put_no_rnd_pixels8_x2_exact, 4,5
      lea          r4, [r2*3]
      pcmpeqb      m6, m6
  .loop:
      mova         m0, [r1]
      mova         m2, [r1+r2]
      mova         m1, [r1+1]
      mova         m3, [r1+r2+1]
      pxor         m0, m6
      pxor         m2, m6
      pxor         m1, m6
      pxor         m3, m6
      PAVGB        m0, m1
      PAVGB        m2, m3
      pxor         m0, m6
      pxor         m2, m6
      mova       [r0], m0
      mova    [r0+r2], m2
      mova         m0, [r1+r2*2]
      mova         m1, [r1+r2*2+1]
      mova         m2, [r1+r4]
      mova         m3, [r1+r4+1]
      pxor         m0, m6
      pxor         m1, m6
      pxor         m2, m6
      pxor         m3, m6
      PAVGB        m0, m1
      PAVGB        m2, m3
      pxor         m0, m6
      pxor         m2, m6
      mova  [r0+r2*2], m0
      mova    [r0+r4], m2
      lea          r1, [r1+r2*4]
      lea          r0, [r0+r2*4]
      sub         r3d, 4
      jg .loop
      REP_RET
  %endmacro
  
  INIT_MMX mmxext
  PUT_NO_RND_PIXELS8_X2_EXACT
  INIT_MMX 3dnow
  PUT_NO_RND_PIXELS8_X2_EXACT
  
  
  ; void ff_put_no_rnd_pixels8_y2_exact(uint8_t *block, const uint8_t *pixels, ptrdiff_t line_size, int h)
  %macro PUT_NO_RND_PIXELS8_Y2_EXACT 0
  cglobal put_no_rnd_pixels8_y2_exact, 4,5
      lea          r4, [r2*3]
      mova         m0, [r1]
      pcmpeqb      m6, m6
      add          r1, r2
      pxor         m0, m6
  .loop:
      mova         m1, [r1]
      mova         m2, [r1+r2]
      pxor         m1, m6
      pxor         m2, m6
      PAVGB        m0, m1
      PAVGB        m1, m2
      pxor         m0, m6
      pxor         m1, m6
      mova       [r0], m0
      mova    [r0+r2], m1
      mova         m1, [r1+r2*2]
      mova         m0, [r1+r4]
      pxor         m1, m6
      pxor         m0, m6
      PAVGB        m2, m1
      PAVGB        m1, m0
      pxor         m2, m6
      pxor         m1, m6
      mova  [r0+r2*2], m2
      mova    [r0+r4], m1
      lea          r1, [r1+r2*4]
      lea          r0, [r0+r2*4]
      sub         r3d, 4
      jg .loop
      REP_RET
  %endmacro
  
  INIT_MMX mmxext
  PUT_NO_RND_PIXELS8_Y2_EXACT
  INIT_MMX 3dnow
  PUT_NO_RND_PIXELS8_Y2_EXACT