Mercurial > libavcodec.hg
annotate ppc/h264_template_altivec.c @ 5818:e0a872dd3ea1 libavcodec
Fix MJPEG decoder for AMV files.
Since decoding is doing from the end and aligned by 16
previous code worked correctly only when picture height was dividable by 16,
otherwise it provides garbage in top lines and truncates bottom.
New code adjusts data[] pointers taking in account alignment issue.
author | voroshil |
---|---|
date | Sat, 13 Oct 2007 17:38:58 +0000 |
parents | 861eb234e6ba |
children | 93089aed00cb |
rev | line source |
---|---|
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
1 /* |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
2 * Copyright (c) 2004 Romain Dolbeau <romain@dolbeau.org> |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
3 * |
3947
c8c591fe26f8
Change license headers to say 'FFmpeg' instead of 'this program/this library'
diego
parents:
3577
diff
changeset
|
4 * This file is part of FFmpeg. |
c8c591fe26f8
Change license headers to say 'FFmpeg' instead of 'this program/this library'
diego
parents:
3577
diff
changeset
|
5 * |
c8c591fe26f8
Change license headers to say 'FFmpeg' instead of 'this program/this library'
diego
parents:
3577
diff
changeset
|
6 * FFmpeg is free software; you can redistribute it and/or |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
7 * modify it under the terms of the GNU Lesser General Public |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
8 * License as published by the Free Software Foundation; either |
3947
c8c591fe26f8
Change license headers to say 'FFmpeg' instead of 'this program/this library'
diego
parents:
3577
diff
changeset
|
9 * version 2.1 of the License, or (at your option) any later version. |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
10 * |
3947
c8c591fe26f8
Change license headers to say 'FFmpeg' instead of 'this program/this library'
diego
parents:
3577
diff
changeset
|
11 * FFmpeg is distributed in the hope that it will be useful, |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
12 * but WITHOUT ANY WARRANTY; without even the implied warranty of |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
13 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
14 * Lesser General Public License for more details. |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
15 * |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
16 * You should have received a copy of the GNU Lesser General Public |
3947
c8c591fe26f8
Change license headers to say 'FFmpeg' instead of 'this program/this library'
diego
parents:
3577
diff
changeset
|
17 * License along with FFmpeg; if not, write to the Free Software |
3036
0b546eab515d
Update licensing information: The FSF changed postal address.
diego
parents:
2967
diff
changeset
|
18 * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
19 */ |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
20 |
5603
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
21 //#define DEBUG_ALIGNMENT |
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
22 #ifdef DEBUG_ALIGNMENT |
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
23 #define ASSERT_ALIGNED(ptr) assert(((unsigned long)ptr&0x0000000F)); |
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
24 #else |
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
25 #define ASSERT_ALIGNED(ptr) ; |
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
26 #endif |
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
27 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
28 /* this code assume that stride % 16 == 0 */ |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
29 void PREFIX_h264_chroma_mc8_altivec(uint8_t * dst, uint8_t * src, int stride, int h, int x, int y) { |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
30 POWERPC_PERF_DECLARE(PREFIX_h264_chroma_mc8_num, 1); |
5019
41cabe79ba25
use macro Use DECLARE_ALIGNED_16 to align stack-allocated variables
gpoirier
parents:
3947
diff
changeset
|
31 DECLARE_ALIGNED_16(signed int, ABCD[4]) = |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
32 {((8 - x) * (8 - y)), |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
33 ((x) * (8 - y)), |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
34 ((8 - x) * (y)), |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
35 ((x) * (y))}; |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
36 register int i; |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
37 vec_u8_t fperm; |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
38 const vec_s32_t vABCD = vec_ld(0, ABCD); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
39 const vec_s16_t vA = vec_splat((vec_s16_t)vABCD, 1); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
40 const vec_s16_t vB = vec_splat((vec_s16_t)vABCD, 3); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
41 const vec_s16_t vC = vec_splat((vec_s16_t)vABCD, 5); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
42 const vec_s16_t vD = vec_splat((vec_s16_t)vABCD, 7); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
43 LOAD_ZERO; |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
44 const vec_s16_t v32ss = vec_sl(vec_splat_s16(1),vec_splat_u16(5)); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
45 const vec_u16_t v6us = vec_splat_u16(6); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
46 register int loadSecond = (((unsigned long)src) % 16) <= 7 ? 0 : 1; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
47 register int reallyBadAlign = (((unsigned long)src) % 16) == 15 ? 1 : 0; |
2967 | 48 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
49 vec_u8_t vsrcAuc, vsrcBuc, vsrcperm0, vsrcperm1; |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
50 vec_u8_t vsrc0uc, vsrc1uc; |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
51 vec_s16_t vsrc0ssH, vsrc1ssH; |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
52 vec_u8_t vsrcCuc, vsrc2uc, vsrc3uc; |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
53 vec_s16_t vsrc2ssH, vsrc3ssH, psum; |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
54 vec_u8_t vdst, ppsum, vfdst, fsum; |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
55 |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
56 POWERPC_PERF_START_COUNT(PREFIX_h264_chroma_mc8_num, 1); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
57 |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
58 if (((unsigned long)dst) % 16 == 0) { |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
59 fperm = (vec_u8_t)AVV(0x10, 0x11, 0x12, 0x13, |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
60 0x14, 0x15, 0x16, 0x17, |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
61 0x08, 0x09, 0x0A, 0x0B, |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
62 0x0C, 0x0D, 0x0E, 0x0F); |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
63 } else { |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
64 fperm = (vec_u8_t)AVV(0x00, 0x01, 0x02, 0x03, |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
65 0x04, 0x05, 0x06, 0x07, |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
66 0x18, 0x19, 0x1A, 0x1B, |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
67 0x1C, 0x1D, 0x1E, 0x1F); |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
68 } |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
69 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
70 vsrcAuc = vec_ld(0, src); |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
71 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
72 if (loadSecond) |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
73 vsrcBuc = vec_ld(16, src); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
74 vsrcperm0 = vec_lvsl(0, src); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
75 vsrcperm1 = vec_lvsl(1, src); |
2967 | 76 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
77 vsrc0uc = vec_perm(vsrcAuc, vsrcBuc, vsrcperm0); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
78 if (reallyBadAlign) |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
79 vsrc1uc = vsrcBuc; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
80 else |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
81 vsrc1uc = vec_perm(vsrcAuc, vsrcBuc, vsrcperm1); |
2967 | 82 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
83 vsrc0ssH = (vec_s16_t)vec_mergeh(zero_u8v,(vec_u8_t)vsrc0uc); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
84 vsrc1ssH = (vec_s16_t)vec_mergeh(zero_u8v,(vec_u8_t)vsrc1uc); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
85 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
86 if (!loadSecond) {// -> !reallyBadAlign |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
87 for (i = 0 ; i < h ; i++) { |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
88 |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
89 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
90 vsrcCuc = vec_ld(stride + 0, src); |
2967 | 91 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
92 vsrc2uc = vec_perm(vsrcCuc, vsrcCuc, vsrcperm0); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
93 vsrc3uc = vec_perm(vsrcCuc, vsrcCuc, vsrcperm1); |
2967 | 94 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
95 vsrc2ssH = (vec_s16_t)vec_mergeh(zero_u8v,(vec_u8_t)vsrc2uc); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
96 vsrc3ssH = (vec_s16_t)vec_mergeh(zero_u8v,(vec_u8_t)vsrc3uc); |
2967 | 97 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
98 psum = vec_mladd(vA, vsrc0ssH, vec_splat_s16(0)); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
99 psum = vec_mladd(vB, vsrc1ssH, psum); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
100 psum = vec_mladd(vC, vsrc2ssH, psum); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
101 psum = vec_mladd(vD, vsrc3ssH, psum); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
102 psum = vec_add(v32ss, psum); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
103 psum = vec_sra(psum, v6us); |
2967 | 104 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
105 vdst = vec_ld(0, dst); |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
106 ppsum = (vec_u8_t)vec_packsu(psum, psum); |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
107 vfdst = vec_perm(vdst, ppsum, fperm); |
2967 | 108 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
109 OP_U8_ALTIVEC(fsum, vfdst, vdst); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
110 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
111 vec_st(fsum, 0, dst); |
2967 | 112 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
113 vsrc0ssH = vsrc2ssH; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
114 vsrc1ssH = vsrc3ssH; |
2967 | 115 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
116 dst += stride; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
117 src += stride; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
118 } |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
119 } else { |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
120 vec_u8_t vsrcDuc; |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
121 for (i = 0 ; i < h ; i++) { |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
122 vsrcCuc = vec_ld(stride + 0, src); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
123 vsrcDuc = vec_ld(stride + 16, src); |
2967 | 124 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
125 vsrc2uc = vec_perm(vsrcCuc, vsrcDuc, vsrcperm0); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
126 if (reallyBadAlign) |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
127 vsrc3uc = vsrcDuc; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
128 else |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
129 vsrc3uc = vec_perm(vsrcCuc, vsrcDuc, vsrcperm1); |
2967 | 130 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
131 vsrc2ssH = (vec_s16_t)vec_mergeh(zero_u8v,(vec_u8_t)vsrc2uc); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
132 vsrc3ssH = (vec_s16_t)vec_mergeh(zero_u8v,(vec_u8_t)vsrc3uc); |
2967 | 133 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
134 psum = vec_mladd(vA, vsrc0ssH, vec_splat_s16(0)); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
135 psum = vec_mladd(vB, vsrc1ssH, psum); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
136 psum = vec_mladd(vC, vsrc2ssH, psum); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
137 psum = vec_mladd(vD, vsrc3ssH, psum); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
138 psum = vec_add(v32ss, psum); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
139 psum = vec_sr(psum, v6us); |
2967 | 140 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
141 vdst = vec_ld(0, dst); |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
142 ppsum = (vec_u8_t)vec_pack(psum, psum); |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
143 vfdst = vec_perm(vdst, ppsum, fperm); |
2967 | 144 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
145 OP_U8_ALTIVEC(fsum, vfdst, vdst); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
146 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
147 vec_st(fsum, 0, dst); |
2967 | 148 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
149 vsrc0ssH = vsrc2ssH; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
150 vsrc1ssH = vsrc3ssH; |
2967 | 151 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
152 dst += stride; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
153 src += stride; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
154 } |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
155 } |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
156 POWERPC_PERF_STOP_COUNT(PREFIX_h264_chroma_mc8_num, 1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
157 } |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
158 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
159 /* this code assume stride % 16 == 0 */ |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
160 static void PREFIX_h264_qpel16_h_lowpass_altivec(uint8_t * dst, uint8_t * src, int dstStride, int srcStride) { |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
161 POWERPC_PERF_DECLARE(PREFIX_h264_qpel16_h_lowpass_num, 1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
162 register int i; |
2967 | 163 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
164 LOAD_ZERO; |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
165 const vec_u8_t permM2 = vec_lvsl(-2, src); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
166 const vec_u8_t permM1 = vec_lvsl(-1, src); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
167 const vec_u8_t permP0 = vec_lvsl(+0, src); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
168 const vec_u8_t permP1 = vec_lvsl(+1, src); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
169 const vec_u8_t permP2 = vec_lvsl(+2, src); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
170 const vec_u8_t permP3 = vec_lvsl(+3, src); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
171 const vec_s16_t v5ss = vec_splat_s16(5); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
172 const vec_u16_t v5us = vec_splat_u16(5); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
173 const vec_s16_t v20ss = vec_sl(vec_splat_s16(5),vec_splat_u16(2)); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
174 const vec_s16_t v16ss = vec_sl(vec_splat_s16(1),vec_splat_u16(4)); |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
175 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
176 vec_u8_t srcM2, srcM1, srcP0, srcP1, srcP2, srcP3; |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
177 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
178 register int align = ((((unsigned long)src) - 2) % 16); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
179 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
180 vec_s16_t srcP0A, srcP0B, srcP1A, srcP1B, |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
181 srcP2A, srcP2B, srcP3A, srcP3B, |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
182 srcM1A, srcM1B, srcM2A, srcM2B, |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
183 sum1A, sum1B, sum2A, sum2B, sum3A, sum3B, |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
184 pp1A, pp1B, pp2A, pp2B, pp3A, pp3B, |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
185 psumA, psumB, sumA, sumB; |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
186 |
5603
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
187 vec_u8_t sum, vdst, fsum; |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
188 |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
189 POWERPC_PERF_START_COUNT(PREFIX_h264_qpel16_h_lowpass_num, 1); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
190 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
191 for (i = 0 ; i < 16 ; i ++) { |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
192 vec_u8_t srcR1 = vec_ld(-2, src); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
193 vec_u8_t srcR2 = vec_ld(14, src); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
194 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
195 switch (align) { |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
196 default: { |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
197 srcM2 = vec_perm(srcR1, srcR2, permM2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
198 srcM1 = vec_perm(srcR1, srcR2, permM1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
199 srcP0 = vec_perm(srcR1, srcR2, permP0); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
200 srcP1 = vec_perm(srcR1, srcR2, permP1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
201 srcP2 = vec_perm(srcR1, srcR2, permP2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
202 srcP3 = vec_perm(srcR1, srcR2, permP3); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
203 } break; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
204 case 11: { |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
205 srcM2 = vec_perm(srcR1, srcR2, permM2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
206 srcM1 = vec_perm(srcR1, srcR2, permM1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
207 srcP0 = vec_perm(srcR1, srcR2, permP0); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
208 srcP1 = vec_perm(srcR1, srcR2, permP1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
209 srcP2 = vec_perm(srcR1, srcR2, permP2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
210 srcP3 = srcR2; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
211 } break; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
212 case 12: { |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
213 vec_u8_t srcR3 = vec_ld(30, src); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
214 srcM2 = vec_perm(srcR1, srcR2, permM2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
215 srcM1 = vec_perm(srcR1, srcR2, permM1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
216 srcP0 = vec_perm(srcR1, srcR2, permP0); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
217 srcP1 = vec_perm(srcR1, srcR2, permP1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
218 srcP2 = srcR2; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
219 srcP3 = vec_perm(srcR2, srcR3, permP3); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
220 } break; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
221 case 13: { |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
222 vec_u8_t srcR3 = vec_ld(30, src); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
223 srcM2 = vec_perm(srcR1, srcR2, permM2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
224 srcM1 = vec_perm(srcR1, srcR2, permM1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
225 srcP0 = vec_perm(srcR1, srcR2, permP0); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
226 srcP1 = srcR2; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
227 srcP2 = vec_perm(srcR2, srcR3, permP2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
228 srcP3 = vec_perm(srcR2, srcR3, permP3); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
229 } break; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
230 case 14: { |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
231 vec_u8_t srcR3 = vec_ld(30, src); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
232 srcM2 = vec_perm(srcR1, srcR2, permM2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
233 srcM1 = vec_perm(srcR1, srcR2, permM1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
234 srcP0 = srcR2; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
235 srcP1 = vec_perm(srcR2, srcR3, permP1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
236 srcP2 = vec_perm(srcR2, srcR3, permP2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
237 srcP3 = vec_perm(srcR2, srcR3, permP3); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
238 } break; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
239 case 15: { |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
240 vec_u8_t srcR3 = vec_ld(30, src); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
241 srcM2 = vec_perm(srcR1, srcR2, permM2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
242 srcM1 = srcR2; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
243 srcP0 = vec_perm(srcR2, srcR3, permP0); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
244 srcP1 = vec_perm(srcR2, srcR3, permP1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
245 srcP2 = vec_perm(srcR2, srcR3, permP2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
246 srcP3 = vec_perm(srcR2, srcR3, permP3); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
247 } break; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
248 } |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
249 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
250 srcP0A = (vec_s16_t) vec_mergeh(zero_u8v, srcP0); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
251 srcP0B = (vec_s16_t) vec_mergel(zero_u8v, srcP0); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
252 srcP1A = (vec_s16_t) vec_mergeh(zero_u8v, srcP1); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
253 srcP1B = (vec_s16_t) vec_mergel(zero_u8v, srcP1); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
254 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
255 srcP2A = (vec_s16_t) vec_mergeh(zero_u8v, srcP2); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
256 srcP2B = (vec_s16_t) vec_mergel(zero_u8v, srcP2); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
257 srcP3A = (vec_s16_t) vec_mergeh(zero_u8v, srcP3); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
258 srcP3B = (vec_s16_t) vec_mergel(zero_u8v, srcP3); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
259 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
260 srcM1A = (vec_s16_t) vec_mergeh(zero_u8v, srcM1); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
261 srcM1B = (vec_s16_t) vec_mergel(zero_u8v, srcM1); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
262 srcM2A = (vec_s16_t) vec_mergeh(zero_u8v, srcM2); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
263 srcM2B = (vec_s16_t) vec_mergel(zero_u8v, srcM2); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
264 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
265 sum1A = vec_adds(srcP0A, srcP1A); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
266 sum1B = vec_adds(srcP0B, srcP1B); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
267 sum2A = vec_adds(srcM1A, srcP2A); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
268 sum2B = vec_adds(srcM1B, srcP2B); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
269 sum3A = vec_adds(srcM2A, srcP3A); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
270 sum3B = vec_adds(srcM2B, srcP3B); |
2967 | 271 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
272 pp1A = vec_mladd(sum1A, v20ss, v16ss); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
273 pp1B = vec_mladd(sum1B, v20ss, v16ss); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
274 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
275 pp2A = vec_mladd(sum2A, v5ss, zero_s16v); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
276 pp2B = vec_mladd(sum2B, v5ss, zero_s16v); |
2967 | 277 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
278 pp3A = vec_add(sum3A, pp1A); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
279 pp3B = vec_add(sum3B, pp1B); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
280 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
281 psumA = vec_sub(pp3A, pp2A); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
282 psumB = vec_sub(pp3B, pp2B); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
283 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
284 sumA = vec_sra(psumA, v5us); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
285 sumB = vec_sra(psumB, v5us); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
286 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
287 sum = vec_packsu(sumA, sumB); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
288 |
5603
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
289 ASSERT_ALIGNED(dst); |
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
290 vdst = vec_ld(0, dst); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
291 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
292 OP_U8_ALTIVEC(fsum, sum, vdst); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
293 |
5603
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
294 vec_st(fsum, 0, dst); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
295 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
296 src += srcStride; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
297 dst += dstStride; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
298 } |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
299 POWERPC_PERF_STOP_COUNT(PREFIX_h264_qpel16_h_lowpass_num, 1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
300 } |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
301 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
302 /* this code assume stride % 16 == 0 */ |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
303 static void PREFIX_h264_qpel16_v_lowpass_altivec(uint8_t * dst, uint8_t * src, int dstStride, int srcStride) { |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
304 POWERPC_PERF_DECLARE(PREFIX_h264_qpel16_v_lowpass_num, 1); |
2967 | 305 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
306 register int i; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
307 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
308 LOAD_ZERO; |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
309 const vec_u8_t perm = vec_lvsl(0, src); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
310 const vec_s16_t v20ss = vec_sl(vec_splat_s16(5),vec_splat_u16(2)); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
311 const vec_u16_t v5us = vec_splat_u16(5); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
312 const vec_s16_t v5ss = vec_splat_s16(5); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
313 const vec_s16_t v16ss = vec_sl(vec_splat_s16(1),vec_splat_u16(4)); |
2967 | 314 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
315 uint8_t *srcbis = src - (srcStride * 2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
316 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
317 const vec_u8_t srcM2a = vec_ld(0, srcbis); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
318 const vec_u8_t srcM2b = vec_ld(16, srcbis); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
319 const vec_u8_t srcM2 = vec_perm(srcM2a, srcM2b, perm); |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
320 // srcbis += srcStride; |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
321 const vec_u8_t srcM1a = vec_ld(0, srcbis += srcStride); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
322 const vec_u8_t srcM1b = vec_ld(16, srcbis); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
323 const vec_u8_t srcM1 = vec_perm(srcM1a, srcM1b, perm); |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
324 // srcbis += srcStride; |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
325 const vec_u8_t srcP0a = vec_ld(0, srcbis += srcStride); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
326 const vec_u8_t srcP0b = vec_ld(16, srcbis); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
327 const vec_u8_t srcP0 = vec_perm(srcP0a, srcP0b, perm); |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
328 // srcbis += srcStride; |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
329 const vec_u8_t srcP1a = vec_ld(0, srcbis += srcStride); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
330 const vec_u8_t srcP1b = vec_ld(16, srcbis); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
331 const vec_u8_t srcP1 = vec_perm(srcP1a, srcP1b, perm); |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
332 // srcbis += srcStride; |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
333 const vec_u8_t srcP2a = vec_ld(0, srcbis += srcStride); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
334 const vec_u8_t srcP2b = vec_ld(16, srcbis); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
335 const vec_u8_t srcP2 = vec_perm(srcP2a, srcP2b, perm); |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
336 // srcbis += srcStride; |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
337 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
338 vec_s16_t srcM2ssA = (vec_s16_t) vec_mergeh(zero_u8v, srcM2); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
339 vec_s16_t srcM2ssB = (vec_s16_t) vec_mergel(zero_u8v, srcM2); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
340 vec_s16_t srcM1ssA = (vec_s16_t) vec_mergeh(zero_u8v, srcM1); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
341 vec_s16_t srcM1ssB = (vec_s16_t) vec_mergel(zero_u8v, srcM1); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
342 vec_s16_t srcP0ssA = (vec_s16_t) vec_mergeh(zero_u8v, srcP0); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
343 vec_s16_t srcP0ssB = (vec_s16_t) vec_mergel(zero_u8v, srcP0); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
344 vec_s16_t srcP1ssA = (vec_s16_t) vec_mergeh(zero_u8v, srcP1); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
345 vec_s16_t srcP1ssB = (vec_s16_t) vec_mergel(zero_u8v, srcP1); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
346 vec_s16_t srcP2ssA = (vec_s16_t) vec_mergeh(zero_u8v, srcP2); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
347 vec_s16_t srcP2ssB = (vec_s16_t) vec_mergel(zero_u8v, srcP2); |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
348 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
349 vec_s16_t pp1A, pp1B, pp2A, pp2B, pp3A, pp3B, |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
350 psumA, psumB, sumA, sumB, |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
351 srcP3ssA, srcP3ssB, |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
352 sum1A, sum1B, sum2A, sum2B, sum3A, sum3B; |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
353 |
5603
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
354 vec_u8_t sum, vdst, fsum, srcP3a, srcP3b, srcP3; |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
355 |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
356 POWERPC_PERF_START_COUNT(PREFIX_h264_qpel16_v_lowpass_num, 1); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
357 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
358 for (i = 0 ; i < 16 ; i++) { |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
359 srcP3a = vec_ld(0, srcbis += srcStride); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
360 srcP3b = vec_ld(16, srcbis); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
361 srcP3 = vec_perm(srcP3a, srcP3b, perm); |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
362 srcP3ssA = (vec_s16_t) vec_mergeh(zero_u8v, srcP3); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
363 srcP3ssB = (vec_s16_t) vec_mergel(zero_u8v, srcP3); |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
364 // srcbis += srcStride; |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
365 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
366 sum1A = vec_adds(srcP0ssA, srcP1ssA); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
367 sum1B = vec_adds(srcP0ssB, srcP1ssB); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
368 sum2A = vec_adds(srcM1ssA, srcP2ssA); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
369 sum2B = vec_adds(srcM1ssB, srcP2ssB); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
370 sum3A = vec_adds(srcM2ssA, srcP3ssA); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
371 sum3B = vec_adds(srcM2ssB, srcP3ssB); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
372 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
373 srcM2ssA = srcM1ssA; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
374 srcM2ssB = srcM1ssB; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
375 srcM1ssA = srcP0ssA; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
376 srcM1ssB = srcP0ssB; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
377 srcP0ssA = srcP1ssA; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
378 srcP0ssB = srcP1ssB; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
379 srcP1ssA = srcP2ssA; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
380 srcP1ssB = srcP2ssB; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
381 srcP2ssA = srcP3ssA; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
382 srcP2ssB = srcP3ssB; |
2967 | 383 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
384 pp1A = vec_mladd(sum1A, v20ss, v16ss); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
385 pp1B = vec_mladd(sum1B, v20ss, v16ss); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
386 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
387 pp2A = vec_mladd(sum2A, v5ss, zero_s16v); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
388 pp2B = vec_mladd(sum2B, v5ss, zero_s16v); |
2967 | 389 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
390 pp3A = vec_add(sum3A, pp1A); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
391 pp3B = vec_add(sum3B, pp1B); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
392 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
393 psumA = vec_sub(pp3A, pp2A); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
394 psumB = vec_sub(pp3B, pp2B); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
395 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
396 sumA = vec_sra(psumA, v5us); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
397 sumB = vec_sra(psumB, v5us); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
398 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
399 sum = vec_packsu(sumA, sumB); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
400 |
5603
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
401 ASSERT_ALIGNED(dst); |
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
402 vdst = vec_ld(0, dst); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
403 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
404 OP_U8_ALTIVEC(fsum, sum, vdst); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
405 |
5603
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
406 vec_st(fsum, 0, dst); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
407 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
408 dst += dstStride; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
409 } |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
410 POWERPC_PERF_STOP_COUNT(PREFIX_h264_qpel16_v_lowpass_num, 1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
411 } |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
412 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
413 /* this code assume stride % 16 == 0 *and* tmp is properly aligned */ |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
414 static void PREFIX_h264_qpel16_hv_lowpass_altivec(uint8_t * dst, int16_t * tmp, uint8_t * src, int dstStride, int tmpStride, int srcStride) { |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
415 POWERPC_PERF_DECLARE(PREFIX_h264_qpel16_hv_lowpass_num, 1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
416 register int i; |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
417 LOAD_ZERO; |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
418 const vec_u8_t permM2 = vec_lvsl(-2, src); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
419 const vec_u8_t permM1 = vec_lvsl(-1, src); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
420 const vec_u8_t permP0 = vec_lvsl(+0, src); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
421 const vec_u8_t permP1 = vec_lvsl(+1, src); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
422 const vec_u8_t permP2 = vec_lvsl(+2, src); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
423 const vec_u8_t permP3 = vec_lvsl(+3, src); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
424 const vec_s16_t v20ss = vec_sl(vec_splat_s16(5),vec_splat_u16(2)); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
425 const vec_u32_t v10ui = vec_splat_u32(10); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
426 const vec_s16_t v5ss = vec_splat_s16(5); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
427 const vec_s16_t v1ss = vec_splat_s16(1); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
428 const vec_s32_t v512si = vec_sl(vec_splat_s32(1),vec_splat_u32(9)); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
429 const vec_u32_t v16ui = vec_sl(vec_splat_u32(1),vec_splat_u32(4)); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
430 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
431 register int align = ((((unsigned long)src) - 2) % 16); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
432 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
433 vec_s16_t srcP0A, srcP0B, srcP1A, srcP1B, |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
434 srcP2A, srcP2B, srcP3A, srcP3B, |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
435 srcM1A, srcM1B, srcM2A, srcM2B, |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
436 sum1A, sum1B, sum2A, sum2B, sum3A, sum3B, |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
437 pp1A, pp1B, pp2A, pp2B, psumA, psumB; |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
438 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
439 const vec_u8_t mperm = (const vec_u8_t) |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
440 AVV(0x00, 0x08, 0x01, 0x09, 0x02, 0x0A, 0x03, 0x0B, |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
441 0x04, 0x0C, 0x05, 0x0D, 0x06, 0x0E, 0x07, 0x0F); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
442 int16_t *tmpbis = tmp; |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
443 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
444 vec_s16_t tmpM1ssA, tmpM1ssB, tmpM2ssA, tmpM2ssB, |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
445 tmpP0ssA, tmpP0ssB, tmpP1ssA, tmpP1ssB, |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
446 tmpP2ssA, tmpP2ssB; |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
447 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
448 vec_s32_t pp1Ae, pp1Ao, pp1Be, pp1Bo, pp2Ae, pp2Ao, pp2Be, pp2Bo, |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
449 pp3Ae, pp3Ao, pp3Be, pp3Bo, pp1cAe, pp1cAo, pp1cBe, pp1cBo, |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
450 pp32Ae, pp32Ao, pp32Be, pp32Bo, sumAe, sumAo, sumBe, sumBo, |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
451 ssumAe, ssumAo, ssumBe, ssumBo; |
5603
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
452 vec_u8_t fsum, sumv, sum, vdst; |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
453 vec_s16_t ssume, ssumo; |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
454 |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
455 POWERPC_PERF_START_COUNT(PREFIX_h264_qpel16_hv_lowpass_num, 1); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
456 src -= (2 * srcStride); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
457 for (i = 0 ; i < 21 ; i ++) { |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
458 vec_u8_t srcM2, srcM1, srcP0, srcP1, srcP2, srcP3; |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
459 vec_u8_t srcR1 = vec_ld(-2, src); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
460 vec_u8_t srcR2 = vec_ld(14, src); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
461 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
462 switch (align) { |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
463 default: { |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
464 srcM2 = vec_perm(srcR1, srcR2, permM2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
465 srcM1 = vec_perm(srcR1, srcR2, permM1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
466 srcP0 = vec_perm(srcR1, srcR2, permP0); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
467 srcP1 = vec_perm(srcR1, srcR2, permP1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
468 srcP2 = vec_perm(srcR1, srcR2, permP2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
469 srcP3 = vec_perm(srcR1, srcR2, permP3); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
470 } break; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
471 case 11: { |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
472 srcM2 = vec_perm(srcR1, srcR2, permM2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
473 srcM1 = vec_perm(srcR1, srcR2, permM1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
474 srcP0 = vec_perm(srcR1, srcR2, permP0); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
475 srcP1 = vec_perm(srcR1, srcR2, permP1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
476 srcP2 = vec_perm(srcR1, srcR2, permP2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
477 srcP3 = srcR2; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
478 } break; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
479 case 12: { |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
480 vec_u8_t srcR3 = vec_ld(30, src); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
481 srcM2 = vec_perm(srcR1, srcR2, permM2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
482 srcM1 = vec_perm(srcR1, srcR2, permM1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
483 srcP0 = vec_perm(srcR1, srcR2, permP0); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
484 srcP1 = vec_perm(srcR1, srcR2, permP1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
485 srcP2 = srcR2; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
486 srcP3 = vec_perm(srcR2, srcR3, permP3); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
487 } break; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
488 case 13: { |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
489 vec_u8_t srcR3 = vec_ld(30, src); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
490 srcM2 = vec_perm(srcR1, srcR2, permM2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
491 srcM1 = vec_perm(srcR1, srcR2, permM1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
492 srcP0 = vec_perm(srcR1, srcR2, permP0); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
493 srcP1 = srcR2; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
494 srcP2 = vec_perm(srcR2, srcR3, permP2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
495 srcP3 = vec_perm(srcR2, srcR3, permP3); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
496 } break; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
497 case 14: { |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
498 vec_u8_t srcR3 = vec_ld(30, src); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
499 srcM2 = vec_perm(srcR1, srcR2, permM2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
500 srcM1 = vec_perm(srcR1, srcR2, permM1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
501 srcP0 = srcR2; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
502 srcP1 = vec_perm(srcR2, srcR3, permP1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
503 srcP2 = vec_perm(srcR2, srcR3, permP2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
504 srcP3 = vec_perm(srcR2, srcR3, permP3); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
505 } break; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
506 case 15: { |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
507 vec_u8_t srcR3 = vec_ld(30, src); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
508 srcM2 = vec_perm(srcR1, srcR2, permM2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
509 srcM1 = srcR2; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
510 srcP0 = vec_perm(srcR2, srcR3, permP0); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
511 srcP1 = vec_perm(srcR2, srcR3, permP1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
512 srcP2 = vec_perm(srcR2, srcR3, permP2); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
513 srcP3 = vec_perm(srcR2, srcR3, permP3); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
514 } break; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
515 } |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
516 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
517 srcP0A = (vec_s16_t) vec_mergeh(zero_u8v, srcP0); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
518 srcP0B = (vec_s16_t) vec_mergel(zero_u8v, srcP0); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
519 srcP1A = (vec_s16_t) vec_mergeh(zero_u8v, srcP1); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
520 srcP1B = (vec_s16_t) vec_mergel(zero_u8v, srcP1); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
521 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
522 srcP2A = (vec_s16_t) vec_mergeh(zero_u8v, srcP2); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
523 srcP2B = (vec_s16_t) vec_mergel(zero_u8v, srcP2); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
524 srcP3A = (vec_s16_t) vec_mergeh(zero_u8v, srcP3); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
525 srcP3B = (vec_s16_t) vec_mergel(zero_u8v, srcP3); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
526 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
527 srcM1A = (vec_s16_t) vec_mergeh(zero_u8v, srcM1); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
528 srcM1B = (vec_s16_t) vec_mergel(zero_u8v, srcM1); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
529 srcM2A = (vec_s16_t) vec_mergeh(zero_u8v, srcM2); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
530 srcM2B = (vec_s16_t) vec_mergel(zero_u8v, srcM2); |
2967 | 531 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
532 sum1A = vec_adds(srcP0A, srcP1A); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
533 sum1B = vec_adds(srcP0B, srcP1B); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
534 sum2A = vec_adds(srcM1A, srcP2A); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
535 sum2B = vec_adds(srcM1B, srcP2B); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
536 sum3A = vec_adds(srcM2A, srcP3A); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
537 sum3B = vec_adds(srcM2B, srcP3B); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
538 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
539 pp1A = vec_mladd(sum1A, v20ss, sum3A); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
540 pp1B = vec_mladd(sum1B, v20ss, sum3B); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
541 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
542 pp2A = vec_mladd(sum2A, v5ss, zero_s16v); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
543 pp2B = vec_mladd(sum2B, v5ss, zero_s16v); |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
544 |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
545 psumA = vec_sub(pp1A, pp2A); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
546 psumB = vec_sub(pp1B, pp2B); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
547 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
548 vec_st(psumA, 0, tmp); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
549 vec_st(psumB, 16, tmp); |
2967 | 550 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
551 src += srcStride; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
552 tmp += tmpStride; /* int16_t*, and stride is 16, so it's OK here */ |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
553 } |
2967 | 554 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
555 tmpM2ssA = vec_ld(0, tmpbis); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
556 tmpM2ssB = vec_ld(16, tmpbis); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
557 tmpbis += tmpStride; |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
558 tmpM1ssA = vec_ld(0, tmpbis); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
559 tmpM1ssB = vec_ld(16, tmpbis); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
560 tmpbis += tmpStride; |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
561 tmpP0ssA = vec_ld(0, tmpbis); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
562 tmpP0ssB = vec_ld(16, tmpbis); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
563 tmpbis += tmpStride; |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
564 tmpP1ssA = vec_ld(0, tmpbis); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
565 tmpP1ssB = vec_ld(16, tmpbis); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
566 tmpbis += tmpStride; |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
567 tmpP2ssA = vec_ld(0, tmpbis); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
568 tmpP2ssB = vec_ld(16, tmpbis); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
569 tmpbis += tmpStride; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
570 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
571 for (i = 0 ; i < 16 ; i++) { |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
572 const vec_s16_t tmpP3ssA = vec_ld(0, tmpbis); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
573 const vec_s16_t tmpP3ssB = vec_ld(16, tmpbis); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
574 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
575 const vec_s16_t sum1A = vec_adds(tmpP0ssA, tmpP1ssA); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
576 const vec_s16_t sum1B = vec_adds(tmpP0ssB, tmpP1ssB); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
577 const vec_s16_t sum2A = vec_adds(tmpM1ssA, tmpP2ssA); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
578 const vec_s16_t sum2B = vec_adds(tmpM1ssB, tmpP2ssB); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
579 const vec_s16_t sum3A = vec_adds(tmpM2ssA, tmpP3ssA); |
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
580 const vec_s16_t sum3B = vec_adds(tmpM2ssB, tmpP3ssB); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
581 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
582 tmpbis += tmpStride; |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
583 |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
584 tmpM2ssA = tmpM1ssA; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
585 tmpM2ssB = tmpM1ssB; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
586 tmpM1ssA = tmpP0ssA; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
587 tmpM1ssB = tmpP0ssB; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
588 tmpP0ssA = tmpP1ssA; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
589 tmpP0ssB = tmpP1ssB; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
590 tmpP1ssA = tmpP2ssA; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
591 tmpP1ssB = tmpP2ssB; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
592 tmpP2ssA = tmpP3ssA; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
593 tmpP2ssB = tmpP3ssB; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
594 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
595 pp1Ae = vec_mule(sum1A, v20ss); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
596 pp1Ao = vec_mulo(sum1A, v20ss); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
597 pp1Be = vec_mule(sum1B, v20ss); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
598 pp1Bo = vec_mulo(sum1B, v20ss); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
599 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
600 pp2Ae = vec_mule(sum2A, v5ss); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
601 pp2Ao = vec_mulo(sum2A, v5ss); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
602 pp2Be = vec_mule(sum2B, v5ss); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
603 pp2Bo = vec_mulo(sum2B, v5ss); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
604 |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
605 pp3Ae = vec_sra((vec_s32_t)sum3A, v16ui); |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
606 pp3Ao = vec_mulo(sum3A, v1ss); |
5530
cd266411b11a
use shorter types vec_"type" instead of the too long vector "type"
gpoirier
parents:
5019
diff
changeset
|
607 pp3Be = vec_sra((vec_s32_t)sum3B, v16ui); |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
608 pp3Bo = vec_mulo(sum3B, v1ss); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
609 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
610 pp1cAe = vec_add(pp1Ae, v512si); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
611 pp1cAo = vec_add(pp1Ao, v512si); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
612 pp1cBe = vec_add(pp1Be, v512si); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
613 pp1cBo = vec_add(pp1Bo, v512si); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
614 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
615 pp32Ae = vec_sub(pp3Ae, pp2Ae); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
616 pp32Ao = vec_sub(pp3Ao, pp2Ao); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
617 pp32Be = vec_sub(pp3Be, pp2Be); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
618 pp32Bo = vec_sub(pp3Bo, pp2Bo); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
619 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
620 sumAe = vec_add(pp1cAe, pp32Ae); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
621 sumAo = vec_add(pp1cAo, pp32Ao); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
622 sumBe = vec_add(pp1cBe, pp32Be); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
623 sumBo = vec_add(pp1cBo, pp32Bo); |
2967 | 624 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
625 ssumAe = vec_sra(sumAe, v10ui); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
626 ssumAo = vec_sra(sumAo, v10ui); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
627 ssumBe = vec_sra(sumBe, v10ui); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
628 ssumBo = vec_sra(sumBo, v10ui); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
629 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
630 ssume = vec_packs(ssumAe, ssumBe); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
631 ssumo = vec_packs(ssumAo, ssumBo); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
632 |
3346
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
633 sumv = vec_packsu(ssume, ssumo); |
052765f11f1c
Cosmetics: should not hurt performance, scream if are
lu_zero
parents:
3153
diff
changeset
|
634 sum = vec_perm(sumv, sumv, mperm); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
635 |
5603
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
636 ASSERT_ALIGNED(dst); |
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
637 vdst = vec_ld(0, dst); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
638 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
639 OP_U8_ALTIVEC(fsum, sum, vdst); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
640 |
5603
861eb234e6ba
remove alignment correction of the destination pointers in luma_16x6
gpoirier
parents:
5530
diff
changeset
|
641 vec_st(fsum, 0, dst); |
2236
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
642 |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
643 dst += dstStride; |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
644 } |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
645 POWERPC_PERF_STOP_COUNT(PREFIX_h264_qpel16_hv_lowpass_num, 1); |
b0102ea621dd
h264 qpel mc, size 16 patch by (Romain Dolbeau <dolbeau at caps-entreprise dot com>)
michael
parents:
diff
changeset
|
646 } |