Mercurial > mplayer.hg
annotate liba52/imdct.c @ 23572:a00685941686
demux_mkv very long seek fix
The seek code searching for the closest position in the index used
"int64_t min_diff=0xFFFFFFFL" as the initial "further from the goal
than any real alternative" value. The unit is milliseconds so seeks more
than about 75 hours past the end of the file would fail to recognize the
last index position as the best match. This was triggered in practice by
chapter seek code which apparently uses a seek of 1000000000 seconds
forward to mean "seek to the end". The practical effect was that trying
to seek to the next chapter in a file without chapters made MPlayer
block until it finished reading the file from the current position to
the end.
Fixed by increasing the initial value from FFFFFFF to FFFFFFFFFFFFFFF.
author | uau |
---|---|
date | Wed, 20 Jun 2007 18:19:03 +0000 |
parents | 6334c14b38eb |
children | a7b716b53e9f |
rev | line source |
---|---|
3394 | 1 /* |
2 * imdct.c | |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
3 * Copyright (C) 2000-2002 Michel Lespinasse <walken@zoy.org> |
3394 | 4 * Copyright (C) 1999-2000 Aaron Holtzman <aholtzma@ess.engr.uvic.ca> |
5 * | |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
6 * The ifft algorithms in this file have been largely inspired by Dan |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
7 * Bernstein's work, djbfft, available at http://cr.yp.to/djbfft.html |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
8 * |
3394 | 9 * This file is part of a52dec, a free ATSC A-52 stream decoder. |
10 * See http://liba52.sourceforge.net/ for updates. | |
11 * | |
14991
07f1e7669772
Mark modified files as such to comply more closely with GPL ¡ø2a.
diego
parents:
12303
diff
changeset
|
12 * Modified for use with MPlayer, changes contained in liba52_changes.diff. |
18783 | 13 * detailed changelog at http://svn.mplayerhq.hu/mplayer/trunk/ |
14991
07f1e7669772
Mark modified files as such to comply more closely with GPL ¡ø2a.
diego
parents:
12303
diff
changeset
|
14 * $Id$ |
07f1e7669772
Mark modified files as such to comply more closely with GPL ¡ø2a.
diego
parents:
12303
diff
changeset
|
15 * |
3394 | 16 * a52dec is free software; you can redistribute it and/or modify |
17 * it under the terms of the GNU General Public License as published by | |
18 * the Free Software Foundation; either version 2 of the License, or | |
19 * (at your option) any later version. | |
20 * | |
21 * a52dec is distributed in the hope that it will be useful, | |
22 * but WITHOUT ANY WARRANTY; without even the implied warranty of | |
23 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the | |
24 * GNU General Public License for more details. | |
25 * | |
26 * You should have received a copy of the GNU General Public License | |
27 * along with this program; if not, write to the Free Software | |
28 * Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA 02111-1307 USA | |
3579 | 29 * |
30 * SSE optimizations from Michael Niedermayer (michaelni@gmx.at) | |
3884 | 31 * 3DNOW optimizations from Nick Kurshev <nickols_k@mail.ru> |
32 * michael did port them from libac3 (untested, perhaps totally broken) | |
9001
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
33 * AltiVec optimizations from Romain Dolbeau (romain@dolbeau.org) |
3394 | 34 */ |
35 | |
36 #include "config.h" | |
37 | |
38 #include <math.h> | |
39 #include <stdio.h> | |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
40 #ifdef LIBA52_DJBFFT |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
41 #include <fftc4.h> |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
42 #endif |
3394 | 43 #ifndef M_PI |
44 #define M_PI 3.1415926535897932384626433832795029 | |
45 #endif | |
46 #include <inttypes.h> | |
47 | |
48 #include "a52.h" | |
49 #include "a52_internal.h" | |
50 #include "mm_accel.h" | |
4247
2dbd637ffe05
mangle for win32 in liba52 (includes dummy mangle.h pointing to the one in main)
atmos4
parents:
3908
diff
changeset
|
51 #include "mangle.h" |
3394 | 52 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
53 void (*a52_imdct_512) (sample_t * data, sample_t * delay, sample_t bias); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
54 |
3884 | 55 #ifdef RUNTIME_CPUDETECT |
56 #undef HAVE_3DNOWEX | |
57 #endif | |
58 | |
3394 | 59 typedef struct complex_s { |
60 sample_t real; | |
61 sample_t imag; | |
62 } complex_t; | |
63 | |
12303
f881c918739b
attribute_used patch by (VMiklos <mamajom at axelero dot hu>)
michael
parents:
9122
diff
changeset
|
64 static const int pm128[128] attribute_used __attribute__((aligned(16))) = |
3884 | 65 { |
66 0, 16, 32, 48, 64, 80, 96, 112, 8, 40, 72, 104, 24, 56, 88, 120, | |
67 4, 20, 36, 52, 68, 84, 100, 116, 12, 28, 44, 60, 76, 92, 108, 124, | |
68 2, 18, 34, 50, 66, 82, 98, 114, 10, 42, 74, 106, 26, 58, 90, 122, | |
69 6, 22, 38, 54, 70, 86, 102, 118, 14, 46, 78, 110, 30, 62, 94, 126, | |
70 1, 17, 33, 49, 65, 81, 97, 113, 9, 41, 73, 105, 25, 57, 89, 121, | |
71 5, 21, 37, 53, 69, 85, 101, 117, 13, 29, 45, 61, 77, 93, 109, 125, | |
72 3, 19, 35, 51, 67, 83, 99, 115, 11, 43, 75, 107, 27, 59, 91, 123, | |
73 7, 23, 39, 55, 71, 87, 103, 119, 15, 31, 47, 63, 79, 95, 111, 127 | |
74 }; | |
3394 | 75 |
12303
f881c918739b
attribute_used patch by (VMiklos <mamajom at axelero dot hu>)
michael
parents:
9122
diff
changeset
|
76 static uint8_t attribute_used bit_reverse_512[] = { |
3394 | 77 0x00, 0x40, 0x20, 0x60, 0x10, 0x50, 0x30, 0x70, |
78 0x08, 0x48, 0x28, 0x68, 0x18, 0x58, 0x38, 0x78, | |
79 0x04, 0x44, 0x24, 0x64, 0x14, 0x54, 0x34, 0x74, | |
80 0x0c, 0x4c, 0x2c, 0x6c, 0x1c, 0x5c, 0x3c, 0x7c, | |
81 0x02, 0x42, 0x22, 0x62, 0x12, 0x52, 0x32, 0x72, | |
82 0x0a, 0x4a, 0x2a, 0x6a, 0x1a, 0x5a, 0x3a, 0x7a, | |
83 0x06, 0x46, 0x26, 0x66, 0x16, 0x56, 0x36, 0x76, | |
84 0x0e, 0x4e, 0x2e, 0x6e, 0x1e, 0x5e, 0x3e, 0x7e, | |
85 0x01, 0x41, 0x21, 0x61, 0x11, 0x51, 0x31, 0x71, | |
86 0x09, 0x49, 0x29, 0x69, 0x19, 0x59, 0x39, 0x79, | |
87 0x05, 0x45, 0x25, 0x65, 0x15, 0x55, 0x35, 0x75, | |
88 0x0d, 0x4d, 0x2d, 0x6d, 0x1d, 0x5d, 0x3d, 0x7d, | |
89 0x03, 0x43, 0x23, 0x63, 0x13, 0x53, 0x33, 0x73, | |
90 0x0b, 0x4b, 0x2b, 0x6b, 0x1b, 0x5b, 0x3b, 0x7b, | |
91 0x07, 0x47, 0x27, 0x67, 0x17, 0x57, 0x37, 0x77, | |
92 0x0f, 0x4f, 0x2f, 0x6f, 0x1f, 0x5f, 0x3f, 0x7f}; | |
93 | |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
94 static uint8_t fftorder[] = { |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
95 0,128, 64,192, 32,160,224, 96, 16,144, 80,208,240,112, 48,176, |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
96 8,136, 72,200, 40,168,232,104,248,120, 56,184, 24,152,216, 88, |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
97 4,132, 68,196, 36,164,228,100, 20,148, 84,212,244,116, 52,180, |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
98 252,124, 60,188, 28,156,220, 92, 12,140, 76,204,236,108, 44,172, |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
99 2,130, 66,194, 34,162,226, 98, 18,146, 82,210,242,114, 50,178, |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
100 10,138, 74,202, 42,170,234,106,250,122, 58,186, 26,154,218, 90, |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
101 254,126, 62,190, 30,158,222, 94, 14,142, 78,206,238,110, 46,174, |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
102 6,134, 70,198, 38,166,230,102,246,118, 54,182, 22,150,214, 86 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
103 }; |
3394 | 104 |
3508 | 105 static complex_t __attribute__((aligned(16))) buf[128]; |
3394 | 106 |
107 /* Twiddle factor LUT */ | |
9001
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
108 static complex_t __attribute__((aligned(16))) w_1[1]; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
109 static complex_t __attribute__((aligned(16))) w_2[2]; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
110 static complex_t __attribute__((aligned(16))) w_4[4]; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
111 static complex_t __attribute__((aligned(16))) w_8[8]; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
112 static complex_t __attribute__((aligned(16))) w_16[16]; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
113 static complex_t __attribute__((aligned(16))) w_32[32]; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
114 static complex_t __attribute__((aligned(16))) w_64[64]; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
115 static complex_t __attribute__((aligned(16))) * w[7] = {w_1, w_2, w_4, w_8, w_16, w_32, w_64}; |
3394 | 116 |
117 /* Twiddle factors for IMDCT */ | |
9001
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
118 static sample_t __attribute__((aligned(16))) xcos1[128]; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
119 static sample_t __attribute__((aligned(16))) xsin1[128]; |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
120 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
121 #if defined(ARCH_X86) || defined(ARCH_X86_64) |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
122 // NOTE: SSE needs 16byte alignment or it will segfault |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
123 // |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
124 static float __attribute__((aligned(16))) sseSinCos1c[256]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
125 static float __attribute__((aligned(16))) sseSinCos1d[256]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
126 static float attribute_used __attribute__((aligned(16))) ps111_1[4]={1,1,1,-1}; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
127 //static float __attribute__((aligned(16))) sseW0[4]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
128 static float __attribute__((aligned(16))) sseW1[8]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
129 static float __attribute__((aligned(16))) sseW2[16]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
130 static float __attribute__((aligned(16))) sseW3[32]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
131 static float __attribute__((aligned(16))) sseW4[64]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
132 static float __attribute__((aligned(16))) sseW5[128]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
133 static float __attribute__((aligned(16))) sseW6[256]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
134 static float __attribute__((aligned(16))) *sseW[7]= |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
135 {NULL /*sseW0*/,sseW1,sseW2,sseW3,sseW4,sseW5,sseW6}; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
136 static float __attribute__((aligned(16))) sseWindow[512]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
137 #endif |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
138 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
139 /* Root values for IFFT */ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
140 static sample_t roots16[3]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
141 static sample_t roots32[7]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
142 static sample_t roots64[15]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
143 static sample_t roots128[31]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
144 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
145 /* Twiddle factors for IMDCT */ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
146 static complex_t pre1[128]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
147 static complex_t post1[64]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
148 static complex_t pre2[64]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
149 static complex_t post2[32]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
150 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
151 static sample_t a52_imdct_window[256]; |
3394 | 152 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
153 static void (* ifft128) (complex_t * buf); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
154 static void (* ifft64) (complex_t * buf); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
155 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
156 static inline void ifft2 (complex_t * buf) |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
157 { |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
158 double r, i; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
159 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
160 r = buf[0].real; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
161 i = buf[0].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
162 buf[0].real += buf[1].real; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
163 buf[0].imag += buf[1].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
164 buf[1].real = r - buf[1].real; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
165 buf[1].imag = i - buf[1].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
166 } |
3394 | 167 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
168 static inline void ifft4 (complex_t * buf) |
3394 | 169 { |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
170 double tmp1, tmp2, tmp3, tmp4, tmp5, tmp6, tmp7, tmp8; |
3394 | 171 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
172 tmp1 = buf[0].real + buf[1].real; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
173 tmp2 = buf[3].real + buf[2].real; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
174 tmp3 = buf[0].imag + buf[1].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
175 tmp4 = buf[2].imag + buf[3].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
176 tmp5 = buf[0].real - buf[1].real; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
177 tmp6 = buf[0].imag - buf[1].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
178 tmp7 = buf[2].imag - buf[3].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
179 tmp8 = buf[3].real - buf[2].real; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
180 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
181 buf[0].real = tmp1 + tmp2; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
182 buf[0].imag = tmp3 + tmp4; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
183 buf[2].real = tmp1 - tmp2; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
184 buf[2].imag = tmp3 - tmp4; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
185 buf[1].real = tmp5 + tmp7; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
186 buf[1].imag = tmp6 + tmp8; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
187 buf[3].real = tmp5 - tmp7; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
188 buf[3].imag = tmp6 - tmp8; |
3394 | 189 } |
190 | |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
191 /* the basic split-radix ifft butterfly */ |
3394 | 192 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
193 #define BUTTERFLY(a0,a1,a2,a3,wr,wi) do { \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
194 tmp5 = a2.real * wr + a2.imag * wi; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
195 tmp6 = a2.imag * wr - a2.real * wi; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
196 tmp7 = a3.real * wr - a3.imag * wi; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
197 tmp8 = a3.imag * wr + a3.real * wi; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
198 tmp1 = tmp5 + tmp7; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
199 tmp2 = tmp6 + tmp8; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
200 tmp3 = tmp6 - tmp8; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
201 tmp4 = tmp7 - tmp5; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
202 a2.real = a0.real - tmp1; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
203 a2.imag = a0.imag - tmp2; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
204 a3.real = a1.real - tmp3; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
205 a3.imag = a1.imag - tmp4; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
206 a0.real += tmp1; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
207 a0.imag += tmp2; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
208 a1.real += tmp3; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
209 a1.imag += tmp4; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
210 } while (0) |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
211 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
212 /* split-radix ifft butterfly, specialized for wr=1 wi=0 */ |
3394 | 213 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
214 #define BUTTERFLY_ZERO(a0,a1,a2,a3) do { \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
215 tmp1 = a2.real + a3.real; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
216 tmp2 = a2.imag + a3.imag; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
217 tmp3 = a2.imag - a3.imag; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
218 tmp4 = a3.real - a2.real; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
219 a2.real = a0.real - tmp1; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
220 a2.imag = a0.imag - tmp2; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
221 a3.real = a1.real - tmp3; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
222 a3.imag = a1.imag - tmp4; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
223 a0.real += tmp1; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
224 a0.imag += tmp2; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
225 a1.real += tmp3; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
226 a1.imag += tmp4; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
227 } while (0) |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
228 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
229 /* split-radix ifft butterfly, specialized for wr=wi */ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
230 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
231 #define BUTTERFLY_HALF(a0,a1,a2,a3,w) do { \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
232 tmp5 = (a2.real + a2.imag) * w; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
233 tmp6 = (a2.imag - a2.real) * w; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
234 tmp7 = (a3.real - a3.imag) * w; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
235 tmp8 = (a3.imag + a3.real) * w; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
236 tmp1 = tmp5 + tmp7; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
237 tmp2 = tmp6 + tmp8; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
238 tmp3 = tmp6 - tmp8; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
239 tmp4 = tmp7 - tmp5; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
240 a2.real = a0.real - tmp1; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
241 a2.imag = a0.imag - tmp2; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
242 a3.real = a1.real - tmp3; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
243 a3.imag = a1.imag - tmp4; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
244 a0.real += tmp1; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
245 a0.imag += tmp2; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
246 a1.real += tmp3; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
247 a1.imag += tmp4; \ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
248 } while (0) |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
249 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
250 static inline void ifft8 (complex_t * buf) |
3394 | 251 { |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
252 double tmp1, tmp2, tmp3, tmp4, tmp5, tmp6, tmp7, tmp8; |
3394 | 253 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
254 ifft4 (buf); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
255 ifft2 (buf + 4); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
256 ifft2 (buf + 6); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
257 BUTTERFLY_ZERO (buf[0], buf[2], buf[4], buf[6]); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
258 BUTTERFLY_HALF (buf[1], buf[3], buf[5], buf[7], roots16[1]); |
3394 | 259 } |
260 | |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
261 static void ifft_pass (complex_t * buf, sample_t * weight, int n) |
3394 | 262 { |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
263 complex_t * buf1; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
264 complex_t * buf2; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
265 complex_t * buf3; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
266 double tmp1, tmp2, tmp3, tmp4, tmp5, tmp6, tmp7, tmp8; |
8254
772d6d27fd66
warning patch by (Dominik Mierzejewski <dominik at rangers dot eu dot org>)
michael
parents:
4497
diff
changeset
|
267 int i; |
3394 | 268 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
269 buf++; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
270 buf1 = buf + n; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
271 buf2 = buf + 2 * n; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
272 buf3 = buf + 3 * n; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
273 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
274 BUTTERFLY_ZERO (buf[-1], buf1[-1], buf2[-1], buf3[-1]); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
275 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
276 i = n - 1; |
3394 | 277 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
278 do { |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
279 BUTTERFLY (buf[0], buf1[0], buf2[0], buf3[0], weight[n], weight[2*i]); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
280 buf++; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
281 buf1++; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
282 buf2++; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
283 buf3++; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
284 weight++; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
285 } while (--i); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
286 } |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
287 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
288 static void ifft16 (complex_t * buf) |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
289 { |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
290 ifft8 (buf); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
291 ifft4 (buf + 8); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
292 ifft4 (buf + 12); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
293 ifft_pass (buf, roots16 - 4, 4); |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
294 } |
3394 | 295 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
296 static void ifft32 (complex_t * buf) |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
297 { |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
298 ifft16 (buf); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
299 ifft8 (buf + 16); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
300 ifft8 (buf + 24); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
301 ifft_pass (buf, roots32 - 8, 8); |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
302 } |
3579 | 303 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
304 static void ifft64_c (complex_t * buf) |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
305 { |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
306 ifft32 (buf); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
307 ifft16 (buf + 32); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
308 ifft16 (buf + 48); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
309 ifft_pass (buf, roots64 - 16, 16); |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
310 } |
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
311 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
312 static void ifft128_c (complex_t * buf) |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
313 { |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
314 ifft32 (buf); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
315 ifft16 (buf + 32); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
316 ifft16 (buf + 48); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
317 ifft_pass (buf, roots64 - 16, 16); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
318 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
319 ifft32 (buf + 64); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
320 ifft32 (buf + 96); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
321 ifft_pass (buf, roots128 - 32, 32); |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
322 } |
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
323 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
324 void imdct_do_512 (sample_t * data, sample_t * delay, sample_t bias) |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
325 { |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
326 int i, k; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
327 sample_t t_r, t_i, a_r, a_i, b_r, b_i, w_1, w_2; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
328 const sample_t * window = a52_imdct_window; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
329 complex_t buf[128]; |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
330 |
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
331 for (i = 0; i < 128; i++) { |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
332 k = fftorder[i]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
333 t_r = pre1[i].real; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
334 t_i = pre1[i].imag; |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
335 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
336 buf[i].real = t_i * data[255-k] + t_r * data[k]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
337 buf[i].imag = t_r * data[255-k] - t_i * data[k]; |
3579 | 338 } |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
339 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
340 ifft128 (buf); |
3579 | 341 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
342 /* Post IFFT complex multiply plus IFFT complex conjugate*/ |
3579 | 343 /* Window and convert to real valued signal */ |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
344 for (i = 0; i < 64; i++) { |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
345 /* y[n] = z[n] * (xcos1[n] + j * xsin1[n]) ; */ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
346 t_r = post1[i].real; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
347 t_i = post1[i].imag; |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
348 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
349 a_r = t_r * buf[i].real + t_i * buf[i].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
350 a_i = t_i * buf[i].real - t_r * buf[i].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
351 b_r = t_i * buf[127-i].real + t_r * buf[127-i].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
352 b_i = t_r * buf[127-i].real - t_i * buf[127-i].imag; |
3579 | 353 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
354 w_1 = window[2*i]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
355 w_2 = window[255-2*i]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
356 data[2*i] = delay[2*i] * w_2 - a_r * w_1 + bias; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
357 data[255-2*i] = delay[2*i] * w_1 + a_r * w_2 + bias; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
358 delay[2*i] = a_i; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
359 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
360 w_1 = window[2*i+1]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
361 w_2 = window[254-2*i]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
362 data[2*i+1] = delay[2*i+1] * w_2 + b_r * w_1 + bias; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
363 data[254-2*i] = delay[2*i+1] * w_1 - b_r * w_2 + bias; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
364 delay[2*i+1] = b_i; |
3579 | 365 } |
366 } | |
367 | |
9001
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
368 #ifdef HAVE_ALTIVEC |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
369 |
9122 | 370 #ifndef SYS_DARWIN |
371 #include <altivec.h> | |
372 #endif | |
373 | |
9001
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
374 // used to build registers permutation vectors (vcprm) |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
375 // the 's' are for words in the _s_econd vector |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
376 #define WORD_0 0x00,0x01,0x02,0x03 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
377 #define WORD_1 0x04,0x05,0x06,0x07 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
378 #define WORD_2 0x08,0x09,0x0a,0x0b |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
379 #define WORD_3 0x0c,0x0d,0x0e,0x0f |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
380 #define WORD_s0 0x10,0x11,0x12,0x13 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
381 #define WORD_s1 0x14,0x15,0x16,0x17 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
382 #define WORD_s2 0x18,0x19,0x1a,0x1b |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
383 #define WORD_s3 0x1c,0x1d,0x1e,0x1f |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
384 |
9122 | 385 #ifdef SYS_DARWIN |
9001
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
386 #define vcprm(a,b,c,d) (const vector unsigned char)(WORD_ ## a, WORD_ ## b, WORD_ ## c, WORD_ ## d) |
9122 | 387 #else |
388 #define vcprm(a,b,c,d) (const vector unsigned char){WORD_ ## a, WORD_ ## b, WORD_ ## c, WORD_ ## d} | |
389 #endif | |
9001
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
390 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
391 // vcprmle is used to keep the same index as in the SSE version. |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
392 // it's the same as vcprm, with the index inversed |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
393 // ('le' is Little Endian) |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
394 #define vcprmle(a,b,c,d) vcprm(d,c,b,a) |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
395 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
396 // used to build inverse/identity vectors (vcii) |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
397 // n is _n_egative, p is _p_ositive |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
398 #define FLOAT_n -1. |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
399 #define FLOAT_p 1. |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
400 |
9122 | 401 #ifdef SYS_DARWIN |
9001
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
402 #define vcii(a,b,c,d) (const vector float)(FLOAT_ ## a, FLOAT_ ## b, FLOAT_ ## c, FLOAT_ ## d) |
9122 | 403 #else |
404 #define vcii(a,b,c,d) (const vector float){FLOAT_ ## a, FLOAT_ ## b, FLOAT_ ## c, FLOAT_ ## d} | |
405 #endif | |
406 | |
407 #ifdef SYS_DARWIN | |
408 #define FOUROF(a) (a) | |
409 #else | |
410 #define FOUROF(a) {a,a,a,a} | |
411 #endif | |
412 | |
9001
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
413 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
414 void |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
415 imdct_do_512_altivec(sample_t data[],sample_t delay[], sample_t bias) |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
416 { |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
417 int i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
418 int k; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
419 int p,q; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
420 int m; |
16173 | 421 long two_m; |
422 long two_m_plus_one; | |
9001
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
423 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
424 sample_t tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
425 sample_t tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
426 sample_t tmp_a_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
427 sample_t tmp_a_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
428 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
429 sample_t *data_ptr; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
430 sample_t *delay_ptr; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
431 sample_t *window_ptr; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
432 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
433 /* 512 IMDCT with source and dest data in 'data' */ |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
434 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
435 /* Pre IFFT complex multiply plus IFFT cmplx conjugate & reordering*/ |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
436 for( i=0; i < 128; i++) { |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
437 /* z[i] = (X[256-2*i-1] + j * X[2*i]) * (xcos1[i] + j * xsin1[i]) ; */ |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
438 int j= bit_reverse_512[i]; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
439 buf[i].real = (data[256-2*j-1] * xcos1[j]) - (data[2*j] * xsin1[j]); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
440 buf[i].imag = -1.0 * ((data[2*j] * xcos1[j]) + (data[256-2*j-1] * xsin1[j])); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
441 } |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
442 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
443 /* 1. iteration */ |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
444 for(i = 0; i < 128; i += 2) { |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
445 #if 0 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
446 tmp_a_r = buf[i].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
447 tmp_a_i = buf[i].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
448 tmp_b_r = buf[i+1].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
449 tmp_b_i = buf[i+1].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
450 buf[i].real = tmp_a_r + tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
451 buf[i].imag = tmp_a_i + tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
452 buf[i+1].real = tmp_a_r - tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
453 buf[i+1].imag = tmp_a_i - tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
454 #else |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
455 vector float temp, bufv; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
456 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
457 bufv = vec_ld(i << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
458 temp = vec_perm(bufv, bufv, vcprm(2,3,0,1)); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
459 bufv = vec_madd(bufv, vcii(p,p,n,n), temp); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
460 vec_st(bufv, i << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
461 #endif |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
462 } |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
463 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
464 /* 2. iteration */ |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
465 // Note w[1]={{1,0}, {0,-1}} |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
466 for(i = 0; i < 128; i += 4) { |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
467 #if 0 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
468 tmp_a_r = buf[i].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
469 tmp_a_i = buf[i].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
470 tmp_b_r = buf[i+2].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
471 tmp_b_i = buf[i+2].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
472 buf[i].real = tmp_a_r + tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
473 buf[i].imag = tmp_a_i + tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
474 buf[i+2].real = tmp_a_r - tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
475 buf[i+2].imag = tmp_a_i - tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
476 tmp_a_r = buf[i+1].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
477 tmp_a_i = buf[i+1].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
478 /* WARNING: im <-> re here ! */ |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
479 tmp_b_r = buf[i+3].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
480 tmp_b_i = buf[i+3].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
481 buf[i+1].real = tmp_a_r + tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
482 buf[i+1].imag = tmp_a_i - tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
483 buf[i+3].real = tmp_a_r - tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
484 buf[i+3].imag = tmp_a_i + tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
485 #else |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
486 vector float buf01, buf23, temp1, temp2; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
487 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
488 buf01 = vec_ld((i + 0) << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
489 buf23 = vec_ld((i + 2) << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
490 buf23 = vec_perm(buf23,buf23,vcprm(0,1,3,2)); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
491 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
492 temp1 = vec_madd(buf23, vcii(p,p,p,n), buf01); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
493 temp2 = vec_madd(buf23, vcii(n,n,n,p), buf01); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
494 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
495 vec_st(temp1, (i + 0) << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
496 vec_st(temp2, (i + 2) << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
497 #endif |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
498 } |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
499 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
500 /* 3. iteration */ |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
501 for(i = 0; i < 128; i += 8) { |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
502 #if 0 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
503 tmp_a_r = buf[i].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
504 tmp_a_i = buf[i].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
505 tmp_b_r = buf[i+4].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
506 tmp_b_i = buf[i+4].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
507 buf[i].real = tmp_a_r + tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
508 buf[i].imag = tmp_a_i + tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
509 buf[i+4].real = tmp_a_r - tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
510 buf[i+4].imag = tmp_a_i - tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
511 tmp_a_r = buf[1+i].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
512 tmp_a_i = buf[1+i].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
513 tmp_b_r = (buf[i+5].real + buf[i+5].imag) * w[2][1].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
514 tmp_b_i = (buf[i+5].imag - buf[i+5].real) * w[2][1].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
515 buf[1+i].real = tmp_a_r + tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
516 buf[1+i].imag = tmp_a_i + tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
517 buf[i+5].real = tmp_a_r - tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
518 buf[i+5].imag = tmp_a_i - tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
519 tmp_a_r = buf[i+2].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
520 tmp_a_i = buf[i+2].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
521 /* WARNING re <-> im & sign */ |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
522 tmp_b_r = buf[i+6].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
523 tmp_b_i = - buf[i+6].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
524 buf[i+2].real = tmp_a_r + tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
525 buf[i+2].imag = tmp_a_i + tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
526 buf[i+6].real = tmp_a_r - tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
527 buf[i+6].imag = tmp_a_i - tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
528 tmp_a_r = buf[i+3].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
529 tmp_a_i = buf[i+3].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
530 tmp_b_r = (buf[i+7].real - buf[i+7].imag) * w[2][3].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
531 tmp_b_i = (buf[i+7].imag + buf[i+7].real) * w[2][3].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
532 buf[i+3].real = tmp_a_r + tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
533 buf[i+3].imag = tmp_a_i + tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
534 buf[i+7].real = tmp_a_r - tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
535 buf[i+7].imag = tmp_a_i - tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
536 #else |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
537 vector float buf01, buf23, buf45, buf67; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
538 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
539 buf01 = vec_ld((i + 0) << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
540 buf23 = vec_ld((i + 2) << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
541 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
542 tmp_b_r = (buf[i+5].real + buf[i+5].imag) * w[2][1].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
543 tmp_b_i = (buf[i+5].imag - buf[i+5].real) * w[2][1].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
544 buf[i+5].real = tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
545 buf[i+5].imag = tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
546 tmp_b_r = (buf[i+7].real - buf[i+7].imag) * w[2][3].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
547 tmp_b_i = (buf[i+7].imag + buf[i+7].real) * w[2][3].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
548 buf[i+7].real = tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
549 buf[i+7].imag = tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
550 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
551 buf23 = vec_ld((i + 2) << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
552 buf45 = vec_ld((i + 4) << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
553 buf67 = vec_ld((i + 6) << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
554 buf67 = vec_perm(buf67, buf67, vcprm(1,0,2,3)); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
555 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
556 vec_st(vec_add(buf01, buf45), (i + 0) << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
557 vec_st(vec_madd(buf67, vcii(p,n,p,p), buf23), (i + 2) << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
558 vec_st(vec_sub(buf01, buf45), (i + 4) << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
559 vec_st(vec_nmsub(buf67, vcii(p,n,p,p), buf23), (i + 6) << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
560 #endif |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
561 } |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
562 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
563 /* 4-7. iterations */ |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
564 for (m=3; m < 7; m++) { |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
565 two_m = (1 << m); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
566 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
567 two_m_plus_one = two_m<<1; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
568 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
569 for(i = 0; i < 128; i += two_m_plus_one) { |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
570 for(k = 0; k < two_m; k+=2) { |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
571 #if 0 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
572 int p = k + i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
573 int q = p + two_m; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
574 tmp_a_r = buf[p].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
575 tmp_a_i = buf[p].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
576 tmp_b_r = |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
577 buf[q].real * w[m][k].real - |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
578 buf[q].imag * w[m][k].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
579 tmp_b_i = |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
580 buf[q].imag * w[m][k].real + |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
581 buf[q].real * w[m][k].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
582 buf[p].real = tmp_a_r + tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
583 buf[p].imag = tmp_a_i + tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
584 buf[q].real = tmp_a_r - tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
585 buf[q].imag = tmp_a_i - tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
586 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
587 tmp_a_r = buf[(p + 1)].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
588 tmp_a_i = buf[(p + 1)].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
589 tmp_b_r = |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
590 buf[(q + 1)].real * w[m][(k + 1)].real - |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
591 buf[(q + 1)].imag * w[m][(k + 1)].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
592 tmp_b_i = |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
593 buf[(q + 1)].imag * w[m][(k + 1)].real + |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
594 buf[(q + 1)].real * w[m][(k + 1)].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
595 buf[(p + 1)].real = tmp_a_r + tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
596 buf[(p + 1)].imag = tmp_a_i + tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
597 buf[(q + 1)].real = tmp_a_r - tmp_b_r; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
598 buf[(q + 1)].imag = tmp_a_i - tmp_b_i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
599 #else |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
600 int p = k + i; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
601 int q = p + two_m; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
602 vector float vecp, vecq, vecw, temp1, temp2, temp3, temp4; |
9122 | 603 const vector float vczero = (const vector float)FOUROF(0.); |
9001
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
604 // first compute buf[q] and buf[q+1] |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
605 vecq = vec_ld(q << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
606 vecw = vec_ld(0, (float*)&(w[m][k])); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
607 temp1 = vec_madd(vecq, vecw, vczero); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
608 temp2 = vec_perm(vecq, vecq, vcprm(1,0,3,2)); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
609 temp2 = vec_madd(temp2, vecw, vczero); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
610 temp3 = vec_perm(temp1, temp2, vcprm(0,s0,2,s2)); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
611 temp4 = vec_perm(temp1, temp2, vcprm(1,s1,3,s3)); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
612 vecq = vec_madd(temp4, vcii(n,p,n,p), temp3); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
613 // then butterfly with buf[p] and buf[p+1] |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
614 vecp = vec_ld(p << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
615 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
616 temp1 = vec_add(vecp, vecq); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
617 temp2 = vec_sub(vecp, vecq); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
618 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
619 vec_st(temp1, p << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
620 vec_st(temp2, q << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
621 #endif |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
622 } |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
623 } |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
624 } |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
625 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
626 /* Post IFFT complex multiply plus IFFT complex conjugate*/ |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
627 for( i=0; i < 128; i+=4) { |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
628 /* y[n] = z[n] * (xcos1[n] + j * xsin1[n]) ; */ |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
629 #if 0 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
630 tmp_a_r = buf[(i + 0)].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
631 tmp_a_i = -1.0 * buf[(i + 0)].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
632 buf[(i + 0)].real = |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
633 (tmp_a_r * xcos1[(i + 0)]) - (tmp_a_i * xsin1[(i + 0)]); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
634 buf[(i + 0)].imag = |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
635 (tmp_a_r * xsin1[(i + 0)]) + (tmp_a_i * xcos1[(i + 0)]); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
636 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
637 tmp_a_r = buf[(i + 1)].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
638 tmp_a_i = -1.0 * buf[(i + 1)].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
639 buf[(i + 1)].real = |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
640 (tmp_a_r * xcos1[(i + 1)]) - (tmp_a_i * xsin1[(i + 1)]); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
641 buf[(i + 1)].imag = |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
642 (tmp_a_r * xsin1[(i + 1)]) + (tmp_a_i * xcos1[(i + 1)]); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
643 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
644 tmp_a_r = buf[(i + 2)].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
645 tmp_a_i = -1.0 * buf[(i + 2)].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
646 buf[(i + 2)].real = |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
647 (tmp_a_r * xcos1[(i + 2)]) - (tmp_a_i * xsin1[(i + 2)]); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
648 buf[(i + 2)].imag = |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
649 (tmp_a_r * xsin1[(i + 2)]) + (tmp_a_i * xcos1[(i + 2)]); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
650 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
651 tmp_a_r = buf[(i + 3)].real; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
652 tmp_a_i = -1.0 * buf[(i + 3)].imag; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
653 buf[(i + 3)].real = |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
654 (tmp_a_r * xcos1[(i + 3)]) - (tmp_a_i * xsin1[(i + 3)]); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
655 buf[(i + 3)].imag = |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
656 (tmp_a_r * xsin1[(i + 3)]) + (tmp_a_i * xcos1[(i + 3)]); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
657 #else |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
658 vector float bufv_0, bufv_2, cosv, sinv, temp1, temp2; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
659 vector float temp0022, temp1133, tempCS01; |
9122 | 660 const vector float vczero = (const vector float)FOUROF(0.); |
9001
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
661 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
662 bufv_0 = vec_ld((i + 0) << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
663 bufv_2 = vec_ld((i + 2) << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
664 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
665 cosv = vec_ld(i << 2, xcos1); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
666 sinv = vec_ld(i << 2, xsin1); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
667 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
668 temp0022 = vec_perm(bufv_0, bufv_0, vcprm(0,0,2,2)); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
669 temp1133 = vec_perm(bufv_0, bufv_0, vcprm(1,1,3,3)); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
670 tempCS01 = vec_perm(cosv, sinv, vcprm(0,s0,1,s1)); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
671 temp1 = vec_madd(temp0022, tempCS01, vczero); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
672 tempCS01 = vec_perm(cosv, sinv, vcprm(s0,0,s1,1)); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
673 temp2 = vec_madd(temp1133, tempCS01, vczero); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
674 bufv_0 = vec_madd(temp2, vcii(p,n,p,n), temp1); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
675 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
676 vec_st(bufv_0, (i + 0) << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
677 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
678 /* idem with bufv_2 and high-order cosv/sinv */ |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
679 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
680 temp0022 = vec_perm(bufv_2, bufv_2, vcprm(0,0,2,2)); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
681 temp1133 = vec_perm(bufv_2, bufv_2, vcprm(1,1,3,3)); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
682 tempCS01 = vec_perm(cosv, sinv, vcprm(2,s2,3,s3)); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
683 temp1 = vec_madd(temp0022, tempCS01, vczero); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
684 tempCS01 = vec_perm(cosv, sinv, vcprm(s2,2,s3,3)); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
685 temp2 = vec_madd(temp1133, tempCS01, vczero); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
686 bufv_2 = vec_madd(temp2, vcii(p,n,p,n), temp1); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
687 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
688 vec_st(bufv_2, (i + 2) << 3, (float*)buf); |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
689 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
690 #endif |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
691 } |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
692 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
693 data_ptr = data; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
694 delay_ptr = delay; |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
695 window_ptr = a52_imdct_window; |
9001
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
696 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
697 /* Window and convert to real valued signal */ |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
698 for(i=0; i< 64; i++) { |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
699 *data_ptr++ = -buf[64+i].imag * *window_ptr++ + *delay_ptr++ + bias; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
700 *data_ptr++ = buf[64-i-1].real * *window_ptr++ + *delay_ptr++ + bias; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
701 } |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
702 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
703 for(i=0; i< 64; i++) { |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
704 *data_ptr++ = -buf[i].real * *window_ptr++ + *delay_ptr++ + bias; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
705 *data_ptr++ = buf[128-i-1].imag * *window_ptr++ + *delay_ptr++ + bias; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
706 } |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
707 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
708 /* The trailing edge of the window goes into the delay line */ |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
709 delay_ptr = delay; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
710 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
711 for(i=0; i< 64; i++) { |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
712 *delay_ptr++ = -buf[64+i].real * *--window_ptr; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
713 *delay_ptr++ = buf[64-i-1].imag * *--window_ptr; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
714 } |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
715 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
716 for(i=0; i<64; i++) { |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
717 *delay_ptr++ = buf[i].imag * *--window_ptr; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
718 *delay_ptr++ = -buf[128-i-1].real * *--window_ptr; |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
719 } |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
720 } |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
721 #endif |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
722 |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
723 |
4497 | 724 // Stuff below this line is borrowed from libac3 |
725 #include "srfftp.h" | |
16173 | 726 #if defined(ARCH_X86) || defined(ARCH_X86_64) |
4497 | 727 #ifndef HAVE_3DNOW |
728 #define HAVE_3DNOW 1 | |
729 #endif | |
3884 | 730 #include "srfftp_3dnow.h" |
731 | |
8451 | 732 const i_cmplx_t x_plus_minus_3dnow __attribute__ ((aligned (8))) = {{ 0x00000000UL, 0x80000000UL }}; |
733 const i_cmplx_t x_minus_plus_3dnow __attribute__ ((aligned (8))) = {{ 0x80000000UL, 0x00000000UL }}; | |
3884 | 734 const complex_t HSQRT2_3DNOW __attribute__ ((aligned (8))) = { 0.707106781188, 0.707106781188 }; |
735 | |
4497 | 736 #undef HAVE_3DNOWEX |
737 #include "imdct_3dnow.h" | |
738 #define HAVE_3DNOWEX | |
739 #include "imdct_3dnow.h" | |
3884 | 740 |
3579 | 741 void |
742 imdct_do_512_sse(sample_t data[],sample_t delay[], sample_t bias) | |
743 { | |
8254
772d6d27fd66
warning patch by (Dominik Mierzejewski <dominik at rangers dot eu dot org>)
michael
parents:
4497
diff
changeset
|
744 /* int i,k; |
772d6d27fd66
warning patch by (Dominik Mierzejewski <dominik at rangers dot eu dot org>)
michael
parents:
4497
diff
changeset
|
745 int p,q;*/ |
3579 | 746 int m; |
16173 | 747 long two_m; |
748 long two_m_plus_one; | |
749 long two_m_plus_one_shl3; | |
15617
130dd060f723
one bugfix and a few gcc4 bug workaorunds by (Gianluigi Tiesi: mplayer, netfarm it)
michael
parents:
14991
diff
changeset
|
750 complex_t *buf_offset; |
3579 | 751 |
8254
772d6d27fd66
warning patch by (Dominik Mierzejewski <dominik at rangers dot eu dot org>)
michael
parents:
4497
diff
changeset
|
752 /* sample_t tmp_a_i; |
3579 | 753 sample_t tmp_a_r; |
754 sample_t tmp_b_i; | |
8254
772d6d27fd66
warning patch by (Dominik Mierzejewski <dominik at rangers dot eu dot org>)
michael
parents:
4497
diff
changeset
|
755 sample_t tmp_b_r;*/ |
3579 | 756 |
757 sample_t *data_ptr; | |
758 sample_t *delay_ptr; | |
759 sample_t *window_ptr; | |
760 | |
761 /* 512 IMDCT with source and dest data in 'data' */ | |
3623 | 762 /* see the c version (dct_do_512()), its allmost identical, just in C */ |
763 | |
3579 | 764 /* Pre IFFT complex multiply plus IFFT cmplx conjugate */ |
765 /* Bit reversed shuffling */ | |
766 asm volatile( | |
16173 | 767 "xor %%"REG_S", %%"REG_S" \n\t" |
768 "lea "MANGLE(bit_reverse_512)", %%"REG_a"\n\t" | |
769 "mov $1008, %%"REG_D" \n\t" | |
770 "push %%"REG_BP" \n\t" //use ebp without telling gcc | |
19372
6334c14b38eb
Replace asmalign.h hack by ASMALIGN cpp macros from config.h.
diego
parents:
18783
diff
changeset
|
771 ASMALIGN(4) |
3579 | 772 "1: \n\t" |
16173 | 773 "movlps (%0, %%"REG_S"), %%xmm0 \n\t" // XXXI |
774 "movhps 8(%0, %%"REG_D"), %%xmm0 \n\t" // RXXI | |
775 "movlps 8(%0, %%"REG_S"), %%xmm1 \n\t" // XXXi | |
776 "movhps (%0, %%"REG_D"), %%xmm1 \n\t" // rXXi | |
3584 | 777 "shufps $0x33, %%xmm1, %%xmm0 \n\t" // irIR |
16173 | 778 "movaps "MANGLE(sseSinCos1c)"(%%"REG_S"), %%xmm2\n\t" |
3584 | 779 "mulps %%xmm0, %%xmm2 \n\t" |
780 "shufps $0xB1, %%xmm0, %%xmm0 \n\t" // riRI | |
16173 | 781 "mulps "MANGLE(sseSinCos1d)"(%%"REG_S"), %%xmm0\n\t" |
3584 | 782 "subps %%xmm0, %%xmm2 \n\t" |
16173 | 783 "movzb (%%"REG_a"), %%"REG_d" \n\t" |
784 "movzb 1(%%"REG_a"), %%"REG_BP" \n\t" | |
785 "movlps %%xmm2, (%1, %%"REG_d", 8) \n\t" | |
786 "movhps %%xmm2, (%1, %%"REG_BP", 8) \n\t" | |
787 "add $16, %%"REG_S" \n\t" | |
788 "add $2, %%"REG_a" \n\t" // avoid complex addressing for P4 crap | |
789 "sub $16, %%"REG_D" \n\t" | |
790 "jnc 1b \n\t" | |
791 "pop %%"REG_BP" \n\t"//no we didnt touch ebp *g* | |
16189
72764c0dad8a
Fixes segfault on IA-32 machines caused by the ASM patch for AMD-64 for a52.
gpoirier
parents:
16173
diff
changeset
|
792 :: "b" (data), "c" (buf) |
16173 | 793 : "%"REG_S, "%"REG_D, "%"REG_a, "%"REG_d |
3579 | 794 ); |
795 | |
796 | |
797 /* FFT Merge */ | |
798 /* unoptimized variant | |
799 for (m=1; m < 7; m++) { | |
800 if(m) | |
801 two_m = (1 << m); | |
802 else | |
803 two_m = 1; | |
804 | |
805 two_m_plus_one = (1 << (m+1)); | |
806 | |
807 for(i = 0; i < 128; i += two_m_plus_one) { | |
808 for(k = 0; k < two_m; k++) { | |
809 p = k + i; | |
810 q = p + two_m; | |
811 tmp_a_r = buf[p].real; | |
812 tmp_a_i = buf[p].imag; | |
813 tmp_b_r = buf[q].real * w[m][k].real - buf[q].imag * w[m][k].imag; | |
814 tmp_b_i = buf[q].imag * w[m][k].real + buf[q].real * w[m][k].imag; | |
815 buf[p].real = tmp_a_r + tmp_b_r; | |
816 buf[p].imag = tmp_a_i + tmp_b_i; | |
817 buf[q].real = tmp_a_r - tmp_b_r; | |
818 buf[q].imag = tmp_a_i - tmp_b_i; | |
819 } | |
820 } | |
821 } | |
822 */ | |
823 | |
3623 | 824 /* 1. iteration */ |
3549 | 825 // Note w[0][0]={1,0} |
3508 | 826 asm volatile( |
827 "xorps %%xmm1, %%xmm1 \n\t" | |
828 "xorps %%xmm2, %%xmm2 \n\t" | |
16173 | 829 "mov %0, %%"REG_S" \n\t" |
19372
6334c14b38eb
Replace asmalign.h hack by ASMALIGN cpp macros from config.h.
diego
parents:
18783
diff
changeset
|
830 ASMALIGN(4) |
3508 | 831 "1: \n\t" |
16173 | 832 "movlps (%%"REG_S"), %%xmm0\n\t" //buf[p] |
833 "movlps 8(%%"REG_S"), %%xmm1\n\t" //buf[q] | |
834 "movhps (%%"REG_S"), %%xmm0\n\t" //buf[p] | |
835 "movhps 8(%%"REG_S"), %%xmm2\n\t" //buf[q] | |
3508 | 836 "addps %%xmm1, %%xmm0 \n\t" |
837 "subps %%xmm2, %%xmm0 \n\t" | |
16173 | 838 "movaps %%xmm0, (%%"REG_S")\n\t" |
839 "add $16, %%"REG_S" \n\t" | |
840 "cmp %1, %%"REG_S" \n\t" | |
3508 | 841 " jb 1b \n\t" |
842 :: "g" (buf), "r" (buf + 128) | |
16173 | 843 : "%"REG_S |
3508 | 844 ); |
3549 | 845 |
3623 | 846 /* 2. iteration */ |
3512 | 847 // Note w[1]={{1,0}, {0,-1}} |
848 asm volatile( | |
4247
2dbd637ffe05
mangle for win32 in liba52 (includes dummy mangle.h pointing to the one in main)
atmos4
parents:
3908
diff
changeset
|
849 "movaps "MANGLE(ps111_1)", %%xmm7\n\t" // 1,1,1,-1 |
16173 | 850 "mov %0, %%"REG_S" \n\t" |
19372
6334c14b38eb
Replace asmalign.h hack by ASMALIGN cpp macros from config.h.
diego
parents:
18783
diff
changeset
|
851 ASMALIGN(4) |
3512 | 852 "1: \n\t" |
16173 | 853 "movaps 16(%%"REG_S"), %%xmm2 \n\t" //r2,i2,r3,i3 |
3512 | 854 "shufps $0xB4, %%xmm2, %%xmm2 \n\t" //r2,i2,i3,r3 |
855 "mulps %%xmm7, %%xmm2 \n\t" //r2,i2,i3,-r3 | |
16173 | 856 "movaps (%%"REG_S"), %%xmm0 \n\t" //r0,i0,r1,i1 |
857 "movaps (%%"REG_S"), %%xmm1 \n\t" //r0,i0,r1,i1 | |
3512 | 858 "addps %%xmm2, %%xmm0 \n\t" |
859 "subps %%xmm2, %%xmm1 \n\t" | |
16173 | 860 "movaps %%xmm0, (%%"REG_S") \n\t" |
861 "movaps %%xmm1, 16(%%"REG_S") \n\t" | |
862 "add $32, %%"REG_S" \n\t" | |
863 "cmp %1, %%"REG_S" \n\t" | |
3512 | 864 " jb 1b \n\t" |
865 :: "g" (buf), "r" (buf + 128) | |
16173 | 866 : "%"REG_S |
3512 | 867 ); |
3549 | 868 |
3623 | 869 /* 3. iteration */ |
3534 | 870 /* |
871 Note sseW2+0={1,1,sqrt(2),sqrt(2)) | |
872 Note sseW2+16={0,0,sqrt(2),-sqrt(2)) | |
873 Note sseW2+32={0,0,-sqrt(2),-sqrt(2)) | |
874 Note sseW2+48={1,-1,sqrt(2),-sqrt(2)) | |
875 */ | |
876 asm volatile( | |
4247
2dbd637ffe05
mangle for win32 in liba52 (includes dummy mangle.h pointing to the one in main)
atmos4
parents:
3908
diff
changeset
|
877 "movaps 48+"MANGLE(sseW2)", %%xmm6\n\t" |
2dbd637ffe05
mangle for win32 in liba52 (includes dummy mangle.h pointing to the one in main)
atmos4
parents:
3908
diff
changeset
|
878 "movaps 16+"MANGLE(sseW2)", %%xmm7\n\t" |
3534 | 879 "xorps %%xmm5, %%xmm5 \n\t" |
880 "xorps %%xmm2, %%xmm2 \n\t" | |
16173 | 881 "mov %0, %%"REG_S" \n\t" |
19372
6334c14b38eb
Replace asmalign.h hack by ASMALIGN cpp macros from config.h.
diego
parents:
18783
diff
changeset
|
882 ASMALIGN(4) |
3534 | 883 "1: \n\t" |
16173 | 884 "movaps 32(%%"REG_S"), %%xmm2 \n\t" //r4,i4,r5,i5 |
885 "movaps 48(%%"REG_S"), %%xmm3 \n\t" //r6,i6,r7,i7 | |
4247
2dbd637ffe05
mangle for win32 in liba52 (includes dummy mangle.h pointing to the one in main)
atmos4
parents:
3908
diff
changeset
|
886 "movaps "MANGLE(sseW2)", %%xmm4 \n\t" //r4,i4,r5,i5 |
2dbd637ffe05
mangle for win32 in liba52 (includes dummy mangle.h pointing to the one in main)
atmos4
parents:
3908
diff
changeset
|
887 "movaps 32+"MANGLE(sseW2)", %%xmm5\n\t" //r6,i6,r7,i7 |
3537 | 888 "mulps %%xmm2, %%xmm4 \n\t" |
889 "mulps %%xmm3, %%xmm5 \n\t" | |
3534 | 890 "shufps $0xB1, %%xmm2, %%xmm2 \n\t" //i4,r4,i5,r5 |
891 "shufps $0xB1, %%xmm3, %%xmm3 \n\t" //i6,r6,i7,r7 | |
3537 | 892 "mulps %%xmm6, %%xmm3 \n\t" |
3534 | 893 "mulps %%xmm7, %%xmm2 \n\t" |
16173 | 894 "movaps (%%"REG_S"), %%xmm0 \n\t" //r0,i0,r1,i1 |
895 "movaps 16(%%"REG_S"), %%xmm1 \n\t" //r2,i2,r3,i3 | |
3534 | 896 "addps %%xmm4, %%xmm2 \n\t" |
897 "addps %%xmm5, %%xmm3 \n\t" | |
898 "movaps %%xmm2, %%xmm4 \n\t" | |
899 "movaps %%xmm3, %%xmm5 \n\t" | |
900 "addps %%xmm0, %%xmm2 \n\t" | |
901 "addps %%xmm1, %%xmm3 \n\t" | |
902 "subps %%xmm4, %%xmm0 \n\t" | |
903 "subps %%xmm5, %%xmm1 \n\t" | |
16173 | 904 "movaps %%xmm2, (%%"REG_S") \n\t" |
905 "movaps %%xmm3, 16(%%"REG_S") \n\t" | |
906 "movaps %%xmm0, 32(%%"REG_S") \n\t" | |
907 "movaps %%xmm1, 48(%%"REG_S") \n\t" | |
908 "add $64, %%"REG_S" \n\t" | |
909 "cmp %1, %%"REG_S" \n\t" | |
3534 | 910 " jb 1b \n\t" |
911 :: "g" (buf), "r" (buf + 128) | |
16173 | 912 : "%"REG_S |
3534 | 913 ); |
3508 | 914 |
3623 | 915 /* 4-7. iterations */ |
3546 | 916 for (m=3; m < 7; m++) { |
917 two_m = (1 << m); | |
918 two_m_plus_one = two_m<<1; | |
15617
130dd060f723
one bugfix and a few gcc4 bug workaorunds by (Gianluigi Tiesi: mplayer, netfarm it)
michael
parents:
14991
diff
changeset
|
919 two_m_plus_one_shl3 = (two_m_plus_one<<3); |
130dd060f723
one bugfix and a few gcc4 bug workaorunds by (Gianluigi Tiesi: mplayer, netfarm it)
michael
parents:
14991
diff
changeset
|
920 buf_offset = buf+128; |
3546 | 921 asm volatile( |
16173 | 922 "mov %0, %%"REG_S" \n\t" |
19372
6334c14b38eb
Replace asmalign.h hack by ASMALIGN cpp macros from config.h.
diego
parents:
18783
diff
changeset
|
923 ASMALIGN(4) |
3546 | 924 "1: \n\t" |
16173 | 925 "xor %%"REG_D", %%"REG_D" \n\t" // k |
926 "lea (%%"REG_S", %3), %%"REG_d" \n\t" | |
3546 | 927 "2: \n\t" |
16173 | 928 "movaps (%%"REG_d", %%"REG_D"), %%xmm1 \n\t" |
929 "movaps (%4, %%"REG_D", 2), %%xmm2 \n\t" | |
3546 | 930 "mulps %%xmm1, %%xmm2 \n\t" |
931 "shufps $0xB1, %%xmm1, %%xmm1 \n\t" | |
16173 | 932 "mulps 16(%4, %%"REG_D", 2), %%xmm1 \n\t" |
933 "movaps (%%"REG_S", %%"REG_D"), %%xmm0 \n\t" | |
3546 | 934 "addps %%xmm2, %%xmm1 \n\t" |
935 "movaps %%xmm1, %%xmm2 \n\t" | |
936 "addps %%xmm0, %%xmm1 \n\t" | |
937 "subps %%xmm2, %%xmm0 \n\t" | |
16173 | 938 "movaps %%xmm1, (%%"REG_S", %%"REG_D") \n\t" |
939 "movaps %%xmm0, (%%"REG_d", %%"REG_D") \n\t" | |
940 "add $16, %%"REG_D" \n\t" | |
941 "cmp %3, %%"REG_D" \n\t" //FIXME (opt) count against 0 | |
942 "jb 2b \n\t" | |
943 "add %2, %%"REG_S" \n\t" | |
944 "cmp %1, %%"REG_S" \n\t" | |
3546 | 945 " jb 1b \n\t" |
15617
130dd060f723
one bugfix and a few gcc4 bug workaorunds by (Gianluigi Tiesi: mplayer, netfarm it)
michael
parents:
14991
diff
changeset
|
946 :: "g" (buf), "m" (buf_offset), "m" (two_m_plus_one_shl3), "r" (two_m<<3), |
3546 | 947 "r" (sseW[m]) |
16173 | 948 : "%"REG_S, "%"REG_D, "%"REG_d |
3546 | 949 ); |
950 } | |
951 | |
3623 | 952 /* Post IFFT complex multiply plus IFFT complex conjugate*/ |
3581 | 953 asm volatile( |
16173 | 954 "mov $-1024, %%"REG_S" \n\t" |
19372
6334c14b38eb
Replace asmalign.h hack by ASMALIGN cpp macros from config.h.
diego
parents:
18783
diff
changeset
|
955 ASMALIGN(4) |
3581 | 956 "1: \n\t" |
16173 | 957 "movaps (%0, %%"REG_S"), %%xmm0 \n\t" |
958 "movaps (%0, %%"REG_S"), %%xmm1 \n\t" | |
3581 | 959 "shufps $0xB1, %%xmm0, %%xmm0 \n\t" |
16173 | 960 "mulps 1024+"MANGLE(sseSinCos1c)"(%%"REG_S"), %%xmm1\n\t" |
961 "mulps 1024+"MANGLE(sseSinCos1d)"(%%"REG_S"), %%xmm0\n\t" | |
3581 | 962 "addps %%xmm1, %%xmm0 \n\t" |
16173 | 963 "movaps %%xmm0, (%0, %%"REG_S") \n\t" |
964 "add $16, %%"REG_S" \n\t" | |
3581 | 965 " jnz 1b \n\t" |
966 :: "r" (buf+128) | |
16173 | 967 : "%"REG_S |
3581 | 968 ); |
969 | |
3394 | 970 |
971 data_ptr = data; | |
972 delay_ptr = delay; | |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
973 window_ptr = a52_imdct_window; |
3394 | 974 |
975 /* Window and convert to real valued signal */ | |
3552 | 976 asm volatile( |
16173 | 977 "xor %%"REG_D", %%"REG_D" \n\t" // 0 |
978 "xor %%"REG_S", %%"REG_S" \n\t" // 0 | |
3552 | 979 "movss %3, %%xmm2 \n\t" // bias |
980 "shufps $0x00, %%xmm2, %%xmm2 \n\t" // bias, bias, ... | |
19372
6334c14b38eb
Replace asmalign.h hack by ASMALIGN cpp macros from config.h.
diego
parents:
18783
diff
changeset
|
981 ASMALIGN(4) |
3552 | 982 "1: \n\t" |
16173 | 983 "movlps (%0, %%"REG_S"), %%xmm0 \n\t" // ? ? A ? |
984 "movlps 8(%0, %%"REG_S"), %%xmm1 \n\t" // ? ? C ? | |
985 "movhps -16(%0, %%"REG_D"), %%xmm1 \n\t" // ? D C ? | |
986 "movhps -8(%0, %%"REG_D"), %%xmm0 \n\t" // ? B A ? | |
3552 | 987 "shufps $0x99, %%xmm1, %%xmm0 \n\t" // D C B A |
16173 | 988 "mulps "MANGLE(sseWindow)"(%%"REG_S"), %%xmm0\n\t" |
989 "addps (%2, %%"REG_S"), %%xmm0 \n\t" | |
3552 | 990 "addps %%xmm2, %%xmm0 \n\t" |
16173 | 991 "movaps %%xmm0, (%1, %%"REG_S") \n\t" |
992 "add $16, %%"REG_S" \n\t" | |
993 "sub $16, %%"REG_D" \n\t" | |
994 "cmp $512, %%"REG_S" \n\t" | |
3552 | 995 " jb 1b \n\t" |
996 :: "r" (buf+64), "r" (data_ptr), "r" (delay_ptr), "m" (bias) | |
16173 | 997 : "%"REG_S, "%"REG_D |
3552 | 998 ); |
999 data_ptr+=128; | |
1000 delay_ptr+=128; | |
3553 | 1001 // window_ptr+=128; |
3579 | 1002 |
3552 | 1003 asm volatile( |
16173 | 1004 "mov $1024, %%"REG_D" \n\t" // 512 |
1005 "xor %%"REG_S", %%"REG_S" \n\t" // 0 | |
3552 | 1006 "movss %3, %%xmm2 \n\t" // bias |
1007 "shufps $0x00, %%xmm2, %%xmm2 \n\t" // bias, bias, ... | |
19372
6334c14b38eb
Replace asmalign.h hack by ASMALIGN cpp macros from config.h.
diego
parents:
18783
diff
changeset
|
1008 ASMALIGN(4) |
3552 | 1009 "1: \n\t" |
16173 | 1010 "movlps (%0, %%"REG_S"), %%xmm0 \n\t" // ? ? ? A |
1011 "movlps 8(%0, %%"REG_S"), %%xmm1 \n\t" // ? ? ? C | |
1012 "movhps -16(%0, %%"REG_D"), %%xmm1 \n\t" // D ? ? C | |
1013 "movhps -8(%0, %%"REG_D"), %%xmm0 \n\t" // B ? ? A | |
3552 | 1014 "shufps $0xCC, %%xmm1, %%xmm0 \n\t" // D C B A |
16173 | 1015 "mulps 512+"MANGLE(sseWindow)"(%%"REG_S"), %%xmm0\n\t" |
1016 "addps (%2, %%"REG_S"), %%xmm0 \n\t" | |
3552 | 1017 "addps %%xmm2, %%xmm0 \n\t" |
16173 | 1018 "movaps %%xmm0, (%1, %%"REG_S") \n\t" |
1019 "add $16, %%"REG_S" \n\t" | |
1020 "sub $16, %%"REG_D" \n\t" | |
1021 "cmp $512, %%"REG_S" \n\t" | |
3552 | 1022 " jb 1b \n\t" |
1023 :: "r" (buf), "r" (data_ptr), "r" (delay_ptr), "m" (bias) | |
16173 | 1024 : "%"REG_S, "%"REG_D |
3552 | 1025 ); |
1026 data_ptr+=128; | |
3553 | 1027 // window_ptr+=128; |
3394 | 1028 |
1029 /* The trailing edge of the window goes into the delay line */ | |
1030 delay_ptr = delay; | |
1031 | |
3553 | 1032 asm volatile( |
16173 | 1033 "xor %%"REG_D", %%"REG_D" \n\t" // 0 |
1034 "xor %%"REG_S", %%"REG_S" \n\t" // 0 | |
19372
6334c14b38eb
Replace asmalign.h hack by ASMALIGN cpp macros from config.h.
diego
parents:
18783
diff
changeset
|
1035 ASMALIGN(4) |
3553 | 1036 "1: \n\t" |
16173 | 1037 "movlps (%0, %%"REG_S"), %%xmm0 \n\t" // ? ? ? A |
1038 "movlps 8(%0, %%"REG_S"), %%xmm1 \n\t" // ? ? ? C | |
1039 "movhps -16(%0, %%"REG_D"), %%xmm1 \n\t" // D ? ? C | |
1040 "movhps -8(%0, %%"REG_D"), %%xmm0 \n\t" // B ? ? A | |
3553 | 1041 "shufps $0xCC, %%xmm1, %%xmm0 \n\t" // D C B A |
16173 | 1042 "mulps 1024+"MANGLE(sseWindow)"(%%"REG_S"), %%xmm0\n\t" |
1043 "movaps %%xmm0, (%1, %%"REG_S") \n\t" | |
1044 "add $16, %%"REG_S" \n\t" | |
1045 "sub $16, %%"REG_D" \n\t" | |
1046 "cmp $512, %%"REG_S" \n\t" | |
3553 | 1047 " jb 1b \n\t" |
1048 :: "r" (buf+64), "r" (delay_ptr) | |
16173 | 1049 : "%"REG_S, "%"REG_D |
3553 | 1050 ); |
1051 delay_ptr+=128; | |
1052 // window_ptr-=128; | |
3579 | 1053 |
3553 | 1054 asm volatile( |
16173 | 1055 "mov $1024, %%"REG_D" \n\t" // 1024 |
1056 "xor %%"REG_S", %%"REG_S" \n\t" // 0 | |
19372
6334c14b38eb
Replace asmalign.h hack by ASMALIGN cpp macros from config.h.
diego
parents:
18783
diff
changeset
|
1057 ASMALIGN(4) |
3553 | 1058 "1: \n\t" |
16173 | 1059 "movlps (%0, %%"REG_S"), %%xmm0 \n\t" // ? ? A ? |
1060 "movlps 8(%0, %%"REG_S"), %%xmm1 \n\t" // ? ? C ? | |
1061 "movhps -16(%0, %%"REG_D"), %%xmm1 \n\t" // ? D C ? | |
1062 "movhps -8(%0, %%"REG_D"), %%xmm0 \n\t" // ? B A ? | |
3553 | 1063 "shufps $0x99, %%xmm1, %%xmm0 \n\t" // D C B A |
16173 | 1064 "mulps 1536+"MANGLE(sseWindow)"(%%"REG_S"), %%xmm0\n\t" |
1065 "movaps %%xmm0, (%1, %%"REG_S") \n\t" | |
1066 "add $16, %%"REG_S" \n\t" | |
1067 "sub $16, %%"REG_D" \n\t" | |
1068 "cmp $512, %%"REG_S" \n\t" | |
3553 | 1069 " jb 1b \n\t" |
1070 :: "r" (buf), "r" (delay_ptr) | |
16173 | 1071 : "%"REG_S, "%"REG_D |
3553 | 1072 ); |
3394 | 1073 } |
16173 | 1074 #endif // ARCH_X86 || ARCH_X86_64 |
3394 | 1075 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1076 void a52_imdct_256(sample_t * data, sample_t * delay, sample_t bias) |
3394 | 1077 { |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1078 int i, k; |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1079 sample_t t_r, t_i, a_r, a_i, b_r, b_i, c_r, c_i, d_r, d_i, w_1, w_2; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1080 const sample_t * window = a52_imdct_window; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1081 complex_t buf1[64], buf2[64]; |
3394 | 1082 |
1083 /* Pre IFFT complex multiply plus IFFT cmplx conjugate */ | |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1084 for (i = 0; i < 64; i++) { |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1085 k = fftorder[i]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1086 t_r = pre2[i].real; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1087 t_i = pre2[i].imag; |
3394 | 1088 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1089 buf1[i].real = t_i * data[254-k] + t_r * data[k]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1090 buf1[i].imag = t_r * data[254-k] - t_i * data[k]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1091 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1092 buf2[i].real = t_i * data[255-k] + t_r * data[k+1]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1093 buf2[i].imag = t_r * data[255-k] - t_i * data[k+1]; |
3394 | 1094 } |
1095 | |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1096 ifft64 (buf1); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1097 ifft64 (buf2); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1098 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1099 /* Post IFFT complex multiply */ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1100 /* Window and convert to real valued signal */ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1101 for (i = 0; i < 32; i++) { |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1102 /* y1[n] = z1[n] * (xcos2[n] + j * xs in2[n]) ; */ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1103 t_r = post2[i].real; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1104 t_i = post2[i].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1105 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1106 a_r = t_r * buf1[i].real + t_i * buf1[i].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1107 a_i = t_i * buf1[i].real - t_r * buf1[i].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1108 b_r = t_i * buf1[63-i].real + t_r * buf1[63-i].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1109 b_i = t_r * buf1[63-i].real - t_i * buf1[63-i].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1110 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1111 c_r = t_r * buf2[i].real + t_i * buf2[i].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1112 c_i = t_i * buf2[i].real - t_r * buf2[i].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1113 d_r = t_i * buf2[63-i].real + t_r * buf2[63-i].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1114 d_i = t_r * buf2[63-i].real - t_i * buf2[63-i].imag; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1115 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1116 w_1 = window[2*i]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1117 w_2 = window[255-2*i]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1118 data[2*i] = delay[2*i] * w_2 - a_r * w_1 + bias; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1119 data[255-2*i] = delay[2*i] * w_1 + a_r * w_2 + bias; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1120 delay[2*i] = c_i; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1121 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1122 w_1 = window[128+2*i]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1123 w_2 = window[127-2*i]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1124 data[128+2*i] = delay[127-2*i] * w_2 + a_i * w_1 + bias; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1125 data[127-2*i] = delay[127-2*i] * w_1 - a_i * w_2 + bias; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1126 delay[127-2*i] = c_r; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1127 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1128 w_1 = window[2*i+1]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1129 w_2 = window[254-2*i]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1130 data[2*i+1] = delay[2*i+1] * w_2 - b_i * w_1 + bias; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1131 data[254-2*i] = delay[2*i+1] * w_1 + b_i * w_2 + bias; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1132 delay[2*i+1] = d_r; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1133 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1134 w_1 = window[129+2*i]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1135 w_2 = window[126-2*i]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1136 data[129+2*i] = delay[126-2*i] * w_2 + b_r * w_1 + bias; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1137 data[126-2*i] = delay[126-2*i] * w_1 - b_r * w_2 + bias; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1138 delay[126-2*i] = d_i; |
3394 | 1139 } |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1140 } |
3394 | 1141 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1142 static double besselI0 (double x) |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1143 { |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1144 double bessel = 1; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1145 int i = 100; |
3394 | 1146 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1147 do |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1148 bessel = bessel * x / (i * i) + 1; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1149 while (--i); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1150 return bessel; |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1151 } |
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1152 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1153 void a52_imdct_init (uint32_t mm_accel) |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1154 { |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1155 int i, j, k; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1156 double sum; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1157 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1158 /* compute imdct window - kaiser-bessel derived window, alpha = 5.0 */ |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1159 sum = 0; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1160 for (i = 0; i < 256; i++) { |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1161 sum += besselI0 (i * (256 - i) * (5 * M_PI / 256) * (5 * M_PI / 256)); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1162 a52_imdct_window[i] = sum; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1163 } |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1164 sum++; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1165 for (i = 0; i < 256; i++) |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1166 a52_imdct_window[i] = sqrt (a52_imdct_window[i] / sum); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1167 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1168 for (i = 0; i < 3; i++) |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1169 roots16[i] = cos ((M_PI / 8) * (i + 1)); |
3394 | 1170 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1171 for (i = 0; i < 7; i++) |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1172 roots32[i] = cos ((M_PI / 16) * (i + 1)); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1173 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1174 for (i = 0; i < 15; i++) |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1175 roots64[i] = cos ((M_PI / 32) * (i + 1)); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1176 |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1177 for (i = 0; i < 31; i++) |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1178 roots128[i] = cos ((M_PI / 64) * (i + 1)); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1179 |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1180 for (i = 0; i < 64; i++) { |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1181 k = fftorder[i] / 2 + 64; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1182 pre1[i].real = cos ((M_PI / 256) * (k - 0.25)); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1183 pre1[i].imag = sin ((M_PI / 256) * (k - 0.25)); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1184 } |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1185 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1186 for (i = 64; i < 128; i++) { |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1187 k = fftorder[i] / 2 + 64; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1188 pre1[i].real = -cos ((M_PI / 256) * (k - 0.25)); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1189 pre1[i].imag = -sin ((M_PI / 256) * (k - 0.25)); |
3394 | 1190 } |
1191 | |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1192 for (i = 0; i < 64; i++) { |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1193 post1[i].real = cos ((M_PI / 256) * (i + 0.5)); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1194 post1[i].imag = sin ((M_PI / 256) * (i + 0.5)); |
3394 | 1195 } |
1196 | |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1197 for (i = 0; i < 64; i++) { |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1198 k = fftorder[i] / 4; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1199 pre2[i].real = cos ((M_PI / 128) * (k - 0.25)); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1200 pre2[i].imag = sin ((M_PI / 128) * (k - 0.25)); |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1201 } |
3394 | 1202 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1203 for (i = 0; i < 32; i++) { |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1204 post2[i].real = cos ((M_PI / 128) * (i + 0.5)); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1205 post2[i].imag = sin ((M_PI / 128) * (i + 0.5)); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1206 } |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1207 for (i = 0; i < 128; i++) { |
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1208 xcos1[i] = -cos ((M_PI / 2048) * (8 * i + 1)); |
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1209 xsin1[i] = -sin ((M_PI / 2048) * (8 * i + 1)); |
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1210 } |
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1211 for (i = 0; i < 7; i++) { |
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1212 j = 1 << i; |
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1213 for (k = 0; k < j; k++) { |
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1214 w[i][k].real = cos (-M_PI * k / j); |
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1215 w[i][k].imag = sin (-M_PI * k / j); |
3394 | 1216 } |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1217 } |
16173 | 1218 #if defined(ARCH_X86) || defined(ARCH_X86_64) |
3527 | 1219 for (i = 0; i < 128; i++) { |
3581 | 1220 sseSinCos1c[2*i+0]= xcos1[i]; |
1221 sseSinCos1c[2*i+1]= -xcos1[i]; | |
1222 sseSinCos1d[2*i+0]= xsin1[i]; | |
1223 sseSinCos1d[2*i+1]= xsin1[i]; | |
3527 | 1224 } |
3534 | 1225 for (i = 1; i < 7; i++) { |
1226 j = 1 << i; | |
1227 for (k = 0; k < j; k+=2) { | |
1228 | |
1229 sseW[i][4*k + 0] = w[i][k+0].real; | |
1230 sseW[i][4*k + 1] = w[i][k+0].real; | |
1231 sseW[i][4*k + 2] = w[i][k+1].real; | |
1232 sseW[i][4*k + 3] = w[i][k+1].real; | |
1233 | |
1234 sseW[i][4*k + 4] = -w[i][k+0].imag; | |
1235 sseW[i][4*k + 5] = w[i][k+0].imag; | |
1236 sseW[i][4*k + 6] = -w[i][k+1].imag; | |
1237 sseW[i][4*k + 7] = w[i][k+1].imag; | |
1238 | |
1239 //we multiply more or less uninitalized numbers so we need to use exactly 0.0 | |
1240 if(k==0) | |
1241 { | |
1242 // sseW[i][4*k + 0]= sseW[i][4*k + 1]= 1.0; | |
1243 sseW[i][4*k + 4]= sseW[i][4*k + 5]= 0.0; | |
1244 } | |
1245 | |
1246 if(2*k == j) | |
1247 { | |
1248 sseW[i][4*k + 0]= sseW[i][4*k + 1]= 0.0; | |
1249 // sseW[i][4*k + 4]= -(sseW[i][4*k + 5]= -1.0); | |
1250 } | |
1251 } | |
1252 } | |
3552 | 1253 |
1254 for(i=0; i<128; i++) | |
1255 { | |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1256 sseWindow[2*i+0]= -a52_imdct_window[2*i+0]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1257 sseWindow[2*i+1]= a52_imdct_window[2*i+1]; |
3552 | 1258 } |
3553 | 1259 |
1260 for(i=0; i<64; i++) | |
1261 { | |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1262 sseWindow[256 + 2*i+0]= -a52_imdct_window[254 - 2*i+1]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1263 sseWindow[256 + 2*i+1]= a52_imdct_window[254 - 2*i+0]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1264 sseWindow[384 + 2*i+0]= a52_imdct_window[126 - 2*i+1]; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1265 sseWindow[384 + 2*i+1]= -a52_imdct_window[126 - 2*i+0]; |
3553 | 1266 } |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1267 #endif |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1268 a52_imdct_512 = imdct_do_512; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1269 ifft128 = ifft128_c; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1270 ifft64 = ifft64_c; |
3579 | 1271 |
16173 | 1272 #if defined(ARCH_X86) || defined(ARCH_X86_64) |
4497 | 1273 if(mm_accel & MM_ACCEL_X86_SSE) |
1274 { | |
1275 fprintf (stderr, "Using SSE optimized IMDCT transform\n"); | |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1276 a52_imdct_512 = imdct_do_512_sse; |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1277 } |
4497 | 1278 else |
1279 if(mm_accel & MM_ACCEL_X86_3DNOWEXT) | |
1280 { | |
1281 fprintf (stderr, "Using 3DNowEx optimized IMDCT transform\n"); | |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1282 a52_imdct_512 = imdct_do_512_3dnowex; |
4497 | 1283 } |
1284 else | |
1285 if(mm_accel & MM_ACCEL_X86_3DNOW) | |
1286 { | |
1287 fprintf (stderr, "Using 3DNow optimized IMDCT transform\n"); | |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1288 a52_imdct_512 = imdct_do_512_3dnow; |
4497 | 1289 } |
1290 else | |
16173 | 1291 #endif // ARCH_X86 || ARCH_X86_64 |
9001
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
1292 #ifdef HAVE_ALTIVEC |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
1293 if (mm_accel & MM_ACCEL_PPC_ALTIVEC) |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
1294 { |
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
1295 fprintf(stderr, "Using AltiVec optimized IMDCT transform\n"); |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1296 a52_imdct_512 = imdct_do_512_altivec; |
9001
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
1297 } |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1298 else |
9001
01a9cf43074c
An AltiVec-enhanced IMDCT for liba52 (liba52/imdct.c)
arpi
parents:
8451
diff
changeset
|
1299 #endif |
3884 | 1300 |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1301 #ifdef LIBA52_DJBFFT |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1302 if (mm_accel & MM_ACCEL_DJBFFT) { |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1303 fprintf (stderr, "Using djbfft for IMDCT transform\n"); |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1304 ifft128 = (void (*) (complex_t *)) fftc4_un128; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1305 ifft64 = (void (*) (complex_t *)) fftc4_un64; |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1306 } else |
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1307 #endif |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1308 { |
18720
4bad7f00556e
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18104
diff
changeset
|
1309 fprintf (stderr, "No accelerated IMDCT transform found\n"); |
18721
722ac20fac5f
sync with liba52 0.7.4, patch by Emanuele Giaquinta >emanuele.giaquinta ! gmail * com<
rathann
parents:
18720
diff
changeset
|
1310 } |
3884 | 1311 } |