NFFT 3.6.0
nfft.c
1/*
2 * Copyright (c) 2002, 2017 Jens Keiner, Stefan Kunis, Daniel Potts
3 *
4 * This program is free software; you can redistribute it and/or modify it under
5 * the terms of the GNU General Public License as published by the Free Software
6 * Foundation; either version 2 of the License, or (at your option) any later
7 * version.
8 *
9 * This program is distributed in the hope that it will be useful, but WITHOUT
10 * ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or FITNESS
11 * FOR A PARTICULAR PURPOSE. See the GNU General Public License for more
12 * details.
13 *
14 * You should have received a copy of the GNU General Public License along with
15 * this program; if not, write to the Free Software Foundation, Inc., 51
16 * Franklin Street, Fifth Floor, Boston, MA 02110-1301, USA.
17 */
18
19/* Nonequispaced FFT */
20
21/* Authors: D. Potts, S. Kunis 2002-2009, Jens Keiner 2009, Toni Volkmer 2012 */
22
23/* configure header */
24#include "config.h"
25
26/* complex datatype (maybe) */
27#ifdef HAVE_COMPLEX_H
28#include<complex.h>
29#endif
30
31/* NFFT headers */
32#include "nfft3.h"
33#include "infft.h"
34
35#ifdef _OPENMP
36#include <omp.h>
37#endif
38
39#ifdef OMP_ASSERT
40#include <assert.h>
41#endif
42
43#undef X
44#define X(name) NFFT(name)
45
47static inline INT intprod(const INT *vec, const INT a, const INT d)
48{
49 INT t, p;
50
51 p = 1;
52 for (t = 0; t < d; t++)
53 p *= vec[t] - a;
54
55 return p;
56}
57
58/* handy shortcuts */
59#define BASE(x) CEXP(x)
60
75static inline void sort0(const INT d, const INT *n, const INT m,
76 const INT local_x_num, const R *local_x, INT *ar_x)
77{
78 INT u_j[d], i, j, help, rhigh;
79 INT *ar_x_temp;
80 INT nprod;
81
82 for (i = 0; i < local_x_num; i++)
83 {
84 ar_x[2 * i] = 0;
85 ar_x[2 *i + 1] = i;
86 for (j = 0; j < d; j++)
87 {
88 help = (INT) LRINT(FLOOR((R)(n[j]) * local_x[d * i + j] - (R)(m)));
89 u_j[j] = (help % n[j] + n[j]) % n[j];
90
91 ar_x[2 * i] += u_j[j];
92 if (j + 1 < d)
93 ar_x[2 * i] *= n[j + 1];
94 }
95 }
96
97 for (j = 0, nprod = 1; j < d; j++)
98 nprod *= n[j];
99
100 rhigh = (INT) LRINT(CEIL(LOG2((R)nprod))) - 1;
101
102 ar_x_temp = (INT*) Y(malloc)(2 * (size_t)(local_x_num) * sizeof(INT));
103 Y(sort_node_indices_radix_lsdf)(local_x_num, ar_x, ar_x_temp, rhigh);
104#ifdef OMP_ASSERT
105 for (i = 1; i < local_x_num; i++)
106 assert(ar_x[2 * (i - 1)] <= ar_x[2 * i]);
107#endif
108 Y(free)(ar_x_temp);
109}
110
119static inline void sort(const X(plan) *ths)
120{
121 if (ths->flags & NFFT_SORT_NODES)
122 sort0(ths->d, ths->n, ths->m, ths->M_total, ths->x, ths->index_x);
123}
124
145void X(trafo_direct)(const X(plan) *ths)
146{
147 C *f_hat = (C*)ths->f_hat, *f = (C*)ths->f;
148
149 if (ths->d == 1)
150 {
151 /* specialize for univariate case, rationale: faster */
152 INT j;
153#ifdef _OPENMP
154 #pragma omp parallel for default(shared) private(j)
155#endif
156 for (j = 0; j < ths->M_total; j++)
157 {
158 C v = K(0.0);
159 INT k_L;
160 for (k_L = 0; k_L < ths->N_total; k_L++)
161 {
162 R omega = K2PI * ((R)(k_L - ths->N_total/2)) * ths->x[j];
163 v += f_hat[k_L] * (COS(omega) - II * SIN(omega));
164 }
165
166 f[j] = v;
167 }
168 }
169 else
170 {
171 /* multivariate case */
172 INT j;
173#ifdef _OPENMP
174 #pragma omp parallel for default(shared) private(j)
175#endif
176 for (j = 0; j < ths->M_total; j++)
177 {
178 C v = K(0.0);
179 R x[ths->d], omega, Omega[ths->d + 1];
180 INT t, t2, k_L, k[ths->d];
181 Omega[0] = K(0.0);
182 for (t = 0; t < ths->d; t++)
183 {
184 k[t] = -ths->N[t]/2;
185 x[t] = K2PI * ths->x[j * ths->d + t];
186 Omega[t+1] = ((R)k[t]) * x[t] + Omega[t];
187 }
188 omega = Omega[ths->d];
189
190 for (k_L = 0; k_L < ths->N_total; k_L++)
191 {
192 v += f_hat[k_L] * (COS(omega) - II * SIN(omega));
193 {
194 for (t = ths->d - 1; (t >= 1) && (k[t] == ths->N[t]/2 - 1); t--)
195 k[t]-= ths->N[t]-1;
196
197 k[t]++;
198
199 for (t2 = t; t2 < ths->d; t2++)
200 Omega[t2+1] = ((R)k[t2]) * x[t2] + Omega[t2];
201
202 omega = Omega[ths->d];
203 }
204 }
205
206 f[j] = v;
207 }
208 }
209}
210
211void X(adjoint_direct)(const X(plan) *ths)
212{
213 C *f_hat = (C*)ths->f_hat, *f = (C*)ths->f;
214
215 memset(f_hat, 0, (size_t)(ths->N_total) * sizeof(C));
216
217 if (ths->d == 1)
218 {
219 /* specialize for univariate case, rationale: faster */
220#ifdef _OPENMP
221 INT k_L;
222 #pragma omp parallel for default(shared) private(k_L)
223 for (k_L = 0; k_L < ths->N_total; k_L++)
224 {
225 INT j;
226 for (j = 0; j < ths->M_total; j++)
227 {
228 R omega = K2PI * ((R)(k_L - (ths->N_total/2))) * ths->x[j];
229 f_hat[k_L] += f[j] * (COS(omega) + II * SIN(omega));
230 }
231 }
232#else
233 INT j;
234 for (j = 0; j < ths->M_total; j++)
235 {
236 INT k_L;
237 for (k_L = 0; k_L < ths->N_total; k_L++)
238 {
239 R omega = K2PI * ((R)(k_L - ths->N_total / 2)) * ths->x[j];
240 f_hat[k_L] += f[j] * (COS(omega) + II * SIN(omega));
241 }
242 }
243#endif
244 }
245 else
246 {
247 /* multivariate case */
248 INT j, k_L;
249#ifdef _OPENMP
250 #pragma omp parallel for default(shared) private(j, k_L)
251 for (k_L = 0; k_L < ths->N_total; k_L++)
252 {
253 INT k[ths->d], k_temp, t;
254
255 k_temp = k_L;
256
257 for (t = ths->d - 1; t >= 0; t--)
258 {
259 k[t] = k_temp % ths->N[t] - ths->N[t]/2;
260 k_temp /= ths->N[t];
261 }
262
263 for (j = 0; j < ths->M_total; j++)
264 {
265 R omega = K(0.0);
266 for (t = 0; t < ths->d; t++)
267 omega += k[t] * K2PI * ths->x[j * ths->d + t];
268 f_hat[k_L] += f[j] * (COS(omega) + II * SIN(omega));
269 }
270 }
271#else
272 for (j = 0; j < ths->M_total; j++)
273 {
274 R x[ths->d], omega, Omega[ths->d+1];
275 INT t, t2, k[ths->d];
276 Omega[0] = K(0.0);
277 for (t = 0; t < ths->d; t++)
278 {
279 k[t] = -ths->N[t]/2;
280 x[t] = K2PI * ths->x[j * ths->d + t];
281 Omega[t+1] = ((R)k[t]) * x[t] + Omega[t];
282 }
283 omega = Omega[ths->d];
284 for (k_L = 0; k_L < ths->N_total; k_L++)
285 {
286 f_hat[k_L] += f[j] * (COS(omega) + II * SIN(omega));
287
288 for (t = ths->d-1; (t >= 1) && (k[t] == ths->N[t]/2-1); t--)
289 k[t]-= ths->N[t]-1;
290
291 k[t]++;
292
293 for (t2 = t; t2 < ths->d; t2++)
294 Omega[t2+1] = ((R)k[t2]) * x[t2] + Omega[t2];
295
296 omega = Omega[ths->d];
297 }
298 }
299#endif
300 }
301}
302
328static inline void uo(const X(plan) *ths, const INT j, INT *up, INT *op,
329 const INT act_dim)
330{
331 const R xj = ths->x[j * ths->d + act_dim];
332 INT c = LRINT(FLOOR(xj * (R)(ths->n[act_dim])));
333
334 (*up) = c - (ths->m);
335 (*op) = c + 1 + (ths->m);
336}
337
338static inline void uo2(INT *u, INT *o, const R x, const INT n, const INT m)
339{
340 INT c = LRINT(FLOOR(x * (R)(n)));
341
342 *u = (c - m + n) % n;
343 *o = (c + 1 + m + n) % n;
344}
345
346#define MACRO_D_compute_A \
347{ \
348 g_hat[k_plain[ths->d]] = f_hat[ks_plain[ths->d]] * c_phi_inv_k[ths->d]; \
349}
350
351#define MACRO_D_compute_T \
352{ \
353 f_hat[ks_plain[ths->d]] = g_hat[k_plain[ths->d]] * c_phi_inv_k[ths->d]; \
354}
355
356#define MACRO_D_init_result_A memset(g_hat, 0, (size_t)(ths->n_total) * sizeof(C));
357
358#define MACRO_D_init_result_T memset(f_hat, 0, (size_t)(ths->N_total) * sizeof(C));
359
360#define MACRO_with_PRE_PHI_HUT * ths->c_phi_inv[t2][ks[t2]];
361
362#define MACRO_without_PRE_PHI_HUT / (PHI_HUT(ths->n[t2],ks[t2]-(ths->N[t2]/2),t2));
363
364#define MACRO_init_k_ks \
365{ \
366 for (t = ths->d-1; 0 <= t; t--) \
367 { \
368 kp[t] = k[t] = 0; \
369 ks[t] = ths->N[t]/2; \
370 } \
371 t++; \
372}
373
374#define MACRO_update_c_phi_inv_k(which_one) \
375{ \
376 for (t2 = t; t2 < ths->d; t2++) \
377 { \
378 c_phi_inv_k[t2+1] = c_phi_inv_k[t2] MACRO_ ##which_one; \
379 ks_plain[t2+1] = ks_plain[t2]*ths->N[t2] + ks[t2]; \
380 k_plain[t2+1] = k_plain[t2]*ths->n[t2] + k[t2]; \
381 } \
382}
383
384#define MACRO_count_k_ks \
385{ \
386 for (t = ths->d-1; (t > 0) && (kp[t] == ths->N[t]-1); t--) \
387 { \
388 kp[t] = k[t] = 0; \
389 ks[t]= ths->N[t]/2; \
390 } \
391\
392 kp[t]++; k[t]++; ks[t]++; \
393 if(kp[t] == ths->N[t]/2) \
394 { \
395 k[t] = ths->n[t] - ths->N[t]/2; \
396 ks[t] = 0; \
397 } \
398} \
399
400/* sub routines for the fast transforms matrix vector multiplication with D, D^T */
401#define MACRO_D(which_one) \
402static inline void D_serial_ ## which_one (X(plan) *ths) \
403{ \
404 C *f_hat, *g_hat; /* local copy */ \
405 R c_phi_inv_k[ths->d+1]; /* postfix product of PHI_HUT */ \
406 INT t, t2; /* index dimensions */ \
407 INT k_L; /* plain index */ \
408 INT kp[ths->d]; /* multi index (simple) */ \
409 INT k[ths->d]; /* multi index in g_hat */ \
410 INT ks[ths->d]; /* multi index in f_hat, c_phi_inv*/ \
411 INT k_plain[ths->d+1]; /* postfix plain index */ \
412 INT ks_plain[ths->d+1]; /* postfix plain index */ \
413 \
414 f_hat = (C*)ths->f_hat; g_hat = (C*)ths->g_hat; \
415 MACRO_D_init_result_ ## which_one; \
416\
417 c_phi_inv_k[0] = K(1.0); \
418 k_plain[0] = 0; \
419 ks_plain[0] = 0; \
420\
421 MACRO_init_k_ks; \
422\
423 if (ths->flags & PRE_PHI_HUT) \
424 { \
425 for (k_L = 0; k_L < ths->N_total; k_L++) \
426 { \
427 MACRO_update_c_phi_inv_k(with_PRE_PHI_HUT); \
428 MACRO_D_compute_ ## which_one; \
429 MACRO_count_k_ks; \
430 } \
431 } \
432 else \
433 { \
434 for (k_L = 0; k_L < ths->N_total; k_L++) \
435 { \
436 MACRO_update_c_phi_inv_k(without_PRE_PHI_HUT); \
437 MACRO_D_compute_ ## which_one; \
438 MACRO_count_k_ks; \
439 } \
440 } \
441}
442
443#ifdef _OPENMP
444static inline void D_openmp_A(X(plan) *ths)
445{
446 C *f_hat, *g_hat;
447 INT k_L;
449 f_hat = (C*)ths->f_hat; g_hat = (C*)ths->g_hat;
450 memset(g_hat, 0, ths->n_total * sizeof(C));
451
452 if (ths->flags & PRE_PHI_HUT)
453 {
454 #pragma omp parallel for default(shared) private(k_L)
455 for (k_L = 0; k_L < ths->N_total; k_L++)
456 {
457 INT kp[ths->d]; //0..N-1
458 INT k[ths->d];
459 INT ks[ths->d];
460 R c_phi_inv_k_val = K(1.0);
461 INT k_plain_val = 0;
462 INT ks_plain_val = 0;
463 INT t;
464 INT k_temp = k_L;
465
466 for (t = ths->d-1; t >= 0; t--)
467 {
468 kp[t] = k_temp % ths->N[t];
469 if (kp[t] >= ths->N[t]/2)
470 k[t] = ths->n[t] - ths->N[t] + kp[t];
471 else
472 k[t] = kp[t];
473 ks[t] = (kp[t] + ths->N[t]/2) % ths->N[t];
474 k_temp /= ths->N[t];
475 }
476
477 for (t = 0; t < ths->d; t++)
478 {
479 c_phi_inv_k_val *= ths->c_phi_inv[t][ks[t]];
480 ks_plain_val = ks_plain_val*ths->N[t] + ks[t];
481 k_plain_val = k_plain_val*ths->n[t] + k[t];
482 }
483
484 g_hat[k_plain_val] = f_hat[ks_plain_val] * c_phi_inv_k_val;
485 } /* for(k_L) */
486 } /* if(PRE_PHI_HUT) */
487 else
488 {
489 #pragma omp parallel for default(shared) private(k_L)
490 for (k_L = 0; k_L < ths->N_total; k_L++)
491 {
492 INT kp[ths->d]; //0..N-1
493 INT k[ths->d];
494 INT ks[ths->d];
495 R c_phi_inv_k_val = K(1.0);
496 INT k_plain_val = 0;
497 INT ks_plain_val = 0;
498 INT t;
499 INT k_temp = k_L;
500
501 for (t = ths->d-1; t >= 0; t--)
502 {
503 kp[t] = k_temp % ths->N[t];
504 if (kp[t] >= ths->N[t]/2)
505 k[t] = ths->n[t] - ths->N[t] + kp[t];
506 else
507 k[t] = kp[t];
508 ks[t] = (kp[t] + ths->N[t]/2) % ths->N[t];
509 k_temp /= ths->N[t];
510 }
511
512 for (t = 0; t < ths->d; t++)
513 {
514 c_phi_inv_k_val /= (PHI_HUT(ths->n[t],ks[t]-(ths->N[t]/2),t));
515 ks_plain_val = ks_plain_val*ths->N[t] + ks[t];
516 k_plain_val = k_plain_val*ths->n[t] + k[t];
517 }
518
519 g_hat[k_plain_val] = f_hat[ks_plain_val] * c_phi_inv_k_val;
520 } /* for(k_L) */
521 } /* else(PRE_PHI_HUT) */
522}
523#endif
524
525#ifndef _OPENMP
526MACRO_D(A)
527#endif
528
529static inline void D_A(X(plan) *ths)
530{
531#ifdef _OPENMP
532 D_openmp_A(ths);
533#else
534 D_serial_A(ths);
535#endif
536}
537
538#ifdef _OPENMP
539static void D_openmp_T(X(plan) *ths)
540{
541 C *f_hat, *g_hat;
542 INT k_L;
544 f_hat = (C*)ths->f_hat; g_hat = (C*)ths->g_hat;
545 memset(f_hat, 0, ths->N_total * sizeof(C));
546
547 if (ths->flags & PRE_PHI_HUT)
548 {
549 #pragma omp parallel for default(shared) private(k_L)
550 for (k_L = 0; k_L < ths->N_total; k_L++)
551 {
552 INT kp[ths->d]; //0..N-1
553 INT k[ths->d];
554 INT ks[ths->d];
555 R c_phi_inv_k_val = K(1.0);
556 INT k_plain_val = 0;
557 INT ks_plain_val = 0;
558 INT t;
559 INT k_temp = k_L;
560
561 for (t = ths->d - 1; t >= 0; t--)
562 {
563 kp[t] = k_temp % ths->N[t];
564 if (kp[t] >= ths->N[t]/2)
565 k[t] = ths->n[t] - ths->N[t] + kp[t];
566 else
567 k[t] = kp[t];
568 ks[t] = (kp[t] + ths->N[t]/2) % ths->N[t];
569 k_temp /= ths->N[t];
570 }
571
572 for (t = 0; t < ths->d; t++)
573 {
574 c_phi_inv_k_val *= ths->c_phi_inv[t][ks[t]];
575 ks_plain_val = ks_plain_val*ths->N[t] + ks[t];
576 k_plain_val = k_plain_val*ths->n[t] + k[t];
577 }
578
579 f_hat[ks_plain_val] = g_hat[k_plain_val] * c_phi_inv_k_val;
580 } /* for(k_L) */
581 } /* if(PRE_PHI_HUT) */
582 else
583 {
584 #pragma omp parallel for default(shared) private(k_L)
585 for (k_L = 0; k_L < ths->N_total; k_L++)
586 {
587 INT kp[ths->d]; //0..N-1
588 INT k[ths->d];
589 INT ks[ths->d];
590 R c_phi_inv_k_val = K(1.0);
591 INT k_plain_val = 0;
592 INT ks_plain_val = 0;
593 INT t;
594 INT k_temp = k_L;
595
596 for (t = ths->d-1; t >= 0; t--)
597 {
598 kp[t] = k_temp % ths->N[t];
599 if (kp[t] >= ths->N[t]/2)
600 k[t] = ths->n[t] - ths->N[t] + kp[t];
601 else
602 k[t] = kp[t];
603 ks[t] = (kp[t] + ths->N[t]/2) % ths->N[t];
604 k_temp /= ths->N[t];
605 }
606
607 for (t = 0; t < ths->d; t++)
608 {
609 c_phi_inv_k_val /= (PHI_HUT(ths->n[t],ks[t]-(ths->N[t]/2),t));
610 ks_plain_val = ks_plain_val*ths->N[t] + ks[t];
611 k_plain_val = k_plain_val*ths->n[t] + k[t];
612 }
613
614 f_hat[ks_plain_val] = g_hat[k_plain_val] * c_phi_inv_k_val;
615 } /* for(k_L) */
616 } /* else(PRE_PHI_HUT) */
617}
618#endif
619
620#ifndef _OPENMP
621MACRO_D(T)
622#endif
623
624static void D_T(X(plan) *ths)
625{
626#ifdef _OPENMP
627 D_openmp_T(ths);
628#else
629 D_serial_T(ths);
630#endif
631}
632
633/* sub routines for the fast transforms matrix vector multiplication with B, B^T */
634#define MACRO_B_init_result_A memset(ths->f, 0, (size_t)(ths->M_total) * sizeof(C));
635#define MACRO_B_init_result_T memset(ths->g, 0, (size_t)(ths->n_total) * sizeof(C));
636
637#define MACRO_B_PRE_FULL_PSI_compute_A \
638{ \
639 (*fj) += ths->psi[ix] * g[ths->psi_index_g[ix]]; \
640}
641
642#define MACRO_B_PRE_FULL_PSI_compute_T \
643{ \
644 g[ths->psi_index_g[ix]] += ths->psi[ix] * (*fj); \
645}
646
647#define MACRO_B_compute_A \
648{ \
649 ths->f[j] += phi_prod[ths->d] * ths->g[ll_plain[ths->d]]; \
650}
651
652#define MACRO_B_compute_T \
653{ \
654 ths->g[ll_plain[ths->d]] += phi_prod[ths->d] * ths->f[j]; \
655}
656
657#define MACRO_with_FG_PSI fg_psi[t2][lj[t2]]
658
659#define MACRO_with_PRE_PSI ths->psi[(j*ths->d+t2) * (2*ths->m+2)+lj[t2]]
660
661#define MACRO_without_PRE_PSI_improved psij_const[t2 * (2*ths->m+2) + lj[t2]]
662
663#define MACRO_without_PRE_PSI PHI(ths->n[t2], ths->x[j*ths->d+t2] \
664 - ((R) (lj[t2]+u[t2]))/((R)ths->n[t2]), t2)
665
666#define MACRO_init_uo_l_lj_t \
667INT l_all[ths->d*(2*ths->m+2)]; \
668{ \
669 for (t = ths->d-1; t >= 0; t--) \
670 { \
671 uo(ths,j,&u[t],&o[t],t); \
672 INT lj_t; \
673 for (lj_t = 0; lj_t < 2*ths->m+2; lj_t++) \
674 l_all[t*(2*ths->m+2) + lj_t] = (u[t] + lj_t + ths->n[t]) % ths->n[t]; \
675 lj[t] = 0; \
676 } \
677 t++; \
678}
679
680#define MACRO_update_phi_prod_ll_plain(which_one) { \
681 for (t2 = t; t2 < ths->d; t2++) \
682 { \
683 phi_prod[t2+1] = phi_prod[t2] * MACRO_ ## which_one; \
684 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
685 } \
686}
687
688#define MACRO_count_uo_l_lj_t \
689{ \
690 for (t = ths->d-1; (t > 0) && (lj[t] == o[t]-u[t]); t--) \
691 { \
692 lj[t] = 0; \
693 } \
694 \
695 lj[t]++; \
696}
697
698#define MACRO_COMPUTE_with_PRE_PSI MACRO_with_PRE_PSI
699#define MACRO_COMPUTE_with_PRE_FG_PSI MACRO_with_FG_PSI
700#define MACRO_COMPUTE_with_FG_PSI MACRO_with_FG_PSI
701#define MACRO_COMPUTE_with_PRE_LIN_PSI MACRO_with_FG_PSI
702#define MACRO_COMPUTE_without_PRE_PSI MACRO_without_PRE_PSI_improved
703#define MACRO_COMPUTE_without_PRE_PSI_improved MACRO_without_PRE_PSI_improved
704
705#define MACRO_B_COMPUTE_ONE_NODE(whichone_AT,whichone_FLAGS) \
706 if (ths->d == 4) \
707 { \
708 INT l0, l1, l2, l3; \
709 for (l0 = 0; l0 < 2*ths->m+2; l0++) \
710 { \
711 lj[0] = l0; \
712 t2 = 0; \
713 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone_FLAGS; \
714 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
715 for (l1 = 0; l1 < 2*ths->m+2; l1++) \
716 { \
717 lj[1] = l1; \
718 t2 = 1; \
719 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone_FLAGS; \
720 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
721 for (l2 = 0; l2 < 2*ths->m+2; l2++) \
722 { \
723 lj[2] = l2; \
724 t2 = 2; \
725 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone_FLAGS; \
726 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
727 for (l3 = 0; l3 < 2*ths->m+2; l3++) \
728 { \
729 lj[3] = l3; \
730 t2 = 3; \
731 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone_FLAGS; \
732 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
733 \
734 MACRO_B_compute_ ## whichone_AT; \
735 } \
736 } \
737 } \
738 } \
739 } /* if(d==4) */ \
740 else if (ths->d == 5) \
741 { \
742 INT l0, l1, l2, l3, l4; \
743 for (l0 = 0; l0 < 2*ths->m+2; l0++) \
744 { \
745 lj[0] = l0; \
746 t2 = 0; \
747 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone_FLAGS; \
748 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
749 for (l1 = 0; l1 < 2*ths->m+2; l1++) \
750 { \
751 lj[1] = l1; \
752 t2 = 1; \
753 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone_FLAGS; \
754 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
755 for (l2 = 0; l2 < 2*ths->m+2; l2++) \
756 { \
757 lj[2] = l2; \
758 t2 = 2; \
759 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone_FLAGS; \
760 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
761 for (l3 = 0; l3 < 2*ths->m+2; l3++) \
762 { \
763 lj[3] = l3; \
764 t2 = 3; \
765 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone_FLAGS; \
766 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
767 for (l4 = 0; l4 < 2*ths->m+2; l4++) \
768 { \
769 lj[4] = l4; \
770 t2 = 4; \
771 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone_FLAGS; \
772 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
773 \
774 MACRO_B_compute_ ## whichone_AT; \
775 } \
776 } \
777 } \
778 } \
779 } \
780 } /* if(d==5) */ \
781 else { \
782 for (l_L = 0; l_L < lprod; l_L++) \
783 { \
784 MACRO_update_phi_prod_ll_plain(whichone_FLAGS); \
785 \
786 MACRO_B_compute_ ## whichone_AT; \
787 \
788 MACRO_count_uo_l_lj_t; \
789 } /* for(l_L) */ \
790 }
791
792#define MACRO_B(which_one) \
793static inline void B_serial_ ## which_one (X(plan) *ths) \
794{ \
795 INT lprod; /* 'regular bandwidth' of matrix B */ \
796 INT u[ths->d], o[ths->d]; /* multi band with respect to x_j */ \
797 INT t, t2; /* index dimensions */ \
798 INT k; /* index nodes */ \
799 INT l_L, ix; /* index one row of B */ \
800 INT lj[ths->d]; /* multi index 0<=lj<u+o+1 */ \
801 INT ll_plain[ths->d+1]; /* postfix plain index in g */ \
802 R phi_prod[ths->d+1]; /* postfix product of PHI */ \
803 R y[ths->d]; \
804 R fg_psi[ths->d][2*ths->m+2]; \
805 R fg_exp_l[ths->d][2*ths->m+2+1]; \
806 INT l_fg,lj_fg; \
807 R tmpEXP1, tmpEXP2, tmpEXP2sq, tmp1, tmp2, tmp3; \
808 R ip_w; \
809 INT ip_u; \
810 INT ip_s = ths->K/(ths->m+2); \
811 \
812 MACRO_B_init_result_ ## which_one; \
813 \
814 if (ths->flags & PRE_FULL_PSI) \
815 { \
816 INT j; \
817 C *f, *g; /* local copy */ \
818 C *fj; /* local copy */ \
819 f = (C*)ths->f; g = (C*)ths->g; \
820 \
821 for (ix = 0, j = 0, fj = f; j < ths->M_total; j++, fj++) \
822 { \
823 for (l_L = 0; l_L < ths->psi_index_f[j]; l_L++, ix++) \
824 { \
825 MACRO_B_PRE_FULL_PSI_compute_ ## which_one; \
826 } \
827 } \
828 return; \
829 } \
830\
831 phi_prod[0] = K(1.0); \
832 ll_plain[0] = 0; \
833\
834 for (t = 0, lprod = 1; t < ths->d; t++) \
835 lprod *= (2 * ths->m + 2); \
836\
837 if (ths->flags & PRE_PSI) \
838 { \
839 sort(ths); \
840 \
841 for (k = 0; k < ths->M_total; k++) \
842 { \
843 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k; \
844 \
845 MACRO_init_uo_l_lj_t; \
846 \
847 MACRO_B_COMPUTE_ONE_NODE(which_one,with_PRE_PSI); \
848 } /* for(j) */ \
849 return; \
850 } /* if(PRE_PSI) */ \
851 \
852 if (ths->flags & PRE_FG_PSI) \
853 { \
854 sort(ths); \
855 \
856 for(t2 = 0; t2 < ths->d; t2++) \
857 { \
858 tmpEXP2 = EXP(K(-1.0) / ths->b[t2]); \
859 tmpEXP2sq = tmpEXP2*tmpEXP2; \
860 tmp2 = K(1.0); \
861 tmp3 = K(1.0); \
862 fg_exp_l[t2][0] = K(1.0); \
863 for (lj_fg = 1; lj_fg <= (2 * ths->m + 2); lj_fg++) \
864 { \
865 tmp3 = tmp2*tmpEXP2; \
866 tmp2 *= tmpEXP2sq; \
867 fg_exp_l[t2][lj_fg] = fg_exp_l[t2][lj_fg-1] * tmp3; \
868 } \
869 } \
870 for (k = 0; k < ths->M_total; k++) \
871 { \
872 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k; \
873 \
874 MACRO_init_uo_l_lj_t; \
875 \
876 for (t2 = 0; t2 < ths->d; t2++) \
877 { \
878 fg_psi[t2][0] = ths->psi[2*(j*ths->d+t2)]; \
879 tmpEXP1 = ths->psi[2*(j*ths->d+t2)+1]; \
880 tmp1 = K(1.0); \
881 for (l_fg = u[t2]+1, lj_fg = 1; l_fg <= o[t2]; l_fg++, lj_fg++) \
882 { \
883 tmp1 *= tmpEXP1; \
884 fg_psi[t2][lj_fg] = fg_psi[t2][0]*tmp1*fg_exp_l[t2][lj_fg]; \
885 } \
886 } \
887 \
888 MACRO_B_COMPUTE_ONE_NODE(which_one,with_FG_PSI); \
889 } /* for(j) */ \
890 return; \
891 } /* if(PRE_FG_PSI) */ \
892 \
893 if (ths->flags & FG_PSI) \
894 { \
895 sort(ths); \
896 \
897 for (t2 = 0; t2 < ths->d; t2++) \
898 { \
899 tmpEXP2 = EXP(K(-1.0)/ths->b[t2]); \
900 tmpEXP2sq = tmpEXP2*tmpEXP2; \
901 tmp2 = K(1.0); \
902 tmp3 = K(1.0); \
903 fg_exp_l[t2][0] = K(1.0); \
904 for (lj_fg = 1; lj_fg <= (2*ths->m+2); lj_fg++) \
905 { \
906 tmp3 = tmp2*tmpEXP2; \
907 tmp2 *= tmpEXP2sq; \
908 fg_exp_l[t2][lj_fg] = fg_exp_l[t2][lj_fg-1]*tmp3; \
909 } \
910 } \
911 for (k = 0; k < ths->M_total; k++) \
912 { \
913 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k; \
914 \
915 MACRO_init_uo_l_lj_t; \
916 \
917 for (t2 = 0; t2 < ths->d; t2++) \
918 { \
919 fg_psi[t2][0] = (PHI(ths->n[t2], (ths->x[j*ths->d+t2] - ((R)u[t2])/((R)(ths->n[t2]))), t2));\
920 \
921 tmpEXP1 = EXP(K(2.0) * ((R)(ths->n[t2]) * ths->x[j * ths->d + t2] - (R)(u[t2])) \
922 /ths->b[t2]); \
923 tmp1 = K(1.0); \
924 for (l_fg = u[t2] + 1, lj_fg = 1; l_fg <= o[t2]; l_fg++, lj_fg++) \
925 { \
926 tmp1 *= tmpEXP1; \
927 fg_psi[t2][lj_fg] = fg_psi[t2][0]*tmp1*fg_exp_l[t2][lj_fg]; \
928 } \
929 } \
930 \
931 MACRO_B_COMPUTE_ONE_NODE(which_one,with_FG_PSI); \
932 } /* for(j) */ \
933 return; \
934 } /* if(FG_PSI) */ \
935 \
936 if (ths->flags & PRE_LIN_PSI) \
937 { \
938 sort(ths); \
939 \
940 for (k = 0; k<ths->M_total; k++) \
941 { \
942 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k; \
943 \
944 MACRO_init_uo_l_lj_t; \
945 \
946 for (t2 = 0; t2 < ths->d; t2++) \
947 { \
948 y[t2] = (((R)(ths->n[t2]) * ths->x[j * ths->d + t2] - (R)(u[t2])) \
949 * ((R)(ths->K))) / (R)(ths->m + 2); \
950 ip_u = LRINT(FLOOR(y[t2])); \
951 ip_w = y[t2]-ip_u; \
952 for (l_fg = u[t2], lj_fg = 0; l_fg <= o[t2]; l_fg++, lj_fg++) \
953 { \
954 fg_psi[t2][lj_fg] = ths->psi[(ths->K+1)*t2 + ABS(ip_u-lj_fg*ip_s)] \
955 * (1-ip_w) + ths->psi[(ths->K+1)*t2 + ABS(ip_u-lj_fg*ip_s+1)] \
956 * (ip_w); \
957 } \
958 } \
959 \
960 MACRO_B_COMPUTE_ONE_NODE(which_one,with_FG_PSI); \
961 } /* for(j) */ \
962 return; \
963 } /* if(PRE_LIN_PSI) */ \
964 \
965 sort(ths); \
966 \
967 /* no precomputed psi at all */ \
968 for (k = 0; k < ths->M_total; k++) \
969 { \
970 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k; \
971 \
972 R psij_const[ths->d * (2*ths->m+2)]; \
973 \
974 MACRO_init_uo_l_lj_t; \
975 \
976 for (t2 = 0; t2 < ths->d; t2++) \
977 { \
978 INT lj_t; \
979 for (lj_t = 0; lj_t < 2*ths->m+2; lj_t++) \
980 psij_const[t2 * (2*ths->m+2) + lj_t] = PHI(ths->n[t2], ths->x[j*ths->d+t2] \
981 - ((R) (lj_t+u[t2]))/((R)ths->n[t2]), t2); \
982 } \
983 \
984 MACRO_B_COMPUTE_ONE_NODE(which_one,without_PRE_PSI_improved); \
985 } /* for(j) */ \
986} /* nfft_B */ \
987
988#ifndef _OPENMP
989MACRO_B(A)
990#endif
991
992#ifdef _OPENMP
993#define MACRO_B_openmp_A_COMPUTE_BEFORE_LOOP_with_PRE_PSI
994#define MACRO_B_openmp_A_COMPUTE_UPDATE_with_PRE_PSI \
995 MACRO_update_phi_prod_ll_plain(with_PRE_PSI);
996
997#define MACRO_B_openmp_A_COMPUTE_INIT_FG_PSI \
998 for (t2 = 0; t2 < ths->d; t2++) \
999 { \
1000 INT lj_fg; \
1001 R tmpEXP2 = EXP(K(-1.0)/ths->b[t2]); \
1002 R tmpEXP2sq = tmpEXP2*tmpEXP2; \
1003 R tmp2 = K(1.0); \
1004 R tmp3 = K(1.0); \
1005 fg_exp_l[t2][0] = K(1.0); \
1006 for(lj_fg = 1; lj_fg <= (2*ths->m+2); lj_fg++) \
1007 { \
1008 tmp3 = tmp2*tmpEXP2; \
1009 tmp2 *= tmpEXP2sq; \
1010 fg_exp_l[t2][lj_fg] = fg_exp_l[t2][lj_fg-1]*tmp3; \
1011 } \
1012 }
1013#define MACRO_B_openmp_A_COMPUTE_BEFORE_LOOP_with_PRE_FG_PSI \
1014 for (t2 = 0; t2 < ths->d; t2++) \
1015 { \
1016 fg_psi[t2][0] = ths->psi[2*(j*ths->d+t2)]; \
1017 tmpEXP1 = ths->psi[2*(j*ths->d+t2)+1]; \
1018 tmp1 = K(1.0); \
1019 for (l_fg = u[t2]+1, lj_fg = 1; l_fg <= o[t2]; l_fg++, lj_fg++) \
1020 { \
1021 tmp1 *= tmpEXP1; \
1022 fg_psi[t2][lj_fg] = fg_psi[t2][0]*tmp1*fg_exp_l[t2][lj_fg]; \
1023 } \
1024 }
1025#define MACRO_B_openmp_A_COMPUTE_UPDATE_with_PRE_FG_PSI \
1026 MACRO_update_phi_prod_ll_plain(with_FG_PSI);
1027
1028#define MACRO_B_openmp_A_COMPUTE_BEFORE_LOOP_with_FG_PSI \
1029 for (t2 = 0; t2 < ths->d; t2++) \
1030 { \
1031 fg_psi[t2][0] = (PHI(ths->n[t2],(ths->x[j*ths->d+t2]-((R)u[t2])/((R)ths->n[t2])),t2)); \
1032 \
1033 tmpEXP1 = EXP(K(2.0)*(ths->n[t2]*ths->x[j*ths->d+t2] - u[t2]) \
1034 /ths->b[t2]); \
1035 tmp1 = K(1.0); \
1036 for (l_fg = u[t2] + 1, lj_fg = 1; l_fg <= o[t2]; l_fg++, lj_fg++) \
1037 { \
1038 tmp1 *= tmpEXP1; \
1039 fg_psi[t2][lj_fg] = fg_psi[t2][0]*tmp1*fg_exp_l[t2][lj_fg]; \
1040 } \
1041 }
1042#define MACRO_B_openmp_A_COMPUTE_UPDATE_with_FG_PSI \
1043 MACRO_update_phi_prod_ll_plain(with_FG_PSI);
1044
1045#define MACRO_B_openmp_A_COMPUTE_BEFORE_LOOP_with_PRE_LIN_PSI \
1046 for (t2 = 0; t2 < ths->d; t2++) \
1047 { \
1048 y[t2] = ((ths->n[t2]*ths->x[j*ths->d+t2]-(R)u[t2]) \
1049 * ((R)ths->K))/(ths->m+2); \
1050 ip_u = LRINT(FLOOR(y[t2])); \
1051 ip_w = y[t2]-ip_u; \
1052 for (l_fg = u[t2], lj_fg = 0; l_fg <= o[t2]; l_fg++, lj_fg++) \
1053 { \
1054 fg_psi[t2][lj_fg] = ths->psi[(ths->K+1)*t2 + ABS(ip_u-lj_fg*ip_s)] \
1055 * (1-ip_w) + ths->psi[(ths->K+1)*t2 + ABS(ip_u-lj_fg*ip_s+1)] \
1056 * (ip_w); \
1057 } \
1058 }
1059#define MACRO_B_openmp_A_COMPUTE_UPDATE_with_PRE_LIN_PSI \
1060 MACRO_update_phi_prod_ll_plain(with_FG_PSI);
1061
1062#define MACRO_B_openmp_A_COMPUTE_BEFORE_LOOP_without_PRE_PSI \
1063 for (t2 = 0; t2 < ths->d; t2++) \
1064 { \
1065 INT lj_t; \
1066 for (lj_t = 0; lj_t < 2*ths->m+2; lj_t++) \
1067 psij_const[t2 * (2*ths->m+2) + lj_t] = PHI(ths->n[t2], ths->x[j*ths->d+t2] \
1068 - ((R) (lj_t+u[t2]))/((R)ths->n[t2]), t2); \
1069 }
1070#define MACRO_B_openmp_A_COMPUTE_UPDATE_without_PRE_PSI \
1071 MACRO_update_phi_prod_ll_plain(without_PRE_PSI_improved);
1072
1073#define MACRO_B_openmp_A_COMPUTE(whichone) \
1074{ \
1075 INT u[ths->d], o[ths->d]; /* multi band with respect to x_j */ \
1076 INT l_L; /* index one row of B */ \
1077 INT lj[ths->d]; /* multi index 0<=lj<u+o+1 */ \
1078 INT ll_plain[ths->d+1]; /* postfix plain index in g */ \
1079 R phi_prod[ths->d+1]; /* postfix product of PHI */ \
1080 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k; \
1081 \
1082 phi_prod[0] = K(1.0); \
1083 ll_plain[0] = 0; \
1084 \
1085 MACRO_init_uo_l_lj_t; \
1086 \
1087 MACRO_B_openmp_A_COMPUTE_BEFORE_LOOP_ ##whichone \
1088 \
1089 if (ths->d == 4) \
1090 { \
1091 INT l0, l1, l2, l3; \
1092 for (l0 = 0; l0 < 2*ths->m+2; l0++) \
1093 { \
1094 lj[0] = l0; \
1095 t2 = 0; \
1096 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1097 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1098 for (l1 = 0; l1 < 2*ths->m+2; l1++) \
1099 { \
1100 lj[1] = l1; \
1101 t2 = 1; \
1102 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1103 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1104 for (l2 = 0; l2 < 2*ths->m+2; l2++) \
1105 { \
1106 lj[2] = l2; \
1107 t2 = 2; \
1108 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1109 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1110 for (l3 = 0; l3 < 2*ths->m+2; l3++) \
1111 { \
1112 lj[3] = l3; \
1113 t2 = 3; \
1114 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1115 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1116 \
1117 ths->f[j] += phi_prod[ths->d] * ths->g[ll_plain[ths->d]]; \
1118 } \
1119 } \
1120 } \
1121 } \
1122 } /* if(d==4) */ \
1123 else if (ths->d == 5) \
1124 { \
1125 INT l0, l1, l2, l3, l4; \
1126 for (l0 = 0; l0 < 2*ths->m+2; l0++) \
1127 { \
1128 lj[0] = l0; \
1129 t2 = 0; \
1130 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1131 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1132 for (l1 = 0; l1 < 2*ths->m+2; l1++) \
1133 { \
1134 lj[1] = l1; \
1135 t2 = 1; \
1136 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1137 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1138 for (l2 = 0; l2 < 2*ths->m+2; l2++) \
1139 { \
1140 lj[2] = l2; \
1141 t2 = 2; \
1142 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1143 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1144 for (l3 = 0; l3 < 2*ths->m+2; l3++) \
1145 { \
1146 lj[3] = l3; \
1147 t2 = 3; \
1148 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1149 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1150 for (l4 = 0; l4 < 2*ths->m+2; l4++) \
1151 { \
1152 lj[4] = l4; \
1153 t2 = 4; \
1154 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1155 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1156 \
1157 ths->f[j] += phi_prod[ths->d] * ths->g[ll_plain[ths->d]]; \
1158 } \
1159 } \
1160 } \
1161 } \
1162 } \
1163 } /* if(d==5) */ \
1164 else { \
1165 for (l_L = 0; l_L < lprod; l_L++) \
1166 { \
1167 MACRO_B_openmp_A_COMPUTE_UPDATE_ ##whichone \
1168 \
1169 ths->f[j] += phi_prod[ths->d] * ths->g[ll_plain[ths->d]]; \
1170 \
1171 MACRO_count_uo_l_lj_t; \
1172 } /* for(l_L) */ \
1173 } \
1174}
1175
1176static inline void B_openmp_A (X(plan) *ths)
1177{
1178 INT lprod; /* 'regular bandwidth' of matrix B */
1179 INT k;
1180
1181 memset(ths->f, 0, ths->M_total * sizeof(C));
1182
1183 for (k = 0, lprod = 1; k < ths->d; k++)
1184 lprod *= (2*ths->m+2);
1185
1186 if (ths->flags & PRE_FULL_PSI)
1187 {
1188 #pragma omp parallel for default(shared) private(k)
1189 for (k = 0; k < ths->M_total; k++)
1190 {
1191 INT l;
1192 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
1193 ths->f[j] = K(0.0);
1194 for (l = 0; l < lprod; l++)
1195 ths->f[j] += ths->psi[j*lprod+l] * ths->g[ths->psi_index_g[j*lprod+l]];
1196 }
1197 return;
1198 }
1199
1200 if (ths->flags & PRE_PSI)
1201 {
1202 #pragma omp parallel for default(shared) private(k)
1203 for (k = 0; k < ths->M_total; k++)
1204 {
1205 INT t, t2; /* index dimensions */
1206 MACRO_B_openmp_A_COMPUTE(with_PRE_PSI);
1207 } /* for(j) */
1208 return;
1209 } /* if(PRE_PSI) */
1210
1211 if (ths->flags & PRE_FG_PSI)
1212 {
1213 INT t, t2; /* index dimensions */
1214 R fg_exp_l[ths->d][2*ths->m+2+1];
1215
1216 MACRO_B_openmp_A_COMPUTE_INIT_FG_PSI
1217
1218 #pragma omp parallel for default(shared) private(k,t,t2)
1219 for (k = 0; k < ths->M_total; k++)
1220 {
1221 R fg_psi[ths->d][2*ths->m+2];
1222 R tmpEXP1, tmp1;
1223 INT l_fg,lj_fg;
1224
1225 MACRO_B_openmp_A_COMPUTE(with_PRE_FG_PSI);
1226 } /* for(j) */
1227 return;
1228 } /* if(PRE_FG_PSI) */
1229
1230 if (ths->flags & FG_PSI)
1231 {
1232 INT t, t2; /* index dimensions */
1233 R fg_exp_l[ths->d][2*ths->m+2+1];
1234
1235 sort(ths);
1236
1237 MACRO_B_openmp_A_COMPUTE_INIT_FG_PSI
1238
1239 #pragma omp parallel for default(shared) private(k,t,t2)
1240 for (k = 0; k < ths->M_total; k++)
1241 {
1242 R fg_psi[ths->d][2*ths->m+2];
1243 R tmpEXP1, tmp1;
1244 INT l_fg,lj_fg;
1245
1246 MACRO_B_openmp_A_COMPUTE(with_FG_PSI);
1247 } /* for(j) */
1248 return;
1249 } /* if(FG_PSI) */
1250
1251 if (ths->flags & PRE_LIN_PSI)
1252 {
1253 sort(ths);
1254
1255 #pragma omp parallel for default(shared) private(k)
1256 for (k = 0; k<ths->M_total; k++)
1257 {
1258 INT t, t2; /* index dimensions */
1259 R y[ths->d];
1260 R fg_psi[ths->d][2*ths->m+2];
1261 INT l_fg,lj_fg;
1262 R ip_w;
1263 INT ip_u;
1264 INT ip_s = ths->K/(ths->m+2);
1265
1266 MACRO_B_openmp_A_COMPUTE(with_PRE_LIN_PSI);
1267 } /* for(j) */
1268 return;
1269 } /* if(PRE_LIN_PSI) */
1270
1271 /* no precomputed psi at all */
1272 sort(ths);
1273
1274 #pragma omp parallel for default(shared) private(k)
1275 for (k = 0; k < ths->M_total; k++)
1276 {
1277 INT t, t2; /* index dimensions */
1278 R psij_const[ths->d * (2*ths->m+2)];
1279
1280 MACRO_B_openmp_A_COMPUTE(without_PRE_PSI);
1281 } /* for(j) */
1282}
1283#endif
1284
1285static void B_A(X(plan) *ths)
1286{
1287#ifdef _OPENMP
1288 B_openmp_A(ths);
1289#else
1290 B_serial_A(ths);
1291#endif
1292}
1293
1294#ifdef _OPENMP
1310static inline INT index_x_binary_search(const INT *ar_x, const INT len, const INT key)
1311{
1312 INT left = 0, right = len - 1;
1313
1314 if (len == 1)
1315 return 0;
1316
1317 while (left < right - 1)
1318 {
1319 INT i = (left + right) / 2;
1320 if (ar_x[2*i] >= key)
1321 right = i;
1322 else if (ar_x[2*i] < key)
1323 left = i;
1324 }
1325
1326 if (ar_x[2*left] < key && left != len-1)
1327 return left+1;
1328
1329 return left;
1330}
1331#endif
1332
1333#ifdef _OPENMP
1349static void nfft_adjoint_B_omp_blockwise_init(INT *my_u0, INT *my_o0,
1350 INT *min_u_a, INT *max_u_a, INT *min_u_b, INT *max_u_b, const INT d,
1351 const INT *n, const INT m)
1352{
1353 const INT n0 = n[0];
1354 INT k;
1355 INT nthreads = omp_get_num_threads();
1356 INT nthreads_used = MIN(nthreads, n0);
1357 INT size_per_thread = n0 / nthreads_used;
1358 INT size_left = n0 - size_per_thread * nthreads_used;
1359 INT size_g[nthreads_used];
1360 INT offset_g[nthreads_used];
1361 INT my_id = omp_get_thread_num();
1362 INT n_prod_rest = 1;
1363
1364 for (k = 1; k < d; k++)
1365 n_prod_rest *= n[k];
1366
1367 *min_u_a = -1;
1368 *max_u_a = -1;
1369 *min_u_b = -1;
1370 *max_u_b = -1;
1371 *my_u0 = -1;
1372 *my_o0 = -1;
1373
1374 if (my_id < nthreads_used)
1375 {
1376 const INT m22 = 2 * m + 2;
1377
1378 offset_g[0] = 0;
1379 for (k = 0; k < nthreads_used; k++)
1380 {
1381 if (k > 0)
1382 offset_g[k] = offset_g[k-1] + size_g[k-1];
1383 size_g[k] = size_per_thread;
1384 if (size_left > 0)
1385 {
1386 size_g[k]++;
1387 size_left--;
1388 }
1389 }
1390
1391 *my_u0 = offset_g[my_id];
1392 *my_o0 = offset_g[my_id] + size_g[my_id] - 1;
1393
1394 if (nthreads_used > 1)
1395 {
1396 *max_u_a = n_prod_rest*(offset_g[my_id] + size_g[my_id]) - 1;
1397 *min_u_a = n_prod_rest*(offset_g[my_id] - m22 + 1);
1398 }
1399 else
1400 {
1401 *min_u_a = 0;
1402 *max_u_a = n_prod_rest * n0 - 1;
1403 }
1404
1405 if (*min_u_a < 0)
1406 {
1407 *min_u_b = n_prod_rest * (offset_g[my_id] - m22 + 1 + n0);
1408 *max_u_b = n_prod_rest * n0 - 1;
1409 *min_u_a = 0;
1410 }
1411
1412 if (*min_u_b != -1 && *min_u_b <= *max_u_a)
1413 {
1414 *max_u_a = *max_u_b;
1415 *min_u_b = -1;
1416 *max_u_b = -1;
1417 }
1418#ifdef OMP_ASSERT
1419 assert(*min_u_a <= *max_u_a);
1420 assert(*min_u_b <= *max_u_b);
1421 assert(*min_u_b == -1 || *max_u_a < *min_u_b);
1422#endif
1423 }
1424}
1425#endif
1426
1435static void nfft_adjoint_B_compute_full_psi(C *g, const INT *psi_index_g,
1436 const R *psi, const C *f, const INT M, const INT d, const INT *n,
1437 const INT m, const unsigned flags, const INT *index_x)
1438{
1439 INT k;
1440 INT lprod;
1441#ifdef _OPENMP
1442 INT lprod_m1;
1443#endif
1444#ifndef _OPENMP
1445 UNUSED(n);
1446#endif
1447 {
1448 INT t;
1449 for(t = 0, lprod = 1; t < d; t++)
1450 lprod *= 2 * m + 2;
1451 }
1452#ifdef _OPENMP
1453 lprod_m1 = lprod / (2 * m + 2);
1454#endif
1455
1456#ifdef _OPENMP
1457 if (flags & NFFT_OMP_BLOCKWISE_ADJOINT)
1458 {
1459 #pragma omp parallel private(k)
1460 {
1461 INT my_u0, my_o0, min_u_a, max_u_a, min_u_b, max_u_b;
1462 const INT *ar_x = index_x;
1463 INT n_prod_rest = 1;
1464
1465 for (k = 1; k < d; k++)
1466 n_prod_rest *= n[k];
1467
1468 nfft_adjoint_B_omp_blockwise_init(&my_u0, &my_o0, &min_u_a, &max_u_a, &min_u_b, &max_u_b, d, n, m);
1469
1470 if (min_u_a != -1)
1471 {
1472 k = index_x_binary_search(ar_x, M, min_u_a);
1473#ifdef OMP_ASSERT
1474 assert(ar_x[2*k] >= min_u_a || k == M-1);
1475 if (k > 0)
1476 assert(ar_x[2*k-2] < min_u_a);
1477#endif
1478 while (k < M)
1479 {
1480 INT l0, lrest;
1481 INT u_prod = ar_x[2*k];
1482 INT j = ar_x[2*k+1];
1483
1484 if (u_prod < min_u_a || u_prod > max_u_a)
1485 break;
1486
1487 for (l0 = 0; l0 < 2 * m + 2; l0++)
1488 {
1489 const INT start_index = psi_index_g[j * lprod + l0 * lprod_m1];
1490
1491 if (start_index < my_u0 * n_prod_rest || start_index > (my_o0+1) * n_prod_rest - 1)
1492 continue;
1493
1494 for (lrest = 0; lrest < lprod_m1; lrest++)
1495 {
1496 const INT l = l0 * lprod_m1 + lrest;
1497 g[psi_index_g[j * lprod + l]] += psi[j * lprod + l] * f[j];
1498 }
1499 }
1500
1501 k++;
1502 }
1503 }
1504
1505 if (min_u_b != -1)
1506 {
1507 k = index_x_binary_search(ar_x, M, min_u_b);
1508#ifdef OMP_ASSERT
1509 assert(ar_x[2*k] >= min_u_b || k == M-1);
1510 if (k > 0)
1511 assert(ar_x[2*k-2] < min_u_b);
1512#endif
1513 while (k < M)
1514 {
1515 INT l0, lrest;
1516 INT u_prod = ar_x[2*k];
1517 INT j = ar_x[2*k+1];
1518
1519 if (u_prod < min_u_b || u_prod > max_u_b)
1520 break;
1521
1522 for (l0 = 0; l0 < 2 * m + 2; l0++)
1523 {
1524 const INT start_index = psi_index_g[j * lprod + l0 * lprod_m1];
1525
1526 if (start_index < my_u0 * n_prod_rest || start_index > (my_o0+1) * n_prod_rest - 1)
1527 continue;
1528 for (lrest = 0; lrest < lprod_m1; lrest++)
1529 {
1530 const INT l = l0 * lprod_m1 + lrest;
1531 g[psi_index_g[j * lprod + l]] += psi[j * lprod + l] * f[j];
1532 }
1533 }
1534
1535 k++;
1536 }
1537 }
1538 } /* omp parallel */
1539 return;
1540 } /* if(NFFT_OMP_BLOCKWISE_ADJOINT) */
1541#endif
1542
1543#ifdef _OPENMP
1544 #pragma omp parallel for default(shared) private(k)
1545#endif
1546 for (k = 0; k < M; k++)
1547 {
1548 INT l;
1549 INT j = (flags & NFFT_SORT_NODES) ? index_x[2*k+1] : k;
1550
1551 for (l = 0; l < lprod; l++)
1552 {
1553#ifdef _OPENMP
1554 C val = psi[j * lprod + l] * f[j];
1555 C *gref = g + psi_index_g[j * lprod + l];
1556 R *gref_real = (R*) gref;
1557
1558 #pragma omp atomic
1559 gref_real[0] += CREAL(val);
1560
1561 #pragma omp atomic
1562 gref_real[1] += CIMAG(val);
1563#else
1564 g[psi_index_g[j * lprod + l]] += psi[j * lprod + l] * f[j];
1565#endif
1566 }
1567 }
1568}
1569
1570#ifndef _OPENMP
1571MACRO_B(T)
1572#endif
1573
1574
1575#ifdef _OPENMP
1576
1577#ifdef OMP_ASSERT
1578#define MACRO_adjoint_nd_B_OMP_BLOCKWISE_ASSERT_A \
1579{ \
1580 assert(ar_x[2*k] >= min_u_a || k == M-1); \
1581 if (k > 0) \
1582 assert(ar_x[2*k-2] < min_u_a); \
1583}
1584#else
1585#define MACRO_adjoint_nd_B_OMP_BLOCKWISE_ASSERT_A
1586#endif
1587
1588#ifdef OMP_ASSERT
1589#define MACRO_adjoint_nd_B_OMP_BLOCKWISE_ASSERT_B \
1590{ \
1591 assert(ar_x[2*k] >= min_u_b || k == M-1); \
1592 if (k > 0) \
1593 assert(ar_x[2*k-2] < min_u_b); \
1594}
1595#else
1596#define MACRO_adjoint_nd_B_OMP_BLOCKWISE_ASSERT_B
1597#endif
1598
1599#define MACRO_adjoint_nd_B_OMP_COMPUTE_BEFORE_LOOP_with_PRE_PSI
1600#define MACRO_adjoint_nd_B_OMP_COMPUTE_UPDATE_with_PRE_PSI \
1601 MACRO_update_phi_prod_ll_plain(with_PRE_PSI);
1602
1603#define MACRO_adjoint_nd_B_OMP_COMPUTE_BEFORE_LOOP_with_PRE_FG_PSI \
1604 R fg_psi[ths->d][2*ths->m+2]; \
1605 R tmpEXP1, tmp1; \
1606 INT l_fg,lj_fg; \
1607 for (t2 = 0; t2 < ths->d; t2++) \
1608 { \
1609 fg_psi[t2][0] = ths->psi[2*(j*ths->d+t2)]; \
1610 tmpEXP1 = ths->psi[2*(j*ths->d+t2)+1]; \
1611 tmp1 = K(1.0); \
1612 for (l_fg = u[t2]+1, lj_fg = 1; l_fg <= o[t2]; l_fg++, lj_fg++) \
1613 { \
1614 tmp1 *= tmpEXP1; \
1615 fg_psi[t2][lj_fg] = fg_psi[t2][0]*tmp1*fg_exp_l[t2][lj_fg]; \
1616 } \
1617 }
1618#define MACRO_adjoint_nd_B_OMP_COMPUTE_UPDATE_with_PRE_FG_PSI \
1619 MACRO_update_phi_prod_ll_plain(with_FG_PSI);
1620
1621#define MACRO_adjoint_nd_B_OMP_COMPUTE_BEFORE_LOOP_with_FG_PSI \
1622 R fg_psi[ths->d][2*ths->m+2]; \
1623 R tmpEXP1, tmp1; \
1624 INT l_fg,lj_fg; \
1625 for (t2 = 0; t2 < ths->d; t2++) \
1626 { \
1627 fg_psi[t2][0] = (PHI(ths->n[t2],(ths->x[j*ths->d+t2]-((R)u[t2])/((R)ths->n[t2])),t2)); \
1628 \
1629 tmpEXP1 = EXP(K(2.0)*((R)ths->n[t2]*ths->x[j*ths->d+t2] - (R)u[t2]) \
1630 /ths->b[t2]); \
1631 tmp1 = K(1.0); \
1632 for (l_fg = u[t2] + 1, lj_fg = 1; l_fg <= o[t2]; l_fg++, lj_fg++) \
1633 { \
1634 tmp1 *= tmpEXP1; \
1635 fg_psi[t2][lj_fg] = fg_psi[t2][0]*tmp1*fg_exp_l[t2][lj_fg]; \
1636 } \
1637 }
1638#define MACRO_adjoint_nd_B_OMP_COMPUTE_UPDATE_with_FG_PSI \
1639 MACRO_update_phi_prod_ll_plain(with_FG_PSI);
1640
1641#define MACRO_adjoint_nd_B_OMP_COMPUTE_BEFORE_LOOP_with_PRE_LIN_PSI \
1642 R y[ths->d]; \
1643 R fg_psi[ths->d][2*ths->m+2]; \
1644 INT l_fg,lj_fg; \
1645 R ip_w; \
1646 INT ip_u; \
1647 INT ip_s = ths->K/(ths->m+2); \
1648 for (t2 = 0; t2 < ths->d; t2++) \
1649 { \
1650 y[t2] = ((((R)ths->n[t2])*ths->x[j*ths->d+t2]-(R)u[t2]) \
1651 * ((R)ths->K))/((R)ths->m+2); \
1652 ip_u = LRINT(FLOOR(y[t2])); \
1653 ip_w = y[t2]-ip_u; \
1654 for (l_fg = u[t2], lj_fg = 0; l_fg <= o[t2]; l_fg++, lj_fg++) \
1655 { \
1656 fg_psi[t2][lj_fg] = ths->psi[(ths->K+1)*t2 + ABS(ip_u-lj_fg*ip_s)] \
1657 * (1-ip_w) + ths->psi[(ths->K+1)*t2 + ABS(ip_u-lj_fg*ip_s+1)] \
1658 * (ip_w); \
1659 } \
1660 }
1661#define MACRO_adjoint_nd_B_OMP_COMPUTE_UPDATE_with_PRE_LIN_PSI \
1662 MACRO_update_phi_prod_ll_plain(with_FG_PSI);
1663
1664#define MACRO_adjoint_nd_B_OMP_COMPUTE_BEFORE_LOOP_without_PRE_PSI \
1665 R psij_const[ths->d * (2*ths->m+2)]; \
1666 for (t2 = 0; t2 < ths->d; t2++) \
1667 { \
1668 INT lj_t; \
1669 for (lj_t = 0; lj_t < 2*ths->m+2; lj_t++) \
1670 psij_const[t2 * (2*ths->m+2) + lj_t] = PHI(ths->n[t2], ths->x[j*ths->d+t2] \
1671 - ((R) (lj_t+u[t2]))/((R)ths->n[t2]), t2); \
1672 }
1673#define MACRO_adjoint_nd_B_OMP_COMPUTE_UPDATE_without_PRE_PSI \
1674 MACRO_update_phi_prod_ll_plain(without_PRE_PSI_improved);
1675
1676#define MACRO_adjoint_nd_B_OMP_BLOCKWISE_COMPUTE(whichone) \
1677{ \
1678 INT u[ths->d], o[ths->d]; /* multi band with respect to x_j */ \
1679 INT t, t2; /* index dimensions */ \
1680 INT l_L; /* index one row of B */ \
1681 INT lj[ths->d]; /* multi index 0<=lj<u+o+1 */ \
1682 INT ll_plain[ths->d+1]; /* postfix plain index in g */ \
1683 R phi_prod[ths->d+1]; /* postfix product of PHI */ \
1684 \
1685 phi_prod[0] = K(1.0); \
1686 ll_plain[0] = 0; \
1687 \
1688 MACRO_init_uo_l_lj_t; \
1689 \
1690 MACRO_adjoint_nd_B_OMP_COMPUTE_BEFORE_LOOP_ ##whichone \
1691 \
1692 if (ths->d == 4) \
1693 { \
1694 INT l0, l1, l2, l3; \
1695 for (l0 = 0; l0 < 2*ths->m+2; l0++) \
1696 { \
1697 lj[0] = l0; \
1698 t2 = 0; \
1699 if (l_all[lj[0]] < my_u0 || l_all[lj[0]] > my_o0) \
1700 continue; \
1701 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1702 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1703 for (l1 = 0; l1 < 2*ths->m+2; l1++) \
1704 { \
1705 lj[1] = l1; \
1706 t2 = 1; \
1707 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1708 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1709 for (l2 = 0; l2 < 2*ths->m+2; l2++) \
1710 { \
1711 lj[2] = l2; \
1712 t2 = 2; \
1713 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1714 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1715 for (l3 = 0; l3 < 2*ths->m+2; l3++) \
1716 { \
1717 lj[3] = l3; \
1718 t2 = 3; \
1719 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1720 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1721 \
1722 ths->g[ll_plain[ths->d]] += phi_prod[ths->d] * ths->f[j]; \
1723 } \
1724 } \
1725 } \
1726 } \
1727 } /* if(d==4) */ \
1728 else if (ths->d == 5) \
1729 { \
1730 INT l0, l1, l2, l3, l4; \
1731 for (l0 = 0; l0 < 2*ths->m+2; l0++) \
1732 { \
1733 lj[0] = l0; \
1734 t2 = 0; \
1735 if (l_all[lj[0]] < my_u0 || l_all[lj[0]] > my_o0) \
1736 continue; \
1737 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1738 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1739 for (l1 = 0; l1 < 2*ths->m+2; l1++) \
1740 { \
1741 lj[1] = l1; \
1742 t2 = 1; \
1743 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1744 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1745 for (l2 = 0; l2 < 2*ths->m+2; l2++) \
1746 { \
1747 lj[2] = l2; \
1748 t2 = 2; \
1749 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1750 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1751 for (l3 = 0; l3 < 2*ths->m+2; l3++) \
1752 { \
1753 lj[3] = l3; \
1754 t2 = 3; \
1755 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1756 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1757 for (l4 = 0; l4 < 2*ths->m+2; l4++) \
1758 { \
1759 lj[4] = l4; \
1760 t2 = 4; \
1761 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1762 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1763 \
1764 ths->g[ll_plain[ths->d]] += phi_prod[ths->d] * ths->f[j]; \
1765 } \
1766 } \
1767 } \
1768 } \
1769 } \
1770 } /* if(d==5) */ \
1771 else { \
1772 l_L = 0; \
1773 while (l_L < lprod) \
1774 { \
1775 if (t == 0 && (l_all[lj[0]] < my_u0 || l_all[lj[0]] > my_o0)) \
1776 { \
1777 lj[0]++; \
1778 l_L += lprodrest; \
1779 continue; \
1780 } \
1781 MACRO_adjoint_nd_B_OMP_COMPUTE_UPDATE_ ##whichone \
1782 ths->g[ll_plain[ths->d]] += phi_prod[ths->d] * ths->f[j]; \
1783 MACRO_count_uo_l_lj_t; \
1784 l_L++; \
1785 } /* for(l_L) */ \
1786 } \
1787}
1788
1789#define MACRO_adjoint_nd_B_OMP_BLOCKWISE(whichone) \
1790{ \
1791 if (ths->flags & NFFT_OMP_BLOCKWISE_ADJOINT) \
1792 { \
1793 INT lprodrest = 1; \
1794 for (k = 1; k < ths->d; k++) \
1795 lprodrest *= (2*ths->m+2); \
1796 _Pragma("omp parallel private(k)") \
1797 { \
1798 INT my_u0, my_o0, min_u_a, max_u_a, min_u_b, max_u_b; \
1799 INT *ar_x = ths->index_x; \
1800 \
1801 nfft_adjoint_B_omp_blockwise_init(&my_u0, &my_o0, &min_u_a, &max_u_a, \
1802 &min_u_b, &max_u_b, ths->d, ths->n, ths->m); \
1803 \
1804 if (min_u_a != -1) \
1805 { \
1806 k = index_x_binary_search(ar_x, ths->M_total, min_u_a); \
1807 \
1808 MACRO_adjoint_nd_B_OMP_BLOCKWISE_ASSERT_A \
1809 \
1810 while (k < ths->M_total) \
1811 { \
1812 INT u_prod = ar_x[2*k]; \
1813 INT j = ar_x[2*k+1]; \
1814 \
1815 if (u_prod < min_u_a || u_prod > max_u_a) \
1816 break; \
1817 \
1818 MACRO_adjoint_nd_B_OMP_BLOCKWISE_COMPUTE(whichone) \
1819 \
1820 k++; \
1821 } \
1822 } \
1823 \
1824 if (min_u_b != -1) \
1825 { \
1826 INT k = index_x_binary_search(ar_x, ths->M_total, min_u_b); \
1827 \
1828 MACRO_adjoint_nd_B_OMP_BLOCKWISE_ASSERT_B \
1829 \
1830 while (k < ths->M_total) \
1831 { \
1832 INT u_prod = ar_x[2*k]; \
1833 INT j = ar_x[2*k+1]; \
1834 \
1835 if (u_prod < min_u_b || u_prod > max_u_b) \
1836 break; \
1837 \
1838 MACRO_adjoint_nd_B_OMP_BLOCKWISE_COMPUTE(whichone) \
1839 \
1840 k++; \
1841 } \
1842 } \
1843 } /* omp parallel */ \
1844 return; \
1845 } /* if(NFFT_OMP_BLOCKWISE_ADJOINT) */ \
1846}
1847
1848#define MACRO_adjoint_nd_B_OMP_COMPUTE(whichone) \
1849{ \
1850 INT u[ths->d], o[ths->d]; /* multi band with respect to x_j */ \
1851 INT l_L; /* index one row of B */ \
1852 INT lj[ths->d]; /* multi index 0<=lj<u+o+1 */ \
1853 INT ll_plain[ths->d+1]; /* postfix plain index in g */ \
1854 R phi_prod[ths->d+1]; /* postfix product of PHI */ \
1855 \
1856 phi_prod[0] = K(1.0); \
1857 ll_plain[0] = 0; \
1858 \
1859 MACRO_init_uo_l_lj_t; \
1860 \
1861 MACRO_adjoint_nd_B_OMP_COMPUTE_BEFORE_LOOP_ ## whichone \
1862 \
1863 if (ths->d == 4) \
1864 { \
1865 INT l0, l1, l2, l3; \
1866 for (l0 = 0; l0 < 2*ths->m+2; l0++) \
1867 { \
1868 lj[0] = l0; \
1869 t2 = 0; \
1870 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1871 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1872 for (l1 = 0; l1 < 2*ths->m+2; l1++) \
1873 { \
1874 lj[1] = l1; \
1875 t2 = 1; \
1876 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1877 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1878 for (l2 = 0; l2 < 2*ths->m+2; l2++) \
1879 { \
1880 lj[2] = l2; \
1881 t2 = 2; \
1882 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1883 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1884 for (l3 = 0; l3 < 2*ths->m+2; l3++) \
1885 { \
1886 lj[3] = l3; \
1887 t2 = 3; \
1888 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1889 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1890 \
1891 C *lhs = ths->g + ll_plain[ths->d]; \
1892 R *lhs_real = (R*)lhs; \
1893 C val = phi_prod[ths->d] * ths->f[j]; \
1894 \
1895 _Pragma("omp atomic") \
1896 lhs_real[0] += CREAL(val); \
1897 \
1898 _Pragma("omp atomic") \
1899 lhs_real[1] += CIMAG(val); \
1900 } \
1901 } \
1902 } \
1903 } \
1904 } /* if(d==4) */ \
1905 else if (ths->d == 5) \
1906 { \
1907 INT l0, l1, l2, l3, l4; \
1908 for (l0 = 0; l0 < 2*ths->m+2; l0++) \
1909 { \
1910 lj[0] = l0; \
1911 t2 = 0; \
1912 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1913 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1914 for (l1 = 0; l1 < 2*ths->m+2; l1++) \
1915 { \
1916 lj[1] = l1; \
1917 t2 = 1; \
1918 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1919 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1920 for (l2 = 0; l2 < 2*ths->m+2; l2++) \
1921 { \
1922 lj[2] = l2; \
1923 t2 = 2; \
1924 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1925 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1926 for (l3 = 0; l3 < 2*ths->m+2; l3++) \
1927 { \
1928 lj[3] = l3; \
1929 t2 = 3; \
1930 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1931 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1932 for (l4 = 0; l4 < 2*ths->m+2; l4++) \
1933 { \
1934 lj[4] = l4; \
1935 t2 = 4; \
1936 phi_prod[t2+1] = phi_prod[t2] * MACRO_COMPUTE_ ## whichone; \
1937 ll_plain[t2+1] = ll_plain[t2] * ths->n[t2] + l_all[t2*(2*ths->m+2) + lj[t2]]; \
1938 \
1939 C *lhs = ths->g + ll_plain[ths->d]; \
1940 R *lhs_real = (R*)lhs; \
1941 C val = phi_prod[ths->d] * ths->f[j]; \
1942 \
1943 _Pragma("omp atomic") \
1944 lhs_real[0] += CREAL(val); \
1945 \
1946 _Pragma("omp atomic") \
1947 lhs_real[1] += CIMAG(val); \
1948 } \
1949 } \
1950 } \
1951 } \
1952 } \
1953 } /* if(d==5) */ \
1954 else { \
1955 for (l_L = 0; l_L < lprod; l_L++) \
1956 { \
1957 C *lhs; \
1958 R *lhs_real; \
1959 C val; \
1960 \
1961 MACRO_adjoint_nd_B_OMP_COMPUTE_UPDATE_ ## whichone \
1962 \
1963 lhs = ths->g + ll_plain[ths->d]; \
1964 lhs_real = (R*)lhs; \
1965 val = phi_prod[ths->d] * ths->f[j]; \
1966 \
1967 _Pragma("omp atomic") \
1968 lhs_real[0] += CREAL(val); \
1969 \
1970 _Pragma("omp atomic") \
1971 lhs_real[1] += CIMAG(val); \
1972 \
1973 MACRO_count_uo_l_lj_t; \
1974 } /* for(l_L) */ \
1975 } \
1976}
1977
1978static inline void B_openmp_T(X(plan) *ths)
1979{
1980 INT lprod; /* 'regular bandwidth' of matrix B */
1981 INT k;
1982
1983 memset(ths->g, 0, (size_t)(ths->n_total) * sizeof(C));
1984
1985 for (k = 0, lprod = 1; k < ths->d; k++)
1986 lprod *= (2*ths->m+2);
1987
1988 if (ths->flags & PRE_FULL_PSI)
1989 {
1990 nfft_adjoint_B_compute_full_psi(ths->g, ths->psi_index_g, ths->psi, ths->f,
1991 ths->M_total, ths->d, ths->n, ths->m, ths->flags, ths->index_x);
1992 return;
1993 }
1994
1995 if (ths->flags & PRE_PSI)
1996 {
1997 MACRO_adjoint_nd_B_OMP_BLOCKWISE(with_PRE_PSI);
1998
1999 #pragma omp parallel for default(shared) private(k)
2000 for (k = 0; k < ths->M_total; k++)
2001 {
2002 INT t, t2; /* index dimensions */ \
2003 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
2004 MACRO_adjoint_nd_B_OMP_COMPUTE(with_PRE_PSI);
2005 } /* for(j) */
2006 return;
2007 } /* if(PRE_PSI) */
2008
2009 if (ths->flags & PRE_FG_PSI)
2010 {
2011 INT t, t2; /* index dimensions */
2012 R fg_exp_l[ths->d][2*ths->m+2+1];
2013 for(t2 = 0; t2 < ths->d; t2++)
2014 {
2015 INT lj_fg;
2016 R tmpEXP2 = EXP(K(-1.0)/ths->b[t2]);
2017 R tmpEXP2sq = tmpEXP2*tmpEXP2;
2018 R tmp2 = K(1.0);
2019 R tmp3 = K(1.0);
2020 fg_exp_l[t2][0] = K(1.0);
2021 for(lj_fg = 1; lj_fg <= (2*ths->m+2); lj_fg++)
2022 {
2023 tmp3 = tmp2*tmpEXP2;
2024 tmp2 *= tmpEXP2sq;
2025 fg_exp_l[t2][lj_fg] = fg_exp_l[t2][lj_fg-1]*tmp3;
2026 }
2027 }
2028
2029 MACRO_adjoint_nd_B_OMP_BLOCKWISE(with_PRE_FG_PSI);
2030
2031 #pragma omp parallel for default(shared) private(k,t,t2)
2032 for (k = 0; k < ths->M_total; k++)
2033 {
2034 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
2035 MACRO_adjoint_nd_B_OMP_COMPUTE(with_PRE_FG_PSI);
2036 } /* for(j) */
2037 return;
2038 } /* if(PRE_FG_PSI) */
2039
2040 if (ths->flags & FG_PSI)
2041 {
2042 INT t, t2; /* index dimensions */
2043 R fg_exp_l[ths->d][2*ths->m+2+1];
2044
2045 sort(ths);
2046
2047 for (t2 = 0; t2 < ths->d; t2++)
2048 {
2049 INT lj_fg;
2050 R tmpEXP2 = EXP(K(-1.0)/ths->b[t2]);
2051 R tmpEXP2sq = tmpEXP2*tmpEXP2;
2052 R tmp2 = K(1.0);
2053 R tmp3 = K(1.0);
2054 fg_exp_l[t2][0] = K(1.0);
2055 for (lj_fg = 1; lj_fg <= (2*ths->m+2); lj_fg++)
2056 {
2057 tmp3 = tmp2*tmpEXP2;
2058 tmp2 *= tmpEXP2sq;
2059 fg_exp_l[t2][lj_fg] = fg_exp_l[t2][lj_fg-1]*tmp3;
2060 }
2061 }
2062
2063 MACRO_adjoint_nd_B_OMP_BLOCKWISE(with_FG_PSI);
2064
2065 #pragma omp parallel for default(shared) private(k,t,t2)
2066 for (k = 0; k < ths->M_total; k++)
2067 {
2068 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
2069 MACRO_adjoint_nd_B_OMP_COMPUTE(with_FG_PSI);
2070 } /* for(j) */
2071 return;
2072 } /* if(FG_PSI) */
2073
2074 if (ths->flags & PRE_LIN_PSI)
2075 {
2076 sort(ths);
2077
2078 MACRO_adjoint_nd_B_OMP_BLOCKWISE(with_PRE_LIN_PSI);
2079
2080 #pragma omp parallel for default(shared) private(k)
2081 for (k = 0; k<ths->M_total; k++)
2082 {
2083 INT t, t2; /* index dimensions */
2084 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
2085 MACRO_adjoint_nd_B_OMP_COMPUTE(with_PRE_LIN_PSI);
2086 } /* for(j) */
2087 return;
2088 } /* if(PRE_LIN_PSI) */
2089
2090 /* no precomputed psi at all */
2091 sort(ths);
2092
2093 MACRO_adjoint_nd_B_OMP_BLOCKWISE(without_PRE_PSI);
2094
2095 #pragma omp parallel for default(shared) private(k)
2096 for (k = 0; k < ths->M_total; k++)
2097 {
2098 INT t, t2; /* index dimensions */
2099 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
2100 MACRO_adjoint_nd_B_OMP_COMPUTE(without_PRE_PSI);
2101 } /* for(j) */
2102}
2103#endif
2104
2105static void B_T(X(plan) *ths)
2106{
2107#ifdef _OPENMP
2108 B_openmp_T(ths);
2109#else
2110 B_serial_T(ths);
2111#endif
2112}
2113
2114/* ## specialized version for d=1 ########################################### */
2115
2116static void nfft_1d_init_fg_exp_l(R *fg_exp_l, const INT m, const R b)
2117{
2118 const INT tmp2 = 2*m+2;
2119 INT l;
2120 R fg_exp_b0, fg_exp_b1, fg_exp_b2, fg_exp_b0_sq;
2121
2122 fg_exp_b0 = EXP(K(-1.0)/b);
2123 fg_exp_b0_sq = fg_exp_b0*fg_exp_b0;
2124 fg_exp_b1 = fg_exp_b2 =fg_exp_l[0] = K(1.0);
2125
2126 for (l = 1; l < tmp2; l++)
2127 {
2128 fg_exp_b2 = fg_exp_b1*fg_exp_b0;
2129 fg_exp_b1 *= fg_exp_b0_sq;
2130 fg_exp_l[l] = fg_exp_l[l-1]*fg_exp_b2;
2131 }
2132}
2133
2134
2135static void nfft_trafo_1d_compute(C *fj, const C *g,const R *psij_const,
2136 const R *xj, const INT n, const INT m)
2137{
2138 INT u, o, l;
2139 const C *gj;
2140 const R *psij;
2141 psij = psij_const;
2142
2143 uo2(&u, &o, *xj, n, m);
2144
2145 if (u < o)
2146 {
2147 for (l = 1, gj = g + u, (*fj) = (*psij++) * (*gj++); l <= 2*m+1; l++)
2148 (*fj) += (*psij++) * (*gj++);
2149 }
2150 else
2151 {
2152 for (l = 1, gj = g + u, (*fj) = (*psij++) * (*gj++); l < 2*m+1 - o; l++)
2153 (*fj) += (*psij++) * (*gj++);
2154 for (l = 0, gj = g; l <= o; l++)
2155 (*fj) += (*psij++) * (*gj++);
2156 }
2157}
2158
2159#ifndef _OPENMP
2160static void nfft_adjoint_1d_compute_serial(const C *fj, C *g,
2161 const R *psij_const, const R *xj, const INT n, const INT m)
2162{
2163 INT u,o,l;
2164 C *gj;
2165 const R *psij;
2166 psij = psij_const;
2167
2168 uo2(&u,&o,*xj, n, m);
2169
2170 if (u < o)
2171 {
2172 for (l = 0, gj = g+u; l <= 2*m+1; l++)
2173 (*gj++) += (*psij++) * (*fj);
2174 }
2175 else
2176 {
2177 for (l = 0, gj = g+u; l < 2*m+1-o; l++)
2178 (*gj++) += (*psij++) * (*fj);
2179 for (l = 0, gj = g; l <= o; l++)
2180 (*gj++) += (*psij++) * (*fj);
2181 }
2182}
2183#endif
2184
2185#ifdef _OPENMP
2186/* adjoint NFFT one-dimensional case with OpenMP atomic operations */
2187static void nfft_adjoint_1d_compute_omp_atomic(const C f, C *g,
2188 const R *psij_const, const R *xj, const INT n, const INT m)
2189{
2190 INT u,o,l;
2191 C *gj;
2192 INT index_temp[2*m+2];
2193
2194 uo2(&u,&o,*xj, n, m);
2195
2196 for (l=0; l<=2*m+1; l++)
2197 index_temp[l] = (l+u)%n;
2198
2199 for (l = 0, gj = g+u; l <= 2*m+1; l++)
2200 {
2201 INT i = index_temp[l];
2202 C *lhs = g+i;
2203 R *lhs_real = (R*)lhs;
2204 C val = psij_const[l] * f;
2205 #pragma omp atomic
2206 lhs_real[0] += CREAL(val);
2207
2208 #pragma omp atomic
2209 lhs_real[1] += CIMAG(val);
2210 }
2211}
2212#endif
2213
2214#ifdef _OPENMP
2230static void nfft_adjoint_1d_compute_omp_blockwise(const C f, C *g,
2231 const R *psij_const, const R *xj, const INT n, const INT m,
2232 const INT my_u0, const INT my_o0)
2233{
2234 INT ar_u,ar_o,l;
2235
2236 uo2(&ar_u,&ar_o,*xj, n, m);
2237
2238 if (ar_u < ar_o)
2239 {
2240 INT u = MAX(my_u0,ar_u);
2241 INT o = MIN(my_o0,ar_o);
2242 INT offset_psij = u-ar_u;
2243#ifdef OMP_ASSERT
2244 assert(offset_psij >= 0);
2245 assert(o-u <= 2*m+1);
2246 assert(offset_psij+o-u <= 2*m+1);
2247#endif
2248
2249 for (l = 0; l <= o-u; l++)
2250 g[u+l] += psij_const[offset_psij+l] * f;
2251 }
2252 else
2253 {
2254 INT u = MAX(my_u0,ar_u);
2255 INT o = my_o0;
2256 INT offset_psij = u-ar_u;
2257#ifdef OMP_ASSERT
2258 assert(offset_psij >= 0);
2259 assert(o-u <= 2*m+1);
2260 assert(offset_psij+o-u <= 2*m+1);
2261#endif
2262
2263 for (l = 0; l <= o-u; l++)
2264 g[u+l] += psij_const[offset_psij+l] * f;
2265
2266 u = my_u0;
2267 o = MIN(my_o0,ar_o);
2268 offset_psij += my_u0-ar_u+n;
2269
2270#ifdef OMP_ASSERT
2271 if (u <= o)
2272 {
2273 assert(o-u <= 2*m+1);
2274 if (offset_psij+o-u > 2*m+1)
2275 {
2276 fprintf(stderr, "ERR: %d %d %d %d %d %d %d\n", ar_u, ar_o, my_u0, my_o0, u, o, offset_psij);
2277 }
2278 assert(offset_psij+o-u <= 2*m+1);
2279 }
2280#endif
2281 for (l = 0; l <= o-u; l++)
2282 g[u+l] += psij_const[offset_psij+l] * f;
2283 }
2284}
2285#endif
2286
2287/* Window buffers come from the heap, once per thread: GCC 13 and newer abort
2288 * expanding this function's OpenMP regions on aarch64-apple-darwin when they
2289 * hold a run-time sized stack array. */
2290static void nfft_trafo_1d_B(X(plan) *ths)
2291{
2292 const INT n = ths->n[0], M = ths->M_total, m = ths->m, m2p2 = 2*m+2;
2293 const C *g = (C*)ths->g;
2294
2295 if (ths->flags & PRE_FULL_PSI)
2296 {
2297 INT k;
2298#ifdef _OPENMP
2299 #pragma omp parallel for default(shared) private(k)
2300#endif
2301 for (k = 0; k < M; k++)
2302 {
2303 INT l;
2304 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
2305 ths->f[j] = K(0.0);
2306 for (l = 0; l < m2p2; l++)
2307 ths->f[j] += ths->psi[j*m2p2+l] * g[ths->psi_index_g[j*m2p2+l]];
2308 }
2309 return;
2310 } /* if(PRE_FULL_PSI) */
2311
2312 if (ths->flags & PRE_PSI)
2313 {
2314 INT k;
2315#ifdef _OPENMP
2316 #pragma omp parallel for default(shared) private(k)
2317#endif
2318 for (k = 0; k < M; k++)
2319 {
2320 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
2321 nfft_trafo_1d_compute(&ths->f[j], g, ths->psi + j * (2 * m + 2),
2322 &ths->x[j], n, m);
2323 }
2324 return;
2325 } /* if(PRE_PSI) */
2326
2327 if (ths->flags & PRE_FG_PSI)
2328 {
2329 R *fg_exp_l = (R*)Y(malloc)((size_t)(m2p2+1) * sizeof(R));
2330
2331 nfft_1d_init_fg_exp_l(fg_exp_l, m, ths->b[0]);
2332
2333#ifdef _OPENMP
2334 #pragma omp parallel default(shared)
2335#endif
2336 {
2337 INT k;
2338 R *psij_const = (R*)Y(malloc)((size_t)(m2p2) * sizeof(R));
2339#ifdef _OPENMP
2340 #pragma omp for
2341#endif
2342 for (k = 0; k < M; k++)
2343 {
2344 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
2345 const R fg_psij0 = ths->psi[2 * j], fg_psij1 = ths->psi[2 * j + 1];
2346 R fg_psij2 = K(1.0);
2347 INT l;
2348
2349 psij_const[0] = fg_psij0;
2350
2351 for (l = 1; l < m2p2; l++)
2352 {
2353 fg_psij2 *= fg_psij1;
2354 psij_const[l] = fg_psij0 * fg_psij2 * fg_exp_l[l];
2355 }
2356
2357 nfft_trafo_1d_compute(&ths->f[j], g, psij_const, &ths->x[j], n, m);
2358 }
2359 Y(free)(psij_const);
2360 }
2361
2362 Y(free)(fg_exp_l);
2363 return;
2364 } /* if(PRE_FG_PSI) */
2365
2366 if (ths->flags & FG_PSI)
2367 {
2368 R *fg_exp_l = (R*)Y(malloc)((size_t)(m2p2+1) * sizeof(R));
2369
2370 sort(ths);
2371
2372 nfft_1d_init_fg_exp_l(fg_exp_l, m, ths->b[0]);
2373
2374#ifdef _OPENMP
2375 #pragma omp parallel default(shared)
2376#endif
2377 {
2378 INT k;
2379 R *psij_const = (R*)Y(malloc)((size_t)(m2p2) * sizeof(R));
2380#ifdef _OPENMP
2381 #pragma omp for
2382#endif
2383 for (k = 0; k < M; k++)
2384 {
2385 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
2386 INT u, o, l;
2387 R fg_psij0, fg_psij1, fg_psij2;
2388
2389 uo(ths, (INT)j, &u, &o, (INT)0);
2390 fg_psij0 = (PHI(ths->n[0], ths->x[j] - ((R)(u))/(R)(n), 0));
2391 fg_psij1 = EXP(K(2.0) * ((R)(n) * ths->x[j] - (R)(u)) / ths->b[0]);
2392 fg_psij2 = K(1.0);
2393
2394 psij_const[0] = fg_psij0;
2395
2396 for (l = 1; l < m2p2; l++)
2397 {
2398 fg_psij2 *= fg_psij1;
2399 psij_const[l] = fg_psij0 * fg_psij2 * fg_exp_l[l];
2400 }
2401
2402 nfft_trafo_1d_compute(&ths->f[j], g, psij_const, &ths->x[j], n, m);
2403 }
2404 Y(free)(psij_const);
2405 }
2406
2407 Y(free)(fg_exp_l);
2408 return;
2409 } /* if(FG_PSI) */
2410
2411 if (ths->flags & PRE_LIN_PSI)
2412 {
2413 const INT K = ths->K, ip_s = K / (m + 2);
2414
2415 sort(ths);
2416
2417#ifdef _OPENMP
2418 #pragma omp parallel default(shared)
2419#endif
2420 {
2421 INT k;
2422 R *psij_const = (R*)Y(malloc)((size_t)(m2p2) * sizeof(R));
2423#ifdef _OPENMP
2424 #pragma omp for
2425#endif
2426 for (k = 0; k < M; k++)
2427 {
2428 INT u, o, l;
2429 R ip_y, ip_w;
2430 INT ip_u;
2431 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
2432
2433 uo(ths, (INT)j, &u, &o, (INT)0);
2434
2435 ip_y = FABS((R)(n) * ths->x[j] - (R)(u)) * ((R)ip_s);
2436 ip_u = (INT)(LRINT(FLOOR(ip_y)));
2437 ip_w = ip_y - (R)(ip_u);
2438
2439 for (l = 0; l < m2p2; l++)
2440 psij_const[l] = ths->psi[ABS(ip_u-l*ip_s)] * (K(1.0) - ip_w)
2441 + ths->psi[ABS(ip_u-l*ip_s+1)] * (ip_w);
2442
2443 nfft_trafo_1d_compute(&ths->f[j], g, psij_const, &ths->x[j], n, m);
2444 }
2445 Y(free)(psij_const);
2446 }
2447 return;
2448 } /* if(PRE_LIN_PSI) */
2449 else
2450 {
2451 /* no precomputed psi at all */
2452
2453 sort(ths);
2454
2455#ifdef _OPENMP
2456 #pragma omp parallel default(shared)
2457#endif
2458 {
2459 INT k;
2460 R *psij_const = (R*)Y(malloc)((size_t)(m2p2) * sizeof(R));
2461#ifdef _OPENMP
2462 #pragma omp for
2463#endif
2464 for (k = 0; k < M; k++)
2465 {
2466 INT u, o, l;
2467 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
2468
2469 uo(ths, (INT)j, &u, &o, (INT)0);
2470
2471 for (l = 0; l < m2p2; l++)
2472 psij_const[l] = (PHI(ths->n[0], ths->x[j] - ((R)((u+l))) / (R)(n), 0));
2473
2474 nfft_trafo_1d_compute(&ths->f[j], g, psij_const, &ths->x[j], n, m);
2475 }
2476 Y(free)(psij_const);
2477 }
2478 }
2479}
2480
2481
2482#define MACRO_adjoint_1d_B_OMP_BLOCKWISE_COMPUTE_PRE_PSI \
2483{ \
2484 nfft_adjoint_1d_compute_omp_blockwise(ths->f[j], g, \
2485 ths->psi + j * (2 * m + 2), ths->x + j, n, m, my_u0, my_o0); \
2486}
2487
2488#define MACRO_adjoint_1d_B_OMP_BLOCKWISE_COMPUTE_PRE_FG_PSI \
2489{ \
2490 R psij_const[2 * m + 2]; \
2491 INT l; \
2492 R fg_psij0 = ths->psi[2 * j]; \
2493 R fg_psij1 = ths->psi[2 * j + 1]; \
2494 R fg_psij2 = K(1.0); \
2495 \
2496 psij_const[0] = fg_psij0; \
2497 for (l = 1; l <= 2 * m + 1; l++) \
2498 { \
2499 fg_psij2 *= fg_psij1; \
2500 psij_const[l] = fg_psij0 * fg_psij2 * fg_exp_l[l]; \
2501 } \
2502 \
2503 nfft_adjoint_1d_compute_omp_blockwise(ths->f[j], g, psij_const, \
2504 ths->x + j, n, m, my_u0, my_o0); \
2505}
2506
2507#define MACRO_adjoint_1d_B_OMP_BLOCKWISE_COMPUTE_FG_PSI \
2508{ \
2509 R psij_const[2 * m + 2]; \
2510 R fg_psij0, fg_psij1, fg_psij2; \
2511 INT u, o, l; \
2512 \
2513 uo(ths, j, &u, &o, (INT)0); \
2514 fg_psij0 = (PHI(ths->n[0],ths->x[j]-((R)u)/((R)n),0)); \
2515 fg_psij1 = EXP(K(2.0) * (((R)n) * (ths->x[j]) - (R)u) / ths->b[0]); \
2516 fg_psij2 = K(1.0); \
2517 psij_const[0] = fg_psij0; \
2518 for (l = 1; l <= 2 * m + 1; l++) \
2519 { \
2520 fg_psij2 *= fg_psij1; \
2521 psij_const[l] = fg_psij0 * fg_psij2 * fg_exp_l[l]; \
2522 } \
2523 \
2524 nfft_adjoint_1d_compute_omp_blockwise(ths->f[j], g, psij_const, \
2525 ths->x + j, n, m, my_u0, my_o0); \
2526}
2527
2528#define MACRO_adjoint_1d_B_OMP_BLOCKWISE_COMPUTE_PRE_LIN_PSI \
2529{ \
2530 R psij_const[2 * m + 2]; \
2531 INT ip_u; \
2532 R ip_y, ip_w; \
2533 INT u, o, l; \
2534 \
2535 uo(ths, j, &u, &o, (INT)0); \
2536 \
2537 ip_y = FABS(((R)n) * ths->x[j] - (R)u) * ((R)ip_s); \
2538 ip_u = LRINT(FLOOR(ip_y)); \
2539 ip_w = ip_y - ip_u; \
2540 for (l = 0; l < 2 * m + 2; l++) \
2541 psij_const[l] \
2542 = ths->psi[ABS(ip_u-l*ip_s)] * (K(1.0) - ip_w) \
2543 + ths->psi[ABS(ip_u-l*ip_s+1)] * (ip_w); \
2544 \
2545 nfft_adjoint_1d_compute_omp_blockwise(ths->f[j], g, psij_const, \
2546 ths->x + j, n, m, my_u0, my_o0); \
2547}
2548
2549#define MACRO_adjoint_1d_B_OMP_BLOCKWISE_COMPUTE_NO_PSI \
2550{ \
2551 R psij_const[2 * m + 2]; \
2552 INT u, o, l; \
2553 \
2554 uo(ths, j, &u, &o, (INT)0); \
2555 \
2556 for (l = 0; l <= 2 * m + 1; l++) \
2557 psij_const[l] = (PHI(ths->n[0],ths->x[j]-((R)((u+l)))/((R)n),0)); \
2558 \
2559 nfft_adjoint_1d_compute_omp_blockwise(ths->f[j], g, psij_const, \
2560 ths->x + j, n, m, my_u0, my_o0); \
2561}
2562
2563#define MACRO_adjoint_1d_B_OMP_BLOCKWISE(whichone) \
2564{ \
2565 if (ths->flags & NFFT_OMP_BLOCKWISE_ADJOINT) \
2566 { \
2567 _Pragma("omp parallel private(k)") \
2568 { \
2569 INT my_u0, my_o0, min_u_a, max_u_a, min_u_b, max_u_b; \
2570 INT *ar_x = ths->index_x; \
2571 \
2572 nfft_adjoint_B_omp_blockwise_init(&my_u0, &my_o0, &min_u_a, &max_u_a, \
2573 &min_u_b, &max_u_b, 1, &n, m); \
2574 \
2575 if (min_u_a != -1) \
2576 { \
2577 k = index_x_binary_search(ar_x, M, min_u_a); \
2578 \
2579 MACRO_adjoint_nd_B_OMP_BLOCKWISE_ASSERT_A \
2580 \
2581 while (k < M) \
2582 { \
2583 INT u_prod = ar_x[2*k]; \
2584 INT j = ar_x[2*k+1]; \
2585 \
2586 if (u_prod < min_u_a || u_prod > max_u_a) \
2587 break; \
2588 \
2589 MACRO_adjoint_1d_B_OMP_BLOCKWISE_COMPUTE_ ##whichone \
2590 \
2591 k++; \
2592 } \
2593 } \
2594 \
2595 if (min_u_b != -1) \
2596 { \
2597 k = index_x_binary_search(ar_x, M, min_u_b); \
2598 \
2599 MACRO_adjoint_nd_B_OMP_BLOCKWISE_ASSERT_B \
2600 \
2601 while (k < M) \
2602 { \
2603 INT u_prod = ar_x[2*k]; \
2604 INT j = ar_x[2*k+1]; \
2605 \
2606 if (u_prod < min_u_b || u_prod > max_u_b) \
2607 break; \
2608 \
2609 MACRO_adjoint_1d_B_OMP_BLOCKWISE_COMPUTE_ ##whichone \
2610 \
2611 k++; \
2612 } \
2613 } \
2614 } /* omp parallel */ \
2615 return; \
2616 } /* if(NFFT_OMP_BLOCKWISE_ADJOINT) */ \
2617}
2618
2619static void nfft_adjoint_1d_B(X(plan) *ths)
2620{
2621 const INT n = ths->n[0], M = ths->M_total, m = ths->m;
2622 INT k;
2623 C *g = (C*)ths->g;
2624
2625 memset(g, 0, (size_t)(ths->n_total) * sizeof(C));
2626
2627 if (ths->flags & PRE_FULL_PSI)
2628 {
2629 nfft_adjoint_B_compute_full_psi(g, ths->psi_index_g, ths->psi, ths->f, M,
2630 (INT)1, ths->n, m, ths->flags, ths->index_x);
2631 return;
2632 } /* if(PRE_FULL_PSI) */
2633
2634 if (ths->flags & PRE_PSI)
2635 {
2636#ifdef _OPENMP
2637 MACRO_adjoint_1d_B_OMP_BLOCKWISE(PRE_PSI)
2638#endif
2639
2640#ifdef _OPENMP
2641 #pragma omp parallel for default(shared) private(k)
2642#endif
2643 for (k = 0; k < M; k++)
2644 {
2645 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
2646#ifdef _OPENMP
2647 nfft_adjoint_1d_compute_omp_atomic(ths->f[j], g, ths->psi + j * (2 * m + 2), ths->x + j, n, m);
2648#else
2649 nfft_adjoint_1d_compute_serial(ths->f + j, g, ths->psi + j * (2 * m + 2), ths->x + j, n, m);
2650#endif
2651 }
2652
2653 return;
2654 } /* if(PRE_PSI) */
2655
2656 if (ths->flags & PRE_FG_PSI)
2657 {
2658 R fg_exp_l[2 * m + 2 + 1];
2659
2660 nfft_1d_init_fg_exp_l(fg_exp_l, m, ths->b[0]);
2661
2662#ifdef _OPENMP
2663 MACRO_adjoint_1d_B_OMP_BLOCKWISE(PRE_FG_PSI)
2664#endif
2665
2666
2667#ifdef _OPENMP
2668 #pragma omp parallel for default(shared) private(k)
2669#endif
2670 for (k = 0; k < M; k++)
2671 {
2672 R psij_const[2 * m + 2];
2673 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
2674 INT l;
2675 R fg_psij0 = ths->psi[2 * j];
2676 R fg_psij1 = ths->psi[2 * j + 1];
2677 R fg_psij2 = K(1.0);
2678
2679 psij_const[0] = fg_psij0;
2680 for (l = 1; l <= 2 * m + 1; l++)
2681 {
2682 fg_psij2 *= fg_psij1;
2683 psij_const[l] = fg_psij0 * fg_psij2 * fg_exp_l[l];
2684 }
2685
2686#ifdef _OPENMP
2687 nfft_adjoint_1d_compute_omp_atomic(ths->f[j], g, psij_const, ths->x + j, n, m);
2688#else
2689 nfft_adjoint_1d_compute_serial(ths->f + j, g, psij_const, ths->x + j, n, m);
2690#endif
2691 }
2692
2693 return;
2694 } /* if(PRE_FG_PSI) */
2695
2696 if (ths->flags & FG_PSI)
2697 {
2698 R fg_exp_l[2 * m + 2 + 1];
2699
2700 nfft_1d_init_fg_exp_l(fg_exp_l, m, ths->b[0]);
2701
2702 sort(ths);
2703
2704#ifdef _OPENMP
2705 MACRO_adjoint_1d_B_OMP_BLOCKWISE(FG_PSI)
2706#endif
2707
2708#ifdef _OPENMP
2709 #pragma omp parallel for default(shared) private(k)
2710#endif
2711 for (k = 0; k < M; k++)
2712 {
2713 INT u,o,l;
2714 R psij_const[2 * m + 2];
2715 R fg_psij0, fg_psij1, fg_psij2;
2716 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
2717
2718 uo(ths, j, &u, &o, (INT)0);
2719 fg_psij0 = (PHI(ths->n[0], ths->x[j] - ((R)u) / (R)(n),0));
2720 fg_psij1 = EXP(K(2.0) * ((R)(n) * (ths->x[j]) - (R)(u)) / ths->b[0]);
2721 fg_psij2 = K(1.0);
2722 psij_const[0] = fg_psij0;
2723 for (l = 1; l <= 2 * m + 1; l++)
2724 {
2725 fg_psij2 *= fg_psij1;
2726 psij_const[l] = fg_psij0 * fg_psij2 * fg_exp_l[l];
2727 }
2728
2729#ifdef _OPENMP
2730 nfft_adjoint_1d_compute_omp_atomic(ths->f[j], g, psij_const, ths->x + j, n, m);
2731#else
2732 nfft_adjoint_1d_compute_serial(ths->f + j, g, psij_const, ths->x + j, n, m);
2733#endif
2734 }
2735
2736 return;
2737 } /* if(FG_PSI) */
2738
2739 if (ths->flags & PRE_LIN_PSI)
2740 {
2741 const INT K = ths->K;
2742 const INT ip_s = K / (m + 2);
2743
2744 sort(ths);
2745
2746#ifdef _OPENMP
2747 MACRO_adjoint_1d_B_OMP_BLOCKWISE(PRE_LIN_PSI)
2748#endif
2749
2750#ifdef _OPENMP
2751 #pragma omp parallel for default(shared) private(k)
2752#endif
2753 for (k = 0; k < M; k++)
2754 {
2755 INT u,o,l;
2756 INT ip_u;
2757 R ip_y, ip_w;
2758 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
2759 R psij_const[2 * m + 2];
2760
2761 uo(ths, j, &u, &o, (INT)0);
2762
2763 ip_y = FABS((R)(n) * ths->x[j] - (R)(u)) * ((R)ip_s);
2764 ip_u = (INT)(LRINT(FLOOR(ip_y)));
2765 ip_w = ip_y - (R)(ip_u);
2766 for (l = 0; l < 2 * m + 2; l++)
2767 psij_const[l]
2768 = ths->psi[ABS(ip_u-l*ip_s)] * (K(1.0) - ip_w)
2769 + ths->psi[ABS(ip_u-l*ip_s+1)] * (ip_w);
2770
2771#ifdef _OPENMP
2772 nfft_adjoint_1d_compute_omp_atomic(ths->f[j], g, psij_const, ths->x + j, n, m);
2773#else
2774 nfft_adjoint_1d_compute_serial(ths->f + j, g, psij_const, ths->x + j, n, m);
2775#endif
2776 }
2777 return;
2778 } /* if(PRE_LIN_PSI) */
2779
2780 /* no precomputed psi at all */
2781 sort(ths);
2782
2783#ifdef _OPENMP
2784 MACRO_adjoint_1d_B_OMP_BLOCKWISE(NO_PSI)
2785#endif
2786
2787#ifdef _OPENMP
2788 #pragma omp parallel for default(shared) private(k)
2789#endif
2790 for (k = 0; k < M; k++)
2791 {
2792 INT u,o,l;
2793 R psij_const[2 * m + 2];
2794 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
2795
2796 uo(ths, j, &u, &o, (INT)0);
2797
2798 for (l = 0; l <= 2 * m + 1; l++)
2799 psij_const[l] = (PHI(ths->n[0], ths->x[j] - ((R)((u+l))) / (R)(n),0));
2800
2801#ifdef _OPENMP
2802 nfft_adjoint_1d_compute_omp_atomic(ths->f[j], g, psij_const, ths->x + j, n, m);
2803#else
2804 nfft_adjoint_1d_compute_serial(ths->f + j, g, psij_const, ths->x + j, n, m);
2805#endif
2806 }
2807}
2808
2809void X(trafo_1d)(X(plan) *ths)
2810{
2811 if((ths->N[0] <= ths->m) || (ths->n[0] <= 2*ths->m+2))
2812 {
2813 X(trafo_direct)(ths);
2814 return;
2815 }
2816
2817 const INT N = ths->N[0], N2 = N/2, n = ths->n[0];
2818 C *f_hat1 = (C*)ths->f_hat, *f_hat2 = (C*)&ths->f_hat[N2];
2819
2820 ths->g_hat = ths->g1;
2821 ths->g = ths->g2;
2822
2823 {
2824 C *g_hat1 = (C*)&ths->g_hat[n-N/2], *g_hat2 = (C*)ths->g_hat;
2825 R *c_phi_inv1, *c_phi_inv2;
2826
2827 TIC(0)
2828#ifdef _OPENMP
2829 {
2830 INT k;
2831 #pragma omp parallel for default(shared) private(k)
2832 for (k = 0; k < ths->n_total; k++)
2833 ths->g_hat[k] = 0.0;
2834 }
2835#else
2836 memset(ths->g_hat, 0, (size_t)(ths->n_total) * sizeof(C));
2837#endif
2838 if(ths->flags & PRE_PHI_HUT)
2839 {
2840 INT k;
2841 c_phi_inv1 = ths->c_phi_inv[0];
2842 c_phi_inv2 = &ths->c_phi_inv[0][N2];
2843
2844#ifdef _OPENMP
2845 #pragma omp parallel for default(shared) private(k)
2846#endif
2847 for (k = 0; k < N2; k++)
2848 {
2849 g_hat1[k] = f_hat1[k] * c_phi_inv1[k];
2850 g_hat2[k] = f_hat2[k] * c_phi_inv2[k];
2851 }
2852 }
2853 else
2854 {
2855 INT k;
2856#ifdef _OPENMP
2857 #pragma omp parallel for default(shared) private(k)
2858#endif
2859 for (k = 0; k < N2; k++)
2860 {
2861 g_hat1[k] = f_hat1[k] / (PHI_HUT(ths->n[0],k-N2,0));
2862 g_hat2[k] = f_hat2[k] / (PHI_HUT(ths->n[0],k,0));
2863 }
2864 }
2865 TOC(0)
2866
2867 TIC_FFTW(1)
2868 FFTW(execute)(ths->my_fftw_plan1);
2869 TOC_FFTW(1);
2870
2871 TIC(2);
2872 nfft_trafo_1d_B(ths);
2873 TOC(2);
2874 }
2875}
2876
2877void X(adjoint_1d)(X(plan) *ths)
2878{
2879 if((ths->N[0] <= ths->m) || (ths->n[0] <= 2*ths->m+2))
2880 {
2881 X(adjoint_direct)(ths);
2882 return;
2883 }
2884
2885 INT n,N;
2886 C *g_hat1,*g_hat2,*f_hat1,*f_hat2;
2887 R *c_phi_inv1, *c_phi_inv2;
2888
2889 N=ths->N[0];
2890 n=ths->n[0];
2891
2892 ths->g_hat=ths->g1;
2893 ths->g=ths->g2;
2894
2895 f_hat1=(C*)ths->f_hat;
2896 f_hat2=(C*)&ths->f_hat[N/2];
2897 g_hat1=(C*)&ths->g_hat[n-N/2];
2898 g_hat2=(C*)ths->g_hat;
2899
2900 TIC(2)
2901 nfft_adjoint_1d_B(ths);
2902 TOC(2)
2903
2904 TIC_FFTW(1)
2905 FFTW(execute)(ths->my_fftw_plan2);
2906 TOC_FFTW(1);
2907
2908 TIC(0)
2909 if(ths->flags & PRE_PHI_HUT)
2910 {
2911 INT k;
2912 c_phi_inv1=ths->c_phi_inv[0];
2913 c_phi_inv2=&ths->c_phi_inv[0][N/2];
2914
2915#ifdef _OPENMP
2916 #pragma omp parallel for default(shared) private(k)
2917#endif
2918 for (k = 0; k < N/2; k++)
2919 {
2920 f_hat1[k] = g_hat1[k] * c_phi_inv1[k];
2921 f_hat2[k] = g_hat2[k] * c_phi_inv2[k];
2922 }
2923 }
2924 else
2925 {
2926 INT k;
2927
2928#ifdef _OPENMP
2929 #pragma omp parallel for default(shared) private(k)
2930#endif
2931 for (k = 0; k < N/2; k++)
2932 {
2933 f_hat1[k] = g_hat1[k] / (PHI_HUT(ths->n[0],k-N/2,0));
2934 f_hat2[k] = g_hat2[k] / (PHI_HUT(ths->n[0],k,0));
2935 }
2936 }
2937 TOC(0)
2938}
2939
2940
2941/* ################################################ SPECIFIC VERSIONS FOR d=2 */
2942
2943static void nfft_2d_init_fg_exp_l(R *fg_exp_l, const INT m, const R b)
2944{
2945 INT l;
2946 R fg_exp_b0, fg_exp_b1, fg_exp_b2, fg_exp_b0_sq;
2947
2948 fg_exp_b0 = EXP(K(-1.0)/b);
2949 fg_exp_b0_sq = fg_exp_b0*fg_exp_b0;
2950 fg_exp_b1 = K(1.0);
2951 fg_exp_b2 = K(1.0);
2952 fg_exp_l[0] = K(1.0);
2953 for(l=1; l <= 2*m+1; l++)
2954 {
2955 fg_exp_b2 = fg_exp_b1*fg_exp_b0;
2956 fg_exp_b1 *= fg_exp_b0_sq;
2957 fg_exp_l[l] = fg_exp_l[l-1]*fg_exp_b2;
2958 }
2959}
2960
2961static void nfft_trafo_2d_compute(C *fj, const C *g, const R *psij_const0,
2962 const R *psij_const1, const R *xj0, const R *xj1, const INT n0,
2963 const INT n1, const INT m)
2964{
2965 INT u0,o0,l0,u1,o1,l1;
2966 const C *gj;
2967 const R *psij0,*psij1;
2968
2969 psij0=psij_const0;
2970 psij1=psij_const1;
2971
2972 uo2(&u0,&o0,*xj0, n0, m);
2973 uo2(&u1,&o1,*xj1, n1, m);
2974
2975 *fj=0;
2976
2977 if (u0 < o0)
2978 if(u1 < o1)
2979 for(l0=0; l0<=2*m+1; l0++,psij0++)
2980 {
2981 psij1=psij_const1;
2982 gj=g+(u0+l0)*n1+u1;
2983 for(l1=0; l1<=2*m+1; l1++)
2984 (*fj) += (*psij0) * (*psij1++) * (*gj++);
2985 }
2986 else
2987 for(l0=0; l0<=2*m+1; l0++,psij0++)
2988 {
2989 psij1=psij_const1;
2990 gj=g+(u0+l0)*n1+u1;
2991 for(l1=0; l1<2*m+1-o1; l1++)
2992 (*fj) += (*psij0) * (*psij1++) * (*gj++);
2993 gj=g+(u0+l0)*n1;
2994 for(l1=0; l1<=o1; l1++)
2995 (*fj) += (*psij0) * (*psij1++) * (*gj++);
2996 }
2997 else
2998 if(u1<o1)
2999 {
3000 for(l0=0; l0<2*m+1-o0; l0++,psij0++)
3001 {
3002 psij1=psij_const1;
3003 gj=g+(u0+l0)*n1+u1;
3004 for(l1=0; l1<=2*m+1; l1++)
3005 (*fj) += (*psij0) * (*psij1++) * (*gj++);
3006 }
3007 for(l0=0; l0<=o0; l0++,psij0++)
3008 {
3009 psij1=psij_const1;
3010 gj=g+l0*n1+u1;
3011 for(l1=0; l1<=2*m+1; l1++)
3012 (*fj) += (*psij0) * (*psij1++) * (*gj++);
3013 }
3014 }
3015 else
3016 {
3017 for(l0=0; l0<2*m+1-o0; l0++,psij0++)
3018 {
3019 psij1=psij_const1;
3020 gj=g+(u0+l0)*n1+u1;
3021 for(l1=0; l1<2*m+1-o1; l1++)
3022 (*fj) += (*psij0) * (*psij1++) * (*gj++);
3023 gj=g+(u0+l0)*n1;
3024 for(l1=0; l1<=o1; l1++)
3025 (*fj) += (*psij0) * (*psij1++) * (*gj++);
3026 }
3027 for(l0=0; l0<=o0; l0++,psij0++)
3028 {
3029 psij1=psij_const1;
3030 gj=g+l0*n1+u1;
3031 for(l1=0; l1<2*m+1-o1; l1++)
3032 (*fj) += (*psij0) * (*psij1++) * (*gj++);
3033 gj=g+l0*n1;
3034 for(l1=0; l1<=o1; l1++)
3035 (*fj) += (*psij0) * (*psij1++) * (*gj++);
3036 }
3037 }
3038}
3039
3040#ifdef _OPENMP
3041/* adjoint NFFT two-dimensional case with OpenMP atomic operations */
3042static void nfft_adjoint_2d_compute_omp_atomic(const C f, C *g,
3043 const R *psij_const0, const R *psij_const1, const R *xj0,
3044 const R *xj1, const INT n0, const INT n1, const INT m)
3045{
3046 INT u0,o0,l0,u1,o1,l1;
3047
3048 INT index_temp0[2*m+2];
3049 INT index_temp1[2*m+2];
3050
3051 uo2(&u0,&o0,*xj0, n0, m);
3052 uo2(&u1,&o1,*xj1, n1, m);
3053
3054 for (l0=0; l0<=2*m+1; l0++)
3055 index_temp0[l0] = (u0+l0)%n0;
3056
3057 for (l1=0; l1<=2*m+1; l1++)
3058 index_temp1[l1] = (u1+l1)%n1;
3059
3060 for(l0=0; l0<=2*m+1; l0++)
3061 {
3062 for(l1=0; l1<=2*m+1; l1++)
3063 {
3064 INT i = index_temp0[l0] * n1 + index_temp1[l1];
3065 C *lhs = g+i;
3066 R *lhs_real = (R*)lhs;
3067 C val = psij_const0[l0] * psij_const1[l1] * f;
3068
3069 #pragma omp atomic
3070 lhs_real[0] += CREAL(val);
3071
3072 #pragma omp atomic
3073 lhs_real[1] += CIMAG(val);
3074 }
3075 }
3076}
3077#endif
3078
3079#ifdef _OPENMP
3098static void nfft_adjoint_2d_compute_omp_blockwise(const C f, C *g,
3099 const R *psij_const0, const R *psij_const1, const R *xj0,
3100 const R *xj1, const INT n0, const INT n1, const INT m,
3101 const INT my_u0, const INT my_o0)
3102{
3103 INT ar_u0,ar_o0,l0,u1,o1,l1;
3104 INT index_temp1[2*m+2];
3105
3106 uo2(&ar_u0,&ar_o0,*xj0, n0, m);
3107 uo2(&u1,&o1,*xj1, n1, m);
3108
3109 for (l1 = 0; l1 <= 2*m+1; l1++)
3110 index_temp1[l1] = (u1+l1)%n1;
3111
3112 if(ar_u0 < ar_o0)
3113 {
3114 INT u0 = MAX(my_u0,ar_u0);
3115 INT o0 = MIN(my_o0,ar_o0);
3116 INT offset_psij = u0-ar_u0;
3117#ifdef OMP_ASSERT
3118 assert(offset_psij >= 0);
3119 assert(o0-u0 <= 2*m+1);
3120 assert(offset_psij+o0-u0 <= 2*m+1);
3121#endif
3122
3123 for (l0 = 0; l0 <= o0-u0; l0++)
3124 {
3125 INT i0 = (u0+l0) * n1;
3126 const C val0 = psij_const0[offset_psij+l0];
3127
3128 for(l1=0; l1<=2*m+1; l1++)
3129 g[i0 + index_temp1[l1]] += val0 * psij_const1[l1] * f;
3130 }
3131 }
3132 else
3133 {
3134 INT u0 = MAX(my_u0,ar_u0);
3135 INT o0 = my_o0;
3136 INT offset_psij = u0-ar_u0;
3137#ifdef OMP_ASSERT
3138 assert(offset_psij >= 0);
3139 assert(o0-u0 <= 2*m+1);
3140 assert(offset_psij+o0-u0 <= 2*m+1);
3141#endif
3142
3143 for (l0 = 0; l0 <= o0-u0; l0++)
3144 {
3145 INT i0 = (u0+l0) * n1;
3146 const C val0 = psij_const0[offset_psij+l0];
3147
3148 for(l1=0; l1<=2*m+1; l1++)
3149 g[i0 + index_temp1[l1]] += val0 * psij_const1[l1] * f;
3150 }
3151
3152 u0 = my_u0;
3153 o0 = MIN(my_o0,ar_o0);
3154 offset_psij += my_u0-ar_u0+n0;
3155
3156#ifdef OMP_ASSERT
3157 if (u0<=o0)
3158 {
3159 assert(o0-u0 <= 2*m+1);
3160 assert(offset_psij+o0-u0 <= 2*m+1);
3161 }
3162#endif
3163
3164 for (l0 = 0; l0 <= o0-u0; l0++)
3165 {
3166 INT i0 = (u0+l0) * n1;
3167 const C val0 = psij_const0[offset_psij+l0];
3168
3169 for(l1=0; l1<=2*m+1; l1++)
3170 g[i0 + index_temp1[l1]] += val0 * psij_const1[l1] * f;
3171 }
3172 }
3173}
3174#endif
3175
3176#ifndef _OPENMP
3177static void nfft_adjoint_2d_compute_serial(const C *fj, C *g,
3178 const R *psij_const0, const R *psij_const1, const R *xj0,
3179 const R *xj1, const INT n0, const INT n1, const INT m)
3180{
3181 INT u0,o0,l0,u1,o1,l1;
3182 C *gj;
3183 const R *psij0,*psij1;
3184
3185 psij0=psij_const0;
3186 psij1=psij_const1;
3187
3188 uo2(&u0,&o0,*xj0, n0, m);
3189 uo2(&u1,&o1,*xj1, n1, m);
3190
3191 if(u0<o0)
3192 if(u1<o1)
3193 for(l0=0; l0<=2*m+1; l0++,psij0++)
3194 {
3195 psij1=psij_const1;
3196 gj=g+(u0+l0)*n1+u1;
3197 for(l1=0; l1<=2*m+1; l1++)
3198 (*gj++) += (*psij0) * (*psij1++) * (*fj);
3199 }
3200 else
3201 for(l0=0; l0<=2*m+1; l0++,psij0++)
3202 {
3203 psij1=psij_const1;
3204 gj=g+(u0+l0)*n1+u1;
3205 for(l1=0; l1<2*m+1-o1; l1++)
3206 (*gj++) += (*psij0) * (*psij1++) * (*fj);
3207 gj=g+(u0+l0)*n1;
3208 for(l1=0; l1<=o1; l1++)
3209 (*gj++) += (*psij0) * (*psij1++) * (*fj);
3210 }
3211 else
3212 if(u1<o1)
3213 {
3214 for(l0=0; l0<2*m+1-o0; l0++,psij0++)
3215 {
3216 psij1=psij_const1;
3217 gj=g+(u0+l0)*n1+u1;
3218 for(l1=0; l1<=2*m+1; l1++)
3219 (*gj++) += (*psij0) * (*psij1++) * (*fj);
3220 }
3221 for(l0=0; l0<=o0; l0++,psij0++)
3222 {
3223 psij1=psij_const1;
3224 gj=g+l0*n1+u1;
3225 for(l1=0; l1<=2*m+1; l1++)
3226 (*gj++) += (*psij0) * (*psij1++) * (*fj);
3227 }
3228 }
3229 else
3230 {
3231 for(l0=0; l0<2*m+1-o0; l0++,psij0++)
3232 {
3233 psij1=psij_const1;
3234 gj=g+(u0+l0)*n1+u1;
3235 for(l1=0; l1<2*m+1-o1; l1++)
3236 (*gj++) += (*psij0) * (*psij1++) * (*fj);
3237 gj=g+(u0+l0)*n1;
3238 for(l1=0; l1<=o1; l1++)
3239 (*gj++) += (*psij0) * (*psij1++) * (*fj);
3240 }
3241 for(l0=0; l0<=o0; l0++,psij0++)
3242 {
3243 psij1=psij_const1;
3244 gj=g+l0*n1+u1;
3245 for(l1=0; l1<2*m+1-o1; l1++)
3246 (*gj++) += (*psij0) * (*psij1++) * (*fj);
3247 gj=g+l0*n1;
3248 for(l1=0; l1<=o1; l1++)
3249 (*gj++) += (*psij0) * (*psij1++) * (*fj);
3250 }
3251 }
3252}
3253#endif
3254
3255static void nfft_trafo_2d_B(X(plan) *ths)
3256{
3257 const C *g = (C*)ths->g;
3258 const INT n0 = ths->n[0];
3259 const INT n1 = ths->n[1];
3260 const INT M = ths->M_total;
3261 const INT m = ths->m;
3262
3263 INT k;
3264
3265 if(ths->flags & PRE_FULL_PSI)
3266 {
3267 const INT lprod = (2*m+2) * (2*m+2);
3268#ifdef _OPENMP
3269 #pragma omp parallel for default(shared) private(k)
3270#endif
3271 for (k = 0; k < M; k++)
3272 {
3273 INT l;
3274 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
3275 ths->f[j] = K(0.0);
3276 for (l = 0; l < lprod; l++)
3277 ths->f[j] += ths->psi[j*lprod+l] * g[ths->psi_index_g[j*lprod+l]];
3278 }
3279 return;
3280 } /* if(PRE_FULL_PSI) */
3281
3282 if(ths->flags & PRE_PSI)
3283 {
3284#ifdef _OPENMP
3285 #pragma omp parallel for default(shared) private(k)
3286#endif
3287 for (k = 0; k < M; k++)
3288 {
3289 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
3290 nfft_trafo_2d_compute(ths->f+j, g, ths->psi+j*2*(2*m+2), ths->psi+(j*2+1)*(2*m+2), ths->x+2*j, ths->x+2*j+1, n0, n1, m);
3291 }
3292
3293 return;
3294 } /* if(PRE_PSI) */
3295
3296 if(ths->flags & PRE_FG_PSI)
3297 {
3298 R fg_exp_l[2*(2*m+2+1)];
3299
3300 nfft_2d_init_fg_exp_l(fg_exp_l, m, ths->b[0]);
3301 nfft_2d_init_fg_exp_l(fg_exp_l+2*m+2, m, ths->b[1]);
3302
3303#ifdef _OPENMP
3304 #pragma omp parallel for default(shared) private(k)
3305#endif
3306 for (k = 0; k < M; k++)
3307 {
3308 R psij_const[2*(2*m+2)];
3309 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
3310 INT l;
3311 R fg_psij0 = ths->psi[2*j*2];
3312 R fg_psij1 = ths->psi[2*j*2+1];
3313 R fg_psij2 = K(1.0);
3314
3315 psij_const[0] = fg_psij0;
3316 for (l = 1; l <= 2*m+1; l++)
3317 {
3318 fg_psij2 *= fg_psij1;
3319 psij_const[l] = fg_psij0*fg_psij2*fg_exp_l[l];
3320 }
3321
3322 fg_psij0 = ths->psi[2*(j*2+1)];
3323 fg_psij1 = ths->psi[2*(j*2+1)+1];
3324 fg_psij2 = K(1.0);
3325 psij_const[2*m+2] = fg_psij0;
3326 for (l = 1; l <= 2*m+1; l++)
3327 {
3328 fg_psij2 *= fg_psij1;
3329 psij_const[2*m+2+l] = fg_psij0*fg_psij2*fg_exp_l[2*m+2+l];
3330 }
3331
3332 nfft_trafo_2d_compute(ths->f+j, g, psij_const, psij_const+2*m+2, ths->x+2*j, ths->x+2*j+1, n0, n1, m);
3333 }
3334
3335 return;
3336 } /* if(PRE_FG_PSI) */
3337
3338 if(ths->flags & FG_PSI)
3339 {
3340 R fg_exp_l[2*(2*m+2+1)];
3341
3342 nfft_2d_init_fg_exp_l(fg_exp_l, m, ths->b[0]);
3343 nfft_2d_init_fg_exp_l(fg_exp_l+2*m+2, m, ths->b[1]);
3344
3345 sort(ths);
3346
3347#ifdef _OPENMP
3348 #pragma omp parallel for default(shared) private(k)
3349#endif
3350 for (k = 0; k < M; k++)
3351 {
3352 INT u, o, l;
3353 R fg_psij0, fg_psij1, fg_psij2;
3354 R psij_const[2*(2*m+2)];
3355 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
3356
3357 uo(ths, j, &u, &o, (INT)0);
3358 fg_psij0 = (PHI(ths->n[0], ths->x[2*j] - ((R)u) / (R)(n0),0));
3359 fg_psij1 = EXP(K(2.0) * ((R)(n0) * (ths->x[2*j]) - (R)(u)) / ths->b[0]);
3360 fg_psij2 = K(1.0);
3361 psij_const[0] = fg_psij0;
3362 for (l = 1; l <= 2*m+1; l++)
3363 {
3364 fg_psij2 *= fg_psij1;
3365 psij_const[l] = fg_psij0*fg_psij2*fg_exp_l[l];
3366 }
3367
3368 uo(ths,j,&u,&o, (INT)1);
3369 fg_psij0 = (PHI(ths->n[1], ths->x[2*j+1] - ((R)u) / (R)(n1),1));
3370 fg_psij1 = EXP(K(2.0) * ((R)(n1) * (ths->x[2*j+1]) - (R)(u)) / ths->b[1]);
3371 fg_psij2 = K(1.0);
3372 psij_const[2*m+2] = fg_psij0;
3373 for(l=1; l<=2*m+1; l++)
3374 {
3375 fg_psij2 *= fg_psij1;
3376 psij_const[2*m+2+l] = fg_psij0*fg_psij2*fg_exp_l[2*m+2+l];
3377 }
3378
3379 nfft_trafo_2d_compute(ths->f+j, g, psij_const, psij_const+2*m+2, ths->x+2*j, ths->x+2*j+1, n0, n1, m);
3380 }
3381
3382 return;
3383 } /* if(FG_PSI) */
3384
3385 if(ths->flags & PRE_LIN_PSI)
3386 {
3387 const INT K = ths->K, ip_s = K / (m + 2);
3388
3389 sort(ths);
3390
3391#ifdef _OPENMP
3392 #pragma omp parallel for default(shared) private(k)
3393#endif
3394 for (k = 0; k < M; k++)
3395 {
3396 INT u, o, l;
3397 R ip_y, ip_w;
3398 INT ip_u;
3399 R psij_const[2*(2*m+2)];
3400 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
3401
3402 uo(ths,j,&u,&o,(INT)0);
3403 ip_y = FABS((R)(n0) * ths->x[2*j] - (R)(u)) * ((R)ip_s);
3404 ip_u = (INT)LRINT(FLOOR(ip_y));
3405 ip_w = ip_y - (R)(ip_u);
3406 for (l = 0; l < 2*m+2; l++)
3407 psij_const[l] = ths->psi[ABS(ip_u-l*ip_s)]*(K(1.0)-ip_w) + ths->psi[ABS(ip_u-l*ip_s+1)]*(ip_w);
3408
3409 uo(ths,j,&u,&o,(INT)1);
3410 ip_y = FABS((R)(n1) * ths->x[2*j+1] - (R)(u)) * ((R)ip_s);
3411 ip_u = (INT)(LRINT(FLOOR(ip_y)));
3412 ip_w = ip_y - (R)(ip_u);
3413 for (l = 0; l < 2*m+2; l++)
3414 psij_const[2*m+2+l] = ths->psi[(K+1)+ABS(ip_u-l*ip_s)]*(K(1.0)-ip_w) + ths->psi[(K+1)+ABS(ip_u-l*ip_s+1)]*(ip_w);
3415
3416 nfft_trafo_2d_compute(ths->f+j, g, psij_const, psij_const+2*m+2, ths->x+2*j, ths->x+2*j+1, n0, n1, m);
3417 }
3418 return;
3419 } /* if(PRE_LIN_PSI) */
3420
3421 /* no precomputed psi at all */
3422
3423 sort(ths);
3424
3425#ifdef _OPENMP
3426 #pragma omp parallel for default(shared) private(k)
3427#endif
3428 for (k = 0; k < M; k++)
3429 {
3430 R psij_const[2*(2*m+2)];
3431 INT u, o, l;
3432 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
3433
3434 uo(ths,j,&u,&o,(INT)0);
3435 for (l = 0; l <= 2*m+1; l++)
3436 psij_const[l]=(PHI(ths->n[0], ths->x[2*j] - ((R)((u+l))) / (R)(n0),0));
3437
3438 uo(ths,j,&u,&o,(INT)1);
3439 for (l = 0; l <= 2*m+1; l++)
3440 psij_const[2*m+2+l] = (PHI(ths->n[1], ths->x[2*j+1] - ((R)((u+l)))/(R)(n1),1));
3441
3442 nfft_trafo_2d_compute(ths->f+j, g, psij_const, psij_const+2*m+2, ths->x+2*j, ths->x+2*j+1, n0, n1, m);
3443 }
3444}
3445
3446#define MACRO_adjoint_2d_B_OMP_BLOCKWISE_COMPUTE_PRE_PSI \
3447 nfft_adjoint_2d_compute_omp_blockwise(ths->f[j], g, \
3448 ths->psi+j*2*(2*m+2), ths->psi+(j*2+1)*(2*m+2), \
3449 ths->x+2*j, ths->x+2*j+1, n0, n1, m, my_u0, my_o0);
3450
3451#define MACRO_adjoint_2d_B_OMP_BLOCKWISE_COMPUTE_PRE_FG_PSI \
3452{ \
3453 R psij_const[2*(2*m+2)]; \
3454 INT l; \
3455 R fg_psij0 = ths->psi[2*j*2]; \
3456 R fg_psij1 = ths->psi[2*j*2+1]; \
3457 R fg_psij2 = K(1.0); \
3458 \
3459 psij_const[0] = fg_psij0; \
3460 for(l=1; l<=2*m+1; l++) \
3461 { \
3462 fg_psij2 *= fg_psij1; \
3463 psij_const[l] = fg_psij0*fg_psij2*fg_exp_l[l]; \
3464 } \
3465 \
3466 fg_psij0 = ths->psi[2*(j*2+1)]; \
3467 fg_psij1 = ths->psi[2*(j*2+1)+1]; \
3468 fg_psij2 = K(1.0); \
3469 psij_const[2*m+2] = fg_psij0; \
3470 for(l=1; l<=2*m+1; l++) \
3471 { \
3472 fg_psij2 *= fg_psij1; \
3473 psij_const[2*m+2+l] = fg_psij0*fg_psij2*fg_exp_l[2*m+2+l]; \
3474 } \
3475 \
3476 nfft_adjoint_2d_compute_omp_blockwise(ths->f[j], g, \
3477 psij_const, psij_const+2*m+2, ths->x+2*j, ths->x+2*j+1, \
3478 n0, n1, m, my_u0, my_o0); \
3479}
3480
3481#define MACRO_adjoint_2d_B_OMP_BLOCKWISE_COMPUTE_FG_PSI \
3482{ \
3483 R psij_const[2*(2*m+2)]; \
3484 R fg_psij0, fg_psij1, fg_psij2; \
3485 INT u, o, l; \
3486 \
3487 uo(ths,j,&u,&o,(INT)0); \
3488 fg_psij0 = (PHI(ths->n[0],ths->x[2*j]-((R)u)/((R)n0),0)); \
3489 fg_psij1 = EXP(K(2.0)*(((R)n0)*(ths->x[2*j]) - (R)u)/ths->b[0]); \
3490 fg_psij2 = K(1.0); \
3491 psij_const[0] = fg_psij0; \
3492 for(l=1; l<=2*m+1; l++) \
3493 { \
3494 fg_psij2 *= fg_psij1; \
3495 psij_const[l] = fg_psij0*fg_psij2*fg_exp_l[l]; \
3496 } \
3497 \
3498 uo(ths,j,&u,&o,(INT)1); \
3499 fg_psij0 = (PHI(ths->n[1],ths->x[2*j+1]-((R)u)/((R)n1),1)); \
3500 fg_psij1 = EXP(K(2.0)*(((R)n1)*(ths->x[2*j+1]) - (R)u)/ths->b[1]); \
3501 fg_psij2 = K(1.0); \
3502 psij_const[2*m+2] = fg_psij0; \
3503 for(l=1; l<=2*m+1; l++) \
3504 { \
3505 fg_psij2 *= fg_psij1; \
3506 psij_const[2*m+2+l] = fg_psij0*fg_psij2*fg_exp_l[2*m+2+l]; \
3507 } \
3508 \
3509 nfft_adjoint_2d_compute_omp_blockwise(ths->f[j], g, \
3510 psij_const, psij_const+2*m+2, ths->x+2*j, ths->x+2*j+1, \
3511 n0, n1, m, my_u0, my_o0); \
3512}
3513
3514#define MACRO_adjoint_2d_B_OMP_BLOCKWISE_COMPUTE_PRE_LIN_PSI \
3515{ \
3516 R psij_const[2*(2*m+2)]; \
3517 INT u, o, l; \
3518 INT ip_u; \
3519 R ip_y, ip_w; \
3520 \
3521 uo(ths,j,&u,&o,(INT)0); \
3522 ip_y = FABS(((R)n0)*(ths->x[2*j]) - (R)u)*((R)ip_s); \
3523 ip_u = LRINT(FLOOR(ip_y)); \
3524 ip_w = ip_y-ip_u; \
3525 for(l=0; l < 2*m+2; l++) \
3526 psij_const[l] = ths->psi[ABS(ip_u-l*ip_s)]*(K(1.0)-ip_w) + \
3527 ths->psi[ABS(ip_u-l*ip_s+1)]*(ip_w); \
3528 \
3529 uo(ths,j,&u,&o,(INT)1); \
3530 ip_y = FABS(((R)n1)*(ths->x[2*j+1]) - (R)u)*((R)ip_s); \
3531 ip_u = LRINT(FLOOR(ip_y)); \
3532 ip_w = ip_y-ip_u; \
3533 for(l=0; l < 2*m+2; l++) \
3534 psij_const[2*m+2+l] = ths->psi[(K+1)+ABS(ip_u-l*ip_s)]*(K(1.0)-ip_w) + \
3535 ths->psi[(K+1)+ABS(ip_u-l*ip_s+1)]*(ip_w); \
3536 \
3537 nfft_adjoint_2d_compute_omp_blockwise(ths->f[j], g, \
3538 psij_const, psij_const+2*m+2, ths->x+2*j, ths->x+2*j+1, \
3539 n0, n1, m, my_u0, my_o0); \
3540}
3541
3542#define MACRO_adjoint_2d_B_OMP_BLOCKWISE_COMPUTE_NO_PSI \
3543{ \
3544 R psij_const[2*(2*m+2)]; \
3545 INT u, o, l; \
3546 \
3547 uo(ths,j,&u,&o,(INT)0); \
3548 for(l=0;l<=2*m+1;l++) \
3549 psij_const[l]=(PHI(ths->n[0],ths->x[2*j]-((R)((u+l)))/((R)n0),0)); \
3550 \
3551 uo(ths,j,&u,&o,(INT)1); \
3552 for(l=0;l<=2*m+1;l++) \
3553 psij_const[2*m+2+l]=(PHI(ths->n[1],ths->x[2*j+1]-((R)((u+l)))/((R)n1),1)); \
3554 \
3555 nfft_adjoint_2d_compute_omp_blockwise(ths->f[j], g, \
3556 psij_const, psij_const+2*m+2, ths->x+2*j, ths->x+2*j+1, \
3557 n0, n1, m, my_u0, my_o0); \
3558}
3559
3560#define MACRO_adjoint_2d_B_OMP_BLOCKWISE(whichone) \
3561{ \
3562 if (ths->flags & NFFT_OMP_BLOCKWISE_ADJOINT) \
3563 { \
3564 _Pragma("omp parallel private(k)") \
3565 { \
3566 INT my_u0, my_o0, min_u_a, max_u_a, min_u_b, max_u_b; \
3567 INT *ar_x = ths->index_x; \
3568 \
3569 nfft_adjoint_B_omp_blockwise_init(&my_u0, &my_o0, &min_u_a, &max_u_a, \
3570 &min_u_b, &max_u_b, 2, ths->n, m); \
3571 \
3572 if (min_u_a != -1) \
3573 { \
3574 k = index_x_binary_search(ar_x, M, min_u_a); \
3575 \
3576 MACRO_adjoint_nd_B_OMP_BLOCKWISE_ASSERT_A \
3577 \
3578 while (k < M) \
3579 { \
3580 INT u_prod = ar_x[2*k]; \
3581 INT j = ar_x[2*k+1]; \
3582 \
3583 if (u_prod < min_u_a || u_prod > max_u_a) \
3584 break; \
3585 \
3586 MACRO_adjoint_2d_B_OMP_BLOCKWISE_COMPUTE_ ##whichone \
3587 \
3588 k++; \
3589 } \
3590 } \
3591 \
3592 if (min_u_b != -1) \
3593 { \
3594 INT k = index_x_binary_search(ar_x, M, min_u_b); \
3595 \
3596 MACRO_adjoint_nd_B_OMP_BLOCKWISE_ASSERT_B \
3597 \
3598 while (k < M) \
3599 { \
3600 INT u_prod = ar_x[2*k]; \
3601 INT j = ar_x[2*k+1]; \
3602 \
3603 if (u_prod < min_u_b || u_prod > max_u_b) \
3604 break; \
3605 \
3606 MACRO_adjoint_2d_B_OMP_BLOCKWISE_COMPUTE_ ##whichone \
3607 \
3608 k++; \
3609 } \
3610 } \
3611 } /* omp parallel */ \
3612 return; \
3613 } /* if(NFFT_OMP_BLOCKWISE_ADJOINT) */ \
3614}
3615
3616
3617static void nfft_adjoint_2d_B(X(plan) *ths)
3618{
3619 const INT n0 = ths->n[0];
3620 const INT n1 = ths->n[1];
3621 const INT M = ths->M_total;
3622 const INT m = ths->m;
3623 C* g = (C*) ths->g;
3624 INT k;
3625
3626 memset(g, 0, (size_t)(ths->n_total) * sizeof(C));
3627
3628 if(ths->flags & PRE_FULL_PSI)
3629 {
3630 nfft_adjoint_B_compute_full_psi(g, ths->psi_index_g, ths->psi, ths->f, M,
3631 (INT)2, ths->n, m, ths->flags, ths->index_x);
3632 return;
3633 } /* if(PRE_FULL_PSI) */
3634
3635 if(ths->flags & PRE_PSI)
3636 {
3637#ifdef _OPENMP
3638 MACRO_adjoint_2d_B_OMP_BLOCKWISE(PRE_PSI)
3639#endif
3640
3641#ifdef _OPENMP
3642 #pragma omp parallel for default(shared) private(k)
3643#endif
3644 for (k = 0; k < M; k++)
3645 {
3646 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
3647#ifdef _OPENMP
3648 nfft_adjoint_2d_compute_omp_atomic(ths->f[j], g, ths->psi+j*2*(2*m+2), ths->psi+(j*2+1)*(2*m+2), ths->x+2*j, ths->x+2*j+1, n0, n1, m);
3649#else
3650 nfft_adjoint_2d_compute_serial(ths->f+j, g, ths->psi+j*2*(2*m+2), ths->psi+(j*2+1)*(2*m+2), ths->x+2*j, ths->x+2*j+1, n0, n1, m);
3651#endif
3652 }
3653 return;
3654 } /* if(PRE_PSI) */
3655
3656 if(ths->flags & PRE_FG_PSI)
3657 {
3658 R fg_exp_l[2*(2*m+2+1)];
3659
3660 nfft_2d_init_fg_exp_l(fg_exp_l, m, ths->b[0]);
3661 nfft_2d_init_fg_exp_l(fg_exp_l+2*m+2, m, ths->b[1]);
3662
3663#ifdef _OPENMP
3664 MACRO_adjoint_2d_B_OMP_BLOCKWISE(PRE_FG_PSI)
3665#endif
3666
3667
3668#ifdef _OPENMP
3669 #pragma omp parallel for default(shared) private(k)
3670#endif
3671 for (k = 0; k < M; k++)
3672 {
3673 R psij_const[2*(2*m+2)];
3674 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
3675 INT l;
3676 R fg_psij0 = ths->psi[2*j*2];
3677 R fg_psij1 = ths->psi[2*j*2+1];
3678 R fg_psij2 = K(1.0);
3679
3680 psij_const[0] = fg_psij0;
3681 for(l=1; l<=2*m+1; l++)
3682 {
3683 fg_psij2 *= fg_psij1;
3684 psij_const[l] = fg_psij0*fg_psij2*fg_exp_l[l];
3685 }
3686
3687 fg_psij0 = ths->psi[2*(j*2+1)];
3688 fg_psij1 = ths->psi[2*(j*2+1)+1];
3689 fg_psij2 = K(1.0);
3690 psij_const[2*m+2] = fg_psij0;
3691 for(l=1; l<=2*m+1; l++)
3692 {
3693 fg_psij2 *= fg_psij1;
3694 psij_const[2*m+2+l] = fg_psij0*fg_psij2*fg_exp_l[2*m+2+l];
3695 }
3696
3697#ifdef _OPENMP
3698 nfft_adjoint_2d_compute_omp_atomic(ths->f[j], g, psij_const, psij_const+2*m+2, ths->x+2*j, ths->x+2*j+1, n0, n1, m);
3699#else
3700 nfft_adjoint_2d_compute_serial(ths->f+j, g, psij_const, psij_const+2*m+2, ths->x+2*j, ths->x+2*j+1, n0, n1, m);
3701#endif
3702 }
3703
3704 return;
3705 } /* if(PRE_FG_PSI) */
3706
3707 if(ths->flags & FG_PSI)
3708 {
3709 R fg_exp_l[2*(2*m+2+1)];
3710
3711 nfft_2d_init_fg_exp_l(fg_exp_l, m, ths->b[0]);
3712 nfft_2d_init_fg_exp_l(fg_exp_l+2*m+2, m, ths->b[1]);
3713
3714 sort(ths);
3715
3716#ifdef _OPENMP
3717 MACRO_adjoint_2d_B_OMP_BLOCKWISE(FG_PSI)
3718#endif
3719
3720#ifdef _OPENMP
3721 #pragma omp parallel for default(shared) private(k)
3722#endif
3723 for (k = 0; k < M; k++)
3724 {
3725 INT u, o, l;
3726 R fg_psij0, fg_psij1, fg_psij2;
3727 R psij_const[2*(2*m+2)];
3728 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
3729
3730 uo(ths,j,&u,&o,(INT)0);
3731 fg_psij0 = (PHI(ths->n[0], ths->x[2*j] - ((R)u)/(R)(n0),0));
3732 fg_psij1 = EXP(K(2.0) * ((R)(n0) * (ths->x[2*j]) - (R)(u)) / ths->b[0]);
3733 fg_psij2 = K(1.0);
3734 psij_const[0] = fg_psij0;
3735 for(l=1; l<=2*m+1; l++)
3736 {
3737 fg_psij2 *= fg_psij1;
3738 psij_const[l] = fg_psij0*fg_psij2*fg_exp_l[l];
3739 }
3740
3741 uo(ths,j,&u,&o,(INT)1);
3742 fg_psij0 = (PHI(ths->n[1], ths->x[2*j+1] - ((R)u) / (R)(n1),1));
3743 fg_psij1 = EXP(K(2.0) * ((R)(n1) * (ths->x[2*j+1]) - (R)(u)) / ths->b[1]);
3744 fg_psij2 = K(1.0);
3745 psij_const[2*m+2] = fg_psij0;
3746 for(l=1; l<=2*m+1; l++)
3747 {
3748 fg_psij2 *= fg_psij1;
3749 psij_const[2*m+2+l] = fg_psij0*fg_psij2*fg_exp_l[2*m+2+l];
3750 }
3751
3752#ifdef _OPENMP
3753 nfft_adjoint_2d_compute_omp_atomic(ths->f[j], g, psij_const, psij_const+2*m+2, ths->x+2*j, ths->x+2*j+1, n0, n1, m);
3754#else
3755 nfft_adjoint_2d_compute_serial(ths->f+j, g, psij_const, psij_const+2*m+2, ths->x+2*j, ths->x+2*j+1, n0, n1, m);
3756#endif
3757 }
3758
3759 return;
3760 } /* if(FG_PSI) */
3761
3762 if(ths->flags & PRE_LIN_PSI)
3763 {
3764 const INT K = ths->K;
3765 const INT ip_s = K / (m + 2);
3766
3767 sort(ths);
3768
3769#ifdef _OPENMP
3770 MACRO_adjoint_2d_B_OMP_BLOCKWISE(PRE_LIN_PSI)
3771#endif
3772
3773#ifdef _OPENMP
3774 #pragma omp parallel for default(shared) private(k)
3775#endif
3776 for (k = 0; k < M; k++)
3777 {
3778 INT u,o,l;
3779 INT ip_u;
3780 R ip_y, ip_w;
3781 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
3782 R psij_const[2*(2*m+2)];
3783
3784 uo(ths,j,&u,&o,(INT)0);
3785 ip_y = FABS((R)(n0) * (ths->x[2*j]) - (R)(u)) * ((R)ip_s);
3786 ip_u = (INT)(LRINT(FLOOR(ip_y)));
3787 ip_w = ip_y - (R)(ip_u);
3788 for(l=0; l < 2*m+2; l++)
3789 psij_const[l] = ths->psi[ABS(ip_u-l*ip_s)]*(K(1.0)-ip_w) +
3790 ths->psi[ABS(ip_u-l*ip_s+1)]*(ip_w);
3791
3792 uo(ths,j,&u,&o,(INT)1);
3793 ip_y = FABS((R)(n1) * (ths->x[2*j+1]) - (R)(u)) * ((R)ip_s);
3794 ip_u = (INT)(LRINT(FLOOR(ip_y)));
3795 ip_w = ip_y - (R)(ip_u);
3796 for(l=0; l < 2*m+2; l++)
3797 psij_const[2*m+2+l] = ths->psi[(K+1)+ABS(ip_u-l*ip_s)]*(K(1.0)-ip_w) +
3798 ths->psi[(K+1)+ABS(ip_u-l*ip_s+1)]*(ip_w);
3799
3800#ifdef _OPENMP
3801 nfft_adjoint_2d_compute_omp_atomic(ths->f[j], g, psij_const, psij_const+2*m+2, ths->x+2*j, ths->x+2*j+1, n0, n1, m);
3802#else
3803 nfft_adjoint_2d_compute_serial(ths->f+j, g, psij_const, psij_const+2*m+2, ths->x+2*j, ths->x+2*j+1, n0, n1, m);
3804#endif
3805 }
3806 return;
3807 } /* if(PRE_LIN_PSI) */
3808
3809 /* no precomputed psi at all */
3810 sort(ths);
3811
3812#ifdef _OPENMP
3813 MACRO_adjoint_2d_B_OMP_BLOCKWISE(NO_PSI)
3814#endif
3815
3816#ifdef _OPENMP
3817 #pragma omp parallel for default(shared) private(k)
3818#endif
3819 for (k = 0; k < M; k++)
3820 {
3821 INT u,o,l;
3822 R psij_const[2*(2*m+2)];
3823 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
3824
3825 uo(ths,j,&u,&o,(INT)0);
3826 for(l=0;l<=2*m+1;l++)
3827 psij_const[l]=(PHI(ths->n[0], ths->x[2*j] - ((R)((u+l))) / (R)(n0),0));
3828
3829 uo(ths,j,&u,&o,(INT)1);
3830 for(l=0;l<=2*m+1;l++)
3831 psij_const[2*m+2+l]=(PHI(ths->n[1], ths->x[2*j+1] - ((R)((u+l))) / (R)(n1),1));
3832
3833#ifdef _OPENMP
3834 nfft_adjoint_2d_compute_omp_atomic(ths->f[j], g, psij_const, psij_const+2*m+2, ths->x+2*j, ths->x+2*j+1, n0, n1, m);
3835#else
3836 nfft_adjoint_2d_compute_serial(ths->f+j, g, psij_const, psij_const+2*m+2, ths->x+2*j, ths->x+2*j+1, n0, n1, m);
3837#endif
3838 }
3839}
3840
3841
3842void X(trafo_2d)(X(plan) *ths)
3843{
3844 if((ths->N[0] <= ths->m) || (ths->N[1] <= ths->m) || (ths->n[0] <= 2*ths->m+2) || (ths->n[1] <= 2*ths->m+2))
3845 {
3846 X(trafo_direct)(ths);
3847 return;
3848 }
3849
3850 INT k0,k1,n0,n1,N0,N1;
3851 C *g_hat,*f_hat;
3852 R *c_phi_inv01, *c_phi_inv02, *c_phi_inv11, *c_phi_inv12;
3853 R ck01, ck02, ck11, ck12;
3854 C *g_hat11,*f_hat11,*g_hat21,*f_hat21,*g_hat12,*f_hat12,*g_hat22,*f_hat22;
3855
3856 ths->g_hat=ths->g1;
3857 ths->g=ths->g2;
3858
3859 N0=ths->N[0];
3860 N1=ths->N[1];
3861 n0=ths->n[0];
3862 n1=ths->n[1];
3863
3864 f_hat=(C*)ths->f_hat;
3865 g_hat=(C*)ths->g_hat;
3866
3867 TIC(0)
3868#ifdef _OPENMP
3869 #pragma omp parallel for default(shared) private(k0)
3870 for (k0 = 0; k0 < ths->n_total; k0++)
3871 ths->g_hat[k0] = 0.0;
3872#else
3873 memset(ths->g_hat, 0, (size_t)(ths->n_total) * sizeof(C));
3874#endif
3875 if(ths->flags & PRE_PHI_HUT)
3876 {
3877 c_phi_inv01=ths->c_phi_inv[0];
3878 c_phi_inv02=&ths->c_phi_inv[0][N0/2];
3879
3880#ifdef _OPENMP
3881 #pragma omp parallel for default(shared) private(k0,k1,ck01,ck02,c_phi_inv11,c_phi_inv12,g_hat11,f_hat11,g_hat21,f_hat21,g_hat12,f_hat12,g_hat22,f_hat22,ck11,ck12)
3882#endif
3883 for(k0=0;k0<N0/2;k0++)
3884 {
3885 ck01=c_phi_inv01[k0];
3886 ck02=c_phi_inv02[k0];
3887
3888 c_phi_inv11=ths->c_phi_inv[1];
3889 c_phi_inv12=&ths->c_phi_inv[1][N1/2];
3890
3891 g_hat11=g_hat + (n0-(N0/2)+k0)*n1+n1-(N1/2);
3892 f_hat11=f_hat + k0*N1;
3893 g_hat21=g_hat + k0*n1+n1-(N1/2);
3894 f_hat21=f_hat + ((N0/2)+k0)*N1;
3895 g_hat12=g_hat + (n0-(N0/2)+k0)*n1;
3896 f_hat12=f_hat + k0*N1+(N1/2);
3897 g_hat22=g_hat + k0*n1;
3898 f_hat22=f_hat + ((N0/2)+k0)*N1+(N1/2);
3899
3900 for(k1=0;k1<N1/2;k1++)
3901 {
3902 ck11=c_phi_inv11[k1];
3903 ck12=c_phi_inv12[k1];
3904
3905 g_hat11[k1] = f_hat11[k1] * ck01 * ck11;
3906 g_hat21[k1] = f_hat21[k1] * ck02 * ck11;
3907 g_hat12[k1] = f_hat12[k1] * ck01 * ck12;
3908 g_hat22[k1] = f_hat22[k1] * ck02 * ck12;
3909 }
3910 }
3911 }
3912 else
3913#ifdef _OPENMP
3914 #pragma omp parallel for default(shared) private(k0,k1,ck01,ck02,ck11,ck12)
3915#endif
3916 for(k0=0;k0<N0/2;k0++)
3917 {
3918 ck01=K(1.0)/(PHI_HUT(ths->n[0],k0-N0/2,0));
3919 ck02=K(1.0)/(PHI_HUT(ths->n[0],k0,0));
3920 for(k1=0;k1<N1/2;k1++)
3921 {
3922 ck11=K(1.0)/(PHI_HUT(ths->n[1],k1-N1/2,1));
3923 ck12=K(1.0)/(PHI_HUT(ths->n[1],k1,1));
3924 g_hat[(n0-N0/2+k0)*n1+n1-N1/2+k1] = f_hat[k0*N1+k1] * ck01 * ck11;
3925 g_hat[k0*n1+n1-N1/2+k1] = f_hat[(N0/2+k0)*N1+k1] * ck02 * ck11;
3926 g_hat[(n0-N0/2+k0)*n1+k1] = f_hat[k0*N1+N1/2+k1] * ck01 * ck12;
3927 g_hat[k0*n1+k1] = f_hat[(N0/2+k0)*N1+N1/2+k1] * ck02 * ck12;
3928 }
3929 }
3930
3931 TOC(0)
3932
3933 TIC_FFTW(1)
3934 FFTW(execute)(ths->my_fftw_plan1);
3935 TOC_FFTW(1);
3936
3937 TIC(2);
3938 nfft_trafo_2d_B(ths);
3939 TOC(2);
3940}
3941
3942void X(adjoint_2d)(X(plan) *ths)
3943{
3944 if((ths->N[0] <= ths->m) || (ths->N[1] <= ths->m) || (ths->n[0] <= 2*ths->m+2) || (ths->n[1] <= 2*ths->m+2))
3945 {
3946 X(adjoint_direct)(ths);
3947 return;
3948 }
3949
3950 INT k0,k1,n0,n1,N0,N1;
3951 C *g_hat,*f_hat;
3952 R *c_phi_inv01, *c_phi_inv02, *c_phi_inv11, *c_phi_inv12;
3953 R ck01, ck02, ck11, ck12;
3954 C *g_hat11,*f_hat11,*g_hat21,*f_hat21,*g_hat12,*f_hat12,*g_hat22,*f_hat22;
3955
3956 ths->g_hat=ths->g1;
3957 ths->g=ths->g2;
3958
3959 N0=ths->N[0];
3960 N1=ths->N[1];
3961 n0=ths->n[0];
3962 n1=ths->n[1];
3963
3964 f_hat=(C*)ths->f_hat;
3965 g_hat=(C*)ths->g_hat;
3966
3967 TIC(2);
3968 nfft_adjoint_2d_B(ths);
3969 TOC(2);
3970
3971 TIC_FFTW(1)
3972 FFTW(execute)(ths->my_fftw_plan2);
3973 TOC_FFTW(1);
3974
3975 TIC(0)
3976 if(ths->flags & PRE_PHI_HUT)
3977 {
3978 c_phi_inv01=ths->c_phi_inv[0];
3979 c_phi_inv02=&ths->c_phi_inv[0][N0/2];
3980
3981#ifdef _OPENMP
3982 #pragma omp parallel for default(shared) private(k0,k1,ck01,ck02,c_phi_inv11,c_phi_inv12,g_hat11,f_hat11,g_hat21,f_hat21,g_hat12,f_hat12,g_hat22,f_hat22,ck11,ck12)
3983#endif
3984 for(k0=0;k0<N0/2;k0++)
3985 {
3986 ck01=c_phi_inv01[k0];
3987 ck02=c_phi_inv02[k0];
3988
3989 c_phi_inv11=ths->c_phi_inv[1];
3990 c_phi_inv12=&ths->c_phi_inv[1][N1/2];
3991
3992 g_hat11=g_hat + (n0-(N0/2)+k0)*n1+n1-(N1/2);
3993 f_hat11=f_hat + k0*N1;
3994 g_hat21=g_hat + k0*n1+n1-(N1/2);
3995 f_hat21=f_hat + ((N0/2)+k0)*N1;
3996 g_hat12=g_hat + (n0-(N0/2)+k0)*n1;
3997 f_hat12=f_hat + k0*N1+(N1/2);
3998 g_hat22=g_hat + k0*n1;
3999 f_hat22=f_hat + ((N0/2)+k0)*N1+(N1/2);
4000
4001 for(k1=0;k1<N1/2;k1++)
4002 {
4003 ck11=c_phi_inv11[k1];
4004 ck12=c_phi_inv12[k1];
4005
4006 f_hat11[k1] = g_hat11[k1] * ck01 * ck11;
4007 f_hat21[k1] = g_hat21[k1] * ck02 * ck11;
4008 f_hat12[k1] = g_hat12[k1] * ck01 * ck12;
4009 f_hat22[k1] = g_hat22[k1] * ck02 * ck12;
4010 }
4011 }
4012 }
4013 else
4014#ifdef _OPENMP
4015 #pragma omp parallel for default(shared) private(k0,k1,ck01,ck02,ck11,ck12)
4016#endif
4017 for(k0=0;k0<N0/2;k0++)
4018 {
4019 ck01=K(1.0)/(PHI_HUT(ths->n[0],k0-N0/2,0));
4020 ck02=K(1.0)/(PHI_HUT(ths->n[0],k0,0));
4021 for(k1=0;k1<N1/2;k1++)
4022 {
4023 ck11=K(1.0)/(PHI_HUT(ths->n[1],k1-N1/2,1));
4024 ck12=K(1.0)/(PHI_HUT(ths->n[1],k1,1));
4025 f_hat[k0*N1+k1] = g_hat[(n0-N0/2+k0)*n1+n1-N1/2+k1] * ck01 * ck11;
4026 f_hat[(N0/2+k0)*N1+k1] = g_hat[k0*n1+n1-N1/2+k1] * ck02 * ck11;
4027 f_hat[k0*N1+N1/2+k1] = g_hat[(n0-N0/2+k0)*n1+k1] * ck01 * ck12;
4028 f_hat[(N0/2+k0)*N1+N1/2+k1] = g_hat[k0*n1+k1] * ck02 * ck12;
4029 }
4030 }
4031 TOC(0)
4032}
4033
4034/* ################################################ SPECIFIC VERSIONS FOR d=3 */
4035
4036static void nfft_3d_init_fg_exp_l(R *fg_exp_l, const INT m, const R b)
4037{
4038 INT l;
4039 R fg_exp_b0, fg_exp_b1, fg_exp_b2, fg_exp_b0_sq;
4040
4041 fg_exp_b0 = EXP(-K(1.0) / b);
4042 fg_exp_b0_sq = fg_exp_b0*fg_exp_b0;
4043 fg_exp_b1 = K(1.0);
4044 fg_exp_b2 = K(1.0);
4045 fg_exp_l[0] = K(1.0);
4046 for(l=1; l <= 2*m+1; l++)
4047 {
4048 fg_exp_b2 = fg_exp_b1*fg_exp_b0;
4049 fg_exp_b1 *= fg_exp_b0_sq;
4050 fg_exp_l[l] = fg_exp_l[l-1]*fg_exp_b2;
4051 }
4052}
4053
4054static void nfft_trafo_3d_compute(C *fj, const C *g, const R *psij_const0,
4055 const R *psij_const1, const R *psij_const2, const R *xj0, const R *xj1,
4056 const R *xj2, const INT n0, const INT n1, const INT n2, const INT m)
4057{
4058 INT u0, o0, l0, u1, o1, l1, u2, o2, l2;
4059 const C *gj;
4060 const R *psij0, *psij1, *psij2;
4061
4062 psij0 = psij_const0;
4063 psij1 = psij_const1;
4064 psij2 = psij_const2;
4065
4066 uo2(&u0, &o0, *xj0, n0, m);
4067 uo2(&u1, &o1, *xj1, n1, m);
4068 uo2(&u2, &o2, *xj2, n2, m);
4069
4070 *fj = 0;
4071
4072 if (u0 < o0)
4073 if (u1 < o1)
4074 if (u2 < o2)
4075 for (l0 = 0; l0 <= 2 * m + 1; l0++, psij0++)
4076 {
4077 psij1 = psij_const1;
4078 for (l1 = 0; l1 <= 2 * m + 1; l1++, psij1++)
4079 {
4080 psij2 = psij_const2;
4081 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2 + u2;
4082 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4083 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4084 }
4085 }
4086 else
4087 /* asserts (u2>o2)*/
4088 for (l0 = 0; l0 <= 2 * m + 1; l0++, psij0++)
4089 {
4090 psij1 = psij_const1;
4091 for (l1 = 0; l1 <= 2 * m + 1; l1++, psij1++)
4092 {
4093 psij2 = psij_const2;
4094 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2 + u2;
4095 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4096 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4097 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2;
4098 for (l2 = 0; l2 <= o2; l2++)
4099 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4100 }
4101 }
4102 else /* asserts (u1>o1)*/
4103 if (u2 < o2)
4104 for (l0 = 0; l0 <= 2 * m + 1; l0++, psij0++)
4105 {
4106 psij1 = psij_const1;
4107 for (l1 = 0; l1 < 2 * m + 1 - o1; l1++, psij1++)
4108 {
4109 psij2 = psij_const2;
4110 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2 + u2;
4111 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4112 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4113 }
4114 for (l1 = 0; l1 <= o1; l1++, psij1++)
4115 {
4116 psij2 = psij_const2;
4117 gj = g + ((u0 + l0) * n1 + l1) * n2 + u2;
4118 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4119 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4120 }
4121 }
4122 else/* asserts (u2>o2) */
4123 {
4124 for (l0 = 0; l0 <= 2 * m + 1; l0++, psij0++)
4125 {
4126 psij1 = psij_const1;
4127 for (l1 = 0; l1 < 2 * m + 1 - o1; l1++, psij1++)
4128 {
4129 psij2 = psij_const2;
4130 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2 + u2;
4131 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4132 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4133 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2;
4134 for (l2 = 0; l2 <= o2; l2++)
4135 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4136 }
4137 for (l1 = 0; l1 <= o1; l1++, psij1++)
4138 {
4139 psij2 = psij_const2;
4140 gj = g + ((u0 + l0) * n1 + l1) * n2 + u2;
4141 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4142 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4143 gj = g + ((u0 + l0) * n1 + l1) * n2;
4144 for (l2 = 0; l2 <= o2; l2++)
4145 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4146 }
4147 }
4148 }
4149 else /* asserts (u0>o0) */
4150 if (u1 < o1)
4151 if (u2 < o2)
4152 {
4153 for (l0 = 0; l0 < 2 * m + 1 - o0; l0++, psij0++)
4154 {
4155 psij1 = psij_const1;
4156 for (l1 = 0; l1 <= 2 * m + 1; l1++, psij1++)
4157 {
4158 psij2 = psij_const2;
4159 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2 + u2;
4160 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4161 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4162 }
4163 }
4164
4165 for (l0 = 0; l0 <= o0; l0++, psij0++)
4166 {
4167 psij1 = psij_const1;
4168 for (l1 = 0; l1 <= 2 * m + 1; l1++, psij1++)
4169 {
4170 psij2 = psij_const2;
4171 gj = g + (l0 * n1 + (u1 + l1)) * n2 + u2;
4172 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4173 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4174 }
4175 }
4176 } else/* asserts (u2>o2) */
4177 {
4178 for (l0 = 0; l0 < 2 * m + 1 - o0; l0++, psij0++)
4179 {
4180 psij1 = psij_const1;
4181 for (l1 = 0; l1 <= 2 * m + 1; l1++, psij1++)
4182 {
4183 psij2 = psij_const2;
4184 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2 + u2;
4185 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4186 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4187 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2;
4188 for (l2 = 0; l2 <= o2; l2++)
4189 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4190 }
4191 }
4192
4193 for (l0 = 0; l0 <= o0; l0++, psij0++)
4194 {
4195 psij1 = psij_const1;
4196 for (l1 = 0; l1 <= 2 * m + 1; l1++, psij1++)
4197 {
4198 psij2 = psij_const2;
4199 gj = g + (l0 * n1 + (u1 + l1)) * n2 + u2;
4200 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4201 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4202 gj = g + (l0 * n1 + (u1 + l1)) * n2;
4203 for (l2 = 0; l2 <= o2; l2++)
4204 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4205 }
4206 }
4207 }
4208 else /* asserts (u1>o1) */
4209 if (u2 < o2)
4210 {
4211 for (l0 = 0; l0 < 2 * m + 1 - o0; l0++, psij0++)
4212 {
4213 psij1 = psij_const1;
4214 for (l1 = 0; l1 < 2 * m + 1 - o1; l1++, psij1++)
4215 {
4216 psij2 = psij_const2;
4217 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2 + u2;
4218 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4219 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4220 }
4221 for (l1 = 0; l1 <= o1; l1++, psij1++)
4222 {
4223 psij2 = psij_const2;
4224 gj = g + ((u0 + l0) * n1 + l1) * n2 + u2;
4225 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4226 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4227 }
4228 }
4229 for (l0 = 0; l0 <= o0; l0++, psij0++)
4230 {
4231 psij1 = psij_const1;
4232 for (l1 = 0; l1 < 2 * m + 1 - o1; l1++, psij1++)
4233 {
4234 psij2 = psij_const2;
4235 gj = g + (l0 * n1 + (u1 + l1)) * n2 + u2;
4236 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4237 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4238 }
4239 for (l1 = 0; l1 <= o1; l1++, psij1++)
4240 {
4241 psij2 = psij_const2;
4242 gj = g + (l0 * n1 + l1) * n2 + u2;
4243 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4244 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4245 }
4246 }
4247 } else/* asserts (u2>o2) */
4248 {
4249 for (l0 = 0; l0 < 2 * m + 1 - o0; l0++, psij0++)
4250 {
4251 psij1 = psij_const1;
4252 for (l1 = 0; l1 < 2 * m + 1 - o1; l1++, psij1++)
4253 {
4254 psij2 = psij_const2;
4255 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2 + u2;
4256 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4257 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4258 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2;
4259 for (l2 = 0; l2 <= o2; l2++)
4260 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4261 }
4262 for (l1 = 0; l1 <= o1; l1++, psij1++)
4263 {
4264 psij2 = psij_const2;
4265 gj = g + ((u0 + l0) * n1 + l1) * n2 + u2;
4266 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4267 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4268 gj = g + ((u0 + l0) * n1 + l1) * n2;
4269 for (l2 = 0; l2 <= o2; l2++)
4270 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4271 }
4272 }
4273
4274 for (l0 = 0; l0 <= o0; l0++, psij0++)
4275 {
4276 psij1 = psij_const1;
4277 for (l1 = 0; l1 < 2 * m + 1 - o1; l1++, psij1++)
4278 {
4279 psij2 = psij_const2;
4280 gj = g + (l0 * n1 + (u1 + l1)) * n2 + u2;
4281 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4282 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4283 gj = g + (l0 * n1 + (u1 + l1)) * n2;
4284 for (l2 = 0; l2 <= o2; l2++)
4285 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4286 }
4287 for (l1 = 0; l1 <= o1; l1++, psij1++)
4288 {
4289 psij2 = psij_const2;
4290 gj = g + (l0 * n1 + l1) * n2 + u2;
4291 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4292 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4293 gj = g + (l0 * n1 + l1) * n2;
4294 for (l2 = 0; l2 <= o2; l2++)
4295 (*fj) += (*psij0) * (*psij1) * (*psij2++) * (*gj++);
4296 }
4297 }
4298 }
4299}
4300
4301#ifdef _OPENMP
4323static void nfft_adjoint_3d_compute_omp_blockwise(const C f, C *g,
4324 const R *psij_const0, const R *psij_const1, const R *psij_const2,
4325 const R *xj0, const R *xj1, const R *xj2,
4326 const INT n0, const INT n1, const INT n2, const INT m,
4327 const INT my_u0, const INT my_o0)
4328{
4329 INT ar_u0,ar_o0,l0,u1,o1,l1,u2,o2,l2;
4330
4331 INT index_temp1[2*m+2];
4332 INT index_temp2[2*m+2];
4333
4334 uo2(&ar_u0,&ar_o0,*xj0, n0, m);
4335 uo2(&u1,&o1,*xj1, n1, m);
4336 uo2(&u2,&o2,*xj2, n2, m);
4337
4338 for (l1=0; l1<=2*m+1; l1++)
4339 index_temp1[l1] = (u1+l1)%n1;
4340
4341 for (l2=0; l2<=2*m+1; l2++)
4342 index_temp2[l2] = (u2+l2)%n2;
4343
4344 if(ar_u0<ar_o0)
4345 {
4346 INT u0 = MAX(my_u0,ar_u0);
4347 INT o0 = MIN(my_o0,ar_o0);
4348 INT offset_psij = u0-ar_u0;
4349#ifdef OMP_ASSERT
4350 assert(offset_psij >= 0);
4351 assert(o0-u0 <= 2*m+1);
4352 assert(offset_psij+o0-u0 <= 2*m+1);
4353#endif
4354
4355 for (l0 = 0; l0 <= o0-u0; l0++)
4356 {
4357 const INT i0 = (u0+l0) * n1;
4358 const C val0 = psij_const0[offset_psij+l0];
4359
4360 for(l1=0; l1<=2*m+1; l1++)
4361 {
4362 const INT i1 = (i0 + index_temp1[l1]) * n2;
4363 const C val1 = psij_const1[l1];
4364
4365 for(l2=0; l2<=2*m+1; l2++)
4366 g[i1 + index_temp2[l2]] += val0 * val1 * psij_const2[l2] * f;
4367 }
4368 }
4369 }
4370 else
4371 {
4372 INT u0 = MAX(my_u0,ar_u0);
4373 INT o0 = my_o0;
4374 INT offset_psij = u0-ar_u0;
4375#ifdef OMP_ASSERT
4376 assert(offset_psij >= 0);
4377 assert(o0-u0 <= 2*m+1);
4378 assert(offset_psij+o0-u0 <= 2*m+1);
4379#endif
4380
4381 for (l0 = 0; l0 <= o0-u0; l0++)
4382 {
4383 INT i0 = (u0+l0) * n1;
4384 const C val0 = psij_const0[offset_psij+l0];
4385
4386 for(l1=0; l1<=2*m+1; l1++)
4387 {
4388 const INT i1 = (i0 + index_temp1[l1]) * n2;
4389 const C val1 = psij_const1[l1];
4390
4391 for(l2=0; l2<=2*m+1; l2++)
4392 g[i1 + index_temp2[l2]] += val0 * val1 * psij_const2[l2] * f;
4393 }
4394 }
4395
4396 u0 = my_u0;
4397 o0 = MIN(my_o0,ar_o0);
4398 offset_psij += my_u0-ar_u0+n0;
4399
4400#ifdef OMP_ASSERT
4401 if (u0<=o0)
4402 {
4403 assert(o0-u0 <= 2*m+1);
4404 assert(offset_psij+o0-u0 <= 2*m+1);
4405 }
4406#endif
4407 for (l0 = 0; l0 <= o0-u0; l0++)
4408 {
4409 INT i0 = (u0+l0) * n1;
4410 const C val0 = psij_const0[offset_psij+l0];
4411
4412 for(l1=0; l1<=2*m+1; l1++)
4413 {
4414 const INT i1 = (i0 + index_temp1[l1]) * n2;
4415 const C val1 = psij_const1[l1];
4416
4417 for(l2=0; l2<=2*m+1; l2++)
4418 g[i1 + index_temp2[l2]] += val0 * val1 * psij_const2[l2] * f;
4419 }
4420 }
4421 }
4422}
4423#endif
4424
4425#ifdef _OPENMP
4426/* adjoint NFFT three-dimensional case with OpenMP atomic operations */
4427static void nfft_adjoint_3d_compute_omp_atomic(const C f, C *g,
4428 const R *psij_const0, const R *psij_const1, const R *psij_const2,
4429 const R *xj0, const R *xj1, const R *xj2,
4430 const INT n0, const INT n1, const INT n2, const INT m)
4431{
4432 INT u0,o0,l0,u1,o1,l1,u2,o2,l2;
4433
4434 INT index_temp0[2*m+2];
4435 INT index_temp1[2*m+2];
4436 INT index_temp2[2*m+2];
4437
4438 uo2(&u0,&o0,*xj0, n0, m);
4439 uo2(&u1,&o1,*xj1, n1, m);
4440 uo2(&u2,&o2,*xj2, n2, m);
4441
4442 for (l0=0; l0<=2*m+1; l0++)
4443 index_temp0[l0] = (u0+l0)%n0;
4444
4445 for (l1=0; l1<=2*m+1; l1++)
4446 index_temp1[l1] = (u1+l1)%n1;
4447
4448 for (l2=0; l2<=2*m+1; l2++)
4449 index_temp2[l2] = (u2+l2)%n2;
4450
4451 for(l0=0; l0<=2*m+1; l0++)
4452 {
4453 for(l1=0; l1<=2*m+1; l1++)
4454 {
4455 for(l2=0; l2<=2*m+1; l2++)
4456 {
4457 INT i = (index_temp0[l0] * n1 + index_temp1[l1]) * n2 + index_temp2[l2];
4458 C *lhs = g+i;
4459 R *lhs_real = (R*)lhs;
4460 C val = psij_const0[l0] * psij_const1[l1] * psij_const2[l2] * f;
4461
4462#pragma omp atomic
4463 lhs_real[0] += CREAL(val);
4464
4465#pragma omp atomic
4466 lhs_real[1] += CIMAG(val);
4467 }
4468 }
4469 }
4470}
4471#endif
4472
4473#ifndef _OPENMP
4474static void nfft_adjoint_3d_compute_serial(const C *fj, C *g,
4475 const R *psij_const0, const R *psij_const1, const R *psij_const2, const R *xj0,
4476 const R *xj1, const R *xj2, const INT n0, const INT n1, const INT n2,
4477 const INT m)
4478{
4479 INT u0, o0, l0, u1, o1, l1, u2, o2, l2;
4480 C *gj;
4481 const R *psij0, *psij1, *psij2;
4482
4483 psij0 = psij_const0;
4484 psij1 = psij_const1;
4485 psij2 = psij_const2;
4486
4487 uo2(&u0, &o0, *xj0, n0, m);
4488 uo2(&u1, &o1, *xj1, n1, m);
4489 uo2(&u2, &o2, *xj2, n2, m);
4490
4491 if (u0 < o0)
4492 if (u1 < o1)
4493 if (u2 < o2)
4494 for (l0 = 0; l0 <= 2 * m + 1; l0++, psij0++)
4495 {
4496 psij1 = psij_const1;
4497 for (l1 = 0; l1 <= 2 * m + 1; l1++, psij1++)
4498 {
4499 psij2 = psij_const2;
4500 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2 + u2;
4501 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4502 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4503 }
4504 }
4505 else
4506 /* asserts (u2>o2)*/
4507 for (l0 = 0; l0 <= 2 * m + 1; l0++, psij0++)
4508 {
4509 psij1 = psij_const1;
4510 for (l1 = 0; l1 <= 2 * m + 1; l1++, psij1++)
4511 {
4512 psij2 = psij_const2;
4513 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2 + u2;
4514 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4515 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4516 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2;
4517 for (l2 = 0; l2 <= o2; l2++)
4518 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4519 }
4520 }
4521 else /* asserts (u1>o1)*/
4522 if (u2 < o2)
4523 for (l0 = 0; l0 <= 2 * m + 1; l0++, psij0++)
4524 {
4525 psij1 = psij_const1;
4526 for (l1 = 0; l1 < 2 * m + 1 - o1; l1++, psij1++)
4527 {
4528 psij2 = psij_const2;
4529 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2 + u2;
4530 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4531 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4532 }
4533 for (l1 = 0; l1 <= o1; l1++, psij1++)
4534 {
4535 psij2 = psij_const2;
4536 gj = g + ((u0 + l0) * n1 + l1) * n2 + u2;
4537 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4538 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4539 }
4540 }
4541 else/* asserts (u2>o2) */
4542 {
4543 for (l0 = 0; l0 <= 2 * m + 1; l0++, psij0++)
4544 {
4545 psij1 = psij_const1;
4546 for (l1 = 0; l1 < 2 * m + 1 - o1; l1++, psij1++)
4547 {
4548 psij2 = psij_const2;
4549 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2 + u2;
4550 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4551 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4552 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2;
4553 for (l2 = 0; l2 <= o2; l2++)
4554 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4555 }
4556 for (l1 = 0; l1 <= o1; l1++, psij1++)
4557 {
4558 psij2 = psij_const2;
4559 gj = g + ((u0 + l0) * n1 + l1) * n2 + u2;
4560 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4561 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4562 gj = g + ((u0 + l0) * n1 + l1) * n2;
4563 for (l2 = 0; l2 <= o2; l2++)
4564 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4565 }
4566 }
4567 }
4568 else /* asserts (u0>o0) */
4569 if (u1 < o1)
4570 if (u2 < o2)
4571 {
4572 for (l0 = 0; l0 < 2 * m + 1 - o0; l0++, psij0++)
4573 {
4574 psij1 = psij_const1;
4575 for (l1 = 0; l1 <= 2 * m + 1; l1++, psij1++)
4576 {
4577 psij2 = psij_const2;
4578 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2 + u2;
4579 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4580 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4581 }
4582 }
4583
4584 for (l0 = 0; l0 <= o0; l0++, psij0++)
4585 {
4586 psij1 = psij_const1;
4587 for (l1 = 0; l1 <= 2 * m + 1; l1++, psij1++)
4588 {
4589 psij2 = psij_const2;
4590 gj = g + (l0 * n1 + (u1 + l1)) * n2 + u2;
4591 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4592 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4593 }
4594 }
4595 } else/* asserts (u2>o2) */
4596 {
4597 for (l0 = 0; l0 < 2 * m + 1 - o0; l0++, psij0++)
4598 {
4599 psij1 = psij_const1;
4600 for (l1 = 0; l1 <= 2 * m + 1; l1++, psij1++)
4601 {
4602 psij2 = psij_const2;
4603 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2 + u2;
4604 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4605 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4606 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2;
4607 for (l2 = 0; l2 <= o2; l2++)
4608 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4609 }
4610 }
4611
4612 for (l0 = 0; l0 <= o0; l0++, psij0++)
4613 {
4614 psij1 = psij_const1;
4615 for (l1 = 0; l1 <= 2 * m + 1; l1++, psij1++)
4616 {
4617 psij2 = psij_const2;
4618 gj = g + (l0 * n1 + (u1 + l1)) * n2 + u2;
4619 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4620 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4621 gj = g + (l0 * n1 + (u1 + l1)) * n2;
4622 for (l2 = 0; l2 <= o2; l2++)
4623 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4624 }
4625 }
4626 }
4627 else /* asserts (u1>o1) */
4628 if (u2 < o2)
4629 {
4630 for (l0 = 0; l0 < 2 * m + 1 - o0; l0++, psij0++)
4631 {
4632 psij1 = psij_const1;
4633 for (l1 = 0; l1 < 2 * m + 1 - o1; l1++, psij1++)
4634 {
4635 psij2 = psij_const2;
4636 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2 + u2;
4637 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4638 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4639 }
4640 for (l1 = 0; l1 <= o1; l1++, psij1++)
4641 {
4642 psij2 = psij_const2;
4643 gj = g + ((u0 + l0) * n1 + l1) * n2 + u2;
4644 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4645 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4646 }
4647 }
4648 for (l0 = 0; l0 <= o0; l0++, psij0++)
4649 {
4650 psij1 = psij_const1;
4651 for (l1 = 0; l1 < 2 * m + 1 - o1; l1++, psij1++)
4652 {
4653 psij2 = psij_const2;
4654 gj = g + (l0 * n1 + (u1 + l1)) * n2 + u2;
4655 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4656 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4657 }
4658 for (l1 = 0; l1 <= o1; l1++, psij1++)
4659 {
4660 psij2 = psij_const2;
4661 gj = g + (l0 * n1 + l1) * n2 + u2;
4662 for (l2 = 0; l2 <= 2 * m + 1; l2++)
4663 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4664 }
4665 }
4666 } else/* asserts (u2>o2) */
4667 {
4668 for (l0 = 0; l0 < 2 * m + 1 - o0; l0++, psij0++)
4669 {
4670 psij1 = psij_const1;
4671 for (l1 = 0; l1 < 2 * m + 1 - o1; l1++, psij1++)
4672 {
4673 psij2 = psij_const2;
4674 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2 + u2;
4675 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4676 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4677 gj = g + ((u0 + l0) * n1 + (u1 + l1)) * n2;
4678 for (l2 = 0; l2 <= o2; l2++)
4679 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4680 }
4681 for (l1 = 0; l1 <= o1; l1++, psij1++)
4682 {
4683 psij2 = psij_const2;
4684 gj = g + ((u0 + l0) * n1 + l1) * n2 + u2;
4685 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4686 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4687 gj = g + ((u0 + l0) * n1 + l1) * n2;
4688 for (l2 = 0; l2 <= o2; l2++)
4689 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4690 }
4691 }
4692
4693 for (l0 = 0; l0 <= o0; l0++, psij0++)
4694 {
4695 psij1 = psij_const1;
4696 for (l1 = 0; l1 < 2 * m + 1 - o1; l1++, psij1++)
4697 {
4698 psij2 = psij_const2;
4699 gj = g + (l0 * n1 + (u1 + l1)) * n2 + u2;
4700 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4701 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4702 gj = g + (l0 * n1 + (u1 + l1)) * n2;
4703 for (l2 = 0; l2 <= o2; l2++)
4704 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4705 }
4706 for (l1 = 0; l1 <= o1; l1++, psij1++)
4707 {
4708 psij2 = psij_const2;
4709 gj = g + (l0 * n1 + l1) * n2 + u2;
4710 for (l2 = 0; l2 < 2 * m + 1 - o2; l2++)
4711 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4712 gj = g + (l0 * n1 + l1) * n2;
4713 for (l2 = 0; l2 <= o2; l2++)
4714 (*gj++) += (*psij0) * (*psij1) * (*psij2++) * (*fj);
4715 }
4716 }
4717 }
4718}
4719#endif
4720
4721static void nfft_trafo_3d_B(X(plan) *ths)
4722{
4723 const INT n0 = ths->n[0];
4724 const INT n1 = ths->n[1];
4725 const INT n2 = ths->n[2];
4726 const INT M = ths->M_total;
4727 const INT m = ths->m;
4728
4729 const C* g = (C*) ths->g;
4730
4731 INT k;
4732
4733 if(ths->flags & PRE_FULL_PSI)
4734 {
4735 const INT lprod = (2*m+2) * (2*m+2) * (2*m+2);
4736#ifdef _OPENMP
4737 #pragma omp parallel for default(shared) private(k)
4738#endif
4739 for (k = 0; k < M; k++)
4740 {
4741 INT l;
4742 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
4743 ths->f[j] = K(0.0);
4744 for (l = 0; l < lprod; l++)
4745 ths->f[j] += ths->psi[j*lprod+l] * g[ths->psi_index_g[j*lprod+l]];
4746 }
4747 return;
4748 } /* if(PRE_FULL_PSI) */
4749
4750 if(ths->flags & PRE_PSI)
4751 {
4752#ifdef _OPENMP
4753 #pragma omp parallel for default(shared) private(k)
4754#endif
4755 for (k = 0; k < M; k++)
4756 {
4757 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
4758 nfft_trafo_3d_compute(ths->f+j, g, ths->psi+j*3*(2*m+2), ths->psi+(j*3+1)*(2*m+2), ths->psi+(j*3+2)*(2*m+2), ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, n0, n1, n2, m);
4759 }
4760 return;
4761 } /* if(PRE_PSI) */
4762
4763 if(ths->flags & PRE_FG_PSI)
4764 {
4765 R fg_exp_l[3*(2*m+2+1)];
4766
4767 nfft_3d_init_fg_exp_l(fg_exp_l, m, ths->b[0]);
4768 nfft_3d_init_fg_exp_l(fg_exp_l+2*m+2, m, ths->b[1]);
4769 nfft_3d_init_fg_exp_l(fg_exp_l+2*(2*m+2), m, ths->b[2]);
4770
4771#ifdef _OPENMP
4772 #pragma omp parallel for default(shared) private(k)
4773#endif
4774 for (k = 0; k < M; k++)
4775 {
4776 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
4777 INT l;
4778 R psij_const[3*(2*m+2)];
4779 R fg_psij0 = ths->psi[2*j*3];
4780 R fg_psij1 = ths->psi[2*j*3+1];
4781 R fg_psij2 = K(1.0);
4782
4783 psij_const[0] = fg_psij0;
4784 for(l=1; l<=2*m+1; l++)
4785 {
4786 fg_psij2 *= fg_psij1;
4787 psij_const[l] = fg_psij0*fg_psij2*fg_exp_l[l];
4788 }
4789
4790 fg_psij0 = ths->psi[2*(j*3+1)];
4791 fg_psij1 = ths->psi[2*(j*3+1)+1];
4792 fg_psij2 = K(1.0);
4793 psij_const[2*m+2] = fg_psij0;
4794 for(l=1; l<=2*m+1; l++)
4795 {
4796 fg_psij2 *= fg_psij1;
4797 psij_const[2*m+2+l] = fg_psij0*fg_psij2*fg_exp_l[2*m+2+l];
4798 }
4799
4800 fg_psij0 = ths->psi[2*(j*3+2)];
4801 fg_psij1 = ths->psi[2*(j*3+2)+1];
4802 fg_psij2 = K(1.0);
4803 psij_const[2*(2*m+2)] = fg_psij0;
4804 for(l=1; l<=2*m+1; l++)
4805 {
4806 fg_psij2 *= fg_psij1;
4807 psij_const[2*(2*m+2)+l] = fg_psij0*fg_psij2*fg_exp_l[2*(2*m+2)+l];
4808 }
4809
4810 nfft_trafo_3d_compute(ths->f+j, g, psij_const, psij_const+2*m+2, psij_const+(2*m+2)*2, ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, n0, n1, n2, m);
4811 }
4812
4813 return;
4814 } /* if(PRE_FG_PSI) */
4815
4816 if(ths->flags & FG_PSI)
4817 {
4818 R fg_exp_l[3*(2*m+2+1)];
4819
4820 nfft_3d_init_fg_exp_l(fg_exp_l, m, ths->b[0]);
4821 nfft_3d_init_fg_exp_l(fg_exp_l+2*m+2, m, ths->b[1]);
4822 nfft_3d_init_fg_exp_l(fg_exp_l+2*(2*m+2), m, ths->b[2]);
4823
4824 sort(ths);
4825
4826#ifdef _OPENMP
4827 #pragma omp parallel for default(shared) private(k)
4828#endif
4829 for (k = 0; k < M; k++)
4830 {
4831 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
4832 INT u, o, l;
4833 R psij_const[3*(2*m+2)];
4834 R fg_psij0, fg_psij1, fg_psij2;
4835
4836 uo(ths,j,&u,&o,(INT)0);
4837 fg_psij0 = (PHI(ths->n[0], ths->x[3*j] - ((R)u) / (R)(n0),0));
4838 fg_psij1 = EXP(K(2.0) * ((R)(n0) * (ths->x[3*j]) - (R)(u)) / ths->b[0]);
4839 fg_psij2 = K(1.0);
4840 psij_const[0] = fg_psij0;
4841 for(l=1; l<=2*m+1; l++)
4842 {
4843 fg_psij2 *= fg_psij1;
4844 psij_const[l] = fg_psij0*fg_psij2*fg_exp_l[l];
4845 }
4846
4847 uo(ths,j,&u,&o,(INT)1);
4848 fg_psij0 = (PHI(ths->n[1], ths->x[3*j+1] - ((R)u) / (R)(n1),1));
4849 fg_psij1 = EXP(K(2.0) * ((R)(n1) * (ths->x[3*j+1]) - (R)(u)) / ths->b[1]);
4850 fg_psij2 = K(1.0);
4851 psij_const[2*m+2] = fg_psij0;
4852 for(l=1; l<=2*m+1; l++)
4853 {
4854 fg_psij2 *= fg_psij1;
4855 psij_const[2*m+2+l] = fg_psij0*fg_psij2*fg_exp_l[2*m+2+l];
4856 }
4857
4858 uo(ths,j,&u,&o,(INT)2);
4859 fg_psij0 = (PHI(ths->n[2], ths->x[3*j+2] - ((R)u) / (R)(n2),2));
4860 fg_psij1 = EXP(K(2.0) * ((R)(n2) * (ths->x[3*j+2]) - (R)(u)) / ths->b[2]);
4861 fg_psij2 = K(1.0);
4862 psij_const[2*(2*m+2)] = fg_psij0;
4863 for(l=1; l<=2*m+1; l++)
4864 {
4865 fg_psij2 *= fg_psij1;
4866 psij_const[2*(2*m+2)+l] = fg_psij0*fg_psij2*fg_exp_l[2*(2*m+2)+l];
4867 }
4868
4869 nfft_trafo_3d_compute(ths->f+j, g, psij_const, psij_const+2*m+2, psij_const+(2*m+2)*2, ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, n0, n1, n2, m);
4870 }
4871
4872 return;
4873 } /* if(FG_PSI) */
4874
4875 if(ths->flags & PRE_LIN_PSI)
4876 {
4877 const INT K = ths->K, ip_s = K / (m + 2);
4878
4879 sort(ths);
4880
4881#ifdef _OPENMP
4882 #pragma omp parallel for default(shared) private(k)
4883#endif
4884 for (k = 0; k < M; k++)
4885 {
4886 INT u, o, l;
4887 R ip_y, ip_w;
4888 INT ip_u;
4889 R psij_const[3*(2*m+2)];
4890 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
4891
4892 uo(ths,j,&u,&o,(INT)0);
4893 ip_y = FABS((R)(n0) * ths->x[3*j+0] - (R)(u)) * ((R)ip_s);
4894 ip_u = (INT)(LRINT(FLOOR(ip_y)));
4895 ip_w = ip_y - (R)(ip_u);
4896 for(l=0; l < 2*m+2; l++)
4897 psij_const[l] = ths->psi[ABS(ip_u-l*ip_s)]*(K(1.0)-ip_w) +
4898 ths->psi[ABS(ip_u-l*ip_s+1)]*(ip_w);
4899
4900 uo(ths,j,&u,&o,(INT)1);
4901 ip_y = FABS((R)(n1) * ths->x[3*j+1] - (R)(u)) * ((R)ip_s);
4902 ip_u = (INT)(LRINT(FLOOR(ip_y)));
4903 ip_w = ip_y - (R)(ip_u);
4904 for(l=0; l < 2*m+2; l++)
4905 psij_const[2*m+2+l] = ths->psi[(K+1)+ABS(ip_u-l*ip_s)]*(K(1.0)-ip_w) +
4906 ths->psi[(K+1)+ABS(ip_u-l*ip_s+1)]*(ip_w);
4907
4908 uo(ths,j,&u,&o,(INT)2);
4909 ip_y = FABS((R)(n2) * ths->x[3*j+2] - (R)(u)) * ((R)ip_s);
4910 ip_u = (INT)(LRINT(FLOOR(ip_y)));
4911 ip_w = ip_y - (R)(ip_u);
4912 for(l=0; l < 2*m+2; l++)
4913 psij_const[2*(2*m+2)+l] = ths->psi[2*(K+1)+ABS(ip_u-l*ip_s)]*(K(1.0)-ip_w) +
4914 ths->psi[2*(K+1)+ABS(ip_u-l*ip_s+1)]*(ip_w);
4915
4916 nfft_trafo_3d_compute(ths->f+j, g, psij_const, psij_const+2*m+2, psij_const+(2*m+2)*2, ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, n0, n1, n2, m);
4917 }
4918 return;
4919 } /* if(PRE_LIN_PSI) */
4920
4921 /* no precomputed psi at all */
4922
4923 sort(ths);
4924
4925#ifdef _OPENMP
4926 #pragma omp parallel for default(shared) private(k)
4927#endif
4928 for (k = 0; k < M; k++)
4929 {
4930 R psij_const[3*(2*m+2)];
4931 INT u, o, l;
4932 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
4933
4934 uo(ths,j,&u,&o,(INT)0);
4935 for(l=0;l<=2*m+1;l++)
4936 psij_const[l]=(PHI(ths->n[0], ths->x[3*j] - ((R)((u+l))) / (R)(n0),0));
4937
4938 uo(ths,j,&u,&o,(INT)1);
4939 for(l=0;l<=2*m+1;l++)
4940 psij_const[2*m+2+l]=(PHI(ths->n[1], ths->x[3*j+1] - ((R)((u+l))) / (R)(n1),1));
4941
4942 uo(ths,j,&u,&o,(INT)2);
4943 for(l=0;l<=2*m+1;l++)
4944 psij_const[2*(2*m+2)+l]=(PHI(ths->n[2], ths->x[3*j+2] - ((R)((u+l))) / (R)(n2),2));
4945
4946 nfft_trafo_3d_compute(ths->f+j, g, psij_const, psij_const+2*m+2, psij_const+(2*m+2)*2, ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, n0, n1, n2, m);
4947 }
4948}
4949
4950#define MACRO_adjoint_3d_B_OMP_BLOCKWISE_COMPUTE_PRE_PSI \
4951 nfft_adjoint_3d_compute_omp_blockwise(ths->f[j], g, \
4952 ths->psi+j*3*(2*m+2), \
4953 ths->psi+(j*3+1)*(2*m+2), \
4954 ths->psi+(j*3+2)*(2*m+2), \
4955 ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, \
4956 n0, n1, n2, m, my_u0, my_o0);
4957
4958#define MACRO_adjoint_3d_B_OMP_BLOCKWISE_COMPUTE_PRE_FG_PSI \
4959{ \
4960 INT l; \
4961 R psij_const[3*(2*m+2)]; \
4962 R fg_psij0 = ths->psi[2*j*3]; \
4963 R fg_psij1 = ths->psi[2*j*3+1]; \
4964 R fg_psij2 = K(1.0); \
4965 \
4966 psij_const[0] = fg_psij0; \
4967 for(l=1; l<=2*m+1; l++) \
4968 { \
4969 fg_psij2 *= fg_psij1; \
4970 psij_const[l] = fg_psij0*fg_psij2*fg_exp_l[l]; \
4971 } \
4972 \
4973 fg_psij0 = ths->psi[2*(j*3+1)]; \
4974 fg_psij1 = ths->psi[2*(j*3+1)+1]; \
4975 fg_psij2 = K(1.0); \
4976 psij_const[2*m+2] = fg_psij0; \
4977 for(l=1; l<=2*m+1; l++) \
4978 { \
4979 fg_psij2 *= fg_psij1; \
4980 psij_const[2*m+2+l] = fg_psij0*fg_psij2*fg_exp_l[2*m+2+l]; \
4981 } \
4982 \
4983 fg_psij0 = ths->psi[2*(j*3+2)]; \
4984 fg_psij1 = ths->psi[2*(j*3+2)+1]; \
4985 fg_psij2 = K(1.0); \
4986 psij_const[2*(2*m+2)] = fg_psij0; \
4987 for(l=1; l<=2*m+1; l++) \
4988 { \
4989 fg_psij2 *= fg_psij1; \
4990 psij_const[2*(2*m+2)+l] = fg_psij0*fg_psij2*fg_exp_l[2*(2*m+2)+l]; \
4991 } \
4992 \
4993 nfft_adjoint_3d_compute_omp_blockwise(ths->f[j], g, \
4994 psij_const, psij_const+2*m+2, psij_const+(2*m+2)*2, \
4995 ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, \
4996 n0, n1, n2, m, my_u0, my_o0); \
4997}
4998
4999#define MACRO_adjoint_3d_B_OMP_BLOCKWISE_COMPUTE_FG_PSI \
5000{ \
5001 INT u, o, l; \
5002 R psij_const[3*(2*m+2)]; \
5003 R fg_psij0, fg_psij1, fg_psij2; \
5004 \
5005 uo(ths,j,&u,&o,(INT)0); \
5006 fg_psij0 = (PHI(ths->n[0],ths->x[3*j]-((R)u)/((R)n0),0)); \
5007 fg_psij1 = EXP(K(2.0)*(((R)n0)*(ths->x[3*j]) - (R)u)/ths->b[0]); \
5008 fg_psij2 = K(1.0); \
5009 psij_const[0] = fg_psij0; \
5010 for(l=1; l<=2*m+1; l++) \
5011 { \
5012 fg_psij2 *= fg_psij1; \
5013 psij_const[l] = fg_psij0*fg_psij2*fg_exp_l[l]; \
5014 } \
5015 \
5016 uo(ths,j,&u,&o,(INT)1); \
5017 fg_psij0 = (PHI(ths->n[1],ths->x[3*j+1]-((R)u)/((R)n1),1)); \
5018 fg_psij1 = EXP(K(2.0)*(((R)n1)*(ths->x[3*j+1]) - (R)u)/ths->b[1]); \
5019 fg_psij2 = K(1.0); \
5020 psij_const[2*m+2] = fg_psij0; \
5021 for(l=1; l<=2*m+1; l++) \
5022 { \
5023 fg_psij2 *= fg_psij1; \
5024 psij_const[2*m+2+l] = fg_psij0*fg_psij2*fg_exp_l[2*m+2+l]; \
5025 } \
5026 \
5027 uo(ths,j,&u,&o,(INT)2); \
5028 fg_psij0 = (PHI(ths->n[2],ths->x[3*j+2]-((R)u)/((R)n2),2)); \
5029 fg_psij1 = EXP(K(2.0)*(((R)n2)*(ths->x[3*j+2]) - (R)u)/ths->b[2]); \
5030 fg_psij2 = K(1.0); \
5031 psij_const[2*(2*m+2)] = fg_psij0; \
5032 for(l=1; l<=2*m+1; l++) \
5033 { \
5034 fg_psij2 *= fg_psij1; \
5035 psij_const[2*(2*m+2)+l] = fg_psij0*fg_psij2*fg_exp_l[2*(2*m+2)+l]; \
5036 } \
5037 \
5038 nfft_adjoint_3d_compute_omp_blockwise(ths->f[j], g, \
5039 psij_const, psij_const+2*m+2, psij_const+(2*m+2)*2, \
5040 ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, \
5041 n0, n1, n2, m, my_u0, my_o0); \
5042}
5043
5044#define MACRO_adjoint_3d_B_OMP_BLOCKWISE_COMPUTE_PRE_LIN_PSI \
5045{ \
5046 INT u, o, l; \
5047 R psij_const[3*(2*m+2)]; \
5048 INT ip_u; \
5049 R ip_y, ip_w; \
5050 \
5051 uo(ths,j,&u,&o,(INT)0); \
5052 ip_y = FABS(((R)n0)*ths->x[3*j+0] - (R)u)*((R)ip_s); \
5053 ip_u = LRINT(FLOOR(ip_y)); \
5054 ip_w = ip_y-ip_u; \
5055 for(l=0; l < 2*m+2; l++) \
5056 psij_const[l] = ths->psi[ABS(ip_u-l*ip_s)]*(K(1.0)-ip_w) + \
5057 ths->psi[ABS(ip_u-l*ip_s+1)]*(ip_w); \
5058 \
5059 uo(ths,j,&u,&o,(INT)1); \
5060 ip_y = FABS(((R)n1)*ths->x[3*j+1] - (R)u)*((R)ip_s); \
5061 ip_u = LRINT(FLOOR(ip_y)); \
5062 ip_w = ip_y-ip_u; \
5063 for(l=0; l < 2*m+2; l++) \
5064 psij_const[2*m+2+l] = ths->psi[(K+1)+ABS(ip_u-l*ip_s)]*(K(1.0)-ip_w) + \
5065 ths->psi[(K+1)+ABS(ip_u-l*ip_s+1)]*(ip_w); \
5066 \
5067 uo(ths,j,&u,&o,(INT)2); \
5068 ip_y = FABS(((R)n2)*ths->x[3*j+2] - (R)u)*((R)ip_s); \
5069 ip_u = LRINT(FLOOR(ip_y)); \
5070 ip_w = ip_y-ip_u; \
5071 for(l=0; l < 2*m+2; l++) \
5072 psij_const[2*(2*m+2)+l] = ths->psi[2*(K+1)+ABS(ip_u-l*ip_s)]*(K(1.0)-ip_w) + \
5073 ths->psi[2*(K+1)+ABS(ip_u-l*ip_s+1)]*(ip_w); \
5074 \
5075 nfft_adjoint_3d_compute_omp_blockwise(ths->f[j], g, \
5076 psij_const, psij_const+2*m+2, psij_const+(2*m+2)*2, \
5077 ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, \
5078 n0, n1, n2, m, my_u0, my_o0); \
5079}
5080
5081#define MACRO_adjoint_3d_B_OMP_BLOCKWISE_COMPUTE_NO_PSI \
5082{ \
5083 INT u, o, l; \
5084 R psij_const[3*(2*m+2)]; \
5085 \
5086 uo(ths,j,&u,&o,(INT)0); \
5087 for(l=0;l<=2*m+1;l++) \
5088 psij_const[l]=(PHI(ths->n[0],ths->x[3*j]-((R)((u+l)))/((R) n0),0)); \
5089 \
5090 uo(ths,j,&u,&o,(INT)1); \
5091 for(l=0;l<=2*m+1;l++) \
5092 psij_const[2*m+2+l]=(PHI(ths->n[1],ths->x[3*j+1]-((R)((u+l)))/((R) n1),1)); \
5093 \
5094 uo(ths,j,&u,&o,(INT)2); \
5095 for(l=0;l<=2*m+1;l++) \
5096 psij_const[2*(2*m+2)+l]=(PHI(ths->n[2],ths->x[3*j+2]-((R)((u+l)))/((R) n2),2)); \
5097 \
5098 nfft_adjoint_3d_compute_omp_blockwise(ths->f[j], g, \
5099 psij_const, psij_const+2*m+2, psij_const+(2*m+2)*2, \
5100 ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, \
5101 n0, n1, n2, m, my_u0, my_o0); \
5102}
5103
5104#define MACRO_adjoint_3d_B_OMP_BLOCKWISE(whichone) \
5105{ \
5106 if (ths->flags & NFFT_OMP_BLOCKWISE_ADJOINT) \
5107 { \
5108 _Pragma("omp parallel private(k)") \
5109 { \
5110 INT my_u0, my_o0, min_u_a, max_u_a, min_u_b, max_u_b; \
5111 INT *ar_x = ths->index_x; \
5112 \
5113 nfft_adjoint_B_omp_blockwise_init(&my_u0, &my_o0, &min_u_a, &max_u_a, \
5114 &min_u_b, &max_u_b, 3, ths->n, m); \
5115 \
5116 if (min_u_a != -1) \
5117 { \
5118 k = index_x_binary_search(ar_x, M, min_u_a); \
5119 \
5120 MACRO_adjoint_nd_B_OMP_BLOCKWISE_ASSERT_A \
5121 \
5122 while (k < M) \
5123 { \
5124 INT u_prod = ar_x[2*k]; \
5125 INT j = ar_x[2*k+1]; \
5126 \
5127 if (u_prod < min_u_a || u_prod > max_u_a) \
5128 break; \
5129 \
5130 MACRO_adjoint_3d_B_OMP_BLOCKWISE_COMPUTE_ ##whichone \
5131 \
5132 k++; \
5133 } \
5134 } \
5135 \
5136 if (min_u_b != -1) \
5137 { \
5138 INT k = index_x_binary_search(ar_x, M, min_u_b); \
5139 \
5140 MACRO_adjoint_nd_B_OMP_BLOCKWISE_ASSERT_B \
5141 \
5142 while (k < M) \
5143 { \
5144 INT u_prod = ar_x[2*k]; \
5145 INT j = ar_x[2*k+1]; \
5146 \
5147 if (u_prod < min_u_b || u_prod > max_u_b) \
5148 break; \
5149 \
5150 MACRO_adjoint_3d_B_OMP_BLOCKWISE_COMPUTE_ ##whichone \
5151 \
5152 k++; \
5153 } \
5154 } \
5155 } /* omp parallel */ \
5156 return; \
5157 } /* if(NFFT_OMP_BLOCKWISE_ADJOINT) */ \
5158}
5159
5160static void nfft_adjoint_3d_B(X(plan) *ths)
5161{
5162 INT k;
5163 const INT n0 = ths->n[0];
5164 const INT n1 = ths->n[1];
5165 const INT n2 = ths->n[2];
5166 const INT M = ths->M_total;
5167 const INT m = ths->m;
5168
5169 C* g = (C*) ths->g;
5170
5171 memset(g, 0, (size_t)(ths->n_total) * sizeof(C));
5172
5173 if(ths->flags & PRE_FULL_PSI)
5174 {
5175 nfft_adjoint_B_compute_full_psi(g, ths->psi_index_g, ths->psi, ths->f, M,
5176 (INT)3, ths->n, m, ths->flags, ths->index_x);
5177 return;
5178 } /* if(PRE_FULL_PSI) */
5179
5180 if(ths->flags & PRE_PSI)
5181 {
5182#ifdef _OPENMP
5183 MACRO_adjoint_3d_B_OMP_BLOCKWISE(PRE_PSI)
5184#endif
5185
5186#ifdef _OPENMP
5187 #pragma omp parallel for default(shared) private(k)
5188#endif
5189 for (k = 0; k < M; k++)
5190 {
5191 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
5192#ifdef _OPENMP
5193 nfft_adjoint_3d_compute_omp_atomic(ths->f[j], g, ths->psi+j*3*(2*m+2), ths->psi+(j*3+1)*(2*m+2), ths->psi+(j*3+2)*(2*m+2), ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, n0, n1, n2, m);
5194#else
5195 nfft_adjoint_3d_compute_serial(ths->f+j, g, ths->psi+j*3*(2*m+2), ths->psi+(j*3+1)*(2*m+2), ths->psi+(j*3+2)*(2*m+2), ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, n0, n1, n2, m);
5196#endif
5197 }
5198 return;
5199 } /* if(PRE_PSI) */
5200
5201 if(ths->flags & PRE_FG_PSI)
5202 {
5203 R fg_exp_l[3*(2*m+2+1)];
5204
5205 nfft_3d_init_fg_exp_l(fg_exp_l, m, ths->b[0]);
5206 nfft_3d_init_fg_exp_l(fg_exp_l+2*m+2, m, ths->b[1]);
5207 nfft_3d_init_fg_exp_l(fg_exp_l+2*(2*m+2), m, ths->b[2]);
5208
5209#ifdef _OPENMP
5210 MACRO_adjoint_3d_B_OMP_BLOCKWISE(PRE_FG_PSI)
5211#endif
5212
5213#ifdef _OPENMP
5214 #pragma omp parallel for default(shared) private(k)
5215#endif
5216 for (k = 0; k < M; k++)
5217 {
5218 R psij_const[3*(2*m+2)];
5219 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
5220 INT l;
5221 R fg_psij0 = ths->psi[2*j*3];
5222 R fg_psij1 = ths->psi[2*j*3+1];
5223 R fg_psij2 = K(1.0);
5224
5225 psij_const[0] = fg_psij0;
5226 for(l=1; l<=2*m+1; l++)
5227 {
5228 fg_psij2 *= fg_psij1;
5229 psij_const[l] = fg_psij0*fg_psij2*fg_exp_l[l];
5230 }
5231
5232 fg_psij0 = ths->psi[2*(j*3+1)];
5233 fg_psij1 = ths->psi[2*(j*3+1)+1];
5234 fg_psij2 = K(1.0);
5235 psij_const[2*m+2] = fg_psij0;
5236 for(l=1; l<=2*m+1; l++)
5237 {
5238 fg_psij2 *= fg_psij1;
5239 psij_const[2*m+2+l] = fg_psij0*fg_psij2*fg_exp_l[2*m+2+l];
5240 }
5241
5242 fg_psij0 = ths->psi[2*(j*3+2)];
5243 fg_psij1 = ths->psi[2*(j*3+2)+1];
5244 fg_psij2 = K(1.0);
5245 psij_const[2*(2*m+2)] = fg_psij0;
5246 for(l=1; l<=2*m+1; l++)
5247 {
5248 fg_psij2 *= fg_psij1;
5249 psij_const[2*(2*m+2)+l] = fg_psij0*fg_psij2*fg_exp_l[2*(2*m+2)+l];
5250 }
5251
5252#ifdef _OPENMP
5253 nfft_adjoint_3d_compute_omp_atomic(ths->f[j], g, psij_const, psij_const+2*m+2, psij_const+(2*m+2)*2, ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, n0, n1, n2, m);
5254#else
5255 nfft_adjoint_3d_compute_serial(ths->f+j, g, psij_const, psij_const+2*m+2, psij_const+(2*m+2)*2, ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, n0, n1, n2, m);
5256#endif
5257 }
5258
5259 return;
5260 } /* if(PRE_FG_PSI) */
5261
5262 if(ths->flags & FG_PSI)
5263 {
5264 R fg_exp_l[3*(2*m+2+1)];
5265
5266 nfft_3d_init_fg_exp_l(fg_exp_l, m, ths->b[0]);
5267 nfft_3d_init_fg_exp_l(fg_exp_l+2*m+2, m, ths->b[1]);
5268 nfft_3d_init_fg_exp_l(fg_exp_l+2*(2*m+2), m, ths->b[2]);
5269
5270 sort(ths);
5271
5272#ifdef _OPENMP
5273 MACRO_adjoint_3d_B_OMP_BLOCKWISE(FG_PSI)
5274#endif
5275
5276#ifdef _OPENMP
5277 #pragma omp parallel for default(shared) private(k)
5278#endif
5279 for (k = 0; k < M; k++)
5280 {
5281 INT u,o,l;
5282 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
5283 R psij_const[3*(2*m+2)];
5284 R fg_psij0, fg_psij1, fg_psij2;
5285
5286 uo(ths,j,&u,&o,(INT)0);
5287 fg_psij0 = (PHI(ths->n[0], ths->x[3*j] - ((R)u) / (R)(n0),0));
5288 fg_psij1 = EXP(K(2.0) * ((R)(n0) * (ths->x[3*j]) - (R)(u))/ths->b[0]);
5289 fg_psij2 = K(1.0);
5290 psij_const[0] = fg_psij0;
5291 for(l=1; l<=2*m+1; l++)
5292 {
5293 fg_psij2 *= fg_psij1;
5294 psij_const[l] = fg_psij0*fg_psij2*fg_exp_l[l];
5295 }
5296
5297 uo(ths,j,&u,&o,(INT)1);
5298 fg_psij0 = (PHI(ths->n[1], ths->x[3*j+1] - ((R)u) / (R)(n1),1));
5299 fg_psij1 = EXP(K(2.0) * ((R)(n1) * (ths->x[3*j+1]) - (R)(u))/ths->b[1]);
5300 fg_psij2 = K(1.0);
5301 psij_const[2*m+2] = fg_psij0;
5302 for(l=1; l<=2*m+1; l++)
5303 {
5304 fg_psij2 *= fg_psij1;
5305 psij_const[2*m+2+l] = fg_psij0*fg_psij2*fg_exp_l[2*m+2+l];
5306 }
5307
5308 uo(ths,j,&u,&o,(INT)2);
5309 fg_psij0 = (PHI(ths->n[2], ths->x[3*j+2] - ((R)u) / (R)(n2),2));
5310 fg_psij1 = EXP(K(2.0) * ((R)(n2) * (ths->x[3*j+2]) - (R)(u))/ths->b[2]);
5311 fg_psij2 = K(1.0);
5312 psij_const[2*(2*m+2)] = fg_psij0;
5313 for(l=1; l<=2*m+1; l++)
5314 {
5315 fg_psij2 *= fg_psij1;
5316 psij_const[2*(2*m+2)+l] = fg_psij0*fg_psij2*fg_exp_l[2*(2*m+2)+l];
5317 }
5318
5319#ifdef _OPENMP
5320 nfft_adjoint_3d_compute_omp_atomic(ths->f[j], g, psij_const, psij_const+2*m+2, psij_const+(2*m+2)*2, ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, n0, n1, n2, m);
5321#else
5322 nfft_adjoint_3d_compute_serial(ths->f+j, g, psij_const, psij_const+2*m+2, psij_const+(2*m+2)*2, ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, n0, n1, n2, m);
5323#endif
5324 }
5325
5326 return;
5327 } /* if(FG_PSI) */
5328
5329 if(ths->flags & PRE_LIN_PSI)
5330 {
5331 const INT K = ths->K;
5332 const INT ip_s = K / (m + 2);
5333
5334 sort(ths);
5335
5336#ifdef _OPENMP
5337 MACRO_adjoint_3d_B_OMP_BLOCKWISE(PRE_LIN_PSI)
5338#endif
5339
5340#ifdef _OPENMP
5341 #pragma omp parallel for default(shared) private(k)
5342#endif
5343 for (k = 0; k < M; k++)
5344 {
5345 INT u,o,l;
5346 INT ip_u;
5347 R ip_y, ip_w;
5348 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
5349 R psij_const[3*(2*m+2)];
5350
5351 uo(ths,j,&u,&o,(INT)0);
5352 ip_y = FABS((R)(n0) * ths->x[3*j+0] - (R)(u)) * ((R)ip_s);
5353 ip_u = (INT)(LRINT(FLOOR(ip_y)));
5354 ip_w = ip_y - (R)(ip_u);
5355 for(l=0; l < 2*m+2; l++)
5356 psij_const[l] = ths->psi[ABS(ip_u-l*ip_s)]*(K(1.0)-ip_w) +
5357 ths->psi[ABS(ip_u-l*ip_s+1)]*(ip_w);
5358
5359 uo(ths,j,&u,&o,(INT)1);
5360 ip_y = FABS((R)(n1) * ths->x[3*j+1] - (R)(u)) * ((R)ip_s);
5361 ip_u = (INT)(LRINT(FLOOR(ip_y)));
5362 ip_w = ip_y - (R)(ip_u);
5363 for(l=0; l < 2*m+2; l++)
5364 psij_const[2*m+2+l] = ths->psi[(K+1)+ABS(ip_u-l*ip_s)]*(K(1.0)-ip_w) +
5365 ths->psi[(K+1)+ABS(ip_u-l*ip_s+1)]*(ip_w);
5366
5367 uo(ths,j,&u,&o,(INT)2);
5368 ip_y = FABS((R)(n2) * ths->x[3*j+2] - (R)(u))*((R)ip_s);
5369 ip_u = (INT)(LRINT(FLOOR(ip_y)));
5370 ip_w = ip_y - (R)(ip_u);
5371 for(l=0; l < 2*m+2; l++)
5372 psij_const[2*(2*m+2)+l] = ths->psi[2*(K+1)+ABS(ip_u-l*ip_s)]*(K(1.0)-ip_w) +
5373 ths->psi[2*(K+1)+ABS(ip_u-l*ip_s+1)]*(ip_w);
5374
5375#ifdef _OPENMP
5376 nfft_adjoint_3d_compute_omp_atomic(ths->f[j], g, psij_const, psij_const+2*m+2, psij_const+(2*m+2)*2, ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, n0, n1, n2, m);
5377#else
5378 nfft_adjoint_3d_compute_serial(ths->f+j, g, psij_const, psij_const+2*m+2, psij_const+(2*m+2)*2, ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, n0, n1, n2, m);
5379#endif
5380 }
5381 return;
5382 } /* if(PRE_LIN_PSI) */
5383
5384 /* no precomputed psi at all */
5385 sort(ths);
5386
5387#ifdef _OPENMP
5388 MACRO_adjoint_3d_B_OMP_BLOCKWISE(NO_PSI)
5389#endif
5390
5391#ifdef _OPENMP
5392 #pragma omp parallel for default(shared) private(k)
5393#endif
5394 for (k = 0; k < M; k++)
5395 {
5396 INT u,o,l;
5397 R psij_const[3*(2*m+2)];
5398 INT j = (ths->flags & NFFT_SORT_NODES) ? ths->index_x[2*k+1] : k;
5399
5400 uo(ths,j,&u,&o,(INT)0);
5401 for(l=0;l<=2*m+1;l++)
5402 psij_const[l]=(PHI(ths->n[0], ths->x[3*j] - ((R)((u+l))) / (R)(n0),0));
5403
5404 uo(ths,j,&u,&o,(INT)1);
5405 for(l=0;l<=2*m+1;l++)
5406 psij_const[2*m+2+l]=(PHI(ths->n[1], ths->x[3*j+1] - ((R)((u+l))) / (R)(n1),1));
5407
5408 uo(ths,j,&u,&o,(INT)2);
5409 for(l=0;l<=2*m+1;l++)
5410 psij_const[2*(2*m+2)+l]=(PHI(ths->n[2], ths->x[3*j+2] - ((R)((u+l))) / (R)(n2),2));
5411
5412#ifdef _OPENMP
5413 nfft_adjoint_3d_compute_omp_atomic(ths->f[j], g, psij_const, psij_const+2*m+2, psij_const+(2*m+2)*2, ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, n0, n1, n2, m);
5414#else
5415 nfft_adjoint_3d_compute_serial(ths->f+j, g, psij_const, psij_const+2*m+2, psij_const+(2*m+2)*2, ths->x+3*j, ths->x+3*j+1, ths->x+3*j+2, n0, n1, n2, m);
5416#endif
5417 }
5418}
5419
5420
5421void X(trafo_3d)(X(plan) *ths)
5422{
5423 if((ths->N[0] <= ths->m) || (ths->N[1] <= ths->m) || (ths->N[2] <= ths->m) || (ths->n[0] <= 2*ths->m+2) || (ths->n[1] <= 2*ths->m+2) || (ths->n[2] <= 2*ths->m+2))
5424 {
5425 X(trafo_direct)(ths);
5426 return;
5427 }
5428
5429 INT k0,k1,k2,n0,n1,n2,N0,N1,N2;
5430 C *g_hat,*f_hat;
5431 R *c_phi_inv01, *c_phi_inv02, *c_phi_inv11, *c_phi_inv12, *c_phi_inv21, *c_phi_inv22;
5432 R ck01, ck02, ck11, ck12, ck21, ck22;
5433 C *g_hat111,*f_hat111,*g_hat211,*f_hat211,*g_hat121,*f_hat121,*g_hat221,*f_hat221;
5434 C *g_hat112,*f_hat112,*g_hat212,*f_hat212,*g_hat122,*f_hat122,*g_hat222,*f_hat222;
5435
5436 ths->g_hat=ths->g1;
5437 ths->g=ths->g2;
5438
5439 N0=ths->N[0];
5440 N1=ths->N[1];
5441 N2=ths->N[2];
5442 n0=ths->n[0];
5443 n1=ths->n[1];
5444 n2=ths->n[2];
5445
5446 f_hat=(C*)ths->f_hat;
5447 g_hat=(C*)ths->g_hat;
5448
5449 TIC(0)
5450#ifdef _OPENMP
5451 #pragma omp parallel for default(shared) private(k0)
5452 for (k0 = 0; k0 < ths->n_total; k0++)
5453 ths->g_hat[k0] = 0.0;
5454#else
5455 memset(ths->g_hat, 0, (size_t)(ths->n_total) * sizeof(C));
5456#endif
5457
5458 if(ths->flags & PRE_PHI_HUT)
5459 {
5460 c_phi_inv01=ths->c_phi_inv[0];
5461 c_phi_inv02=&ths->c_phi_inv[0][N0/2];
5462
5463#ifdef _OPENMP
5464 #pragma omp parallel for default(shared) private(k0,k1,k2,ck01,ck02,c_phi_inv11,c_phi_inv12,ck11,ck12,c_phi_inv21,c_phi_inv22,g_hat111,f_hat111,g_hat211,f_hat211,g_hat121,f_hat121,g_hat221,f_hat221,g_hat112,f_hat112,g_hat212,f_hat212,g_hat122,f_hat122,g_hat222,f_hat222,ck21,ck22)
5465#endif
5466 for(k0=0;k0<N0/2;k0++)
5467 {
5468 ck01=c_phi_inv01[k0];
5469 ck02=c_phi_inv02[k0];
5470 c_phi_inv11=ths->c_phi_inv[1];
5471 c_phi_inv12=&ths->c_phi_inv[1][N1/2];
5472
5473 for(k1=0;k1<N1/2;k1++)
5474 {
5475 ck11=c_phi_inv11[k1];
5476 ck12=c_phi_inv12[k1];
5477 c_phi_inv21=ths->c_phi_inv[2];
5478 c_phi_inv22=&ths->c_phi_inv[2][N2/2];
5479
5480 g_hat111=g_hat + ((n0-(N0/2)+k0)*n1+n1-(N1/2)+k1)*n2+n2-(N2/2);
5481 f_hat111=f_hat + (k0*N1+k1)*N2;
5482 g_hat211=g_hat + (k0*n1+n1-(N1/2)+k1)*n2+n2-(N2/2);
5483 f_hat211=f_hat + (((N0/2)+k0)*N1+k1)*N2;
5484 g_hat121=g_hat + ((n0-(N0/2)+k0)*n1+k1)*n2+n2-(N2/2);
5485 f_hat121=f_hat + (k0*N1+(N1/2)+k1)*N2;
5486 g_hat221=g_hat + (k0*n1+k1)*n2+n2-(N2/2);
5487 f_hat221=f_hat + (((N0/2)+k0)*N1+(N1/2)+k1)*N2;
5488
5489 g_hat112=g_hat + ((n0-(N0/2)+k0)*n1+n1-(N1/2)+k1)*n2;
5490 f_hat112=f_hat + (k0*N1+k1)*N2+(N2/2);
5491 g_hat212=g_hat + (k0*n1+n1-(N1/2)+k1)*n2;
5492 f_hat212=f_hat + (((N0/2)+k0)*N1+k1)*N2+(N2/2);
5493 g_hat122=g_hat + ((n0-(N0/2)+k0)*n1+k1)*n2;
5494 f_hat122=f_hat + (k0*N1+N1/2+k1)*N2+(N2/2);
5495 g_hat222=g_hat + (k0*n1+k1)*n2;
5496 f_hat222=f_hat + (((N0/2)+k0)*N1+(N1/2)+k1)*N2+(N2/2);
5497
5498 for(k2=0;k2<N2/2;k2++)
5499 {
5500 ck21=c_phi_inv21[k2];
5501 ck22=c_phi_inv22[k2];
5502
5503 g_hat111[k2] = f_hat111[k2] * ck01 * ck11 * ck21;
5504 g_hat211[k2] = f_hat211[k2] * ck02 * ck11 * ck21;
5505 g_hat121[k2] = f_hat121[k2] * ck01 * ck12 * ck21;
5506 g_hat221[k2] = f_hat221[k2] * ck02 * ck12 * ck21;
5507
5508 g_hat112[k2] = f_hat112[k2] * ck01 * ck11 * ck22;
5509 g_hat212[k2] = f_hat212[k2] * ck02 * ck11 * ck22;
5510 g_hat122[k2] = f_hat122[k2] * ck01 * ck12 * ck22;
5511 g_hat222[k2] = f_hat222[k2] * ck02 * ck12 * ck22;
5512 }
5513 }
5514 }
5515 }
5516 else
5517#ifdef _OPENMP
5518 #pragma omp parallel for default(shared) private(k0,k1,k2,ck01,ck02,ck11,ck12,ck21,ck22)
5519#endif
5520 for(k0=0;k0<N0/2;k0++)
5521 {
5522 ck01=K(1.0)/(PHI_HUT(ths->n[0],k0-N0/2,0));
5523 ck02=K(1.0)/(PHI_HUT(ths->n[0],k0,0));
5524 for(k1=0;k1<N1/2;k1++)
5525 {
5526 ck11=K(1.0)/(PHI_HUT(ths->n[1],k1-N1/2,1));
5527 ck12=K(1.0)/(PHI_HUT(ths->n[1],k1,1));
5528
5529 for(k2=0;k2<N2/2;k2++)
5530 {
5531 ck21=K(1.0)/(PHI_HUT(ths->n[2],k2-N2/2,2));
5532 ck22=K(1.0)/(PHI_HUT(ths->n[2],k2,2));
5533
5534 g_hat[((n0-N0/2+k0)*n1+n1-N1/2+k1)*n2+n2-N2/2+k2] = f_hat[(k0*N1+k1)*N2+k2] * ck01 * ck11 * ck21;
5535 g_hat[(k0*n1+n1-N1/2+k1)*n2+n2-N2/2+k2] = f_hat[((N0/2+k0)*N1+k1)*N2+k2] * ck02 * ck11 * ck21;
5536 g_hat[((n0-N0/2+k0)*n1+k1)*n2+n2-N2/2+k2] = f_hat[(k0*N1+N1/2+k1)*N2+k2] * ck01 * ck12 * ck21;
5537 g_hat[(k0*n1+k1)*n2+n2-N2/2+k2] = f_hat[((N0/2+k0)*N1+N1/2+k1)*N2+k2] * ck02 * ck12 * ck21;
5538
5539 g_hat[((n0-N0/2+k0)*n1+n1-N1/2+k1)*n2+k2] = f_hat[(k0*N1+k1)*N2+N2/2+k2] * ck01 * ck11 * ck22;
5540 g_hat[(k0*n1+n1-N1/2+k1)*n2+k2] = f_hat[((N0/2+k0)*N1+k1)*N2+N2/2+k2] * ck02 * ck11 * ck22;
5541 g_hat[((n0-N0/2+k0)*n1+k1)*n2+k2] = f_hat[(k0*N1+N1/2+k1)*N2+N2/2+k2] * ck01 * ck12 * ck22;
5542 g_hat[(k0*n1+k1)*n2+k2] = f_hat[((N0/2+k0)*N1+N1/2+k1)*N2+N2/2+k2] * ck02 * ck12 * ck22;
5543 }
5544 }
5545 }
5546
5547 TOC(0)
5548
5549 TIC_FFTW(1)
5550 FFTW(execute)(ths->my_fftw_plan1);
5551 TOC_FFTW(1);
5552
5553 TIC(2);
5554 nfft_trafo_3d_B(ths);
5555 TOC(2);
5556}
5557
5558void X(adjoint_3d)(X(plan) *ths)
5559{
5560 if((ths->N[0] <= ths->m) || (ths->N[1] <= ths->m) || (ths->N[2] <= ths->m) || (ths->n[0] <= 2*ths->m+2) || (ths->n[1] <= 2*ths->m+2) || (ths->n[2] <= 2*ths->m+2))
5561 {
5562 X(adjoint_direct)(ths);
5563 return;
5564 }
5565
5566 INT k0,k1,k2,n0,n1,n2,N0,N1,N2;
5567 C *g_hat,*f_hat;
5568 R *c_phi_inv01, *c_phi_inv02, *c_phi_inv11, *c_phi_inv12, *c_phi_inv21, *c_phi_inv22;
5569 R ck01, ck02, ck11, ck12, ck21, ck22;
5570 C *g_hat111,*f_hat111,*g_hat211,*f_hat211,*g_hat121,*f_hat121,*g_hat221,*f_hat221;
5571 C *g_hat112,*f_hat112,*g_hat212,*f_hat212,*g_hat122,*f_hat122,*g_hat222,*f_hat222;
5572
5573 ths->g_hat=ths->g1;
5574 ths->g=ths->g2;
5575
5576 N0=ths->N[0];
5577 N1=ths->N[1];
5578 N2=ths->N[2];
5579 n0=ths->n[0];
5580 n1=ths->n[1];
5581 n2=ths->n[2];
5582
5583 f_hat=(C*)ths->f_hat;
5584 g_hat=(C*)ths->g_hat;
5585
5586 TIC(2);
5587 nfft_adjoint_3d_B(ths);
5588 TOC(2);
5589
5590 TIC_FFTW(1)
5591 FFTW(execute)(ths->my_fftw_plan2);
5592 TOC_FFTW(1);
5593
5594 TIC(0)
5595 if(ths->flags & PRE_PHI_HUT)
5596 {
5597 c_phi_inv01=ths->c_phi_inv[0];
5598 c_phi_inv02=&ths->c_phi_inv[0][N0/2];
5599
5600#ifdef _OPENMP
5601 #pragma omp parallel for default(shared) private(k0,k1,k2,ck01,ck02,c_phi_inv11,c_phi_inv12,ck11,ck12,c_phi_inv21,c_phi_inv22,g_hat111,f_hat111,g_hat211,f_hat211,g_hat121,f_hat121,g_hat221,f_hat221,g_hat112,f_hat112,g_hat212,f_hat212,g_hat122,f_hat122,g_hat222,f_hat222,ck21,ck22)
5602#endif
5603 for(k0=0;k0<N0/2;k0++)
5604 {
5605 ck01=c_phi_inv01[k0];
5606 ck02=c_phi_inv02[k0];
5607 c_phi_inv11=ths->c_phi_inv[1];
5608 c_phi_inv12=&ths->c_phi_inv[1][N1/2];
5609
5610 for(k1=0;k1<N1/2;k1++)
5611 {
5612 ck11=c_phi_inv11[k1];
5613 ck12=c_phi_inv12[k1];
5614 c_phi_inv21=ths->c_phi_inv[2];
5615 c_phi_inv22=&ths->c_phi_inv[2][N2/2];
5616
5617 g_hat111=g_hat + ((n0-(N0/2)+k0)*n1+n1-(N1/2)+k1)*n2+n2-(N2/2);
5618 f_hat111=f_hat + (k0*N1+k1)*N2;
5619 g_hat211=g_hat + (k0*n1+n1-(N1/2)+k1)*n2+n2-(N2/2);
5620 f_hat211=f_hat + (((N0/2)+k0)*N1+k1)*N2;
5621 g_hat121=g_hat + ((n0-(N0/2)+k0)*n1+k1)*n2+n2-(N2/2);
5622 f_hat121=f_hat + (k0*N1+(N1/2)+k1)*N2;
5623 g_hat221=g_hat + (k0*n1+k1)*n2+n2-(N2/2);
5624 f_hat221=f_hat + (((N0/2)+k0)*N1+(N1/2)+k1)*N2;
5625
5626 g_hat112=g_hat + ((n0-(N0/2)+k0)*n1+n1-(N1/2)+k1)*n2;
5627 f_hat112=f_hat + (k0*N1+k1)*N2+(N2/2);
5628 g_hat212=g_hat + (k0*n1+n1-(N1/2)+k1)*n2;
5629 f_hat212=f_hat + (((N0/2)+k0)*N1+k1)*N2+(N2/2);
5630 g_hat122=g_hat + ((n0-(N0/2)+k0)*n1+k1)*n2;
5631 f_hat122=f_hat + (k0*N1+(N1/2)+k1)*N2+(N2/2);
5632 g_hat222=g_hat + (k0*n1+k1)*n2;
5633 f_hat222=f_hat + (((N0/2)+k0)*N1+(N1/2)+k1)*N2+(N2/2);
5634
5635 for(k2=0;k2<N2/2;k2++)
5636 {
5637 ck21=c_phi_inv21[k2];
5638 ck22=c_phi_inv22[k2];
5639
5640 f_hat111[k2] = g_hat111[k2] * ck01 * ck11 * ck21;
5641 f_hat211[k2] = g_hat211[k2] * ck02 * ck11 * ck21;
5642 f_hat121[k2] = g_hat121[k2] * ck01 * ck12 * ck21;
5643 f_hat221[k2] = g_hat221[k2] * ck02 * ck12 * ck21;
5644
5645 f_hat112[k2] = g_hat112[k2] * ck01 * ck11 * ck22;
5646 f_hat212[k2] = g_hat212[k2] * ck02 * ck11 * ck22;
5647 f_hat122[k2] = g_hat122[k2] * ck01 * ck12 * ck22;
5648 f_hat222[k2] = g_hat222[k2] * ck02 * ck12 * ck22;
5649 }
5650 }
5651 }
5652 }
5653 else
5654#ifdef _OPENMP
5655 #pragma omp parallel for default(shared) private(k0,k1,k2,ck01,ck02,ck11,ck12,ck21,ck22)
5656#endif
5657 for(k0=0;k0<N0/2;k0++)
5658 {
5659 ck01=K(1.0)/(PHI_HUT(ths->n[0],k0-N0/2,0));
5660 ck02=K(1.0)/(PHI_HUT(ths->n[0],k0,0));
5661 for(k1=0;k1<N1/2;k1++)
5662 {
5663 ck11=K(1.0)/(PHI_HUT(ths->n[1],k1-N1/2,1));
5664 ck12=K(1.0)/(PHI_HUT(ths->n[1],k1,1));
5665
5666 for(k2=0;k2<N2/2;k2++)
5667 {
5668 ck21=K(1.0)/(PHI_HUT(ths->n[2],k2-N2/2,2));
5669 ck22=K(1.0)/(PHI_HUT(ths->n[2],k2,2));
5670
5671 f_hat[(k0*N1+k1)*N2+k2] = g_hat[((n0-N0/2+k0)*n1+n1-N1/2+k1)*n2+n2-N2/2+k2] * ck01 * ck11 * ck21;
5672 f_hat[((N0/2+k0)*N1+k1)*N2+k2] = g_hat[(k0*n1+n1-N1/2+k1)*n2+n2-N2/2+k2] * ck02 * ck11 * ck21;
5673 f_hat[(k0*N1+N1/2+k1)*N2+k2] = g_hat[((n0-N0/2+k0)*n1+k1)*n2+n2-N2/2+k2] * ck01 * ck12 * ck21;
5674 f_hat[((N0/2+k0)*N1+N1/2+k1)*N2+k2] = g_hat[(k0*n1+k1)*n2+n2-N2/2+k2] * ck02 * ck12 * ck21;
5675
5676 f_hat[(k0*N1+k1)*N2+N2/2+k2] = g_hat[((n0-N0/2+k0)*n1+n1-N1/2+k1)*n2+k2] * ck01 * ck11 * ck22;
5677 f_hat[((N0/2+k0)*N1+k1)*N2+N2/2+k2] = g_hat[(k0*n1+n1-N1/2+k1)*n2+k2] * ck02 * ck11 * ck22;
5678 f_hat[(k0*N1+N1/2+k1)*N2+N2/2+k2] = g_hat[((n0-N0/2+k0)*n1+k1)*n2+k2] * ck01 * ck12 * ck22;
5679 f_hat[((N0/2+k0)*N1+N1/2+k1)*N2+N2/2+k2] = g_hat[(k0*n1+k1)*n2+k2] * ck02 * ck12 * ck22;
5680 }
5681 }
5682 }
5683
5684 TOC(0)
5685}
5686
5689void X(trafo)(X(plan) *ths)
5690{
5691 /* use direct transform if degree N is too low */
5692 for (int j = 0; j < ths->d; j++)
5693 {
5694 if((ths->N[j] <= ths->m) || (ths->n[j] <= 2*ths->m+2))
5695 {
5696 X(trafo_direct)(ths);
5697 return;
5698 }
5699 }
5700
5701 switch(ths->d)
5702 {
5703 case 1: X(trafo_1d)(ths); break;
5704 case 2: X(trafo_2d)(ths); break;
5705 case 3: X(trafo_3d)(ths); break;
5706 default:
5707 {
5708 /* use ths->my_fftw_plan1 */
5709 ths->g_hat = ths->g1;
5710 ths->g = ths->g2;
5711
5715 TIC(0)
5716 D_A(ths);
5717 TOC(0)
5718
5723 TIC_FFTW(1)
5724 FFTW(execute)(ths->my_fftw_plan1);
5725 TOC_FFTW(1)
5726
5730 TIC(2)
5731 B_A(ths);
5732 TOC(2)
5733 }
5734 }
5735} /* nfft_trafo */
5736
5737void X(adjoint)(X(plan) *ths)
5738{
5739 /* use direct transform if degree N is too low */
5740 for (int j = 0; j < ths->d; j++)
5741 {
5742 if((ths->N[j] <= ths->m) || (ths->n[j] <= 2*ths->m+2))
5743 {
5744 X(adjoint_direct)(ths);
5745 return;
5746 }
5747 }
5748
5749 switch(ths->d)
5750 {
5751 case 1: X(adjoint_1d)(ths); break;
5752 case 2: X(adjoint_2d)(ths); break;
5753 case 3: X(adjoint_3d)(ths); break;
5754 default:
5755 {
5756 /* use ths->my_fftw_plan2 */
5757 ths->g_hat=ths->g1;
5758 ths->g=ths->g2;
5759
5763 TIC(2)
5764 B_T(ths);
5765 TOC(2)
5766
5771 TIC_FFTW(1)
5772 FFTW(execute)(ths->my_fftw_plan2);
5773 TOC_FFTW(1)
5774
5778 TIC(0)
5779 D_T(ths);
5780 TOC(0)
5781 }
5782 }
5783} /* nfft_adjoint */
5784
5785
5788static void precompute_phi_hut(X(plan) *ths)
5789{
5790 INT ks[ths->d]; /* index over all frequencies */
5791 INT t; /* index over all dimensions */
5792
5793 ths->c_phi_inv = (R**) Y(malloc)((size_t)(ths->d) * sizeof(R*));
5794
5795 for (t = 0; t < ths->d; t++)
5796 {
5797 ths->c_phi_inv[t] = (R*)Y(malloc)((size_t)(ths->N[t]) * sizeof(R));
5798
5799 for (ks[t] = 0; ks[t] < ths->N[t]; ks[t]++)
5800 {
5801 ths->c_phi_inv[t][ks[t]]= K(1.0) / (PHI_HUT(ths->n[t], ks[t] - ths->N[t] / 2,t));
5802 }
5803 }
5804} /* nfft_phi_hut */
5805
5810void X(precompute_lin_psi)(X(plan) *ths)
5811{
5812 INT t;
5813 INT j;
5814 R step;
5816 for (t=0; t<ths->d; t++)
5817 {
5818 step = ((R)(ths->m+2)) / ((R)(ths->K * ths->n[t]));
5819 for(j = 0;j <= ths->K; j++)
5820 {
5821 ths->psi[(ths->K+1)*t + j] = PHI(ths->n[t], (R)(j) * step,t);
5822 } /* for(j) */
5823 } /* for(t) */
5824}
5825
5826void X(precompute_fg_psi)(X(plan) *ths)
5827{
5828 INT t;
5829 INT u, o;
5831 sort(ths);
5832
5833 for (t=0; t<ths->d; t++)
5834 {
5835 INT j;
5836#ifdef _OPENMP
5837 #pragma omp parallel for default(shared) private(j,u,o)
5838#endif
5839 for (j = 0; j < ths->M_total; j++)
5840 {
5841 uo(ths,j,&u,&o,t);
5842
5843 ths->psi[2*(j*ths->d+t)]=
5844 (PHI(ths->n[t] ,(ths->x[j*ths->d+t] - ((R)u) / (R)(ths->n[t])),t));
5845
5846 ths->psi[2*(j*ths->d+t)+1]=
5847 EXP(K(2.0) * ((R)(ths->n[t]) * ths->x[j*ths->d+t] - (R)(u)) / ths->b[t]);
5848 } /* for(j) */
5849 }
5850 /* for(t) */
5851} /* nfft_precompute_fg_psi */
5852
5853void X(precompute_psi)(X(plan) *ths)
5854{
5855 INT t; /* index over all dimensions */
5856 INT l; /* index u<=l<=o */
5857 INT lj; /* index 0<=lj<u+o+1 */
5858 INT u, o; /* depends on x_j */
5859
5860 sort(ths);
5861
5862 for (t=0; t<ths->d; t++)
5863 {
5864 INT j;
5865#ifdef _OPENMP
5866 #pragma omp parallel for default(shared) private(j,l,lj,u,o)
5867#endif
5868 for (j = 0; j < ths->M_total; j++)
5869 {
5870 uo(ths,j,&u,&o,t);
5871
5872 for(l = u, lj = 0; l <= o; l++, lj++)
5873 ths->psi[(j * ths->d + t) * (2 * ths->m + 2) + lj] =
5874 (PHI(ths->n[t], (ths->x[j*ths->d+t] - ((R)l) / (R)(ths->n[t])), t));
5875 } /* for(j) */
5876 }
5877 /* for(t) */
5878} /* nfft_precompute_psi */
5879
5880#ifdef _OPENMP
5881static void nfft_precompute_full_psi_omp(X(plan) *ths)
5882{
5883 INT j;
5884 INT lprod;
5886 {
5887 INT t;
5888 for(t=0,lprod = 1; t<ths->d; t++)
5889 lprod *= 2*ths->m+2;
5890 }
5891
5892 #pragma omp parallel for default(shared) private(j)
5893 for(j=0; j<ths->M_total; j++)
5894 {
5895 INT t,t2;
5896 INT l_L;
5897 INT lj[ths->d];
5898 INT ll_plain[ths->d+1];
5900 INT u[ths->d], o[ths->d];
5902 R phi_prod[ths->d+1];
5903 INT ix = j*lprod;
5904
5905 phi_prod[0]=1;
5906 ll_plain[0]=0;
5907
5908 MACRO_init_uo_l_lj_t;
5909
5910 for(l_L=0; l_L<lprod; l_L++, ix++)
5911 {
5912 MACRO_update_phi_prod_ll_plain(without_PRE_PSI);
5913
5914 ths->psi_index_g[ix]=ll_plain[ths->d];
5915 ths->psi[ix]=phi_prod[ths->d];
5916
5917 MACRO_count_uo_l_lj_t;
5918 } /* for(l_L) */
5919
5920 ths->psi_index_f[j]=lprod;
5921 } /* for(j) */
5922}
5923#endif
5924
5925void X(precompute_full_psi)(X(plan) *ths)
5926{
5927#ifdef _OPENMP
5928 sort(ths);
5929
5930 nfft_precompute_full_psi_omp(ths);
5931#else
5932 INT t, t2; /* index over all dimensions */
5933 INT j; /* index over all nodes */
5934 INT l_L; /* plain index 0 <= l_L < lprod */
5935 INT lj[ths->d]; /* multi index 0<=lj<u+o+1 */
5936 INT ll_plain[ths->d+1]; /* postfix plain index */
5937 INT lprod; /* 'bandwidth' of matrix B */
5938 INT u[ths->d], o[ths->d]; /* depends on x_j */
5939
5940 R phi_prod[ths->d+1];
5941
5942 INT ix, ix_old;
5943
5944 sort(ths);
5945
5946 phi_prod[0] = K(1.0);
5947 ll_plain[0] = 0;
5948
5949 for (t = 0, lprod = 1; t < ths->d; t++)
5950 lprod *= 2 * ths->m + 2;
5951
5952 for (j = 0, ix = 0, ix_old = 0; j < ths->M_total; j++)
5953 {
5954 MACRO_init_uo_l_lj_t;
5955
5956 for (l_L = 0; l_L < lprod; l_L++, ix++)
5957 {
5958 MACRO_update_phi_prod_ll_plain(without_PRE_PSI);
5959
5960 ths->psi_index_g[ix] = ll_plain[ths->d];
5961 ths->psi[ix] = phi_prod[ths->d];
5962
5963 MACRO_count_uo_l_lj_t;
5964 } /* for(l_L) */
5965
5966 ths->psi_index_f[j] = ix - ix_old;
5967 ix_old = ix;
5968 } /* for(j) */
5969#endif
5970}
5971
5972void X(precompute_one_psi)(X(plan) *ths)
5973{
5974 if(ths->flags & PRE_LIN_PSI)
5975 X(precompute_lin_psi)(ths);
5976 if(ths->flags & PRE_FG_PSI)
5977 X(precompute_fg_psi)(ths);
5978 if(ths->flags & PRE_PSI)
5979 X(precompute_psi)(ths);
5980 if(ths->flags & PRE_FULL_PSI)
5981 X(precompute_full_psi)(ths);
5982}
5983
5984static void init_help(X(plan) *ths)
5985{
5986 INT t; /* index over all dimensions */
5987 INT lprod; /* 'bandwidth' of matrix B */
5988
5989 if (ths->flags & NFFT_OMP_BLOCKWISE_ADJOINT)
5990 ths->flags |= NFFT_SORT_NODES;
5991
5992 ths->N_total = intprod(ths->N, 0, ths->d);
5993 ths->n_total = intprod(ths->n, 0, ths->d);
5994
5995 ths->sigma = (R*) Y(malloc)((size_t)(ths->d) * sizeof(R));
5996
5997 for(t = 0;t < ths->d; t++)
5998 ths->sigma[t] = ((R)ths->n[t]) / (R)(ths->N[t]);
5999
6000 WINDOW_HELP_INIT;
6001
6002 if(ths->flags & MALLOC_X)
6003 ths->x = (R*)Y(malloc)((size_t)(ths->d * ths->M_total) * sizeof(R));
6004
6005 if(ths->flags & MALLOC_F_HAT)
6006 ths->f_hat = (C*)Y(malloc)((size_t)(ths->N_total) * sizeof(C));
6007
6008 if(ths->flags & MALLOC_F)
6009 ths->f = (C*)Y(malloc)((size_t)(ths->M_total) * sizeof(C));
6010
6011 if(ths->flags & PRE_PHI_HUT)
6012 precompute_phi_hut(ths);
6013
6014 if (ths->flags & PRE_LIN_PSI)
6015 {
6016 if (ths->K == 0)
6017 {
6018 ths->K = Y(m2K)(ths->m);
6019 }
6020 ths->psi = (R*) Y(malloc)((size_t)((ths->K+1) * ths->d) * sizeof(R));
6021 }
6022
6023 if(ths->flags & PRE_FG_PSI)
6024 ths->psi = (R*) Y(malloc)((size_t)(ths->M_total * ths->d * 2) * sizeof(R));
6025
6026 if(ths->flags & PRE_PSI)
6027 ths->psi = (R*) Y(malloc)((size_t)(ths->M_total * ths->d * (2 * ths->m + 2)) * sizeof(R));
6028
6029 if(ths->flags & PRE_FULL_PSI)
6030 {
6031 for (t = 0, lprod = 1; t < ths->d; t++)
6032 lprod *= 2 * ths->m + 2;
6033
6034 ths->psi = (R*) Y(malloc)((size_t)(ths->M_total * lprod) * sizeof(R));
6035
6036 ths->psi_index_f = (INT*) Y(malloc)((size_t)(ths->M_total) * sizeof(INT));
6037 ths->psi_index_g = (INT*) Y(malloc)((size_t)(ths->M_total * lprod) * sizeof(INT));
6038 }
6039
6040 if(ths->flags & FFTW_INIT)
6041 {
6042#ifdef _OPENMP
6043 INT nthreads = Y(get_num_threads)();
6044#endif
6045
6046 ths->g1 = (C*)Y(malloc)((size_t)(ths->n_total) * sizeof(C));
6047
6048 if(ths->flags & FFT_OUT_OF_PLACE)
6049 ths->g2 = (C*) Y(malloc)((size_t)(ths->n_total) * sizeof(C));
6050 else
6051 ths->g2 = ths->g1;
6052
6053#if defined(_OPENMP) && defined(HAVE_FFTW_THREADS)
6054#pragma omp critical (nfft_omp_critical_fftw_plan)
6055{
6056 FFTW(plan_with_nthreads)(nthreads);
6057#endif
6058 {
6059 int *_n = Y(malloc)((size_t)(ths->d) * sizeof(int));
6060
6061 for (t = 0; t < ths->d; t++)
6062 _n[t] = (int)(ths->n[t]);
6063
6064 ths->my_fftw_plan1 = FFTW(plan_dft)((int)ths->d, _n, ths->g1, ths->g2, FFTW_FORWARD, ths->fftw_flags);
6065 ths->my_fftw_plan2 = FFTW(plan_dft)((int)ths->d, _n, ths->g2, ths->g1, FFTW_BACKWARD, ths->fftw_flags);
6066 Y(free)(_n);
6067 }
6068#if defined(_OPENMP) && defined(HAVE_FFTW_THREADS)
6069}
6070#endif
6071 }
6072
6073 if(ths->flags & NFFT_SORT_NODES)
6074 ths->index_x = (INT*) Y(malloc)(sizeof(INT) * 2U * (size_t)(ths->M_total));
6075 else
6076 ths->index_x = NULL;
6077
6078 ths->mv_trafo = (void (*) (void* ))X(trafo);
6079 ths->mv_adjoint = (void (*) (void* ))X(adjoint);
6080}
6081
6082void X(init)(X(plan) *ths, int d, int *N, int M_total)
6083{
6084 INT t; /* index over all dimensions */
6085
6086 ths->d = (INT)d;
6087
6088 ths->N = (INT*) Y(malloc)((size_t)(d) * sizeof(INT));
6089
6090 for (t = 0; t < d; t++)
6091 ths->N[t] = (INT)N[t];
6092
6093 ths->M_total = (INT)M_total;
6094
6095 ths->n = (INT*) Y(malloc)((size_t)(d) * sizeof(INT));
6096
6097 for (t = 0; t < d; t++)
6098 ths->n[t] = 2 * (Y(next_power_of_2)(ths->N[t]));
6099
6100 ths->m = WINDOW_HELP_ESTIMATE_m;
6101
6102 if (d > 1)
6103 {
6104#ifdef _OPENMP
6105 ths->flags = PRE_PHI_HUT | PRE_PSI | MALLOC_X| MALLOC_F_HAT | MALLOC_F |
6106 FFTW_INIT | NFFT_SORT_NODES |
6107 NFFT_OMP_BLOCKWISE_ADJOINT;
6108#else
6109 ths->flags = PRE_PHI_HUT | PRE_PSI | MALLOC_X| MALLOC_F_HAT | MALLOC_F |
6110 FFTW_INIT | NFFT_SORT_NODES;
6111#endif
6112 }
6113 else
6114 ths->flags = PRE_PHI_HUT | PRE_PSI | MALLOC_X| MALLOC_F_HAT | MALLOC_F |
6116
6117 ths->fftw_flags= FFTW_ESTIMATE| FFTW_DESTROY_INPUT;
6118
6119 ths->K = 0;
6120 init_help(ths);
6121}
6122
6123void X(init_guru)(X(plan) *ths, int d, int *N, int M_total, int *n, int m,
6124 unsigned flags, unsigned fftw_flags)
6125{
6126 INT t; /* index over all dimensions */
6127
6128 ths->d = (INT)d;
6129 ths->M_total = (INT)M_total;
6130 ths->N = (INT*)Y(malloc)((size_t)(ths->d) * sizeof(INT));
6131
6132 for (t = 0; t < d; t++)
6133 ths->N[t] = (INT)N[t];
6134
6135 ths->n = (INT*)Y(malloc)((size_t)(ths->d) * sizeof(INT));
6136
6137 for (t = 0; t < d; t++)
6138 ths->n[t] = (INT)n[t];
6139
6140 ths->m = (INT)m;
6141
6142 ths->flags = flags;
6143 ths->fftw_flags = fftw_flags;
6144
6145 ths->K = 0;
6146 init_help(ths);
6147}
6148
6149void X(init_lin)(X(plan) *ths, int d, int *N, int M_total, int *n, int m, int K,
6150 unsigned flags, unsigned fftw_flags)
6151{
6152 INT t; /* index over all dimensions */
6153
6154 ths->d = (INT)d;
6155 ths->M_total = (INT)M_total;
6156 ths->N = (INT*)Y(malloc)((size_t)(ths->d) * sizeof(INT));
6157
6158 for (t = 0; t < d; t++)
6159 ths->N[t] = (INT)N[t];
6160
6161 ths->n = (INT*)Y(malloc)((size_t)(ths->d) * sizeof(INT));
6162
6163 for (t = 0; t < d; t++)
6164 ths->n[t] = (INT)n[t];
6165
6166 ths->m = (INT)m;
6167
6168 ths->flags = flags;
6169 ths->fftw_flags = fftw_flags;
6170
6171 ths->K = K;
6172 init_help(ths);
6173}
6174
6175void X(init_1d)(X(plan) *ths, int N1, int M_total)
6176{
6177 int N[1];
6178
6179 N[0] = N1;
6180
6181 X(init)(ths, 1, N, M_total);
6182}
6183
6184void X(init_2d)(X(plan) *ths, int N1, int N2, int M_total)
6185{
6186 int N[2];
6187
6188 N[0] = N1;
6189 N[1] = N2;
6190 X(init)(ths, 2, N, M_total);
6191}
6192
6193void X(init_3d)(X(plan) *ths, int N1, int N2, int N3, int M_total)
6194{
6195 int N[3];
6196
6197 N[0] = N1;
6198 N[1] = N2;
6199 N[2] = N3;
6200 X(init)(ths, 3, N, M_total);
6201}
6202
6203const char* X(check)(X(plan) *ths)
6204{
6205 INT j;
6206
6207 if (!ths->f)
6208 return "Member f not initialized.";
6209
6210 if (!ths->x)
6211 return "Member x not initialized.";
6212
6213 if (!ths->f_hat)
6214 return "Member f_hat not initialized.";
6215
6216 if ((ths->flags & PRE_LIN_PSI) && ths->K < ths->M_total)
6217 return "Number of nodes too small to use PRE_LIN_PSI.";
6218
6219 for (j = 0; j < ths->M_total * ths->d; j++)
6220 {
6221 if ((ths->x[j]<-K(0.5)) || (ths->x[j]>= K(0.5)))
6222 {
6223 return "ths->x out of range [-0.5,0.5)";
6224 }
6225 }
6226
6227 for (j = 0; j < ths->d; j++)
6228 {
6229 if (ths->sigma[j] <= 1)
6230 return "Oversampling factor too small";
6231
6232 /* Automatically calls trafo_direct if
6233 if(ths->N[j] <= ths->m)
6234 return "Polynomial degree N is <= cut-off m";
6235 */
6236
6237 if(ths->N[j]%2 == 1)
6238 return "polynomial degree N has to be even";
6239 }
6240 return 0;
6241}
6242
6243void X(finalize)(X(plan) *ths)
6244{
6245 INT t; /* index over dimensions */
6246
6247 if(ths->flags & NFFT_SORT_NODES)
6248 Y(free)(ths->index_x);
6249
6250 if(ths->flags & FFTW_INIT)
6251 {
6252#ifdef _OPENMP
6253 #pragma omp critical (nfft_omp_critical_fftw_plan)
6254#endif
6255 FFTW(destroy_plan)(ths->my_fftw_plan2);
6256#ifdef _OPENMP
6257 #pragma omp critical (nfft_omp_critical_fftw_plan)
6258#endif
6259 FFTW(destroy_plan)(ths->my_fftw_plan1);
6260
6261 if(ths->flags & FFT_OUT_OF_PLACE)
6262 Y(free)(ths->g2);
6263
6264 Y(free)(ths->g1);
6265 }
6266
6267 if(ths->flags & PRE_FULL_PSI)
6268 {
6269 Y(free)(ths->psi_index_g);
6270 Y(free)(ths->psi_index_f);
6271 Y(free)(ths->psi);
6272 }
6273
6274 if(ths->flags & PRE_PSI)
6275 Y(free)(ths->psi);
6276
6277 if(ths->flags & PRE_FG_PSI)
6278 Y(free)(ths->psi);
6279
6280 if(ths->flags & PRE_LIN_PSI)
6281 Y(free)(ths->psi);
6282
6283 if(ths->flags & PRE_PHI_HUT)
6284 {
6285 for (t = 0; t < ths->d; t++)
6286 Y(free)(ths->c_phi_inv[t]);
6287 Y(free)(ths->c_phi_inv);
6288 }
6289
6290 if(ths->flags & MALLOC_F)
6291 Y(free)(ths->f);
6292
6293 if(ths->flags & MALLOC_F_HAT)
6294 Y(free)(ths->f_hat);
6295
6296 if(ths->flags & MALLOC_X)
6297 Y(free)(ths->x);
6298
6299 WINDOW_HELP_FINALIZE;
6300
6301 Y(free)(ths->sigma);
6302 Y(free)(ths->n);
6303 Y(free)(ths->N);
6304}
#define FG_PSI
Definition nfft3.h:188
#define MALLOC_F_HAT
Definition nfft3.h:194
#define MALLOC_X
Definition nfft3.h:193
#define PRE_FULL_PSI
Definition nfft3.h:192
#define FFT_OUT_OF_PLACE
Definition nfft3.h:196
#define PRE_PSI
Definition nfft3.h:191
#define PRE_FG_PSI
Definition nfft3.h:190
#define MALLOC_F
Definition nfft3.h:195
#define PRE_LIN_PSI
Definition nfft3.h:189
#define FFTW_INIT
Definition nfft3.h:197
#define PRE_PHI_HUT
Definition nfft3.h:187
#define TIC(a)
Timing, method works since the inaccurate timer is updated mostly in the measured function.
Definition infft.h:1457
#define UNUSED(x)
Dummy use of unused parameters to silence compiler warnings.
Definition infft.h:1376
Internal header file for auxiliary definitions and functions.
Header file for the nfft3 library.