MADNESS 0.10.1
funcimpl.h
Go to the documentation of this file.
1/*
2 This file is part of MADNESS.
3
4 Copyright (C) 2007,2010 Oak Ridge National Laboratory
5
6 This program is free software; you can redistribute it and/or modify
7 it under the terms of the GNU General Public License as published by
8 the Free Software Foundation; either version 2 of the License, or
9 (at your option) any later version.
10
11 This program is distributed in the hope that it will be useful,
12 but WITHOUT ANY WARRANTY; without even the implied warranty of
13 MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
14 GNU General Public License for more details.
15
16 You should have received a copy of the GNU General Public License
17 along with this program; if not, write to the Free Software
18 Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA 02111-1307 USA
19
20 For more information please contact:
21
22 Robert J. Harrison
23 Oak Ridge National Laboratory
24 One Bethel Valley Road
25 P.O. Box 2008, MS-6367
26
27 email: harrisonrj@ornl.gov
28 tel: 865-241-3937
29 fax: 865-572-0680
30*/
31
32#ifndef MADNESS_MRA_FUNCIMPL_H__INCLUDED
33#define MADNESS_MRA_FUNCIMPL_H__INCLUDED
34
35/// \file funcimpl.h
36/// \brief Provides FunctionCommonData, FunctionImpl and FunctionFactory
37
39#include <madness/world/print.h>
40#include <madness/misc/misc.h>
43
45#include <madness/mra/indexit.h>
46#include <madness/mra/key.h>
50
51#include <madness/mra/leafop.h>
52
53#include <array>
54#include <iostream>
55#include <type_traits>
56
57namespace madness {
58 template <typename T, std::size_t NDIM>
59 class DerivativeBase;
60
61 template<typename T, std::size_t NDIM>
62 class FunctionImpl;
63
64 template<typename T, std::size_t NDIM>
65 class FunctionNode;
66
67 template<typename T, std::size_t NDIM>
68 class Function;
69
70 template<typename T, std::size_t NDIM>
71 class FunctionFactory;
72
73 template<typename T, std::size_t NDIM, std::size_t MDIM>
74 class CompositeFunctorInterface;
75
76 template<int D>
78
79}
80
81namespace madness {
82
83
84 /// A simple process map
85 template<typename keyT>
86 class SimplePmap : public WorldDCPmapInterface<keyT> {
87 private:
88 const int nproc;
90
91 public:
92 SimplePmap(World& world) : nproc(world.nproc()), me(world.rank())
93 { }
94
95 ProcessID owner(const keyT& key) const {
96 if (key.level() == 0)
97 return 0;
98 else
99 return key.hash() % nproc;
100 }
101 };
102
103 /// A pmap that locates children on odd levels with their even level parents
104 template <typename keyT>
105 class LevelPmap : public WorldDCPmapInterface<keyT> {
106 private:
107 const int nproc;
108 public:
109 LevelPmap() : nproc(0) {};
110
111 LevelPmap(World& world) : nproc(world.nproc()) {}
112
113 /// Find the owner of a given key
114 ProcessID owner(const keyT& key) const {
115 Level n = key.level();
116 if (n == 0) return 0;
117 hashT hash;
118 if (n <= 3 || (n&0x1)) hash = key.hash();
119 else hash = key.parent().hash();
120 return hash%nproc;
121 }
122 };
123
124
125 /// Sentinel for norm_tree/dnorm_tree meaning "not computed". Overflows the screening
126 /// criterion, so nodes carrying it are descended rather than screened.
127 static constexpr double NORM_TREE_UNCOMPUTED = 1e300;
128
129 /// Safety margin on the tolerance of the screened multiplication: the criterion estimates
130 /// the neglected cross terms rather than bounding them. Tightening is cheap and buys
131 /// accuracy, hence the extra factor. Applied once, in mulXXvec.
132 static constexpr double MUL_SCREENING_SAFETY = 0.1;
133
134 /// FunctionNode holds the coefficients, etc., at each node of the 2^NDIM-tree
135 template<typename T, std::size_t NDIM>
137 public:
140 private:
141 // Should compile OK with these volatile but there should
142 // be no need to set as volatile since the container internally
143 // stores the entire entry as volatile
144
145 coeffT _coeffs; ///< The coefficients, if any
146 double _norm_tree; ///< After norm_tree will contain norm of sum coefficients summed up tree
147 double _dnorm_tree=NORM_TREE_UNCOMPUTED; ///< norm of the difference coefficients summed up the tree
148 bool _has_children; ///< True if there are children
149 coeffT buffer; ///< The coefficients, if any
150 double dnorm=-1.0; ///< norm of the d coefficients, also defined if there are no d coefficients
151 double snorm=-1.0; ///< norm of the s coefficients
152
153 public:
154 typedef WorldContainer<Key<NDIM> , FunctionNode<T, NDIM> > dcT; ///< Type of container holding the nodes
155 /// Default constructor makes node without coeff or children
159
160 /// Constructor from given coefficients with optional children
161
162 /// Note that only a shallow copy of the coeff are taken so
163 /// you should pass in a deep copy if you want the node to
164 /// take ownership.
165 explicit
169
170 explicit
174
175 explicit
176 FunctionNode(const coeffT& coeff, double norm_tree, double dnorm_tree, double snorm, double dnorm, bool has_children) :
178 }
179
182 _has_children(other._has_children), dnorm(other.dnorm), snorm(other.snorm) {
183 }
184
187 if (this != &other) {
188 coeff() = copy(other.coeff());
189 _norm_tree = other._norm_tree;
190 _dnorm_tree = other._dnorm_tree;
192 dnorm=other.dnorm;
193 snorm=other.snorm;
194 }
195 return *this;
196 }
197
198 /// Copy with possible type conversion of coefficients, copying all other state
199
200 /// Choose to not overload copy and type conversion operators
201 /// so there are no automatic type conversions.
202 template<typename Q>
204 convert() const {
205 return FunctionNode<Q, NDIM> (madness::convert<Q,T>(coeff()), _norm_tree, _dnorm_tree, snorm, dnorm, _has_children);
206 }
207
208 /// Returns true if there are coefficients in this node
209 bool
210 has_coeff() const {
211 return _coeffs.has_data();
212 }
213
214
215 /// Returns true if this node has children
216 bool
217 has_children() const {
218 return _has_children;
219 }
220
221 /// Returns true if this does not have children
222 bool
223 is_leaf() const {
224 return !_has_children;
225 }
226
227 /// Returns true if this node is invalid (no coeffs and no children)
228 bool
229 is_invalid() const {
230 return !(has_coeff() || has_children());
231 }
232
233 /// Returns a non-const reference to the tensor containing the coeffs
234
235 /// Returns an empty tensor if there are no coefficients.
236 coeffT&
238 MADNESS_ASSERT(_coeffs.ndim() == -1 || (_coeffs.dim(0) <= 2
239 * MAXK && _coeffs.dim(0) >= 0));
240 return const_cast<coeffT&>(_coeffs);
241 }
242
243 /// Returns a const reference to the tensor containing the coeffs
244
245 /// Returns an empty tensor if there are no coefficeints.
246 const coeffT&
247 coeff() const {
248 return const_cast<const coeffT&>(_coeffs);
249 }
250
251 /// Returns the number of coefficients in this node
252 size_t size() const {
253 return _coeffs.size();
254 }
255
256 public:
257
258 /// reduces the rank of the coefficients (if applicable)
259 void reduceRank(const double& eps) {
260 _coeffs.reduce_rank(eps);
261 }
262
263 /// Sets \c has_children attribute to value of \c flag.
264 void set_has_children(bool flag) {
265 _has_children = flag;
266 }
267
268 /// Sets \c has_children attribute to true recurring up to ensure connected
270 //madness::print(" set_chi_recu: ", key, *this);
271 //PROFILE_MEMBER_FUNC(FunctionNode); // Too fine grain for routine profiling
272 if (!(has_children() || has_coeff() || key.level()==0)) {
273 // If node already knows it has children or it has
274 // coefficients then it must already be connected to
275 // its parent. If not, the node was probably just
276 // created for this operation and must be connected to
277 // its parent.
278 Key<NDIM> parent = key.parent();
279 // Task on next line used to be TaskAttributes::hipri()) ... but deferring execution of this
280 // makes sense since it is not urgent and lazy connection will likely mean that less forwarding
281 // will happen since the upper level task will have already made the connection.
282 const_cast<dcT&>(c).task(parent, &FunctionNode<T,NDIM>::set_has_children_recursive, c, parent);
283 //const_cast<dcT&>(c).send(parent, &FunctionNode<T,NDIM>::set_has_children_recursive, c, parent);
284 //madness::print(" set_chi_recu: forwarding",key,parent);
285 }
286 _has_children = true;
287 }
288
289 /// Sets \c has_children attribute to value of \c !flag
290 void set_is_leaf(bool flag) {
291 _has_children = !flag;
292 }
293
294 /// Takes a \em shallow copy of the coeff --- same as \c this->coeff()=coeff
295 void set_coeff(const coeffT& coeffs) {
296 coeff() = coeffs;
297 if ((_coeffs.has_data()) and ((_coeffs.dim(0) < 0) || (_coeffs.dim(0)>2*MAXK))) {
298 print("set_coeff: may have a problem");
299 print("set_coeff: coeff.dim[0] =", coeffs.dim(0), ", 2* MAXK =", 2*MAXK);
300 }
301 MADNESS_ASSERT(coeffs.dim(0)<=2*MAXK && coeffs.dim(0)>=0);
302 }
303
304 /// Clears the coefficients (has_coeff() will subsequently return false)
305 void clear_coeff() {
306 coeff()=coeffT();
307 }
308
309 /// Scale the coefficients of this node
310 template <typename Q>
311 void scale(Q a) {
312 _coeffs.scale(a);
313 }
314
315 /// Sets the value of norm_tree
318 }
319
320 /// Sets the value of dnorm_tree
321 void set_dnorm_tree(double dnorm_tree) {
322 _dnorm_tree = dnorm_tree;
323 }
324
325 /// Gets the value of norm_tree
326 double get_norm_tree() const {
327 return _norm_tree;
328 }
329
330 /// Gets the value of dnorm_tree
331 double get_dnorm_tree() const {
332 return _dnorm_tree;
333 }
334
335 /// return the precomputed norm of the (virtual) d coefficients
336 double get_dnorm() const {
337 return dnorm;
338 }
339
340 /// set the precomputed norm of the (virtual) s coefficients
341 void set_snorm(const double sn) {
342 snorm=sn;
343 }
344
345 /// set the precomputed norm of the (virtual) d coefficients
346 void set_dnorm(const double dn) {
347 dnorm=dn;
348 }
349
350 /// get the precomputed norm of the (virtual) s coefficients
351 double get_snorm() const {
352 return snorm;
353 }
354
356 snorm = 0.0;
357 dnorm = 0.0;
358 if (coeff().size() == 0) { ;
359 } else if (coeff().dim(0) == cdata.vk[0]) {
360 snorm = coeff().normf();
361
362 } else if (coeff().is_full_tensor()) {
363 Tensor<T> c = copy(coeff().get_tensor());
364 snorm = c(cdata.s0).normf();
365 c(cdata.s0) = 0.0;
366 dnorm = c.normf();
367
368 } else if (coeff().is_svd_tensor()) {
369 coeffT c= coeff()(cdata.s0);
370 snorm = c.normf();
371 double norm = coeff().normf();
372 dnorm = sqrt(norm * norm - snorm * snorm);
373
374 } else {
375 MADNESS_EXCEPTION("cannot use compute_dnorm", 1);
376 }
377 }
378
379
380 /// General bi-linear operation --- this = this*alpha + other*beta
381
382 /// This/other may not have coefficients. Has_children will be
383 /// true in the result if either this/other have children.
384 template <typename Q, typename R>
385 void gaxpy_inplace(const T& alpha, const FunctionNode<Q,NDIM>& other, const R& beta) {
386 //PROFILE_MEMBER_FUNC(FuncNode); // Too fine grain for routine profiling
387 if (other.has_children())
388 _has_children = true;
389 if (has_coeff()) {
390 if (other.has_coeff()) {
391 coeff().gaxpy(alpha,other.coeff(),beta);
392 }
393 else {
394 coeff().scale(alpha);
395 }
396 }
397 else if (other.has_coeff()) {
398 coeff() = other.coeff()*beta; //? Is this the correct type conversion?
399 }
400 }
401
402 /// Accumulate inplace and if necessary connect node to parent
403 void accumulate2(const tensorT& t, const typename FunctionNode<T,NDIM>::dcT& c,
404 const Key<NDIM>& key) {
405 // double cpu0=cpu_time();
406 if (has_coeff()) {
407 MADNESS_ASSERT(coeff().is_full_tensor());
408 // if (coeff().type==TT_FULL) {
409 coeff() += coeffT(t,-1.0,TT_FULL);
410 // } else {
411 // tensorT cc=coeff().full_tensor_copy();;
412 // cc += t;
413 // coeff()=coeffT(cc,args);
414 // }
415 }
416 else {
417 // No coeff and no children means the node is newly
418 // created for this operation and therefore we must
419 // tell its parent that it exists.
420 coeff() = coeffT(t,-1.0,TT_FULL);
421 // coeff() = copy(t);
422 // coeff() = coeffT(t,args);
423 if ((!_has_children) && key.level()> 0) {
424 Key<NDIM> parent = key.parent();
425 if (c.is_local(parent))
426 const_cast<dcT&>(c).send(parent, &FunctionNode<T,NDIM>::set_has_children_recursive, c, parent);
427 else
428 const_cast<dcT&>(c).task(parent, &FunctionNode<T,NDIM>::set_has_children_recursive, c, parent);
429 }
430 }
431 //double cpu1=cpu_time();
432 }
433
434
435 /// Accumulate inplace and if necessary connect node to parent
436 void accumulate(const coeffT& t, const typename FunctionNode<T,NDIM>::dcT& c,
437 const Key<NDIM>& key, const TensorArgs& args) {
438 if (has_coeff()) {
439 coeff().add_SVD(t,args.thresh);
440 if (buffer.rank()<coeff().rank()) {
441 if (buffer.has_data()) {
442 buffer.add_SVD(coeff(),args.thresh);
443 } else {
444 buffer=copy(coeff());
445 }
446 coeff()=coeffT();
447 }
448
449 } else {
450 // No coeff and no children means the node is newly
451 // created for this operation and therefore we must
452 // tell its parent that it exists.
453 coeff() = copy(t);
454 if ((!_has_children) && key.level()> 0) {
455 Key<NDIM> parent = key.parent();
456 if (c.is_local(parent))
457 const_cast<dcT&>(c).send(parent, &FunctionNode<T,NDIM>::set_has_children_recursive, c, parent);
458 else
459 const_cast<dcT&>(c).task(parent, &FunctionNode<T,NDIM>::set_has_children_recursive, c, parent);
460 }
461 }
462 }
463
464 void consolidate_buffer(const TensorArgs& args) {
465 if ((coeff().has_data()) and (buffer.has_data())) {
466 coeff().add_SVD(buffer,args.thresh);
467 } else if (buffer.has_data()) {
468 coeff()=buffer;
469 }
470 buffer=coeffT();
471 }
472
473 T trace_conj(const FunctionNode<T,NDIM>& rhs) const {
474 return this->_coeffs.trace_conj((rhs._coeffs));
475 }
476
477 template <typename Archive>
478 void serialize(Archive& ar) {
479 // changing this list changes the on-disk format: bump FUNCTION_ARCHIVE_MAGIC
481 }
482
483 /// like operator<<(ostream&, const FunctionNode<T,NDIM>&) but
484 /// produces a sequence JSON-formatted key-value pairs
485 /// @warning enclose the output in curly braces to make
486 /// a valid JSON object
487 void print_json(std::ostream& s) const {
488 s << "\"has_coeff\":" << this->has_coeff()
489 << ",\"has_children\":" << this->has_children() << ",\"norm\":";
490 double norm = this->has_coeff() ? this->coeff().normf() : 0.0;
491 if (norm < 1e-12)
492 norm = 0.0;
493 double nt = this->get_norm_tree();
494 if (nt == NORM_TREE_UNCOMPUTED)
495 nt = 0.0;
496 s << norm << ",\"norm_tree\":" << nt << ",\"snorm\":"
497 << this->get_snorm() << ",\"dnorm\":" << this->get_dnorm()
498 << ",\"rank\":" << this->coeff().rank();
499 if (this->coeff().is_assigned())
500 s << ",\"dim\":" << this->coeff().dim(0);
501 }
502
503 };
504
505 template <typename T, std::size_t NDIM>
506 std::ostream& operator<<(std::ostream& s, const FunctionNode<T,NDIM>& node) {
507 s << "(has_coeff=" << node.has_coeff() << ", has_children=" << node.has_children() << ", norm=";
508 double norm = node.has_coeff() ? node.coeff().normf() : 0.0;
509 if (norm < 1e-12)
510 norm = 0.0;
511 double nt = node.get_norm_tree();
512 double dnt = node.get_dnorm_tree();
513 if (nt == NORM_TREE_UNCOMPUTED) nt = 0.0;
514 if (dnt == NORM_TREE_UNCOMPUTED) dnt = 0.0;
515 s << norm << ", norm_tree = " << nt << ", dnorm_tree = " << dnt << ", s/dnorm =" << node.get_snorm() << " " << node.get_dnorm() << "), rank="<< node.coeff().rank()<<")";
516 if (node.coeff().is_assigned()) s << " dim " << node.coeff().dim(0) << " ";
517 return s;
518 }
519
520
521 /// returns true if the result of a hartree_product is a leaf node (compute norm & error)
522 template<typename T, size_t NDIM>
524
527 long k;
528 bool do_error_leaf_op() const {return false;}
529
530 hartree_leaf_op() = default;
531 hartree_leaf_op(const implT* f, const long& k) : f(f), k(k) {}
532
533 /// no pre-determination
534 bool operator()(const Key<NDIM>& key) const {return false;}
535
536 /// no post-determination
537 bool operator()(const Key<NDIM>& key, const GenTensor<T>& coeff) const {
538 MADNESS_EXCEPTION("no post-determination in hartree_leaf_op",1);
539 return true;
540 }
541
542 /// post-determination: true if f is a leaf and the result is well-represented
543
544 /// @param[in] key the hi-dimensional key (breaks into keys for f and g)
545 /// @param[in] fcoeff coefficients of f of its appropriate key in NS form
546 /// @param[in] gcoeff coefficients of g of its appropriate key in NS form
547 bool operator()(const Key<NDIM>& key, const Tensor<T>& fcoeff, const Tensor<T>& gcoeff) const {
548
549 if (key.level()<2) return false;
550 Slice s = Slice(0,k-1);
551 std::vector<Slice> s0(NDIM/2,s);
552
553 const double tol=f->get_thresh();
554 const double thresh=f->truncate_tol(tol, key)*0.3; // custom factor to "ensure" accuracy
555 // include the wavelets in the norm, makes it much more accurate
556 const double fnorm=fcoeff.normf();
557 const double gnorm=gcoeff.normf();
558
559 // if the final norm is small, perform the hartree product and return
560 const double norm=fnorm*gnorm; // computing the outer product
561 if (norm < thresh) return true;
562
563 // norm of the scaling function coefficients
564 const double sfnorm=fcoeff(s0).normf();
565 const double sgnorm=gcoeff(s0).normf();
566
567 // get the error of both functions and of the pair function;
568 // need the abs for numerics: sfnorm might be equal fnorm.
569 const double ferror=sqrt(std::abs(fnorm*fnorm-sfnorm*sfnorm));
570 const double gerror=sqrt(std::abs(gnorm*gnorm-sgnorm*sgnorm));
571
572 // if the expected error is small, perform the hartree product and return
573 const double error=fnorm*gerror + ferror*gnorm + ferror*gerror;
574 // const double error=sqrt(fnorm*fnorm*gnorm*gnorm - sfnorm*sfnorm*sgnorm*sgnorm);
575
576 if (error < thresh) return true;
577 return false;
578 }
579 template <typename Archive> void serialize (Archive& ar) {
580 ar & f & k;
581 }
582 };
583
584 /// returns true if the result of the convolution operator op with some provided
585 /// coefficients will be small
586 template<typename T, size_t NDIM, typename opT>
587 struct op_leaf_op {
589
590 const opT* op; ///< the convolution operator
591 const implT* f; ///< the source or result function, needed for truncate_tol
592 bool do_error_leaf_op() const {return true;}
593
594 op_leaf_op() = default;
595 op_leaf_op(const opT* op, const implT* f) : op(op), f(f) {}
596
597 /// pre-determination: we can't know if this will be a leaf node before we got the final coeffs
598 bool operator()(const Key<NDIM>& key) const {return true;}
599
600 /// post-determination: return true if operator and coefficient norms are small
601 bool operator()(const Key<NDIM>& key, const GenTensor<T>& coeff) const {
602 if (key.level()<2) return false;
603 const double cnorm=coeff.normf();
604 return this->operator()(key,cnorm);
605 }
606
607 /// post-determination: return true if operator and coefficient norms are small
608 bool operator()(const Key<NDIM>& key, const double& cnorm) const {
609 if (key.level()<2) return false;
610
611 typedef Key<opT::opdim> opkeyT;
612 const opkeyT source=op->get_source_key(key);
613
614 const double thresh=f->truncate_tol(f->get_thresh(),key);
615 const std::vector<opkeyT>& disp = op->get_disp(key.level());
616 const opkeyT& d = *disp.begin(); // use the zero-displacement for screening
617 const double opnorm = op->norm(key.level(), d, source);
618 const double norm=opnorm*cnorm;
619 return norm<thresh;
620
621 }
622
623 template <typename Archive> void serialize (Archive& ar) {
624 ar & op & f;
625 }
626
627 };
628
629
630 /// returns true if the result of a hartree_product is a leaf node
631 /// criteria are error, norm and its effect on a convolution operator
632 template<typename T, size_t NDIM, size_t LDIM, typename opT>
634
637
639 const implL* g; // for use of its cdata only
640 const opT* op;
641 bool do_error_leaf_op() const {return false;}
642
644 hartree_convolute_leaf_op(const implT* f, const implL* g, const opT* op)
645 : f(f), g(g), op(op) {}
646
647 /// no pre-determination
648 bool operator()(const Key<NDIM>& key) const {return true;}
649
650 /// no post-determination
651 bool operator()(const Key<NDIM>& key, const GenTensor<T>& coeff) const {
652 MADNESS_EXCEPTION("no post-determination in hartree_convolute_leaf_op",1);
653 return true;
654 }
655
656 /// post-determination: true if f is a leaf and the result is well-represented
657
658 /// @param[in] key the hi-dimensional key (breaks into keys for f and g)
659 /// @param[in] fcoeff coefficients of f of its appropriate key in NS form
660 /// @param[in] gcoeff coefficients of g of its appropriate key in NS form
661 bool operator()(const Key<NDIM>& key, const Tensor<T>& fcoeff, const Tensor<T>& gcoeff) const {
662 // bool operator()(const Key<NDIM>& key, const GenTensor<T>& coeff) const {
663
664 if (key.level()<2) return false;
665
666 const double tol=f->get_thresh();
667 const double thresh=f->truncate_tol(tol, key);
668 // include the wavelets in the norm, makes it much more accurate
669 const double fnorm=fcoeff.normf();
670 const double gnorm=gcoeff.normf();
671
672 // norm of the scaling function coefficients
673 const double sfnorm=fcoeff(g->get_cdata().s0).normf();
674 const double sgnorm=gcoeff(g->get_cdata().s0).normf();
675
676 // if the final norm is small, perform the hartree product and return
677 const double norm=fnorm*gnorm; // computing the outer product
678 if (norm < thresh) return true;
679
680 // get the error of both functions and of the pair function
681 const double ferror=sqrt(fnorm*fnorm-sfnorm*sfnorm);
682 const double gerror=sqrt(gnorm*gnorm-sgnorm*sgnorm);
683
684 // if the expected error is small, perform the hartree product and return
685 const double error=fnorm*gerror + ferror*gnorm + ferror*gerror;
686 if (error < thresh) return true;
687
688 // now check if the norm of this and the norm of the operator are significant
689 const std::vector<Key<NDIM> >& disp = op->get_disp(key.level());
690 const Key<NDIM>& d = *disp.begin(); // use the zero-displacement for screening
691 const double opnorm = op->norm(key.level(), d, key);
692 const double final_norm=opnorm*sfnorm*sgnorm;
693 if (final_norm < thresh) return true;
694
695 return false;
696 }
697 template <typename Archive> void serialize (Archive& ar) {
698 ar & f & op;
699 }
700 };
701
702 template<typename T, size_t NDIM>
703 struct noop {
704 void operator()(const Key<NDIM>& key, const GenTensor<T>& coeff, const bool& is_leaf) const {}
705 bool operator()(const Key<NDIM>& key, const GenTensor<T>& fcoeff, const GenTensor<T>& gcoeff) const {
706 MADNESS_EXCEPTION("in noop::operator()",1);
707 return true;
708 }
709 template <typename Archive> void serialize (Archive& ar) {}
710
711 };
712
713 /// insert/replaces the coefficients into the function
714 template<typename T, std::size_t NDIM>
715 struct insert_op {
720
724 insert_op(const insert_op& other) : impl(other.impl) {}
725 void operator()(const keyT& key, const coeffT& coeff, const bool& is_leaf) const {
727 impl->get_coeffs().replace(key,nodeT(coeff,not is_leaf));
728 }
729 template <typename Archive> void serialize (Archive& ar) {
730 ar & impl;
731 }
732
733 };
734
735 /// inserts/accumulates coefficients into impl's tree
736
737 /// NOTE: will use buffer and will need consolidation after operation ended !! NOTE !!
738 template<typename T, std::size_t NDIM>
742
744 accumulate_op() = default;
746 accumulate_op(const accumulate_op& other) = default;
747 void operator()(const Key<NDIM>& key, const coeffT& coeff, const bool& is_leaf) const {
748 if (coeff.has_data())
749 impl->get_coeffs().task(key, &nodeT::accumulate, coeff, impl->get_coeffs(), key, impl->get_tensor_args());
750 }
751 template <typename Archive> void serialize (Archive& ar) {
752 ar & impl;
753 }
754
755 };
756
757
758template<size_t NDIM>
759 struct true_op {
760
761 template<typename T>
762 bool operator()(const Key<NDIM>& key, const T& t) const {return true;}
763
764 template<typename T, typename R>
765 bool operator()(const Key<NDIM>& key, const T& t, const R& r) const {return true;}
766 template <typename Archive> void serialize (Archive& ar) {}
767
768 };
769
770 /// shallow-copy, pared-down version of FunctionNode, for special purpose only
771 template<typename T, std::size_t NDIM>
772 struct ShallowNode {
776 double dnorm=-1.0;
779 : _coeffs(node.coeff()), _has_children(node.has_children()),
780 dnorm(node.get_dnorm()) {}
782 : _coeffs(node.coeff()), _has_children(node._has_children),
783 dnorm(node.dnorm) {}
784
785 const coeffT& coeff() const {return _coeffs;}
786 coeffT& coeff() {return _coeffs;}
787 bool has_children() const {return _has_children;}
788 bool is_leaf() const {return not _has_children;}
789 template <typename Archive>
790 void serialize(Archive& ar) {
791 ar & coeff() & _has_children & dnorm;
792 }
793 };
794
795
796 /// a class to track where relevant (parent) coeffs are
797
798 /// E.g. if a 6D function is composed of two 3D functions their coefficients must be tracked.
799 /// We might need coeffs from a box that does not exist, and to avoid searching for
800 /// parents we track which are their required respective boxes.
801 /// - CoeffTracker will refer either to a requested key, if it exists, or to its
802 /// outermost parent.
803 /// - Children must be made in sequential order to be able to track correctly.
804 ///
805 /// Usage: 1. make the child of a given CoeffTracker.
806 /// If the parent CoeffTracker refers to a leaf node (flag is_leaf)
807 /// the child will refer to the same node. Otherwise it will refer
808 /// to the child node.
809 /// 2. retrieve its coefficients (possible communication/ returns a Future).
810 /// Member variable key always refers to an existing node,
811 /// so we can fetch it. Once we have the node we can determine
812 /// if it has children which allows us to make a child (see 1. )
813 template<typename T, size_t NDIM>
815
819 typedef std::pair<Key<NDIM>,ShallowNode<T,NDIM> > datumT;
821
822 /// the funcimpl that has the coeffs
823 const implT* impl;
824 /// the current key, which must exists in impl
826 /// flag if key is a leaf node
828 /// the coefficients belonging to key
830 /// norm of d coefficients corresponding to key
831 double dnorm_=-1.0;
832
833 public:
834
835 /// default ctor
836 CoeffTracker() : impl(), key_(0), is_leaf_(unknown), coeff_() {} // Initialize key to avoid warnings of possible unititialied use
837
838 /// the initial ctor making the root key
840 if (impl) key_=impl->get_cdata().key0;
841 }
842
843 /// ctor with a pair<keyT,nodeT>
844 explicit CoeffTracker(const CoeffTracker& other, const datumT& datum)
845 : impl(other.impl), key_(other.key_), coeff_(datum.second.coeff()),
846 dnorm_(datum.second.dnorm) {
847 if (datum.second.is_leaf()) is_leaf_=yes;
848 else is_leaf_=no;
849 }
850
851 /// copy ctor
852 CoeffTracker(const CoeffTracker& other) : impl(other.impl), key_(other.key_),
853 is_leaf_(other.is_leaf_), coeff_(other.coeff_), dnorm_(other.dnorm_) {};
854
855 CoeffTracker& operator=(const CoeffTracker& other) = default;
856
857 /// const reference to impl
858 const implT* get_impl() const {return impl;}
859
860 /// const reference to the coeffs
861 const coeffT& coeff() const {return coeff_;}
862
863 /// const reference to the key
864 const keyT& key() const {return key_;}
865
866 /// return the coefficients belonging to the passed-in key
867
868 /// if key equals tracked key just return the coeffs, otherwise
869 /// make the child coefficients.
870 /// @param[in] key return coeffs corresponding to this key
871 /// @return coefficients belonging to key
879
880 /// return the s and dnorm belonging to the passed-in key
881 double dnorm(const keyT& key) const {
882 if (key==key_) return dnorm_;
883 MADNESS_ASSERT(key.is_child_of(key_));
884 return 0.0;
885 }
886
887 /// const reference to is_leaf flag
888 const LeafStatus& is_leaf() const {return is_leaf_;}
889
890 /// make a child of this, ignoring the coeffs
891 CoeffTracker make_child(const keyT& child) const {
892
893 // fast return
894 if ((not impl) or impl->is_on_demand()) return CoeffTracker(*this);
895
896 // can't make a child without knowing if this is a leaf -- activate first
898
899 CoeffTracker result;
900 if (impl) {
901 result.impl=impl;
902 if (is_leaf_==yes) result.key_=key_;
903 if (is_leaf_==no) {
904 result.key_=child;
905 // check if child is direct descendent of this, but root node is special case
906 if (child.level()>0) MADNESS_ASSERT(result.key().level()==key().level()+1);
907 }
908 result.is_leaf_=unknown;
909 }
910 return result;
911 }
912
913 /// find the coefficients
914
915 /// this involves communication to a remote node
916 /// @return a Future<CoeffTracker> with the coefficients that key refers to
918
919 // fast return
920 if (not impl) return Future<CoeffTracker>(CoeffTracker());
922
923 // this will return a <keyT,nodeT> from a remote node
926
927 // construct a new CoeffTracker locally
928 return impl->world.taskq.add(*const_cast<CoeffTracker*> (this),
929 &CoeffTracker::forward_ctor,*this,datum1);
930 }
931
932 private:
933 /// taskq-compatible forwarding to the ctor
934 CoeffTracker forward_ctor(const CoeffTracker& other, const datumT& datum) const {
935 return CoeffTracker(other,datum);
936 }
937
938 public:
939 /// serialization
940 template <typename Archive> void serialize(const Archive& ar) {
941 int il=int(is_leaf_);
942 ar & impl & key_ & il & coeff_ & dnorm_;
944 }
945 };
946
947 template<typename T, std::size_t NDIM>
948 std::ostream&
949 operator<<(std::ostream& s, const CoeffTracker<T,NDIM>& ct) {
950 s << ct.key() << ct.is_leaf() << " " << ct.get_impl();
951 return s;
952 }
953
954 /// FunctionImpl holds all Function state to facilitate shallow copy semantics
955
956 /// Since Function assignment and copy constructors are shallow it
957 /// greatly simplifies maintaining consistent state to have all
958 /// (permanent) state encapsulated in a single class. The state
959 /// is shared between instances using a shared_ptr<FunctionImpl>.
960 ///
961 /// The FunctionImpl inherits all of the functionality of WorldContainer
962 /// (to store the coefficients) and WorldObject<WorldContainer> (used
963 /// for RMI and for its unqiue id).
964 ///
965 /// The class methods are public to avoid painful multiple friend template
966 /// declarations for Function and FunctionImpl ... but this trust should not be
967 /// abused ... NOTHING except FunctionImpl methods should mess with FunctionImplData.
968 /// The LB stuff might have to be an exception.
969 template <typename T, std::size_t NDIM>
970 class FunctionImpl : public WorldObject< FunctionImpl<T,NDIM> > {
971 private:
972 typedef WorldObject< FunctionImpl<T,NDIM> > woT; ///< Base class world object type
973 public:
974 typedef T typeT;
975 typedef FunctionImpl<T,NDIM> implT; ///< Type of this class (implementation)
976 typedef std::shared_ptr< FunctionImpl<T,NDIM> > pimplT; ///< pointer to this class
977 typedef Tensor<T> tensorT; ///< Type of tensor for anything but to hold coeffs
978 typedef Vector<Translation,NDIM> tranT; ///< Type of array holding translation
979 typedef Key<NDIM> keyT; ///< Type of key
980 typedef FunctionNode<T,NDIM> nodeT; ///< Type of node
981 typedef GenTensor<T> coeffT; ///< Type of tensor used to hold coeffs
982 typedef WorldContainer<keyT,nodeT> dcT; ///< Type of container holding the coefficients
983 typedef std::pair<const keyT,nodeT> datumT; ///< Type of entry in container
984 typedef Vector<double,NDIM> coordT; ///< Type of vector holding coordinates
985
986 //template <typename Q, int D> friend class Function;
987 template <typename Q, std::size_t D> friend class FunctionImpl;
988
990
991 /// getter
994 const std::vector<Vector<double,NDIM> >& get_special_points()const{return special_points;}
995
996 private:
997 int k; ///< Wavelet order
998 double thresh; ///< Screening threshold
999 int initial_level; ///< Initial level for refinement
1000 int special_level; ///< Minimium level for refinement on special points
1001 std::vector<Vector<double,NDIM> > special_points; ///< special points for further refinement (needed for composite functions or multiplication)
1002 const Tensor<double> cell; ///< the size of the root cell in each dimension, unchangeable
1003 int max_refine_level; ///< Do not refine below this level
1004 int truncate_mode; ///< 0=default=(|d|<thresh), 1=(|d|<thresh/2^n), 2=(|d|<thresh/4^n);
1005 bool autorefine; ///< If true, autorefine where appropriate
1006 bool truncate_on_project; ///< If true projection inserts at level n-1 not n
1007 TensorArgs targs; ///< type of tensor to be used in the FunctionNodes
1008
1010
1011 std::shared_ptr< FunctionFunctorInterface<T,NDIM> > functor;
1013
1014 dcT coeffs; ///< The coefficients
1015
1016 /// Neighbor coefficients pushed here by whoever owns them; null until something stages.
1017
1018 /// Values mirror what `sock_it_to_me` returns for the same key: coefficients for a same-level
1019 /// leaf, empty for an interior node, absent when the neighbor is coarser and the consumer
1020 /// must walk up. Which nodes get pushed is the operator's business, not the table's --- see
1021 /// `DerivativeBase::stage_halo`.
1022 ///
1023 /// Allocated on the first push, so functions that never stage one do not carry it: a
1024 /// `ConcurrentHashMap` default-constructs 1021 bins, ~32 kB per function.
1025 mutable std::atomic<ConcurrentHashMap<keyT,coeffT>*> neighbor_halo_{nullptr};
1026
1027 // Disable the default copy constructor
1029
1030 public:
1031 /// Is a neighbor halo staged on this function?
1032 bool halo_enabled() const {
1033 return neighbor_halo_.load(std::memory_order_acquire) != nullptr;
1034 }
1035
1036 /// Discard the neighbor halo, freeing the staged coefficients.
1037
1038 /// Requires a quiescent window: it frees a table that `halo_probe` may be reading.
1039 void halo_clear() const {
1040 delete neighbor_halo_.exchange(nullptr, std::memory_order_acq_rel);
1041 }
1042
1043 /// How many neighbor nodes are staged on this rank; zero if no halo.
1044 std::size_t halo_size() const {
1045 const auto* h = neighbor_halo_.load(std::memory_order_acquire);
1046 return h ? h->size() : 0;
1047 }
1048
1049 /// Insert pushed neighbor nodes into the halo; runs as a task, concurrently with other pushes.
1050
1051 /// Allocates the table on the first push, so staging needs no collective set-up.
1052 void receive_halo(const std::vector<std::pair<keyT,coeffT> >& buf) const {
1053 auto* h = neighbor_halo_.load(std::memory_order_acquire);
1054 if (!h) {
1055 auto* fresh = new ConcurrentHashMap<keyT,coeffT>();
1056 if (neighbor_halo_.compare_exchange_strong(h, fresh, std::memory_order_acq_rel,
1057 std::memory_order_acquire))
1058 h = fresh;
1059 else
1060 delete fresh; // lost the race; the failed CAS put the winner's table in h
1061 }
1062 for (const auto& kv : buf) {
1064 (void) h->insert(acc, kv.first);
1065 acc->second = kv.second;
1066 }
1067 }
1068
1069 /// Look up a staged neighbor; on a hit copy its coefficients, which are empty for an interior node.
1070 bool halo_probe(const keyT& key, coeffT& out) const {
1071 const auto* h = neighbor_halo_.load(std::memory_order_acquire);
1072 if (!h) return false;
1074 if (h->find(acc, key)) { out = acc->second; return true; }
1075 return false;
1076 }
1077
1086
1087 /// Initialize function impl from data in factory
1089 : WorldObject<implT>(factory._world)
1090 , world(factory._world)
1091 , k(factory._k)
1092 , thresh(factory._thresh)
1093 , initial_level(factory._initial_level)
1094 , special_level(factory._special_level)
1095 , special_points(factory._special_points)
1097 , max_refine_level(factory._max_refine_level)
1098 , truncate_mode(factory._truncate_mode)
1099 , autorefine(factory._autorefine)
1100 , truncate_on_project(factory._truncate_on_project)
1101 , targs(factory._thresh,FunctionDefaults<NDIM>::get_tensor_type())
1102 , cdata(FunctionCommonData<T,NDIM>::get(k))
1103 , functor(factory.get_functor())
1104 , tree_state(factory._tree_state)
1105 , coeffs(world,factory._pmap,false)
1106 //, bc(factory._bc)
1107 {
1108 // PROFILE_MEMBER_FUNC(FunctionImpl); // No need to profile this
1109 // !!! Ensure that all local state is correctly formed
1110 // before invoking process_pending for the coeffs and
1111 // for this. Otherwise, there is a race condition.
1112 MADNESS_ASSERT(k>0 && k<=MAXK);
1113
1114 bool empty = (factory._empty or is_on_demand());
1115 bool do_refine = factory._refine;
1116
1117 if (do_refine)
1118 initial_level = std::max(0,initial_level - 1);
1119
1120 if (empty) { // Do not set any coefficients at all
1121 // additional functors are only evaluated on-demand
1122 } else if (functor) { // Project function and optionally refine
1124 // set the union of the special points of functor and the ones explicitly given to FunctionFactory
1125 std::vector<coordT> functor_special_points=functor->special_points();
1126 if (!functor_special_points.empty()) special_points.insert(special_points.end(), functor_special_points.begin(), functor_special_points.end());
1127 // near special points refine as deeply as requested by the factory AND the functor
1128 special_level = std::max(special_level, functor->special_level());
1129
1130 typename dcT::const_iterator end = coeffs.end();
1131 for (typename dcT::const_iterator it=coeffs.begin(); it!=end; ++it) {
1132 if (it->second.is_leaf())
1133 woT::task(coeffs.owner(it->first), &implT::project_refine_op, it->first, do_refine,
1135 }
1136 }
1137 else { // Set as if a zero function
1138 initial_level = 1;
1140 }
1141
1143 this->process_pending();
1144 if (factory._fence && (functor || !empty)) world.gop.fence();
1145 }
1146
1147 /// Copy constructor
1148
1149 /// Allocates a \em new function in preparation for a deep copy
1150 ///
1151 /// By default takes pmap from other but can also specify a different pmap.
1152 /// Does \em not copy the coefficients ... creates an empty container.
1153 template <typename Q>
1155 const std::shared_ptr< WorldDCPmapInterface< Key<NDIM> > >& pmap,
1156 bool dozero) : FunctionImpl(other.world, other, pmap, dozero) {
1157 }
1158
1159 /// Copy constructor
1160
1161 /// Allocates a \em new function in preparation for a deep copy
1162 ///
1163 /// By default takes pmap from other but can also specify a different pmap.
1164 /// Does \em not copy the coefficients ... creates an empty container.
1165 ///
1166 /// uses a different world for the new function
1167 template <typename Q>
1169 const FunctionImpl<Q,NDIM>& other,
1170 const std::shared_ptr< WorldDCPmapInterface< Key<NDIM> > >& pmap,
1171 bool dozero)
1173 , world(world)
1174 , k(other.k)
1175 , thresh(other.thresh)
1179 , cell(other.cell)
1182 , autorefine(other.autorefine)
1184 , targs(other.targs)
1185 , cdata(FunctionCommonData<T,NDIM>::get(k))
1186 , functor()
1187 , tree_state(other.tree_state)
1188 , coeffs(world, pmap ? pmap : other.coeffs.get_pmap())
1189 {
1190 if (dozero) {
1191 initial_level = 1;
1193 //world.gop.fence(); <<<<<<<<<<<<<<<<<<<<<< needs a fence argument
1194 }
1196 this->process_pending();
1197 }
1198
1199 virtual ~FunctionImpl() { halo_clear(); }
1200
1201 const std::shared_ptr< WorldDCPmapInterface< Key<NDIM> > >& get_pmap() const;
1202
1203 void replicate(bool fence=true) {
1204 coeffs.replicate(fence);
1205 }
1206
1207 void replicate_on_hosts(bool fence=true) {
1209 }
1210
1211 // remove all coeffs that are not local according to pmap
1212 void undo_replicate(bool fence=true) {
1213 std::list<keyT> keys;
1214 for (const auto& [key, node] : coeffs) if (not coeffs.is_local(key)) keys.push_back(key);
1215 for (const auto& key : keys) coeffs.erase(key);
1216 if (fence) world.gop.fence();
1217 }
1218
1219 void distribute(std::shared_ptr< WorldDCPmapInterface< Key<NDIM> > > newmap) const {
1220 auto currentmap=coeffs.get_pmap();
1221 currentmap->redistribute(world,newmap);
1222 }
1223
1224 /// Copy coeffs from other into self
1225
1226 /// this and other might live in different worlds
1227 template <typename Q>
1228 void copy_coeffs(const FunctionImpl<Q,NDIM>& other, bool fence) {
1229 if (world.id()==other.world.id())
1230 copy_coeffs_same_world(other,false);
1231 else
1233 if (fence) world.gop.fence();
1234 }
1235
1236 /// Copy coefficients from other funcimpl with possibly different world and on a different node
1237 template<typename Q>
1239
1240 // copy coeffs from (a subset of) other's world
1241
1242 // single-owner pmap: fetch from that rank only -- the all-ranks poll costs
1243 // N-1 empty round-trips queued behind the owner's compute
1244 const ProcessID single_owner = other.get_pmap()->single_owner();
1245 if (single_owner >= 0) {
1246 copy_remote_coeffs_from_pid<Q>(single_owner, other);
1247
1248 // if other's data is distributed, we need to fetch from all ranks
1249 } else if (other.get_coeffs().is_distributed()) {
1250 for (ProcessID pid=0; pid<other.world.size(); ++pid) {
1251 copy_remote_coeffs_from_pid<Q>(pid, other);
1252 }
1253
1254 // if other's data is replicated, all coeffs are on the rank that owns key0
1255 } else if (other.get_coeffs().is_replicated() or other.get_coeffs().is_host_replicated()) {
1256 auto key0=other.cdata.key0;
1257 copy_remote_coeffs_from_pid<Q>(other.get_pmap()->owner(key0), other);
1258 }
1259 }
1260
1261 /// Copy coefficients from other funcimpl with possibly different world and on a different node
1262 /// to this
1263 template <typename Q>
1265 typedef FunctionImpl<Q,NDIM> implQ; ///< Type of this class (implementation)
1266 if (pid == other.world.rank()) {
1267 // the shard is local: copy nodes directly, no serialize round-trip
1268 copy_coeffs_same_world(other, false);
1269 } else {
1270 auto v=other.task(pid, &implQ::serialize_remote_coeffs);
1272 }
1273 }
1274
1275 /// invoked by copy_remote_coeffs_from_pid to serialize *local* coeffs
1276 std::vector<unsigned char> serialize_remote_coeffs() {
1277 std::vector<unsigned char> v;
1279 ar & get_coeffs();
1280 return v;
1281 }
1282
1283 /// insert coeffs from vector archive into this
1284 void insert_serialized_coeffs(std::vector<unsigned char>& v) {
1286 ar & get_coeffs();
1287 }
1288
1289 /// Copy coeffs from other into self
1290 template <typename Q>
1291 void copy_coeffs_same_world(const FunctionImpl<Q,NDIM>& other, bool fence) {
1292 for (const auto& [key, node] : other.coeffs) { // iterate over all entries in other
1293 coeffs.replace(key,node. template convert<T>());
1294 }
1295 if (fence)
1296 world.gop.fence();
1297 }
1298
1299 /// perform inplace gaxpy: this = alpha*this + beta*other
1300 /// @param[in] alpha prefactor for this
1301 /// @param[in] beta prefactor for other
1302 /// @param[in] g the other function, reconstructed
1303 /// @return *this = alpha*this + beta*other, in either reconstructed or redundant_after_merge state
1304 template<typename Q, typename R>
1305 void gaxpy_inplace_reconstructed(const T& alpha, const FunctionImpl<Q,NDIM>& g, const R& beta, const bool fence) {
1306 // merge g's tree into this' tree
1307 gaxpy_inplace(alpha,g,beta,fence);
1309 // this->merge_trees(beta,g,alpha,fence);
1310 // tree is now redundant_after_merge
1311 // sum down the sum coeffs into the leafs if possible to keep the state most clean
1312 if (fence) sum_down(fence);
1313 }
1314
1315 /// merge the trees of this and other, while multiplying them with the alpha or beta, resp
1316
1317 /// first step in an inplace gaxpy operation for reconstructed functions; assuming the same
1318 /// distribution for this and other
1319
1320 /// on output, *this = alpha* *this + beta * other
1321 /// @param[in] alpha prefactor for this
1322 /// @param[in] beta prefactor for other
1323 /// @param[in] other the other function, reconstructed
1324 template<typename Q, typename R>
1325 void merge_trees(const T alpha, const FunctionImpl<Q,NDIM>& other, const R beta, const bool fence=true) {
1326 MADNESS_ASSERT(get_pmap() == other.get_pmap());
1329 }
1330
1331 /// merge the trees of this and other, while multiplying them with the alpha or beta, resp
1332
1333 /// result and rhs do not have to have the same distribution or live in the same world
1334 /// result+=alpha* this
1335 /// @param[in] alpha prefactor for this
1336 template<typename Q, typename R>
1337 void accumulate_trees(FunctionImpl<Q,NDIM>& result, const R alpha, const bool fence=true) const {
1339 }
1340
1341 /// perform: this= alpha*f + beta*g, invoked by result
1342
1343 /// f and g are reconstructed, so we can save on the compress operation,
1344 /// walk down the joint tree, and add leaf coefficients; effectively refines
1345 /// to common finest level.
1346
1347 /// nothing returned, but leaves this's tree reconstructed and as sum of f and g
1348 /// @param[in] alpha prefactor for f
1349 /// @param[in] f first addend
1350 /// @param[in] beta prefactor for g
1351 /// @param[in] g second addend
1352 void gaxpy_oop_reconstructed(const double alpha, const implT& f,
1353 const double beta, const implT& g, const bool fence);
1354
1355 /// functor for the gaxpy_inplace method
1356 template <typename Q, typename R>
1359 FunctionImpl<T,NDIM>* f; ///< prefactor for current function impl
1360 T alpha; ///< the current function impl
1361 R beta; ///< prefactor for other function impl
1362 do_gaxpy_inplace() = default;
1364 bool operator()(typename rangeT::iterator& it) const {
1365 const keyT& key = it->first;
1366 const FunctionNode<Q,NDIM>& other_node = it->second;
1367 // Use send to get write accessor and automated construction if missing
1368 f->coeffs.send(key, &nodeT:: template gaxpy_inplace<Q,R>, alpha, other_node, beta);
1369 return true;
1370 }
1371 template <typename Archive>
1372 void serialize(Archive& ar) {
1373 ar & f & alpha & beta;
1374 }
1375 };
1376
1377 /// Inplace general bilinear operation
1378
1379 /// this's world can differ from other's world
1380 /// this = alpha * this + beta * other
1381 /// @param[in] alpha prefactor for the current function impl
1382 /// @param[in] other the other function impl
1383 /// @param[in] beta prefactor for other
1384 template <typename Q, typename R>
1385 void gaxpy_inplace(const T& alpha,const FunctionImpl<Q,NDIM>& other, const R& beta, bool fence) {
1386// MADNESS_ASSERT(get_pmap() == other.get_pmap());
1387 if (alpha != T(1.0)) scale_inplace(alpha,false);
1389 typedef do_gaxpy_inplace<Q,R> opT;
1390 other.world.taskq. template for_each<rangeT,opT>(rangeT(other.coeffs.begin(), other.coeffs.end()), opT(this, T(1.0), beta));
1391 if (fence)
1392 other.world.gop.fence();
1393 }
1394
1395 // loads a function impl from persistence
1396 // @param[in] ar the archive where the function impl is stored
1397 template <typename Archive>
1398 void load(Archive& ar) {
1399 // WE RELY ON K BEING STORED FIRST
1400 int kk = 0;
1401 ar & kk;
1402
1403 MADNESS_ASSERT(kk==k);
1404
1405 // note that functor should not be (re)stored
1407 & autorefine & truncate_on_project & tree_state;//nonstandard & compressed ; //& bc;
1408
1409 ar & coeffs;
1410 world.gop.fence();
1411 }
1412
1413 // saves a function impl to persistence
1414 // @param[in] ar the archive where the function impl is to be stored
1415 template <typename Archive>
1416 void store(Archive& ar) {
1417 // WE RELY ON K BEING STORED FIRST
1418
1419 // note that functor should not be (re)stored
1421 & autorefine & truncate_on_project & tree_state;//nonstandard & compressed ; //& bc;
1422
1423 ar & coeffs;
1424 world.gop.fence();
1425 }
1426
1427 /// Returns true if the function is compressed.
1428 bool is_compressed() const;
1429
1430 /// Returns true if the function is compressed.
1431 bool is_reconstructed() const;
1432
1433 /// Returns true if the function is redundant.
1434 bool is_redundant() const;
1435
1436 /// Returns true if the function is redundant_after_merge.
1437 bool is_redundant_after_merge() const;
1438
1439 bool is_nonstandard() const;
1440
1441 bool is_nonstandard_with_leaves() const;
1442
1443 bool is_on_demand() const;
1444
1445 /// Returns true if only the leaves of this tree carry its coefficients
1446
1447 /// redundant and nonstandard_with_leaves keep s coefficients on the
1448 /// internal nodes as well, so a sum over all nodes counts the function
1449 /// more than once; the leaves alone are exactly the reconstructed tree,
1450 /// which is why change_tree_state(reconstructed) is nothing but
1451 /// remove_internal_coefficients() for these two states.
1453
1454 /// Returns true if summing over the local nodes yields the function
1455
1456 /// The precondition of norm2sq_local() and trace_local(): the tree holds
1457 /// its coefficients exactly once. reconstructed and compressed do so
1458 /// outright, the two states above once the internal nodes are skipped.
1459 /// redundant_after_merge is a sum that has not been collapsed yet, and
1460 /// nonstandard / nonstandard_after_apply have no leaf coefficients to
1461 /// single out, so neither qualifies.
1462 bool has_summable_coefficients() const;
1463
1464 bool has_leaves() const;
1465
1466 void set_tree_state(const TreeState& state) {
1467 tree_state=state;
1468 }
1469
1471
1472 void set_functor(const std::shared_ptr<FunctionFunctorInterface<T,NDIM> > functor1);
1473
1474 std::shared_ptr<FunctionFunctorInterface<T,NDIM> > get_functor();
1475
1476 std::shared_ptr<FunctionFunctorInterface<T,NDIM> > get_functor() const;
1477
1478 void unset_functor();
1479
1480
1482
1484 void set_tensor_args(const TensorArgs& t);
1485
1486 double get_thresh() const;
1487
1488 /// return the simulation cell
1489 const Tensor<double>& get_cell() const { return cell; }
1490
1491 void set_thresh(double value);
1492
1493 bool get_autorefine() const;
1494
1495 void set_autorefine(bool value);
1496
1497 int get_k() const;
1498
1499 const dcT& get_coeffs() const;
1500
1501 dcT& get_coeffs();
1502
1504
1505 void accumulate_timer(const double time) const; // !!!!!!!!!!!! REDUNDANT !!!!!!!!!!!!!!!
1506
1507 void print_timer() const;
1508
1509 void reset_timer();
1510
1511 /// Adds a constant to the function. Local operation, optional fence
1512
1513 /// In scaling function basis must add value to first polyn in
1514 /// each box with appropriate scaling for level. In wavelet basis
1515 /// need only add at level zero.
1516 /// @param[in] t the scalar to be added
1517 void add_scalar_inplace(T t, bool fence);
1518
1519 /// Initialize nodes to zero function at initial_level of refinement.
1520
1521 /// Works for either basis. No communication.
1522 void insert_zero_down_to_initial_level(const keyT& key);
1523
1524 /// Truncate according to the threshold with optional global fence
1525
1526 /// If thresh<=0 the default value of this->thresh is used
1527 /// @param[in] tol the truncation tolerance
1528 void truncate(double tol, bool fence);
1529
1530 /// Returns true if after truncation this node has coefficients
1531
1532 /// Assumed to be invoked on process owning key. Possible non-blocking
1533 /// communication.
1534 /// @param[in] key the key of the current function node
1535 Future<bool> truncate_spawn(const keyT& key, double tol);
1536
1537 /// Actually do the truncate operation
1538 /// @param[in] key the key to the current function node being evaluated for truncation
1539 /// @param[in] tol the tolerance for thresholding
1540 /// @param[in] v vector of Future<bool>'s that specify whether the current nodes children have coeffs
1541 bool truncate_op(const keyT& key, double tol, const std::vector< Future<bool> >& v);
1542
1543 /// Evaluate function at quadrature points in the specified box
1544
1545 /// @param[in] key the key indicating where the quadrature points are located
1546 /// @param[in] f the interface to the elementary function
1547 /// @param[in] qx quadrature points on a level=0 box
1548 /// @param[out] fval values
1549 void fcube(const keyT& key, const FunctionFunctorInterface<T,NDIM>& f, const Tensor<double>& qx, tensorT& fval) const;
1550
1551 /// Evaluate function at quadrature points in the specified box
1552
1553 /// @param[in] key the key indicating where the quadrature points are located
1554 /// @param[in] f the interface to the elementary function
1555 /// @param[in] qx quadrature points on a level=0 box
1556 /// @param[out] fval values
1557 void fcube(const keyT& key, T (*f)(const coordT&), const Tensor<double>& qx, tensorT& fval) const;
1558
1559 /// Returns cdata.key0
1560 const keyT& key0() const;
1561
1562 /// Prints the coeffs tree of the current function impl
1563 /// @param[in] maxlevel the maximum level of the tree for printing
1564 /// @param[out] os the ostream to where the output is sent
1565 void print_tree(std::ostream& os = std::cout, Level maxlevel = 10000) const;
1566
1567 /// Functor for the do_print_tree method
1568 void do_print_tree(const keyT& key, std::ostream& os, Level maxlevel) const;
1569
1570 /// Prints the coeffs tree of the current function impl (using GraphViz)
1571 /// @param[in] maxlevel the maximum level of the tree for printing
1572 /// @param[out] os the ostream to where the output is sent
1573 void print_tree_graphviz(std::ostream& os = std::cout, Level maxlevel = 10000) const;
1574
1575 /// Functor for the do_print_tree method (using GraphViz)
1576 void do_print_tree_graphviz(const keyT& key, std::ostream& os, Level maxlevel) const;
1577
1578 /// Same as print_tree() but in JSON format
1579 /// @param[out] os the ostream to where the output is sent
1580 /// @param[in] maxlevel the maximum level of the tree for printing
1581 void print_tree_json(std::ostream& os = std::cout, Level maxlevel = 10000) const;
1582
1583 /// Functor for the do_print_tree_json method
1584 void do_print_tree_json(const keyT& key, std::multimap<Level, std::tuple<tranT, std::string>>& data, Level maxlevel) const;
1585
1586 /// convert a number [0,limit] to a hue color code [blue,red],
1587 /// or, if log is set, a number [1.e-10,limit]
1589 double limit;
1590 bool log;
1591 static double lower() {return 1.e-10;};
1593 do_convert_to_color(const double limit, const bool log) : limit(limit), log(log) {}
1594 double operator()(double val) const {
1595 double color=0.0;
1596
1597 if (log) {
1598 double val2=log10(val) - log10(lower()); // will yield >0.0
1599 double upper=log10(limit) -log10(lower());
1600 val2=0.7-(0.7/upper)*val2;
1601 color= std::max(0.0,val2);
1602 color= std::min(0.7,color);
1603 } else {
1604 double hue=0.7-(0.7/limit)*(val);
1605 color= std::max(0.0,hue);
1606 }
1607 return color;
1608 }
1609 };
1610
1611
1612 /// Print a plane ("xy", "xz", or "yz") containing the point x to file
1613
1614 /// works for all dimensions; we walk through the tree, and if a leaf node
1615 /// inside the sub-cell touches the plane we print it in pstricks format
1616 void print_plane(const std::string filename, const int xaxis, const int yaxis, const coordT& el2);
1617
1618 /// collect the data for a plot of the MRA structure locally on each node
1619
1620 /// @param[in] xaxis the x-axis in the plot (can be any axis of the MRA box)
1621 /// @param[in] yaxis the y-axis in the plot (can be any axis of the MRA box)
1622 /// @param[in] el2 needs a description
1623 /// \todo Provide a description for el2
1624 Tensor<double> print_plane_local(const int xaxis, const int yaxis, const coordT& el2);
1625
1626 /// Functor for the print_plane method
1627 /// @param[in] filename the filename for the output
1628 /// @param[in] plotinfo plotting parameters
1629 /// @param[in] xaxis the x-axis in the plot (can be any axis of the MRA box)
1630 /// @param[in] yaxis the y-axis in the plot (can be any axis of the MRA box)
1631 void do_print_plane(const std::string filename, std::vector<Tensor<double> > plotinfo,
1632 const int xaxis, const int yaxis, const coordT el2);
1633
1634 /// print the grid (the roots of the quadrature of each leaf box)
1635 /// of this function in user xyz coordinates
1636 /// @param[in] filename the filename for the output
1637 void print_grid(const std::string filename) const;
1638
1639 /// return the keys of the local leaf boxes
1640 std::vector<keyT> local_leaf_keys() const;
1641
1642 /// print the grid in xyz format
1643
1644 /// the quadrature points and the key information will be written to file,
1645 /// @param[in] filename where the quadrature points will be written to
1646 /// @param[in] keys all leaf keys
1647 void do_print_grid(const std::string filename, const std::vector<keyT>& keys) const;
1648
1649 /// read data from a grid
1650
1651 /// @param[in] keyfile file with keys and grid points for each key
1652 /// @param[in] gridfile file with grid points, w/o key, but with same ordering
1653 /// @param[in] vnuc_functor subtract the values of this functor if regularization is needed
1654 template<size_t FDIM>
1655 typename std::enable_if<NDIM==FDIM>::type
1656 read_grid(const std::string keyfile, const std::string gridfile,
1657 std::shared_ptr< FunctionFunctorInterface<double,NDIM> > vnuc_functor) {
1658
1659 std::ifstream kfile(keyfile.c_str());
1660 std::ifstream gfile(gridfile.c_str());
1661 std::string line;
1662
1663 long ndata,ndata1;
1664 if (not (std::getline(kfile,line))) MADNESS_EXCEPTION("failed reading 1st line of key data",0);
1665 if (not (std::istringstream(line) >> ndata)) MADNESS_EXCEPTION("failed reading k",0);
1666 if (not (std::getline(gfile,line))) MADNESS_EXCEPTION("failed reading 1st line of grid data",0);
1667 if (not (std::istringstream(line) >> ndata1)) MADNESS_EXCEPTION("failed reading k",0);
1668 MADNESS_CHECK(ndata==ndata1);
1669 if (not (std::getline(kfile,line))) MADNESS_EXCEPTION("failed reading 2nd line of key data",0);
1670 if (not (std::getline(gfile,line))) MADNESS_EXCEPTION("failed reading 2nd line of grid data",0);
1671
1672 // the quadrature points in simulation coordinates of the root node
1673 const Tensor<double> qx=cdata.quad_x;
1674 const size_t npt = qx.dim(0);
1675
1676 // the number of coordinates (grid point tuples) per box ({x1},{x2},{x3},..,{xNDIM})
1677 long npoints=power<NDIM>(npt);
1678 // the number of boxes
1679 long nboxes=ndata/npoints;
1680 MADNESS_ASSERT(nboxes*npoints==ndata);
1681 print("reading ",nboxes,"boxes from file",gridfile,keyfile);
1682
1683 // these will be the data
1684 Tensor<T> values(cdata.vk,false);
1685
1686 int ii=0;
1687 std::string gline,kline;
1688 // while (1) {
1689 while (std::getline(kfile,kline)) {
1690
1691 double x,y,z,x1,y1,z1,val;
1692
1693 // get the key
1694 long nn;
1695 Translation l1,l2,l3;
1696 // line looks like: # key: n l1 l2 l3
1697 kline.erase(0,7);
1698 std::stringstream(kline) >> nn >> l1 >> l2 >> l3;
1699 // kfile >> s >> nn >> l1 >> l2 >> l3;
1700 const Vector<Translation,3> ll{ l1,l2,l3 };
1701 Key<3> key(nn,ll);
1702
1703 // this is borrowed from fcube
1704 const Vector<Translation,3>& l = key.translation();
1705 const Level n = key.level();
1706 const double h = std::pow(0.5,double(n));
1707 coordT c; // will hold the point in user coordinates
1710
1711
1712 if (NDIM == 3) {
1713 for (size_t i=0; i<npt; ++i) {
1714 c[0] = cell(0,0) + h*cell_width[0]*(l[0] + qx(i)); // x
1715 for (size_t j=0; j<npt; ++j) {
1716 c[1] = cell(1,0) + h*cell_width[1]*(l[1] + qx(j)); // y
1717 for (size_t k=0; k<npt; ++k) {
1718 c[2] = cell(2,0) + h*cell_width[2]*(l[2] + qx(k)); // z
1719 // fprintf(pFile,"%18.12f %18.12f %18.12f\n",c[0],c[1],c[2]);
1720 auto& success1 = std::getline(gfile,gline); MADNESS_CHECK(success1);
1721 auto& success2 = std::getline(kfile,kline); MADNESS_CHECK(success2);
1722 std::istringstream(gline) >> x >> y >> z >> val;
1723 std::istringstream(kline) >> x1 >> y1 >> z1;
1724 MADNESS_CHECK(std::fabs(x-c[0])<1.e-4);
1725 MADNESS_CHECK(std::fabs(x1-c[0])<1.e-4);
1726 MADNESS_CHECK(std::fabs(y-c[1])<1.e-4);
1727 MADNESS_CHECK(std::fabs(y1-c[1])<1.e-4);
1728 MADNESS_CHECK(std::fabs(z-c[2])<1.e-4);
1729 MADNESS_CHECK(std::fabs(z1-c[2])<1.e-4);
1730
1731 // regularize if a functor is given
1732 if (vnuc_functor) val-=(*vnuc_functor)(c);
1733 values(i,j,k)=val;
1734 }
1735 }
1736 }
1737 } else {
1738 MADNESS_EXCEPTION("only NDIM=3 in print_grid",0);
1739 }
1740
1741 // insert the new leaf node
1742 const bool has_children=false;
1743 coeffT coeff=coeffT(this->values2coeffs(key,values),targs);
1744 nodeT node(coeff,has_children);
1745 coeffs.replace(key,node);
1747 ii++;
1748 }
1749
1750 kfile.close();
1751 gfile.close();
1752 MADNESS_CHECK(ii==nboxes);
1753
1754 }
1755
1756
1757 /// read data from a grid
1758
1759 /// @param[in] gridfile file with keys and grid points and values for each key
1760 /// @param[in] vnuc_functor subtract the values of this functor if regularization is needed
1761 template<size_t FDIM>
1762 typename std::enable_if<NDIM==FDIM>::type
1763 read_grid2(const std::string gridfile,
1764 std::shared_ptr< FunctionFunctorInterface<double,NDIM> > vnuc_functor) {
1765
1766 std::ifstream gfile(gridfile.c_str());
1767 std::string line;
1768
1769 long ndata;
1770 if (not (std::getline(gfile,line))) MADNESS_EXCEPTION("failed reading 1st line of grid data",0);
1771 if (not (std::istringstream(line) >> ndata)) MADNESS_EXCEPTION("failed reading k",0);
1772 if (not (std::getline(gfile,line))) MADNESS_EXCEPTION("failed reading 2nd line of grid data",0);
1773
1774 // the quadrature points in simulation coordinates of the root node
1775 const Tensor<double> qx=cdata.quad_x;
1776 const size_t npt = qx.dim(0);
1777
1778 // the number of coordinates (grid point tuples) per box ({x1},{x2},{x3},..,{xNDIM})
1779 long npoints=power<NDIM>(npt);
1780 // the number of boxes
1781 long nboxes=ndata/npoints;
1782 MADNESS_CHECK(nboxes*npoints==ndata);
1783 print("reading ",nboxes,"boxes from file",gridfile);
1784
1785 // these will be the data
1786 Tensor<T> values(cdata.vk,false);
1787
1788 int ii=0;
1789 std::string gline;
1790 // while (1) {
1791 while (std::getline(gfile,gline)) {
1792
1793 double x1,y1,z1,val;
1794
1795 // get the key
1796 long nn;
1797 Translation l1,l2,l3;
1798 // line looks like: # key: n l1 l2 l3
1799 gline.erase(0,7);
1800 std::stringstream(gline) >> nn >> l1 >> l2 >> l3;
1801 const Vector<Translation,3> ll{ l1,l2,l3 };
1802 Key<3> key(nn,ll);
1803
1804 // this is borrowed from fcube
1805 const Vector<Translation,3>& l = key.translation();
1806 const Level n = key.level();
1807 const double h = std::pow(0.5,double(n));
1808 coordT c; // will hold the point in user coordinates
1811
1812
1813 if (NDIM == 3) {
1814 for (int i=0; i<npt; ++i) {
1815 c[0] = cell(0,0) + h*cell_width[0]*(l[0] + qx(i)); // x
1816 for (int j=0; j<npt; ++j) {
1817 c[1] = cell(1,0) + h*cell_width[1]*(l[1] + qx(j)); // y
1818 for (int k=0; k<npt; ++k) {
1819 c[2] = cell(2,0) + h*cell_width[2]*(l[2] + qx(k)); // z
1820
1821 auto& success = std::getline(gfile,gline);
1822 MADNESS_CHECK(success);
1823 std::istringstream(gline) >> x1 >> y1 >> z1 >> val;
1824 MADNESS_CHECK(std::fabs(x1-c[0])<1.e-4);
1825 MADNESS_CHECK(std::fabs(y1-c[1])<1.e-4);
1826 MADNESS_CHECK(std::fabs(z1-c[2])<1.e-4);
1827
1828 // regularize if a functor is given
1829 if (vnuc_functor) val-=(*vnuc_functor)(c);
1830 values(i,j,k)=val;
1831 }
1832 }
1833 }
1834 } else {
1835 MADNESS_EXCEPTION("only NDIM=3 in print_grid",0);
1836 }
1837
1838 // insert the new leaf node
1839 const bool has_children=false;
1840 coeffT coeff=coeffT(this->values2coeffs(key,values),targs);
1841 nodeT node(coeff,has_children);
1842 coeffs.replace(key,node);
1843 const_cast<dcT&>(coeffs).send(key.parent(),
1845 coeffs, key.parent());
1846 ii++;
1847 }
1848
1849 gfile.close();
1850 MADNESS_CHECK(ii==nboxes);
1851
1852 }
1853
1854
1855 /// Compute by projection the scaling function coeffs in specified box
1856 /// @param[in] key the key to the current function node (box)
1857 tensorT project(const keyT& key) const;
1858
1859 /// Returns the truncation threshold according to truncate_method
1860
1861 /// here is our handwaving argument:
1862 /// this threshold will give each FunctionNode an error of less than tol. The
1863 /// total error can then be as high as sqrt(#nodes) * tol. Therefore in order
1864 /// to account for higher dimensions: divide tol by about the root of number
1865 /// of siblings (2^NDIM) that have a large error when we refine along a deep
1866 /// branch of the tree.
1867 double truncate_tol(double tol, const keyT& key) const;
1868
1869 int get_truncate_mode() const { return truncate_mode; };
1870 void set_truncate_mode(int mode) { truncate_mode = mode; };
1871
1872
1873 /// Returns patch referring to coeffs of child in parent box
1874 /// @param[in] child the key to the child function node (box)
1875 std::vector<Slice> child_patch(const keyT& child) const;
1876
1877 /// Projection with optional refinement w/ special points
1878 /// @param[in] key the key to the current function node (box)
1879 /// @param[in] do_refine should we continue refinement?
1880 /// @param[in] specialpts vector of special points in the function where we need
1881 /// to refine at a much finer level
1882 void project_refine_op(const keyT& key, bool do_refine,
1883 const std::vector<Vector<double,NDIM> >& specialpts);
1884
1885 /// Compute the Legendre scaling functions for multiplication
1886
1887 /// Evaluate parent polyn at quadrature points of a child. The prefactor of
1888 /// 2^n/2 is included. The tensor must be preallocated as phi(k,npt).
1889 /// Refer to the implementation notes for more info.
1890 /// @todo Robert please verify this comment. I don't understand this method.
1891 /// @param[in] np level of the parent function node (box)
1892 /// @param[in] nc level of the child function node (box)
1893 /// @param[in] lp translation of the parent function node (box)
1894 /// @param[in] lc translation of the child function node (box)
1895 /// @param[out] phi tensor of the legendre scaling functions
1896 void phi_for_mul(Level np, Translation lp, Level nc, Translation lc, Tensor<double>& phi) const;
1897
1898 /// Directly project parent coeffs to child coeffs
1899
1900 /// Currently used by diff, but other uses can be anticipated
1901
1902 /// @todo is this documentation correct?
1903 /// @param[in] child the key whose coeffs we are requesting
1904 /// @param[in] parent the (leaf) key of our function
1905 /// @param[in] s the (leaf) coeffs belonging to parent
1906 /// @return coeffs
1907 const coeffT parent_to_child(const coeffT& s, const keyT& parent, const keyT& child) const;
1908
1909 /// Directly project parent NS coeffs to child NS coeffs
1910
1911 /// return the NS coefficients if parent and child are the same,
1912 /// or construct sum coeffs from the parents and "add" zero wavelet coeffs
1913 /// @param[in] child the key whose coeffs we are requesting
1914 /// @param[in] parent the (leaf) key of our function
1915 /// @param[in] coeff the (leaf) coeffs belonging to parent
1916 /// @return coeffs in NS form
1917 coeffT parent_to_child_NS(const keyT& child, const keyT& parent,
1918 const coeffT& coeff) const;
1919
1920 /// Return the values when given the coeffs in scaling function basis
1921 /// @param[in] key the key of the function node (box)
1922 /// @param[in] coeff the tensor of scaling function coefficients for function node (box)
1923 /// @return function values for function node (box)
1924 template <typename Q>
1925 GenTensor<Q> coeffs2values(const keyT& key, const GenTensor<Q>& coeff) const {
1926 // PROFILE_MEMBER_FUNC(FunctionImpl); // Too fine grain for routine profiling
1927 double scale = pow(2.0,0.5*NDIM*key.level())/sqrt(FunctionDefaults<NDIM>::get_cell_volume());
1928 return transform(coeff,cdata.quad_phit).scale(scale);
1929 }
1930
1931 /// convert S or NS coeffs to values on a 2k grid of the children
1932
1933 /// equivalent to unfiltering the NS coeffs and then converting all child S-coeffs
1934 /// to values in their respective boxes. If only S coeffs are provided d coeffs are
1935 /// assumed to be zero. Reverse operation to values2NScoeffs().
1936 /// @param[in] key the key of the current S or NS coeffs, level n
1937 /// @param[in] coeff coeffs in S or NS form; if S then d coeffs are assumed zero
1938 /// @param[in] s_only sanity check to avoid unintended discard of d coeffs
1939 /// @return function values on the quadrature points of the children of child (!)
1940 template <typename Q>
1942 const bool s_only) const {
1943 // PROFILE_MEMBER_FUNC(FunctionImpl); // Too fine grain for routine profiling
1944
1945 // sanity checks
1946 MADNESS_ASSERT((coeff.dim(0)==this->get_k()) == s_only);
1947 MADNESS_ASSERT((coeff.dim(0)==this->get_k()) or (coeff.dim(0)==2*this->get_k()));
1948
1949 // this is a block-diagonal matrix with the quadrature points on the diagonal
1950 Tensor<double> quad_phit_2k(2*cdata.k,2*cdata.npt);
1951 quad_phit_2k(cdata.s[0],cdata.s[0])=cdata.quad_phit;
1952 quad_phit_2k(cdata.s[1],cdata.s[1])=cdata.quad_phit;
1953
1954 // the transformation matrix unfilters (cdata.hg) and transforms to values in one step
1955 const Tensor<double> transf = (s_only)
1956 ? inner(cdata.hg(Slice(0,k-1),_),quad_phit_2k) // S coeffs
1957 : inner(cdata.hg,quad_phit_2k); // NS coeffs
1958
1959 // increment the level since the coeffs2values part happens on level n+1
1960 const double scale = pow(2.0,0.5*NDIM*(key.level()+1))/
1962
1963 return transform(coeff,transf).scale(scale);
1964 }
1965
1966 /// Compute the function values for multiplication
1967
1968 /// Given S or NS coefficients from a parent cell, compute the value of
1969 /// the functions at the quadrature points of a child
1970 /// currently restricted to special cases
1971 /// @param[in] child key of the box in which we compute values
1972 /// @param[in] parent key of the parent box holding the coeffs
1973 /// @param[in] coeff coeffs of the parent box
1974 /// @param[in] s_only sanity check to avoid unintended discard of d coeffs
1975 /// @return function values on the quadrature points of the children of child (!)
1976 template <typename Q>
1977 GenTensor<Q> NS_fcube_for_mul(const keyT& child, const keyT& parent,
1978 const GenTensor<Q>& coeff, const bool s_only) const {
1979 // PROFILE_MEMBER_FUNC(FunctionImpl); // Too fine grain for routine profiling
1980
1981 // sanity checks
1982 MADNESS_ASSERT((coeff.dim(0)==this->get_k()) == s_only);
1983 MADNESS_ASSERT((coeff.dim(0)==this->get_k()) or (coeff.dim(0)==2*this->get_k()));
1984
1985 // fast return if possible
1986 // if (child.level()==parent.level()) return NScoeffs2values(child,coeff,s_only);
1987
1988 if (s_only) {
1989
1990 Tensor<double> quad_phi[NDIM];
1991 // tmp tensor
1992 Tensor<double> phi1(cdata.k,cdata.npt);
1993
1994 for (std::size_t d=0; d<NDIM; ++d) {
1995
1996 // input is S coeffs (dimension k), output is values on 2*npt grid points
1997 quad_phi[d]=Tensor<double>(cdata.k,2*cdata.npt);
1998
1999 // for both children of "child" evaluate the Legendre polynomials
2000 // first the left child on level n+1 and translations 2l
2001 phi_for_mul(parent.level(),parent.translation()[d],
2002 child.level()+1, 2*child.translation()[d], phi1);
2003 quad_phi[d](_,Slice(0,k-1))=phi1;
2004
2005 // next the right child on level n+1 and translations 2l+1
2006 phi_for_mul(parent.level(),parent.translation()[d],
2007 child.level()+1, 2*child.translation()[d]+1, phi1);
2008 quad_phi[d](_,Slice(k,2*k-1))=phi1;
2009 }
2010
2011 const double scale = 1.0/sqrt(FunctionDefaults<NDIM>::get_cell_volume());
2012 return general_transform(coeff,quad_phi).scale(scale);
2013 }
2014 MADNESS_EXCEPTION("you should not be here in NS_fcube_for_mul",1);
2015 return GenTensor<Q>();
2016 }
2017
2018 /// convert function values of the a child generation directly to NS coeffs
2019
2020 /// equivalent to converting the function values to 2^NDIM S coeffs and then
2021 /// filtering them to NS coeffs. Reverse operation to NScoeffs2values().
2022 /// @param[in] key key of the parent of the generation
2023 /// @param[in] values tensor holding function values of the 2^NDIM children of key
2024 /// @return NS coeffs belonging to key
2025 template <typename Q>
2026 GenTensor<Q> values2NScoeffs(const keyT& key, const GenTensor<Q>& values) const {
2027 //PROFILE_MEMBER_FUNC(FunctionImpl); // Too fine grain for routine profiling
2028
2029 // sanity checks
2030 MADNESS_ASSERT(values.dim(0)==2*this->get_k());
2031
2032 // this is a block-diagonal matrix with the quadrature points on the diagonal
2033 Tensor<double> quad_phit_2k(2*cdata.npt,2*cdata.k);
2034 quad_phit_2k(cdata.s[0],cdata.s[0])=cdata.quad_phiw;
2035 quad_phit_2k(cdata.s[1],cdata.s[1])=cdata.quad_phiw;
2036
2037 // the transformation matrix unfilters (cdata.hg) and transforms to values in one step
2038 const Tensor<double> transf=inner(quad_phit_2k,cdata.hgT);
2039
2040 // increment the level since the values2coeffs part happens on level n+1
2041 const double scale = pow(0.5,0.5*NDIM*(key.level()+1))
2043
2044 return transform(values,transf).scale(scale);
2045 }
2046
2047 /// Return the scaling function coeffs when given the function values at the quadrature points
2048 /// @param[in] key the key of the function node (box)
2049 /// @return function values for function node (box)
2050 template <typename Q>
2051 Tensor<Q> coeffs2values(const keyT& key, const Tensor<Q>& coeff) const {
2052 // PROFILE_MEMBER_FUNC(FunctionImpl); // Too fine grain for routine profiling
2053 double scale = pow(2.0,0.5*NDIM*key.level())/sqrt(FunctionDefaults<NDIM>::get_cell_volume());
2054 return transform(coeff,cdata.quad_phit).scale(scale);
2055 }
2056
2057 template <typename Q>
2058 GenTensor<Q> values2coeffs(const keyT& key, const GenTensor<Q>& values) const {
2059 // PROFILE_MEMBER_FUNC(FunctionImpl); // Too fine grain for routine profiling
2060 double scale = pow(0.5,0.5*NDIM*key.level())*sqrt(FunctionDefaults<NDIM>::get_cell_volume());
2061 return transform(values,cdata.quad_phiw).scale(scale);
2062 }
2063
2064 template <typename Q>
2065 Tensor<Q> values2coeffs(const keyT& key, const Tensor<Q>& values) const {
2066 // PROFILE_MEMBER_FUNC(FunctionImpl); // Too fine grain for routine profiling
2067 double scale = pow(0.5,0.5*NDIM*key.level())*sqrt(FunctionDefaults<NDIM>::get_cell_volume());
2068 return transform(values,cdata.quad_phiw).scale(scale);
2069 }
2070
2071 /// Compute the function values for multiplication
2072
2073 /// Given coefficients from a parent cell, compute the value of
2074 /// the functions at the quadrature points of a child
2075 /// @param[in] child the key for the child function node (box)
2076 /// @param[in] parent the key for the parent function node (box)
2077 /// @param[in] coeff the coefficients of scaling function basis of the parent box
2078 template <typename Q>
2079 Tensor<Q> fcube_for_mul(const keyT& child, const keyT& parent, const Tensor<Q>& coeff) const {
2080 // PROFILE_MEMBER_FUNC(FunctionImpl); // Too fine grain for routine profiling
2081 if (child.level() == parent.level()) {
2082 return coeffs2values(parent, coeff);
2083 }
2084 else if (child.level() < parent.level()) {
2085 MADNESS_EXCEPTION("FunctionImpl: fcube_for_mul: child-parent relationship bad?",0);
2086 }
2087 else {
2088 Tensor<double> phi[NDIM];
2089 for (std::size_t d=0; d<NDIM; ++d) {
2090 phi[d] = Tensor<double>(cdata.k,cdata.npt);
2091 phi_for_mul(parent.level(),parent.translation()[d],
2092 child.level(), child.translation()[d], phi[d]);
2093 }
2094 return general_transform(coeff,phi).scale(1.0/sqrt(FunctionDefaults<NDIM>::get_cell_volume()));;
2095 }
2096 }
2097
2098
2099 /// Compute the function values for multiplication
2100
2101 /// Given coefficients from a parent cell, compute the value of
2102 /// the functions at the quadrature points of a child
2103 /// @param[in] child the key for the child function node (box)
2104 /// @param[in] parent the key for the parent function node (box)
2105 /// @param[in] coeff the coefficients of scaling function basis of the parent box
2106 template <typename Q>
2107 GenTensor<Q> fcube_for_mul(const keyT& child, const keyT& parent, const GenTensor<Q>& coeff) const {
2108 // PROFILE_MEMBER_FUNC(FunctionImpl); // Too fine grain for routine profiling
2109 if (child.level() == parent.level()) {
2110 return coeffs2values(parent, coeff);
2111 }
2112 else if (child.level() < parent.level()) {
2113 MADNESS_EXCEPTION("FunctionImpl: fcube_for_mul: child-parent relationship bad?",0);
2114 }
2115 else {
2116 Tensor<double> phi[NDIM];
2117 for (size_t d=0; d<NDIM; d++) {
2118 phi[d] = Tensor<double>(cdata.k,cdata.npt);
2119 phi_for_mul(parent.level(),parent.translation()[d],
2120 child.level(), child.translation()[d], phi[d]);
2121 }
2122 return general_transform(coeff,phi).scale(1.0/sqrt(FunctionDefaults<NDIM>::get_cell_volume()));
2123 }
2124 }
2125
2126
2127 /// Functor for the mul method
2128 template <typename L, typename R>
2129 void do_mul(const keyT& key, const Tensor<L>& left, const std::pair< keyT, Tensor<R> >& arg) {
2130 // PROFILE_MEMBER_FUNC(FunctionImpl); // Too fine grain for routine profiling
2131 const keyT& rkey = arg.first;
2132 const Tensor<R>& rcoeff = arg.second;
2133 //madness::print("do_mul: r", rkey, rcoeff.size());
2134 Tensor<R> rcube = fcube_for_mul(key, rkey, rcoeff);
2135 //madness::print("do_mul: l", key, left.size());
2136 Tensor<L> lcube = fcube_for_mul(key, key, left);
2137
2138 Tensor<T> tcube(cdata.vk,false);
2139 TERNARY_OPTIMIZED_ITERATOR(T, tcube, L, lcube, R, rcube, *_p0 = *_p1 * *_p2;);
2140 double scale = pow(0.5,0.5*NDIM*key.level())*sqrt(FunctionDefaults<NDIM>::get_cell_volume());
2141 tcube = transform(tcube,cdata.quad_phiw).scale(scale);
2142 coeffs.replace(key, nodeT(coeffT(tcube,targs),false));
2143 }
2144
2145
2146 /// multiply the values of two coefficient tensors using a custom number of grid points
2147
2148 /// note both coefficient tensors have to refer to the same key!
2149 /// @param[in] c1 a tensor holding coefficients
2150 /// @param[in] c2 another tensor holding coeffs
2151 /// @param[in] npt number of grid points (optional, default is cdata.npt)
2152 /// @return coefficient tensor holding the product of the values of c1 and c2
2153 template<typename R>
2155 const int npt, const keyT& key) const {
2156 typedef TENSOR_RESULT_TYPE(T,R) resultT;
2157
2159
2160 // construct a tensor with the npt coeffs
2161 Tensor<T> c11(cdata2.vk), c22(cdata2.vk);
2162 c11(this->cdata.s0)=c1;
2163 c22(this->cdata.s0)=c2;
2164
2165 // it's sufficient to scale once
2166 double scale = pow(2.0,0.5*NDIM*key.level())/sqrt(FunctionDefaults<NDIM>::get_cell_volume());
2167 Tensor<T> c1value=transform(c11,cdata2.quad_phit).scale(scale);
2168 Tensor<R> c2value=transform(c22,cdata2.quad_phit);
2169 Tensor<resultT> resultvalue(cdata2.vk,false);
2170 TERNARY_OPTIMIZED_ITERATOR(resultT, resultvalue, T, c1value, R, c2value, *_p0 = *_p1 * *_p2;);
2171
2172 Tensor<resultT> result=transform(resultvalue,cdata2.quad_phiw);
2173
2174 // return a copy of the slice to have the tensor contiguous
2175 return copy(result(this->cdata.s0));
2176 }
2177
2178
2179 /// Functor for the binary_op method
2180 template <typename L, typename R, typename opT>
2181 void do_binary_op(const keyT& key, const Tensor<L>& left,
2182 const std::pair< keyT, Tensor<R> >& arg,
2183 const opT& op) {
2184 //PROFILE_MEMBER_FUNC(FunctionImpl); // Too fine grain for routine profiling
2185 const keyT& rkey = arg.first;
2186 const Tensor<R>& rcoeff = arg.second;
2187 Tensor<R> rcube = fcube_for_mul(key, rkey, rcoeff);
2188 Tensor<L> lcube = fcube_for_mul(key, key, left);
2189
2190 Tensor<T> tcube(cdata.vk,false);
2191 op(key, tcube, lcube, rcube);
2192 double scale = pow(0.5,0.5*NDIM*key.level())*sqrt(FunctionDefaults<NDIM>::get_cell_volume());
2193 tcube = transform(tcube,cdata.quad_phiw).scale(scale);
2194 coeffs.replace(key, nodeT(coeffT(tcube,targs),false));
2195 }
2196
2197 /// Invoked by result to perform result += alpha*left+beta*right in wavelet basis
2198
2199 /// Does not assume that any of result, left, right have the same distribution.
2200 /// For most purposes result will start as an empty so actually are implementing
2201 /// out of place gaxpy. If all functions have the same distribution there is
2202 /// no communication except for the optional fence.
2203 template <typename L, typename R>
2204 void gaxpy(T alpha, const FunctionImpl<L,NDIM>& left,
2205 T beta, const FunctionImpl<R,NDIM>& right, bool fence) {
2206 // Loop over local nodes in both functions. Add in left and subtract right.
2207 // Not that efficient in terms of memory bandwidth but ensures we do
2208 // not miss any nodes.
2209 typename FunctionImpl<L,NDIM>::dcT::const_iterator left_end = left.coeffs.end();
2211 it!=left_end;
2212 ++it) {
2213 const keyT& key = it->first;
2214 const typename FunctionImpl<L,NDIM>::nodeT& other_node = it->second;
2215 coeffs.send(key, &nodeT:: template gaxpy_inplace<T,L>, 1.0, other_node, alpha);
2216 }
2217 typename FunctionImpl<R,NDIM>::dcT::const_iterator right_end = right.coeffs.end();
2219 it!=right_end;
2220 ++it) {
2221 const keyT& key = it->first;
2222 const typename FunctionImpl<L,NDIM>::nodeT& other_node = it->second;
2223 coeffs.send(key, &nodeT:: template gaxpy_inplace<T,R>, 1.0, other_node, beta);
2224 }
2225 if (fence)
2226 world.gop.fence();
2227 }
2228
2229 /// Unary operation applied inplace to the coefficients WITHOUT refinement, optional fence
2230 /// @param[in] op the unary operator for the coefficients
2231 template <typename opT>
2232 void unary_op_coeff_inplace(const opT& op, bool fence) {
2233 typename dcT::iterator end = coeffs.end();
2234 for (typename dcT::iterator it=coeffs.begin(); it!=end; ++it) {
2235 const keyT& parent = it->first;
2236 nodeT& node = it->second;
2237 if (node.has_coeff()) {
2238 // op(parent, node.coeff());
2239 TensorArgs full(-1.0,TT_FULL);
2240 change_tensor_type(node.coeff(),full);
2241 op(parent, node.coeff().full_tensor());
2243 // op(parent,node);
2244 }
2245 }
2246 if (fence)
2247 world.gop.fence();
2248 }
2249
2250 /// Unary operation applied inplace to the coefficients WITHOUT refinement, optional fence
2251 /// @param[in] op the unary operator for the coefficients
2252 template <typename opT>
2253 void unary_op_node_inplace(const opT& op, bool fence) {
2254 typename dcT::iterator end = coeffs.end();
2255 for (typename dcT::iterator it=coeffs.begin(); it!=end; ++it) {
2256 const keyT& parent = it->first;
2257 nodeT& node = it->second;
2258 op(parent, node);
2259 }
2260 if (fence)
2261 world.gop.fence();
2262 }
2263
2264 /// Integrate over one particle of a two particle function and get a one particle function
2265 /// bsp \int g(1,2) \delta(2-1) d2 = f(1)
2266 /// The overall dimension of g should be even
2267
2268 /// The operator
2269 template<std::size_t LDIM>
2270 void dirac_convolution_op(const keyT &key, const nodeT &node, FunctionImpl<T,LDIM>* f) const {
2271 // fast return if the node has children (not a leaf node)
2272 if(node.has_children()) return;
2273
2274 const implT* g=this;
2275
2276 // break the 6D key into two 3D keys (may also work for every even dimension)
2277 Key<LDIM> key1, key2;
2278 key.break_apart(key1,key2);
2279
2280 // get the coefficients of the 6D function g
2281 const coeffT& g_coeff = node.coeff();
2282
2283 // get the values of the 6D function g
2284 coeffT g_values = g->coeffs2values(key,g_coeff);
2285
2286 // Determine rank and k
2287 const long rank=g_values.rank();
2288 const long maxk=f->get_k();
2289 MADNESS_ASSERT(maxk==g_coeff.dim(0));
2290
2291 // get tensors for particle 1 and 2 (U and V in SVD)
2292 tensorT vec1=copy(g_values.get_svdtensor().ref_vector(0).reshape(rank,maxk,maxk,maxk));
2293 tensorT vec2=g_values.get_svdtensor().ref_vector(1).reshape(rank,maxk,maxk,maxk);
2294 tensorT result(maxk,maxk,maxk); // should give zero tensor
2295 // Multiply the values of each U and V vector
2296 for (long i=0; i<rank; ++i) {
2297 tensorT c1=vec1(Slice(i,i),_,_,_); // shallow copy (!)
2298 tensorT c2=vec2(Slice(i,i),_,_,_);
2299 c1.emul(c2); // this changes vec1 because of shallow copy, but not the g function because of the deep copy made above
2300 double singular_value_i = g_values.get_svdtensor().weights(i);
2301 result += (singular_value_i*c1);
2302 }
2303
2304 // accumulate coefficients (since only diagonal boxes are used the coefficients get just replaced, but accumulate is needed to create the right tree structure
2305 tensorT f_coeff = f->values2coeffs(key1,result);
2306 f->coeffs.task(key1, &FunctionNode<T,LDIM>::accumulate2, f_coeff, f->coeffs, key1, TaskAttributes::hipri());
2307// coeffs.task(dest, &nodeT::accumulate2, result, coeffs, dest, TaskAttributes::hipri());
2308
2309
2310 return;
2311 }
2312
2313
2314 template<std::size_t LDIM>
2316 typename dcT::const_iterator end = this->coeffs.end();
2317 for (typename dcT::const_iterator it=this->coeffs.begin(); it!=end; ++it) {
2318 // looping through all the leaf(!) coefficients in the NDIM function ("this")
2319 const keyT& key = it->first;
2320 const FunctionNode<T,NDIM>& node = it->second;
2321 if (node.is_leaf()) {
2322 // only process the diagonal boxes
2323 Key<LDIM> key1, key2;
2324 key.break_apart(key1,key2);
2325 if(key1 == key2){
2326 ProcessID p = coeffs.owner(key);
2327 woT::task(p, &implT:: template dirac_convolution_op<LDIM>, key, node, f);
2328 }
2329 }
2330 }
2331 world.gop.fence(); // fence is necessary if trickle down is used afterwards
2332 // trickle down and undo redundand shouldnt change anything if only the diagonal elements are considered above -> check this
2333 f->trickle_down(true); // fence must be true otherwise undo_redundant will have trouble
2334// f->undo_redundant(true);
2335 f->verify_tree();
2336 //if (fence) world.gop.fence(); // unnecessary, fence is activated in undo_redundant
2337
2338 }
2339
2340
2341 /// Unary operation applied inplace to the coefficients WITHOUT refinement, optional fence
2342 /// @param[in] op the unary operator for the coefficients
2343 template <typename opT>
2344 void flo_unary_op_node_inplace(const opT& op, bool fence) {
2346// typedef do_unary_op_value_inplace<opT> xopT;
2348 if (fence) world.gop.fence();
2349 }
2350
2351 /// Unary operation applied inplace to the coefficients WITHOUT refinement, optional fence
2352 /// @param[in] op the unary operator for the coefficients
2353 template <typename opT>
2354 void flo_unary_op_node_inplace(const opT& op, bool fence) const {
2356// typedef do_unary_op_value_inplace<opT> xopT;
2358 if (fence)
2359 world.gop.fence();
2360 }
2361
2362 /// truncate tree at a certain level
2363 /// @param[in] max_level truncate tree below this level
2364 void erase(const Level& max_level);
2365
2366 /// Returns some asymmetry measure ... no comms
2367 double check_symmetry_local() const;
2368
2369 /// given an NS tree resulting from a convolution, truncate leafs if appropriate
2372 const implT* f; // for calling its member functions
2373
2375
2376 bool operator()(typename rangeT::iterator& it) const {
2377
2378 const keyT& key = it->first;
2379 nodeT& node = it->second;
2380
2381 if (node.is_leaf() and node.coeff().has_data()) {
2382 coeffT d = copy(node.coeff());
2383 d(f->cdata.s0)=0.0;
2384 const double error=d.normf();
2385 const double tol=f->truncate_tol(f->get_thresh(),key);
2386 if (error<tol) node.coeff()=copy(node.coeff()(f->cdata.s0));
2387 }
2388 return true;
2389 }
2390 template <typename Archive> void serialize(const Archive& ar) {}
2391
2392 };
2393
2394 /// remove all coefficients of internal nodes
2397
2398 /// constructor need impl for cdata
2400
2401 bool operator()(typename rangeT::iterator& it) const {
2402
2403 nodeT& node = it->second;
2404 if (node.has_children()) node.clear_coeff();
2405 return true;
2406 }
2407 template <typename Archive> void serialize(const Archive& ar) {}
2408
2409 };
2410
2411 /// remove all coefficients of leaf nodes
2414
2415 /// constructor need impl for cdata
2417
2418 bool operator()(typename rangeT::iterator& it) const {
2419 nodeT& node = it->second;
2420 if (not node.has_children()) node.clear_coeff();
2421 return true;
2422 }
2423 template <typename Archive> void serialize(const Archive& ar) {}
2424
2425 };
2426
2427
2428 /// keep only the sum coefficients in each node
2432
2433 /// constructor need impl for cdata
2435
2436 bool operator()(typename rangeT::iterator& it) const {
2437
2438 nodeT& node = it->second;
2439 coeffT s=copy(node.coeff()(impl->cdata.s0));
2440 node.coeff()=s;
2441 return true;
2442 }
2443 template <typename Archive> void serialize(const Archive& ar) {}
2444
2445 };
2446
2447
2448 /// reduce the rank of the nodes, optional fence
2451
2452 // threshold for rank reduction / SVD truncation
2454
2455 // constructor takes target precision
2456 do_reduce_rank() = default;
2458 do_reduce_rank(const double& thresh) {
2460 }
2461
2462 //
2463 bool operator()(typename rangeT::iterator& it) const {
2464
2465 nodeT& node = it->second;
2466 node.reduceRank(args.thresh);
2467 return true;
2468 }
2469 template <typename Archive> void serialize(const Archive& ar) {}
2470 };
2471
2472
2473
2474 /// check symmetry wrt particle exchange
2477 const implT* f;
2480
2481 /// return the norm of the difference of this node and its "mirror" node
2482 double operator()(typename rangeT::iterator& it) const {
2483
2484 // Temporary fix to GCC whining about out of range access for NDIM!=6
2485 if constexpr(NDIM==6) {
2486 const keyT& key = it->first;
2487 const nodeT& fnode = it->second;
2488
2489 // skip internal nodes
2490 if (fnode.has_children()) return 0.0;
2491
2492 if (f->world.size()>1) return 0.0;
2493
2494 // exchange particles
2495 std::vector<long> map(NDIM);
2496 map[0]=3; map[1]=4; map[2]=5;
2497 map[3]=0; map[4]=1; map[5]=2;
2498
2499 // make mapped key
2501 for (std::size_t i=0; i<NDIM; ++i) l[map[i]] = key.translation()[i];
2502 const keyT mapkey(key.level(),l);
2503
2504 double norm=0.0;
2505
2506
2507 // hope it's local
2508 if (f->get_coeffs().probe(mapkey)) {
2509 MADNESS_ASSERT(f->get_coeffs().probe(mapkey));
2510 const nodeT& mapnode=f->get_coeffs().find(mapkey).get()->second;
2511
2512// bool have_c1=fnode.coeff().has_data() and fnode.coeff().config().has_data();
2513// bool have_c2=mapnode.coeff().has_data() and mapnode.coeff().config().has_data();
2514 bool have_c1=fnode.coeff().has_data();
2515 bool have_c2=mapnode.coeff().has_data();
2516
2517 if (have_c1 and have_c2) {
2518 tensorT c1=fnode.coeff().full_tensor_copy();
2519 tensorT c2=mapnode.coeff().full_tensor_copy();
2520 c2 = copy(c2.mapdim(map));
2521 norm=(c1-c2).normf();
2522 } else if (have_c1) {
2523 tensorT c1=fnode.coeff().full_tensor_copy();
2524 norm=c1.normf();
2525 } else if (have_c2) {
2526 tensorT c2=mapnode.coeff().full_tensor_copy();
2527 norm=c2.normf();
2528 } else {
2529 norm=0.0;
2530 }
2531 } else {
2532 norm=fnode.coeff().normf();
2533 }
2534 return norm*norm;
2535 }
2536 else {
2537 MADNESS_EXCEPTION("ONLY FOR DIM 6!", 1);
2538 }
2539 }
2540
2541 double operator()(double a, double b) const {
2542 return (a+b);
2543 }
2544
2545 template <typename Archive> void serialize(const Archive& ar) {
2546 MADNESS_EXCEPTION("no serialization of do_check_symmetry yet",1);
2547 }
2548
2549
2550 };
2551
2552 /// merge the coefficent boxes of this into result's tree
2553
2554 /// result+= alpha*this
2555 /// this and result don't have to have the same distribution or live in the same world
2556 /// no comm, and the tree should be in an consistent state by virtue
2557 template<typename Q, typename R>
2561 T alpha=T(1.0);
2565
2566 /// return the norm of the difference of this node and its "mirror" node
2567 bool operator()(typename rangeT::iterator& it) const {
2568
2569 const keyT& key = it->first;
2570 const nodeT& node = it->second;
2571 if (node.has_coeff()) result->get_coeffs().task(key, &nodeT::accumulate,
2572 alpha*node.coeff(), result->get_coeffs(), key, result->targs);
2573 return true;
2574 }
2575
2576 template <typename Archive> void serialize(const Archive& ar) {
2577 MADNESS_EXCEPTION("no serialization of do_accumulate_trees",1);
2578 }
2579 };
2580
2581
2582 /// merge the coefficient boxes of this into other's tree
2583
2584 /// no comm, and the tree should be in an consistent state by virtue
2585 /// of FunctionNode::gaxpy_inplace
2586 template<typename Q, typename R>
2595
2596 /// return the norm of the difference of this node and its "mirror" node
2597 bool operator()(typename rangeT::iterator& it) const {
2598
2599 const keyT& key = it->first;
2600 const nodeT& fnode = it->second;
2601
2602 // if other's node exists: add this' coeffs to it
2603 // otherwise insert this' node into other's tree
2604 typename dcT::accessor acc;
2605 if (other->get_coeffs().find(acc,key)) {
2606 nodeT& gnode=acc->second;
2607 gnode.gaxpy_inplace(beta,fnode,alpha);
2608 } else {
2609 nodeT gnode=fnode;
2610 gnode.scale(alpha);
2611 other->get_coeffs().replace(key,gnode);
2612 }
2613 return true;
2614 }
2615
2616 template <typename Archive> void serialize(const Archive& ar) {
2617 MADNESS_EXCEPTION("no serialization of do_merge_trees",1);
2618 }
2619 };
2620
2621
2622 /// map this on f
2623 struct do_mapdim {
2625
2626 std::vector<long> map;
2628
2629 do_mapdim() : f(0) {};
2630 do_mapdim(const std::vector<long> map, implT& f) : map(map), f(&f) {}
2631
2632 bool operator()(typename rangeT::iterator& it) const {
2633
2634 const keyT& key = it->first;
2635 const nodeT& node = it->second;
2636
2638 for (std::size_t i=0; i<NDIM; ++i) l[map[i]] = key.translation()[i];
2639 tensorT c = node.coeff().reconstruct_tensor();
2640 if (c.size()) c = copy(c.mapdim(map));
2642 f->get_coeffs().replace(keyT(key.level(),l), nodeT(cc,node.has_children()));
2643
2644 return true;
2645 }
2646 template <typename Archive> void serialize(const Archive& ar) {
2647 MADNESS_EXCEPTION("no serialization of do_mapdim",1);
2648 }
2649
2650 };
2651
2652 /// mirror dimensions of this, write result on f
2653 struct do_mirror {
2655
2656 std::vector<long> mirror;
2658
2659 do_mirror() : f(0) {};
2660 do_mirror(const std::vector<long> mirror, implT& f) : mirror(mirror), f(&f) {}
2661
2662 bool operator()(typename rangeT::iterator& it) const {
2663
2664 const keyT& key = it->first;
2665 const nodeT& node = it->second;
2666
2667 // mirror translation index: l_new + l_old = l_max
2669 Translation lmax = (Translation(1)<<key.level()) - 1;
2670 for (std::size_t i=0; i<NDIM; ++i) {
2671 if (mirror[i]==-1) l[i]= lmax - key.translation()[i];
2672 }
2673
2674 // mirror coefficients: multiply all odd-k slices with -1
2675 tensorT c = node.coeff().full_tensor_copy();
2676 if (c.size()) {
2677 std::vector<Slice> s(___);
2678
2679 // loop over dimensions and over k
2680 for (size_t i=0; i<NDIM; ++i) {
2681 std::size_t kmax=c.dim(i);
2682 if (mirror[i]==-1) {
2683 for (size_t k=1; k<kmax; k+=2) {
2684 s[i]=Slice(k,k,1);
2685 c(s)*=(-1.0);
2686 }
2687 s[i]=_;
2688 }
2689 }
2690 }
2692 f->get_coeffs().replace(keyT(key.level(),l), nodeT(cc,node.has_children()));
2693
2694 return true;
2695 }
2696 template <typename Archive> void serialize(const Archive& ar) {
2697 MADNESS_EXCEPTION("no serialization of do_mirror",1);
2698 }
2699
2700 };
2701
2702 /// mirror dimensions of this, write result on f
2705
2706 std::vector<long> map,mirror;
2708
2710 do_map_and_mirror(const std::vector<long> map, const std::vector<long> mirror, implT& f)
2711 : map(map), mirror(mirror), f(&f) {}
2712
2713 bool operator()(typename rangeT::iterator& it) const {
2714
2715 const keyT& key = it->first;
2716 const nodeT& node = it->second;
2717
2718 tensorT c = node.coeff().full_tensor_copy();
2720
2721 // do the mapping first (if present)
2722 if (map.size()>0) {
2724 for (std::size_t i=0; i<NDIM; ++i) l1[map[i]] = l[i];
2725 std::swap(l,l1);
2726 if (c.size()) c = copy(c.mapdim(map));
2727 }
2728
2729 if (mirror.size()>0) {
2730 // mirror translation index: l_new + l_old = l_max
2732 Translation lmax = (Translation(1)<<key.level()) - 1;
2733 for (std::size_t i=0; i<NDIM; ++i) {
2734 if (mirror[i]==-1) l1[i]= lmax - l[i];
2735 }
2736 std::swap(l,l1);
2737
2738 // mirror coefficients: multiply all odd-k slices with -1
2739 if (c.size()) {
2740 std::vector<Slice> s(___);
2741
2742 // loop over dimensions and over k
2743 for (size_t i=0; i<NDIM; ++i) {
2744 std::size_t kmax=c.dim(i);
2745 if (mirror[i]==-1) {
2746 for (size_t k=1; k<kmax; k+=2) {
2747 s[i]=Slice(k,k,1);
2748 c(s)*=(-1.0);
2749 }
2750 s[i]=_;
2751 }
2752 }
2753 }
2754 }
2755
2757 f->get_coeffs().replace(keyT(key.level(),l), nodeT(cc,node.has_children()));
2758 return true;
2759 }
2760 template <typename Archive> void serialize(const Archive& ar) {
2761 MADNESS_EXCEPTION("no serialization of do_mirror",1);
2762 }
2763
2764 };
2765
2766
2767
2768 /// "put" this on g
2769 struct do_average {
2771
2773
2774 do_average() : g(0) {}
2776
2777 /// iterator it points to this
2778 bool operator()(typename rangeT::iterator& it) const {
2779
2780 const keyT& key = it->first;
2781 const nodeT& fnode = it->second;
2782
2783 // fast return if rhs has no coeff here
2784 if (fnode.has_coeff()) {
2785
2786 // check if there is a node already existing
2787 typename dcT::accessor acc;
2788 if (g->get_coeffs().find(acc,key)) {
2789 nodeT& gnode=acc->second;
2790 if (gnode.has_coeff()) gnode.coeff()+=fnode.coeff();
2791 } else {
2792 g->get_coeffs().replace(key,fnode);
2793 }
2794 }
2795
2796 return true;
2797 }
2798 template <typename Archive> void serialize(const Archive& ar) {}
2799 };
2800
2801 /// change representation of nodes' coeffs to low rank, optional fence
2804
2805 // threshold for rank reduction / SVD truncation
2808
2809 // constructor takes target precision
2811 // do_change_tensor_type(const TensorArgs& targs) : targs(targs) {}
2813
2814 //
2815 bool operator()(typename rangeT::iterator& it) const {
2816
2817 double cpu0=cpu_time();
2818 nodeT& node = it->second;
2820 double cpu1=cpu_time();
2822
2823 return true;
2824
2825 }
2826 template <typename Archive> void serialize(const Archive& ar) {}
2827 };
2828
2831
2832 // threshold for rank reduction / SVD truncation
2834
2835 // constructor takes target precision
2838 bool operator()(typename rangeT::iterator& it) const {
2839 it->second.consolidate_buffer(targs);
2840 return true;
2841 }
2842 template <typename Archive> void serialize(const Archive& ar) {}
2843 };
2844
2845
2846
2847 template <typename opT>
2851 opT op;
2853 bool operator()(typename rangeT::iterator& it) const {
2854 const keyT& key = it->first;
2855 nodeT& node = it->second;
2856 if (node.has_coeff()) {
2857 const TensorArgs full_args(-1.0,TT_FULL);
2858 change_tensor_type(node.coeff(),full_args);
2859 tensorT& t= node.coeff().full_tensor();
2860 //double before = t.normf();
2861 tensorT values = impl->fcube_for_mul(key, key, t);
2862 op(key, values);
2863 double scale = pow(0.5,0.5*NDIM*key.level())*sqrt(FunctionDefaults<NDIM>::get_cell_volume());
2864 t = transform(values,impl->cdata.quad_phiw).scale(scale);
2865 node.coeff()=coeffT(t,impl->get_tensor_args());
2866 //double after = t.normf();
2867 //madness::print("XOP:", key, before, after);
2868 }
2869 return true;
2870 }
2871 template <typename Archive> void serialize(const Archive& ar) {}
2872 };
2873
2874 template <typename Q, typename R>
2875 /// @todo I don't know what this does other than a trasform
2876 void vtransform_doit(const std::shared_ptr< FunctionImpl<R,NDIM> >& right,
2877 const Tensor<Q>& c,
2878 const std::vector< std::shared_ptr< FunctionImpl<T,NDIM> > >& vleft,
2879 double tol) {
2880 // To reduce crunch on vectors being transformed each task
2881 // does them in a random order
2882 std::vector<unsigned int> ind(vleft.size());
2883 for (unsigned int i=0; i<vleft.size(); ++i) {
2884 ind[i] = i;
2885 }
2886 for (unsigned int i=0; i<vleft.size(); ++i) {
2887 unsigned int j = RandomValue<int>()%vleft.size();
2888 std::swap(ind[i],ind[j]);
2889 }
2890
2891 for (const auto& [key, rnode] : right->coeffs) {
2892 if (rnode.has_coeff()) {
2893 const GenTensor<R>& r = rnode.coeff();
2894 double norm = r.normf();
2895 double keytol = truncate_tol(tol,key);
2896
2897 for (unsigned int j=0; j<vleft.size(); ++j) {
2898 unsigned int i = ind[j]; // Random permutation
2899 if (std::abs(norm*c(i)) > keytol) {
2900 implT* left = vleft[i].get();
2901 typename dcT::accessor acc;
2902 bool new_node = left->coeffs.insert(acc,key);
2903 if (new_node) {
2904 /* Notify parent nodes that a new child exists. */
2905 Key<NDIM> parent = key.parent();
2906 if (left->coeffs.is_local(parent))
2907 left->coeffs.send(parent, &nodeT::set_has_children_recursive, left->coeffs, parent);
2908 else
2909 left->coeffs.task(parent, &nodeT::set_has_children_recursive, left->coeffs, parent);
2910 }
2911 nodeT& node = acc->second;
2912 node.gaxpy_inplace(1.0, rnode, c(i));
2913 }
2914 }
2915 }
2916 }
2917 }
2918
2919 /// Refine multiple functions down to the same finest level
2920
2921 /// @param v the vector of functions we are refining.
2922 /// @param key the current node.
2923 /// @param c the vector of coefficients passed from above.
2924 void refine_to_common_level(const std::vector<FunctionImpl<T,NDIM>*>& v,
2925 const std::vector<tensorT>& c,
2926 const keyT key);
2927
2928 /// Inplace operate on many functions (impl's) with an operator within a certain box
2929 /// @param[in] key the key of the current function node (box)
2930 /// @param[in] op the operator
2931 /// @param[in] v the vector of function impl's on which to be operated
2932 template <typename opT>
2933 void multiop_values_doit(const keyT& key, const opT& op, const std::vector<implT*>& v) {
2934 std::vector<tensorT> c(v.size());
2935 for (unsigned int i=0; i<v.size(); i++) {
2936 if (v[i]) {
2937 coeffT cc = coeffs2values(key, v[i]->coeffs.find(key).get()->second.coeff());
2938 c[i]=cc.full_tensor();
2939 }
2940 }
2941 tensorT r = op(key, c);
2942 coeffs.replace(key, nodeT(coeffT(values2coeffs(key, r),targs),false));
2943 }
2944
2945 /// Inplace operate on many functions (impl's) with an operator within a certain box
2946 /// Assumes all functions have been refined down to the same level
2947 /// @param[in] op the operator
2948 /// @param[in] v the vector of function impl's on which to be operated
2949 template <typename opT>
2950 void multiop_values(const opT& op, const std::vector<implT*>& v) {
2951 // rough check on refinement level (ignore non-initialized functions
2952 for (std::size_t i=1; i<v.size(); ++i) {
2953 if (v[i] and v[i-1]) {
2954 MADNESS_ASSERT(v[i]->coeffs.size()==v[i-1]->coeffs.size());
2955 }
2956 }
2957 typename dcT::iterator end = v[0]->coeffs.end();
2958 for (typename dcT::iterator it=v[0]->coeffs.begin(); it!=end; ++it) {
2959 const keyT& key = it->first;
2960 if (it->second.has_coeff())
2961 world.taskq.add(*this, &implT:: template multiop_values_doit<opT>, key, op, v);
2962 else
2963 coeffs.replace(key, nodeT(coeffT(),true));
2964 }
2965 world.gop.fence();
2966 }
2967
2968 /// Inplace operate on many functions (impl's) with an operator within a certain box
2969
2970 /// @param[in] key the key of the current function node (box)
2971 /// @param[in] op the operator
2972 /// @param[in] vin the vector of function impl's on which to be operated
2973 /// @param[out] vout the resulting vector of function impl's
2974 template <typename opT>
2975 void multi_to_multi_op_values_doit(const keyT& key, const opT& op,
2976 const std::vector<implT*>& vin, std::vector<implT*>& vout) {
2977 std::vector<tensorT> c(vin.size());
2978 for (unsigned int i=0; i<vin.size(); i++) {
2979 if (vin[i]) {
2980 coeffT cc = coeffs2values(key, vin[i]->coeffs.find(key).get()->second.coeff());
2981 c[i]=cc.full_tensor();
2982 }
2983 }
2984 std::vector<tensorT> r = op(key, c);
2985 MADNESS_ASSERT(r.size()==vout.size());
2986 for (std::size_t i=0; i<vout.size(); ++i) {
2987 vout[i]->coeffs.replace(key, nodeT(coeffT(values2coeffs(key, r[i]),targs),false));
2988 }
2989 }
2990
2991 /// Inplace operate on many functions (impl's) with an operator within a certain box
2992
2993 /// Assumes all functions have been refined down to the same level
2994 /// @param[in] op the operator
2995 /// @param[in] vin the vector of function impl's on which to be operated
2996 /// @param[out] vout the resulting vector of function impl's
2997 template <typename opT>
2998 void multi_to_multi_op_values(const opT& op, const std::vector<implT*>& vin,
2999 std::vector<implT*>& vout, const bool fence=true) {
3000 // rough check on refinement level (ignore non-initialized functions
3001 for (std::size_t i=1; i<vin.size(); ++i) {
3002 if (vin[i] and vin[i-1]) {
3003 MADNESS_ASSERT(vin[i]->coeffs.size()==vin[i-1]->coeffs.size());
3004 }
3005 }
3006 typename dcT::iterator end = vin[0]->coeffs.end();
3007 for (typename dcT::iterator it=vin[0]->coeffs.begin(); it!=end; ++it) {
3008 const keyT& key = it->first;
3009 if (it->second.has_coeff())
3010 world.taskq.add(*this, &implT:: template multi_to_multi_op_values_doit<opT>,
3011 key, op, vin, vout);
3012 else {
3013 // fill result functions with empty box in this key
3014 for (implT* it2 : vout) {
3015 it2->coeffs.replace(key, nodeT(coeffT(),true));
3016 }
3017 }
3018 }
3019 if (fence) world.gop.fence();
3020 }
3021
3022 /// Transforms a vector of functions left[i] = sum[j] right[j]*c[j,i] using sparsity
3023 /// @param[in] vright vector of functions (impl's) on which to be transformed
3024 /// @param[in] c the tensor (matrix) transformer
3025 /// @param[in] vleft vector of of the *newly* transformed functions (impl's)
3026 template <typename Q, typename R>
3027 void vtransform(const std::vector< std::shared_ptr< FunctionImpl<R,NDIM> > >& vright,
3028 const Tensor<Q>& c,
3029 const std::vector< std::shared_ptr< FunctionImpl<T,NDIM> > >& vleft,
3030 double tol,
3031 bool fence) {
3032 for (unsigned int j=0; j<vright.size(); ++j) {
3033 world.taskq.add(*this, &implT:: template vtransform_doit<Q,R>, vright[j], copy(c(j,_)), vleft, tol);
3034 }
3035 if (fence)
3036 world.gop.fence();
3037 }
3038
3039 /// Unary operation applied inplace to the values with optional refinement and fence
3040 /// @param[in] op the unary operator for the values
3041 template <typename opT>
3042 void unary_op_value_inplace(const opT& op, bool fence) {
3044 typedef do_unary_op_value_inplace<opT> xopT;
3045 world.taskq.for_each<rangeT,xopT>(rangeT(coeffs.begin(), coeffs.end()), xopT(this,op));
3046 if (fence)
3047 world.gop.fence();
3048 }
3049
3050 // Multiplication assuming same distribution and recursive descent
3051 /// Both left and right functions are in the scaling function basis
3052 /// @param[in] key the key to the current function node (box)
3053 /// @param[in] left the function impl associated with the left function
3054 /// @param[in] lcin the scaling function coefficients associated with the
3055 /// current box in the left function
3056 /// @param[in] vrightin the vector of function impl's associated with
3057 /// the vector of right functions
3058 /// @param[in] vrcin the vector scaling function coefficients associated with the
3059 /// current box in the right functions
3060 /// @param[out] vresultin the vector of resulting functions (impl's)
3061 template <typename L, typename R>
3062 void mulXXveca(const keyT& key,
3063 const FunctionImpl<L,NDIM>* left, const Tensor<L>& lcin,
3064 const std::vector<const FunctionImpl<R,NDIM>*> vrightin,
3065 const std::vector< Tensor<R> >& vrcin,
3066 const std::vector<FunctionImpl<T,NDIM>*> vresultin,
3067 double tol) {
3068 typedef typename FunctionImpl<L,NDIM>::dcT::const_iterator literT;
3069 typedef typename FunctionImpl<R,NDIM>::dcT::const_iterator riterT;
3070
3071 double lnorm = 1e99;
3072 double ldnorm = 1e99;
3073 bool l_is_leaf = false;
3074 Tensor<L> lc = lcin;
3075 literT lit = left->coeffs.find(key).get();
3076
3077 if (lc.size() == 0) {
3078 MADNESS_CHECK(lit != left->coeffs.end());
3079 // redundant form puts coefficients and computed norms on every
3080 // node; without them the screen below reads garbage
3081 MADNESS_CHECK(lit->second.has_coeff());
3082 lnorm = lit->second.get_norm_tree();
3083 ldnorm = lit->second.get_dnorm_tree();
3085 l_is_leaf = !lit->second.has_children();
3086 }
3087 else {
3088 lnorm = lc.normf();
3089 ldnorm = 0.0; // node created to match trees; leaves carry no detail
3090 l_is_leaf = true;
3091 }
3092
3093 // Loop thru RHS functions seeing if anything can be multiplied
3094 std::vector<FunctionImpl<T,NDIM>*> vresult;
3095 std::vector<const FunctionImpl<R,NDIM>*> vright;
3096 std::vector< Tensor<R> > vrc;
3097 vresult.reserve(vrightin.size());
3098 vright.reserve(vrightin.size());
3099 vrc.reserve(vrightin.size());
3100
3101 // fetched at most once and shared by every right function; do_mul only reads it
3102 Tensor<L> lc_shared;
3103 bool lc_shared_set = false;
3104 auto left_coeffs = [&]() -> const Tensor<L>& {
3105 if (!lc_shared_set) {
3106 lc_shared = lc.size() ? lc : lit->second.coeff().full_tensor_copy();
3107 lc_shared_set = true;
3108 }
3109 return lc_shared;
3110 };
3111
3112 for (unsigned int i=0; i<vrightin.size(); ++i) {
3113 FunctionImpl<T,NDIM>* result = vresultin[i];
3114 const FunctionImpl<R,NDIM>* right = vrightin[i];
3115 Tensor<R> rc = vrcin[i];
3116 double rnorm, rdnorm;
3117 riterT rit = right->coeffs.find(key).get();
3118 if (rc.size() == 0) {
3119 MADNESS_CHECK(rit != right->coeffs.end());
3120 MADNESS_CHECK(rit->second.has_coeff());
3121 rnorm = rit->second.get_norm_tree();
3122 rdnorm = rit->second.get_dnorm_tree();
3124 }
3125 else {
3126 rnorm = rc.normf();
3127 rdnorm = 0.0;
3128 }
3129
3130 // the neglected cross terms are below threshold: multiply here (requires redundant form)
3131 if (rnorm*ldnorm + lnorm*rdnorm + ldnorm*rdnorm <= truncate_tol(tol, key)) {
3132 // lc/rc must keep their size for the recursion logic, so pass separate tensors
3133 Tensor<R> rc_data = (rc.size() == 0) ? rit->second.coeff().full_tensor_copy() : rc;
3134 result->task(world.rank(), &implT:: template do_mul<L,R>, key, left_coeffs(), std::make_pair(key,rc_data));
3135 }
3136 else { // Interior node
3137 result->coeffs.replace(key, nodeT(coeffT(),true));
3138 vresult.push_back(result);
3139 vright.push_back(right);
3140 vrc.push_back(rc);
3141 }
3142 }
3143
3144 if (vresult.size()) {
3145 Tensor<L> lss;
3146 if (lc.size() || l_is_leaf) {
3147 Tensor<L> ld(cdata.v2k);
3148 ld(cdata.s0) = left_coeffs()(___);
3149 lss = left->unfilter(ld);
3150 }
3151
3152 // invariant across the child loop below, so look it up once per right function
3153 std::vector<char> r_unfiltered(vresult.size(), 0);
3154 std::vector< Tensor<R> > vrss(vresult.size());
3155 for (unsigned int i=0; i<vresult.size(); ++i) {
3156 riterT rit = vright[i]->coeffs.find(key).get();
3157 // coefficients handed down from the parent stand in for a node
3158 // that need not exist here; without them the node must exist
3159 MADNESS_CHECK(vrc[i].size() || rit != vright[i]->coeffs.end());
3160 if (vrc[i].size() || !rit->second.has_children()) {
3161 MADNESS_CHECK(vrc[i].size() || rit->second.has_coeff());
3162 Tensor<R> rd(cdata.v2k);
3163 rd(cdata.s0) = (vrc[i].size() ? vrc[i] : rit->second.coeff().full_tensor_copy())(___);
3164 vrss[i] = vright[i]->unfilter(rd);
3165 r_unfiltered[i] = 1;
3166 }
3167 }
3168
3169 for (KeyChildIterator<NDIM> kit(key); kit; ++kit) {
3170 const keyT& child = kit.key();
3171 Tensor<L> ll;
3172
3173 std::vector<Slice> cp = child_patch(child);
3174
3175 if (lc.size() || l_is_leaf)
3176 ll = copy(lss(cp));
3177
3178 std::vector< Tensor<R> > vv(vresult.size());
3179 for (unsigned int i=0; i<vresult.size(); ++i) {
3180 if (r_unfiltered[i])
3181 vv[i] = copy(vrss[i](cp));
3182 }
3183
3184 woT::task(coeffs.owner(child), &implT:: template mulXXveca<L,R>, child, left, ll, vright, vv, vresult, tol);
3185 }
3186 }
3187 }
3188
3189 /// Multiplication using recursive descent and assuming same distribution
3190 /// Both left and right functions are in the scaling function basis
3191 /// @param[in] key the key to the current function node (box)
3192 /// @param[in] left the function impl associated with the left function
3193 /// @param[in] lcin the scaling function coefficients associated with the
3194 /// current box in the left function
3195 /// @param[in] right the function impl associated with the right function
3196 /// @param[in] rcin the scaling function coefficients associated with the
3197 /// current box in the right function
3198 template <typename L, typename R>
3199 void mulXXa(const keyT& key,
3200 const FunctionImpl<L,NDIM>* left, const Tensor<L>& lcin,
3201 const FunctionImpl<R,NDIM>* right,const Tensor<R>& rcin,
3202 double tol) {
3203 typedef typename FunctionImpl<L,NDIM>::dcT::const_iterator literT;
3204 typedef typename FunctionImpl<R,NDIM>::dcT::const_iterator riterT;
3205
3206 double lnorm=1e99, rnorm=1e99;
3207
3208 Tensor<L> lc = lcin;
3209 if (lc.size() == 0) {
3210 literT it = left->coeffs.find(key).get();
3211 MADNESS_ASSERT(it != left->coeffs.end());
3212 lnorm = it->second.get_norm_tree();
3213 if (it->second.has_coeff())
3214 lc = it->second.coeff().reconstruct_tensor();
3215 }
3216
3217 Tensor<R> rc = rcin;
3218 if (rc.size() == 0) {
3219 riterT it = right->coeffs.find(key).get();
3220 MADNESS_ASSERT(it != right->coeffs.end());
3221 rnorm = it->second.get_norm_tree();
3222 if (it->second.has_coeff())
3223 rc = it->second.coeff().reconstruct_tensor();
3224 }
3225
3226 // both nodes are leaf nodes: multiply and return
3227 if (rc.size() && lc.size()) { // Yipee!
3228 do_mul<L,R>(key, lc, std::make_pair(key,rc));
3229 return;
3230 }
3231
3232 if (tol) {
3233 if (lc.size())
3234 lnorm = lc.normf(); // Otherwise got from norm tree above
3235 if (rc.size())
3236 rnorm = rc.normf();
3237 if (lnorm*rnorm < truncate_tol(tol, key)) {
3238 coeffs.replace(key, nodeT(coeffT(cdata.vk,targs),false)); // Zero leaf node
3239 return;
3240 }
3241 }
3242
3243 // Recur down
3244 coeffs.replace(key, nodeT(coeffT(),true)); // Interior node
3245
3246 Tensor<L> lss;
3247 if (lc.size()) {
3248 Tensor<L> ld(cdata.v2k);
3249 ld(cdata.s0) = lc(___);
3250 lss = left->unfilter(ld);
3251 }
3252
3253 Tensor<R> rss;
3254 if (rc.size()) {
3255 Tensor<R> rd(cdata.v2k);
3256 rd(cdata.s0) = rc(___);
3257 rss = right->unfilter(rd);
3258 }
3259
3260 for (KeyChildIterator<NDIM> kit(key); kit; ++kit) {
3261 const keyT& child = kit.key();
3262 Tensor<L> ll;
3263 Tensor<R> rr;
3264 if (lc.size())
3265 ll = copy(lss(child_patch(child)));
3266 if (rc.size())
3267 rr = copy(rss(child_patch(child)));
3268
3269 woT::task(coeffs.owner(child), &implT:: template mulXXa<L,R>, child, left, ll, right, rr, tol);
3270 }
3271 }
3272
3273
3274 // Binary operation on values using recursive descent and assuming same distribution
3275 /// Both left and right functions are in the scaling function basis
3276 /// @param[in] key the key to the current function node (box)
3277 /// @param[in] left the function impl associated with the left function
3278 /// @param[in] lcin the scaling function coefficients associated with the
3279 /// current box in the left function
3280 /// @param[in] right the function impl associated with the right function
3281 /// @param[in] rcin the scaling function coefficients associated with the
3282 /// current box in the right function
3283 /// @param[in] op the binary operator
3284 template <typename L, typename R, typename opT>
3285 void binaryXXa(const keyT& key,
3286 const FunctionImpl<L,NDIM>* left, const Tensor<L>& lcin,
3287 const FunctionImpl<R,NDIM>* right,const Tensor<R>& rcin,
3288 const opT& op) {
3289 typedef typename FunctionImpl<L,NDIM>::dcT::const_iterator literT;
3290 typedef typename FunctionImpl<R,NDIM>::dcT::const_iterator riterT;
3291
3292 Tensor<L> lc = lcin;
3293 if (lc.size() == 0) {
3294 literT it = left->coeffs.find(key).get();
3295 MADNESS_ASSERT(it != left->coeffs.end());
3296 if (it->second.has_coeff())
3297 lc = it->second.coeff().reconstruct_tensor();
3298 }
3299
3300 Tensor<R> rc = rcin;
3301 if (rc.size() == 0) {
3302 riterT it = right->coeffs.find(key).get();
3303 MADNESS_ASSERT(it != right->coeffs.end());
3304 if (it->second.has_coeff())
3305 rc = it->second.coeff().reconstruct_tensor();
3306 }
3307
3308 if (rc.size() && lc.size()) { // Yipee!
3309 do_binary_op<L,R>(key, lc, std::make_pair(key,rc), op);
3310 return;
3311 }
3312
3313 // Recur down
3314 coeffs.replace(key, nodeT(coeffT(),true)); // Interior node
3315
3316 Tensor<L> lss;
3317 if (lc.size()) {
3318 Tensor<L> ld(cdata.v2k);
3319 ld(cdata.s0) = lc(___);
3320 lss = left->unfilter(ld);
3321 }
3322
3323 Tensor<R> rss;
3324 if (rc.size()) {
3325 Tensor<R> rd(cdata.v2k);
3326 rd(cdata.s0) = rc(___);
3327 rss = right->unfilter(rd);
3328 }
3329
3330 for (KeyChildIterator<NDIM> kit(key); kit; ++kit) {
3331 const keyT& child = kit.key();
3332 Tensor<L> ll;
3333 Tensor<R> rr;
3334 if (lc.size())
3335 ll = copy(lss(child_patch(child)));
3336 if (rc.size())
3337 rr = copy(rss(child_patch(child)));
3338
3339 woT::task(coeffs.owner(child), &implT:: template binaryXXa<L,R,opT>, child, left, ll, right, rr, op);
3340 }
3341 }
3342
3343 template <typename Q, typename opT>
3345 typedef typename opT::resultT resultT;
3347 opT op;
3348
3353
3354 Tensor<resultT> operator()(const Key<NDIM>& key, const Tensor<Q>& t) const {
3355 Tensor<Q> invalues = impl_func->coeffs2values(key, t);
3356
3357 Tensor<resultT> outvalues = op(key, invalues);
3358
3359 return impl_func->values2coeffs(key, outvalues);
3360 }
3361
3362 template <typename Archive>
3363 void serialize(Archive& ar) {
3364 ar & impl_func & op;
3365 }
3366 };
3367
3368 /// Out of place unary operation on function impl
3369 /// The skeleton algorithm should resemble something like
3370 ///
3371 /// *this = op(*func)
3372 ///
3373 /// @param[in] key the key of the current function node (box)
3374 /// @param[in] func the function impl on which to be operated
3375 /// @param[in] op the unary operator
3376 template <typename Q, typename opT>
3377 void unaryXXa(const keyT& key,
3378 const FunctionImpl<Q,NDIM>* func, const opT& op) {
3379
3380 // const Tensor<Q>& fc = func->coeffs.find(key).get()->second.full_tensor_copy();
3381 const Tensor<Q> fc = func->coeffs.find(key).get()->second.coeff().reconstruct_tensor();
3382
3383 if (fc.size() == 0) {
3384 // Recur down
3385 coeffs.replace(key, nodeT(coeffT(),true)); // Interior node
3386 for (KeyChildIterator<NDIM> kit(key); kit; ++kit) {
3387 const keyT& child = kit.key();
3388 woT::task(coeffs.owner(child), &implT:: template unaryXXa<Q,opT>, child, func, op);
3389 }
3390 }
3391 else {
3392 tensorT t=op(key,fc);
3393 coeffs.replace(key, nodeT(coeffT(t,targs),false)); // Leaf node
3394 }
3395 }
3396
3397 /// Multiplies two functions (impl's) together. Delegates to the mulXXa() method
3398 /// @param[in] left pointer to the left function impl
3399 /// @param[in] right pointer to the right function impl
3400 /// @param[in] tol numerical tolerance
3401 template <typename L, typename R>
3402 void mulXX(const FunctionImpl<L,NDIM>* left, const FunctionImpl<R,NDIM>* right, double tol, bool fence) {
3403 if (world.rank() == coeffs.owner(cdata.key0))
3404 mulXXa(cdata.key0, left, Tensor<L>(), right, Tensor<R>(), tol);
3405 if (fence)
3406 world.gop.fence();
3407
3408 //verify_tree();
3409 }
3410
3411 /// Performs binary operation on two functions (impl's). Delegates to the binaryXXa() method
3412 /// @param[in] left pointer to the left function impl
3413 /// @param[in] right pointer to the right function impl
3414 /// @param[in] op the binary operator
3415 template <typename L, typename R, typename opT>
3417 const opT& op, bool fence) {
3418 if (world.rank() == coeffs.owner(cdata.key0))
3419 binaryXXa(cdata.key0, left, Tensor<L>(), right, Tensor<R>(), op);
3420 if (fence)
3421 world.gop.fence();
3422
3423 //verify_tree();
3424 }
3425
3426 /// Performs unary operation on function impl. Delegates to the unaryXXa() method
3427 /// @param[in] func function impl of the operand
3428 /// @param[in] op the unary operator
3429 template <typename Q, typename opT>
3430 void unaryXX(const FunctionImpl<Q,NDIM>* func, const opT& op, bool fence) {
3431 if (world.rank() == coeffs.owner(cdata.key0))
3432 unaryXXa(cdata.key0, func, op);
3433 if (fence)
3434 world.gop.fence();
3435
3436 //verify_tree();
3437 }
3438
3439 /// Performs unary operation on function impl. Delegates to the unaryXXa() method
3440 /// @param[in] func function impl of the operand
3441 /// @param[in] op the unary operator
3442 template <typename Q, typename opT>
3443 void unaryXXvalues(const FunctionImpl<Q,NDIM>* func, const opT& op, bool fence) {
3444 if (world.rank() == coeffs.owner(cdata.key0))
3446 if (fence)
3447 world.gop.fence();
3448
3449 //verify_tree();
3450 }
3451
3452 /// Multiplies a function (impl) with a vector of functions (impl's). Delegates to the
3453 /// mulXXveca() method.
3454 /// @param[in] left pointer to the left function impl
3455 /// @param[in] vright vector of pointers to the right function impl's
3456 /// @param[in] tol numerical tolerance
3457 /// @param[out] vresult vector of pointers to the resulting function impl's
3458 template <typename L, typename R>
3460 const std::vector<const FunctionImpl<R,NDIM>*>& vright,
3461 const std::vector<FunctionImpl<T,NDIM>*>& vresult,
3462 double tol,
3463 bool fence) {
3464 std::vector< Tensor<R> > vr(vright.size());
3465 if (world.rank() == coeffs.owner(cdata.key0))
3466 mulXXveca(cdata.key0, left, Tensor<L>(), vright, vr, vresult,
3468 if (fence)
3469 world.gop.fence();
3470 }
3471
3473
3474 mutable long box_leaf[1000];
3475 mutable long box_interior[1000];
3476
3477 // horrifically non-scalable
3478 void put_in_box(ProcessID from, long nl, long ni) const;
3479
3480 /// Prints summary of data distribution
3481 void print_info() const;
3482
3483 /// Verify tree is properly constructed ... global synchronization involved
3484
3485 /// If an inconsistency is detected, prints a message describing the error and
3486 /// then throws a madness exception.
3487 ///
3488 /// This is a reasonably quick and scalable operation that is
3489 /// useful for debugging and paranoia.
3490 void verify_tree() const;
3491
3492 /// check that parents and children are consistent
3493
3494 /// will not check proper size of coefficients
3495 /// global communication
3496 bool verify_parents_and_children() const;
3497
3498 /// check that the tree state and the coeffs are consistent
3499
3500 /// will not check existence of children and/or parents
3501 /// no communication
3502 bool verify_tree_state_local() const;
3503
3504 /// Walk up the tree returning pair(key,node) for first node with coefficients
3505
3506 /// Three possibilities.
3507 ///
3508 /// 1) The coeffs are present and returned with the key of the containing node.
3509 ///
3510 /// 2) The coeffs are further up the tree ... the request is forwarded up.
3511 ///
3512 /// 3) The coeffs are futher down the tree ... an empty tensor is returned.
3513 ///
3514 /// !! This routine is crying out for an optimization to
3515 /// manage the number of messages being sent ... presently
3516 /// each parent is fetched 2^(n*d) times where n is the no. of
3517 /// levels between the level of evaluation and the parent.
3518 /// Alternatively, reimplement multiply as a downward tree
3519 /// walk and just pass the parent down. Slightly less
3520 /// parallelism but much less communication.
3521 /// @todo Robert .... help!
3522 void sock_it_to_me(const keyT& key,
3523 const RemoteReference< FutureImpl< std::pair<keyT,coeffT> > >& ref) const;
3524 /// As above, except
3525 /// 3) The coeffs are constructed from the avg of nodes further down the tree
3526 /// @todo Robert .... help!
3527 void sock_it_to_me_too(const keyT& key,
3528 const RemoteReference< FutureImpl< std::pair<keyT,coeffT> > >& ref) const;
3529
3530 /// @todo help!
3532 const keyT& key,
3533 const coordT& plotlo, const coordT& plothi, const std::vector<long>& npt,
3534 bool eval_refine) const;
3535
3536
3537 /// Evaluate a cube/slice of points ... plotlo and plothi are already in simulation coordinates
3538 /// No communications
3539 /// @param[in] plotlo the coordinate of the starting point
3540 /// @param[in] plothi the coordinate of the ending point
3541 /// @param[in] npt the number of points in each dimension
3542 Tensor<T> eval_plot_cube(const coordT& plotlo,
3543 const coordT& plothi,
3544 const std::vector<long>& npt,
3545 const bool eval_refine = false) const;
3546
3547
3548 /// Evaluate function only if point is local returning (true,value); otherwise return (false,0.0)
3549
3550 /// maxlevel is the maximum depth to search down to --- the max local depth can be
3551 /// computed with max_local_depth();
3552 std::pair<bool,T> eval_local_only(const Vector<double,NDIM>& xin, Level maxlevel) ;
3553
3554 /// Allocation-free core of the batched eval_local_only: writes one
3555 /// (local?,value) pair per point, in input order, into results[0..npt).
3556 /// Consecutive points in the same leaf box share that box's descent and
3557 /// coefficient fetch (last-box memoization); each point is evaluated by
3558 /// the same eval_cube on the same tensor as the single-point path, so
3559 /// results are bit-for-bit identical. No communications.
3560 void eval_local_only(const Vector<double,NDIM>* xin, std::size_t npt,
3561 Level maxlevel, std::pair<bool,T>* results);
3562
3563 /// Batched eval_local_only returning a fresh vector (see the pointer
3564 /// core above for semantics).
3565 /// maxlevel is the maximum depth to search down to --- the max local depth can
3566 /// be computed with max_local_depth();
3567 std::vector<std::pair<bool,T>>
3568 eval_local_only(const std::vector<Vector<double,NDIM>>& xin, Level maxlevel) ;
3569
3570
3571 /// Evaluate the function at a point in \em simulation coordinates
3572
3573 /// Only the invoking process will get the result via the
3574 /// remote reference to a future. Active messages may be sent
3575 /// to other nodes.
3576 void eval(const Vector<double,NDIM>& xin,
3577 const keyT& keyin,
3578 const typename Future<T>::remote_refT& ref);
3579
3580 /// Get the depth of the tree at a point in \em simulation coordinates
3581
3582 /// Only the invoking process will get the result via the
3583 /// remote reference to a future. Active messages may be sent
3584 /// to other nodes.
3585 ///
3586 /// This function is a minimally-modified version of eval()
3587 void evaldepthpt(const Vector<double,NDIM>& xin,
3588 const keyT& keyin,
3589 const typename Future<Level>::remote_refT& ref);
3590
3591 /// Get the rank of leaf box of the tree at a point in \em simulation coordinates
3592
3593 /// Only the invoking process will get the result via the
3594 /// remote reference to a future. Active messages may be sent
3595 /// to other nodes.
3596 ///
3597 /// This function is a minimally-modified version of eval()
3598 void evalR(const Vector<double,NDIM>& xin,
3599 const keyT& keyin,
3600 const typename Future<long>::remote_refT& ref);
3601
3602
3603 /// Computes norm of low/high-order polyn. coeffs for autorefinement test
3604
3605 /// t is a k^d tensor. In order to screen the autorefinement
3606 /// during multiplication compute the norms of
3607 /// ... lo ... the block of t for all polynomials of order < k/2
3608 /// ... hi ... the block of t for all polynomials of order >= k/2
3609 ///
3610 /// k=5 0,1,2,3,4 --> 0,1,2 ... 3,4
3611 /// k=6 0,1,2,3,4,5 --> 0,1,2 ... 3,4,5
3612 ///
3613 /// k=number of wavelets, so k=5 means max order is 4, so max exactly
3614 /// representable squarable polynomial is of order 2.
3615 void static tnorm(const tensorT& t, double* lo, double* hi);
3616
3617 void static tnorm(const GenTensor<T>& t, double* lo, double* hi);
3618
3619 void static tnorm(const SVDTensor<T>& t, double* lo, double* hi, const int particle);
3620
3621 // This invoked if node has not been autorefined
3622 void do_square_inplace(const keyT& key);
3623
3624 // This invoked if node has been autorefined
3625 void do_square_inplace2(const keyT& parent, const keyT& child, const tensorT& parent_coeff);
3626
3627 /// Always returns false (for when autorefine is not wanted)
3628 bool noautorefine(const keyT& key, const tensorT& t) const;
3629
3630 /// Returns true if this block of coeffs needs autorefining
3631 bool autorefine_square_test(const keyT& key, const nodeT& t) const;
3632
3633 /// Pointwise squaring of function with optional global fence
3634
3635 /// If not autorefining, local computation only if not fencing.
3636 /// If autorefining, may result in asynchronous communication.
3637 void square_inplace(bool fence);
3638 void abs_inplace(bool fence);
3639 void abs_square_inplace(bool fence);
3640
3641 /// is this the same as trickle_down() ?
3642 void sum_down_spawn(const keyT& key, const coeffT& s);
3643
3644 /// After 1d push operator must sum coeffs down the tree to restore correct scaling function coefficients
3645 void sum_down(bool fence);
3646
3647 /// perform this multiplication: h(1,2) = f(1,2) * g(1)
3648 template<size_t LDIM>
3650
3651 static bool randomize() {return false;}
3655
3656 implT* h; ///< the result function h(1,2) = f(1,2) * g(1)
3659 int particle; ///< if g is g(1) or g(2)
3660
3661 multiply_op() : h(), f(), g(), particle(1) {}
3662
3663 multiply_op(implT* h1, const ctT& f1, const ctL& g1, const int particle1)
3664 : h(h1), f(f1), g(g1), particle(particle1) {};
3665
3666 /// return true if this will be a leaf node
3667
3668 /// use generalization of tnorm for a GenTensor
3669 bool screen(const coeffT& fcoeff, const coeffT& gcoeff, const keyT& key) const {
3671 MADNESS_ASSERT(fcoeff.is_svd_tensor());
3674
3675 double glo=0.0, ghi=0.0, flo=0.0, fhi=0.0;
3676 g.get_impl()->tnorm(gcoeff.get_tensor(), &glo, &ghi);
3677 g.get_impl()->tnorm(fcoeff.get_svdtensor(),&flo,&fhi,particle);
3678
3679 double total_hi=glo*fhi + ghi*flo + fhi*ghi;
3680 return (total_hi<h->truncate_tol(h->get_thresh(),key));
3681
3682 }
3683
3684 /// apply this on a FunctionNode of f and g of Key key
3685
3686 /// @param[in] key key for FunctionNode in f and g, (g: broken into particles)
3687 /// @return <this node is a leaf, coefficients of this node>
3688 std::pair<bool,coeffT> operator()(const Key<NDIM>& key) const {
3689
3690 // bool is_leaf=(not fdatum.second.has_children());
3691 // if (not is_leaf) return std::pair<bool,coeffT> (is_leaf,coeffT());
3692
3693 // break key into particles (these are the child keys, with f/gdatum come the parent keys)
3694 Key<LDIM> key1,key2;
3695 key.break_apart(key1,key2);
3696 const Key<LDIM> gkey= (particle==1) ? key1 : key2;
3697
3698 // get coefficients of the actual FunctionNode
3699 coeffT coeff1=f.get_impl()->parent_to_child(f.coeff(),f.key(),key);
3700 coeff1.normalize();
3701 const coeffT coeff2=g.get_impl()->parent_to_child(g.coeff(),g.key(),gkey);
3702
3703 // multiplication is done in TT_2D
3704 coeffT coeff1_2D=coeff1.convert(TensorArgs(h->get_thresh(),TT_2D));
3705 coeff1_2D.normalize();
3706
3707 bool is_leaf=screen(coeff1_2D,coeff2,key);
3708 if (key.level()<2) is_leaf=false;
3709
3710 coeffT hcoeff;
3711 if (is_leaf) {
3712
3713 // convert coefficients to values
3714 coeffT hvalues=f.get_impl()->coeffs2values(key,coeff1_2D);
3715 coeffT gvalues=g.get_impl()->coeffs2values(gkey,coeff2);
3716
3717 // perform multiplication
3718 coeffT result_val=h->multiply(hvalues,gvalues,particle-1);
3719
3720 hcoeff=h->values2coeffs(key,result_val);
3721
3722 // conversion on coeffs, not on values, because it implies truncation!
3723 if (not hcoeff.is_of_tensortype(h->get_tensor_type()))
3724 hcoeff=hcoeff.convert(h->get_tensor_args());
3725 }
3726
3727 return std::pair<bool,coeffT> (is_leaf,hcoeff);
3728 }
3729
3730 this_type make_child(const keyT& child) const {
3731
3732 // break key into particles
3733 Key<LDIM> key1, key2;
3734 child.break_apart(key1,key2);
3735 const Key<LDIM> gkey= (particle==1) ? key1 : key2;
3736
3737 return this_type(h,f.make_child(child),g.make_child(gkey),particle);
3738 }
3739
3741 Future<ctT> f1=f.activate();
3743 return h->world.taskq.add(detail::wrap_mem_fn(*const_cast<this_type *> (this),
3744 &this_type::forward_ctor),h,f1,g1,particle);
3745 }
3746
3747 this_type forward_ctor(implT* h1, const ctT& f1, const ctL& g1, const int particle) {
3748 return this_type(h1,f1,g1,particle);
3749 }
3750
3751 template <typename Archive> void serialize(const Archive& ar) {
3752 ar & h & f & g & particle;
3753 }
3754 };
3755
3756
3757 /// add two functions f and g: result=alpha * f + beta * g
3758 struct add_op {
3759
3762
3763 bool randomize() const {return false;}
3764
3765 /// tracking coeffs of first and second addend
3767 /// prefactor for f, g
3768 double alpha, beta;
3769
3770 add_op() = default;
3771 add_op(const ctT& f, const ctT& g, const double alpha, const double beta)
3772 : f(f), g(g), alpha(alpha), beta(beta){}
3773
3774 /// if we are at the bottom of the trees, return the sum of the coeffs
3775 std::pair<bool,coeffT> operator()(const keyT& key) const {
3776
3777 bool is_leaf=(f.is_leaf() and g.is_leaf());
3778 if (not is_leaf) return std::pair<bool,coeffT> (is_leaf,coeffT());
3779
3780 coeffT fcoeff=f.get_impl()->parent_to_child(f.coeff(),f.key(),key);
3781 coeffT gcoeff=g.get_impl()->parent_to_child(g.coeff(),g.key(),key);
3782 coeffT hcoeff=copy(fcoeff);
3783 hcoeff.gaxpy(alpha,gcoeff,beta);
3784 hcoeff.reduce_rank(f.get_impl()->get_tensor_args().thresh);
3785 return std::pair<bool,coeffT> (is_leaf,hcoeff);
3786 }
3787
3788 this_type make_child(const keyT& child) const {
3789 return this_type(f.make_child(child),g.make_child(child),alpha,beta);
3790 }
3791
3792 /// retrieve the coefficients (parent coeffs might be remote)
3794 Future<ctT> f1=f.activate();
3795 Future<ctT> g1=g.activate();
3796 return f.get_impl()->world.taskq.add(detail::wrap_mem_fn(*const_cast<this_type *> (this),
3798 }
3799
3800 /// taskq-compatible ctor
3801 this_type forward_ctor(const ctT& f1, const ctT& g1, const double alpha, const double beta) {
3802 return this_type(f1,g1,alpha,beta);
3803 }
3804
3805 template <typename Archive> void serialize(const Archive& ar) {
3806 ar & f & g & alpha & beta;
3807 }
3808
3809 };
3810
3811 /// multiply f (a pair function of NDIM) with an orbital g (LDIM=NDIM/2)
3812
3813 /// as in (with h(1,2)=*this) : h(1,2) = g(1) * f(1,2)
3814 /// use tnorm as a measure to determine if f (=*this) must be refined
3815 /// @param[in] f the NDIM function f=f(1,2)
3816 /// @param[in] g the LDIM function g(1) (or g(2))
3817 /// @param[in] particle 1 or 2, as in g(1) or g(2)
3818 template<size_t LDIM>
3819 void multiply(const implT* f, const FunctionImpl<T,LDIM>* g, const int particle) {
3820
3823
3824 typedef multiply_op<LDIM> coeff_opT;
3825 coeff_opT coeff_op(this,ff,gg,particle);
3826
3827 typedef insert_op<T,NDIM> apply_opT;
3828 apply_opT apply_op(this);
3829
3830 keyT key0=f->cdata.key0;
3831 if (world.rank() == coeffs.owner(key0)) {
3833 woT::task(p, &implT:: template forward_traverse<coeff_opT,apply_opT>, coeff_op, apply_op, key0);
3834 }
3835
3837 }
3838
3839 /// Hartree product of two LDIM functions to yield a NDIM = 2*LDIM function
3840 template<size_t LDIM, typename leaf_opT>
3841 struct hartree_op {
3842 bool randomize() const {return false;}
3843
3846
3847 implT* result; ///< where to construct the pair function
3848 ctL p1, p2; ///< tracking coeffs of the two lo-dim functions
3849 leaf_opT leaf_op; ///< determine if a given node will be a leaf node
3850
3851 // ctor
3853 hartree_op(implT* result, const ctL& p11, const ctL& p22, const leaf_opT& leaf_op)
3854 : result(result), p1(p11), p2(p22), leaf_op(leaf_op) {
3855 MADNESS_ASSERT(LDIM+LDIM==NDIM);
3856 }
3857
3858 std::pair<bool,coeffT> operator()(const Key<NDIM>& key) const {
3859
3860 // break key into particles (these are the child keys, with datum1/2 come the parent keys)
3861 Key<LDIM> key1,key2;
3862 key.break_apart(key1,key2);
3863
3864 // this returns the appropriate NS coeffs for key1 and key2 resp.
3865 const coeffT fcoeff=p1.coeff(key1);
3866 const coeffT gcoeff=p2.coeff(key2);
3867 bool is_leaf=leaf_op(key,fcoeff.full_tensor(),gcoeff.full_tensor());
3868 if (not is_leaf) return std::pair<bool,coeffT> (is_leaf,coeffT());
3869
3870 // extract the sum coeffs from the NS coeffs
3871 const coeffT s1=fcoeff(p1.get_impl()->cdata.s0);
3872 const coeffT s2=gcoeff(p2.get_impl()->cdata.s0);
3873
3874 // new coeffs are simply the hartree/kronecker/outer product --
3875 coeffT coeff=outer(s1,s2,result->get_tensor_args());
3876 // no post-determination
3877 // is_leaf=leaf_op(key,coeff);
3878 return std::pair<bool,coeffT>(is_leaf,coeff);
3879 }
3880
3881 this_type make_child(const keyT& child) const {
3882
3883 // break key into particles
3884 Key<LDIM> key1, key2;
3885 child.break_apart(key1,key2);
3886
3887 return this_type(result,p1.make_child(key1),p2.make_child(key2),leaf_op);
3888 }
3889
3891 Future<ctL> p11=p1.activate();
3892 Future<ctL> p22=p2.activate();
3893 return result->world.taskq.add(detail::wrap_mem_fn(*const_cast<this_type *> (this),
3894 &this_type::forward_ctor),result,p11,p22,leaf_op);
3895 }
3896
3897 this_type forward_ctor(implT* result1, const ctL& p11, const ctL& p22, const leaf_opT& leaf_op) {
3898 return this_type(result1,p11,p22,leaf_op);
3899 }
3900
3901 template <typename Archive> void serialize(const Archive& ar) {
3902 ar & result & p1 & p2 & leaf_op;
3903 }
3904 };
3905
3906 /// traverse a non-existing tree
3907
3908 /// part II: activate coeff_op, i.e. retrieve all the necessary remote boxes (communication)
3909 /// @param[in] coeff_op operator making the coefficients that needs activation
3910 /// @param[in] apply_op just passing thru
3911 /// @param[in] key the key we are working on
3912 template<typename coeff_opT, typename apply_opT>
3913 void forward_traverse(const coeff_opT& coeff_op, const apply_opT& apply_op, const keyT& key) const {
3915 Future<coeff_opT> active_coeff=coeff_op.activate();
3916 woT::task(world.rank(), &implT:: template traverse_tree<coeff_opT,apply_opT>, active_coeff, apply_op, key);
3917 }
3918
3919
3920 /// traverse a non-existing tree
3921
3922 /// part I: make the coefficients, process them and continue the recursion if necessary
3923 /// @param[in] coeff_op operator making the coefficients and determining them being leaves
3924 /// @param[in] apply_op operator processing the coefficients
3925 /// @param[in] key the key we are currently working on
3926 template<typename coeff_opT, typename apply_opT>
3927 void traverse_tree(const coeff_opT& coeff_op, const apply_opT& apply_op, const keyT& key) const {
3929
3930 typedef typename std::pair<bool,coeffT> argT;
3931 const argT arg=coeff_op(key);
3932 apply_op.operator()(key,arg.second,arg.first);
3933
3934 const bool has_children=(not arg.first);
3935 if (has_children) {
3936 for (KeyChildIterator<NDIM> kit(key); kit; ++kit) {
3937 const keyT& child=kit.key();
3938 coeff_opT child_op=coeff_op.make_child(child);
3939 // spawn activation where child is local
3940 ProcessID p=coeffs.owner(child);
3941
3942 void (implT::*ft)(const coeff_opT&, const apply_opT&, const keyT&) const = &implT::forward_traverse<coeff_opT,apply_opT>;
3943
3944 woT::task(p, ft, child_op, apply_op, child);
3945 }
3946 }
3947 }
3948
3949
3950 /// given two functions of LDIM, perform the Hartree/Kronecker/outer product
3951
3952 /// |Phi(1,2)> = |phi(1)> x |phi(2)>
3953 /// @param[in] p1 FunctionImpl of particle 1
3954 /// @param[in] p2 FunctionImpl of particle 2
3955 /// @param[in] leaf_op operator determining of a given box will be a leaf
3956 template<std::size_t LDIM, typename leaf_opT>
3957 void hartree_product(const std::vector<std::shared_ptr<FunctionImpl<T,LDIM>>> p1,
3958 const std::vector<std::shared_ptr<FunctionImpl<T,LDIM>>> p2,
3959 const leaf_opT& leaf_op, bool fence) {
3960 MADNESS_CHECK_THROW(p1.size()==p2.size(),"hartree_product: p1 and p2 must have the same size");
3961 for (auto& p : p1) MADNESS_CHECK(p->is_nonstandard() or p->is_nonstandard_with_leaves());
3962 for (auto& p : p2) MADNESS_CHECK(p->is_nonstandard() or p->is_nonstandard_with_leaves());
3963
3964 const keyT key0=cdata.key0;
3965
3966 for (std::size_t i=0; i<p1.size(); ++i) {
3967 if (world.rank() == this->get_coeffs().owner(key0)) {
3968
3969 // prepare the CoeffTracker
3970 CoeffTracker<T,LDIM> iap1(p1[i].get());
3971 CoeffTracker<T,LDIM> iap2(p2[i].get());
3972
3973 // the operator making the coefficients
3974 typedef hartree_op<LDIM,leaf_opT> coeff_opT;
3975 coeff_opT coeff_op(this,iap1,iap2,leaf_op);
3976
3977 // this operator simply inserts the coeffs into this' tree
3978// typedef insert_op<T,NDIM> apply_opT;
3979 typedef accumulate_op<T,NDIM> apply_opT;
3980 apply_opT apply_op(this);
3981
3982 woT::task(world.rank(), &implT:: template forward_traverse<coeff_opT,apply_opT>,
3983 coeff_op, apply_op, cdata.key0);
3984
3985 }
3986 }
3987
3989 if (fence) world.gop.fence();
3990 }
3991
3992
3993 template <typename opT, typename R>
3994 void
3996 const opT* op = pop.ptr;
3997 const Level n = key.level();
3998 const double cnorm = c.normf();
3999 const double tol = truncate_tol(thresh, key)*0.1; // ??? why this value????
4000
4002 const Translation lold = lnew[axis];
4003 const Translation maxs = Translation(1)<<n;
4004
4005 int nsmall = 0; // Counts neglected blocks to terminate s loop
4006 for (Translation s=0; s<maxs; ++s) {
4007 int maxdir = s ? 1 : -1;
4008 for (int direction=-1; direction<=maxdir; direction+=2) {
4009 lnew[axis] = lold + direction*s;
4010 if (lnew[axis] >= 0 && lnew[axis] < maxs) { // NON-ZERO BOUNDARY CONDITIONS IGNORED HERE !!!!!!!!!!!!!!!!!!!!
4011 const Tensor<typename opT::opT>& r = op->rnlij(n, s*direction, true);
4012 double Rnorm = r.normf();
4013
4014 if (Rnorm == 0.0) {
4015 return; // Hard zero means finished!
4016 }
4017
4018 if (s <= 1 || r.normf()*cnorm > tol) { // Always do kernel and neighbor
4019 nsmall = 0;
4020 tensorT result = transform_dir(c,r,axis);
4021
4022 if (result.normf() > tol*0.3) {
4023 Key<NDIM> dest(n,lnew);
4024 coeffs.task(dest, &nodeT::accumulate2, result, coeffs, dest, TaskAttributes::hipri());
4025 }
4026 }
4027 else {
4028 ++nsmall;
4029 }
4030 }
4031 else {
4032 ++nsmall;
4033 }
4034 }
4035 if (nsmall >= 4) {
4036 // If have two negligble blocks in
4037 // succession in each direction interpret
4038 // this as the operator being zero beyond
4039 break;
4040 }
4041 }
4042 }
4043
4044 template <typename opT, typename R>
4045 void
4046 apply_1d_realspace_push(const opT& op, const FunctionImpl<R,NDIM>* f, int axis, bool fence) {
4047 MADNESS_ASSERT(!f->is_compressed());
4048
4049 typedef typename FunctionImpl<R,NDIM>::dcT::const_iterator fiterT;
4050 typedef FunctionNode<R,NDIM> fnodeT;
4051 fiterT end = f->coeffs.end();
4052 ProcessID me = world.rank();
4053 for (fiterT it=f->coeffs.begin(); it!=end; ++it) {
4054 const fnodeT& node = it->second;
4055 if (node.has_coeff()) {
4056 const keyT& key = it->first;
4057 const Tensor<R>& c = node.coeff().full_tensor_copy();
4058 woT::task(me, &implT:: template apply_1d_realspace_push_op<opT,R>,
4060 }
4061 }
4062 if (fence) world.gop.fence();
4063 }
4064
4066 const implT* f,
4067 const keyT& key,
4068 const std::pair<keyT,coeffT>& left,
4069 const std::pair<keyT,coeffT>& center,
4070 const std::pair<keyT,coeffT>& right);
4071
4072 void do_diff1(const DerivativeBase<T,NDIM>* D,
4073 const implT* f,
4074 const keyT& key,
4075 const std::pair<keyT,coeffT>& left,
4076 const std::pair<keyT,coeffT>& center,
4077 const std::pair<keyT,coeffT>& right);
4078
4079 // Called by result function to differentiate f
4080 void diff(const DerivativeBase<T,NDIM>* D, const implT* f, bool fence);
4081
4082 /// Returns key of general neighbor enforcing BC
4083
4084 /// Out of volume keys are mapped to enforce the BC as follows.
4085 /// * Periodic BC map back into the volume and return the correct key
4086 /// * non-periodic BC - returns invalid() to indicate out of volume
4087 keyT neighbor(const keyT& key, const keyT& disp, const array_of_bools<NDIM>& is_periodic) const;
4088
4089 /// Returns key of general neighbor that resides in-volume
4090
4091 /// Out of volume keys are mapped to invalid()
4092 keyT neighbor_in_volume(const keyT& key, const keyT& disp) const;
4093
4094 /// find_me. Called by diff_bdry to get coefficients of boundary function
4095 Future< std::pair<keyT,coeffT> > find_me(const keyT& key) const;
4096
4097 /// return the a std::pair<key, node>, which MUST exist
4098 std::pair<Key<NDIM>,ShallowNode<T,NDIM> > find_datum(keyT key) const;
4099
4100 /// multiply the ket with a one-electron potential rr(1,2)= f(1,2)*g(1)
4101
4102 /// @param[in] val_ket function values of f(1,2)
4103 /// @param[in] val_pot function values of g(1)
4104 /// @param[in] particle if 0 then g(1), if 1 then g(2)
4105 /// @return the resulting function values
4106 coeffT multiply(const coeffT& val_ket, const coeffT& val_pot, int particle) const;
4107
4108
4109 /// given several coefficient tensors, assemble a result tensor
4110
4111 /// the result looks like: (v(1,2) + v(1) + v(2)) |ket(1,2)>
4112 /// or (v(1,2) + v(1) + v(2)) |p(1) p(2)>
4113 /// i.e. coefficients for the ket and coefficients for the two particles are
4114 /// mutually exclusive. All potential terms are optional, just pass in empty coeffs.
4115 /// @param[in] key the key of the FunctionNode to which these coeffs belong
4116 /// @param[in] coeff_ket coefficients of the ket
4117 /// @param[in] vpotential1 function values of the potential for particle 1
4118 /// @param[in] vpotential2 function values of the potential for particle 2
4119 /// @param[in] veri function values for the 2-particle potential
4120 coeffT assemble_coefficients(const keyT& key, const coeffT& coeff_ket,
4121 const coeffT& vpotential1, const coeffT& vpotential2,
4122 const tensorT& veri) const;
4123
4124
4125
4126 template<std::size_t LDIM>
4130 double error=0.0;
4131 double lo=0.0, hi=0.0, lo1=0.0, hi1=0.0, lo2=0.0, hi2=0.0;
4132
4134 pointwise_multiplier(const Key<NDIM> key, const coeffT& clhs) : coeff_lhs(clhs) {
4136 val_lhs=fcf.coeffs2values(key,coeff_lhs);
4137 error=0.0;
4139 if (coeff_lhs.is_svd_tensor()) {
4142 }
4143 }
4144
4145 /// multiply values of rhs and lhs, result on rhs, rhs and lhs are of the same dimensions
4146 tensorT operator()(const Key<NDIM> key, const tensorT& coeff_rhs) {
4147
4148 MADNESS_ASSERT(coeff_rhs.dim(0)==coeff_lhs.dim(0));
4150
4151 // the tnorm estimate is not tight enough to be efficient, better use oversampling
4152 bool use_tnorm=false;
4153 if (use_tnorm) {
4154 double rlo, rhi;
4155 implT::tnorm(coeff_rhs,&rlo,&rhi);
4156 error = hi*rlo + rhi*lo + rhi*hi;
4157 tensorT val_rhs=fcf.coeffs2values(key, coeff_rhs);
4158 val_rhs.emul(val_lhs.full_tensor());
4159 return fcf.values2coeffs(key,val_rhs);
4160 } else { // use quadrature of order k+1
4161
4162 auto& cdata=FunctionCommonData<T,NDIM>::get(coeff_rhs.dim(0)); // npt=k+1
4163 auto& cdata_npt=FunctionCommonData<T,NDIM>::get(coeff_rhs.dim(0)+oversampling); // npt=k+1
4164 FunctionCommonFunctionality<T,NDIM> fcf_hi_npt(cdata_npt);
4165
4166 // coeffs2values for rhs: k -> npt=k+1
4167 tensorT coeff1(cdata_npt.vk);
4168 coeff1(cdata.s0)=coeff_rhs; // s0 is smaller than vk!
4169 tensorT val_rhs_k1=fcf_hi_npt.coeffs2values(key,coeff1);
4170
4171 // coeffs2values for lhs: k -> npt=k+1
4172 tensorT coeff_lhs_k1(cdata_npt.vk);
4173 coeff_lhs_k1(cdata.s0)=std::as_const(coeff_lhs).full_tensor();
4174 tensorT val_lhs_k1=fcf_hi_npt.coeffs2values(key,coeff_lhs_k1);
4175
4176 // multiply
4177 val_lhs_k1.emul(val_rhs_k1);
4178
4179 // values2coeffs: npt = k+1-> k
4180 tensorT result1=fcf_hi_npt.values2coeffs(key,val_lhs_k1);
4181
4182 // extract coeffs up to k
4183 tensorT result=copy(result1(cdata.s0));
4184 result1(cdata.s0)=0.0;
4185 error=result1.normf();
4186 return result;
4187 }
4188 }
4189
4190 /// multiply values of rhs and lhs, result on rhs, rhs and lhs are of differnet dimensions
4191 coeffT operator()(const Key<NDIM> key, const tensorT& coeff_rhs, const int particle) {
4192 Key<LDIM> key1, key2;
4193 key.break_apart(key1,key2);
4194 const long k=coeff_rhs.dim(0);
4196 auto& cdata_lowdim=FunctionCommonData<T,LDIM>::get(k);
4197 FunctionCommonFunctionality<T,LDIM> fcf_lo(cdata_lowdim);
4201
4202
4203 // make hi-dim values from lo-dim coeff_rhs on npt grid points
4204 tensorT ones=tensorT(fcf_lo_npt.cdata.vk);
4205 ones=1.0;
4206
4207 tensorT coeff_rhs_npt1(fcf_lo_npt.cdata.vk);
4208 coeff_rhs_npt1(fcf_lo.cdata.s0)=coeff_rhs;
4209 tensorT val_rhs_npt1=fcf_lo_npt.coeffs2values(key1,coeff_rhs_npt1);
4210
4211 TensorArgs targs(-1.0,TT_2D);
4212 coeffT val_rhs;
4213 if (particle==1) val_rhs=outer(val_rhs_npt1,ones,targs);
4214 if (particle==2) val_rhs=outer(ones,val_rhs_npt1,targs);
4215
4216 // make values from hi-dim coeff_lhs on npt grid points
4217 coeffT coeff_lhs_k1(fcf_hi_npt.cdata.vk,coeff_lhs.tensor_type());
4218 coeff_lhs_k1(fcf_hi.cdata.s0)+=coeff_lhs;
4219 coeffT val_lhs_npt=fcf_hi_npt.coeffs2values(key,coeff_lhs_k1);
4220
4221 // multiply
4222 val_lhs_npt.emul(val_rhs);
4223
4224 // values2coeffs: npt = k+1-> k
4225 coeffT result1=fcf_hi_npt.values2coeffs(key,val_lhs_npt);
4226
4227 // extract coeffs up to k
4228 coeffT result=copy(result1(cdata.s0));
4229 result1(cdata.s0)=0.0;
4230 error=result1.normf();
4231 return result;
4232 }
4233
4234 template <typename Archive> void serialize(const Archive& ar) {
4235 ar & error & lo & lo1 & lo2 & hi & hi1& hi2 & val_lhs & coeff_lhs;
4236 }
4237
4238
4239 };
4240
4241 /// given a ket and the 1- and 2-electron potentials, construct the function V phi
4242
4243 /// small memory footstep version of Vphi_op: use the NS form to have information
4244 /// about parent and children to determine if a box is a leaf. This will require
4245 /// compression of the constituent functions, which will lead to more memory usage
4246 /// there, but will avoid oversampling of the result function.
4247 template<typename opT, size_t LDIM>
4248 struct Vphi_op_NS {
4249
4250 bool randomize() const {return true;}
4251
4255
4256 implT* result; ///< where to construct Vphi, no need to track parents
4257 opT leaf_op; ///< deciding if a given FunctionNode will be a leaf node
4258 ctT iaket; ///< the ket of a pair function (exclusive with p1, p2)
4259 ctL iap1, iap2; ///< the particles 1 and 2 (exclusive with ket)
4260 ctL iav1, iav2; ///< potentials for particles 1 and 2
4261 const implT* eri; ///< 2-particle potential, must be on-demand
4262
4263 bool have_ket() const {return iaket.get_impl();}
4264 bool have_v1() const {return iav1.get_impl();}
4265 bool have_v2() const {return iav2.get_impl();}
4266 bool have_eri() const {return eri;}
4267
4268 void accumulate_into_result(const Key<NDIM>& key, const coeffT& coeff) const {
4270 }
4271
4272 // ctor
4274 Vphi_op_NS(implT* result, const opT& leaf_op, const ctT& iaket,
4275 const ctL& iap1, const ctL& iap2, const ctL& iav1, const ctL& iav2,
4276 const implT* eri)
4278 , iav1(iav1), iav2(iav2), eri(eri) {
4279
4280 // 2-particle potential must be on-demand
4282 }
4283
4284 /// make and insert the coefficients into result's tree
4285 std::pair<bool,coeffT> operator()(const Key<NDIM>& key) const {
4286
4288 if(leaf_op.do_pre_screening()){
4289 // this means that we only construct the boxes which are leaf boxes from the other function in the leaf_op
4290 if(leaf_op.pre_screening(key)){
4291 // construct sum_coefficients, insert them and leave
4292 auto [sum_coeff, error]=make_sum_coeffs(key);
4293 accumulate_into_result(key,sum_coeff);
4294 return std::pair<bool,coeffT> (true,coeffT());
4295 }else{
4296 return continue_recursion(std::vector<bool>(1<<NDIM,false),tensorT(),key);
4297 }
4298 }
4299
4300 // this means that the function has to be completely constructed and not mirrored by another function
4301
4302 // if the initial level is not reached then this must not be a leaf box
4303 size_t il = result->get_initial_level();
4305 if(key.level()<int(il)){
4306 return continue_recursion(std::vector<bool>(1<<NDIM,false),tensorT(),key);
4307 }
4308 // if further refinement is needed (because we are at a special box, special point)
4309 // and the special_level is not reached then this must not be a leaf box
4310 if(key.level()<result->get_special_level() and leaf_op.special_refinement_needed(key)){
4311 return continue_recursion(std::vector<bool>(1<<NDIM,false),tensorT(),key);
4312 }
4313
4314 auto [sum_coeff,error]=make_sum_coeffs(key);
4315
4316 // coeffs are leaf (for whatever reason), insert into tree and stop recursion
4317 if(leaf_op.post_screening(key,sum_coeff)){
4318 accumulate_into_result(key,sum_coeff);
4319 return std::pair<bool,coeffT> (true,coeffT());
4320 }
4321
4322 // coeffs are accurate, insert into tree and stop recursion
4323 if(error<result->truncate_tol(result->get_thresh(),key)){
4324 accumulate_into_result(key,sum_coeff);
4325 return std::pair<bool,coeffT> (true,coeffT());
4326 }
4327
4328 // coeffs are inaccurate, continue recursion
4329 std::vector<bool> child_is_leaf(1<<NDIM,false);
4330 return continue_recursion(child_is_leaf,tensorT(),key);
4331 }
4332
4333
4334 /// loop over all children and either insert their sum coeffs or continue the recursion
4335
4336 /// @param[in] child_is_leaf for each child: is it a leaf?
4337 /// @param[in] coeffs coefficient tensor with 2^N sum coeffs (=unfiltered NS coeffs)
4338 /// @param[in] key the key for the NS coeffs (=parent key of the children)
4339 /// @return to avoid recursion outside this return: std::pair<is_leaf,coeff> = true,coeffT()
4340 std::pair<bool,coeffT> continue_recursion(const std::vector<bool> child_is_leaf,
4341 const tensorT& coeffs, const keyT& key) const {
4342 std::size_t i=0;
4343 for (KeyChildIterator<NDIM> kit(key); kit; ++kit, ++i) {
4344 keyT child=kit.key();
4345 bool is_leaf=child_is_leaf[i];
4346
4347 if (is_leaf) {
4348 // insert the sum coeffs
4350 iop(child,coeffT(copy(coeffs(result->child_patch(child))),result->get_tensor_args()),is_leaf);
4351 } else {
4352 this_type child_op=this->make_child(child);
4353 noop<T,NDIM> no;
4354 // spawn activation where child is local
4355 ProcessID p=result->get_coeffs().owner(child);
4356
4357 void (implT::*ft)(const Vphi_op_NS<opT,LDIM>&, const noop<T,NDIM>&, const keyT&) const = &implT:: template forward_traverse< Vphi_op_NS<opT,LDIM>, noop<T,NDIM> >;
4358 result->task(p, ft, child_op, no, child);
4359 }
4360 }
4361 // return e sum coeffs; also return always is_leaf=true:
4362 // the recursion is continued within this struct, not outside in traverse_tree!
4363 return std::pair<bool,coeffT> (true,coeffT());
4364 }
4365
4366 tensorT eri_coeffs(const keyT& key) const {
4369 if (eri->get_functor()->provides_coeff()) {
4370 return eri->get_functor()->coeff(key).full_tensor();
4371 } else {
4372 tensorT val_eri(eri->cdata.vk);
4373 eri->fcube(key,*(eri->get_functor()),eri->cdata.quad_x,val_eri);
4374 return eri->values2coeffs(key,val_eri);
4375 }
4376 }
4377
4378 /// the error is computed from the d coefficients of the constituent functions
4379
4380 /// the result is h_n = P_n(f g), computed as h_n \approx Pn(f_n g_n)
4381 /// its error is therefore
4382 /// h_n = (f g)_n = ((Pn(f) + Qn(f)) (Pn(g) + Qn(g))
4383 /// = Pn(fn gn) + Qn(fn gn) + Pn(f) Qn(g) + Qn(f) Pn(g) + Qn(f) Pn(g)
4384 /// the first term is what we compute, the second term is estimated by tnorm (in another function),
4385 /// the third to last terms are estimated in this function by e.g.: Qn(f)Pn(g) < ||Qn(f)|| ||Pn(g)||
4387 const tensorT& ceri) const {
4388 double error = 0.0;
4389 Key<LDIM> key1, key2;
4390 key.break_apart(key1,key2);
4391
4392 PROFILE_BLOCK(compute_error);
4393 double dnorm_ket, snorm_ket;
4394 if (have_ket()) {
4395 snorm_ket=iaket.coeff(key).normf();
4396 dnorm_ket=iaket.dnorm(key);
4397 } else {
4398 double s1=iap1.coeff(key1).normf();
4399 double s2=iap2.coeff(key2).normf();
4400 double d1=iap1.dnorm(key1);
4401 double d2=iap2.dnorm(key2);
4402 snorm_ket=s1*s2;
4403 dnorm_ket=s1*d2 + s2*d1 + d1*d2;
4404 }
4405
4406 if (have_v1()) {
4407 double snorm=iav1.coeff(key1).normf();
4408 double dnorm=iav1.dnorm(key1);
4409 error+=snorm*dnorm_ket + dnorm*snorm_ket + dnorm*dnorm_ket;
4410 }
4411 if (have_v2()) {
4412 double snorm=iav2.coeff(key2).normf();
4413 double dnorm=iav2.dnorm(key2);
4414 error+=snorm*dnorm_ket + dnorm*snorm_ket + dnorm*dnorm_ket;
4415 }
4416 if (have_eri()) {
4417 tensorT s_coeffs=ceri(result->cdata.s0);
4418 double snorm=s_coeffs.normf();
4419 tensorT d=copy(ceri);
4420 d(result->cdata.s0)=0.0;
4421 double dnorm=d.normf();
4422 error+=snorm*dnorm_ket + dnorm*snorm_ket + dnorm*dnorm_ket;
4423 }
4424
4425 bool no_potential=not ((have_v1() or have_v2() or have_eri()));
4426 if (no_potential) {
4427 error=dnorm_ket;
4428 }
4429 return error;
4430 }
4431
4432 /// make the sum coeffs for key
4433 std::pair<coeffT,double> make_sum_coeffs(const keyT& key) const {
4435 // break key into particles
4436 Key<LDIM> key1, key2;
4437 key.break_apart(key1,key2);
4438
4439 // bool printme=(int(key.translation()[0])==int(std::pow(key.level(),2)/2)) and
4440 // (int(key.translation()[1])==int(std::pow(key.level(),2)/2)) and
4441 // (int(key.translation()[2])==int(std::pow(key.level(),2)/2));
4442
4443// printme=false;
4444
4445 // get/make all coefficients
4446 const coeffT coeff_ket = (iaket.get_impl()) ? iaket.coeff(key)
4447 : outer(iap1.coeff(key1),iap2.coeff(key2),result->get_tensor_args());
4448 const coeffT cpot1 = (have_v1()) ? iav1.coeff(key1) : coeffT();
4449 const coeffT cpot2 = (have_v2()) ? iav2.coeff(key2) : coeffT();
4450 const tensorT ceri = (have_eri()) ? eri_coeffs(key) : tensorT();
4451
4452 // compute first part of the total error
4453 double refine_error=compute_error_from_inaccurate_refinement(key,ceri);
4454 double error=refine_error;
4455
4456 // prepare the multiplication
4457 pointwise_multiplier<LDIM> pm(key,coeff_ket);
4458
4459 // perform the multiplication, compute tnorm part of the total error
4460 coeffT cresult(result->cdata.vk,result->get_tensor_args());
4461 if (have_v1()) {
4462 cresult+=pm(key,cpot1.get_tensor(),1);
4463 error+=pm.error;
4464 }
4465 if (have_v2()) {
4466 cresult+=pm(key,cpot2.get_tensor(),2);
4467 error+=pm.error;
4468 }
4469
4470 if (have_eri()) {
4471 tensorT result1=cresult.full_tensor_copy();
4472 result1+=pm(key,copy(ceri(result->cdata.s0)));
4473 cresult=coeffT(result1,result->get_tensor_args());
4474 error+=pm.error;
4475 } else {
4477 }
4478 if ((not have_v1()) and (not have_v2()) and (not have_eri())) {
4479 cresult=coeff_ket;
4480 }
4481
4482 return std::make_pair(cresult,error);
4483 }
4484
4485 this_type make_child(const keyT& child) const {
4486
4487 // break key into particles
4488 Key<LDIM> key1, key2;
4489 child.break_apart(key1,key2);
4490
4491 return this_type(result,leaf_op,iaket.make_child(child),
4492 iap1.make_child(key1),iap2.make_child(key2),
4493 iav1.make_child(key1),iav2.make_child(key2),eri);
4494 }
4495
4497 Future<ctT> iaket1=iaket.activate();
4498 Future<ctL> iap11=iap1.activate();
4499 Future<ctL> iap21=iap2.activate();
4500 Future<ctL> iav11=iav1.activate();
4501 Future<ctL> iav21=iav2.activate();
4502 return result->world.taskq.add(detail::wrap_mem_fn(*const_cast<this_type *> (this),
4503 &this_type::forward_ctor),result,leaf_op,
4504 iaket1,iap11,iap21,iav11,iav21,eri);
4505 }
4506
4507 this_type forward_ctor(implT* result1, const opT& leaf_op, const ctT& iaket1,
4508 const ctL& iap11, const ctL& iap21, const ctL& iav11, const ctL& iav21,
4509 const implT* eri1) {
4510 return this_type(result1,leaf_op,iaket1,iap11,iap21,iav11,iav21,eri1);
4511 }
4512
4513 /// serialize this (needed for use in recursive_op)
4514 template <typename Archive> void serialize(const Archive& ar) {
4515 ar & iaket & eri & result & leaf_op & iap1 & iap2 & iav1 & iav2;
4516 }
4517 };
4518
4519 /// assemble the function V*phi using V and phi given from the functor
4520
4521 /// this function must have been constructed using the CompositeFunctorInterface.
4522 /// The interface provides one- and two-electron potentials, and the ket, which are
4523 /// assembled to give V*phi.
4524 /// @param[in] leaf_op operator to decide if a given node is a leaf node
4525 /// @param[in] fence global fence
4526 template<typename opT>
4527 void make_Vphi(const opT& leaf_op, const bool fence=true) {
4528
4529 constexpr size_t LDIM=NDIM/2;
4530 MADNESS_CHECK_THROW(NDIM==LDIM*2,"make_Vphi only works for even dimensions");
4531
4532
4533 // keep the functor available, but remove it from the result
4534 // result will return false upon is_on_demand(), which is necessary for the
4535 // CoeffTracker to track the parent coeffs correctly for error_leaf_op
4536 std::shared_ptr< FunctionFunctorInterface<T,NDIM> > func2(this->get_functor());
4537 this->unset_functor();
4538
4540 dynamic_cast<CompositeFunctorInterface<T,NDIM,LDIM>* >(&(*func2));
4542
4543 // make sure everything is in place if no fence is requested
4544 if (fence) func->make_redundant(true); // no-op if already redundant
4545 MADNESS_CHECK_THROW(func->check_redundant(),"make_Vphi requires redundant functions");
4546
4547 // loop over all functions in the functor (either ket or particles)
4548 for (auto& ket : func->impl_ket_vector) {
4549 FunctionImpl<T,NDIM>* eri=func->impl_eri.get();
4550 FunctionImpl<T,LDIM>* v1=func->impl_m1.get();
4551 FunctionImpl<T,LDIM>* v2=func->impl_m2.get();
4552 FunctionImpl<T,LDIM>* p1=nullptr;
4553 FunctionImpl<T,LDIM>* p2=nullptr;
4554 make_Vphi_only(leaf_op,ket.get(),v1,v2,p1,p2,eri,false);
4555 }
4556
4557 for (std::size_t i=0; i<func->impl_p1_vector.size(); ++i) {
4558 FunctionImpl<T,NDIM>* ket=nullptr;
4559 FunctionImpl<T,NDIM>* eri=func->impl_eri.get();
4560 FunctionImpl<T,LDIM>* v1=func->impl_m1.get();
4561 FunctionImpl<T,LDIM>* v2=func->impl_m2.get();
4562 FunctionImpl<T,LDIM>* p1=func->impl_p1_vector[i].get();
4563 FunctionImpl<T,LDIM>* p2=func->impl_p2_vector[i].get();
4564 make_Vphi_only(leaf_op,ket,v1,v2,p1,p2,eri,false);
4565 }
4566
4567 // some post-processing:
4568 // - FunctionNode::accumulate() uses buffer -> add the buffer contents to the actual coefficients
4569 // - the operation constructs sum coefficients on all scales -> sum down to get a well-defined tree-state
4570 if (fence) {
4571 world.gop.fence();
4573 sum_down(true);
4575 }
4576
4577
4578 }
4579
4580 /// assemble the function V*phi using V and phi given from the functor
4581
4582 /// this function must have been constructed using the CompositeFunctorInterface.
4583 /// The interface provides one- and two-electron potentials, and the ket, which are
4584 /// assembled to give V*phi.
4585 /// @param[in] leaf_op operator to decide if a given node is a leaf node
4586 /// @param[in] fence global fence
4587 template<typename opT, std::size_t LDIM>
4592 const bool fence=true) {
4593
4594 // prepare the CoeffTracker
4595 CoeffTracker<T,NDIM> iaket(ket);
4596 CoeffTracker<T,LDIM> iap1(p1);
4597 CoeffTracker<T,LDIM> iap2(p2);
4598 CoeffTracker<T,LDIM> iav1(v1);
4599 CoeffTracker<T,LDIM> iav2(v2);
4600
4601 // the operator making the coefficients
4602 typedef Vphi_op_NS<opT,LDIM> coeff_opT;
4603 coeff_opT coeff_op(this,leaf_op,iaket,iap1,iap2,iav1,iav2,eri);
4604
4605 // this operator simply inserts the coeffs into this' tree
4606 typedef noop<T,NDIM> apply_opT;
4607 apply_opT apply_op;
4608
4609 if (world.rank() == coeffs.owner(cdata.key0)) {
4610 woT::task(world.rank(), &implT:: template forward_traverse<coeff_opT,apply_opT>,
4611 coeff_op, apply_op, cdata.key0);
4612 }
4613
4615 if (fence) world.gop.fence();
4616
4617 }
4618
4619 /// Permute the dimensions of f according to map, result on this
4620 void mapdim(const implT& f, const std::vector<long>& map, bool fence);
4621
4622 /// mirror the dimensions of f according to map, result on this
4623 void mirror(const implT& f, const std::vector<long>& mirror, bool fence);
4624
4625 /// map and mirror the translation index and the coefficients, result on this
4626
4627 /// first map the dimensions, the mirror!
4628 /// this = mirror(map(f))
4629 void map_and_mirror(const implT& f, const std::vector<long>& map,
4630 const std::vector<long>& mirror, bool fence);
4631
4632 /// take the average of two functions, similar to: this=0.5*(this+rhs)
4633
4634 /// works in either basis and also in nonstandard form
4635 void average(const implT& rhs);
4636
4637 /// change the tensor type of the coefficients in the FunctionNode
4638
4639 /// @param[in] targs target tensor arguments (threshold and full/low rank)
4640 void change_tensor_type1(const TensorArgs& targs, bool fence);
4641
4642 /// reduce the rank of the coefficients tensors
4643
4644 /// @param[in] targs target tensor arguments (threshold and full/low rank)
4645 void reduce_rank(const double thresh, bool fence);
4646
4647
4648 /// remove all nodes with level higher than n
4649 void chop_at_level(const int n, const bool fence=true);
4650
4651 /// compute norm of s and d coefficients for all nodes
4652 void compute_snorm_and_dnorm(bool fence=true);
4653
4654 /// compute the norm of the wavelet coefficients
4657
4661
4662 bool operator()(typename rangeT::iterator& it) const {
4663 auto& node=it->second;
4664 node.recompute_snorm_and_dnorm(cdata);
4665 return true;
4666 }
4667 };
4668
4669
4670 T eval_cube(Level n, coordT& x, const tensorT& c) const;
4671
4672 /// Transform sum coefficients at level n to sums+differences at level n-1
4673
4674 /// Given scaling function coefficients s[n][l][i] and s[n][l+1][i]
4675 /// return the scaling function and wavelet coefficients at the
4676 /// coarser level. I.e., decompose Vn using Vn = Vn-1 + Wn-1.
4677 /// \code
4678 /// s_i = sum(j) h0_ij*s0_j + h1_ij*s1_j
4679 /// d_i = sum(j) g0_ij*s0_j + g1_ij*s1_j
4680 // \endcode
4681 /// Returns a new tensor and has no side effects. Works for any
4682 /// number of dimensions.
4683 ///
4684 /// No communication involved.
4685 tensorT filter(const tensorT& s) const;
4686
4687 coeffT filter(const coeffT& s) const;
4688
4689 /// Transform sums+differences at level n to sum coefficients at level n+1
4690
4691 /// Given scaling function and wavelet coefficients (s and d)
4692 /// returns the scaling function coefficients at the next finer
4693 /// level. I.e., reconstruct Vn using Vn = Vn-1 + Wn-1.
4694 /// \code
4695 /// s0 = sum(j) h0_ji*s_j + g0_ji*d_j
4696 /// s1 = sum(j) h1_ji*s_j + g1_ji*d_j
4697 /// \endcode
4698 /// Returns a new tensor and has no side effects
4699 ///
4700 /// If (sonly) ... then ss is only the scaling function coeff (and
4701 /// assume the d are zero). Works for any number of dimensions.
4702 ///
4703 /// No communication involved.
4704 tensorT unfilter(const tensorT& s) const;
4705
4706 coeffT unfilter(const coeffT& s) const;
4707
4708 /// downsample the sum coefficients of level n+1 to sum coeffs on level n
4709
4710 /// specialization of the filter method, will yield only the sum coefficients
4711 /// @param[in] key key of level n
4712 /// @param[in] v vector of sum coefficients of level n+1
4713 /// @return sum coefficients on level n in full tensor format
4714 tensorT downsample(const keyT& key, const std::vector< Future<coeffT > >& v) const;
4715
4716 /// upsample the sum coefficients of level 1 to sum coeffs on level n+1
4717
4718 /// specialization of the unfilter method, will transform only the sum coefficients
4719 /// @param[in] key key of level n+1
4720 /// @param[in] coeff sum coefficients of level n (does NOT belong to key!!)
4721 /// @return sum coefficients on level n+1
4722 coeffT upsample(const keyT& key, const coeffT& coeff) const;
4723
4724 /// Projects old function into new basis (only in reconstructed form)
4725 void project(const implT& old, bool fence);
4726
4728 bool operator()(const implT* f, const keyT& key, const nodeT& t) const {
4729 return true;
4730 }
4731 template <typename Archive> void serialize(Archive& ar) {}
4732 };
4733
4734 template <typename opT>
4735 void refine_op(const opT& op, const keyT& key) {
4736 // Must allow for someone already having autorefined the coeffs
4737 // and we get a write accessor just in case they are already executing
4738 typename dcT::accessor acc;
4739 const auto found = coeffs.find(acc,key);
4740 MADNESS_CHECK(found);
4741 nodeT& node = acc->second;
4742 if (node.has_coeff() && key.level() < max_refine_level && op(this, key, node)) {
4743 coeffT d(cdata.v2k,targs);
4744 d(cdata.s0) += copy(node.coeff());
4745 d = unfilter(d);
4746 node.clear_coeff();
4747 node.set_has_children(true);
4748 for (KeyChildIterator<NDIM> kit(key); kit; ++kit) {
4749 const keyT& child = kit.key();
4750 coeffT ss = copy(d(child_patch(child)));
4752 // coeffs.replace(child,nodeT(ss,-1.0,false).node_to_low_rank());
4753 coeffs.replace(child,nodeT(ss,-1.0,false));
4754 // Note value -1.0 for norm tree to indicate result of refinement
4755 }
4756 }
4757 }
4758
4759 template <typename opT>
4760 void refine_spawn(const opT& op, const keyT& key) {
4761 nodeT& node = coeffs.find(key).get()->second;
4762 if (node.has_children()) {
4763 for (KeyChildIterator<NDIM> kit(key); kit; ++kit)
4764 woT::task(coeffs.owner(kit.key()), &implT:: template refine_spawn<opT>, op, kit.key(), TaskAttributes::hipri());
4765 }
4766 else {
4767 woT::task(coeffs.owner(key), &implT:: template refine_op<opT>, op, key);
4768 }
4769 }
4770
4771 // Refine in real space according to local user-defined criterion
4772 template <typename opT>
4773 void refine(const opT& op, bool fence) {
4774 if (world.rank() == coeffs.owner(cdata.key0))
4775 woT::task(coeffs.owner(cdata.key0), &implT:: template refine_spawn<opT>, op, cdata.key0, TaskAttributes::hipri());
4776 if (fence)
4777 world.gop.fence();
4778 }
4779
4780 bool exists_and_has_children(const keyT& key) const;
4781
4782 bool exists_and_is_leaf(const keyT& key) const;
4783
4784
4785 void broaden_op(const keyT& key, const std::vector< Future <bool> >& v);
4786
4787 // For each local node sets norm_tree, snorm and dnorm to 0.0, and marks
4788 // dnorm_tree as uncomputed.
4789 void zero_norm_tree();
4790
4791 // Broaden tree
4792 void broaden(const array_of_bools<NDIM>& is_periodic, bool fence);
4793
4794 /// sum all the contributions from all scales after applying an operator in mod-NS form
4795 void trickle_down(bool fence);
4796
4797 /// sum all the contributions from all scales after applying an operator in mod-NS form
4798
4799 /// cf reconstruct_op
4800 void trickle_down_op(const keyT& key, const coeffT& s);
4801
4802 /// reconstruct this tree -- respects fence
4803 void reconstruct(bool fence);
4804
4805 void change_tree_state(const TreeState finalstate, bool fence=true);
4806
4807 // Invoked on node where key is local
4808 // void reconstruct_op(const keyT& key, const tensorT& s);
4809 void reconstruct_op(const keyT& key, const coeffT& s, const bool accumulate_NS=true);
4810
4811 /// compress the wave function
4812
4813 /// after application there will be sum coefficients at the root level,
4814 /// and difference coefficients at all other levels; furthermore:
4815 /// @param[in] nonstandard keep sum coeffs at all other levels, except leaves
4816 /// @param[in] keepleaves keep sum coeffs (but no diff coeffs) at leaves
4817 /// @param[in] redundant keep only sum coeffs at all levels, discard difference coeffs
4818// void compress(bool nonstandard, bool keepleaves, bool redundant, bool fence);
4819 void compress(const TreeState newstate, bool fence);
4820
4821 /// s coefficients plus the (snorm_tree, dnorm_tree) pair propagated up by compress
4822 typedef std::pair<coeffT, std::pair<double,double> > compressT;
4823
4824 /// Invoked on node where key is local
4825 Future<compressT> compress_spawn(const keyT& key, bool nonstandard, bool keepleaves,
4826 bool redundant1);
4827
4828 private:
4829 /// convert this to redundant, i.e. have sum coefficients on all levels
4830 void make_redundant(const bool fence);
4831 public:
4832
4833 /// convert this from redundant to standard reconstructed form
4834 void undo_redundant(const bool fence);
4835
4836 void remove_internal_coefficients(const bool fence);
4837 void remove_leaf_coefficients(const bool fence);
4838
4839
4840 /// compute for each FunctionNode the norm of the function inside that node
4841 void norm_tree(bool fence);
4842
4843 double norm_tree_op(const keyT& key, const std::vector< Future<double> >& v);
4844
4846
4847 /// truncate using a tree in reconstructed form
4848
4849 /// must be invoked where key is local
4850 Future<coeffT> truncate_reconstructed_spawn(const keyT& key, const double tol);
4851
4852 /// given the sum coefficients of all children, truncate or not
4853
4854 /// @return new sum coefficients (empty if internal, not empty, if new leaf); might delete its children
4855 coeffT truncate_reconstructed_op(const keyT& key, const std::vector< Future<coeffT > >& v, const double tol);
4856
4857 /// calculate the wavelet coefficients using the sum coefficients of all child nodes
4858
4859 /// also propagates norm_tree and dnorm_tree for all nodes
4860 /// @param[in] key this's key
4861 /// @param[in] v sum coefficients and propagated norms of the child nodes
4862 /// @param[in] nonstandard keep the sum coefficients with the wavelet coefficients
4863 /// @return the sum coefficients and propagated norms
4864 compressT compress_op(const keyT& key, const std::vector< Future<compressT> >& v, bool nonstandard);
4865
4866
4867 /// similar to compress_op, but insert only the sum coefficients in the tree
4868
4869 /// also propagates norm_tree and dnorm_tree for all nodes
4870 /// @param[in] key this's key
4871 /// @param[in] v sum coefficients and propagated norms of the child nodes
4872 /// @return the sum coefficients and propagated norms
4873 compressT make_redundant_op(const keyT& key,const std::vector< Future<compressT> >& v);
4874
4875 /// Changes non-standard compressed form to standard compressed form
4876 void standard(bool fence);
4877
4878 /// Changes non-standard compressed form to standard compressed form
4881
4882 // threshold for rank reduction / SVD truncation
4884
4885 // constructor takes target precision
4886 do_standard() = default;
4888
4889 //
4890 bool operator()(typename rangeT::iterator& it) const {
4891
4892 const keyT& key = it->first;
4893 nodeT& node = it->second;
4894 if (key.level()> 0 && node.has_coeff()) {
4895 if (node.has_children()) {
4896 // Zero out scaling coeffs
4897 MADNESS_ASSERT(node.coeff().dim(0)==2*impl->get_k());
4898 node.coeff()(impl->cdata.s0)=0.0;
4899 node.reduceRank(impl->targs.thresh);
4900 } else {
4901 // Deleting both scaling and wavelet coeffs
4902 node.clear_coeff();
4903 }
4904 }
4905 return true;
4906 }
4907 template <typename Archive> void serialize(const Archive& ar) {
4908 MADNESS_EXCEPTION("no serialization of do_standard",1);
4909 }
4910 };
4911
4912
4913 /// laziness
4914 template<size_t OPDIM>
4915 struct do_op_args {
4918 double tol, fac, cnorm;
4919
4920 do_op_args() = default;
4921 do_op_args(const Key<OPDIM>& key, const Key<OPDIM>& d, const keyT& dest, double tol, double fac, double cnorm)
4922 : key(key), d(d), dest(dest), tol(tol), fac(fac), cnorm(cnorm) {}
4923 template <class Archive>
4924 void serialize(Archive& ar) {
4925 ar & archive::wrap_opaque(this,1);
4926 }
4927 };
4928
4929 /// for fine-grain parallelism: call the apply method of an operator in a separate task
4930
4931 /// @param[in] op the operator working on our function
4932 /// @param[in] c full rank tensor holding the NS coefficients
4933 /// @param[in] args laziness holding norm of the coefficients, displacement, destination, ..
4934 template <typename opT, typename R, size_t OPDIM>
4935 void do_apply_kernel(const opT* op, const Tensor<R>& c, const do_op_args<OPDIM>& args) {
4936
4937 tensorT result = op->apply(args.key, args.d, c, args.tol/args.fac/args.cnorm);
4938
4939 // Screen here to reduce communication cost of negligible data
4940 // and also to ensure we don't needlessly widen the tree when
4941 // applying the operator
4942 if (result.normf()> 0.3*args.tol/args.fac) {
4944 //woT::task(world.rank(),&implT::accumulate_timer,time,TaskAttributes::hipri());
4945 // UGLY BUT ADDED THE OPTIMIZATION BACK IN HERE EXPLICITLY/
4946 if (args.dest == world.rank()) {
4947 coeffs.send(args.dest, &nodeT::accumulate, result, coeffs, args.dest);
4948 }
4949 else {
4951 }
4952 }
4953 }
4954
4955 /// same as do_apply_kernel, but use full rank tensors as input and low rank tensors as output
4956
4957 /// @param[in] op the operator working on our function
4958 /// @param[in] c full rank tensor holding the NS coefficients
4959 /// @param[in] args laziness holding norm of the coefficients, displacement, destination, ..
4960 /// @param[in] apply_targs TensorArgs with tightened threshold for accumulation
4961 /// @return nothing, but accumulate the result tensor into the destination node
4962 template <typename opT, typename R, size_t OPDIM>
4963 double do_apply_kernel2(const opT* op, const Tensor<R>& c, const do_op_args<OPDIM>& args,
4964 const TensorArgs& apply_targs) {
4965
4966 tensorT result_full = op->apply(args.key, args.d, c, args.tol/args.fac/args.cnorm);
4967 const double norm=result_full.normf();
4968
4969 // Screen here to reduce communication cost of negligible data
4970 // and also to ensure we don't needlessly widen the tree when
4971 // applying the operator
4972 // OPTIMIZATION NEEDED HERE ... CHANGING THIS TO TASK NOT SEND REMOVED
4973 // BUILTIN OPTIMIZATION TO SHORTCIRCUIT MSG IF DATA IS LOCAL
4974 if (norm > 0.3*args.tol/args.fac) {
4975
4976 small++;
4977 //double cpu0=cpu_time();
4978 coeffT result=coeffT(result_full,apply_targs);
4979 MADNESS_ASSERT(result.is_full_tensor() or result.is_svd_tensor());
4980 //double cpu1=cpu_time();
4981 //timer_lr_result.accumulate(cpu1-cpu0);
4982
4983 coeffs.task(args.dest, &nodeT::accumulate, result, coeffs, args.dest, apply_targs,
4985
4986 //woT::task(world.rank(),&implT::accumulate_timer,time,TaskAttributes::hipri());
4987 }
4988 return norm;
4989 }
4990
4991
4992
4993 /// same as do_apply_kernel2, but use low rank tensors as input and low rank tensors as output
4994
4995 /// @param[in] op the operator working on our function
4996 /// @param[in] coeff full rank tensor holding the NS coefficients
4997 /// @param[in] args laziness holding norm of the coefficients, displacement, destination, ..
4998 /// @param[in] apply_targs TensorArgs with tightened threshold for accumulation
4999 /// @return nothing, but accumulate the result tensor into the destination node
5000 template <typename opT, typename R, size_t OPDIM>
5001 double do_apply_kernel3(const opT* op, const GenTensor<R>& coeff, const do_op_args<OPDIM>& args,
5002 const TensorArgs& apply_targs) {
5003
5004 coeffT result;
5005 if (2*OPDIM==NDIM) result= op->apply2_lowdim(args.key, args.d, coeff,
5006 args.tol/args.fac/args.cnorm, args.tol/args.fac);
5007 if (OPDIM==NDIM) result = op->apply2(args.key, args.d, coeff,
5008 args.tol/args.fac/args.cnorm, args.tol/args.fac);
5009
5010 const double result_norm=result.svd_normf();
5011
5012 if (result_norm> 0.3*args.tol/args.fac) {
5013 small++;
5014
5015 double cpu0=cpu_time();
5016 if (not result.is_of_tensortype(targs.tt)) result=result.convert(targs);
5017 double cpu1=cpu_time();
5018 timer_lr_result.accumulate(cpu1-cpu0);
5019
5020 // accumulate also expects result in SVD form
5021 coeffs.task(args.dest, &nodeT::accumulate, result, coeffs, args.dest, apply_targs,
5023// woT::task(world.rank(),&implT::accumulate_timer,time,TaskAttributes::hipri());
5024
5025 }
5026 return result_norm;
5027
5028 }
5029
5030 // volume of n-dimensional sphere of radius R
5031 double vol_nsphere(int n, double R) {
5032 return std::pow(madness::constants::pi,n*0.5)*std::pow(R,n)/std::tgamma(1+0.5*n);
5033 }
5034
5035
5036 /// apply an operator on the coeffs c (at node key)
5037
5038 /// the result is accumulated inplace to this's tree at various FunctionNodes
5039 /// @param[in] op the operator to act on the source function
5040 /// @param[in] key key of the source FunctionNode of f which is processed
5041 /// @param[in] c coeffs of the FunctionNode of f which is processed
5042 template <typename opT, typename R>
5043 void do_apply(const opT* op, const keyT& key, const Tensor<R>& c) {
5045
5046 // working assumption here WAS that the operator is
5047 // isotropic and monotonically decreasing with distance
5048 // ... however, now we are using derivative Gaussian
5049 // expansions (and also non-cubic boxes) isotropic is
5050 // violated. While not strictly monotonically decreasing,
5051 // the derivative gaussian is still such that once it
5052 // becomes negligible we are in the asymptotic region.
5053
5054 typedef typename opT::keyT opkeyT;
5055 constexpr auto opdim = opT::opdim;
5056 const opkeyT source = op->get_source_key(key);
5057
5058 // Tuning here is based on observation that with
5059 // sufficiently high-order wavelet relative to the
5060 // precision, that only nearest neighbor boxes contribute,
5061 // whereas for low-order wavelets more neighbors will
5062 // contribute. Sufficiently high is picked as
5063 // k>=2-log10(eps) which is our empirical rule for
5064 // efficiency/accuracy and code instrumentation has
5065 // previously indicated that (in 3D) just unit
5066 // displacements are invoked. The error decays as R^-(k+1),
5067 // and the number of boxes increases as R^d.
5068 //
5069 // Fac is the expected number of contributions to a given
5070 // box, so the error permitted per contribution will be
5071 // tol/fac
5072
5073 // radius of shell (nearest neighbor is diameter of 3 boxes, so radius=1.5)
5074 double radius = 1.5 + 0.33 * std::max(0.0, 2 - std::log10(thresh) -
5075 k); // 0.33 was 0.5
5076 //double radius = 2.5;
5077 double fac = vol_nsphere(NDIM, radius);
5078 // previously fac=10.0 selected empirically constrained by qmprop
5079
5080 double cnorm = c.normf();
5081
5082 // BC handling:
5083 // - if operator is lattice-summed then treat this as nonperiodic (i.e. tell neighbor() to stay in simulation cell)
5084 // - if operator is NOT lattice-summed then obey BC (i.e. tell neighbor() to go outside the simulation cell along periodic dimensions)
5085 // - BUT user can force operator to treat its arguments as non-periodic (`op.set_domain_periodicity({true,true,true})`) so ... which dimensions of this function are treated as periodic by op?
5086 const array_of_bools<NDIM> func_is_treated_by_op_as_periodic =
5087 (op->particle() == 1)
5088 ? array_of_bools<NDIM>{false}.or_front(
5089 op->func_domain_is_periodic())
5090 : array_of_bools<NDIM>{false}.or_back(
5091 op->func_domain_is_periodic());
5092
5093 const auto default_real_distance_squared = [&](const auto &displacement)
5094 -> double {
5095 return displacement.real_distsq_bc(op->lattice_summed(), FunctionDefaults<NDIM>::get_cell_width());
5096 };
5097 const auto default_lattice_distance_squared = [&](const auto &displacement)
5098 -> std::uint64_t {
5099 return displacement.distsq_bc(op->lattice_summed());
5100 };
5101 const auto default_skip_predicate = [&](const auto &displacement)
5102 -> bool {
5103 return false;
5104 };
5105 const auto for_each = [&](auto &displacements,
5106 const auto &real_distance_squared,
5107 const auto &lattice_distance_squared,
5108 const auto &skip_predicate) -> std::optional<double> {
5109
5110 // used to screen estimated and actual contributions
5111 //const double tol = truncate_tol(thresh, key);
5112 //const double tol = 0.1*truncate_tol(thresh, key);
5113 const double tol = truncate_tol(thresh, key);
5114
5115 // assume isotropic decaying kernel, screen in shell-wise fashion by
5116 // monitoring the decay of magnitude of contribution norms with the
5117 // distance ... as soon as we find a shell of displacements at least
5118 // one of each in simulation domain (see neighbor()) and
5119 // all in-domain shells produce negligible contributions, stop.
5120 // a displacement is negligible if ||op|| * ||c|| > tol / fac
5121 // where fac takes into account
5122 int nvalid = 1; // Counts #valid at each distance
5123 int nused = 1; // Counts #used at each distance
5124 std::optional<double> real_last_distsq;
5125 std::optional<std::uint64_t> lattice_last_distsq;
5126
5127 // displacements to a face of the kernel range boundary are typically same magnitude (modulo variation),
5128 // but faces can be at quite different distances (anisotropic cells, lattice summation along some axes only,
5129 // mixed-parity ranges). Estimate the norm of the contributions from each face using its probing
5130 // displacement, skip the faces whose contributions are negligible, and skip everything if all are.
5131 if constexpr (std::is_same_v<std::decay_t<decltype(displacements)>,BoxSurfaceDisplacementRange<opdim>>) {
5132 bool any_face_survives = false;
5133 const auto &probing_displacements = displacements.probing_displacements();
5134 for (std::size_t d = 0; d != opdim; ++d) {
5135 if (!probing_displacements[d]) continue; // no face normal to an unlimited dimension
5136 const double opnorm =
5137 op->norm(key.level(), *probing_displacements[d], source);
5138 if (cnorm * opnorm <= tol / fac)
5139 displacements.skip_face(d);
5140 else
5141 any_face_survives = true;
5142 }
5143 if (!any_face_survives) {
5144 return {};
5145 }
5146 }
5147
5148 for (const auto& displacement: displacements) {
5149 if (skip_predicate(displacement)) continue;
5150
5151 keyT d;
5152 Key<NDIM - opdim> nullkey(key.level());
5153 MADNESS_ASSERT(op->particle() == 1 || op->particle() == 2);
5154 if (op->particle() == 1)
5155 d = displacement.merge_with(nullkey);
5156 else
5157 d = nullkey.merge_with(displacement);
5158
5159 // Screen out shells. We assume shells are grouped into shells so that the operator decays with shell index.
5160 // Shells are indexed by least distance from box to the central box.
5161 // Cells touching so much as a corner of the central box are further grouped by their lattice distance.
5162 // N.B. lattice-summed decaying kernel is periodic (i.e. does decay w.r.t. r), so loop over shells of displacements sorted by distances modulated by periodicity (Key::distsq_bc)
5163 const auto real_distsq = real_distance_squared(displacement);
5164 const std::uint64_t lattice_distsq = real_distsq ? 0 : lattice_distance_squared(displacement);
5165 if (!real_last_distsq.has_value() ||
5166 !same_displacement_shell(real_distsq, *real_last_distsq) || (*real_last_distsq == 0 && lattice_distsq != *lattice_last_distsq)) { // Moved to next shell of neighbors
5167 if (nvalid > 0 && nused == 0 && (real_distsq > 0 || lattice_distsq > 1)) {
5168 // Have at least done the input box and all first
5169 // nearest neighbors, and none of the last set
5170 // of neighbors made significant contributions. Thus,
5171 // assuming monotonic decrease, we are done.
5172 break;
5173 }
5174 nused = 0;
5175 nvalid = 0;
5176 real_last_distsq = real_distsq;
5177 // After real_last_distsq > 0, we stop caring about keeping lattice_last_distsq up-to-date.
5178 lattice_last_distsq = real_distsq ? std::optional<std::uint64_t>{} : lattice_distsq;
5179 }
5180
5181 keyT dest = neighbor(key, d, func_is_treated_by_op_as_periodic);
5182 if (dest.is_valid()) {
5183 nvalid++;
5184 const double opnorm = op->norm(key.level(), displacement, source);
5185
5186 if (cnorm * opnorm > tol / fac) {
5187 tensorT result =
5188 op->apply(source, displacement, c, tol / fac / cnorm);
5189 if (result.normf() > 0.3 * tol / fac) {
5190 if (coeffs.is_local(dest))
5191 coeffs.send(dest, &nodeT::accumulate2, result, coeffs,
5192 dest);
5193 else
5194 coeffs.task(dest, &nodeT::accumulate2, result, coeffs,
5195 dest);
5196 nused++;
5197 }
5198 }
5199 }
5200 }
5201
5202 return real_last_distsq;
5203 };
5204
5205 // process "standard" displacements, screening assumes monotonic decay of the kernel
5206 // list of displacements sorted in order of increasing distance
5207 // N.B. if op is lattice-summed use periodic displacements, else use
5208 // non-periodic even if op treats any modes of this as periodic
5209 const std::vector<opkeyT> &disp = op->get_disp(key.level());
5210 const auto max_distsq_reached = for_each(disp, default_real_distance_squared, default_lattice_distance_squared, default_skip_predicate);
5211
5212 // for range-restricted kernels displacements to the boundary of the kernel range also need to be included
5213 // N.B. hard range restriction will result in slow decay of operator matrix elements for the displacements
5214 // to the range boundary, should use soft restriction or sacrifice precision
5215 if (op->range_restricted() && key.level() >= 1) {
5216
5217 std::array<std::optional<std::int64_t>, opdim> box_radius;
5218 std::array<std::optional<std::int64_t>, opdim> surface_thickness;
5219 auto &range = op->get_range();
5220 for (int d = 0; d != opdim; ++d) {
5221 if (range[d]) {
5222 box_radius[d] = range[d].N();
5223 surface_thickness[d] = range[d].finite_soft() ? 1 : 0;
5224 }
5225 }
5226
5227 // skip surface displacements that take us outside of the domain and/or were included in regular displacements
5228 // N.B. for lattice-summed axes the "filter" also maps the displacement back into the simulation cell.
5229 // The filter also carries the real-space reach of the standard displacements it discards, outside of which
5230 // the range places its probing displacements. If no standard displacements were processed there is nothing
5231 // to filter and nothing is known about the reach; the probes then screen nothing.
5232 using SurfaceRange = BoxSurfaceDisplacementRange<opdim>;
5233 std::optional<typename SurfaceRange::Validator> validator;
5234 if (max_distsq_reached) {
5235 // N.B. must use the same widths as default_real_distance_squared, i.e. the first opdim axes
5236 const auto &cell_width = FunctionDefaults<NDIM>::get_cell_width();
5237 std::array<double, opdim> widths;
5238 for (std::size_t d = 0; d != opdim; ++d) widths[d] = cell_width(d);
5239 validator.emplace(/* is_infinite_domain= */ op->func_domain_is_periodic(), /* is_lattice_summed= */ op->lattice_summed(),
5240 StandardDisplacementsReach<opdim>{*max_distsq_reached, widths});
5241 }
5242
5243 // this range iterates over the entire surface layer(s), and provides a probing displacement per face that can be used to screen out the face
5244 auto opkey = op->particle() == 1 ? key.template extract_front<opdim>() : key.template extract_back<opdim>();
5245 SurfaceRange range_boundary_face_displacements(opkey, box_radius,
5246 surface_thickness,
5247 op->lattice_summed(),
5248 validator);
5249 for_each(
5250 range_boundary_face_displacements,
5251 // surface displacements are not screened, all are included
5252 [](const auto &displacement) -> double { return 0; },
5253 [](const auto &displacement) -> std::uint64_t { return 0; },
5254 default_skip_predicate);
5255 }
5256 }
5257
5258
5259 /// apply an operator on f to return this
5260 template <typename opT, typename R>
5261 void apply(opT& op, const FunctionImpl<R,NDIM>& f, bool fence) {
5263 MADNESS_ASSERT(!op.modified());
5264 for (const auto& [key, node]: f.coeffs) {
5265 if (node.has_coeff()) {
5266 if (node.coeff().dim(0) != k /* i.e. not a leaf */ || op.doleaves) {
5268// woT::task(p, &implT:: template do_apply<opT,R>, &op, key, node.coeff()); //.full_tensor_copy() ????? why copy ????
5269 woT::task(p, &implT:: template do_apply<opT,R>, &op, key, node.coeff().reconstruct_tensor());
5270 }
5271 }
5272 }
5273 if (fence)
5274 world.gop.fence();
5275
5277// this->compressed=true;
5278// this->nonstandard=true;
5279// this->redundant=false;
5280
5281 }
5282
5283
5284
5285 /// apply an operator on the coeffs c (at node key)
5286
5287 /// invoked by result; the result is accumulated inplace to this's tree at various FunctionNodes
5288 /// @param[in] op the operator to act on the source function
5289 /// @param[in] key key of the source FunctionNode of f which is processed (see "source")
5290 /// @param[in] coeff coeffs of FunctionNode being processed
5291 /// @param[in] do_kernel true: do the 0-disp only; false: do everything but the kernel
5292 /// @return max norm, and will modify or include new nodes in this' tree
5293 template <typename opT, typename R>
5294 double do_apply_directed_screening(const opT* op, const keyT& key, const coeffT& coeff,
5295 const bool& do_kernel) {
5297 // insert timer here
5298 typedef typename opT::keyT opkeyT;
5299
5300 // screening: contains all displacement keys that had small result norms
5301 std::list<opkeyT> blacklist;
5302
5303 constexpr auto opdim=opT::opdim;
5304 Key<NDIM-opdim> nullkey(key.level());
5305
5306 // source is that part of key that corresponds to those dimensions being processed
5307 const opkeyT source=op->get_source_key(key);
5308
5309 const double tol = truncate_tol(thresh, key);
5310
5311 // fac is the root of the number of contributing neighbors (1st shell)
5312 double fac=std::pow(3,NDIM*0.5);
5313 double cnorm = coeff.normf();
5314
5315 // for accumulation: keep slightly tighter TensorArgs
5316 TensorArgs apply_targs(targs);
5317 apply_targs.thresh=tol/fac*0.03;
5318
5319 double maxnorm=0.0;
5320
5321 // for the kernel it may be more efficient to do the convolution in full rank
5322 tensorT coeff_full;
5323 // for partial application (exchange operator) it's more efficient to
5324 // do SVD tensors instead of tensortrains, because addition in apply
5325 // can be done in full form for the specific particle
5326 coeffT coeff_SVD=coeff.convert(TensorArgs(-1.0,TT_2D));
5327#ifdef HAVE_GENTENSOR
5328 coeff_SVD.get_svdtensor().orthonormalize(tol*GenTensor<T>::fac_reduce());
5329#endif
5330
5331 // list of displacements sorted in order of increasing distance
5332 // N.B. if op is lattice-summed gives periodic displacements, else uses
5333 // non-periodic even if op treats any modes of this as periodic
5334 const std::vector<opkeyT>& disp = Displacements<opdim>().get_disp(key.level(), op->lattice_summed());
5335
5336 for (const auto& d: disp) {
5337 const int shell=d.distsq_bc(op->lattice_summed());
5338 if (do_kernel and (shell>0)) break;
5339 if ((not do_kernel) and (shell==0)) continue;
5340
5341 keyT disp1;
5342 if (op->particle()==1) disp1=d.merge_with(nullkey);
5343 else if (op->particle()==2) disp1=nullkey.merge_with(d);
5344 else {
5345 MADNESS_EXCEPTION("confused particle in operator??",1);
5346 }
5347
5348 keyT dest = neighbor_in_volume(key, disp1);
5349
5350 if (not dest.is_valid()) continue;
5351
5352 // directed screening
5353 // working assumption here is that the operator is isotropic and
5354 // monotonically decreasing with distance
5355 bool screened=false;
5356 typename std::list<opkeyT>::const_iterator it2;
5357 for (it2=blacklist.begin(); it2!=blacklist.end(); it2++) {
5358 if (d.is_farther_out_than(*it2)) {
5359 screened=true;
5360 break;
5361 }
5362 }
5363 if (not screened) {
5364
5365 double opnorm = op->norm(key.level(), d, source);
5366 double norm=0.0;
5367
5368 if (cnorm*opnorm> tol/fac) {
5369
5370 double cost_ratio=op->estimate_costs(source, d, coeff_SVD, tol/fac/cnorm, tol/fac);
5371 // cost_ratio=1.5; // force low rank
5372 // cost_ratio=0.5; // force full rank
5373
5374 if (cost_ratio>0.0) {
5375
5376 do_op_args<opdim> args(source, d, dest, tol, fac, cnorm);
5377 norm=0.0;
5378 if (cost_ratio<1.0) {
5379 if (not coeff_full.has_data()) coeff_full=coeff.full_tensor_copy();
5380 norm=do_apply_kernel2(op, coeff_full,args,apply_targs);
5381 } else {
5382 if (2*opdim==NDIM) { // apply operator on one particle only
5383 norm=do_apply_kernel3(op,coeff_SVD,args,apply_targs);
5384 } else {
5385 norm=do_apply_kernel3(op,coeff,args,apply_targs);
5386 }
5387 }
5388 maxnorm=std::max(norm,maxnorm);
5389 }
5390
5391 } else if (shell >= 12) {
5392 break; // Assumes monotonic decay beyond nearest neighbor
5393 }
5394 if (norm<0.3*tol/fac) blacklist.push_back(d);
5395 }
5396 }
5397 return maxnorm;
5398 }
5399
5400
5401 /// similar to apply, but for low rank coeffs
5402 template <typename opT, typename R>
5403 void apply_source_driven(opT& op, const FunctionImpl<R,NDIM>& f, bool fence) {
5405
5406 MADNESS_ASSERT(not op.modified());
5407 // looping through all the coefficients of the source f
5408 typename dcT::const_iterator end = f.get_coeffs().end();
5409 for (typename dcT::const_iterator it=f.get_coeffs().begin(); it!=end; ++it) {
5410
5411 const keyT& key = it->first;
5412 const coeffT& coeff = it->second.coeff();
5413
5414 if (coeff.has_data() and (coeff.rank()!=0)) {
5416 woT::task(p, &implT:: template do_apply_directed_screening<opT,R>, &op, key, coeff, true);
5417 woT::task(p, &implT:: template do_apply_directed_screening<opT,R>, &op, key, coeff, false);
5418 }
5419 }
5420 if (fence) world.gop.fence();
5422 }
5423
5424 /// after apply we need to do some cleanup;
5425
5426 /// forces fence
5427 double finalize_apply();
5428
5429 /// after summing up we need to do some cleanup;
5430
5431 /// forces fence
5432 void finalize_sum();
5433
5434 /// traverse a non-existing tree, make its coeffs and apply an operator
5435
5436 /// invoked by result
5437 /// here we use the fact that the hi-dim NS coefficients on all scales are exactly
5438 /// the outer product of the underlying low-dim functions (also in NS form),
5439 /// so we don't need to construct the full hi-dim tree and then turn it into NS form.
5440 /// @param[in] apply_op the operator acting on the NS tree
5441 /// @param[in] fimpl the funcimpl of the function of particle 1
5442 /// @param[in] gimpl the funcimpl of the function of particle 2
5443 template<typename opT, std::size_t LDIM>
5444 void recursive_apply(opT& apply_op, const FunctionImpl<T,LDIM>* fimpl,
5445 const FunctionImpl<T,LDIM>* gimpl, const bool fence) {
5446
5447 //print("IN RECUR2");
5448 const keyT& key0=cdata.key0;
5449
5450 if (world.rank() == coeffs.owner(key0)) {
5451
5452 CoeffTracker<T,LDIM> ff(fimpl);
5453 CoeffTracker<T,LDIM> gg(gimpl);
5454
5455 typedef recursive_apply_op<opT,LDIM> coeff_opT;
5456 coeff_opT coeff_op(this,ff,gg,&apply_op);
5457
5458 typedef noop<T,NDIM> apply_opT;
5459 apply_opT apply_op;
5460
5462 woT::task(p, &implT:: template forward_traverse<coeff_opT,apply_opT>, coeff_op, apply_op, key0);
5463
5464 }
5465 if (fence) world.gop.fence();
5467 }
5468
5469 /// recursive part of recursive_apply
5470 template<typename opT, std::size_t LDIM>
5472 bool randomize() const {return true;}
5473
5475
5480
5481 // ctor
5485 const opT* apply_op) : result(result), iaf(iaf), iag(iag), apply_op(apply_op)
5486 {
5487 MADNESS_ASSERT(LDIM+LDIM==NDIM);
5488 }
5490 iag(other.iag), apply_op(other.apply_op) {}
5491
5492
5493 /// make the NS-coefficients and send off the application of the operator
5494
5495 /// @return a Future<bool,coeffT>(is_leaf,coeffT())
5496 std::pair<bool,coeffT> operator()(const Key<NDIM>& key) const {
5497
5498 // World& world=result->world;
5499 // break key into particles (these are the child keys, with datum1/2 come the parent keys)
5500 Key<LDIM> key1,key2;
5501 key.break_apart(key1,key2);
5502
5503 // the lo-dim functions should be in full tensor form
5504 const tensorT fcoeff=iaf.coeff(key1).full_tensor();
5505 const tensorT gcoeff=iag.coeff(key2).full_tensor();
5506
5507 // would this be a leaf node? If so, then its sum coeffs have already been
5508 // processed by the parent node's wavelet coeffs. Therefore we won't
5509 // process it any more.
5511 bool is_leaf=leaf_op(key,fcoeff,gcoeff);
5512
5513 if (not is_leaf) {
5514 // new coeffs are simply the hartree/kronecker/outer product --
5515 const std::vector<Slice>& s0=iaf.get_impl()->cdata.s0;
5516 const coeffT coeff = (apply_op->modified())
5517 ? outer(copy(fcoeff(s0)),copy(gcoeff(s0)),result->targs)
5518 : outer(fcoeff,gcoeff,result->targs);
5519
5520 // now send off the application
5521 tensorT coeff_full;
5523 double norm0=result->do_apply_directed_screening<opT,T>(apply_op, key, coeff, true);
5524
5525 result->task(p,&implT:: template do_apply_directed_screening<opT,T>,
5526 apply_op,key,coeff,false);
5527
5528 return finalize(norm0,key,coeff);
5529
5530 } else {
5531 return std::pair<bool,coeffT> (is_leaf,coeffT());
5532 }
5533 }
5534
5535 /// sole purpose is to wait for the kernel norm, wrap it and send it back to caller
5536 std::pair<bool,coeffT> finalize(const double kernel_norm, const keyT& key,
5537 const coeffT& coeff) const {
5538 const double thresh=result->get_thresh()*0.1;
5539 bool is_leaf=(kernel_norm<result->truncate_tol(thresh,key));
5540 if (key.level()<2) is_leaf=false;
5541 return std::pair<bool,coeffT> (is_leaf,coeff);
5542 }
5543
5544
5545 this_type make_child(const keyT& child) const {
5546
5547 // break key into particles
5548 Key<LDIM> key1, key2;
5549 child.break_apart(key1,key2);
5550
5551 return this_type(result,iaf.make_child(key1),iag.make_child(key2),apply_op);
5552 }
5553
5557 return result->world.taskq.add(detail::wrap_mem_fn(*const_cast<this_type *> (this),
5558 &this_type::forward_ctor),result,f1,g1,apply_op);
5559 }
5560
5562 const opT* apply_op1) {
5563 return this_type(r,f1,g1,apply_op1);
5564 }
5565
5566 template <typename Archive> void serialize(const Archive& ar) {
5567 ar & result & iaf & iag & apply_op;
5568 }
5569 };
5570
5571 /// traverse an existing tree and apply an operator
5572
5573 /// invoked by result
5574 /// @param[in] apply_op the operator acting on the NS tree
5575 /// @param[in] fimpl the funcimpl of the source function
5576 /// @param[in] rimpl a dummy function for recursive_op to insert data
5577 template<typename opT>
5578 void recursive_apply(opT& apply_op, const implT* fimpl, implT* rimpl, const bool fence) {
5579
5580 print("IN RECUR1");
5581
5582 const keyT& key0=cdata.key0;
5583
5584 if (world.rank() == coeffs.owner(key0)) {
5585
5586 typedef recursive_apply_op2<opT> coeff_opT;
5587 coeff_opT coeff_op(this,fimpl,&apply_op);
5588
5589 typedef noop<T,NDIM> apply_opT;
5590 apply_opT apply_op;
5591
5592 woT::task(world.rank(), &implT:: template forward_traverse<coeff_opT,apply_opT>,
5593 coeff_op, apply_op, cdata.key0);
5594
5595 }
5596 if (fence) world.gop.fence();
5598 }
5599
5600 /// recursive part of recursive_apply
5601 template<typename opT>
5603 bool randomize() const {return true;}
5604
5607 typedef std::pair<bool,coeffT> argT;
5608
5609 mutable implT* result;
5610 ctT iaf; /// need this for randomization
5611 const opT* apply_op;
5612
5613 // ctor
5617
5619 iaf(other.iaf), apply_op(other.apply_op) {}
5620
5621
5622 /// send off the application of the operator
5623
5624 /// the first (core) neighbor (ie. the box itself) is processed
5625 /// immediately, all other ones are shoved into the taskq
5626 /// @return a pair<bool,coeffT>(is_leaf,coeffT())
5627 argT operator()(const Key<NDIM>& key) const {
5628
5629 const coeffT& coeff=iaf.coeff();
5630
5631 if (coeff.has_data()) {
5632
5633 // now send off the application for all neighbor boxes
5635 result->task(p,&implT:: template do_apply_directed_screening<opT,T>,
5636 apply_op, key, coeff, false);
5637
5638 // process the core box
5639 double norm0=result->do_apply_directed_screening<opT,T>(apply_op,key,coeff,true);
5640
5641 if (iaf.is_leaf()) return argT(true,coeff);
5642 return finalize(norm0,key,coeff,result);
5643
5644 } else {
5645 const bool is_leaf=true;
5646 return argT(is_leaf,coeffT());
5647 }
5648 }
5649
5650 /// sole purpose is to wait for the kernel norm, wrap it and send it back to caller
5651 argT finalize(const double kernel_norm, const keyT& key,
5652 const coeffT& coeff, const implT* r) const {
5653 const double thresh=r->get_thresh()*0.1;
5654 bool is_leaf=(kernel_norm<r->truncate_tol(thresh,key));
5655 if (key.level()<2) is_leaf=false;
5656 return argT(is_leaf,coeff);
5657 }
5658
5659
5660 this_type make_child(const keyT& child) const {
5661 return this_type(result,iaf.make_child(child),apply_op);
5662 }
5663
5664 /// retrieve the coefficients (parent coeffs might be remote)
5666 Future<ctT> f1=iaf.activate();
5667
5668// Future<ctL> g1=g.activate();
5669// return h->world.taskq.add(detail::wrap_mem_fn(*const_cast<this_type *> (this),
5670// &this_type::forward_ctor),h,f1,g1,particle);
5671
5672 return result->world.taskq.add(detail::wrap_mem_fn(*const_cast<this_type *> (this),
5673 &this_type::forward_ctor),result,f1,apply_op);
5674 }
5675
5676 /// taskq-compatible ctor
5677 this_type forward_ctor(implT* result1, const ctT& iaf1, const opT* apply_op1) {
5678 return this_type(result1,iaf1,apply_op1);
5679 }
5680
5681 template <typename Archive> void serialize(const Archive& ar) {
5682 ar & result & iaf & apply_op;
5683 }
5684 };
5685
5686 /// Returns the square of the error norm in the box labeled by key
5687
5688 /// Assumed to be invoked locally but it would be easy to eliminate
5689 /// this assumption
5690 template <typename opT>
5691 double err_box(const keyT& key, const nodeT& node, const opT& func,
5692 int npt, const Tensor<double>& qx, const Tensor<double>& quad_phit,
5693 const Tensor<double>& quad_phiw) const {
5694
5695 std::vector<long> vq(NDIM);
5696 for (std::size_t i=0; i<NDIM; ++i)
5697 vq[i] = npt;
5698 tensorT fval(vq,false), work(vq,false), result(vq,false);
5699
5700 // Compute the "exact" function in this volume at npt points
5701 // where npt is usually this->npt+1.
5702 fcube(key, func, qx, fval);
5703
5704 // Transform into the scaling function basis of order npt
5705 double scale = pow(0.5,0.5*NDIM*key.level())*sqrt(FunctionDefaults<NDIM>::get_cell_volume());
5706 fval = fast_transform(fval,quad_phiw,result,work).scale(scale);
5707
5708 // Subtract to get the error ... the original coeffs are in the order k
5709 // basis but we just computed the coeffs in the order npt(=k+1) basis
5710 // so we can either use slices or an iterator macro.
5711 const tensorT coeff = node.coeff().full_tensor();
5712 ITERATOR(coeff,fval(IND)-=coeff(IND););
5713 // flo note: we do want to keep a full tensor here!
5714
5715 // Compute the norm of what remains
5716 double err = fval.normf();
5717 return err*err;
5718 }
5719
5720 template <typename opT>
5722 const implT* impl;
5723 const opT* func;
5724 int npt;
5728 public:
5729 do_err_box() = default;
5730
5734
5737
5738 double operator()(typename dcT::const_iterator& it) const {
5739 const keyT& key = it->first;
5740 const nodeT& node = it->second;
5741 if (node.has_coeff())
5742 return impl->err_box(key, node, *func, npt, qx, quad_phit, quad_phiw);
5743 else
5744 return 0.0;
5745 }
5746
5747 double operator()(double a, double b) const {
5748 return a+b;
5749 }
5750
5751 template <typename Archive>
5752 void serialize(const Archive& ar) {
5753 MADNESS_EXCEPTION("not yet", 1);
5754 }
5755 };
5756
5757 /// Returns the sum of squares of errors from local info ... no comms
5758 template <typename opT>
5759 double errsq_local(const opT& func) const {
5761 // Make quadrature rule of higher order
5762 const int npt = cdata.npt + 1;
5763 Tensor<double> qx, qw, quad_phi, quad_phiw, quad_phit;
5764 FunctionCommonData<T,NDIM>::_init_quadrature(k+1, npt, qx, qw, quad_phi, quad_phiw, quad_phit);
5765
5768 return world.taskq.reduce< double,rangeT,do_err_box<opT> >(range,
5769 do_err_box<opT>(this, &func, npt, qx, quad_phit, quad_phiw));
5770 }
5771
5772 /// Returns \c int(f(x),x) in local volume
5773 T trace_local() const;
5774
5776 /// skip the internal nodes, cf. has_coefficients_on_leaves_only()
5777 bool leaves_only = false;
5778
5779 do_norm2sq_local() = default;
5781
5782 double operator()(typename dcT::const_iterator& it) const {
5783 const nodeT& node = it->second;
5784 if (leaves_only and node.has_children()) return 0.0;
5785 if (node.has_coeff()) {
5786 double norm = node.coeff().normf();
5787 return norm*norm;
5788 }
5789 else {
5790 return 0.0;
5791 }
5792 }
5793
5794 double operator()(double a, double b) const {
5795 return (a+b);
5796 }
5797
5798 template <typename Archive> void serialize(const Archive& ar) {
5799 MADNESS_EXCEPTION("NOT IMPLEMENTED", 1);
5800 }
5801 };
5802
5803
5804 /// Returns the square of the local norm ... no comms
5805
5806 /// Requires has_summable_coefficients(); throws otherwise, because on a
5807 /// tree that carries its coefficients on more than one level the sum is
5808 /// not the norm. Internal nodes are skipped where they are the
5809 /// duplicates, so no state change is needed to ask for the norm.
5810 double norm2sq_local() const;
5811
5812 /// compute the inner product of this range with other
5813 template<typename R>
5817 typedef TENSOR_RESULT_TYPE(T,R) resultT;
5818
5821 resultT operator()(typename dcT::const_iterator& it) const {
5822
5823 TENSOR_RESULT_TYPE(T,R) sum=0.0;
5824 const keyT& key=it->first;
5825 const nodeT& fnode = it->second;
5826 if (fnode.has_coeff()) {
5827 if (other->coeffs.probe(it->first)) {
5828 const FunctionNode<R,NDIM>& gnode = other->coeffs.find(key).get()->second;
5829 if (gnode.has_coeff()) {
5830 if (gnode.coeff().dim(0) != fnode.coeff().dim(0)) {
5831 madness::print("INNER", it->first, gnode.coeff().dim(0),fnode.coeff().dim(0));
5832 MADNESS_EXCEPTION("functions have different k or compress/reconstruct error", 0);
5833 }
5834 if (leaves_only) {
5835 if (gnode.is_leaf() or fnode.is_leaf()) {
5836 sum += fnode.coeff().trace_conj(gnode.coeff());
5837 }
5838 } else {
5839 sum += fnode.coeff().trace_conj(gnode.coeff());
5840 }
5841 }
5842 }
5843 }
5844 return sum;
5845 }
5846
5847 resultT operator()(resultT a, resultT b) const {
5848 return (a+b);
5849 }
5850
5851 template <typename Archive> void serialize(const Archive& ar) {
5852 MADNESS_EXCEPTION("NOT IMPLEMENTED", 1);
5853 }
5854 };
5855
5856 /// Returns the inner product ASSUMING same distribution
5857
5858 /// handles compressed and redundant form
5859 template <typename R>
5863 typedef TENSOR_RESULT_TYPE(T,R) resultT;
5864
5865 // make sure the states of the trees are consistent
5868 return world.taskq.reduce<resultT,rangeT,do_inner_local<R> >
5870 }
5871
5872
5873 /// compute the inner product of this range with other
5874 template<typename R>
5878 bool leaves_only=true;
5879 typedef TENSOR_RESULT_TYPE(T,R) resultT;
5880
5884 resultT operator()(typename dcT::const_iterator& it) const {
5885
5886 constexpr std::size_t LDIM=std::max(NDIM/2,std::size_t(1));
5887
5888 const keyT& key=it->first;
5889 const nodeT& fnode = it->second;
5890 if (not fnode.has_coeff()) return resultT(0.0); // probably internal nodes
5891
5892 // assuming all boxes (esp the low-dim ones) are local, i.e. the functions are replicated
5893 auto find_valid_parent = [](auto& key, auto& impl, auto&& find_valid_parent) {
5894 MADNESS_CHECK(impl->get_coeffs().owner(key)==impl->world.rank()); // make sure everything is local!
5895 if (impl->get_coeffs().probe(key)) return key;
5896 auto parentkey=key.parent();
5897 return find_valid_parent(parentkey, impl, find_valid_parent);
5898 };
5899
5900 // returns coefficients, empty if no functor present
5901 auto get_coeff = [&find_valid_parent](const auto& key, const auto& v_impl) {
5902 if ((v_impl.size()>0) and v_impl.front().get()) {
5903 auto impl=v_impl.front();
5904
5905// bool have_impl=impl.get();
5906// if (have_impl) {
5907 auto parentkey = find_valid_parent(key, impl, find_valid_parent);
5908 MADNESS_CHECK(impl->get_coeffs().probe(parentkey));
5909 typename decltype(impl->coeffs)::accessor acc;
5910 impl->get_coeffs().find(acc,parentkey);
5911 auto parentcoeff=acc->second.coeff();
5912 auto coeff=impl->parent_to_child(parentcoeff, parentkey, key);
5913 return coeff;
5914 } else {
5915 // get type of vector elements
5916 typedef typename std::decay_t<decltype(v_impl)>::value_type::element_type::typeT S;
5917// typedef typename std::decay_t<decltype(v_impl)>::value_type S;
5918 return GenTensor<S>();
5919// return GenTensor<typename std::decay_t<decltype(*impl)>::typeT>();
5920 }
5921 };
5922
5923 auto make_vector = [](auto& arg) {
5924 return std::vector<std::decay_t<decltype(arg)>>(1,arg);
5925 };
5926
5927
5928 Key<LDIM> key1,key2;
5929 key.break_apart(key1,key2);
5930
5931 auto func=dynamic_cast<CompositeFunctorInterface<R,NDIM,LDIM>* >(ket->functor.get());
5933
5934 MADNESS_CHECK_THROW(func->impl_ket_vector.size()==0 or func->impl_ket_vector.size()==1,
5935 "only one ket function supported in inner_on_demand");
5936 MADNESS_CHECK_THROW(func->impl_p1_vector.size()==0 or func->impl_p1_vector.size()==1,
5937 "only one p1 function supported in inner_on_demand");
5938 MADNESS_CHECK_THROW(func->impl_p2_vector.size()==0 or func->impl_p2_vector.size()==1,
5939 "only one p2 function supported in inner_on_demand");
5940 auto coeff_bra=fnode.coeff();
5941 auto coeff_ket=get_coeff(key,func->impl_ket_vector);
5942 auto coeff_v1=get_coeff(key1,make_vector(func->impl_m1));
5943 auto coeff_v2=get_coeff(key2,make_vector(func->impl_m2));
5944 auto coeff_p1=get_coeff(key1,func->impl_p1_vector);
5945 auto coeff_p2=get_coeff(key2,func->impl_p2_vector);
5946
5947 // construct |ket(1,2)> or |p(1)p(2)> or |p(1)p(2) ket(1,2)>
5948 double error=0.0;
5949 if (coeff_ket.has_data() and coeff_p1.has_data()) {
5950 pointwise_multiplier<LDIM> pm(key,coeff_ket);
5951 coeff_ket=pm(key,outer(coeff_p1,coeff_p2,TensorArgs(TT_FULL,-1.0)).full_tensor());
5952 error+=pm.error;
5953 } else if (coeff_ket.has_data() or coeff_p1.has_data()) {
5954 coeff_ket = (coeff_ket.has_data()) ? coeff_ket : outer(coeff_p1,coeff_p2);
5955 } else { // not ket and no p1p2
5956 MADNESS_EXCEPTION("confused ket/p1p2 in do_inner_local_on_demand",1);
5957 }
5958
5959 // construct (v(1) + v(2)) |ket(1,2)>
5960 coeffT v1v2ket;
5961 if (coeff_v1.has_data()) {
5962 pointwise_multiplier<LDIM> pm(key,coeff_ket);
5963 v1v2ket = pm(key,coeff_v1.full_tensor(), 1);
5964 error+=pm.error;
5965 v1v2ket+= pm(key,coeff_v2.full_tensor(), 2);
5966 error+=pm.error;
5967 } else {
5968 v1v2ket = coeff_ket;
5969 }
5970
5971 resultT result;
5972 if (func->impl_eri) { // project bra*ket onto eri, avoid multiplication with eri
5973 MADNESS_CHECK(func->impl_eri->get_functor()->provides_coeff());
5974 coeffT coeff_eri=func->impl_eri->get_functor()->coeff(key).full_tensor();
5975 pointwise_multiplier<LDIM> pm(key,v1v2ket);
5976 tensorT braket=pm(key,coeff_bra.full_tensor_copy().conj());
5977 error+=pm.error;
5978 if (error>1.e-3) print("error in key",key,error);
5979 result=coeff_eri.full_tensor().trace(braket);
5980
5981 } else { // no eri, project ket onto bra
5982 result=coeff_bra.full_tensor_copy().trace_conj(v1v2ket.full_tensor_copy());
5983 }
5984 return result;
5985 }
5986
5987 resultT operator()(resultT a, resultT b) const {
5988 return (a+b);
5989 }
5990
5991 template <typename Archive> void serialize(const Archive& ar) {
5992 MADNESS_EXCEPTION("NOT IMPLEMENTED", 1);
5993 }
5994 };
5995
5996 /// Returns the inner product of this with function g constructed on-the-fly
5997
5998 /// the leaf boxes of this' MRA tree defines the inner product
5999 template <typename R>
6000 TENSOR_RESULT_TYPE(T,R) inner_local_on_demand(const FunctionImpl<R,NDIM>& gimpl) const {
6003
6007 do_inner_local_on_demand<R>(this, &gimpl));
6008 }
6009
6010 /// compute the inner product of this range with other
6011 template<typename R>
6015 typedef TENSOR_RESULT_TYPE(T,R) resultT;
6016
6019 resultT operator()(typename dcT::const_iterator& it) const {
6020
6021 TENSOR_RESULT_TYPE(T,R) sum=0.0;
6022 const keyT& key=it->first;
6023 const nodeT& fnode = it->second;
6024 if (fnode.has_coeff()) {
6025 if (other->coeffs.probe(it->first)) {
6026 const FunctionNode<R,NDIM>& gnode = other->coeffs.find(key).get()->second;
6027 if (gnode.has_coeff()) {
6028 if (gnode.coeff().dim(0) != fnode.coeff().dim(0)) {
6029 madness::print("DOT", it->first, gnode.coeff().dim(0),fnode.coeff().dim(0));
6030 MADNESS_EXCEPTION("functions have different k or compress/reconstruct error", 0);
6031 }
6032 if (leaves_only) {
6033 if (gnode.is_leaf() or fnode.is_leaf()) {
6034 sum += fnode.coeff().full_tensor().trace(gnode.coeff().full_tensor());
6035 }
6036 } else {
6037 sum += fnode.coeff().full_tensor().trace(gnode.coeff().full_tensor());
6038 }
6039 }
6040 }
6041 }
6042 return sum;
6043 }
6044
6045 resultT operator()(resultT a, resultT b) const {
6046 return (a+b);
6047 }
6048
6049 template <typename Archive> void serialize(const Archive& ar) {
6050 MADNESS_EXCEPTION("NOT IMPLEMENTED", 1);
6051 }
6052 };
6053
6054 /// Returns the dot product ASSUMING same distribution
6055
6056 /// handles compressed and redundant form
6057 template <typename R>
6061 typedef TENSOR_RESULT_TYPE(T,R) resultT;
6062
6063 // make sure the states of the trees are consistent
6065 bool leaves_only=(this->is_redundant());
6066 return world.taskq.reduce<resultT,rangeT,do_dot_local<R> >
6068 }
6069
6070 /// Type of the entry in the map returned by make_key_vec_map
6071 typedef std::vector< std::pair<int,const coeffT*> > mapvecT;
6072
6073 /// Type of the map returned by make_key_vec_map
6075
6076 /// Adds keys to union of local keys with specified index
6077 void add_keys_to_map(mapT* map, int index) const {
6078 typename dcT::const_iterator end = coeffs.end();
6079 for (typename dcT::const_iterator it=coeffs.begin(); it!=end; ++it) {
6080 typename mapT::accessor acc;
6081 const keyT& key = it->first;
6082 const FunctionNode<T,NDIM>& node = it->second;
6083 if (node.has_coeff()) {
6084 [[maybe_unused]] auto inserted = map->insert(acc,key);
6085 acc->second.push_back(std::make_pair(index,&(node.coeff())));
6086 }
6087 }
6088 }
6089
6090 /// Returns map of union of local keys to vector of indexes of functions containing that key
6091
6092 /// Local concurrency and synchronization only; no communication
6093 static
6094 mapT
6095 make_key_vec_map(const std::vector<const FunctionImpl<T,NDIM>*>& v) {
6096 mapT map(100000);
6097 // This loop must be parallelized
6098 for (unsigned int i=0; i<v.size(); i++) {
6099 //v[i]->add_keys_to_map(&map,i);
6100 v[i]->world.taskq.add(*(v[i]), &FunctionImpl<T,NDIM>::add_keys_to_map, &map, int(i));
6101 }
6102 if (v.size()) v[0]->world.taskq.fence();
6103 return map;
6104 }
6105
6106#if 0
6107// Original
6108 template <typename R>
6109 static void do_inner_localX(const typename mapT::iterator lstart,
6110 const typename mapT::iterator lend,
6111 typename FunctionImpl<R,NDIM>::mapT* rmap_ptr,
6112 const bool sym,
6113 Tensor< TENSOR_RESULT_TYPE(T,R) >* result_ptr,
6114 Mutex* mutex) {
6115 Tensor< TENSOR_RESULT_TYPE(T,R) >& result = *result_ptr;
6116 Tensor< TENSOR_RESULT_TYPE(T,R) > r(result.dim(0),result.dim(1));
6117 for (typename mapT::iterator lit=lstart; lit!=lend; ++lit) {
6118 const keyT& key = lit->first;
6119 typename FunctionImpl<R,NDIM>::mapT::iterator rit=rmap_ptr->find(key);
6120 if (rit != rmap_ptr->end()) {
6121 const mapvecT& leftv = lit->second;
6122 const typename FunctionImpl<R,NDIM>::mapvecT& rightv =rit->second;
6123 const int nleft = leftv.size();
6124 const int nright= rightv.size();
6125
6126 for (int iv=0; iv<nleft; iv++) {
6127 const int i = leftv[iv].first;
6128 const GenTensor<T>* iptr = leftv[iv].second;
6129
6130 for (int jv=0; jv<nright; jv++) {
6131 const int j = rightv[jv].first;
6132 const GenTensor<R>* jptr = rightv[jv].second;
6133
6134 if (!sym || (sym && i<=j))
6135 r(i,j) += iptr->trace_conj(*jptr);
6136 }
6137 }
6138 }
6139 }
6140 mutex->lock();
6141 result += r;
6142 mutex->unlock();
6143 }
6144#else
6145 template <typename R>
6146 static void do_inner_localX(const typename mapT::iterator lstart,
6147 const typename mapT::iterator lend,
6148 typename FunctionImpl<R,NDIM>::mapT* rmap_ptr,
6149 const bool sym,
6150 Tensor< TENSOR_RESULT_TYPE(T,R) >* result_ptr,
6151 Mutex* mutex) {
6152 Tensor< TENSOR_RESULT_TYPE(T,R) >& result = *result_ptr;
6153 //Tensor< TENSOR_RESULT_TYPE(T,R) > r(result.dim(0),result.dim(1));
6154 for (typename mapT::iterator lit=lstart; lit!=lend; ++lit) {
6155 const keyT& key = lit->first;
6156 typename FunctionImpl<R,NDIM>::mapT::iterator rit=rmap_ptr->find(key);
6157 if (rit != rmap_ptr->end()) {
6158 const mapvecT& leftv = lit->second;
6159 const typename FunctionImpl<R,NDIM>::mapvecT& rightv =rit->second;
6160 const size_t nleft = leftv.size();
6161 const size_t nright= rightv.size();
6162
6163 unsigned int size = leftv[0].second->size();
6164 Tensor<T> Left(nleft, size);
6165 Tensor<R> Right(nright, size);
6166 Tensor< TENSOR_RESULT_TYPE(T,R)> r(nleft, nright);
6167 for(unsigned int iv = 0; iv < nleft; ++iv) Left(iv,_) = (*(leftv[iv].second)).full_tensor();
6168 for(unsigned int jv = 0; jv < nright; ++jv) Right(jv,_) = (*(rightv[jv].second)).full_tensor();
6169 // call mxmT from mxm.h in tensor
6170 if(TensorTypeData<T>::iscomplex) Left = Left.conj(); // Should handle complex case and leave real case alone
6171 mxmT(nleft, nright, size, r.ptr(), Left.ptr(), Right.ptr());
6172 mutex->lock();
6173 for(unsigned int iv = 0; iv < nleft; ++iv) {
6174 const int i = leftv[iv].first;
6175 for(unsigned int jv = 0; jv < nright; ++jv) {
6176 const int j = rightv[jv].first;
6177 if (!sym || (sym && i<=j)) result(i,j) += r(iv,jv);
6178 }
6179 }
6180 mutex->unlock();
6181 }
6182 }
6183 }
6184#endif
6185
6186#if 0
6187// Original
6188 template <typename R, typename = std::enable_if_t<std::is_floating_point_v<R>>>
6189 static void do_dot_localX(const typename mapT::iterator lstart,
6190 const typename mapT::iterator lend,
6191 typename FunctionImpl<R, NDIM>::mapT* rmap_ptr,
6192 const bool sym,
6193 Tensor<TENSOR_RESULT_TYPE(T, R)>* result_ptr,
6194 Mutex* mutex) {
6195 if (TensorTypeData<T>::iscomplex) MADNESS_EXCEPTION("no complex trace in LowRankTensor, sorry", 1);
6196 Tensor<TENSOR_RESULT_TYPE(T, R)>& result = *result_ptr;
6197 Tensor<TENSOR_RESULT_TYPE(T, R)> r(result.dim(0), result.dim(1));
6198 for (typename mapT::iterator lit = lstart; lit != lend; ++lit) {
6199 const keyT& key = lit->first;
6200 typename FunctionImpl<R, NDIM>::mapT::iterator rit = rmap_ptr->find(key);
6201 if (rit != rmap_ptr->end()) {
6202 const mapvecT& leftv = lit->second;
6203 const typename FunctionImpl<R, NDIM>::mapvecT& rightv = rit->second;
6204 const int nleft = leftv.size();
6205 const int nright = rightv.size();
6206
6207 for (int iv = 0; iv < nleft; iv++) {
6208 const int i = leftv[iv].first;
6209 const GenTensor<T>* iptr = leftv[iv].second;
6210
6211 for (int jv = 0; jv < nright; jv++) {
6212 const int j = rightv[jv].first;
6213 const GenTensor<R>* jptr = rightv[jv].second;
6214
6215 if (!sym || (sym && i <= j))
6216 r(i, j) += iptr->trace_conj(*jptr);
6217 }
6218 }
6219 }
6220 }
6221 mutex->lock();
6222 result += r;
6223 mutex->unlock();
6224 }
6225#else
6226 template <typename R>
6227 static void do_dot_localX(const typename mapT::iterator lstart,
6228 const typename mapT::iterator lend,
6229 typename FunctionImpl<R, NDIM>::mapT* rmap_ptr,
6230 const bool sym,
6231 Tensor<TENSOR_RESULT_TYPE(T, R)>* result_ptr,
6232 Mutex* mutex) {
6233 Tensor<TENSOR_RESULT_TYPE(T, R)>& result = *result_ptr;
6234 // Tensor<TENSOR_RESULT_TYPE(T, R)> r(result.dim(0), result.dim(1));
6235 for (typename mapT::iterator lit = lstart; lit != lend; ++lit) {
6236 const keyT& key = lit->first;
6237 typename FunctionImpl<R, NDIM>::mapT::iterator rit = rmap_ptr->find(key);
6238 if (rit != rmap_ptr->end()) {
6239 const mapvecT& leftv = lit->second;
6240 const typename FunctionImpl<R, NDIM>::mapvecT& rightv = rit->second;
6241 const size_t nleft = leftv.size();
6242 const size_t nright= rightv.size();
6243
6244 unsigned int size = leftv[0].second->size();
6245 Tensor<T> Left(nleft, size);
6246 Tensor<R> Right(nright, size);
6247 Tensor< TENSOR_RESULT_TYPE(T, R)> r(nleft, nright);
6248 for(unsigned int iv = 0; iv < nleft; ++iv) Left(iv, _) = (*(leftv[iv].second)).full_tensor();
6249 for(unsigned int jv = 0; jv < nright; ++jv) Right(jv, _) = (*(rightv[jv].second)).full_tensor();
6250 // call mxmT from mxm.h in tensor
6251 mxmT(nleft, nright, size, r.ptr(), Left.ptr(), Right.ptr());
6252 mutex->lock();
6253 for(unsigned int iv = 0; iv < nleft; ++iv) {
6254 const int i = leftv[iv].first;
6255 for(unsigned int jv = 0; jv < nright; ++jv) {
6256 const int j = rightv[jv].first;
6257 if (!sym || (sym && i <= j)) result(i, j) += r(iv, jv);
6258 }
6259 }
6260 mutex->unlock();
6261 }
6262 }
6263 }
6264#endif
6265
6266 template <typename Real>
6267 static std::enable_if_t<std::is_floating_point_v<Real>, Real> conj(const Real x) {
6268 return x;
6269 }
6270
6271 template <typename Real>
6272 static std::complex<Real> conj(const std::complex<Real>& x) {
6273 return std::conj(x);
6274 }
6275
6276 template <typename R>
6277 static Tensor< TENSOR_RESULT_TYPE(T,R) >
6278 inner_local(const std::vector<const FunctionImpl<T,NDIM>*>& left,
6279 const std::vector<const FunctionImpl<R,NDIM>*>& right,
6280 bool sym) {
6281
6282 // This is basically a sparse matrix^T * matrix product
6283 // Rij = sum(k) Aki * Bkj
6284 // where i and j index functions and k index the wavelet coeffs
6285 // eventually the goal is this structure (don't have jtile yet)
6286 //
6287 // do in parallel tiles of k (tensors of coeffs)
6288 // do tiles of j
6289 // do i
6290 // do j in jtile
6291 // do k in ktile
6292 // Rij += Aki*Bkj
6293
6294 mapT lmap = make_key_vec_map(left);
6295 typename FunctionImpl<R,NDIM>::mapT rmap;
6296 auto* rmap_ptr = (typename FunctionImpl<R,NDIM>::mapT*)(&lmap);
6297 if ((std::vector<const FunctionImpl<R,NDIM>*>*)(&left) != &right) {
6299 rmap_ptr = &rmap;
6300 }
6301
6302 size_t chunk = (lmap.size()-1)/(3*4*5)+1;
6303
6304 Tensor< TENSOR_RESULT_TYPE(T,R) > r(left.size(), right.size());
6305 Mutex mutex;
6306
6307 typename mapT::iterator lstart=lmap.begin();
6308 while (lstart != lmap.end()) {
6309 typename mapT::iterator lend = lstart;
6310 advance(lend,chunk);
6311 left[0]->world.taskq.add(&FunctionImpl<T,NDIM>::do_inner_localX<R>, lstart, lend, rmap_ptr, sym, &r, &mutex);
6312 lstart = lend;
6313 }
6314 left[0]->world.taskq.fence();
6315
6316 if (sym) {
6317 for (long i=0; i<r.dim(0); i++) {
6318 for (long j=0; j<i; j++) {
6319 TENSOR_RESULT_TYPE(T,R) sum = r(i,j)+conj(r(j,i));
6320 r(i,j) = sum;
6321 r(j,i) = conj(sum);
6322 }
6323 }
6324 }
6325 return r;
6326 }
6327
6328 template <typename R>
6329 static Tensor<TENSOR_RESULT_TYPE(T, R)>
6330 dot_local(const std::vector<const FunctionImpl<T, NDIM>*>& left,
6331 const std::vector<const FunctionImpl<R, NDIM>*>& right,
6332 bool sym) {
6333
6334 // This is basically a sparse matrix * matrix product
6335 // Rij = sum(k) Aik * Bkj
6336 // where i and j index functions and k index the wavelet coeffs
6337 // eventually the goal is this structure (don't have jtile yet)
6338 //
6339 // do in parallel tiles of k (tensors of coeffs)
6340 // do tiles of j
6341 // do i
6342 // do j in jtile
6343 // do k in ktile
6344 // Rij += Aik*Bkj
6345
6346 mapT lmap = make_key_vec_map(left);
6347 typename FunctionImpl<R, NDIM>::mapT rmap;
6348 auto* rmap_ptr = (typename FunctionImpl<R, NDIM>::mapT*)(&lmap);
6349 if ((std::vector<const FunctionImpl<R, NDIM>*>*)(&left) != &right) {
6351 rmap_ptr = &rmap;
6352 }
6353
6354 size_t chunk = (lmap.size() - 1) / (3 * 4 * 5) + 1;
6355
6356 Tensor<TENSOR_RESULT_TYPE(T, R)> r(left.size(), right.size());
6357 Mutex mutex;
6358
6359 typename mapT::iterator lstart=lmap.begin();
6360 while (lstart != lmap.end()) {
6361 typename mapT::iterator lend = lstart;
6362 advance(lend, chunk);
6363 left[0]->world.taskq.add(&FunctionImpl<T, NDIM>::do_dot_localX<R>, lstart, lend, rmap_ptr, sym, &r, &mutex);
6364 lstart = lend;
6365 }
6366 left[0]->world.taskq.fence();
6367
6368 // sym is for hermiticity
6369 if (sym) {
6370 for (long i = 0; i < r.dim(0); i++) {
6371 for (long j = 0; j < i; j++) {
6372 TENSOR_RESULT_TYPE(T, R) sum = r(i, j) + conj(r(j, i));
6373 r(i, j) = sum;
6374 r(j, i) = conj(sum);
6375 }
6376 }
6377 }
6378 return r;
6379 }
6380
6381 template <typename R>
6383 {
6384 static_assert(!std::is_same<R, int>::value &&
6385 std::is_same<R, int>::value,
6386 "Compilation failed because you wanted to know the type; see below:");
6387 }
6388
6389 /// invoked by result
6390
6391 /// contract 2 functions f(x,z) = \int g(x,y) * h(y,z) dy
6392 /// @tparam CDIM: the dimension of the contraction variable (y)
6393 /// @tparam NDIM: the dimension of the result (x,z)
6394 /// @tparam LDIM: the dimension of g(x,y)
6395 /// @tparam KDIM: the dimension of h(y,z)
6396 template<typename Q, std::size_t LDIM, typename R, std::size_t KDIM,
6397 std::size_t CDIM = (KDIM + LDIM - NDIM) / 2>
6399 const std::array<int, CDIM> v1, const std::array<int, CDIM> v2) {
6400
6401 typedef std::multimap<Key<NDIM>, std::list<Key<CDIM>>> contractionmapT;
6402 //double wall_get_lists=0.0;
6403 //double wall_recur=0.0;
6404 //double wall_contract=0.0;
6407
6408 // auto print_map = [](const auto& map) {
6409 // for (const auto& kv : map) print(kv.first,"--",kv.second);
6410 // };
6411 // logical constness, not bitwise constness
6412 FunctionImpl<Q,LDIM>& g_nc=const_cast<FunctionImpl<Q,LDIM>&>(g);
6413 FunctionImpl<R,KDIM>& h_nc=const_cast<FunctionImpl<R,KDIM>&>(h);
6414
6415 std::list<contractionmapT> all_contraction_maps;
6416 for (std::size_t n=0; n<nmax; ++n) {
6417
6418 // list of nodes with d coefficients (and their parents)
6419 //double wall0 = wall_time();
6420 auto [g_ijlist, g_jlist] = g.get_contraction_node_lists(n, v1);
6421 auto [h_ijlist, h_jlist] = h.get_contraction_node_lists(n, v2);
6422 if ((g_ijlist.size() == 0) and (h_ijlist.size() == 0)) break;
6423 //double wall1 = wall_time();
6424 //wall_get_lists += (wall1 - wall0);
6425 //wall0 = wall1;
6426// print("g_jlist");
6427// for (const auto& kv : g_jlist) print(kv.first,kv.second);
6428// print("h_jlist");
6429// for (const auto& kv : h_jlist) print(kv.first,kv.second);
6430
6431 // next lines will insert s nodes into g and h -> possible race condition!
6432 bool this_first = true; // are the remaining indices of g before those of g: f(x,z) = g(x,y) h(y,z)
6433 // CDIM, NDIM, KDIM
6434 contractionmapT contraction_map = g_nc.recur_down_for_contraction_map(
6435 g_nc.key0(), g_nc.get_coeffs().find(g_nc.key0()).get()->second, v1, v2,
6436 h_ijlist, h_jlist, this_first, thresh);
6437
6438 this_first = false;
6439 // CDIM, NDIM, LDIM
6440 auto hnode0=h_nc.get_coeffs().find(h_nc.key0()).get()->second;
6441 contractionmapT contraction_map1 = h_nc.recur_down_for_contraction_map(
6442 h_nc.key0(), hnode0, v2, v1,
6443 g_ijlist, g_jlist, this_first, thresh);
6444
6445 // will contain duplicate entries
6446 contraction_map.merge(contraction_map1);
6447 // turn multimap into a map of list
6448 auto it = contraction_map.begin();
6449 while (it != contraction_map.end()) {
6450 auto it_end = contraction_map.upper_bound(it->first);
6451 auto it2 = it;
6452 it2++;
6453 while (it2 != it_end) {
6454 it->second.splice(it->second.end(), it2->second);
6455 it2 = contraction_map.erase(it2);
6456 }
6457 it = it_end;
6458 }
6459// print("thresh ",thresh);
6460// print("contraction list size",contraction_map.size());
6461
6462 // remove all double entries
6463 for (auto& elem: contraction_map) {
6464 elem.second.sort();
6465 elem.second.unique();
6466 }
6467 //wall1 = wall_time();
6468 //wall_recur += (wall1 - wall0);
6469// if (n==2) {
6470// print("contraction map for n=", n);
6471// print_map(contraction_map);
6472// }
6473 all_contraction_maps.push_back(contraction_map);
6474
6475 long mapsize=contraction_map.size();
6476 if (mapsize==0) break;
6477 }
6478
6479
6480 // finally do the contraction
6481 for (const auto& contraction_map : all_contraction_maps) {
6482 for (const auto& key_list : contraction_map) {
6483 const Key<NDIM>& key=key_list.first;
6484 const std::list<Key<CDIM>>& list=key_list.second;
6485 woT::task(coeffs.owner(key), &implT:: template partial_inner_contract<Q,LDIM,R,KDIM>,
6486 &g,&h,v1,v2,key,list);
6487 }
6488 }
6489 }
6490
6491 /// for contraction two functions f(x,z) = \int g(x,y) h(y,z) dy
6492
6493 /// find all nodes with d coefficients and return a list of complete keys and of
6494 /// keys holding only the y dimension, also the maximum norm of all d for the j dimension
6495 /// @param[in] n the scale
6496 /// @param[in] v array holding the indices of the integration variable
6497 /// @return ijlist: list of all nodes with d coeffs; jlist: j-part of ij list only
6498 template<std::size_t CDIM>
6499 std::tuple<std::set<Key<NDIM>>, std::map<Key<CDIM>,double>>
6500 get_contraction_node_lists(const std::size_t n, const std::array<int, CDIM>& v) const {
6501
6502 const auto& cdata=get_cdata();
6503 auto has_d_coeffs = [&cdata](const coeffT& coeff) {
6504 if (coeff.has_no_data()) return false;
6505 return (coeff.dim(0)==2*cdata.k);
6506 };
6507
6508 // keys to be contracted in g
6509 std::set<Key<NDIM>> ij_list; // full key
6510 std::map<Key<CDIM>,double> j_list; // only that dimension that will be contracted
6511
6512 for (auto it=get_coeffs().begin(); it!=get_coeffs().end(); ++it) {
6513 const Key<NDIM>& key=it->first;
6514 const FunctionNode<T,NDIM>& node=it->second;
6515 if ((key.level()==int(n)) and (has_d_coeffs(node.coeff()))) {
6516 ij_list.insert(key);
6518 for (std::size_t i=0; i<CDIM; ++i) j_trans[i]=key.translation()[v[i]];
6519 Key<CDIM> jkey(n,j_trans);
6520 const double max_d_norm=j_list[jkey];
6521 j_list.insert_or_assign(jkey,std::max(max_d_norm,node.get_dnorm()));
6522 Key<CDIM> parent_jkey=jkey.parent();
6523 while (j_list.count(parent_jkey)==0) {
6524 j_list.insert({parent_jkey,1.0});
6525 parent_jkey=parent_jkey.parent();
6526 }
6527 }
6528 }
6529 return std::make_tuple(ij_list,j_list);
6530 }
6531
6532 /// make a map of all nodes that will contribute to a partial inner product
6533
6534 /// given the list of d coefficient-holding nodes of the other function:
6535 /// recur down h if snorm * dnorm > tol and key n−jx ∈ other−ij-list. Make s
6536 /// coefficients if necessary. Make list of nodes n − ijk as map(n-ik, list(j)).
6537 ///
6538 /// !! WILL ADD NEW S NODES TO THIS TREE THAT MUST BE REMOVED TO AVOID INCONSISTENT TREE STRUCTURE !!
6539 ///
6540 /// @param[in] key for recursion
6541 /// @param[in] node corresponds to key
6542 /// @param[in] v_this this' dimension that are contracted
6543 /// @param[in] v_other other's dimension that are contracted
6544 /// @param[in] ij_other_list list of nodes of the other function that will be contracted (and their parents)
6545 /// @param[in] j_other_list list of column nodes of the other function that will be contracted (and their parents)
6546 /// @param[in] max_d_norm max d coeff norm of the nodes in j_list
6547 /// @param[in] this_first are the remaining coeffs of this functions first or last in the result function
6548 /// @param[in] thresh threshold for including nodes in the contraction: snorm*dnorm > thresh
6549 /// @tparam CDIM dimension to be contracted
6550 /// @tparam ODIM dimensions of the other function
6551 /// @tparam FDIM dimensions of the final function
6552 template<std::size_t CDIM, std::size_t ODIM, std::size_t FDIM=NDIM+ODIM-2*CDIM>
6553 std::multimap<Key<FDIM>, std::list<Key<CDIM>>> recur_down_for_contraction_map(
6554 const keyT& key, const nodeT& node,
6555 const std::array<int,CDIM>& v_this,
6556 const std::array<int,CDIM>& v_other,
6557 const std::set<Key<ODIM>>& ij_other_list,
6558 const std::map<Key<CDIM>,double>& j_other_list,
6559 bool this_first, const double thresh) {
6560
6561 std::multimap<Key<FDIM>, std::list<Key<CDIM>>> contraction_map;
6562
6563 // fast return if the other function has no d coeffs
6564 if (j_other_list.empty()) return contraction_map;
6565
6566 // continue recursion if this node may be contracted with the j column
6567 // extract relevant node translations from this node
6568 const auto j_this_key=key.extract_key(v_this);
6569
6570// print("\nkey, j_this_key", key, j_this_key);
6571 const double max_d_norm=j_other_list.find(j_this_key)->second;
6572 const bool sd_norm_product_large = node.get_snorm() * max_d_norm > truncate_tol(thresh,key);
6573// print("sd_product_norm",node.get_snorm() * max_d_norm, thresh);
6574
6575 // end recursion if we have reached the final scale n
6576 // with which nodes from other will this node be contracted?
6577 bool final_scale=key.level()==ij_other_list.begin()->level();
6578 if (final_scale and sd_norm_product_large) {
6579 for (auto& other_key : ij_other_list) {
6580 const auto j_other_key=other_key.extract_key(v_other);
6581 if (j_this_key != j_other_key) continue;
6582 auto i_key=key.extract_complement_key(v_this);
6583 auto k_key=other_key.extract_complement_key(v_other);
6584// print("key, ij_other_key",key,other_key);
6585// print("i, k, j key",i_key, k_key, j_this_key);
6586 Key<FDIM> ik_key=(this_first) ? i_key.merge_with(k_key) : k_key.merge_with(i_key);
6587// print("ik_key",ik_key);
6588// MADNESS_CHECK(contraction_map.count(ik_key)==0);
6589 contraction_map.insert(std::make_pair(ik_key,std::list<Key<CDIM>>{j_this_key}));
6590 }
6591 return contraction_map;
6592 }
6593
6594 bool continue_recursion = (j_other_list.count(j_this_key)==1);
6595 if (not continue_recursion) return contraction_map;
6596
6597
6598 // continue recursion if norms are large
6599 continue_recursion = (node.has_children() or sd_norm_product_large);
6600
6601 if (continue_recursion) {
6602 // in case we need to compute children's coefficients: unfilter only once
6603 bool compute_child_s_coeffs=true;
6604 coeffT d = node.coeff();
6605// print("continuing recursion from key",key);
6606
6607 for (KeyChildIterator<NDIM> kit(key); kit; ++kit) {
6608 keyT child=kit.key();
6609 typename dcT::accessor acc;
6610
6611 // make child's s coeffs if it doesn't exist or if is has no s coeffs
6612 bool childnode_exists=get_coeffs().find(acc,child);
6613 // snorm > 0 stands in for "this node holds s coefficients":
6614 // every writer zeroes snorm when it clears them, cf.
6615 // FunctionNode::recompute_snorm_and_dnorm() and the
6616 // keepleaves=false path of compress_spawn().
6617 bool need_s_coeffs= childnode_exists ? (acc->second.get_snorm()<=0.0) : true;
6618
6619 coeffT child_s_coeffs;
6620 if (need_s_coeffs) {
6621 if (compute_child_s_coeffs) {
6622 if (d.dim(0)==cdata.vk[0]) { // s coeffs only in this node
6623 coeffT d1(cdata.v2k,get_tensor_args());
6624 d1(cdata.s0)+=d;
6625 d=d1;
6626 }
6627 d = unfilter(d);
6628 compute_child_s_coeffs=false;
6629 }
6630 child_s_coeffs=copy(d(child_patch(child)));
6631 child_s_coeffs.reduce_rank(thresh);
6632 }
6633
6634 if (not childnode_exists) {
6635 get_coeffs().replace(child,nodeT(child_s_coeffs,false));
6636 get_coeffs().find(acc,child);
6637 } else if (childnode_exists and need_s_coeffs) {
6638 acc->second.coeff()=child_s_coeffs;
6639 }
6640 bool exists= get_coeffs().find(acc,child);
6641 MADNESS_CHECK(exists);
6642 nodeT& childnode = acc->second;
6643 if (need_s_coeffs) childnode.recompute_snorm_and_dnorm(get_cdata());
6644// print("recurring down to",child);
6645 contraction_map.merge(recur_down_for_contraction_map(child,childnode, v_this, v_other,
6646 ij_other_list, j_other_list, this_first, thresh));
6647// print("contraction_map.size()",contraction_map.size());
6648 }
6649
6650 }
6651
6652 return contraction_map;
6653 }
6654
6655
6656 /// tensor contraction part of partial_inner
6657
6658 /// @param[in] g rhs of the inner product
6659 /// @param[in] h lhs of the inner product
6660 /// @param[in] v1 dimensions of g to be contracted
6661 /// @param[in] v2 dimensions of h to be contracted
6662 /// @param[in] key key of result's (this) FunctionNode
6663 /// @param[in] j_key_list list of contraction index-j keys contributing to this' node
6664 template<typename Q, std::size_t LDIM, typename R, std::size_t KDIM,
6665 std::size_t CDIM = (KDIM + LDIM - NDIM) / 2>
6667 const std::array<int, CDIM> v1, const std::array<int, CDIM> v2,
6668 const Key<NDIM>& key, const std::list<Key<CDIM>>& j_key_list) {
6669
6670 Key<LDIM - CDIM> i_key;
6671 Key<KDIM - CDIM> k_key;
6672 key.break_apart(i_key, k_key);
6673
6674 coeffT result_coeff(get_cdata().v2k, get_tensor_type());
6675 for (const auto& j_key: j_key_list) {
6676
6677 auto v_complement = [](const auto& v, const auto& vc) {
6678 constexpr std::size_t VDIM = std::tuple_size<std::decay_t<decltype(v)>>::value;
6679 constexpr std::size_t VCDIM = std::tuple_size<std::decay_t<decltype(vc)>>::value;
6680 std::array<int, VCDIM> result;
6681 for (std::size_t i = 0; i < VCDIM; i++) result[i] = (v.back() + i + 1) % (VDIM + VCDIM);
6682 return result;
6683 };
6684 auto make_ij_key = [&v_complement](const auto i_key, const auto j_key, const auto& v) {
6685 constexpr std::size_t IDIM = std::decay_t<decltype(i_key)>::static_size;
6686 constexpr std::size_t JDIM = std::decay_t<decltype(j_key)>::static_size;
6687 static_assert(JDIM == std::tuple_size<std::decay_t<decltype(v)>>::value);
6688
6690 for (std::size_t i = 0; i < v.size(); ++i) l[v[i]] = j_key.translation()[i];
6691 std::array<int, IDIM> vc1;
6692 auto vc = v_complement(v, vc1);
6693 for (std::size_t i = 0; i < vc.size(); ++i) l[vc[i]] = i_key.translation()[i];
6694
6695 return Key<IDIM + JDIM>(i_key.level(), l);
6696 };
6697
6698 Key<LDIM> ij_key = make_ij_key(i_key, j_key, v1);
6699 Key<KDIM> jk_key = make_ij_key(k_key, j_key, v2);
6700
6701 MADNESS_CHECK(g->get_coeffs().probe(ij_key));
6702 MADNESS_CHECK(h->get_coeffs().probe(jk_key));
6703 const coeffT& gcoeff = g->get_coeffs().find(ij_key).get()->second.coeff();
6704 const coeffT& hcoeff = h->get_coeffs().find(jk_key).get()->second.coeff();
6705 coeffT gcoeff1, hcoeff1;
6706 if (gcoeff.dim(0) == g->get_cdata().k) {
6707 gcoeff1 = coeffT(g->get_cdata().v2k, g->get_tensor_args());
6708 gcoeff1(g->get_cdata().s0) += gcoeff;
6709 } else {
6710 gcoeff1 = gcoeff;
6711 }
6712 if (hcoeff.dim(0) == g->get_cdata().k) {
6713 hcoeff1 = coeffT(h->get_cdata().v2k, h->get_tensor_args());
6714 hcoeff1(h->get_cdata().s0) += hcoeff;
6715 } else {
6716 hcoeff1 = hcoeff;
6717 }
6718
6719 // offset: 0 for full tensor, 1 for svd representation with rand being the first dimension (r,d1,d2,d3) -> (r,d1*d2*d3)
6720 auto fuse = [](Tensor<T> tensor, const std::array<int, CDIM>& v, int offset) {
6721 for (std::size_t i = 0; i < CDIM - 1; ++i) {
6722 MADNESS_CHECK((v[i] + 1) == v[i + 1]); // make sure v is contiguous and ascending
6723 tensor = tensor.fusedim(v[0]+offset);
6724 }
6725 return tensor;
6726 };
6727
6728 // use case: partial_projection of 2-electron functions in svd representation f(1) = \int g(2) h(1,2) d2
6729 // c_i = \sum_j a_j b_ij = \sum_jr a_j b_rj b'_rj
6730 // = \sum_jr ( a_j b_rj) b'_rj )
6731 auto contract2 = [](const auto& svdcoeff, const auto& tensor, const int particle) {
6732#if HAVE_GENTENSOR
6733 const int spectator_particle=(particle+1)%2;
6734 Tensor<Q> gtensor = svdcoeff.get_svdtensor().make_vector_with_weights(particle);
6735 gtensor=gtensor.reshape(svdcoeff.rank(),gtensor.size()/svdcoeff.rank());
6736 MADNESS_CHECK(gtensor.ndim()==2);
6737 Tensor<Q> gtensor_other = svdcoeff.get_svdtensor().ref_vector(spectator_particle);
6738 Tensor<T> tmp1=inner(gtensor,tensor.flat(),1,0); // tmp1(r) = sum_j a'_(r,j) b(j)
6739 MADNESS_CHECK(tmp1.ndim()==1);
6740 Tensor<T> tmp2=inner(gtensor_other,tmp1,0,0); // tmp2(i) = sum_r a_(r,i) tmp1(r)
6741 return tmp2;
6742#else
6743 MADNESS_EXCEPTION("no partial_inner using svd without GenTensor",1);
6744 return Tensor<T>();
6745#endif
6746 };
6747
6748 if (gcoeff.is_full_tensor() and hcoeff.is_full_tensor() and result_coeff.is_full_tensor()) {
6749 // merge multiple contraction dimensions into one
6750 int offset = 0;
6751 Tensor<Q> gtensor = fuse(gcoeff1.full_tensor(), v1, offset);
6752 Tensor<R> htensor = fuse(hcoeff1.full_tensor(), v2, offset);
6753 result_coeff.full_tensor() += inner(gtensor, htensor, v1[0], v2[0]);
6754 if (key.level() > 0) {
6755 gtensor = copy(gcoeff1.full_tensor()(g->get_cdata().s0));
6756 htensor = copy(hcoeff1.full_tensor()(h->get_cdata().s0));
6757 gtensor = fuse(gtensor, v1, offset);
6758 htensor = fuse(htensor, v2, offset);
6759 result_coeff.full_tensor()(get_cdata().s0) -= inner(gtensor, htensor, v1[0], v2[0]);
6760 }
6761 }
6762
6763
6764 // use case: 2-electron functions in svd representation f(1,3) = \int g(1,2) h(2,3) d2
6765 // c_ik = \sum_j a_ij b_jk = \sum_jrr' a_ri a'_rj b_r'j b_r'k
6766 // = \sum_jrr' ( a_ri (a'_rj b_r'j) ) b_r'k
6767 // = \sum_jrr' c_r'i b_r'k
6768 else if (gcoeff.is_svd_tensor() and hcoeff.is_svd_tensor() and result_coeff.is_svd_tensor()) {
6769 MADNESS_CHECK(v1[0]==0 or v1[CDIM-1]==LDIM-1);
6770 MADNESS_CHECK(v2[0]==0 or v2[CDIM-1]==KDIM-1);
6771 int gparticle= v1[0]==0 ? 0 : 1; // which particle to integrate over
6772 int hparticle= v2[0]==0 ? 0 : 1; // which particle to integrate over
6773 // merge multiple contraction dimensions into one
6774 Tensor<Q> gtensor = gcoeff1.get_svdtensor().flat_vector_with_weights(gparticle);
6775 Tensor<Q> gtensor_other = gcoeff1.get_svdtensor().flat_vector((gparticle+1)%2);
6776 Tensor<R> htensor = hcoeff1.get_svdtensor().flat_vector_with_weights(hparticle);
6777 Tensor<R> htensor_other = hcoeff1.get_svdtensor().flat_vector((hparticle+1)%2);
6778 Tensor<T> tmp1=inner(gtensor,htensor,1,1); // tmp1(r,r') = sum_j b(r,j) a(r',j)
6779 Tensor<T> tmp2=inner(tmp1,gtensor_other,0,0); // tmp2(r',i) = sum_r tmp1(r,r') a(r,i)
6781 MADNESS_CHECK(tmp2.dim(0)==htensor_other.dim(0));
6782 w=1.0;
6783 coeffT result_tmp(get_cdata().v2k, get_tensor_type());
6784 result_tmp.get_svdtensor().set_vectors_and_weights(w,tmp2,htensor_other);
6785 if (key.level() > 0) {
6786 GenTensor<Q> gcoeff2 = copy(gcoeff1(g->get_cdata().s0));
6787 GenTensor<R> hcoeff2 = copy(hcoeff1(h->get_cdata().s0));
6788 Tensor<Q> gtensor = gcoeff2.get_svdtensor().flat_vector_with_weights(gparticle);
6789 Tensor<Q> gtensor_other = gcoeff2.get_svdtensor().flat_vector((gparticle+1)%2);
6790 Tensor<R> htensor = hcoeff2.get_svdtensor().flat_vector_with_weights(hparticle);
6791 Tensor<R> htensor_other = hcoeff2.get_svdtensor().flat_vector((hparticle+1)%2);
6792 Tensor<T> tmp1=inner(gtensor,htensor,1,1); // tmp1(r,r') = sum_j b(r,j) a(r',j)
6793 Tensor<T> tmp2=inner(tmp1,gtensor_other,0,0); // tmp2(r',i) = sum_r tmp1(r,r') a(r,i)
6795 MADNESS_CHECK(tmp2.dim(0)==htensor_other.dim(0));
6796 w=1.0;
6797 coeffT result_coeff1(get_cdata().vk, get_tensor_type());
6798 result_coeff1.get_svdtensor().set_vectors_and_weights(w,tmp2,htensor_other);
6799 result_tmp(get_cdata().s0)-=result_coeff1;
6800 }
6801 result_coeff+=result_tmp;
6802 }
6803
6804 // use case: partial_projection of 2-electron functions in svd representation f(1) = \int g(2) h(1,2) d2
6805 // c_i = \sum_j a_j b_ij = \sum_jr a_j b_rj b'_rj
6806 // = \sum_jr ( a_j b_rj) b'_rj )
6807 else if (gcoeff.is_full_tensor() and hcoeff.is_svd_tensor() and result_coeff.is_full_tensor()) {
6808 MADNESS_CHECK(v1[0]==0 and v1[CDIM-1]==LDIM-1);
6809 MADNESS_CHECK(v2[0]==0 or v2[CDIM-1]==KDIM-1);
6810 MADNESS_CHECK(LDIM==CDIM);
6811 int hparticle= v2[0]==0 ? 0 : 1; // which particle to integrate over
6812
6813 Tensor<T> r=contract2(hcoeff1,gcoeff1.full_tensor(),hparticle);
6814 if (key.level()>0) r(get_cdata().s0)-=contract2(copy(hcoeff1(h->get_cdata().s0)),copy(gcoeff.full_tensor()(g->get_cdata().s0)),hparticle);
6815 result_coeff.full_tensor()+=r;
6816 }
6817 // use case: partial_projection of 2-electron functions in svd representation f(1) = \int g(1,2) h(2) d2
6818 // c_i = \sum_j a_ij b_j = \sum_jr a_ri a'_rj b_j
6819 // = \sum_jr ( a_ri (a'_rj b_j) )
6820 else if (gcoeff.is_svd_tensor() and hcoeff.is_full_tensor() and result_coeff.is_full_tensor()) {
6821 MADNESS_CHECK(v1[0]==0 or v1[CDIM-1]==LDIM-1);
6822 MADNESS_CHECK(v2[0]==0 and v2[CDIM-1]==KDIM-1);
6823 MADNESS_CHECK(KDIM==CDIM);
6824 int gparticle= v1[0]==0 ? 0 : 1; // which particle to integrate over
6825
6826 Tensor<T> r=contract2(gcoeff1,hcoeff1.full_tensor(),gparticle);
6827 if (key.level()>0) r(get_cdata().s0)-=contract2(copy(gcoeff1(g->get_cdata().s0)),copy(hcoeff.full_tensor()(h->get_cdata().s0)),gparticle);
6828 result_coeff.full_tensor()+=r;
6829
6830 } else {
6831 MADNESS_EXCEPTION("unknown case in partial_inner_contract",1);
6832 }
6833 }
6834
6835 MADNESS_CHECK(result_coeff.is_assigned());
6836 result_coeff.reduce_rank(get_thresh());
6837
6838 if (coeffs.is_local(key))
6839 coeffs.send(key, &nodeT::accumulate, result_coeff, coeffs, key, get_tensor_args());
6840 else
6842 }
6843
6844 /// Return the inner product with an external function on a specified function node.
6845
6846 /// @param[in] key Key of the function node to compute the inner product on. (the domain of integration)
6847 /// @param[in] c Tensor of coefficients for the function at the function node given by key
6848 /// @param[in] f Reference to FunctionFunctorInterface. This is the externally provided function
6849 /// @return Returns the inner product over the domain of a single function node, no guarantee of accuracy.
6850 T inner_ext_node(keyT key, tensorT c, const std::shared_ptr< FunctionFunctorInterface<T,NDIM> > f) const {
6851 tensorT fvals = tensorT(this->cdata.vk);
6852 // Compute the value of the external function at the quadrature points.
6853 fcube(key, *(f), cdata.quad_x, fvals);
6854 // Convert quadrature point values to scaling coefficients.
6855 tensorT fc = tensorT(values2coeffs(key, fvals));
6856 // Return the inner product of the two functions' scaling coefficients.
6857 return c.trace_conj(fc);
6858 }
6859
6860 /// Call inner_ext_node recursively until convergence.
6861 /// @param[in] key Key of the function node on which to compute inner product (the domain of integration)
6862 /// @param[in] c coeffs for the function at the node given by key
6863 /// @param[in] f Reference to FunctionFunctorInterface. This is the externally provided function
6864 /// @param[in] leaf_refine boolean switch to turn on/off refinement past leaf nodes
6865 /// @param[in] old_inner the inner product on the parent function node
6866 /// @return Returns the inner product over the domain of a single function, checks for convergence.
6867 T inner_ext_recursive(keyT key, tensorT c, const std::shared_ptr< FunctionFunctorInterface<T,NDIM> > f, const bool leaf_refine, T old_inner=T(0)) const {
6868 int i = 0;
6869 tensorT c_child, inner_child;
6870 T new_inner, result = 0.0;
6871
6872 c_child = tensorT(cdata.v2k); // tensor of child coeffs
6873 inner_child = Tensor<double>(pow(2, NDIM)); // child inner products
6874
6875 // If old_inner is default value, assume this is the first call
6876 // and compute inner product on this node.
6877 if (old_inner == T(0)) {
6878 old_inner = inner_ext_node(key, c, f);
6879 }
6880
6881 if (coeffs.find(key).get()->second.has_children()) {
6882 // Since the key has children and we know the func is redundant,
6883 // Iterate over all children of this compute node, computing
6884 // the inner product on each child node. new_inner will store
6885 // the sum of these, yielding a more accurate inner product.
6886 for (KeyChildIterator<NDIM> it(key); it; ++it, ++i) {
6887 const keyT& child = it.key();
6888 tensorT cc = coeffs.find(child).get()->second.coeff().full_tensor_copy();
6889 inner_child(i) = inner_ext_node(child, cc, f);
6890 }
6891 new_inner = inner_child.sum();
6892 } else if (leaf_refine) {
6893 // We need the scaling coefficients of the numerical function
6894 // at each of the children nodes. We can't use project because
6895 // there is no guarantee that the numerical function will have
6896 // a functor. Instead, since we know we are at or below the
6897 // leaf nodes, the wavelet coefficients are zero (to within the
6898 // truncate tolerance). Thus, we can use unfilter() to
6899 // get the scaling coefficients at the next level.
6900 tensorT d = tensorT(cdata.v2k);
6901 d = T(0);
6902 d(cdata.s0) = copy(c);
6903 c_child = unfilter(d);
6904
6905 // Iterate over all children of this compute node, computing
6906 // the inner product on each child node. new_inner will store
6907 // the sum of these, yielding a more accurate inner product.
6908 for (KeyChildIterator<NDIM> it(key); it; ++it, ++i) {
6909 const keyT& child = it.key();
6910 tensorT cc = tensorT(c_child(child_patch(child)));
6911 inner_child(i) = inner_ext_node(child, cc, f);
6912 }
6913 new_inner = inner_child.sum();
6914 } else {
6915 // If we get to here, we are at the leaf nodes and the user has
6916 // specified that they do not want refinement past leaf nodes.
6917 new_inner = old_inner;
6918 }
6919
6920 // Check for convergence. If converged...yay, we're done. If not,
6921 // call inner_ext_node_recursive on each child node and accumulate
6922 // the inner product in result.
6923 // if (std::abs(new_inner - old_inner) <= truncate_tol(thresh, key)) {
6924 if (std::abs(new_inner - old_inner) <= thresh) {
6925 result = new_inner;
6926 } else {
6927 i = 0;
6928 for (KeyChildIterator<NDIM> it(key); it; ++it, ++i) {
6929 const keyT& child = it.key();
6930 tensorT cc = tensorT(c_child(child_patch(child)));
6931 result += inner_ext_recursive(child, cc, f, leaf_refine, inner_child(i));
6932 }
6933 }
6934
6935 return result;
6936 }
6937
6939 const std::shared_ptr< FunctionFunctorInterface<T, NDIM> > fref;
6940 const implT * impl;
6941 const bool leaf_refine;
6942 const bool do_leaves; ///< start with leaf nodes instead of initial_level
6943
6945 const implT * impl, const bool leaf_refine, const bool do_leaves)
6946 : fref(f), impl(impl), leaf_refine(leaf_refine), do_leaves(do_leaves) {};
6947
6948 T operator()(typename dcT::const_iterator& it) const {
6949 if (do_leaves and it->second.is_leaf()) {
6950 tensorT cc = it->second.coeff().full_tensor();
6951 return impl->inner_adaptive_recursive(it->first, cc, fref, leaf_refine, T(0));
6952 } else if ((not do_leaves) and (it->first.level() == impl->initial_level)) {
6953 tensorT cc = it->second.coeff().full_tensor();
6954 return impl->inner_ext_recursive(it->first, cc, fref, leaf_refine, T(0));
6955 } else {
6956 return 0.0;
6957 }
6958 }
6959
6960 T operator()(T a, T b) const {
6961 return (a + b);
6962 }
6963
6964 template <typename Archive> void serialize(const Archive& ar) {
6965 MADNESS_EXCEPTION("NOT IMPLEMENTED", 1);
6966 }
6967 };
6968
6969 /// Return the local part of inner product with external function ... no communication.
6970 /// @param[in] f Reference to FunctionFunctorInterface. This is the externally provided function
6971 /// @param[in] leaf_refine boolean switch to turn on/off refinement past leaf nodes
6972 /// @return Returns local part of the inner product, i.e. over the domain of all function nodes on this compute node.
6973 T inner_ext_local(const std::shared_ptr< FunctionFunctorInterface<T,NDIM> > f, const bool leaf_refine) const {
6975
6977 do_inner_ext_local_ffi(f, this, leaf_refine, false));
6978 }
6979
6980 /// Return the local part of inner product with external function ... no communication.
6981 /// @param[in] f Reference to FunctionFunctorInterface. This is the externally provided function
6982 /// @param[in] leaf_refine boolean switch to turn on/off refinement past leaf nodes
6983 /// @return Returns local part of the inner product, i.e. over the domain of all function nodes on this compute node.
6984 T inner_adaptive_local(const std::shared_ptr< FunctionFunctorInterface<T,NDIM> > f, const bool leaf_refine) const {
6986
6988 do_inner_ext_local_ffi(f, this, leaf_refine, true));
6989 }
6990
6991 /// Call inner_ext_node recursively until convergence.
6992 /// @param[in] key Key of the function node on which to compute inner product (the domain of integration)
6993 /// @param[in] c coeffs for the function at the node given by key
6994 /// @param[in] f Reference to FunctionFunctorInterface. This is the externally provided function
6995 /// @param[in] leaf_refine boolean switch to turn on/off refinement past leaf nodes
6996 /// @param[in] old_inner the inner product on the parent function node
6997 /// @return Returns the inner product over the domain of a single function, checks for convergence.
6999 const std::shared_ptr< FunctionFunctorInterface<T,NDIM> > f,
7000 const bool leaf_refine, T old_inner=T(0)) const {
7001
7002 // the inner product in the current node
7003 old_inner = inner_ext_node(key, c, f);
7004 T result=0.0;
7005
7006 // the inner product in the child nodes
7007
7008 // compute the sum coefficients of the MRA function
7009 tensorT d = tensorT(cdata.v2k);
7010 d = T(0);
7011 d(cdata.s0) = copy(c);
7012 tensorT c_child = unfilter(d);
7013
7014 // compute the inner product in the child nodes
7015 T new_inner=0.0; // child inner products
7016 for (KeyChildIterator<NDIM> it(key); it; ++it) {
7017 const keyT& child = it.key();
7018 tensorT cc = tensorT(c_child(child_patch(child)));
7019 new_inner+= inner_ext_node(child, cc, f);
7020 }
7021
7022 // continue recursion if needed
7023 const double tol=truncate_tol(thresh,key);
7024 if (leaf_refine and (std::abs(new_inner - old_inner) > tol)) {
7025 for (KeyChildIterator<NDIM> it(key); it; ++it) {
7026 const keyT& child = it.key();
7027 tensorT cc = tensorT(c_child(child_patch(child)));
7028 result += inner_adaptive_recursive(child, cc, f, leaf_refine, T(0));
7029 }
7030 } else {
7031 result = new_inner;
7032 }
7033 return result;
7034
7035 }
7036
7037
7038 /// Return the gaxpy product with an external function on a specified
7039 /// function node.
7040 /// @param[in] key Key of the function node on which to compute gaxpy
7041 /// @param[in] lc Tensor of coefficients for the function at the
7042 /// function node given by key
7043 /// @param[in] f Pointer to function of type T that takes coordT
7044 /// arguments. This is the externally provided function and
7045 /// the right argument of gaxpy.
7046 /// @param[in] alpha prefactor of c Tensor for gaxpy
7047 /// @param[in] beta prefactor of fcoeffs for gaxpy
7048 /// @return Returns coefficient tensor of the gaxpy product at specified
7049 /// key, no guarantee of accuracy.
7050 template <typename L>
7051 tensorT gaxpy_ext_node(keyT key, Tensor<L> lc, T (*f)(const coordT&), T alpha, T beta) const {
7052 // Compute the value of external function at the quadrature points.
7053 tensorT fvals = madness::fcube(key, f, cdata.quad_x);
7054 // Convert quadrature point values to scaling coefficients.
7055 tensorT fcoeffs = values2coeffs(key, fvals);
7056 // Return the inner product of the two functions' scaling coeffs.
7057 tensorT c2 = copy(lc);
7058 c2.gaxpy(alpha, fcoeffs, beta);
7059 return c2;
7060 }
7061
7062 /// Return out of place gaxpy using recursive descent.
7063 /// @param[in] key Key of the function node on which to compute gaxpy
7064 /// @param[in] left FunctionImpl, left argument of gaxpy
7065 /// @param[in] lcin coefficients of left at this node
7066 /// @param[in] c coefficients of gaxpy product at this node
7067 /// @param[in] f pointer to function of type T that takes coordT
7068 /// arguments. This is the externally provided function and
7069 /// the right argument of gaxpy.
7070 /// @param[in] alpha prefactor of left argument for gaxpy
7071 /// @param[in] beta prefactor of right argument for gaxpy
7072 /// @param[in] tol convergence tolerance...when the norm of the gaxpy's
7073 /// difference coefficients is less than tol, we are done.
7074 template <typename L>
7075 void gaxpy_ext_recursive(const keyT& key, const FunctionImpl<L,NDIM>* left,
7076 Tensor<L> lcin, tensorT c, T (*f)(const coordT&),
7077 T alpha, T beta, double tol, bool below_leaf) {
7078 typedef typename FunctionImpl<L,NDIM>::dcT::const_iterator literT;
7079
7080 // If we haven't yet reached the leaf level, check whether the
7081 // current key is a leaf node of left. If so, set below_leaf to true
7082 // and continue. If not, make this a parent, recur down, return.
7083 if (not below_leaf) {
7084 bool left_leaf = left->coeffs.find(key).get()->second.is_leaf();
7085 if (left_leaf) {
7086 below_leaf = true;
7087 } else {
7088 this->coeffs.replace(key, nodeT(coeffT(), true));
7089 for (KeyChildIterator<NDIM> it(key); it; ++it) {
7090 const keyT& child = it.key();
7091 woT::task(left->coeffs.owner(child), &implT:: template gaxpy_ext_recursive<L>,
7092 child, left, Tensor<L>(), tensorT(), f, alpha, beta, tol, below_leaf);
7093 }
7094 return;
7095 }
7096 }
7097
7098 // Compute left's coefficients if not provided
7099 Tensor<L> lc = lcin;
7100 if (lc.size() == 0) {
7101 literT it = left->coeffs.find(key).get();
7102 MADNESS_ASSERT(it != left->coeffs.end());
7103 if (it->second.has_coeff())
7104 lc = it->second.coeff().reconstruct_tensor();
7105 }
7106
7107 // Compute this node's coefficients if not provided in function call
7108 if (c.size() == 0) {
7109 c = gaxpy_ext_node(key, lc, f, alpha, beta);
7110 }
7111
7112 // We need the scaling coefficients of the numerical function at
7113 // each of the children nodes. We can't use project because there
7114 // is no guarantee that the numerical function will have a functor.
7115 // Instead, since we know we are at or below the leaf nodes, the
7116 // wavelet coefficients are zero (to within the truncate tolerance).
7117 // Thus, we can use unfilter() to get the scaling coefficients at
7118 // the next level.
7119 Tensor<L> lc_child = Tensor<L>(cdata.v2k); // left's child coeffs
7120 Tensor<L> ld = Tensor<L>(cdata.v2k);
7121 ld = L(0);
7122 ld(cdata.s0) = copy(lc);
7123 lc_child = unfilter(ld);
7124
7125 // Iterate over children of this node,
7126 // storing the gaxpy coeffs in c_child
7127 tensorT c_child = tensorT(cdata.v2k); // tensor of child coeffs
7128 for (KeyChildIterator<NDIM> it(key); it; ++it) {
7129 const keyT& child = it.key();
7130 tensorT lcoeff = tensorT(lc_child(child_patch(child)));
7131 c_child(child_patch(child)) = gaxpy_ext_node(child, lcoeff, f, alpha, beta);
7132 }
7133
7134 // Compute the difference coefficients to test for convergence.
7135 tensorT d = tensorT(cdata.v2k);
7136 d = filter(c_child);
7137 // Filter returns both s and d coefficients, so set scaling
7138 // coefficient part of d to 0 so that we take only the
7139 // norm of the difference coefficients.
7140 d(cdata.s0) = T(0);
7141 double dnorm = d.normf();
7142
7143 // Small d.normf means we've reached a good level of resolution
7144 // Store the coefficients and return.
7145 if (dnorm <= truncate_tol(tol,key)) {
7146 this->coeffs.replace(key, nodeT(coeffT(c,targs), false));
7147 } else {
7148 // Otherwise, make this a parent node and recur down
7149 this->coeffs.replace(key, nodeT(coeffT(), true)); // Interior node
7150
7151 for (KeyChildIterator<NDIM> it(key); it; ++it) {
7152 const keyT& child = it.key();
7153 tensorT child_coeff = tensorT(c_child(child_patch(child)));
7154 tensorT left_coeff = tensorT(lc_child(child_patch(child)));
7155 woT::task(left->coeffs.owner(child), &implT:: template gaxpy_ext_recursive<L>,
7156 child, left, left_coeff, child_coeff, f, alpha, beta, tol, below_leaf);
7157 }
7158 }
7159 }
7160
7161 template <typename L>
7162 void gaxpy_ext(const FunctionImpl<L,NDIM>* left, T (*f)(const coordT&), T alpha, T beta, double tol, bool fence) {
7163 if (world.rank() == coeffs.owner(cdata.key0))
7164 gaxpy_ext_recursive<L> (cdata.key0, left, Tensor<L>(), tensorT(), f, alpha, beta, tol, false);
7165 if (fence)
7166 world.gop.fence();
7167 }
7168
7169 /// project the low-dim function g on the hi-dim function f: result(x) = <this(x,y) | g(y)>
7170
7171 /// invoked by the hi-dim function, a function of NDIM+LDIM
7172
7173 /// Upon return, result matches this, with contributions on all scales
7174 /// @param[in] result lo-dim function of NDIM-LDIM \todo Should this be param[out]?
7175 /// @param[in] gimpl lo-dim function of LDIM
7176 /// @param[in] dim over which dimensions to be integrated: 0..LDIM or LDIM..LDIM+NDIM-1
7177 template<size_t LDIM>
7179 const int dim, const bool fence) {
7180
7181 const keyT& key0=cdata.key0;
7182
7183 if (world.rank() == coeffs.owner(key0)) {
7184
7185 // coeff_op will accumulate the result
7186 typedef project_out_op<LDIM> coeff_opT;
7187 coeff_opT coeff_op(this,result,CoeffTracker<T,LDIM>(gimpl),dim);
7188
7189 // don't do anything on this -- coeff_op will accumulate into result
7190 typedef noop<T,NDIM> apply_opT;
7191 apply_opT apply_op;
7192
7193 woT::task(world.rank(), &implT:: template forward_traverse<coeff_opT,apply_opT>,
7194 coeff_op, apply_op, cdata.key0);
7195
7196 }
7197 if (fence) world.gop.fence();
7198
7199 }
7200
7201
7202 /// project the low-dim function g on the hi-dim function f: result(x) = <f(x,y) | g(y)>
7203 template<size_t LDIM>
7205 bool randomize() const {return false;}
7206
7209 typedef FunctionImpl<T,NDIM-LDIM> implL1;
7210 typedef std::pair<bool,coeffT> argT;
7211
7212 const implT* fimpl; ///< the hi dim function f
7213 mutable implL1* result; ///< the low dim result function
7214 ctL iag; ///< the low dim function g
7215 int dim; ///< 0: project 0..LDIM-1, 1: project LDIM..NDIM-1
7216
7217 // ctor
7218 project_out_op() = default;
7219 project_out_op(const implT* fimpl, implL1* result, const ctL& iag, const int dim)
7220 : fimpl(fimpl), result(result), iag(iag), dim(dim) {}
7222 : fimpl(other.fimpl), result(other.result), iag(other.iag), dim(other.dim) {}
7223
7224
7225 /// do the actual contraction
7227
7228 Key<LDIM> key1,key2,dest;
7229 key.break_apart(key1,key2);
7230
7231 // make the right coefficients
7232 coeffT gcoeff;
7233 if (dim==0) {
7234 gcoeff=iag.get_impl()->parent_to_child(iag.coeff(),iag.key(),key1);
7235 dest=key2;
7236 }
7237 if (dim==1) {
7238 gcoeff=iag.get_impl()->parent_to_child(iag.coeff(),iag.key(),key2);
7239 dest=key1;
7240 }
7241
7242 MADNESS_ASSERT(fimpl->get_coeffs().probe(key)); // must be local!
7243 const nodeT& fnode=fimpl->get_coeffs().find(key).get()->second;
7244 const coeffT& fcoeff=fnode.coeff();
7245
7246 // fast return if possible
7247 if (fcoeff.has_no_data() or gcoeff.has_no_data())
7248 return Future<argT> (argT(fnode.is_leaf(),coeffT()));;
7249
7250 MADNESS_CHECK(gcoeff.is_full_tensor());
7251 tensorT final(result->cdata.vk);
7252 const int k=fcoeff.dim(0);
7253 const int k_ldim=std::pow(k,LDIM);
7254 std::vector<long> shape(LDIM, k);
7255
7256 if (fcoeff.is_full_tensor()) {
7257 // result_i = \sum_j g_j f_ji
7258 const tensorT gtensor = gcoeff.full_tensor().reshape(k_ldim);
7259 const tensorT ftensor = fcoeff.full_tensor().reshape(k_ldim,k_ldim);
7260 final=inner(gtensor,ftensor,0,dim).reshape(shape);
7261
7262 } else if (fcoeff.is_svd_tensor()) {
7263 if (fcoeff.rank()>0) {
7264
7265 // result_i = \sum_jr g_j a_rj w_r b_ri
7266 const int otherdim = (dim + 1) % 2;
7267 const tensorT gtensor = gcoeff.full_tensor().flat();
7268 const tensorT atensor = fcoeff.get_svdtensor().flat_vector(dim); // a_rj
7269 const tensorT btensor = fcoeff.get_svdtensor().flat_vector(otherdim);
7270 const tensorT gatensor = inner(gtensor, atensor, 0, 1); // ga_r
7271 tensorT weights = copy(fcoeff.get_svdtensor().weights_);
7272 weights.emul(gatensor); // ga_r * w_r
7273 // sum over all ranks of b, include new weights:
7274 // result_i = \sum_r ga_r * w_r * b_ri
7275 for (int r = 0; r < fcoeff.rank(); ++r) final += weights(r) * btensor(r, _);
7276 final = final.reshape(shape);
7277 }
7278
7279 } else {
7280 MADNESS_EXCEPTION("unsupported tensor type in project_out_op",1);
7281 }
7282
7283 // accumulate the result
7284 result->coeffs.task(dest, &FunctionNode<T,LDIM>::accumulate2, final, result->coeffs, dest, TaskAttributes::hipri());
7285
7286 return Future<argT> (argT(fnode.is_leaf(),coeffT()));
7287 }
7288
7289 this_type make_child(const keyT& child) const {
7290 Key<LDIM> key1,key2;
7291 child.break_apart(key1,key2);
7292 const Key<LDIM> gkey = (dim==0) ? key1 : key2;
7293
7294 return this_type(fimpl,result,iag.make_child(gkey),dim);
7295 }
7296
7297 /// retrieve the coefficients (parent coeffs might be remote)
7300 return result->world.taskq.add(detail::wrap_mem_fn(*const_cast<this_type *> (this),
7301 &this_type::forward_ctor),fimpl,result,g1,dim);
7302 }
7303
7304 /// taskq-compatible ctor
7305 this_type forward_ctor(const implT* fimpl1, implL1* result1, const ctL& iag1, const int dim1) {
7306 return this_type(fimpl1,result1,iag1,dim1);
7307 }
7308
7309 template <typename Archive> void serialize(const Archive& ar) {
7310 ar & result & iag & fimpl & dim;
7311 }
7312
7313 };
7314
7315
7316 /// project the low-dim function g on the hi-dim function f: this(x) = <f(x,y) | g(y)>
7317
7318 /// invoked by result, a function of NDIM
7319
7320 /// @param[in] f hi-dim function of LDIM+NDIM
7321 /// @param[in] g lo-dim function of LDIM
7322 /// @param[in] dim over which dimensions to be integrated: 0..LDIM or LDIM..LDIM+NDIM-1
7323 template<size_t LDIM>
7324 void project_out2(const FunctionImpl<T,LDIM+NDIM>* f, const FunctionImpl<T,LDIM>* g, const int dim) {
7325
7326 typedef std::pair< keyT,coeffT > pairT;
7327 typedef typename FunctionImpl<T,NDIM+LDIM>::dcT::const_iterator fiterator;
7328
7329 // loop over all nodes of hi-dim f, compute the inner products with all
7330 // appropriate nodes of g, and accumulate in result
7331 fiterator end = f->get_coeffs().end();
7332 for (fiterator it=f->get_coeffs().begin(); it!=end; ++it) {
7333 const Key<LDIM+NDIM> key=it->first;
7334 const FunctionNode<T,LDIM+NDIM> fnode=it->second;
7335 const coeffT& fcoeff=fnode.coeff();
7336
7337 if (fnode.is_leaf() and fcoeff.has_data()) {
7338
7339 // break key into particle: over key1 will be summed, over key2 will be
7340 // accumulated, or vice versa, depending on dim
7341 if (dim==0) {
7342 Key<NDIM> key1;
7343 Key<LDIM> key2;
7344 key.break_apart(key1,key2);
7345
7346 Future<pairT> result;
7347 // sock_it_to_me(key1, result.remote_ref(world));
7348 g->task(coeffs.owner(key1), &implT::sock_it_to_me, key1, result.remote_ref(world), TaskAttributes::hipri());
7349 woT::task(world.rank(),&implT:: template do_project_out<LDIM>,fcoeff,result,key1,key2,dim);
7350
7351 } else if (dim==1) {
7352 Key<LDIM> key1;
7353 Key<NDIM> key2;
7354 key.break_apart(key1,key2);
7355
7356 Future<pairT> result;
7357 // sock_it_to_me(key2, result.remote_ref(world));
7358 g->task(coeffs.owner(key2), &implT::sock_it_to_me, key2, result.remote_ref(world), TaskAttributes::hipri());
7359 woT::task(world.rank(),&implT:: template do_project_out<LDIM>,fcoeff,result,key2,key1,dim);
7360
7361 } else {
7362 MADNESS_EXCEPTION("confused dim in project_out",1);
7363 }
7364 }
7365 }
7367// this->compressed=false;
7368// this->nonstandard=false;
7369// this->redundant=true;
7370 }
7371
7372
7373 /// compute the inner product of two nodes of only some dimensions and accumulate on result
7374
7375 /// invoked by result
7376 /// @param[in] fcoeff coefficients of high dimension LDIM+NDIM
7377 /// @param[in] gpair key and coeffs of low dimension LDIM (possibly a parent node)
7378 /// @param[in] gkey key of actual low dim node (possibly the same as gpair.first, iff gnode exists)
7379 /// @param[in] dest destination node for the result
7380 /// @param[in] dim which dimensions should be contracted: 0..LDIM-1 or LDIM..NDIM+LDIM-1
7381 template<size_t LDIM>
7382 void do_project_out(const coeffT& fcoeff, const std::pair<keyT,coeffT> gpair, const keyT& gkey,
7383 const Key<NDIM>& dest, const int dim) const {
7384
7385 const coeffT gcoeff=parent_to_child(gpair.second,gpair.first,gkey);
7386
7387 // fast return if possible
7388 if (fcoeff.has_no_data() or gcoeff.has_no_data()) return;
7389
7390 // let's specialize for the time being on SVD tensors for f and full tensors of half dim for g
7392 MADNESS_ASSERT(fcoeff.tensor_type()==TT_2D);
7393 const tensorT gtensor=gcoeff.full_tensor();
7394 tensorT result(cdata.vk);
7395
7396 const int otherdim=(dim+1)%2;
7397 const int k=fcoeff.dim(0);
7398 std::vector<Slice> s(fcoeff.config().dim_per_vector()+1,_);
7399
7400 // do the actual contraction
7401 for (int r=0; r<fcoeff.rank(); ++r) {
7402 s[0]=Slice(r,r);
7403 const tensorT contracted_tensor=fcoeff.config().ref_vector(dim)(s).reshape(k,k,k);
7404 const tensorT other_tensor=fcoeff.config().ref_vector(otherdim)(s).reshape(k,k,k);
7405 const double ovlp= gtensor.trace_conj(contracted_tensor);
7406 const double fac=ovlp * fcoeff.config().weights(r);
7407 result+=fac*other_tensor;
7408 }
7409
7410 // accumulate the result
7411 coeffs.task(dest, &nodeT::accumulate2, result, coeffs, dest, TaskAttributes::hipri());
7412 }
7413
7414
7415
7416
7417 /// Returns the maximum local depth of the tree ... no communications.
7418 std::size_t max_local_depth() const;
7419
7420
7421 /// Returns the maximum depth of the tree ... collective ... global sum/broadcast
7422 std::size_t max_depth() const;
7423
7424 /// Returns the max number of nodes on a processor
7425 std::size_t max_nodes() const;
7426
7427 /// Returns the min number of nodes on a processor
7428 std::size_t min_nodes() const;
7429
7430 /// Returns the size of the tree structure of the function ... collective global sum
7431 std::size_t tree_size() const;
7432
7433 /// Returns the number of coefficients in the function for each rank
7434 std::size_t size_local() const;
7435
7436 /// Returns the number of coefficients in the function ... collective global sum
7437 std::size_t size() const;
7438
7439 /// Returns the number of coefficients in the function for this MPI rank
7440 std::size_t nCoeff_local() const;
7441
7442 /// Returns the number of coefficients in the function ... collective global sum
7443 std::size_t nCoeff() const;
7444
7445 /// Returns the number of coefficients in the function ... collective global sum
7446 std::size_t real_size() const;
7447
7448 /// print tree size and size
7449 void print_size(const std::string name) const;
7450
7451 /// print the number of configurations per node
7452 void print_stats() const;
7453
7454 /// In-place scale by a constant
7455 void scale_inplace(const T q, bool fence);
7456
7457 /// Out-of-place scale by a constant
7458 template <typename Q, typename F>
7459 void scale_oop(const Q q, const FunctionImpl<F,NDIM>& f, bool fence) {
7460 typedef typename FunctionImpl<F,NDIM>::nodeT fnodeT;
7461 typedef typename FunctionImpl<F,NDIM>::dcT fdcT;
7462 typename fdcT::const_iterator end = f.coeffs.end();
7463 for (typename fdcT::const_iterator it=f.coeffs.begin(); it!=end; ++it) {
7464 const keyT& key = it->first;
7465 const fnodeT& node = it->second;
7466
7467 if (node.has_coeff()) {
7468 coeffs.replace(key,nodeT(node.coeff()*q,node.has_children()));
7469 }
7470 else {
7471 coeffs.replace(key,nodeT(coeffT(),node.has_children()));
7472 }
7473 }
7474 if (fence)
7475 world.gop.fence();
7476 }
7477
7478 /// Hash a pointer to \c FunctionImpl
7479
7480 /// \param[in] impl pointer to a FunctionImpl
7481 /// \return The hash.
7482 inline friend hashT hash_value(const FunctionImpl<T,NDIM>* pimpl) {
7483 hashT seed = hash_value(pimpl->id().get_world_id());
7484 detail::combine_hash(seed, hash_value(pimpl->id().get_obj_id()));
7485 return seed;
7486 }
7487
7488 /// Hash a shared_ptr to \c FunctionImpl
7489
7490 /// \param[in] impl pointer to a FunctionImpl
7491 /// \return The hash.
7492 inline friend hashT hash_value(const std::shared_ptr<FunctionImpl<T,NDIM>> impl) {
7493 return hash_value(impl.get());
7494 }
7495 };
7496
7497 namespace archive {
7498 template <class Archive, class T, std::size_t NDIM>
7499 struct ArchiveLoadImpl<Archive,const FunctionImpl<T,NDIM>*> {
7500 static void load(const Archive& ar, const FunctionImpl<T,NDIM>*& ptr) {
7501 bool exists=false;
7502 ar & exists;
7503 if (exists) {
7504 uniqueidT id;
7505 ar & id;
7506 World* world = World::world_from_id(id.get_world_id());
7507 MADNESS_ASSERT(world);
7508 auto ptr_opt = world->ptr_from_id< WorldObject< FunctionImpl<T,NDIM> > >(id);
7509 if (!ptr_opt)
7510 MADNESS_EXCEPTION("FunctionImpl: remote operation attempting to use a locally uninitialized object",0);
7511 ptr = static_cast< const FunctionImpl<T,NDIM>*>(*ptr_opt);
7512 if (!ptr)
7513 MADNESS_EXCEPTION("FunctionImpl: remote operation attempting to use an unregistered object",0);
7514 } else {
7515 ptr=nullptr;
7516 }
7517 }
7518 };
7519
7520 template <class Archive, class T, std::size_t NDIM>
7521 struct ArchiveStoreImpl<Archive,const FunctionImpl<T,NDIM>*> {
7522 static void store(const Archive& ar, const FunctionImpl<T,NDIM>*const& ptr) {
7523 bool exists=(ptr) ? true : false;
7524 ar & exists;
7525 if (exists) ar & ptr->id();
7526 }
7527 };
7528
7529 template <class Archive, class T, std::size_t NDIM>
7530 struct ArchiveLoadImpl<Archive, FunctionImpl<T,NDIM>*> {
7531 static void load(const Archive& ar, FunctionImpl<T,NDIM>*& ptr) {
7532 bool exists=false;
7533 ar & exists;
7534 if (exists) {
7535 uniqueidT id;
7536 ar & id;
7537 World* world = World::world_from_id(id.get_world_id());
7538 MADNESS_ASSERT(world);
7539 auto ptr_opt = world->ptr_from_id< WorldObject< FunctionImpl<T,NDIM> > >(id);
7540 if (!ptr_opt)
7541 MADNESS_EXCEPTION("FunctionImpl: remote operation attempting to use a locally uninitialized object",0);
7542 ptr = static_cast< FunctionImpl<T,NDIM>*>(*ptr_opt);
7543 if (!ptr) {
7544 auto ids=world->get_object_ids();
7545 print(world->get_world_ids());
7546 MADNESS_EXCEPTION("FunctionImpl: remote operation attempting to use an unregistered object",0);
7547 }
7548 } else {
7549 ptr=nullptr;
7550 }
7551 }
7552 };
7553
7554 template <class Archive, class T, std::size_t NDIM>
7555 struct ArchiveStoreImpl<Archive, FunctionImpl<T,NDIM>*> {
7556 static void store(const Archive& ar, FunctionImpl<T,NDIM>*const& ptr) {
7557 bool exists=(ptr) ? true : false;
7558 ar & exists;
7559 if (exists) ar & ptr->id();
7560 // ar & ptr->id();
7561 }
7562 };
7563
7564 template <class Archive, class T, std::size_t NDIM>
7565 struct ArchiveLoadImpl<Archive, std::shared_ptr<const FunctionImpl<T,NDIM> > > {
7566 static void load(const Archive& ar, std::shared_ptr<const FunctionImpl<T,NDIM> >& ptr) {
7567 const FunctionImpl<T,NDIM>* f = nullptr;
7569 ptr.reset(f, [] (const FunctionImpl<T,NDIM> *p_) -> void {});
7570 }
7571 };
7572
7573 template <class Archive, class T, std::size_t NDIM>
7574 struct ArchiveStoreImpl<Archive, std::shared_ptr<const FunctionImpl<T,NDIM> > > {
7575 static void store(const Archive& ar, const std::shared_ptr<const FunctionImpl<T,NDIM> >& ptr) {
7577 }
7578 };
7579
7580 template <class Archive, class T, std::size_t NDIM>
7581 struct ArchiveLoadImpl<Archive, std::shared_ptr<FunctionImpl<T,NDIM> > > {
7582 static void load(const Archive& ar, std::shared_ptr<FunctionImpl<T,NDIM> >& ptr) {
7583 FunctionImpl<T,NDIM>* f = nullptr;
7585 ptr.reset(f, [] (FunctionImpl<T,NDIM> *p_) -> void {});
7586 }
7587 };
7588
7589 template <class Archive, class T, std::size_t NDIM>
7590 struct ArchiveStoreImpl<Archive, std::shared_ptr<FunctionImpl<T,NDIM> > > {
7591 static void store(const Archive& ar, const std::shared_ptr<FunctionImpl<T,NDIM> >& ptr) {
7593 }
7594 };
7595 }
7596
7597}
7598
7599#endif // MADNESS_MRA_FUNCIMPL_H__INCLUDED
double w(double t, double eps)
Definition DKops.h:22
double q(double t)
Definition DKops.h:18
This header should include pretty much everything needed for the parallel runtime.
An integer with atomic set, get, read+increment, read+decrement, and decrement+test operations.
Definition atomicint.h:126
long dim(int i) const
Returns the size of dimension i.
Definition basetensor.h:147
long ndim() const
Returns the number of dimensions in the tensor.
Definition basetensor.h:144
long size() const
Returns the number of elements in the tensor.
Definition basetensor.h:138
Definition displacements.h:551
a class to track where relevant (parent) coeffs are
Definition funcimpl.h:814
const keyT & key() const
const reference to the key
Definition funcimpl.h:864
CoeffTracker(const CoeffTracker &other, const datumT &datum)
ctor with a pair<keyT,nodeT>
Definition funcimpl.h:844
const LeafStatus & is_leaf() const
const reference to is_leaf flag
Definition funcimpl.h:888
const implT * impl
the funcimpl that has the coeffs
Definition funcimpl.h:823
CoeffTracker & operator=(const CoeffTracker &other)=default
LeafStatus
Definition funcimpl.h:820
@ yes
Definition funcimpl.h:820
@ no
Definition funcimpl.h:820
@ unknown
Definition funcimpl.h:820
CoeffTracker(const CoeffTracker &other)
copy ctor
Definition funcimpl.h:852
double dnorm(const keyT &key) const
return the s and dnorm belonging to the passed-in key
Definition funcimpl.h:881
coeffT coeff_
the coefficients belonging to key
Definition funcimpl.h:829
const implT * get_impl() const
const reference to impl
Definition funcimpl.h:858
const coeffT & coeff() const
const reference to the coeffs
Definition funcimpl.h:861
keyT key_
the current key, which must exists in impl
Definition funcimpl.h:825
double dnorm_
norm of d coefficients corresponding to key
Definition funcimpl.h:831
CoeffTracker(const implT *impl)
the initial ctor making the root key
Definition funcimpl.h:839
void serialize(const Archive &ar)
serialization
Definition funcimpl.h:940
Future< CoeffTracker > activate() const
find the coefficients
Definition funcimpl.h:917
CoeffTracker()
default ctor
Definition funcimpl.h:836
GenTensor< T > coeffT
Definition funcimpl.h:818
CoeffTracker make_child(const keyT &child) const
make a child of this, ignoring the coeffs
Definition funcimpl.h:891
FunctionImpl< T, NDIM > implT
Definition funcimpl.h:816
std::pair< Key< NDIM >, ShallowNode< T, NDIM > > datumT
Definition funcimpl.h:819
CoeffTracker forward_ctor(const CoeffTracker &other, const datumT &datum) const
taskq-compatible forwarding to the ctor
Definition funcimpl.h:934
LeafStatus is_leaf_
flag if key is a leaf node
Definition funcimpl.h:827
coeffT coeff(const keyT &key) const
return the coefficients belonging to the passed-in key
Definition funcimpl.h:872
Key< NDIM > keyT
Definition funcimpl.h:817
CompositeFunctorInterface implements a wrapper of holding several functions and functors.
Definition function_interface.h:172
Definition worldhashmap.h:396
Tri-diagonal operator traversing tree primarily for derivative operator.
Definition derivative.h:73
Holds displacements for applying operators to avoid replicating for all operators.
Definition displacements.h:78
const std::vector< Key< NDIM > > & get_disp(Level n, const array_of_bools< NDIM > &kernel_lattice_sum_axes)
Definition displacements.h:279
FunctionCommonData holds all Function data common for given k.
Definition function_common_data.h:52
Tensor< double > quad_phit
transpose of quad_phi
Definition function_common_data.h:102
Tensor< double > quad_phiw
quad_phiw(i,j) = at x[i] value of w[i]*phi[j]
Definition function_common_data.h:103
std::vector< long > vk
(k,...) used to initialize Tensors
Definition function_common_data.h:93
std::vector< Slice > s0
s[0] in each dimension to get scaling coeff
Definition function_common_data.h:91
static const FunctionCommonData< T, NDIM > & get(int k)
Definition function_common_data.h:111
static void _init_quadrature(int k, int npt, Tensor< double > &quad_x, Tensor< double > &quad_w, Tensor< double > &quad_phi, Tensor< double > &quad_phiw, Tensor< double > &quad_phit)
Initialize the quadrature information.
Definition mraimpl.h:93
collect common functionality does not need to be member function of funcimpl
Definition function_common_data.h:135
const FunctionCommonData< T, NDIM > & cdata
Definition function_common_data.h:138
GenTensor< T > coeffs2values(const Key< NDIM > &key, const GenTensor< T > &coeff) const
Definition function_common_data.h:142
Tensor< T > values2coeffs(const Key< NDIM > &key, const Tensor< T > &values) const
Definition function_common_data.h:155
FunctionDefaults holds default paramaters as static class members.
Definition funcdefaults.h:101
static const double & get_thresh()
Returns the default threshold.
Definition funcdefaults.h:183
static int get_max_refine_level()
Gets the default maximum adaptive refinement level.
Definition funcdefaults.h:220
static const Tensor< double > & get_cell_width()
Returns the width of each user cell dimension.
Definition funcdefaults.h:390
static bool get_apply_randomize()
Gets the random load balancing for integral operators flag.
Definition funcdefaults.h:294
static const Tensor< double > & get_cell()
Gets the user cell for the simulation.
Definition funcdefaults.h:352
FunctionFactory implements the named-parameter idiom for Function.
Definition function_factory.h:86
bool _refine
Definition function_factory.h:99
bool _empty
Definition function_factory.h:100
bool _fence
Definition function_factory.h:103
Abstract base class interface required for functors used as input to Functions.
Definition function_interface.h:68
Definition funcimpl.h:5721
double operator()(double a, double b) const
Definition funcimpl.h:5747
const opT * func
Definition funcimpl.h:5723
Tensor< double > qx
Definition funcimpl.h:5725
double operator()(typename dcT::const_iterator &it) const
Definition funcimpl.h:5738
void serialize(const Archive &ar)
Definition funcimpl.h:5752
do_err_box(const implT *impl, const opT *func, int npt, const Tensor< double > &qx, const Tensor< double > &quad_phit, const Tensor< double > &quad_phiw)
Definition funcimpl.h:5731
int npt
Definition funcimpl.h:5724
Tensor< double > quad_phiw
Definition funcimpl.h:5727
const implT * impl
Definition funcimpl.h:5722
Tensor< double > quad_phit
Definition funcimpl.h:5726
do_err_box(const do_err_box &e)
Definition funcimpl.h:5735
FunctionImpl holds all Function state to facilitate shallow copy semantics.
Definition funcimpl.h:970
std::tuple< std::set< Key< NDIM > >, std::map< Key< CDIM >, double > > get_contraction_node_lists(const std::size_t n, const std::array< int, CDIM > &v) const
for contraction two functions f(x,z) = \int g(x,y) h(y,z) dy
Definition funcimpl.h:6500
void copy_coeffs(const FunctionImpl< Q, NDIM > &other, bool fence)
Copy coeffs from other into self.
Definition funcimpl.h:1228
bool is_nonstandard() const
Definition mraimpl.h:275
void insert_serialized_coeffs(std::vector< unsigned char > &v)
insert coeffs from vector archive into this
Definition funcimpl.h:1284
T eval_cube(Level n, coordT &x, const tensorT &c) const
Definition mraimpl.h:2096
void partial_inner_contract(const FunctionImpl< Q, LDIM > *g, const FunctionImpl< R, KDIM > *h, const std::array< int, CDIM > v1, const std::array< int, CDIM > v2, const Key< NDIM > &key, const std::list< Key< CDIM > > &j_key_list)
tensor contraction part of partial_inner
Definition funcimpl.h:6666
AtomicInt large
Definition funcimpl.h:1085
Timer timer_target_driven
Definition funcimpl.h:1083
void binaryXX(const FunctionImpl< L, NDIM > *left, const FunctionImpl< R, NDIM > *right, const opT &op, bool fence)
Definition funcimpl.h:3416
void do_apply(const opT *op, const keyT &key, const Tensor< R > &c)
apply an operator on the coeffs c (at node key)
Definition funcimpl.h:5043
void do_print_tree_graphviz(const keyT &key, std::ostream &os, Level maxlevel) const
Functor for the do_print_tree method (using GraphViz)
Definition mraimpl.h:2850
void add_keys_to_map(mapT *map, int index) const
Adds keys to union of local keys with specified index.
Definition funcimpl.h:6077
void change_tensor_type1(const TensorArgs &targs, bool fence)
change the tensor type of the coefficients in the FunctionNode
Definition mraimpl.h:1131
void gaxpy_ext_recursive(const keyT &key, const FunctionImpl< L, NDIM > *left, Tensor< L > lcin, tensorT c, T(*f)(const coordT &), T alpha, T beta, double tol, bool below_leaf)
Definition funcimpl.h:7075
int initial_level
Initial level for refinement.
Definition funcimpl.h:999
int max_refine_level
Do not refine below this level.
Definition funcimpl.h:1003
double do_apply_kernel3(const opT *op, const GenTensor< R > &coeff, const do_op_args< OPDIM > &args, const TensorArgs &apply_targs)
same as do_apply_kernel2, but use low rank tensors as input and low rank tensors as output
Definition funcimpl.h:5001
void hartree_product(const std::vector< std::shared_ptr< FunctionImpl< T, LDIM > > > p1, const std::vector< std::shared_ptr< FunctionImpl< T, LDIM > > > p2, const leaf_opT &leaf_op, bool fence)
given two functions of LDIM, perform the Hartree/Kronecker/outer product
Definition funcimpl.h:3957
void traverse_tree(const coeff_opT &coeff_op, const apply_opT &apply_op, const keyT &key) const
traverse a non-existing tree
Definition funcimpl.h:3927
void do_square_inplace(const keyT &key)
int special_level
Minimium level for refinement on special points.
Definition funcimpl.h:1000
void do_apply_kernel(const opT *op, const Tensor< R > &c, const do_op_args< OPDIM > &args)
for fine-grain parallelism: call the apply method of an operator in a separate task
Definition funcimpl.h:4935
compressT compress_op(const keyT &key, const std::vector< Future< compressT > > &v, bool nonstandard)
calculate the wavelet coefficients using the sum coefficients of all child nodes
Definition mraimpl.h:1708
double errsq_local(const opT &func) const
Returns the sum of squares of errors from local info ... no comms.
Definition funcimpl.h:5759
WorldContainer< keyT, nodeT > dcT
Type of container holding the coefficients.
Definition funcimpl.h:982
void evaldepthpt(const Vector< double, NDIM > &xin, const keyT &keyin, const typename Future< Level >::remote_refT &ref)
Get the depth of the tree at a point in simulation coordinates.
Definition mraimpl.h:3121
void scale_inplace(const T q, bool fence)
In-place scale by a constant.
Definition mraimpl.h:3292
void gaxpy_oop_reconstructed(const double alpha, const implT &f, const double beta, const implT &g, const bool fence)
perform: this= alpha*f + beta*g, invoked by result
Definition mraimpl.h:225
void unary_op_coeff_inplace(const opT &op, bool fence)
Definition funcimpl.h:2232
World & world
Definition funcimpl.h:989
void apply_1d_realspace_push_op(const archive::archive_ptr< const opT > &pop, int axis, const keyT &key, const Tensor< R > &c)
Definition funcimpl.h:3995
bool is_redundant() const
Returns true if the function is redundant.
Definition mraimpl.h:264
FunctionNode< T, NDIM > nodeT
Type of node.
Definition funcimpl.h:980
std::size_t nCoeff_local() const
Returns the number of coefficients in the function for this MPI rank.
Definition mraimpl.h:1981
void print_size(const std::string name) const
print tree size and size
Definition mraimpl.h:2000
FunctionImpl(const FunctionImpl< T, NDIM > &p)
void print_info() const
Prints summary of data distribution.
Definition mraimpl.h:851
void abs_inplace(bool fence)
Definition mraimpl.h:3304
void binaryXXa(const keyT &key, const FunctionImpl< L, NDIM > *left, const Tensor< L > &lcin, const FunctionImpl< R, NDIM > *right, const Tensor< R > &rcin, const opT &op)
Definition funcimpl.h:3285
void print_timer() const
Definition mraimpl.h:369
void evalR(const Vector< double, NDIM > &xin, const keyT &keyin, const typename Future< long >::remote_refT &ref)
Get the rank of leaf box of the tree at a point in simulation coordinates.
Definition mraimpl.h:3163
const FunctionCommonData< T, NDIM > & cdata
Definition funcimpl.h:1009
void do_print_grid(const std::string filename, const std::vector< keyT > &keys) const
print the grid in xyz format
Definition mraimpl.h:596
void mulXXa(const keyT &key, const FunctionImpl< L, NDIM > *left, const Tensor< L > &lcin, const FunctionImpl< R, NDIM > *right, const Tensor< R > &rcin, double tol)
Definition funcimpl.h:3199
bool has_coefficients_on_leaves_only() const
Returns true if only the leaves of this tree carry its coefficients.
Definition mraimpl.h:290
int get_truncate_mode() const
Definition funcimpl.h:1869
const std::vector< Vector< double, NDIM > > & get_special_points() const
Definition funcimpl.h:994
Future< compressT > compress_spawn(const keyT &key, bool nonstandard, bool keepleaves, bool redundant1)
Invoked on node where key is local.
Definition mraimpl.h:3451
std::size_t nCoeff() const
Returns the number of coefficients in the function ... collective global sum.
Definition mraimpl.h:1991
double vol_nsphere(int n, double R)
Definition funcimpl.h:5031
keyT neighbor_in_volume(const keyT &key, const keyT &disp) const
Returns key of general neighbor that resides in-volume.
Definition mraimpl.h:3423
void compress(const TreeState newstate, bool fence)
compress the wave function
Definition mraimpl.h:1541
void do_dirac_convolution(FunctionImpl< T, LDIM > *f, bool fence) const
Definition funcimpl.h:2315
Future< bool > truncate_spawn(const keyT &key, double tol)
Returns true if after truncation this node has coefficients.
Definition mraimpl.h:2695
void print_type_in_compilation_error(R &&)
Definition funcimpl.h:6382
Future< double > norm_tree_spawn(const keyT &key)
Definition mraimpl.h:1611
std::vector< keyT > local_leaf_keys() const
return the keys of the local leaf boxes
Definition mraimpl.h:570
MADNESS_ASSERT(this->is_redundant()==g.is_redundant())
void do_print_tree(const keyT &key, std::ostream &os, Level maxlevel) const
Functor for the do_print_tree method.
Definition mraimpl.h:2768
void vtransform(const std::vector< std::shared_ptr< FunctionImpl< R, NDIM > > > &vright, const Tensor< Q > &c, const std::vector< std::shared_ptr< FunctionImpl< T, NDIM > > > &vleft, double tol, bool fence)
Definition funcimpl.h:3027
void unset_functor()
Definition mraimpl.h:324
void refine_spawn(const opT &op, const keyT &key)
Definition funcimpl.h:4760
void apply_1d_realspace_push(const opT &op, const FunctionImpl< R, NDIM > *f, int axis, bool fence)
Definition funcimpl.h:4046
void set_truncate_mode(int mode)
Definition funcimpl.h:1870
void do_print_plane(const std::string filename, std::vector< Tensor< double > > plotinfo, const int xaxis, const int yaxis, const coordT el2)
print the MRA structure
Definition mraimpl.h:511
std::pair< Key< NDIM >, ShallowNode< T, NDIM > > find_datum(keyT key) const
return the a std::pair<key, node>, which MUST exist
Definition mraimpl.h:997
void set_functor(const std::shared_ptr< FunctionFunctorInterface< T, NDIM > > functor1)
Definition mraimpl.h:305
std::enable_if< NDIM==FDIM >::type read_grid2(const std::string gridfile, std::shared_ptr< FunctionFunctorInterface< double, NDIM > > vnuc_functor)
read data from a grid
Definition funcimpl.h:1763
bool verify_tree_state_local() const
check that the tree state and the coeffs are consistent
Definition mraimpl.h:171
const std::shared_ptr< WorldDCPmapInterface< Key< NDIM > > > & get_pmap() const
Definition mraimpl.h:209
Tensor< Q > fcube_for_mul(const keyT &child, const keyT &parent, const Tensor< Q > &coeff) const
Compute the function values for multiplication.
Definition funcimpl.h:2079
Timer timer_filter
Definition funcimpl.h:1081
void sock_it_to_me(const keyT &key, const RemoteReference< FutureImpl< std::pair< keyT, coeffT > > > &ref) const
Walk up the tree returning pair(key,node) for first node with coefficients.
Definition mraimpl.h:2908
void recursive_apply(opT &apply_op, const implT *fimpl, implT *rimpl, const bool fence)
traverse an existing tree and apply an operator
Definition funcimpl.h:5578
double get_thresh() const
Definition mraimpl.h:340
void trickle_down(bool fence)
sum all the contributions from all scales after applying an operator in mod-NS form
Definition mraimpl.h:1388
bool autorefine
If true, autorefine where appropriate.
Definition funcimpl.h:1005
bool halo_enabled() const
Is a neighbor halo staged on this function?
Definition funcimpl.h:1032
void set_autorefine(bool value)
Definition mraimpl.h:349
tensorT filter(const tensorT &s) const
Transform sum coefficients at level n to sums+differences at level n-1.
Definition mraimpl.h:1184
void chop_at_level(const int n, const bool fence=true)
remove all nodes with level higher than n
Definition mraimpl.h:1147
void unaryXXvalues(const FunctionImpl< Q, NDIM > *func, const opT &op, bool fence)
Definition funcimpl.h:3443
void partial_inner(const FunctionImpl< Q, LDIM > &g, const FunctionImpl< R, KDIM > &h, const std::array< int, CDIM > v1, const std::array< int, CDIM > v2)
invoked by result
Definition funcimpl.h:6398
TreeState tree_state
Definition funcimpl.h:1012
void print_tree_json(std::ostream &os=std::cout, Level maxlevel=10000) const
Definition mraimpl.h:2788
coeffT parent_to_child_NS(const keyT &child, const keyT &parent, const coeffT &coeff) const
Directly project parent NS coeffs to child NS coeffs.
Definition mraimpl.h:725
void copy_coeffs_different_world(const FunctionImpl< Q, NDIM > &other)
Copy coefficients from other funcimpl with possibly different world and on a different node.
Definition funcimpl.h:1238
void mapdim(const implT &f, const std::vector< long > &map, bool fence)
Permute the dimensions of f according to map, result on this.
Definition mraimpl.h:1089
bool is_compressed() const
Returns true if the function is compressed.
Definition mraimpl.h:252
void receive_halo(const std::vector< std::pair< keyT, coeffT > > &buf) const
Insert pushed neighbor nodes into the halo; runs as a task, concurrently with other pushes.
Definition funcimpl.h:1052
Vector< double, NDIM > coordT
Type of vector holding coordinates.
Definition funcimpl.h:984
void apply(opT &op, const FunctionImpl< R, NDIM > &f, bool fence)
apply an operator on f to return this
Definition funcimpl.h:5261
Tensor< T > tensorT
Type of tensor for anything but to hold coeffs.
Definition funcimpl.h:977
void mirror(const implT &f, const std::vector< long > &mirror, bool fence)
mirror the dimensions of f according to map, result on this
Definition mraimpl.h:1098
T inner_adaptive_recursive(keyT key, const tensorT &c, const std::shared_ptr< FunctionFunctorInterface< T, NDIM > > f, const bool leaf_refine, T old_inner=T(0)) const
Definition funcimpl.h:6998
void store(Archive &ar)
Definition funcimpl.h:1416
void do_binary_op(const keyT &key, const Tensor< L > &left, const std::pair< keyT, Tensor< R > > &arg, const opT &op)
Functor for the binary_op method.
Definition funcimpl.h:2181
void gaxpy_ext(const FunctionImpl< L, NDIM > *left, T(*f)(const coordT &), T alpha, T beta, double tol, bool fence)
Definition funcimpl.h:7162
void accumulate_trees(FunctionImpl< Q, NDIM > &result, const R alpha, const bool fence=true) const
merge the trees of this and other, while multiplying them with the alpha or beta, resp
Definition funcimpl.h:1337
void print_stats() const
print the number of configurations per node
Definition mraimpl.h:2040
void broaden(const array_of_bools< NDIM > &is_periodic, bool fence)
Definition mraimpl.h:1337
coeffT truncate_reconstructed_op(const keyT &key, const std::vector< Future< coeffT > > &v, const double tol)
given the sum coefficients of all children, truncate or not
Definition mraimpl.h:1658
void refine_op(const opT &op, const keyT &key)
Definition funcimpl.h:4735
static Tensor< TENSOR_RESULT_TYPE(T, R) > inner_local(const std::vector< const FunctionImpl< T, NDIM > * > &left, const std::vector< const FunctionImpl< R, NDIM > * > &right, bool sym)
Definition funcimpl.h:6278
void fcube(const keyT &key, const FunctionFunctorInterface< T, NDIM > &f, const Tensor< double > &qx, tensorT &fval) const
Evaluate function at quadrature points in the specified box.
Definition mraimpl.h:2530
Timer timer_change_tensor_type
Definition funcimpl.h:1079
void forward_do_diff1(const DerivativeBase< T, NDIM > *D, const implT *f, const keyT &key, const std::pair< keyT, coeffT > &left, const std::pair< keyT, coeffT > &center, const std::pair< keyT, coeffT > &right)
Definition mraimpl.h:950
std::vector< Slice > child_patch(const keyT &child) const
Returns patch referring to coeffs of child in parent box.
Definition mraimpl.h:714
void print_tree_graphviz(std::ostream &os=std::cout, Level maxlevel=10000) const
Definition mraimpl.h:2841
void set_tree_state(const TreeState &state)
Definition funcimpl.h:1466
std::size_t min_nodes() const
Returns the min number of nodes on a processor.
Definition mraimpl.h:1932
void copy_coeffs_same_world(const FunctionImpl< Q, NDIM > &other, bool fence)
Copy coeffs from other into self.
Definition funcimpl.h:1291
std::shared_ptr< FunctionFunctorInterface< T, NDIM > > functor
Definition funcimpl.h:1011
Timer timer_compress_svd
Definition funcimpl.h:1082
Tensor< TENSOR_RESULT_TYPE(T, R)> mul(const Tensor< T > &c1, const Tensor< R > &c2, const int npt, const keyT &key) const
multiply the values of two coefficient tensors using a custom number of grid points
Definition funcimpl.h:2154
void make_redundant(const bool fence)
convert this to redundant, i.e. have sum coefficients on all levels
Definition mraimpl.h:1569
void load(Archive &ar)
Definition funcimpl.h:1398
std::size_t max_nodes() const
Returns the max number of nodes on a processor.
Definition mraimpl.h:1923
T inner_ext_local(const std::shared_ptr< FunctionFunctorInterface< T, NDIM > > f, const bool leaf_refine) const
Definition funcimpl.h:6973
coeffT upsample(const keyT &key, const coeffT &coeff) const
upsample the sum coefficients of level 1 to sum coeffs on level n+1
Definition mraimpl.h:1263
TensorArgs targs
type of tensor to be used in the FunctionNodes
Definition funcimpl.h:1007
void flo_unary_op_node_inplace(const opT &op, bool fence)
Definition funcimpl.h:2344
std::size_t size_local() const
Returns the number of coefficients in the function for each rank.
Definition mraimpl.h:1950
GenTensor< Q > values2coeffs(const keyT &key, const GenTensor< Q > &values) const
Definition funcimpl.h:2058
void plot_cube_kernel(archive::archive_ptr< Tensor< T > > ptr, const keyT &key, const coordT &plotlo, const coordT &plothi, const std::vector< long > &npt, bool eval_refine) const
Definition mraimpl.h:3526
T trace_local() const
Returns int(f(x),x) in local volume.
Definition mraimpl.h:3346
void print_grid(const std::string filename) const
Definition mraimpl.h:554
void replicate_on_hosts(bool fence=true)
Definition funcimpl.h:1207
bool get_autorefine() const
Definition mraimpl.h:346
int k
Wavelet order.
Definition funcimpl.h:997
void vtransform_doit(const std::shared_ptr< FunctionImpl< R, NDIM > > &right, const Tensor< Q > &c, const std::vector< std::shared_ptr< FunctionImpl< T, NDIM > > > &vleft, double tol)
Definition funcimpl.h:2876
MADNESS_CHECK(this->is_reconstructed())
void phi_for_mul(Level np, Translation lp, Level nc, Translation lc, Tensor< double > &phi) const
Compute the Legendre scaling functions for multiplication.
Definition mraimpl.h:3314
Future< std::pair< keyT, coeffT > > find_me(const keyT &key) const
find_me. Called by diff_bdry to get coefficients of boundary function
Definition mraimpl.h:3438
TensorType get_tensor_type() const
Definition mraimpl.h:331
void do_project_out(const coeffT &fcoeff, const std::pair< keyT, coeffT > gpair, const keyT &gkey, const Key< NDIM > &dest, const int dim) const
compute the inner product of two nodes of only some dimensions and accumulate on result
Definition funcimpl.h:7382
void remove_leaf_coefficients(const bool fence)
Definition mraimpl.h:1563
void insert_zero_down_to_initial_level(const keyT &key)
Initialize nodes to zero function at initial_level of refinement.
Definition mraimpl.h:2664
void do_diff1(const DerivativeBase< T, NDIM > *D, const implT *f, const keyT &key, const std::pair< keyT, coeffT > &left, const std::pair< keyT, coeffT > &center, const std::pair< keyT, coeffT > &right)
Definition mraimpl.h:961
typedef TENSOR_RESULT_TYPE(T, R) resultT
void unary_op_node_inplace(const opT &op, bool fence)
Definition funcimpl.h:2253
T inner_adaptive_local(const std::shared_ptr< FunctionFunctorInterface< T, NDIM > > f, const bool leaf_refine) const
Definition funcimpl.h:6984
void do_print_tree_json(const keyT &key, std::multimap< Level, std::tuple< tranT, std::string > > &data, Level maxlevel) const
Functor for the do_print_tree_json method.
Definition mraimpl.h:2819
std::multimap< Key< FDIM >, std::list< Key< CDIM > > > recur_down_for_contraction_map(const keyT &key, const nodeT &node, const std::array< int, CDIM > &v_this, const std::array< int, CDIM > &v_other, const std::set< Key< ODIM > > &ij_other_list, const std::map< Key< CDIM >, double > &j_other_list, bool this_first, const double thresh)
make a map of all nodes that will contribute to a partial inner product
Definition funcimpl.h:6553
std::shared_ptr< FunctionImpl< T, NDIM > > pimplT
pointer to this class
Definition funcimpl.h:976
TENSOR_RESULT_TYPE(T, R) dot_local(const FunctionImpl< R
Returns the dot product ASSUMING same distribution.
void finalize_sum()
after summing up we need to do some cleanup;
Definition mraimpl.h:1878
std::enable_if< NDIM==FDIM >::type read_grid(const std::string keyfile, const std::string gridfile, std::shared_ptr< FunctionFunctorInterface< double, NDIM > > vnuc_functor)
read data from a grid
Definition funcimpl.h:1656
dcT coeffs
The coefficients.
Definition funcimpl.h:1014
bool exists_and_is_leaf(const keyT &key) const
Definition mraimpl.h:1307
static std::complex< Real > conj(const std::complex< Real > &x)
Definition funcimpl.h:6272
void make_Vphi(const opT &leaf_op, const bool fence=true)
assemble the function V*phi using V and phi given from the functor
Definition funcimpl.h:4527
void unaryXX(const FunctionImpl< Q, NDIM > *func, const opT &op, bool fence)
Definition funcimpl.h:3430
std::vector< std::pair< int, const coeffT * > > mapvecT
Type of the entry in the map returned by make_key_vec_map.
Definition funcimpl.h:6071
void project_out(FunctionImpl< T, NDIM-LDIM > *result, const FunctionImpl< T, LDIM > *gimpl, const int dim, const bool fence)
project the low-dim function g on the hi-dim function f: result(x) = <this(x,y) | g(y)>
Definition funcimpl.h:7178
void verify_tree() const
Verify tree is properly constructed ... global synchronization involved.
Definition mraimpl.h:113
void do_square_inplace2(const keyT &parent, const keyT &child, const tensorT &parent_coeff)
void gaxpy_inplace_reconstructed(const T &alpha, const FunctionImpl< Q, NDIM > &g, const R &beta, const bool fence)
Definition funcimpl.h:1305
void undo_replicate(bool fence=true)
Definition funcimpl.h:1212
void set_tensor_args(const TensorArgs &t)
Definition mraimpl.h:337
GenTensor< Q > fcube_for_mul(const keyT &child, const keyT &parent, const GenTensor< Q > &coeff) const
Compute the function values for multiplication.
Definition funcimpl.h:2107
Range< typename dcT::const_iterator > rangeT
Definition funcimpl.h:5862
std::size_t real_size() const
Returns the number of coefficients in the function ... collective global sum.
Definition mraimpl.h:1968
bool exists_and_has_children(const keyT &key) const
Definition mraimpl.h:1302
void sum_down_spawn(const keyT &key, const coeffT &s)
is this the same as trickle_down() ?
Definition mraimpl.h:894
void multi_to_multi_op_values(const opT &op, const std::vector< implT * > &vin, std::vector< implT * > &vout, const bool fence=true)
Inplace operate on many functions (impl's) with an operator within a certain box.
Definition funcimpl.h:2998
long box_interior[1000]
Definition funcimpl.h:3475
std::atomic< ConcurrentHashMap< keyT, coeffT > * > neighbor_halo_
Neighbor coefficients pushed here by whoever owns them; null until something stages.
Definition funcimpl.h:1025
keyT neighbor(const keyT &key, const keyT &disp, const array_of_bools< NDIM > &is_periodic) const
Returns key of general neighbor enforcing BC.
Definition mraimpl.h:3408
GenTensor< Q > NS_fcube_for_mul(const keyT &child, const keyT &parent, const GenTensor< Q > &coeff, const bool s_only) const
Compute the function values for multiplication.
Definition funcimpl.h:1977
rangeT range(coeffs.begin(), coeffs.end())
void norm_tree(bool fence)
compute for each FunctionNode the norm of the function inside that node
Definition mraimpl.h:1588
void gaxpy_inplace(const T &alpha, const FunctionImpl< Q, NDIM > &other, const R &beta, bool fence)
Inplace general bilinear operation.
Definition funcimpl.h:1385
const Tensor< double > cell
the size of the root cell in each dimension, unchangeable
Definition funcimpl.h:1002
bool has_leaves() const
Definition mraimpl.h:300
bool verify_parents_and_children() const
check that parents and children are consistent
Definition mraimpl.h:121
void apply_source_driven(opT &op, const FunctionImpl< R, NDIM > &f, bool fence)
similar to apply, but for low rank coeffs
Definition funcimpl.h:5403
std::size_t halo_size() const
How many neighbor nodes are staged on this rank; zero if no halo.
Definition funcimpl.h:1044
void distribute(std::shared_ptr< WorldDCPmapInterface< Key< NDIM > > > newmap) const
Definition funcimpl.h:1219
int get_special_level() const
Definition funcimpl.h:993
void reconstruct_op(const keyT &key, const coeffT &s, const bool accumulate_NS=true)
Definition mraimpl.h:2163
tensorT gaxpy_ext_node(keyT key, Tensor< L > lc, T(*f)(const coordT &), T alpha, T beta) const
Definition funcimpl.h:7051
const coeffT parent_to_child(const coeffT &s, const keyT &parent, const keyT &child) const
Directly project parent coeffs to child coeffs.
Definition mraimpl.h:3329
WorldObject< FunctionImpl< T, NDIM > > woT
Base class world object type.
Definition funcimpl.h:972
void undo_redundant(const bool fence)
convert this from redundant to standard reconstructed form
Definition mraimpl.h:1579
GenTensor< T > coeffT
Type of tensor used to hold coeffs.
Definition funcimpl.h:981
const keyT & key0() const
Returns cdata.key0.
Definition mraimpl.h:406
double finalize_apply()
after apply we need to do some cleanup;
Definition mraimpl.h:1835
bool leaves_only
Definition funcimpl.h:5867
friend hashT hash_value(const FunctionImpl< T, NDIM > *pimpl)
Hash a pointer to FunctionImpl.
Definition funcimpl.h:7482
const dcT & get_coeffs() const
Definition mraimpl.h:355
FunctionImpl(World &world, const FunctionImpl< Q, NDIM > &other, const std::shared_ptr< WorldDCPmapInterface< Key< NDIM > > > &pmap, bool dozero)
Copy constructor.
Definition funcimpl.h:1168
compressT make_redundant_op(const keyT &key, const std::vector< Future< compressT > > &v)
similar to compress_op, but insert only the sum coefficients in the tree
Definition mraimpl.h:1780
T inner_ext_node(keyT key, tensorT c, const std::shared_ptr< FunctionFunctorInterface< T, NDIM > > f) const
Return the inner product with an external function on a specified function node.
Definition funcimpl.h:6850
double norm2sq_local() const
Returns the square of the local norm ... no comms.
Definition mraimpl.h:1887
const FunctionCommonData< T, NDIM > & get_cdata() const
Definition mraimpl.h:361
void sum_down(bool fence)
After 1d push operator must sum coeffs down the tree to restore correct scaling function coefficients...
Definition mraimpl.h:940
T inner_ext_recursive(keyT key, tensorT c, const std::shared_ptr< FunctionFunctorInterface< T, NDIM > > f, const bool leaf_refine, T old_inner=T(0)) const
Definition funcimpl.h:6867
bool noautorefine(const keyT &key, const tensorT &t) const
Always returns false (for when autorefine is not wanted)
Definition mraimpl.h:877
double truncate_tol(double tol, const keyT &key) const
Returns the truncation threshold according to truncate_method.
Definition mraimpl.h:667
void flo_unary_op_node_inplace(const opT &op, bool fence) const
Definition funcimpl.h:2354
bool autorefine_square_test(const keyT &key, const nodeT &t) const
Returns true if this block of coeffs needs autorefining.
Definition mraimpl.h:883
void erase(const Level &max_level)
truncate tree at a certain level
Definition mraimpl.h:757
void mulXX(const FunctionImpl< L, NDIM > *left, const FunctionImpl< R, NDIM > *right, double tol, bool fence)
Definition funcimpl.h:3402
std::pair< coeffT, std::pair< double, double > > compressT
s coefficients plus the (snorm_tree, dnorm_tree) pair propagated up by compress
Definition funcimpl.h:4822
void reconstruct(bool fence)
reconstruct this tree – respects fence
Definition mraimpl.h:1509
void multiply(const implT *f, const FunctionImpl< T, LDIM > *g, const int particle)
multiply f (a pair function of NDIM) with an orbital g (LDIM=NDIM/2)
Definition funcimpl.h:3819
coeffT assemble_coefficients(const keyT &key, const coeffT &coeff_ket, const coeffT &vpotential1, const coeffT &vpotential2, const tensorT &veri) const
given several coefficient tensors, assemble a result tensor
Definition mraimpl.h:1045
static void tnorm(const tensorT &t, double *lo, double *hi)
Computes norm of low/high-order polyn. coeffs for autorefinement test.
Definition mraimpl.h:3206
std::pair< bool, T > eval_local_only(const Vector< double, NDIM > &xin, Level maxlevel)
Evaluate function only if point is local returning (true,value); otherwise return (false,...
Definition mraimpl.h:3001
bool halo_probe(const keyT &key, coeffT &out) const
Look up a staged neighbor; on a hit copy its coefficients, which are empty for an interior node.
Definition funcimpl.h:1070
std::size_t max_depth() const
Returns the maximum depth of the tree ... collective ... global sum/broadcast.
Definition mraimpl.h:1915
std::size_t size() const
Returns the number of coefficients in the function ... collective global sum.
Definition mraimpl.h:1960
void reduce_rank(const double thresh, bool fence)
reduce the rank of the coefficients tensors
Definition mraimpl.h:1139
TreeState get_tree_state() const
Definition funcimpl.h:1470
void merge_trees(const T alpha, const FunctionImpl< Q, NDIM > &other, const R beta, const bool fence=true)
merge the trees of this and other, while multiplying them with the alpha or beta, resp
Definition funcimpl.h:1325
const Tensor< double > & get_cell() const
return the simulation cell
Definition funcimpl.h:1489
void halo_clear() const
Discard the neighbor halo, freeing the staged coefficients.
Definition funcimpl.h:1039
std::shared_ptr< FunctionFunctorInterface< T, NDIM > > get_functor()
Definition mraimpl.h:312
double do_apply_directed_screening(const opT *op, const keyT &key, const coeffT &coeff, const bool &do_kernel)
apply an operator on the coeffs c (at node key)
Definition funcimpl.h:5294
tensorT unfilter(const tensorT &s) const
Transform sums+differences at level n to sum coefficients at level n+1.
Definition mraimpl.h:1213
int get_initial_level() const
getter
Definition funcimpl.h:992
Tensor< T > eval_plot_cube(const coordT &plotlo, const coordT &plothi, const std::vector< long > &npt, const bool eval_refine=false) const
Definition mraimpl.h:3618
virtual ~FunctionImpl()
Definition funcimpl.h:1199
Vector< Translation, NDIM > tranT
Type of array holding translation.
Definition funcimpl.h:978
void change_tree_state(const TreeState finalstate, bool fence=true)
change the tree state of this function, might or might not respect fence!
Definition mraimpl.h:1441
Future< coeffT > truncate_reconstructed_spawn(const keyT &key, const double tol)
truncate using a tree in reconstructed form
Definition mraimpl.h:1634
GenTensor< Q > coeffs2values(const keyT &key, const GenTensor< Q > &coeff) const
Definition funcimpl.h:1925
FunctionImpl(const FunctionFactory< T, NDIM > &factory)
Initialize function impl from data in factory.
Definition funcimpl.h:1088
void map_and_mirror(const implT &f, const std::vector< long > &map, const std::vector< long > &mirror, bool fence)
map and mirror the translation index and the coefficients, result on this
Definition mraimpl.h:1108
Timer timer_lr_result
Definition funcimpl.h:1080
void gaxpy(T alpha, const FunctionImpl< L, NDIM > &left, T beta, const FunctionImpl< R, NDIM > &right, bool fence)
Invoked by result to perform result += alpha*left+beta*right in wavelet basis.
Definition funcimpl.h:2204
void truncate(double tol, bool fence)
Truncate according to the threshold with optional global fence.
Definition mraimpl.h:390
void do_mul(const keyT &key, const Tensor< L > &left, const std::pair< keyT, Tensor< R > > &arg)
Functor for the mul method.
Definition funcimpl.h:2129
void copy_remote_coeffs_from_pid(const ProcessID pid, const FunctionImpl< Q, NDIM > &other)
Definition funcimpl.h:1264
void project_out2(const FunctionImpl< T, LDIM+NDIM > *f, const FunctionImpl< T, LDIM > *g, const int dim)
project the low-dim function g on the hi-dim function f: this(x) = <f(x,y) | g(y)>
Definition funcimpl.h:7324
double do_apply_kernel2(const opT *op, const Tensor< R > &c, const do_op_args< OPDIM > &args, const TensorArgs &apply_targs)
same as do_apply_kernel, but use full rank tensors as input and low rank tensors as output
Definition funcimpl.h:4963
static Tensor< TENSOR_RESULT_TYPE(T, R)> dot_local(const std::vector< const FunctionImpl< T, NDIM > * > &left, const std::vector< const FunctionImpl< R, NDIM > * > &right, bool sym)
Definition funcimpl.h:6330
Tensor< Q > coeffs2values(const keyT &key, const Tensor< Q > &coeff) const
Definition funcimpl.h:2051
Tensor< Q > values2coeffs(const keyT &key, const Tensor< Q > &values) const
Definition funcimpl.h:2065
void multi_to_multi_op_values_doit(const keyT &key, const opT &op, const std::vector< implT * > &vin, std::vector< implT * > &vout)
Inplace operate on many functions (impl's) with an operator within a certain box.
Definition funcimpl.h:2975
bool is_reconstructed() const
Returns true if the function is compressed.
Definition mraimpl.h:258
void replicate(bool fence=true)
Definition funcimpl.h:1203
double norm_tree_op(const keyT &key, const std::vector< Future< double > > &v)
Definition mraimpl.h:1596
void reset_timer()
Definition mraimpl.h:378
void refine_to_common_level(const std::vector< FunctionImpl< T, NDIM > * > &v, const std::vector< tensorT > &c, const keyT key)
Refine multiple functions down to the same finest level.
Definition mraimpl.h:787
int get_k() const
Definition mraimpl.h:352
void dirac_convolution_op(const keyT &key, const nodeT &node, FunctionImpl< T, LDIM > *f) const
The operator.
Definition funcimpl.h:2270
FunctionImpl< T, NDIM > implT
Type of this class (implementation)
Definition funcimpl.h:975
void eval(const Vector< double, NDIM > &xin, const keyT &keyin, const typename Future< T >::remote_refT &ref)
Evaluate the function at a point in simulation coordinates.
Definition mraimpl.h:2957
bool truncate_op(const keyT &key, double tol, const std::vector< Future< bool > > &v)
Definition mraimpl.h:2731
void zero_norm_tree()
Definition mraimpl.h:1325
std::size_t max_local_depth() const
Returns the maximum local depth of the tree ... no communications.
Definition mraimpl.h:1901
tensorT project(const keyT &key) const
Definition mraimpl.h:2876
double thresh
Screening threshold.
Definition funcimpl.h:998
double check_symmetry_local() const
Returns some asymmetry measure ... no comms.
Definition mraimpl.h:773
Future< double > get_norm_tree_recursive(const keyT &key) const
Definition mraimpl.h:2897
bool is_redundant_after_merge() const
Returns true if the function is redundant_after_merge.
Definition mraimpl.h:270
void mulXXvec(const FunctionImpl< L, NDIM > *left, const std::vector< const FunctionImpl< R, NDIM > * > &vright, const std::vector< FunctionImpl< T, NDIM > * > &vresult, double tol, bool fence)
Definition funcimpl.h:3459
Key< NDIM > keyT
Type of key.
Definition funcimpl.h:979
friend hashT hash_value(const std::shared_ptr< FunctionImpl< T, NDIM > > impl)
Hash a shared_ptr to FunctionImpl.
Definition funcimpl.h:7492
std::vector< Vector< double, NDIM > > special_points
special points for further refinement (needed for composite functions or multiplication)
Definition funcimpl.h:1001
bool truncate_on_project
If true projection inserts at level n-1 not n.
Definition funcimpl.h:1006
AtomicInt small
Definition funcimpl.h:1084
static void do_dot_localX(const typename mapT::iterator lstart, const typename mapT::iterator lend, typename FunctionImpl< R, NDIM >::mapT *rmap_ptr, const bool sym, Tensor< TENSOR_RESULT_TYPE(T, R)> *result_ptr, Mutex *mutex)
Definition funcimpl.h:6227
bool is_on_demand() const
Definition mraimpl.h:285
double err_box(const keyT &key, const nodeT &node, const opT &func, int npt, const Tensor< double > &qx, const Tensor< double > &quad_phit, const Tensor< double > &quad_phiw) const
Returns the square of the error norm in the box labeled by key.
Definition funcimpl.h:5691
void accumulate_timer(const double time) const
Definition mraimpl.h:364
void trickle_down_op(const keyT &key, const coeffT &s)
sum all the contributions from all scales after applying an operator in mod-NS form
Definition mraimpl.h:1399
static void do_inner_localX(const typename mapT::iterator lstart, const typename mapT::iterator lend, typename FunctionImpl< R, NDIM >::mapT *rmap_ptr, const bool sym, Tensor< TENSOR_RESULT_TYPE(T, R) > *result_ptr, Mutex *mutex)
Definition funcimpl.h:6146
void mulXXveca(const keyT &key, const FunctionImpl< L, NDIM > *left, const Tensor< L > &lcin, const std::vector< const FunctionImpl< R, NDIM > * > vrightin, const std::vector< Tensor< R > > &vrcin, const std::vector< FunctionImpl< T, NDIM > * > vresultin, double tol)
Definition funcimpl.h:3062
void set_thresh(double value)
Definition mraimpl.h:343
Tensor< double > print_plane_local(const int xaxis, const int yaxis, const coordT &el2)
collect the data for a plot of the MRA structure locally on each node
Definition mraimpl.h:435
void sock_it_to_me_too(const keyT &key, const RemoteReference< FutureImpl< std::pair< keyT, coeffT > > > &ref) const
Definition mraimpl.h:2935
void broaden_op(const keyT &key, const std::vector< Future< bool > > &v)
Definition mraimpl.h:1313
void print_plane(const std::string filename, const int xaxis, const int yaxis, const coordT &el2)
Print a plane ("xy", "xz", or "yz") containing the point x to file.
Definition mraimpl.h:415
void print_tree(std::ostream &os=std::cout, Level maxlevel=10000) const
Definition mraimpl.h:2759
bool has_summable_coefficients() const
Returns true if summing over the local nodes yields the function.
Definition mraimpl.h:295
void project_refine_op(const keyT &key, bool do_refine, const std::vector< Vector< double, NDIM > > &specialpts)
Definition mraimpl.h:2542
void scale_oop(const Q q, const FunctionImpl< F, NDIM > &f, bool fence)
Out-of-place scale by a constant.
Definition funcimpl.h:7459
T typeT
Definition funcimpl.h:974
std::size_t tree_size() const
Returns the size of the tree structure of the function ... collective global sum.
Definition mraimpl.h:1941
ConcurrentHashMap< keyT, mapvecT > mapT
Type of the map returned by make_key_vec_map.
Definition funcimpl.h:6074
void add_scalar_inplace(T t, bool fence)
Adds a constant to the function. Local operation, optional fence.
Definition mraimpl.h:2623
void forward_traverse(const coeff_opT &coeff_op, const apply_opT &apply_op, const keyT &key) const
traverse a non-existing tree
Definition funcimpl.h:3913
tensorT downsample(const keyT &key, const std::vector< Future< coeffT > > &v) const
downsample the sum coefficients of level n+1 to sum coeffs on level n
Definition mraimpl.h:1233
void abs_square_inplace(bool fence)
Definition mraimpl.h:3309
FunctionImpl(const FunctionImpl< Q, NDIM > &other, const std::shared_ptr< WorldDCPmapInterface< Key< NDIM > > > &pmap, bool dozero)
Copy constructor.
Definition funcimpl.h:1154
void refine(const opT &op, bool fence)
Definition funcimpl.h:4773
static mapT make_key_vec_map(const std::vector< const FunctionImpl< T, NDIM > * > &v)
Returns map of union of local keys to vector of indexes of functions containing that key.
Definition funcimpl.h:6095
void put_in_box(ProcessID from, long nl, long ni) const
Definition mraimpl.h:842
void unary_op_value_inplace(const opT &op, bool fence)
Definition funcimpl.h:3042
std::pair< const keyT, nodeT > datumT
Type of entry in container.
Definition funcimpl.h:983
Timer timer_accumulate
Definition funcimpl.h:1078
TensorArgs get_tensor_args() const
Definition mraimpl.h:334
void unaryXXa(const keyT &key, const FunctionImpl< Q, NDIM > *func, const opT &op)
Definition funcimpl.h:3377
void make_Vphi_only(const opT &leaf_op, FunctionImpl< T, NDIM > *ket, FunctionImpl< T, LDIM > *v1, FunctionImpl< T, LDIM > *v2, FunctionImpl< T, LDIM > *p1, FunctionImpl< T, LDIM > *p2, FunctionImpl< T, NDIM > *eri, const bool fence=true)
assemble the function V*phi using V and phi given from the functor
Definition funcimpl.h:4588
void average(const implT &rhs)
take the average of two functions, similar to: this=0.5*(this+rhs)
Definition mraimpl.h:1120
void recursive_apply(opT &apply_op, const FunctionImpl< T, LDIM > *fimpl, const FunctionImpl< T, LDIM > *gimpl, const bool fence)
traverse a non-existing tree, make its coeffs and apply an operator
Definition funcimpl.h:5444
void diff(const DerivativeBase< T, NDIM > *D, const implT *f, bool fence)
Definition mraimpl.h:973
void square_inplace(bool fence)
Pointwise squaring of function with optional global fence.
Definition mraimpl.h:3298
void remove_internal_coefficients(const bool fence)
Definition mraimpl.h:1558
void compute_snorm_and_dnorm(bool fence=true)
compute norm of s and d coefficients for all nodes
Definition mraimpl.h:1163
std::vector< unsigned char > serialize_remote_coeffs()
invoked by copy_remote_coeffs_from_pid to serialize local coeffs
Definition funcimpl.h:1276
long box_leaf[1000]
Definition funcimpl.h:3474
void standard(bool fence)
Changes non-standard compressed form to standard compressed form.
Definition mraimpl.h:1822
void multiop_values_doit(const keyT &key, const opT &op, const std::vector< implT * > &v)
Definition funcimpl.h:2933
bool is_nonstandard_with_leaves() const
Definition mraimpl.h:280
GenTensor< Q > values2NScoeffs(const keyT &key, const GenTensor< Q > &values) const
convert function values of the a child generation directly to NS coeffs
Definition funcimpl.h:2026
int truncate_mode
0=default=(|d|<thresh), 1=(|d|<thresh/2^n), 2=(|d|<thresh/4^n);
Definition funcimpl.h:1004
void multiop_values(const opT &op, const std::vector< implT * > &v)
Definition funcimpl.h:2950
GenTensor< Q > NScoeffs2values(const keyT &key, const GenTensor< Q > &coeff, const bool s_only) const
convert S or NS coeffs to values on a 2k grid of the children
Definition funcimpl.h:1941
static std::enable_if_t< std::is_floating_point_v< Real >, Real > conj(const Real x)
Definition funcimpl.h:6267
FunctionNode holds the coefficients, etc., at each node of the 2^NDIM-tree.
Definition funcimpl.h:136
FunctionNode< Q, NDIM > convert() const
Copy with possible type conversion of coefficients, copying all other state.
Definition funcimpl.h:204
GenTensor< T > coeffT
Definition funcimpl.h:138
bool has_coeff() const
Returns true if there are coefficients in this node.
Definition funcimpl.h:210
void recompute_snorm_and_dnorm(const FunctionCommonData< T, NDIM > &cdata)
Definition funcimpl.h:355
FunctionNode(const coeffT &coeff, bool has_children=false)
Constructor from given coefficients with optional children.
Definition funcimpl.h:166
FunctionNode()
Default constructor makes node without coeff or children.
Definition funcimpl.h:156
void serialize(Archive &ar)
Definition funcimpl.h:478
double _dnorm_tree
norm of the difference coefficients summed up the tree
Definition funcimpl.h:147
void consolidate_buffer(const TensorArgs &args)
Definition funcimpl.h:464
double get_dnorm() const
return the precomputed norm of the (virtual) d coefficients
Definition funcimpl.h:336
size_t size() const
Returns the number of coefficients in this node.
Definition funcimpl.h:252
void set_has_children_recursive(const typename FunctionNode< T, NDIM >::dcT &c, const Key< NDIM > &key)
Sets has_children attribute to true recurring up to ensure connected.
Definition funcimpl.h:269
FunctionNode< T, NDIM > & operator=(const FunctionNode< T, NDIM > &other)
Definition funcimpl.h:186
FunctionNode(const coeffT &coeff, double norm_tree, double dnorm_tree, double snorm, double dnorm, bool has_children)
Definition funcimpl.h:176
double snorm
norm of the s coefficients
Definition funcimpl.h:151
void clear_coeff()
Clears the coefficients (has_coeff() will subsequently return false)
Definition funcimpl.h:305
Tensor< T > tensorT
Definition funcimpl.h:139
coeffT buffer
The coefficients, if any.
Definition funcimpl.h:149
T trace_conj(const FunctionNode< T, NDIM > &rhs) const
Definition funcimpl.h:473
void scale(Q a)
Scale the coefficients of this node.
Definition funcimpl.h:311
bool is_leaf() const
Returns true if this does not have children.
Definition funcimpl.h:223
void set_has_children(bool flag)
Sets has_children attribute to value of flag.
Definition funcimpl.h:264
void accumulate(const coeffT &t, const typename FunctionNode< T, NDIM >::dcT &c, const Key< NDIM > &key, const TensorArgs &args)
Accumulate inplace and if necessary connect node to parent.
Definition funcimpl.h:436
double get_norm_tree() const
Gets the value of norm_tree.
Definition funcimpl.h:326
bool _has_children
True if there are children.
Definition funcimpl.h:148
void set_snorm(const double sn)
set the precomputed norm of the (virtual) s coefficients
Definition funcimpl.h:341
coeffT _coeffs
The coefficients, if any.
Definition funcimpl.h:145
void accumulate2(const tensorT &t, const typename FunctionNode< T, NDIM >::dcT &c, const Key< NDIM > &key)
Accumulate inplace and if necessary connect node to parent.
Definition funcimpl.h:403
void reduceRank(const double &eps)
reduces the rank of the coefficients (if applicable)
Definition funcimpl.h:259
WorldContainer< Key< NDIM >, FunctionNode< T, NDIM > > dcT
Definition funcimpl.h:154
void gaxpy_inplace(const T &alpha, const FunctionNode< Q, NDIM > &other, const R &beta)
General bi-linear operation — this = this*alpha + other*beta.
Definition funcimpl.h:385
double get_dnorm_tree() const
Gets the value of dnorm_tree.
Definition funcimpl.h:331
double _norm_tree
After norm_tree will contain norm of sum coefficients summed up tree.
Definition funcimpl.h:146
void set_is_leaf(bool flag)
Sets has_children attribute to value of !flag.
Definition funcimpl.h:290
void print_json(std::ostream &s) const
Definition funcimpl.h:487
double get_snorm() const
get the precomputed norm of the (virtual) s coefficients
Definition funcimpl.h:351
void set_dnorm_tree(double dnorm_tree)
Sets the value of dnorm_tree.
Definition funcimpl.h:321
const coeffT & coeff() const
Returns a const reference to the tensor containing the coeffs.
Definition funcimpl.h:247
FunctionNode(const coeffT &coeff, double norm_tree, bool has_children)
Definition funcimpl.h:171
bool has_children() const
Returns true if this node has children.
Definition funcimpl.h:217
void set_coeff(const coeffT &coeffs)
Takes a shallow copy of the coeff — same as this->coeff()=coeff.
Definition funcimpl.h:295
void set_dnorm(const double dn)
set the precomputed norm of the (virtual) d coefficients
Definition funcimpl.h:346
double dnorm
norm of the d coefficients, also defined if there are no d coefficients
Definition funcimpl.h:150
bool is_invalid() const
Returns true if this node is invalid (no coeffs and no children)
Definition funcimpl.h:229
FunctionNode(const FunctionNode< T, NDIM > &other)
Definition funcimpl.h:180
coeffT & coeff()
Returns a non-const reference to the tensor containing the coeffs.
Definition funcimpl.h:237
void set_norm_tree(double norm_tree)
Sets the value of norm_tree.
Definition funcimpl.h:316
Implements the functionality of futures.
Definition future.h:75
A future is a possibly yet unevaluated value.
Definition future.h:370
remote_refT remote_ref(World &world) const
Returns a structure used to pass references to another process.
Definition future.h:672
RemoteReference< FutureImpl< T > > remote_refT
Definition future.h:395
Definition lowranktensor.h:59
bool is_of_tensortype(const TensorType &tt) const
Definition gentensor.h:225
GenTensor convert(const TensorArgs &) const
Definition gentensor.h:198
long dim(const int i) const
return the number of entries in dimension i
Definition lowranktensor.h:391
Tensor< T > full_tensor_copy() const
Definition gentensor.h:206
long ndim() const
Definition lowranktensor.h:386
constexpr bool is_full_tensor() const
Definition gentensor.h:224
void add_SVD(const GenTensor< T > &rhs, const double &)
Definition gentensor.h:235
const Tensor< T > & get_tensor() const
Definition gentensor.h:203
void reduce_rank(const double &)
Definition gentensor.h:217
bool has_no_data() const
Definition gentensor.h:211
void normalize()
Definition gentensor.h:218
GenTensor< T > & emul(const GenTensor< T > &other)
Inplace multiply by corresponding elements of argument Tensor.
Definition lowranktensor.h:637
float_scalar_type normf() const
Definition lowranktensor.h:406
double svd_normf() const
Definition gentensor.h:213
SRConf< T > config() const
Definition gentensor.h:237
long rank() const
Definition gentensor.h:212
const Tensor< T > & full_tensor() const
Definition gentensor.h:200
long size() const
Definition lowranktensor.h:488
SVDTensor< T > & get_svdtensor()
Definition gentensor.h:228
TensorType tensor_type() const
Definition gentensor.h:221
bool has_data() const
Definition gentensor.h:210
Tensor< T > reconstruct_tensor() const
Definition gentensor.h:199
GenTensor & gaxpy(const T alpha, const GenTensor &other, const T beta)
Definition lowranktensor.h:586
bool is_assigned() const
Definition gentensor.h:209
IsSupported< TensorTypeData< Q >, GenTensor< T > & >::type scale(Q fac)
Inplace multiplication by scalar of supported type (legacy name)
Definition lowranktensor.h:426
constexpr bool is_svd_tensor() const
Definition gentensor.h:222
Definition worldhashmap.h:330
Iterates in lexical order thru all children of a key.
Definition key.h:548
Key is the index for a node of the 2^NDIM-tree.
Definition key.h:70
Key< NDIM+LDIM > merge_with(const Key< LDIM > &rhs) const
merge with other key (ie concatenate), use level of rhs, not of this
Definition key.h:487
Level level() const
Definition key.h:169
bool is_valid() const
Checks if a key is valid.
Definition key.h:124
hashT hash() const
Definition key.h:158
Key< NDIM-VDIM > extract_complement_key(const std::array< int, VDIM > &v) const
extract a new key with the Translations complementary to the ones indicated in the v array
Definition key.h:473
Key< VDIM > extract_key(const std::array< int, VDIM > &v) const
extract a new key with the Translations indicated in the v array
Definition key.h:465
Key parent(int generation=1) const
Returns the key of the parent.
Definition key.h:290
const Vector< Translation, NDIM > & translation() const
Definition key.h:174
void break_apart(Key< LDIM > &key1, Key< KDIM > &key2) const
break key into two low-dimensional keys
Definition key.h:424
A pmap that locates children on odd levels with their even level parents.
Definition funcimpl.h:105
LevelPmap(World &world)
Definition funcimpl.h:111
const int nproc
Definition funcimpl.h:107
LevelPmap()
Definition funcimpl.h:109
ProcessID owner(const keyT &key) const
Find the owner of a given key.
Definition funcimpl.h:114
Definition funcimpl.h:77
Mutex using pthread mutex operations.
Definition worldmutex.h:150
void unlock() const
Free a mutex owned by this thread.
Definition worldmutex.h:184
void lock() const
Acquire the mutex waiting if necessary.
Definition worldmutex.h:174
Range, vaguely a la Intel TBB, to encapsulate a random-access, STL-like start and end iterator with c...
Definition range.h:64
Simple structure used to manage references/pointers to remote instances.
Definition worldref.h:394
Definition SVDTensor.h:42
A simple process map.
Definition funcimpl.h:86
SimplePmap(World &world)
Definition funcimpl.h:92
const int nproc
Definition funcimpl.h:88
const ProcessID me
Definition funcimpl.h:89
ProcessID owner(const keyT &key) const
Maps key to processor.
Definition funcimpl.h:95
A slice defines a sub-range or patch of a dimension.
Definition slice.h:103
static TaskAttributes hipri()
Definition thread.h:457
Traits class to specify support of numeric types.
Definition type_data.h:56
A tensor is a multidimensional array.
Definition tensor.h:318
float_scalar_type normf() const
Returns the Frobenius norm of the tensor.
Definition tensor.h:1727
Tensor< T > & gaxpy(T alpha, const Tensor< T > &other, T beta)
Inplace generalized saxpy ... this = this*alpha + other*beta.
Definition tensor.h:1806
T sum() const
Returns the sum of all elements of the tensor.
Definition tensor.h:1663
Tensor< T > reshape(int ndimnew, const long *d)
Returns new view/tensor reshaping size/number of dimensions to conforming tensor.
Definition tensor.h:1385
T * ptr()
Returns a pointer to the internal data.
Definition tensor.h:1841
Tensor< T > mapdim(const std::vector< long > &map)
Returns new view/tensor permuting the dimensions.
Definition tensor.h:1625
IsSupported< TensorTypeData< Q >, Tensor< T > & >::type scale(Q x)
Inplace multiplication by scalar of supported type (legacy name)
Definition tensor.h:687
Tensor< T > & emul(const Tensor< T > &t)
Inplace multiply by corresponding elements of argument Tensor.
Definition tensor.h:1800
bool has_data() const
Definition tensor.h:1903
Tensor< T > fusedim(long i)
Returns new view/tensor fusing contiguous dimensions i and i+1.
Definition tensor.h:1588
Tensor< T > flat()
Returns new view/tensor rehshaping to flat (1-d) tensor.
Definition tensor.h:1556
Tensor< T > & conj()
Inplace complex conjugate.
Definition tensor.h:717
Definition function_common_data.h:169
void accumulate(const double time) const
accumulate timer
Definition function_common_data.h:183
A simple, fixed dimension vector.
Definition vector.h:64
Makes a distributed container with specified attributes.
Definition worlddc.h:1299
void process_pending()
Process pending messages.
Definition worlddc.h:1645
bool find(accessor &acc, const keyT &key)
Write access to LOCAL value by key. Returns true if found, false otherwise (always false for remote).
Definition worlddc.h:1466
bool probe(const keyT &key) const
Returns true if local data is immediately available (no communication)
Definition worlddc.h:1503
iterator begin()
Returns an iterator to the beginning of the local data (no communication)
Definition worlddc.h:1549
bool is_replicated() const
Definition worlddc.h:1399
ProcessID owner(const keyT &key) const
Returns processor that logically owns key (no communication)
Definition worlddc.h:1513
implT::const_iterator const_iterator
Definition worlddc.h:1307
void replicate(bool fence=true)
Definition worlddc.h:1424
void erase(const keyT &key)
Erases entry from container (non-blocking comm if remote)
Definition worlddc.h:1584
void replace(const pairT &datum)
Inserts/replaces key+value pair (non-blocking communication if key not local)
Definition worlddc.h:1453
iterator end()
Returns an iterator past the end of the local data (no communication)
Definition worlddc.h:1563
const std::shared_ptr< WorldDCPmapInterface< keyT > > & get_pmap() const
Returns shared pointer to the process mapping.
Definition worlddc.h:1621
bool insert(accessor &acc, const keyT &key)
Write access to LOCAL value by key. Returns true if inserted, false if already exists (throws if remo...
Definition worlddc.h:1480
bool is_distributed() const
Definition worlddc.h:1395
implT::iterator iterator
Definition worlddc.h:1306
std::size_t size() const
Returns the number of local entries (no communication)
Definition worlddc.h:1614
Future< REMFUTURE(MEMFUN_RETURNT(memfunT))> task(const keyT &key, memfunT memfun, const TaskAttributes &attr=TaskAttributes())
Adds task "resultT memfun()" in process owning item (non-blocking comm if remote)
Definition worlddc.h:1905
bool is_local(const keyT &key) const
Returns true if the key maps to the local processor (no communication)
Definition worlddc.h:1520
bool is_host_replicated() const
Definition worlddc.h:1403
Future< MEMFUN_RETURNT(memfunT)> send(const keyT &key, memfunT memfun)
Sends message "resultT memfun()" to item (non-blocking comm if remote)
Definition worlddc.h:1662
void replicate_on_hosts(bool fence=true)
replicates this WorldContainer on all hosts (one PID per host)
Definition worlddc.h:1430
implT::accessor accessor
Definition worlddc.h:1308
Interface to be provided by any process map.
Definition worlddc.h:125
void fence(bool debug=false)
Synchronizes all processes in communicator AND globally ensures no pending AM or tasks.
Definition worldgop.cc:177
Implements most parts of a globally addressable object (via unique ID).
Definition world_object.h:491
void process_pending()
To be called from derived constructor to process pending messages.
Definition world_object.h:787
ProcessID me
Rank of self.
Definition world_object.h:514
detail::task_result_type< memfnT >::futureT send(ProcessID dest, memfnT memfn) const
Definition world_object.h:858
detail::task_result_type< memfnT >::futureT task(ProcessID dest, memfnT memfn, const TaskAttributes &attr=TaskAttributes()) const
Sends task to derived class method returnT (this->*memfn)().
Definition world_object.h:1132
Future< bool > for_each(const rangeT &range, const opT &op)
Apply op(item) on all items in range.
Definition world_task_queue.h:572
void add(TaskInterface *t)
Add a new local task, taking ownership of the pointer.
Definition world_task_queue.h:466
Future< resultT > reduce(const rangeT &range, const opT &op)
Reduce op(item) for all items in range using op(sum,op(item)).
Definition world_task_queue.h:527
A parallel world class.
Definition world.h:134
static World * world_from_id(std::uint64_t id)
Convert a World ID to a World pointer.
Definition world.h:516
WorldTaskQueue & taskq
Task queue.
Definition world.h:215
std::vector< uniqueidT > get_object_ids() const
Returns a vector of all unique IDs in this World.
Definition world.h:492
ProcessID rank() const
Returns the process rank in this World (same as MPI_Comm_rank()).
Definition world.h:344
static std::vector< unsigned long > get_world_ids()
return a vector containing all world ids
Definition world.h:500
ProcessID size() const
Returns the number of processes in this World (same as MPI_Comm_size()).
Definition world.h:354
unsigned long id() const
Definition world.h:324
WorldGopInterface & gop
Global operations.
Definition world.h:216
std::optional< T * > ptr_from_id(uniqueidT id) const
Look up a local pointer from a world-wide unique ID.
Definition world.h:440
ProcessID random_proc()
Returns a random process number; that is, an integer in [0,world.size()).
Definition world.h:615
Wraps an archive around an STL vector for input.
Definition vector_archive.h:101
Wraps an archive around an STL vector for output.
Definition vector_archive.h:55
Wrapper for an opaque pointer for serialization purposes.
Definition archive.h:851
syntactic sugar for std::array<bool, N>
Definition array_of_bools.h:19
Class for unique global IDs.
Definition uniqueid.h:53
unsigned long get_obj_id() const
Access the object ID.
Definition uniqueid.h:97
unsigned long get_world_id() const
Access the World ID.
Definition uniqueid.h:90
static const double R
Definition csqrt.cc:46
double(* f1)(const coord_3d &)
Definition derivatives.cc:55
char * p(char *buf, const char *name, int k, int initial_level, double thresh, int order)
Definition derivatives.cc:72
static double lo
Definition dirac-hatom.cc:23
@ upper
Definition dirac-hatom.cc:15
Provides FunctionDefaults and utilities for coordinate transformation.
archive_array< unsigned char > wrap_opaque(const T *, unsigned int)
Factory function to wrap a pointer to contiguous data as an opaque (uchar) archive_array.
Definition archive.h:926
Tensor< typename Tensor< T >::scalar_type > arg(const Tensor< T > &t)
Return a new tensor holding the argument of each element of t (complex types only)
Definition tensor.h:2757
Tensor< TENSOR_RESULT_TYPE(T, Q) > & fast_transform(const Tensor< T > &t, const Tensor< Q > &c, Tensor< TENSOR_RESULT_TYPE(T, Q) > &result, Tensor< TENSOR_RESULT_TYPE(T, Q) > &workspace)
Restricted but heavily optimized form of transform()
Definition tensor.h:2460
const double beta
Definition gygi_soltion.cc:62
static const double v
Definition hatom_sf_dirac.cc:20
Provides IndexIterator.
Tensor< double > op(const Tensor< double > &x)
Definition kain.cc:508
Multidimension Key for MRA tree and associated iterators.
static double pow(const double *a, const double *b)
Definition lda.h:74
#define MADNESS_CHECK(condition)
Check a condition — even in a release build the condition is always evaluated so it can have side eff...
Definition madness_exception.h:182
#define MADNESS_EXCEPTION(msg, value)
Macro for throwing a MADNESS exception.
Definition madness_exception.h:119
#define MADNESS_ASSERT(condition)
Assert a condition that should be free of side-effects since in release builds this might be a no-op.
Definition madness_exception.h:134
#define MADNESS_CHECK_THROW(condition, msg)
Check a condition — even in a release build the condition is always evaluated so it can have side eff...
Definition madness_exception.h:207
Header to declare stuff which has not yet found a home.
constexpr double pi
Mathematical constant .
Definition constants.h:48
MemFuncWrapper< objT *, memfnT, typename result_of< memfnT >::type > wrap_mem_fn(objT &obj, memfnT memfn)
Create a member function wrapper (MemFuncWrapper) from an object and a member function pointer.
Definition mem_func_wrapper.h:251
void combine_hash(hashT &seed, hashT hash)
Internal use only.
Definition worldhash.h:249
Namespace for all elements and tools of MADNESS.
Definition DFConvergence.h:9
std::ostream & operator<<(std::ostream &os, const particle< PDIM > &p)
Definition lowrankfunction.h:401
static const char * filename
Definition legendre.cc:96
static const std::vector< Slice > ___
Entire dimension.
Definition slice.h:124
static double cpu_time()
Returns the cpu time in seconds relative to an arbitrary origin.
Definition timers.h:128
GenTensor< TENSOR_RESULT_TYPE(R, Q)> general_transform(const GenTensor< R > &t, const Tensor< Q > c[])
Definition gentensor.h:274
void finalize()
Call this once at the very end of your main program instead of MPI_Finalize().
Definition world.cc:246
void norm_tree(World &world, const std::vector< Function< T, NDIM > > &v, bool fence=true)
Makes the norm tree for all functions in a vector.
Definition vmra.h:1345
std::vector< Function< TENSOR_RESULT_TYPE(T, R), NDIM > > transform(World &world, const std::vector< Function< T, NDIM > > &v, const Tensor< R > &c, bool fence=true)
Transforms a vector of functions according to new[i] = sum[j] old[j]*c[j,i].
Definition vmra.h:758
TreeState
Definition funcdefaults.h:60
@ nonstandard_after_apply
s and d coeffs, state after operator application
Definition funcdefaults.h:65
@ redundant_after_merge
s coeffs everywhere, must be summed up to yield the result
Definition funcdefaults.h:67
@ reconstructed
s coeffs at the leaves only
Definition funcdefaults.h:61
@ nonstandard
s and d coeffs in internal nodes
Definition funcdefaults.h:63
@ redundant
s coeffs everywhere
Definition funcdefaults.h:66
static Tensor< double > weights[max_npt+1]
Definition legendre.cc:99
int64_t Translation
Definition key.h:58
Key< NDIM > displacement(const Key< NDIM > &source, const Key< NDIM > &target)
given a source and a target, return the displacement in translation
Definition key.h:533
static const Slice _(0,-1, 1)
std::shared_ptr< FunctionFunctorInterface< double, 3 > > func(new opT(g))
int Level
Definition key.h:59
std::enable_if< std::is_base_of< ProjectorBase, projT >::value, OuterProjector< projT, projQ > >::type outer(const projT &p0, const projQ &p1)
Definition projector.h:457
static constexpr double NORM_TREE_UNCOMPUTED
Definition funcimpl.h:127
int RandomValue< int >()
Random int.
Definition ran.cc:250
bool has_data(const PropertyResults &p)
Definition Results.h:413
static constexpr double MUL_SCREENING_SAFETY
Definition funcimpl.h:132
static double pop(std::vector< double > &v)
Definition SCF.cc:117
void print(const T &t, const Ts &... ts)
Print items to std::cout (items separated by spaces) and terminate with a new line.
Definition print.h:227
Tensor< T > fcube(const Key< NDIM > &, T(*f)(const Vector< double, NDIM > &), const Tensor< double > &)
Definition mraimpl.h:2217
TensorType
low rank representations of tensors (see gentensor.h)
Definition gentensor.h:120
@ TT_2D
Definition gentensor.h:120
@ TT_FULL
Definition gentensor.h:120
NDIM & f
Definition mra.h:2668
void error(const char *msg)
Definition world.cc:147
NDIM const Function< R, NDIM > & g
Definition mra.h:2668
std::size_t hashT
The hash value type.
Definition worldhash.h:146
void change_tensor_type(GenTensor< T > &, const TensorArgs &targs)
change representation to targ.tt
Definition gentensor.h:284
static const int kmax
Definition twoscale.cc:52
GenTensor< TENSOR_RESULT_TYPE(R, Q)> transform_dir(const GenTensor< R > &t, const Tensor< Q > &c, const int axis)
Definition lowranktensor.h:1106
Function< T, CCPairFunction< T, NDIM >::LDIM > inner(const CCPairFunction< T, NDIM > &c, const Function< T, CCPairFunction< T, NDIM >::LDIM > &f, const std::tuple< int, int, int > v1, const std::tuple< int, int, int > v2)
Definition ccpairfunction.h:993
void scale(World &world, std::vector< Function< T, NDIM > > &v, const std::vector< Q > &factors, bool fence=true)
Scales inplace a vector of functions by distinct values.
Definition vmra.h:874
std::string name(const FuncType &type, const int ex=-1)
Definition ccpairfunction.h:28
void mxmT(long dimi, long dimj, long dimk, T *MADNESS_RESTRICT c, const T *a, const T *b)
Matrix += Matrix * matrix transpose ... MKL interface version.
Definition mxm.h:191
bool same_displacement_shell(double a, double b)
Definition displacements.h:60
Function< T, NDIM > copy(const Function< T, NDIM > &f, const std::shared_ptr< WorldDCPmapInterface< Key< NDIM > > > &pmap, bool fence=true)
Create a new copy of the function with different distribution and optional fence.
Definition mra.h:2233
static const int MAXK
The maximum wavelet order presently supported.
Definition funcdefaults.h:55
Definition mraimpl.h:53
static long abs(long a)
Definition tensor.h:219
const double cc
Definition navstokes_cosines.cc:107
static const double b
Definition nonlinschro.cc:119
static const double d
Definition nonlinschro.cc:121
static const double a
Definition nonlinschro.cc:118
Defines simple templates for printing to std::cout "a la Python".
double Q(double a)
Definition relops.cc:20
static const double c
Definition relops.cc:10
static const double L
Definition rk.cc:46
static const double thresh
Definition rk.cc:45
Definition test_ar.cc:204
Definition test_dc.cc:47
Key parent() const
Definition test_tree.cc:68
hashT hash() const
Definition test_dc.cc:54
Definition test_ccpairfunction.cc:22
given a ket and the 1- and 2-electron potentials, construct the function V phi
Definition funcimpl.h:4248
implT * result
where to construct Vphi, no need to track parents
Definition funcimpl.h:4256
bool have_v2() const
Definition funcimpl.h:4265
ctL iav1
Definition funcimpl.h:4260
Vphi_op_NS(implT *result, const opT &leaf_op, const ctT &iaket, const ctL &iap1, const ctL &iap2, const ctL &iav1, const ctL &iav2, const implT *eri)
Definition funcimpl.h:4274
ctL iap1
Definition funcimpl.h:4259
bool have_v1() const
Definition funcimpl.h:4264
std::pair< bool, coeffT > continue_recursion(const std::vector< bool > child_is_leaf, const tensorT &coeffs, const keyT &key) const
loop over all children and either insert their sum coeffs or continue the recursion
Definition funcimpl.h:4340
opT leaf_op
deciding if a given FunctionNode will be a leaf node
Definition funcimpl.h:4257
std::pair< coeffT, double > make_sum_coeffs(const keyT &key) const
make the sum coeffs for key
Definition funcimpl.h:4433
CoeffTracker< T, NDIM > ctT
Definition funcimpl.h:4253
ctL iap2
the particles 1 and 2 (exclusive with ket)
Definition funcimpl.h:4259
bool have_ket() const
Definition funcimpl.h:4263
const implT * eri
2-particle potential, must be on-demand
Definition funcimpl.h:4261
CoeffTracker< T, LDIM > ctL
Definition funcimpl.h:4254
std::pair< bool, coeffT > operator()(const Key< NDIM > &key) const
make and insert the coefficients into result's tree
Definition funcimpl.h:4285
void serialize(const Archive &ar)
serialize this (needed for use in recursive_op)
Definition funcimpl.h:4514
Vphi_op_NS< opT, LDIM > this_type
Definition funcimpl.h:4252
ctT iaket
the ket of a pair function (exclusive with p1, p2)
Definition funcimpl.h:4258
double compute_error_from_inaccurate_refinement(const keyT &key, const tensorT &ceri) const
the error is computed from the d coefficients of the constituent functions
Definition funcimpl.h:4386
void accumulate_into_result(const Key< NDIM > &key, const coeffT &coeff) const
Definition funcimpl.h:4268
this_type make_child(const keyT &child) const
Definition funcimpl.h:4485
tensorT eri_coeffs(const keyT &key) const
Definition funcimpl.h:4366
ctL iav2
potentials for particles 1 and 2
Definition funcimpl.h:4260
bool have_eri() const
Definition funcimpl.h:4266
this_type forward_ctor(implT *result1, const opT &leaf_op, const ctT &iaket1, const ctL &iap11, const ctL &iap21, const ctL &iav11, const ctL &iav21, const implT *eri1)
Definition funcimpl.h:4507
Vphi_op_NS()
Definition funcimpl.h:4273
Future< this_type > activate() const
Definition funcimpl.h:4496
bool randomize() const
Definition funcimpl.h:4250
add two functions f and g: result=alpha * f + beta * g
Definition funcimpl.h:3758
bool randomize() const
Definition funcimpl.h:3763
Future< this_type > activate() const
retrieve the coefficients (parent coeffs might be remote)
Definition funcimpl.h:3793
add_op(const ctT &f, const ctT &g, const double alpha, const double beta)
Definition funcimpl.h:3771
ctT f
tracking coeffs of first and second addend
Definition funcimpl.h:3766
double alpha
prefactor for f, g
Definition funcimpl.h:3768
add_op this_type
Definition funcimpl.h:3761
CoeffTracker< T, NDIM > ctT
Definition funcimpl.h:3760
void serialize(const Archive &ar)
Definition funcimpl.h:3805
ctT g
Definition funcimpl.h:3766
std::pair< bool, coeffT > operator()(const keyT &key) const
if we are at the bottom of the trees, return the sum of the coeffs
Definition funcimpl.h:3775
double beta
Definition funcimpl.h:3768
this_type make_child(const keyT &child) const
Definition funcimpl.h:3788
this_type forward_ctor(const ctT &f1, const ctT &g1, const double alpha, const double beta)
taskq-compatible ctor
Definition funcimpl.h:3801
opT op
Definition funcimpl.h:3347
opT::resultT resultT
Definition funcimpl.h:3345
Tensor< resultT > operator()(const Key< NDIM > &key, const Tensor< Q > &t) const
Definition funcimpl.h:3354
coeff_value_adaptor(const FunctionImpl< Q, NDIM > *impl_func, const opT &op)
Definition funcimpl.h:3350
const FunctionImpl< Q, NDIM > * impl_func
Definition funcimpl.h:3346
void serialize(Archive &ar)
Definition funcimpl.h:3363
merge the coefficent boxes of this into result's tree
Definition funcimpl.h:2558
Range< typename dcT::const_iterator > rangeT
Definition funcimpl.h:2559
void serialize(const Archive &ar)
Definition funcimpl.h:2576
FunctionImpl< Q, NDIM > * result
Definition funcimpl.h:2560
do_accumulate_trees(FunctionImpl< Q, NDIM > &result, const T alpha)
Definition funcimpl.h:2563
T alpha
Definition funcimpl.h:2561
bool operator()(typename rangeT::iterator &it) const
return the norm of the difference of this node and its "mirror" node
Definition funcimpl.h:2567
"put" this on g
Definition funcimpl.h:2769
Range< typename dcT::const_iterator > rangeT
Definition funcimpl.h:2770
void serialize(const Archive &ar)
Definition funcimpl.h:2798
implT * g
Definition funcimpl.h:2772
do_average()
Definition funcimpl.h:2774
bool operator()(typename rangeT::iterator &it) const
iterator it points to this
Definition funcimpl.h:2778
do_average(implT &g)
Definition funcimpl.h:2775
change representation of nodes' coeffs to low rank, optional fence
Definition funcimpl.h:2802
Range< typename dcT::iterator > rangeT
Definition funcimpl.h:2803
void serialize(const Archive &ar)
Definition funcimpl.h:2826
TensorArgs targs
Definition funcimpl.h:2806
do_change_tensor_type(const TensorArgs &targs, implT &g)
Definition funcimpl.h:2812
bool operator()(typename rangeT::iterator &it) const
Definition funcimpl.h:2815
implT * f
Definition funcimpl.h:2807
check symmetry wrt particle exchange
Definition funcimpl.h:2475
Range< typename dcT::const_iterator > rangeT
Definition funcimpl.h:2476
double operator()(typename rangeT::iterator &it) const
return the norm of the difference of this node and its "mirror" node
Definition funcimpl.h:2482
do_check_symmetry_local()
Definition funcimpl.h:2478
void serialize(const Archive &ar)
Definition funcimpl.h:2545
double operator()(double a, double b) const
Definition funcimpl.h:2541
do_check_symmetry_local(const implT &f)
Definition funcimpl.h:2479
const implT * f
Definition funcimpl.h:2477
compute the norm of the wavelet coefficients
Definition funcimpl.h:4655
Range< typename dcT::iterator > rangeT
Definition funcimpl.h:4656
bool operator()(typename rangeT::iterator &it) const
Definition funcimpl.h:4662
do_compute_snorm_and_dnorm(const FunctionCommonData< T, NDIM > &cdata)
Definition funcimpl.h:4659
const FunctionCommonData< T, NDIM > & cdata
Definition funcimpl.h:4658
TensorArgs targs
Definition funcimpl.h:2833
bool operator()(typename rangeT::iterator &it) const
Definition funcimpl.h:2838
Range< typename dcT::iterator > rangeT
Definition funcimpl.h:2830
do_consolidate_buffer(const TensorArgs &targs)
Definition funcimpl.h:2837
void serialize(const Archive &ar)
Definition funcimpl.h:2842
double operator()(double val) const
Definition funcimpl.h:1594
double limit
Definition funcimpl.h:1589
do_convert_to_color(const double limit, const bool log)
Definition funcimpl.h:1593
bool log
Definition funcimpl.h:1590
static double lower()
Definition funcimpl.h:1591
compute the inner product of this range with other
Definition funcimpl.h:6012
do_dot_local(const FunctionImpl< R, NDIM > *other, const bool leaves_only)
Definition funcimpl.h:6017
bool leaves_only
Definition funcimpl.h:6014
typedef TENSOR_RESULT_TYPE(T, R) resultT
resultT operator()(resultT a, resultT b) const
Definition funcimpl.h:6045
const FunctionImpl< R, NDIM > * other
Definition funcimpl.h:6013
void serialize(const Archive &ar)
Definition funcimpl.h:6049
resultT operator()(typename dcT::const_iterator &it) const
Definition funcimpl.h:6019
functor for the gaxpy_inplace method
Definition funcimpl.h:1357
FunctionImpl< T, NDIM > * f
prefactor for current function impl
Definition funcimpl.h:1359
do_gaxpy_inplace(FunctionImpl< T, NDIM > *f, T alpha, R beta)
Definition funcimpl.h:1363
bool operator()(typename rangeT::iterator &it) const
Definition funcimpl.h:1364
R beta
prefactor for other function impl
Definition funcimpl.h:1361
void serialize(Archive &ar)
Definition funcimpl.h:1372
Range< typename FunctionImpl< Q, NDIM >::dcT::const_iterator > rangeT
Definition funcimpl.h:1358
T alpha
the current function impl
Definition funcimpl.h:1360
const bool do_leaves
start with leaf nodes instead of initial_level
Definition funcimpl.h:6942
T operator()(T a, T b) const
Definition funcimpl.h:6960
do_inner_ext_local_ffi(const std::shared_ptr< FunctionFunctorInterface< T, NDIM > > f, const implT *impl, const bool leaf_refine, const bool do_leaves)
Definition funcimpl.h:6944
void serialize(const Archive &ar)
Definition funcimpl.h:6964
const bool leaf_refine
Definition funcimpl.h:6941
const std::shared_ptr< FunctionFunctorInterface< T, NDIM > > fref
Definition funcimpl.h:6939
T operator()(typename dcT::const_iterator &it) const
Definition funcimpl.h:6948
const implT * impl
Definition funcimpl.h:6940
compute the inner product of this range with other
Definition funcimpl.h:5875
const FunctionImpl< T, NDIM > * bra
Definition funcimpl.h:5876
void serialize(const Archive &ar)
Definition funcimpl.h:5991
const FunctionImpl< R, NDIM > * ket
Definition funcimpl.h:5877
bool leaves_only
Definition funcimpl.h:5878
do_inner_local_on_demand(const FunctionImpl< T, NDIM > *bra, const FunctionImpl< R, NDIM > *ket, const bool leaves_only=true)
Definition funcimpl.h:5881
resultT operator()(resultT a, resultT b) const
Definition funcimpl.h:5987
resultT operator()(typename dcT::const_iterator &it) const
Definition funcimpl.h:5884
compute the inner product of this range with other
Definition funcimpl.h:5814
resultT operator()(resultT a, resultT b) const
Definition funcimpl.h:5847
bool leaves_only
Definition funcimpl.h:5816
void serialize(const Archive &ar)
Definition funcimpl.h:5851
do_inner_local(const FunctionImpl< R, NDIM > *other, const bool leaves_only)
Definition funcimpl.h:5819
const FunctionImpl< R, NDIM > * other
Definition funcimpl.h:5815
resultT operator()(typename dcT::const_iterator &it) const
Definition funcimpl.h:5821
typedef TENSOR_RESULT_TYPE(T, R) resultT
keep only the sum coefficients in each node
Definition funcimpl.h:2429
Range< typename dcT::iterator > rangeT
Definition funcimpl.h:2430
do_keep_sum_coeffs(implT *impl)
constructor need impl for cdata
Definition funcimpl.h:2434
implT * impl
Definition funcimpl.h:2431
void serialize(const Archive &ar)
Definition funcimpl.h:2443
bool operator()(typename rangeT::iterator &it) const
Definition funcimpl.h:2436
mirror dimensions of this, write result on f
Definition funcimpl.h:2703
bool operator()(typename rangeT::iterator &it) const
Definition funcimpl.h:2713
implT * f
Definition funcimpl.h:2707
std::vector< long > mirror
Definition funcimpl.h:2706
void serialize(const Archive &ar)
Definition funcimpl.h:2760
Range< typename dcT::iterator > rangeT
Definition funcimpl.h:2704
std::vector< long > map
Definition funcimpl.h:2706
do_map_and_mirror(const std::vector< long > map, const std::vector< long > mirror, implT &f)
Definition funcimpl.h:2710
map this on f
Definition funcimpl.h:2623
do_mapdim(const std::vector< long > map, implT &f)
Definition funcimpl.h:2630
void serialize(const Archive &ar)
Definition funcimpl.h:2646
Range< typename dcT::iterator > rangeT
Definition funcimpl.h:2624
bool operator()(typename rangeT::iterator &it) const
Definition funcimpl.h:2632
std::vector< long > map
Definition funcimpl.h:2626
do_mapdim()
Definition funcimpl.h:2629
implT * f
Definition funcimpl.h:2627
merge the coefficient boxes of this into other's tree
Definition funcimpl.h:2587
bool operator()(typename rangeT::iterator &it) const
return the norm of the difference of this node and its "mirror" node
Definition funcimpl.h:2597
Range< typename dcT::const_iterator > rangeT
Definition funcimpl.h:2588
FunctionImpl< Q, NDIM > * other
Definition funcimpl.h:2589
do_merge_trees(const T alpha, const R beta, FunctionImpl< Q, NDIM > &other)
Definition funcimpl.h:2593
T alpha
Definition funcimpl.h:2590
do_merge_trees()
Definition funcimpl.h:2592
R beta
Definition funcimpl.h:2591
void serialize(const Archive &ar)
Definition funcimpl.h:2616
mirror dimensions of this, write result on f
Definition funcimpl.h:2653
bool operator()(typename rangeT::iterator &it) const
Definition funcimpl.h:2662
implT * f
Definition funcimpl.h:2657
Range< typename dcT::iterator > rangeT
Definition funcimpl.h:2654
do_mirror()
Definition funcimpl.h:2659
do_mirror(const std::vector< long > mirror, implT &f)
Definition funcimpl.h:2660
void serialize(const Archive &ar)
Definition funcimpl.h:2696
std::vector< long > mirror
Definition funcimpl.h:2656
Definition funcimpl.h:5775
double operator()(typename dcT::const_iterator &it) const
Definition funcimpl.h:5782
bool leaves_only
skip the internal nodes, cf. has_coefficients_on_leaves_only()
Definition funcimpl.h:5777
do_norm2sq_local(bool leaves_only)
Definition funcimpl.h:5780
void serialize(const Archive &ar)
Definition funcimpl.h:5798
double operator()(double a, double b) const
Definition funcimpl.h:5794
laziness
Definition funcimpl.h:4915
void serialize(Archive &ar)
Definition funcimpl.h:4924
Key< OPDIM > d
Definition funcimpl.h:4916
Key< OPDIM > key
Definition funcimpl.h:4916
keyT dest
Definition funcimpl.h:4917
double fac
Definition funcimpl.h:4918
do_op_args(const Key< OPDIM > &key, const Key< OPDIM > &d, const keyT &dest, double tol, double fac, double cnorm)
Definition funcimpl.h:4921
double cnorm
Definition funcimpl.h:4918
double tol
Definition funcimpl.h:4918
reduce the rank of the nodes, optional fence
Definition funcimpl.h:2449
do_reduce_rank(const TensorArgs &targs)
Definition funcimpl.h:2457
TensorArgs args
Definition funcimpl.h:2453
bool operator()(typename rangeT::iterator &it) const
Definition funcimpl.h:2463
Range< typename dcT::iterator > rangeT
Definition funcimpl.h:2450
do_reduce_rank(const double &thresh)
Definition funcimpl.h:2458
void serialize(const Archive &ar)
Definition funcimpl.h:2469
Changes non-standard compressed form to standard compressed form.
Definition funcimpl.h:4879
bool operator()(typename rangeT::iterator &it) const
Definition funcimpl.h:4890
do_standard(implT *impl)
Definition funcimpl.h:4887
Range< typename dcT::iterator > rangeT
Definition funcimpl.h:4880
void serialize(const Archive &ar)
Definition funcimpl.h:4907
implT * impl
Definition funcimpl.h:4883
given an NS tree resulting from a convolution, truncate leafs if appropriate
Definition funcimpl.h:2370
void serialize(const Archive &ar)
Definition funcimpl.h:2390
const implT * f
Definition funcimpl.h:2372
bool operator()(typename rangeT::iterator &it) const
Definition funcimpl.h:2376
Range< typename dcT::iterator > rangeT
Definition funcimpl.h:2371
do_truncate_NS_leafs(const implT *f)
Definition funcimpl.h:2374
Range< typename dcT::iterator > rangeT
Definition funcimpl.h:2849
bool operator()(typename rangeT::iterator &it) const
Definition funcimpl.h:2853
implT * impl
Definition funcimpl.h:2850
void serialize(const Archive &ar)
Definition funcimpl.h:2871
do_unary_op_value_inplace(implT *impl, const opT &op)
Definition funcimpl.h:2852
Hartree product of two LDIM functions to yield a NDIM = 2*LDIM function.
Definition funcimpl.h:3841
this_type forward_ctor(implT *result1, const ctL &p11, const ctL &p22, const leaf_opT &leaf_op)
Definition funcimpl.h:3897
bool randomize() const
Definition funcimpl.h:3842
void serialize(const Archive &ar)
Definition funcimpl.h:3901
hartree_op(implT *result, const ctL &p11, const ctL &p22, const leaf_opT &leaf_op)
Definition funcimpl.h:3853
CoeffTracker< T, LDIM > ctL
Definition funcimpl.h:3845
ctL p2
tracking coeffs of the two lo-dim functions
Definition funcimpl.h:3848
leaf_opT leaf_op
determine if a given node will be a leaf node
Definition funcimpl.h:3849
hartree_op()
Definition funcimpl.h:3852
implT * result
where to construct the pair function
Definition funcimpl.h:3847
hartree_op< LDIM, leaf_opT > this_type
Definition funcimpl.h:3844
std::pair< bool, coeffT > operator()(const Key< NDIM > &key) const
Definition funcimpl.h:3858
ctL p1
Definition funcimpl.h:3848
this_type make_child(const keyT &child) const
Definition funcimpl.h:3881
Future< this_type > activate() const
Definition funcimpl.h:3890
perform this multiplication: h(1,2) = f(1,2) * g(1)
Definition funcimpl.h:3649
multiply_op()
Definition funcimpl.h:3661
ctL g
Definition funcimpl.h:3658
Future< this_type > activate() const
Definition funcimpl.h:3740
CoeffTracker< T, LDIM > ctL
Definition funcimpl.h:3653
implT * h
the result function h(1,2) = f(1,2) * g(1)
Definition funcimpl.h:3656
CoeffTracker< T, NDIM > ctT
Definition funcimpl.h:3652
std::pair< bool, coeffT > operator()(const Key< NDIM > &key) const
apply this on a FunctionNode of f and g of Key key
Definition funcimpl.h:3688
this_type forward_ctor(implT *h1, const ctT &f1, const ctL &g1, const int particle)
Definition funcimpl.h:3747
static bool randomize()
Definition funcimpl.h:3651
int particle
if g is g(1) or g(2)
Definition funcimpl.h:3659
ctT f
Definition funcimpl.h:3657
multiply_op< LDIM > this_type
Definition funcimpl.h:3654
multiply_op(implT *h1, const ctT &f1, const ctL &g1, const int particle1)
Definition funcimpl.h:3663
bool screen(const coeffT &fcoeff, const coeffT &gcoeff, const keyT &key) const
return true if this will be a leaf node
Definition funcimpl.h:3669
this_type make_child(const keyT &child) const
Definition funcimpl.h:3730
void serialize(const Archive &ar)
Definition funcimpl.h:3751
coeffT val_lhs
Definition funcimpl.h:4128
double lo
Definition funcimpl.h:4131
double lo1
Definition funcimpl.h:4131
long oversampling
Definition funcimpl.h:4129
double error
Definition funcimpl.h:4130
tensorT operator()(const Key< NDIM > key, const tensorT &coeff_rhs)
multiply values of rhs and lhs, result on rhs, rhs and lhs are of the same dimensions
Definition funcimpl.h:4146
coeffT coeff_lhs
Definition funcimpl.h:4128
void serialize(const Archive &ar)
Definition funcimpl.h:4234
double lo2
Definition funcimpl.h:4131
double hi1
Definition funcimpl.h:4131
pointwise_multiplier(const Key< NDIM > key, const coeffT &clhs)
Definition funcimpl.h:4134
coeffT operator()(const Key< NDIM > key, const tensorT &coeff_rhs, const int particle)
multiply values of rhs and lhs, result on rhs, rhs and lhs are of differnet dimensions
Definition funcimpl.h:4191
double hi2
Definition funcimpl.h:4131
double hi
Definition funcimpl.h:4131
project the low-dim function g on the hi-dim function f: result(x) = <f(x,y) | g(y)>
Definition funcimpl.h:7204
project_out_op(const implT *fimpl, implL1 *result, const ctL &iag, const int dim)
Definition funcimpl.h:7219
ctL iag
the low dim function g
Definition funcimpl.h:7214
FunctionImpl< T, NDIM-LDIM > implL1
Definition funcimpl.h:7209
Future< this_type > activate() const
retrieve the coefficients (parent coeffs might be remote)
Definition funcimpl.h:7298
std::pair< bool, coeffT > argT
Definition funcimpl.h:7210
const implT * fimpl
the hi dim function f
Definition funcimpl.h:7212
this_type forward_ctor(const implT *fimpl1, implL1 *result1, const ctL &iag1, const int dim1)
taskq-compatible ctor
Definition funcimpl.h:7305
this_type make_child(const keyT &child) const
Definition funcimpl.h:7289
project_out_op< LDIM > this_type
Definition funcimpl.h:7207
implL1 * result
the low dim result function
Definition funcimpl.h:7213
Future< argT > operator()(const Key< NDIM > &key) const
do the actual contraction
Definition funcimpl.h:7226
void serialize(const Archive &ar)
Definition funcimpl.h:7309
project_out_op(const project_out_op &other)
Definition funcimpl.h:7221
int dim
0: project 0..LDIM-1, 1: project LDIM..NDIM-1
Definition funcimpl.h:7215
bool randomize() const
Definition funcimpl.h:7205
CoeffTracker< T, LDIM > ctL
Definition funcimpl.h:7208
recursive part of recursive_apply
Definition funcimpl.h:5602
ctT iaf
Definition funcimpl.h:5610
recursive_apply_op2< opT > this_type
Definition funcimpl.h:5605
Future< this_type > activate() const
retrieve the coefficients (parent coeffs might be remote)
Definition funcimpl.h:5665
const opT * apply_op
need this for randomization
Definition funcimpl.h:5611
bool randomize() const
Definition funcimpl.h:5603
recursive_apply_op2(const recursive_apply_op2 &other)
Definition funcimpl.h:5618
void serialize(const Archive &ar)
Definition funcimpl.h:5681
argT finalize(const double kernel_norm, const keyT &key, const coeffT &coeff, const implT *r) const
sole purpose is to wait for the kernel norm, wrap it and send it back to caller
Definition funcimpl.h:5651
this_type make_child(const keyT &child) const
Definition funcimpl.h:5660
recursive_apply_op2(implT *result, const ctT &iaf, const opT *apply_op)
Definition funcimpl.h:5615
std::pair< bool, coeffT > argT
Definition funcimpl.h:5607
implT * result
Definition funcimpl.h:5609
CoeffTracker< T, NDIM > ctT
Definition funcimpl.h:5606
argT operator()(const Key< NDIM > &key) const
send off the application of the operator
Definition funcimpl.h:5627
this_type forward_ctor(implT *result1, const ctT &iaf1, const opT *apply_op1)
taskq-compatible ctor
Definition funcimpl.h:5677
recursive part of recursive_apply
Definition funcimpl.h:5471
std::pair< bool, coeffT > operator()(const Key< NDIM > &key) const
make the NS-coefficients and send off the application of the operator
Definition funcimpl.h:5496
this_type forward_ctor(implT *r, const CoeffTracker< T, LDIM > &f1, const CoeffTracker< T, LDIM > &g1, const opT *apply_op1)
Definition funcimpl.h:5561
opT * apply_op
Definition funcimpl.h:5479
recursive_apply_op(const recursive_apply_op &other)
Definition funcimpl.h:5489
recursive_apply_op< opT, LDIM > this_type
Definition funcimpl.h:5474
Future< this_type > activate() const
Definition funcimpl.h:5554
bool randomize() const
Definition funcimpl.h:5472
implT * result
Definition funcimpl.h:5476
CoeffTracker< T, LDIM > iaf
Definition funcimpl.h:5477
void serialize(const Archive &ar)
Definition funcimpl.h:5566
std::pair< bool, coeffT > finalize(const double kernel_norm, const keyT &key, const coeffT &coeff) const
sole purpose is to wait for the kernel norm, wrap it and send it back to caller
Definition funcimpl.h:5536
recursive_apply_op(implT *result, const CoeffTracker< T, LDIM > &iaf, const CoeffTracker< T, LDIM > &iag, const opT *apply_op)
Definition funcimpl.h:5483
this_type make_child(const keyT &child) const
Definition funcimpl.h:5545
CoeffTracker< T, LDIM > iag
Definition funcimpl.h:5478
remove all coefficients of internal nodes
Definition funcimpl.h:2395
Range< typename dcT::iterator > rangeT
Definition funcimpl.h:2396
remove_internal_coeffs()=default
constructor need impl for cdata
bool operator()(typename rangeT::iterator &it) const
Definition funcimpl.h:2401
void serialize(const Archive &ar)
Definition funcimpl.h:2407
remove all coefficients of leaf nodes
Definition funcimpl.h:2412
bool operator()(typename rangeT::iterator &it) const
Definition funcimpl.h:2418
remove_leaf_coeffs()=default
constructor need impl for cdata
void serialize(const Archive &ar)
Definition funcimpl.h:2423
Range< typename dcT::iterator > rangeT
Definition funcimpl.h:2413
Definition funcimpl.h:4727
void serialize(Archive &ar)
Definition funcimpl.h:4731
bool operator()(const implT *f, const keyT &key, const nodeT &t) const
Definition funcimpl.h:4728
shallow-copy, pared-down version of FunctionNode, for special purpose only
Definition funcimpl.h:772
coeffT & coeff()
Definition funcimpl.h:786
GenTensor< T > coeffT
Definition funcimpl.h:773
bool is_leaf() const
Definition funcimpl.h:788
void serialize(Archive &ar)
Definition funcimpl.h:790
ShallowNode(const ShallowNode< T, NDIM > &node)
Definition funcimpl.h:781
ShallowNode(const FunctionNode< T, NDIM > &node)
Definition funcimpl.h:778
bool has_children() const
Definition funcimpl.h:787
ShallowNode()
Definition funcimpl.h:777
bool _has_children
Definition funcimpl.h:775
double dnorm
Definition funcimpl.h:776
const coeffT & coeff() const
Definition funcimpl.h:785
coeffT _coeffs
Definition funcimpl.h:774
Definition displacements.h:390
TensorArgs holds the arguments for creating a LowRankTensor.
Definition gentensor.h:134
double thresh
Definition gentensor.h:135
TensorType tt
Definition gentensor.h:136
const uniqueidT & id() const
Returns the globally unique object ID.
Definition world_object.h:424
inserts/accumulates coefficients into impl's tree
Definition funcimpl.h:739
FunctionImpl< T, NDIM > * impl
Definition funcimpl.h:743
FunctionNode< T, NDIM > nodeT
Definition funcimpl.h:741
accumulate_op(const accumulate_op &other)=default
void operator()(const Key< NDIM > &key, const coeffT &coeff, const bool &is_leaf) const
Definition funcimpl.h:747
void serialize(Archive &ar)
Definition funcimpl.h:751
GenTensor< T > coeffT
Definition funcimpl.h:740
accumulate_op(FunctionImpl< T, NDIM > *f)
Definition funcimpl.h:745
static void load(const Archive &ar, FunctionImpl< T, NDIM > *&ptr)
Definition funcimpl.h:7531
static void load(const Archive &ar, const FunctionImpl< T, NDIM > *&ptr)
Definition funcimpl.h:7500
static void load(const Archive &ar, std::shared_ptr< FunctionImpl< T, NDIM > > &ptr)
Definition funcimpl.h:7582
static void load(const Archive &ar, std::shared_ptr< const FunctionImpl< T, NDIM > > &ptr)
Definition funcimpl.h:7566
Default load of an object via serialize(ar, t).
Definition archive.h:667
static void load(const A &ar, const U &t)
Load an object.
Definition archive.h:679
static void store(const Archive &ar, FunctionImpl< T, NDIM > *const &ptr)
Definition funcimpl.h:7556
static void store(const Archive &ar, const FunctionImpl< T, NDIM > *const &ptr)
Definition funcimpl.h:7522
static void store(const Archive &ar, const std::shared_ptr< FunctionImpl< T, NDIM > > &ptr)
Definition funcimpl.h:7591
static void store(const Archive &ar, const std::shared_ptr< const FunctionImpl< T, NDIM > > &ptr)
Definition funcimpl.h:7575
Default store of an object via serialize(ar, t).
Definition archive.h:612
static std::enable_if_t< is_output_archive_v< A > &&!std::is_function< U >::value &&(has_member_serialize_v< U, A >||has_nonmember_serialize_v< U, A >||has_freestanding_serialize_v< U, A >||has_freestanding_default_serialize_v< U, A >), void > store(const A &ar, const U &t)
Definition archive.h:622
Definition funcimpl.h:633
void serialize(Archive &ar)
Definition funcimpl.h:697
const opT * op
Definition funcimpl.h:640
hartree_convolute_leaf_op(const implT *f, const implL *g, const opT *op)
Definition funcimpl.h:644
bool operator()(const Key< NDIM > &key) const
no pre-determination
Definition funcimpl.h:648
bool operator()(const Key< NDIM > &key, const Tensor< T > &fcoeff, const Tensor< T > &gcoeff) const
post-determination: true if f is a leaf and the result is well-represented
Definition funcimpl.h:661
const implL * g
Definition funcimpl.h:639
const FunctionImpl< T, NDIM > * f
Definition funcimpl.h:638
FunctionImpl< T, LDIM > implL
Definition funcimpl.h:636
bool do_error_leaf_op() const
Definition funcimpl.h:641
FunctionImpl< T, NDIM > implT
Definition funcimpl.h:635
bool operator()(const Key< NDIM > &key, const GenTensor< T > &coeff) const
no post-determination
Definition funcimpl.h:651
returns true if the result of a hartree_product is a leaf node (compute norm & error)
Definition funcimpl.h:523
bool do_error_leaf_op() const
Definition funcimpl.h:528
const FunctionImpl< T, NDIM > * f
Definition funcimpl.h:526
hartree_leaf_op(const implT *f, const long &k)
Definition funcimpl.h:531
long k
Definition funcimpl.h:527
void serialize(Archive &ar)
Definition funcimpl.h:579
bool operator()(const Key< NDIM > &key, const GenTensor< T > &coeff) const
no post-determination
Definition funcimpl.h:537
bool operator()(const Key< NDIM > &key, const Tensor< T > &fcoeff, const Tensor< T > &gcoeff) const
post-determination: true if f is a leaf and the result is well-represented
Definition funcimpl.h:547
bool operator()(const Key< NDIM > &key) const
no pre-determination
Definition funcimpl.h:534
FunctionImpl< T, NDIM > implT
Definition funcimpl.h:525
insert/replaces the coefficients into the function
Definition funcimpl.h:715
insert_op()
Definition funcimpl.h:722
implT * impl
Definition funcimpl.h:721
void operator()(const keyT &key, const coeffT &coeff, const bool &is_leaf) const
Definition funcimpl.h:725
FunctionNode< T, NDIM > nodeT
Definition funcimpl.h:719
Key< NDIM > keyT
Definition funcimpl.h:717
insert_op(const insert_op &other)
Definition funcimpl.h:724
FunctionImpl< T, NDIM > implT
Definition funcimpl.h:716
GenTensor< T > coeffT
Definition funcimpl.h:718
insert_op(implT *f)
Definition funcimpl.h:723
void serialize(Archive &ar)
Definition funcimpl.h:729
Definition mra.h:112
Definition funcimpl.h:703
bool operator()(const Key< NDIM > &key, const GenTensor< T > &fcoeff, const GenTensor< T > &gcoeff) const
Definition funcimpl.h:705
void serialize(Archive &ar)
Definition funcimpl.h:709
void operator()(const Key< NDIM > &key, const GenTensor< T > &coeff, const bool &is_leaf) const
Definition funcimpl.h:704
Definition funcimpl.h:587
bool operator()(const Key< NDIM > &key, const double &cnorm) const
post-determination: return true if operator and coefficient norms are small
Definition funcimpl.h:608
void serialize(Archive &ar)
Definition funcimpl.h:623
const implT * f
the source or result function, needed for truncate_tol
Definition funcimpl.h:591
op_leaf_op(const opT *op, const implT *f)
Definition funcimpl.h:595
FunctionImpl< T, NDIM > implT
Definition funcimpl.h:588
const opT * op
the convolution operator
Definition funcimpl.h:590
bool do_error_leaf_op() const
Definition funcimpl.h:592
bool operator()(const Key< NDIM > &key) const
pre-determination: we can't know if this will be a leaf node before we got the final coeffs
Definition funcimpl.h:598
bool operator()(const Key< NDIM > &key, const GenTensor< T > &coeff) const
post-determination: return true if operator and coefficient norms are small
Definition funcimpl.h:601
Definition lowrankfunction.h:336
Definition funcimpl.h:759
void serialize(Archive &ar)
Definition funcimpl.h:766
bool operator()(const Key< NDIM > &key, const T &t, const R &r) const
Definition funcimpl.h:765
bool operator()(const Key< NDIM > &key, const T &t) const
Definition funcimpl.h:762
int np
Definition tdse1d.cc:165
static const double s0
Definition tdse4.cc:83
Defines and implements most of Tensor.
#define ITERATOR(t, exp)
Definition tensor_macros.h:249
#define IND
Definition tensor_macros.h:204
#define TERNARY_OPTIMIZED_ITERATOR(X, x, Y, y, Z, z, exp)
Definition tensor_macros.h:719
AtomicInt sum
Definition test_atomicint.cc:46
double norm(const T i1)
Definition test_cloud.cc:85
int task(int i)
Definition test_runtime.cpp:4
void e()
Definition test_sig.cc:75
static double g1(const Vector< double, D > &r)
Definition test_state_archive_hdf5.cpp:34
static const double alpha
Definition testcosine.cc:10
const double offset
Definition testfuns.cc:143
constexpr std::size_t NDIM
Definition testgconv.cc:54
double h(const coord_1d &r)
Definition testgconv.cc:175
std::size_t axis
Definition testpdiff.cc:59
double source(const coordT &r)
Definition testperiodic.cc:48
#define TENSOR_RESULT_TYPE(L, R)
This macro simplifies access to TensorResultType.
Definition type_data.h:205
#define PROFILE_MEMBER_FUNC(classname)
Definition worldprofile.h:210
#define PROFILE_BLOCK(name)
Definition worldprofile.h:208
int ProcessID
Used to clearly identify process number/rank.
Definition worldtypes.h:43