~/bend-docscommunity

par.bend source

par.bend on the hub · documented module

import Baseimport ./geom.bend as Gimport ./protein.bend as Pimport ./topology.bend as Timport ./force.bend as F# --- Parallel kernels v0: 4-wide unrolled fork-join. ---# Bend has no forward references, so divide-and-conquer halving (which must# match a computed split) is inexpressible: it needs a helper cycle. Instead# each step runs 4 independent head computations as parallel calls and folds# one tail, which is a single self-recursion. Call an entry with `!` to run# every nested parallel call on the GPU (falls back to CPU without one;# verify codegen with `bend <file> -o <file>.c`). Remainders under 4 fall# back to the sequential kernels.def count_par4(xs: List<&2, P.Atom>, +qp: G.Vec3, +c2: F32) -> Nat:  match xs:    case h1 <> h2 <> h3 <> h4 <> t:      a b = T.head_hit_nat(h1, qp, c2) T.head_hit_nat(h2, qp, c2)      c d = T.head_hit_nat(h3, qp, c2) T.head_hit_nat(h4, qp, c2)      Nat.add(Nat.add(a, b), Nat.add(c, Nat.add(d, count_par4(t, qp, c2))))    case rest:      T.count_below(rest, qp, c2)# Plain-cutoff wrapper, mirroring T.count_within.def count_within_par(xs: List<&2, P.Atom>, +qp: G.Vec3, +cutoff: F32) -> Nat:  count_par4(xs, qp, (cutoff * cutoff : F32))def lj_row_par4(+pi: G.Vec3, +eps: F32, +sig2: F32, +c2: F32, ys: List<&2, P.Atom>) -> F32:  match ys:    case h1 <> h2 <> h3 <> h4 <> t:      a b = F.lj_head(pi, eps, sig2, c2, h1) F.lj_head(pi, eps, sig2, c2, h2)      c d = F.lj_head(pi, eps, sig2, c2, h3) F.lj_head(pi, eps, sig2, c2, h4)      ((a + b : F32) + ((c + d : F32) + lj_row_par4(pi, eps, sig2, c2, t) : F32) : F32)    case rest:      F.lj_row(pi, eps, sig2, c2, rest)