import Base import ./geom.bend as G import ./protein.bend as P import ./topology.bend as T import ./force.bend as F # --- Parallel kernels v0: 4-wide unrolled fork-join. --- # Bend has no forward references, so divide-and-conquer halving (which must # match a computed split) is inexpressible: it needs a helper cycle. Instead # each step runs 4 independent head computations as parallel calls and folds # one tail, which is a single self-recursion. Call an entry with `!` to run # every nested parallel call on the GPU (falls back to CPU without one; # verify codegen with `bend -o .c`). Remainders under 4 fall # back to the sequential kernels. def count_par4(xs: List<&2, P.Atom>, +qp: G.Vec3, +c2: F32) -> Nat: match xs: case h1 <> h2 <> h3 <> h4 <> t: a b = T.head_hit_nat(h1, qp, c2) T.head_hit_nat(h2, qp, c2) c d = T.head_hit_nat(h3, qp, c2) T.head_hit_nat(h4, qp, c2) Nat.add(Nat.add(a, b), Nat.add(c, Nat.add(d, count_par4(t, qp, c2)))) case rest: T.count_below(rest, qp, c2) # Plain-cutoff wrapper, mirroring T.count_within. def count_within_par(xs: List<&2, P.Atom>, +qp: G.Vec3, +cutoff: F32) -> Nat: count_par4(xs, qp, (cutoff * cutoff : F32)) def lj_row_par4(+pi: G.Vec3, +eps: F32, +sig2: F32, +c2: F32, ys: List<&2, P.Atom>) -> F32: match ys: case h1 <> h2 <> h3 <> h4 <> t: a b = F.lj_head(pi, eps, sig2, c2, h1) F.lj_head(pi, eps, sig2, c2, h2) c d = F.lj_head(pi, eps, sig2, c2, h3) F.lj_head(pi, eps, sig2, c2, h4) ((a + b : F32) + ((c + d : F32) + lj_row_par4(pi, eps, sig2, c2, t) : F32) : F32) case rest: F.lj_row(pi, eps, sig2, c2, rest)