# MPEG-1 Layer III synthesis: antialias, IMDCT-36, sign flip, DCT-II, polyphase. import Base import ./mp3_tab.bend as Tab import ./mp3_enc.bend as Enc # add two binary32 values def f.add(+aa: F32, +bb: F32) -> F32: (aa + bb : F32) # subtract two binary32 values def f.sub(+aa: F32, +bb: F32) -> F32: (aa - bb : F32) # multiply two binary32 values def f.mul(+aa: F32, +bb: F32) -> F32: (aa * bb : F32) # sample at an index, or 0 past the end def f.at(+xs: List<&2, F32>, +ii: U32) -> F32: Enc.f.nth(xs, ii) # replace one sample def f.set(xs: List<&2, F32>, +ii: U32, +vv: F32) -> List<&2, F32>: Enc.f.put(xs, ii, vv) # unsigned add def u.add(+aa: U32, +bb: U32) -> U32: (aa + bb : U32) # unsigned subtract def u.sub(+aa: U32, +bb: U32) -> U32: (aa - bb : U32) # unsigned multiply def u.mul(+aa: U32, +bb: U32) -> U32: (aa * bb : U32) # low bit def u.and(+kk: U32) -> U32: (kk .&. 1 : U32) # 0.5 def c.half() -> F32: Enc.f32.bits(1056964608) # 0.93969262 def c.c939() -> F32: Enc.f32.bits(1064341426) # 0.76604444 def c.c766() -> F32: Enc.f32.bits(1061428093) # 0.17364818 def c.c173() -> F32: Enc.f32.bits(1043452116) # 0.86602540 def c.c866() -> F32: Enc.f32.bits(1063105495) # 0.98480775 def c.c984() -> F32: Enc.f32.bits(1065098332) # 0.34202014 def c.c342() -> F32: Enc.f32.bits(1051663684) # 0.64278761 def c.c642() -> F32: Enc.f32.bits(1059360187) # 0.70710677 def c.c707() -> F32: Enc.f32.bits(1060439283) # 0.198912367 def c.c198() -> F32: Enc.f32.bits(1045147567) # 0.382683432 def c.c382() -> F32: Enc.f32.bits(1053028117) # 0.50979561 def c.c509() -> F32: Enc.f32.bits(1057128951) # 0.54119611 def c.c541() -> F32: Enc.f32.bits(1057655764) # 0.60134488 def c.c601() -> F32: Enc.f32.bits(1058664893) # 0.89997619 def c.c899() -> F32: Enc.f32.bits(1063675095) # 1.30656302 def c.c130() -> F32: Enc.f32.bits(1067924853) # 2.56291556 def c.c256() -> F32: Enc.f32.bits(1076102863) # 1/32768, the polyphase scale def c.inv() -> F32: Enc.f32.bits(939524096) # drop a prefix def ls.drop(hop: Nat, xs: List<&2, F32>) -> List<&2, F32>: match hop xs: case 0n ys: ys case _ Nil{}: [] case 1n+p _hd <> tl: ls.drop(p, tl) # take a prefix, reversed into acc def ls.take(hop: Nat, xs: List<&2, F32>, acc: List<&2, F32>) -> List<&2, F32>: match hop xs: case 0n _: List.reverse(&2, F32, acc) case _ Nil{}: List.reverse(&2, F32, acc) case 1n+p +hd <> tl: ls.take(p, tl, hd <> acc) # a slice of length hop starting at an index def ls.slice(+xs: List<&2, F32>, +at: U32, hop: Nat) -> List<&2, F32>: ls.take(hop, ls.drop(U32.to_nat(at), xs), []) # write a short list over a destination starting at an index def ls.poke(hop: Nat, dst: List<&2, F32>, src: List<&2, F32>, +at: U32) -> List<&2, F32>: match hop src: case 0n _: dst case _ Nil{}: dst case 1n+p +hd <> tl: ls.poke(p, f.set(dst, at, hd), tl, u.add(at, 1)) # nine samples, index 0 first def cs.pack( +v0: F32, +v1: F32, +v2: F32, +v3: F32, +v4: F32, +v5: F32, +v6: F32, +v7: F32, +v8: F32 ) -> List<&2, F32>: v0 <> (v1 <> (v2 <> (v3 <> (v4 <> (v5 <> (v6 <> (v7 <> (v8 <> [])))))))) # cosine half of one IMDCT-36 band def cs.co(+gg: List<&2, F32>) -> List<&2, F32>: cs.pack(F32.neg(f.at(gg, 0)), f.add(f.at(gg, 1), f.at(gg, 2)), F32.neg(f.add(f.at(gg, 3), f.at(gg, 4))), f.add(f.at(gg, 5), f.at(gg, 6)), F32.neg(f.add(f.at(gg, 7), f.at(gg, 8))), f.add(f.at(gg, 9), f.at(gg, 10)), F32.neg(f.add(f.at(gg, 11), f.at(gg, 12))), f.add(f.at(gg, 13), f.at(gg, 14)), F32.neg(f.add(f.at(gg, 15), f.at(gg, 16)))) # sine half of one IMDCT-36 band def cs.si(+gg: List<&2, F32>) -> List<&2, F32>: cs.pack(f.at(gg, 17), f.sub(f.at(gg, 16), f.at(gg, 15)), f.sub(f.at(gg, 13), f.at(gg, 14)), f.sub(f.at(gg, 12), f.at(gg, 11)), f.sub(f.at(gg, 9), f.at(gg, 10)), f.sub(f.at(gg, 8), f.at(gg, 7)), f.sub(f.at(gg, 5), f.at(gg, 6)), f.sub(f.at(gg, 4), f.at(gg, 3)), f.sub(f.at(gg, 1), f.at(gg, 2))) # even-part sum used by the 9-point DCT def d9.t0(+y0: F32, +y6: F32) -> F32: f.add(y0, f.mul(y6, c.half())) # even-part difference def d9.s0(+y0: F32, +y6: F32) -> F32: f.sub(y0, y6) # twiddle on bins 4 and 2 def d9.t4(+y4: F32, +y2: F32) -> F32: f.mul(f.add(y4, y2), c.c939()) # twiddle on bins 8 and 2 def d9.t2(+y8: F32, +y2: F32) -> F32: f.mul(f.add(y8, y2), c.c766()) # twiddle on bins 4 and 8 def d9.s6(+y4: F32, +y8: F32) -> F32: f.mul(f.sub(y4, y8), c.c173()) # even residual def d9.s4(+y4: F32, +y8: F32, +y2: F32) -> F32: f.sub(f.add(y4, y8), y2) # scaled residual def d9.s2(+s0: F32, +s4: F32) -> F32: f.sub(s0, f.mul(s4, c.half())) # centre bin of the even DCT def d9.y4(+s4: F32, +s0: F32) -> F32: f.add(s4, s0) # high even output def d9.s8c(+t0: F32, +t2: F32, +s6: F32) -> F32: f.add(f.sub(t0, t2), s6) # low even output def d9.s0c(+t0: F32, +t4: F32, +t2: F32) -> F32: f.add(f.sub(t0, t4), t2) # mid even output def d9.s4c(+t0: F32, +t4: F32, +s6: F32) -> F32: f.sub(f.add(t0, t4), s6) # odd centre def d9.s3(+y3: F32) -> F32: f.mul(y3, c.c866()) # odd low twiddle def d9.tb0(+y5: F32, +y1: F32) -> F32: f.mul(f.add(y5, y1), c.c984()) # odd high twiddle def d9.tb4(+y5: F32, +y7: F32) -> F32: f.mul(f.sub(y5, y7), c.c342()) # odd mid twiddle def d9.tb2(+y1: F32, +y7: F32) -> F32: f.mul(f.add(y1, y7), c.c642()) # odd residual def d9.s1(+y1: F32, +y5: F32, +y7: F32) -> F32: f.mul(f.sub(f.sub(y1, y5), y7), c.c866()) # odd combination def d9.s5c(+t0: F32, +s3: F32, +t2: F32) -> F32: f.sub(f.sub(t0, s3), t2) # odd combination def d9.s7c(+t4: F32, +s3: F32, +t0: F32) -> F32: f.sub(f.sub(t4, s3), t0) # odd combination def d9.s3c(+t4: F32, +s3: F32, +t2: F32) -> F32: f.sub(f.add(t4, s3), t2) # even and odd 9-point DCT results, ready to interleave type D9 is Data: D9{ s2b: F32, y4o: F32, s8c: F32, s0c: F32, s4c: F32, s1b: F32, s5c: F32, s7c: F32, s3c: F32 } # assemble the nine DCT outputs from the even and odd parts def d9.from( +t0: F32, +s0: F32, +t4: F32, +t2: F32, +s6: F32, +s4: F32, +s3: F32, +u0: F32, +u4: F32, +u2: F32, +s1: F32 ) -> D9: D9{d9.s2(s0, s4), d9.y4(s4, s0), d9.s8c(t0, t2, s6), d9.s0c(t0, t4, t2), d9.s4c(t0, t4, s6), s1, d9.s5c(u0, s3, u2), d9.s7c(u4, s3, u0), d9.s3c(u4, s3, u2)} # 9-point DCT of one half def d9.build( +y0: F32, +y1: F32, +y2: F32, +y3: F32, +y4: F32, +y5: F32, +y6: F32, +y7: F32, +y8: F32 ) -> D9: d9.from(d9.t0(y0, y6), d9.s0(y0, y6), d9.t4(y4, y2), d9.t2(y8, y2), d9.s6(y4, y8), d9.s4(y4, y8, y2), d9.s3(y3), d9.tb0(y5, y1), d9.tb4(y5, y7), d9.tb2(y1, y7), d9.s1(y1, y5, y7)) # nine DCT outputs in bin order def d9.list(dd: D9) -> List<&2, F32>: match dd: case D9{+s2b, +y4o, +s8c, +s0c, +s4c, +s1b, +s5c, +s7c, +s3c}: cs.pack(f.sub(s4c, s7c), f.add(s2b, s1b), f.sub(s0c, s3c), f.add(s8c, s5c), y4o, f.sub(s8c, s5c), f.add(s0c, s3c), f.sub(s2b, s1b), f.add(s4c, s7c)) # 9-point DCT def d9.run(+ys: List<&2, F32>) -> List<&2, F32>: d9.list(d9.build(f.at(ys, 0), f.at(ys, 1), f.at(ys, 2), f.at(ys, 3), f.at(ys, 4), f.at(ys, 5), f.at(ys, 6), f.at(ys, 7), f.at(ys, 8))) # negate the odd bins of a sine DCT def d9.neg(+ys: List<&2, F32>) -> List<&2, F32>: f.set(f.set(f.set(f.set(ys, 1, F32.neg(f.at(ys, 1))), 3, F32.neg(f.at(ys, 3))), 5, F32.neg(f.at(ys, 5))), 7, F32.neg(f.at(ys, 7))) # one windowed band: 18 time samples and 9 new overlap samples type Wn is Data: Wn{out: List<&2, F32>, nov: List<&2, F32>} # window product and the overlap it saves type Ws is Data: Ws{sm: F32, nv: F32} # aliasing pair for one window index def wn.pair(+co: List<&2, F32>, +si: List<&2, F32>, +tw: List<&2, F32>, +ii: U32) -> Ws: Ws{f.add(f.mul(f.at(co, ii), f.at(tw, u.add(ii, 9))), f.mul(f.at(si, ii), f.at(tw, ii))), f.sub(f.mul(f.at(co, ii), f.at(tw, ii)), f.mul(f.at(si, ii), f.at(tw, u.add(ii, 9))))} # write one window index into the band def wn.put(st: Wn, +ov: List<&2, F32>, +win: List<&2, F32>, +ii: U32, ws: Ws) -> Wn: match st ws: case Wn{out, nov} Ws{+sm, +nv}: Wn{f.set(f.set(out, ii, f.sub(f.mul(f.at(ov, ii), f.at(win, ii)), f.mul(sm, f.at(win, u.add(ii, 9))))), u.sub(17, ii), f.add(f.mul(f.at(ov, ii), f.at(win, u.add(ii, 9))), f.mul(sm, f.at(win, ii)))), f.set(nov, ii, nv)} # one of the nine window steps def wn.one( st: Wn, +co: List<&2, F32>, +si: List<&2, F32>, +ov: List<&2, F32>, +tw: List<&2, F32>, +win: List<&2, F32>, +ii: U32 ) -> Wn: wn.put(st, ov, win, ii, wn.pair(co, si, tw, ii)) # window all nine indices def wn.go( hop: Nat, st: Wn, +co: List<&2, F32>, +si: List<&2, F32>, +ov: List<&2, F32>, +tw: List<&2, F32>, +win: List<&2, F32>, +ii: U32 ) -> Wn: match hop: case 0n: st case 1n+p: wn.go(p, wn.one(st, co, si, ov, tw, win, ii), co, si, ov, tw, win, u.add(ii, 1)) # IMDCT-36 of one 18-point band def im.band(+gg: List<&2, F32>, +ov: List<&2, F32>, +tw: List<&2, F32>, +win: List<&2, F32>) -> Wn: wn.go(9n, Wn{List.replicate(F32, 18n, 0.0), List.replicate(F32, 9n, 0.0)}, d9.run(cs.co(gg)), d9.neg(d9.run(cs.si(gg))), ov, tw, win, 0) # one F32 read from the IMDCT/DCT spectrum Array type Dx is Type: Dx{acc: Array, +v: F32} def dx.of(got: Array & F32) -> Dx: (acc, v) = got Dx{acc, v} def dx.at(acc: Array, +ii: U32) -> Dx: dx.of(Array.get(F32, acc, ii)) # spectrum Array beside the samples just read type Il is Type: Il{acc: Array, xs: List<&2, F32>} # pull one sample; hop is remaining after this read def ix.pull(hop: Nat, got: Dx, +ii: U32, ys: List<&2, F32>) -> Il: match hop got: case 0n Dx{acc, +cur}: Il{acc, List.reverse(&2, F32, cur <> ys)} case 1n+p Dx{acc, +cur}: ix.pull(p, dx.at(acc, u.add(ii, 1)), u.add(ii, 1), cur <> ys) # hop+1 samples starting at `at`. The spectrum Array is handed back. def ix.take(acc: Array, +at: U32, hop: Nat) -> Il: ix.pull(hop, dx.at(acc, at), at, []) # write a short list into the spectrum Array def ix.poke(hop: Nat, acc: Array, src: List<&2, F32>, +at: U32) -> Array: match hop src: case 0n _: acc case _ Nil{}: acc case 1n+p +hd <> tl: ix.poke(p, Array.set(F32, acc, at, hd), tl, u.add(at, 1)) # 576 lines in the spectrum Array, and 288 overlap samples type Im is Type: Im{buf: Array, ov: List<&2, F32>} # write one band back into the spectrum Array and the overlap def im.store(buf: Array, +ov: List<&2, F32>, ww: Wn, +jj: U32, +base: U32) -> Im: match ww: case Wn{out, nov}: Im{ix.poke(18n, buf, out, u.add(base, u.mul(jj, 18))), ls.poke(9n, ov, nov, u.mul(jj, 9))} # band samples in hand; window them and store def im.got( got: Il, +ov: List<&2, F32>, +tw: List<&2, F32>, +win: List<&2, F32>, +jj: U32, +base: U32 ) -> Im: match got: case Il{buf, gg}: im.store(buf, ov, im.band(gg, ls.slice(ov, u.mul(jj, 9), 9n), tw, win), jj, base) # IMDCT-36 of band jj. Each of the 18 samples is an Array.get. def im.one(st: Im, +tw: List<&2, F32>, +win: List<&2, F32>, +jj: U32, +base: U32) -> Im: match st: case Im{buf, +ov}: im.got(ix.take(buf, u.add(base, u.mul(jj, 18)), 17n), ov, tw, win, jj, base) # IMDCT-36 over 32 bands. base is the channel offset in the spectrum Array. def im.go(hop: Nat, st: Im, +tw: List<&2, F32>, +win: List<&2, F32>, +jj: U32, +base: U32) -> Im: match hop: case 0n: st case 1n+p: im.go(p, im.one(st, tw, win, jj, base), tw, win, u.add(jj, 1), base) # high index of one aliasing pair def aa.ixh(+base: U32, +bb: U32, +ii: U32) -> U32: u.add(base, u.add(u.mul(u.add(bb, 1), 18), ii)) # low index of one aliasing pair def aa.ixl(+base: U32, +bb: U32, +ii: U32) -> U32: u.add(base, u.add(u.mul(bb, 18), u.sub(17, ii))) # spectrum Array beside the two aliasing samples type Ia is Type: Ia{acc: Array, +up: F32, +dp: F32} # low sample, then both indices are in hand def aa.lo(got: Dx, +up: F32) -> Ia: match got: case Dx{acc, dp}: Ia{acc, up, dp} # high sample in hand; read the low sample next def aa.rd1(got: Dx, +lo: U32) -> Ia: match got: case Dx{acc, up}: aa.lo(dx.at(acc, lo), up) # read both samples of one aliasing pair. Each read is an Array.get. def aa.rd(acc: Array, +hi: U32, +lo: U32) -> Ia: aa.rd1(dx.at(acc, hi), lo) # butterfly of one aliasing pair. c0 and c1 are the two cosine rows. def aa.wr( acc: Array, +up: F32, +dp: F32, +c0: F32, +c1: F32, +hi: U32, +lo: U32 ) -> Array: Array.set(F32, Array.set(F32, acc, hi, f.sub(f.mul(c0, up), f.mul(c1, dp))), lo, f.add(f.mul(c1, up), f.mul(c0, dp))) # write the pair once both samples have been read def aa.use(got: Ia, +c0: F32, +c1: F32, +hi: U32, +lo: U32) -> Array: match got: case Ia{acc, up, dp}: aa.wr(acc, up, dp, c0, c1, hi, lo) # one butterfly at index ii of band bb def aa.one(acc: Array, +aa: List<&2, F32>, +bb: U32, +ii: U32, +base: U32) -> Array: aa.use(aa.rd(acc, aa.ixh(base, bb, ii), aa.ixl(base, bb, ii)), f.at(aa, ii), f.at(aa, u.add(ii, 8)), aa.ixh(base, bb, ii), aa.ixl(base, bb, ii)) # eight butterflies between band bb and the next def aa.i(hop: Nat, acc: Array, +aa: List<&2, F32>, +bb: U32, +ii: U32, +base: U32) -> Array: match hop: case 0n: acc case 1n+p: aa.i(p, aa.one(acc, aa, bb, ii, base), aa, bb, u.add(ii, 1), base) # antialias from low bands toward high bands def aa.b(hop: Nat, acc: Array, +aa: List<&2, F32>, +bb: U32, +base: U32) -> Array: match hop: case 0n: acc case 1n+p: aa.b(p, aa.i(8n, acc, aa, bb, 0, base), aa, u.add(bb, 1), base) # 31 long-block aliasing pairs. base is the channel offset in the spectrum Array. def aa.run(acc: Array, +aa: List<&2, F32>, +base: U32) -> Array: aa.b(31n, acc, aa, 0, base) # negate the sample just read def sg.neg(got: Dx, +at: U32) -> Array: match got: case Dx{acc, v}: Array.set(F32, acc, at, F32.neg(v)) # negate one odd sample of an odd band def sg.i(hop: Nat, acc: Array, +base: U32, +ii: U32) -> Array: match hop: case 0n: acc case 1n+p: sg.i(p, sg.neg(dx.at(acc, u.add(base, ii)), u.add(base, ii)), base, u.add(ii, 2)) # odd samples of 16 odd bands, starting at sample 18 def sg.b(hop: Nat, acc: Array, +base: U32) -> Array: match hop: case 0n: acc case 1n+p: sg.b(p, sg.i(9n, acc, base, 1), u.add(base, 36)) # frequency inversion after the IMDCT. base is the channel offset. def im.sign(st: Im, +base: U32) -> Im: match st: case Im{buf, ov}: Im{sg.b(16n, buf, u.add(base, 18)), ov} # four DCT-II column results type Q4 is Data: Q4{a0: F32, a1: F32, a2: F32, a3: F32} # column sample i of the 32-band vector def dc.x0(acc: Array, +kk: U32, +ii: U32) -> Dx: dx.at(acc, u.add(kk, u.mul(ii, 18))) # mirrored column def dc.x1(acc: Array, +kk: U32, +ii: U32) -> Dx: dx.at(acc, u.add(kk, u.mul(u.sub(15, ii), 18))) # upper column def dc.x2(acc: Array, +kk: U32, +ii: U32) -> Dx: dx.at(acc, u.add(kk, u.mul(u.add(16, ii), 18))) # upper mirror def dc.x3(acc: Array, +kk: U32, +ii: U32) -> Dx: dx.at(acc, u.add(kk, u.mul(u.sub(31, ii), 18))) # finish the column butterfly def dc.q2(+t0: F32, +t1: F32, +t2: F32, +t3: F32, +s2: F32) -> Q4: Q4{f.add(t0, t1), f.mul(f.sub(t0, t1), s2), f.add(t3, t2), f.mul(f.sub(t3, t2), s2)} # one DCT-II column from four samples and three secants def dc.q(+x0: F32, +x1: F32, +x2: F32, +x3: F32, +s0: F32, +s1: F32, +s2: F32) -> Q4: dc.q2(f.add(x0, x3), f.add(x1, x2), f.mul(f.sub(x1, x2), s0), f.mul(f.sub(x0, x3), s1), s2) # spectrum Array beside one finished column type Dq is Type: Dq{acc: Array, q: Q4} # fourth sample, then the column butterfly def dc.rd3(got: Dx, +x0: F32, +x1: F32, +x2: F32, +s0: F32, +s1: F32, +s2: F32) -> Dq: match got: case Dx{acc, x3}: Dq{acc, dc.q(x0, x1, x2, x3, s0, s1, s2)} # third sample and the three secants def dc.rd2(got: Dx, +sec: List<&2, F32>, +kk: U32, +ii: U32, +x0: F32, +x1: F32) -> Dq: match got: case Dx{acc, x2}: dc.rd3(dc.x3(acc, kk, ii), x0, x1, x2, f.at(sec, u.mul(ii, 3)), f.at(sec, u.add(u.mul(ii, 3), 1)), f.at(sec, u.add(u.mul(ii, 3), 2))) # second sample def dc.rd1(got: Dx, +sec: List<&2, F32>, +kk: U32, +ii: U32, +x0: F32) -> Dq: match got: case Dx{acc, x1}: dc.rd2(dc.x2(acc, kk, ii), sec, kk, ii, x0, x1) # read column ii. Each of the four samples is an Array.get. def dc.cell(got: Dx, +sec: List<&2, F32>, +kk: U32, +ii: U32) -> Dq: match got: case Dx{acc, x0}: dc.rd1(dc.x1(acc, kk, ii), sec, kk, ii, x0) # store one column into the 4 by 8 working Array def dc.store(xs: Array, qq: Q4, +ii: U32) -> Array: match qq: case Q4{a0, a1, a2, a3}: Array.set(F32, Array.set(F32, Array.set(F32, Array.set(F32, xs, ii, a0), u.add(ii, 8), a1), u.add(ii, 16), a2), u.add(ii, 24), a3) # spectrum Array beside the 4 by 8 working rows type Dr is Type: Dr{acc: Array, xs: Array} # write one gathered column and keep both Arrays def dc.stored(got: Dq, xs: Array, +ii: U32) -> Dr: match got: case Dq{acc, q}: Dr{acc, dc.store(xs, q, ii)} # gather all eight columns before any write-back def dc.cols(hop: Nat, got: Dr, +sec: List<&2, F32>, +kk: U32, +ii: U32) -> Dr: match hop got: case 0n Dr{acc, xs}: Dr{acc, xs} case 1n+p Dr{acc, xs}: dc.cols(p, dc.stored(dc.cell(dc.x0(acc, kk, ii), sec, kk, ii), xs, ii), sec, kk, u.add(ii, 1)) # eight floats carried through the DCT-II row butterfly type E8 is Data: E8{z0: F32, z1: F32, z2: F32, z3: F32, z4: F32, z5: F32, z6: F32, z7: F32} # finish the first stage def but.sum2( +xt: F32, +y0: F32, +y7: F32, +y1: F32, +y6: F32, +y2: F32, +y5: F32, +y3: F32 ) -> E8: E8{f.add(y0, y3), f.add(y1, y2), y6, f.sub(y1, y2), f.sub(y0, y3), y5, y7, xt} # first add/sub stage of a DCT-II row def but.sum(+x0: F32, +x1: F32, +x2: F32, +x3: F32, +x4: F32, +x5: F32, +x6: F32, +x7: F32) -> E8: but.sum2(f.sub(x0, x7), f.add(x0, x7), f.sub(x1, x6), f.add(x1, x6), f.sub(x2, x5), f.add(x2, x5), f.sub(x3, x4), f.add(x3, x4)) # second stage: the 45-degree rotations def but.mid(ee: E8) -> E8: match ee: case E8{+z0, +z1, +z2, +z3, +z4, +z5, +z6, +z7}: E8{f.add(z0, z1), f.mul(f.add(z3, z4), c.c707()), f.mul(f.add(z2, z6), c.c707()), z4, f.mul(f.sub(z0, z1), c.c707()), f.add(z5, z2), f.add(z6, z7), z7} # last rotation update def but.rot3( +o0: F32, +w3: F32, +w6: F32, +y4: F32, +o4: F32, +p5: F32, +p7: F32, +xt: F32 ) -> E8: E8{o0, p7, w3, y4, o4, f.sub(p5, f.mul(p7, c.c198())), f.sub(xt, w6), f.add(xt, w6)} # complete the rotation def but.rot2( +o0: F32, +w3: F32, +w6: F32, +y4: F32, +o4: F32, +p5: F32, +w7: F32, +xt: F32 ) -> E8: but.rot3(o0, w3, w6, y4, o4, p5, f.add(w7, f.mul(p5, c.c382())), xt) # third stage: the small-angle rotation, using the updated values def but.rot(ee: E8) -> E8: match ee: case E8{+z0, +z1, +z2, +z3, +z4, +z5, +z6, +z7}: but.rot2(z0, z1, z2, z3, z4, f.sub(z5, f.mul(z6, c.c198())), z6, z7) # eight row outputs def but.end(ee: E8) -> List<&2, F32>: match ee: case E8{+z0, +z1, +z2, +z3, +z4, +z5, +z6, +z7}: cs.pack(z0, f.mul(f.add(z7, z1), c.c509()), f.mul(f.add(z3, z2), c.c541()), f.mul(f.sub(z6, z5), c.c601()), z4, f.mul(f.add(z6, z5), c.c899()), f.mul(f.sub(z3, z2), c.c130()), f.mul(f.sub(z7, z1), c.c256()), 0.0) # cs.pack takes 9; the row is 8. Drop the padding below. def but.out(ee: E8) -> List<&2, F32>: ls.take(8n, but.end(ee), []) # butterfly the eight samples and write the row back def but.sum8(acc: Array, +ys: List<&2, F32>, +base: U32) -> Array: ix.poke(8n, acc, but.out(but.rot(but.mid(but.sum(f.at(ys, 0), f.at(ys, 1), f.at(ys, 2), f.at(ys, 3), f.at(ys, 4), f.at(ys, 5), f.at(ys, 6), f.at(ys, 7))))), base) # butterfly one row. The eight inputs are one slice of the working Array. def but.use(got: Il, +base: U32) -> Array: match got: case Il{acc, ys}: but.sum8(acc, ys, base) # butterfly one row of eight and write it back def but.row(xs: Array, +base: U32) -> Array: but.use(ix.take(xs, base, 7n), base) # butterfly all four rows def dc.fly(hop: Nat, xs: Array, +base: U32) -> Array: match hop: case 0n: xs case 1n+p: dc.fly(p, but.row(xs, base), u.add(base, 8)) # seven samples of one scatter column, row-major type Sv is Type: Sv{xs: Array, +p0: F32, +p1: F32, +p2: F32, +p3: F32, +p4: F32, +p5: F32, +p6: F32} # row 3, column ii+1 def sc.v6(got: Dx, +r0: F32, +r1: F32, +r1n: F32, +r2: F32, +r2n: F32, +r3: F32) -> Sv: match got: case Dx{acc, r3n}: Sv{acc, r0, r1, r1n, r2, r2n, r3, r3n} # row 3, column ii def sc.v5(got: Dx, +ii: U32, +r0: F32, +r1: F32, +r1n: F32, +r2: F32, +r2n: F32) -> Sv: match got: case Dx{acc, r3}: sc.v6(dx.at(acc, u.add(ii, 25)), r0, r1, r1n, r2, r2n, r3) # row 2, column ii+1 def sc.v4(got: Dx, +ii: U32, +r0: F32, +r1: F32, +r1n: F32, +r2: F32) -> Sv: match got: case Dx{acc, r2n}: sc.v5(dx.at(acc, u.add(ii, 24)), ii, r0, r1, r1n, r2, r2n) # row 2, column ii def sc.v3(got: Dx, +ii: U32, +r0: F32, +r1: F32, +r1n: F32) -> Sv: match got: case Dx{acc, r2}: sc.v4(dx.at(acc, u.add(ii, 17)), ii, r0, r1, r1n, r2) # row 1, column ii+1 def sc.v2(got: Dx, +ii: U32, +r0: F32, +r1: F32) -> Sv: match got: case Dx{acc, r1n}: sc.v3(dx.at(acc, u.add(ii, 16)), ii, r0, r1, r1n) # row 1, column ii def sc.v1(got: Dx, +ii: U32, +r0: F32) -> Sv: match got: case Dx{acc, r1}: sc.v2(dx.at(acc, u.add(ii, 9)), ii, r0, r1) # row 0, column ii, then the six neighbour slots def sc.v0(got: Dx, +ii: U32) -> Sv: match got: case Dx{acc, r0}: sc.v1(dx.at(acc, u.add(ii, 8)), ii, r0) # the seven samples of column ii, each an Array.get def sc.vals(xs: Array, +ii: U32) -> Sv: sc.v0(dx.at(xs, ii), ii) # spectrum Array beside the rows Array type Sc is Type: Sc{acc: Array, xs: Array} # four band values from the seven column samples def sc.bands( +p0: F32, +p1: F32, +p2: F32, +p3: F32, +p4: F32, +p5: F32, +p6: F32 ) -> Q4: Q4{p0, f.add(f.add(p3, p5), p6), f.add(p1, p2), f.add(f.add(p4, p5), p6)} # store the four bands and hand both Arrays back def sc.put4(xs: Array, acc: Array, qq: Q4, +yy: U32) -> Sc: match qq: case Q4{a0, a1, a2, a3}: Sc{Array.set(F32, Array.set(F32, Array.set(F32, Array.set(F32, acc, yy, a0), u.add(yy, 18), a1), u.add(yy, 36), a2), u.add(yy, 54), a3), xs} # write one scatter group from the seven samples def sc.put(got: Sv, acc: Array, +yy: U32) -> Sc: match got: case Sv{xs, p0, p1, p2, p3, p4, p5, p6}: sc.put4(xs, acc, sc.bands(p0, p1, p2, p3, p4, p5, p6), yy) # scatter one group of four bands into the spectrum Array def sc.one(acc: Array, xs: Array, +yy: U32, +ii: U32) -> Sc: sc.put(sc.vals(xs, ii), acc, yy) # seven scatter steps; the eighth is the tail def sc.go(hop: Nat, got: Sc, +yy: U32, +ii: U32) -> Sc: match hop got: case 0n Sc{acc, xs}: Sc{acc, xs} case 1n+p Sc{acc, xs}: sc.go(p, sc.one(acc, xs, yy, ii), u.add(yy, 72), u.add(ii, 1)) # row 0, 1, 2 and 3 of column 7 type Sf is Type: Sf{xs: Array, +r0: F32, +r1: F32, +r2: F32, +r3: F32} # fourth tail sample, index 31 def sc.e3(got: Dx, +r0: F32, +r1: F32, +r2: F32) -> Sf: match got: case Dx{acc, r3}: Sf{acc, r0, r1, r2, r3} # third tail sample, index 23 def sc.e2(got: Dx, +r0: F32, +r1: F32) -> Sf: match got: case Dx{acc, r2}: sc.e3(dx.at(acc, 31), r0, r1, r2) # second tail sample, index 15 def sc.e1(got: Dx, +r0: F32) -> Sf: match got: case Dx{acc, r1}: sc.e2(dx.at(acc, 23), r0, r1) # first tail sample, index 7, then the other three rows def sc.e0(got: Dx) -> Sf: match got: case Dx{acc, r0}: sc.e1(dx.at(acc, 15), r0) # four tail bands. r3 is both the high sum and the last band. def sc.tset(acc: Array, +r0: F32, +r1: F32, +r2: F32, +r3: F32, +yy: U32) -> Array: Array.set(F32, Array.set(F32, Array.set(F32, Array.set(F32, acc, yy, r0), u.add(yy, 18), f.add(r2, r3)), u.add(yy, 36), r1), u.add(yy, 54), r3) # drop the rows Array once the four tail samples have been read def sc.tend(got: Sf, acc: Array, +yy: U32) -> Array: match got: case Sf{xs, r0, r1, r2, r3}: match xs: case _xs: sc.tset(acc, r0, r1, r2, r3, yy) # last scatter group, which has no i+1 neighbour. The rows Array ends here. def sc.tail(got: Sc, +yy: U32) -> Array: match got: case Sc{acc, xs}: sc.tend(sc.e0(dx.at(xs, 7)), acc, yy) # scatter every band of frequency bin kk def sc.run(acc: Array, xs: Array, +kk: U32) -> Array: sc.tail(sc.go(7n, Sc{acc, xs}, kk, 0), u.add(kk, 504)) # butterfly the gathered rows, then scatter them back into the spectrum def dc.rows(got: Dr) -> Dr: match got: case Dr{acc, xs}: Dr{acc, dc.fly(4n, xs, 0)} # scatter one frequency bin. The working rows are consumed here. def dc.scat(got: Dr, +kk: U32) -> Array: match got: case Dr{acc, xs}: sc.run(acc, xs, kk) # one frequency bin of the 32-point DCT-II. The working rows stay an Array. def dc.one(acc: Array, +sec: List<&2, F32>, +kk: U32) -> Array: dc.scat(dc.rows(dc.cols(8n, Dr{acc, Array.new(F32, 5n, 0.0)}, sec, kk, 0)), kk) # DCT-II of 18 frequency bins. kk is the first bin's index in the spectrum Array. def dc.run(hop: Nat, acc: Array, +sec: List<&2, F32>, +kk: U32) -> Array: match hop: case 0n: acc case 1n+p: dc.run(p, dc.one(acc, sec, kk), sec, u.add(kk, 1)) # one F32 read from the polyphase delay Array type La is Type: La{lin: Array, +v: F32} def la.of(got: Array & F32) -> La: (lin, v) = got La{lin, v} def ln.at(lin: Array, +ii: U32) -> La: la.of(Array.get(F32, lin, ii)) def ln.set(lin: Array, +ii: U32, +vv: F32) -> Array: Array.set(F32, lin, ii, vv) def ln.fill(xs: List<&2, F32>, a: Array, +ii: U32) -> Array: match xs: case Nil{}: a case +hd <> tl: ln.fill(tl, Array.set(F32, a, ii, hd), u.add(ii, 1)) # pull one sample into an accumulator; hop is remaining after this read def ln.pull(hop: Nat, got: La, +ii: U32, acc: List<&2, F32>) -> List<&2, F32>: match hop got: case 0n La{lin, +cur}: match lin: case _l: List.reverse(&2, F32, cur <> acc) case 1n+p La{lin, +cur}: ln.pull(p, ln.at(lin, u.add(ii, 1)), u.add(ii, 1), cur <> acc) # first hop+1 slots of an Array as a list def ln.list(a: Array, hop: Nat) -> List<&2, F32>: ln.pull(hop, ln.at(a, 0), 0, []) # four running polyphase sums, a and b type Mac is Data: Mac{aa: List<&2, F32>, bb: List<&2, F32>} # MAC state plus the delay line handle mac.* only reads type Mk is Type: Mk{st: Mac, lin: Array} # mode 0 replaces, mode 2 flips the a product, otherwise both add def mac.av(+mode: U32, +zv: F32, +yv: F32, +w0: F32, +w1: F32) -> F32: match mode: case 2: f.sub(f.mul(yv, w1), f.mul(zv, w0)) case _: f.sub(f.mul(zv, w0), f.mul(yv, w1)) # apply one lane of a window load def mac.wr( +aa: List<&2, F32>, +bb: List<&2, F32>, +mode: U32, +jj: U32, +bv: F32, +av: F32 ) -> Mac: match mode: case 0: Mac{f.set(aa, jj, av), f.set(bb, jj, bv)} case _: Mac{f.set(aa, jj, f.add(f.at(aa, jj), av)), f.set(bb, jj, f.add(f.at(bb, jj), bv))} # one of the four lanes, y slot def mac.jy( st: Mac, got: La, +w0: F32, +w1: F32, +mode: U32, +jj: U32, +zv: F32 ) -> Mk: match st got: case Mac{aa, bb} La{lin2, yv}: Mk{mac.wr(aa, bb, mode, jj, f.add(f.mul(zv, w1), f.mul(yv, w0)), mac.av(mode, zv, yv, w0, w1)), lin2} # one of the four lanes, z slot def mac.jz( st: Mac, got: La, +w0: F32, +w1: F32, +vy: U32, +mode: U32, +jj: U32 ) -> Mk: match got: case La{lin1, zv}: mac.jy(st, ln.at(lin1, u.add(vy, jj)), w0, w1, mode, jj, zv) # one of the four lanes def mac.j( st: Mac, lin: Array, +w0: F32, +w1: F32, +vz: U32, +vy: U32, +mode: U32, +jj: U32 ) -> Mk: mac.jz(st, ln.at(lin, u.add(vz, jj)), w0, w1, vy, mode, jj) # four lanes of one load, advance one lane def mac.js_next(mk: Mk, +w0: F32, +w1: F32, +vz: U32, +vy: U32, +mode: U32, +jj: U32) -> Mk: match mk: case Mk{st, lin}: mac.j(st, lin, w0, w1, vz, vy, mode, jj) # four lanes of one load def mac.js( hop: Nat, mk: Mk, +w0: F32, +w1: F32, +vz: U32, +vy: U32, +mode: U32, +jj: U32 ) -> Mk: match hop: case 0n: mk case 1n+p: mac.js(p, mac.js_next(mk, w0, w1, vz, vy, mode, jj), w0, w1, vz, vy, mode, u.add(jj, 1)) # odd or even load def mac.mode3(odd: Bool) -> U32: match odd: case True{}: 2 case False{}: 1 # pick the load mode def mac.mode2(zero: Bool, odd: Bool) -> U32: match zero: case True{}: 0 case False{}: mac.mode3(odd) # k = 0 replaces, odd k uses mode 2, other even k uses mode 1 def mac.mode(+kk: U32) -> U32: mac.mode2(U32.is_eq(kk, 0), U32.is_eq(u.and(kk), 1)) # window-coefficient index for block row ii and load kk def mac.pos(+ii: U32, +kk: U32) -> U32: u.add(u.mul(u.sub(14, ii), 16), u.mul(kk, 2)) # one window coefficient handed back beside the table Array.get just read type Sw is Type: Sw{syn: Array, +w: F32} def sw.of(got: Array & F32) -> Sw: (syn, w) = got Sw{syn, w} def sw.at(syn: Array, +ii: U32) -> Sw: sw.of(Array.get(F32, syn, ii)) # 242 synthesis-window coefficients. 256 slots cover every mac.pos index. def sw.tab() -> Array: ln.fill(Enc.f.of(Tab.tab.syn()), Array.new(F32, 8n, 0.0), 0) # MAC state plus the window table the next load still reads type Mw is Type: Mw{mk: Mk, syn: Array} # second coefficient, then the eight-lane load def mac.step2(mk: Mk, got: Sw, +w0: F32, +zlin: U32, +ii: U32, +kk: U32) -> Mw: match mk got: case Mk{st, lin} Sw{syn, w1}: Mw{mac.js(4n, Mk{st, lin}, w0, w1, u.sub(u.add(zlin, u.mul(ii, 4)), u.mul(kk, 64)), u.sub(u.add(zlin, u.mul(ii, 4)), u.mul(u.sub(15, kk), 64)), mac.mode(kk), 0), syn} # first coefficient, then its neighbour def mac.step1(mk: Mk, got: Sw, +pos: U32, +zlin: U32, +ii: U32, +kk: U32) -> Mw: match got: case Sw{syn, w0}: mac.step2(mk, sw.at(syn, u.add(pos, 1)), w0, zlin, ii, kk) # one of the eight loads. Each tap reads the window Array. def mac.step(mk: Mk, syn: Array, +zlin: U32, +ii: U32, +kk: U32) -> Mw: mac.step1(mk, sw.at(syn, mac.pos(ii, kk)), mac.pos(ii, kk), zlin, ii, kk) # eight loads def mac.k(hop: Nat, got: Mw, +zlin: U32, +ii: U32, +kk: U32) -> Mw: match hop got: case 0n Mw{mk, syn}: Mw{mk, syn} case 1n+p Mw{mk, syn}: mac.k(p, mac.step(mk, syn, zlin, ii, kk), zlin, ii, u.add(kk, 1)) # scale a polyphase sum into a sample def mac.sc(+vv: F32) -> F32: f.mul(vv, c.inv()) # write the eight samples this row produces def mac.p8( pcm: Array, +aa: List<&2, F32>, +bb: List<&2, F32>, +nch: U32, +pcmi: U32, +ii: U32 ) -> Array: ln.set(ln.set(ln.set(ln.set(ln.set(ln.set(ln.set(ln.set(pcm, u.add(pcmi, u.add(u.mul(u.sub(15, ii), nch), u.sub(nch, 1))), mac.sc(f.at(aa, 1))), u.add(pcmi, u.add(u.mul(u.add(17, ii), nch), u.sub(nch, 1))), mac.sc(f.at(bb, 1))), u.add(pcmi, u.mul(u.sub(15, ii), nch)), mac.sc(f.at(aa, 0))), u.add(pcmi, u.mul(u.add(17, ii), nch)), mac.sc(f.at(bb, 0))), u.add(pcmi, u.add(u.mul(u.sub(47, ii), nch), u.sub(nch, 1))), mac.sc(f.at(aa, 3))), u.add(pcmi, u.add(u.mul(u.add(49, ii), nch), u.sub(nch, 1))), mac.sc(f.at(bb, 3))), u.add(pcmi, u.mul(u.sub(47, ii), nch)), mac.sc(f.at(aa, 2))), u.add(pcmi, u.mul(u.add(49, ii), nch)), mac.sc(f.at(bb, 2))) # store the row def mac.put(pcm: Array, st: Mac, +nch: U32, +pcmi: U32, +ii: U32) -> Array: match st: case Mac{+aa, +bb}: mac.p8(pcm, aa, bb, nch, pcmi, ii) # fresh accumulators def mac.z() -> Mac: Mac{List.replicate(F32, 4n, 0.0), List.replicate(F32, 4n, 0.0)} # channel offset: the right channel starts at 576 when there are two def blk.off(+nch: U32, right: Bool) -> U32: match right: case False{}: 0 case True{}: u.mul(u.sub(nch, 1), 576) # one F32 read from the post-DCT granule Array type Ga is Type: Ga{gr: Array, +v: F32} def ga.of(got: Array & F32) -> Ga: (gr, v) = got Ga{gr, v} def ga.at(gr: Array, +ii: U32) -> Ga: ga.of(Array.get(F32, gr, ii)) # delay line handed back beside the granule Array a write just read type Lg is Type: Lg{lin: Array, gr: Array} # subband index. right selects the second channel, which starts at 576. def blk.ix(+nch: U32, +band: U32, +tm: U32, right: Bool) -> U32: u.add(u.add(blk.off(nch, right), u.mul(band, 18)), tm) def blk.rd(gr: Array, +nch: U32, +band: U32, +tm: U32, right: Bool) -> Ga: ga.at(gr, blk.ix(nch, band, tm, right)) # fourth sample of a four-lane write def blk.w4d(g: Ga, lin: Array, +at: U32, +v0: F32, +v1: F32, +v2: F32) -> Lg: match g: case Ga{gr, v3}: Lg{ln.set(ln.set(ln.set(ln.set(lin, at, v0), u.add(at, 1), v1), u.add(at, 2), v2), u.add(at, 3), v3), gr} # third sample: band 0, this channel's left slot def blk.w4c(g: Ga, lin: Array, +nch: U32, +at: U32, +tm: U32, +v0: F32, +v1: F32) -> Lg: match g: case Ga{gr, v2}: blk.w4d(blk.rd(gr, nch, 0, tm, True{}), lin, at, v0, v1, v2) # second sample: the right channel of this band def blk.w4b(g: Ga, lin: Array, +nch: U32, +at: U32, +tm: U32, +v0: F32) -> Lg: match g: case Ga{gr, v1}: blk.w4c(blk.rd(gr, nch, 0, tm, False{}), lin, nch, at, tm, v0, v1) # first sample: the left channel of this band def blk.w4a(g: Ga, lin: Array, +nch: U32, +at: U32, +band: U32, +tm: U32) -> Lg: match g: case Ga{gr, v0}: blk.w4b(blk.rd(gr, nch, band, tm, True{}), lin, nch, at, tm, v0) # write the four samples at zlin + 4*slot def blk.w4( lin: Array, gr: Array, +nch: U32, +at: U32, +band: U32, +tm: U32 ) -> Lg: blk.w4a(blk.rd(gr, nch, band, tm, False{}), lin, nch, at, band, tm) # right channel of a two-lane write def blk.w2b(g: Ga, lin: Array, +at: U32, +v0: F32) -> Lg: match g: case Ga{gr, v1}: Lg{ln.set(ln.set(lin, at, v0), u.add(at, 1), v1), gr} # left channel of a two-lane write def blk.w2a(g: Ga, lin: Array, +nch: U32, +at: U32, +band: U32, +tm: U32) -> Lg: match g: case Ga{gr, v0}: blk.w2b(blk.rd(gr, nch, band, tm, True{}), lin, at, v0) # the second write of a row uses band (1+ii), not band 0 def blk.w2( lin: Array, gr: Array, +nch: U32, +at: U32, +band: U32, +tm: U32 ) -> Lg: blk.w2a(blk.rd(gr, nch, band, tm, False{}), lin, nch, at, band, tm) # second preamble write: band 16 at the next time def blk.pre2(got: Lg, +nch: U32, +zlin: U32, +tm: U32) -> Lg: match got: case Lg{lin, gr}: blk.w4(lin, gr, nch, u.add(zlin, 124), 16, u.add(tm, 1)) # preamble: bands 16 and 0 at the current and next time def blk.pre( lin: Array, gr: Array, +nch: U32, +zlin: U32, +tm: U32 ) -> Lg: blk.pre2(blk.w4(lin, gr, nch, u.add(zlin, 60), 16, tm), nch, zlin, tm) # index of the history write, which steps back 16 slots def blk.hist(+zlin: U32, +ii: U32) -> U32: u.add(u.sub(zlin, u.mul(u.sub(16, ii), 4)), 2) # fourth lane: this band at the next time, right channel def blk.wsd(g: Ga, lin: Array, +at: U32, +v0: F32, +v1: F32, +v2: F32) -> Lg: match g: case Ga{gr, v3}: Lg{ln.set(ln.set(ln.set(ln.set(lin, at, v0), u.add(at, 1), v1), u.add(at, 2), v2), u.add(at, 3), v3), gr} # third lane: this band at the next time, left channel def blk.wsc( g: Ga, lin: Array, +nch: U32, +at: U32, +band: U32, +tm: U32, +v0: F32, +v1: F32 ) -> Lg: match g: case Ga{gr, v2}: blk.wsd(blk.rd(gr, nch, band, u.add(tm, 1), True{}), lin, at, v0, v1, v2) # second lane: this band at this time, right channel def blk.wsb(g: Ga, lin: Array, +nch: U32, +at: U32, +band: U32, +tm: U32, +v0: F32) -> Lg: match g: case Ga{gr, v1}: blk.wsc(blk.rd(gr, nch, band, u.add(tm, 1), False{}), lin, nch, at, band, tm, v0, v1) # first lane: this band at this time, left channel def blk.wsa(g: Ga, lin: Array, +nch: U32, +at: U32, +band: U32, +tm: U32) -> Lg: match g: case Ga{gr, v0}: blk.wsb(blk.rd(gr, nch, band, tm, True{}), lin, nch, at, band, tm, v0) # same band at this time and the next, four lanes def blk.ws( lin: Array, gr: Array, +nch: U32, +at: U32, +band: U32, +tm: U32 ) -> Lg: blk.wsa(blk.rd(gr, nch, band, tm, False{}), lin, nch, at, band, tm) # current-slot writes for row ii def blk.i1( lin: Array, gr: Array, +nch: U32, +zlin: U32, +tm: U32, +ii: U32 ) -> Lg: blk.ws(lin, gr, nch, u.add(zlin, u.mul(ii, 4)), u.sub(31, ii), tm) # history write, after the next-slot write has returned the granule def blk.i2b(got: Lg, +nch: U32, +zlin: U32, +tm: U32, +ii: U32) -> Lg: match got: case Lg{lin, gr}: blk.w2(lin, gr, nch, blk.hist(zlin, ii), u.add(1, ii), tm) # next-slot and history writes for row ii def blk.i2( lin: Array, gr: Array, +nch: U32, +zlin: U32, +tm: U32, +ii: U32 ) -> Lg: blk.i2b(blk.w2(lin, gr, nch, u.add(u.add(zlin, u.mul(ii, 4)), 64), u.add(1, ii), u.add(tm, 1)), nch, zlin, tm, ii) # history writes, after the current-slot write has returned the granule def blk.ib(got: Lg, +nch: U32, +zlin: U32, +tm: U32, +ii: U32) -> Lg: match got: case Lg{lin, gr}: blk.i2(lin, gr, nch, zlin, tm, ii) # both writes for row ii def blk.i( lin: Array, gr: Array, +nch: U32, +zlin: U32, +tm: U32, +ii: U32 ) -> Lg: blk.ib(blk.i1(lin, gr, nch, zlin, tm, ii), nch, zlin, tm, ii) # (z+896) - z, times 29 def sp.a1a3(g1: La, +v896: F32) -> La: match g1: case La{lin2, vz}: La{lin2, f.mul(f.sub(v896, vz), Enc.f32.bits(1105723392))} def sp.a1a2(g0: La, +zz: U32) -> La: match g0: case La{lin1, v896}: sp.a1a3(ln.at(lin1, zz), v896) def sp.a1a(lin: Array, +zz: U32) -> La: sp.a1a2(ln.at(lin, u.add(zz, 896)), zz) # slots 1 and 13, times 213 def sp.a1b2(g1: La, +acc: F32, +v64: F32) -> La: match g1: case La{lin2, v832}: La{lin2, f.add(acc, f.mul(f.add(v64, v832), Enc.f32.bits(1129644032)))} def sp.a1b1(g0: La, +acc: F32, +zz: U32) -> La: match g0: case La{lin1, v64}: sp.a1b2(ln.at(lin1, u.add(zz, 832)), acc, v64) def sp.a1b(+acc: F32, lin: Array, +zz: U32) -> La: sp.a1b1(ln.at(lin, u.add(zz, 64)), acc, zz) # slots 12 and 2, times 459 def sp.a1c2(g1: La, +acc: F32, +v768: F32) -> La: match g1: case La{lin2, v128}: La{lin2, f.add(acc, f.mul(f.sub(v768, v128), Enc.f32.bits(1139113984)))} def sp.a1c1(g0: La, +acc: F32, +zz: U32) -> La: match g0: case La{lin1, v768}: sp.a1c2(ln.at(lin1, u.add(zz, 128)), acc, v768) def sp.a1c(+acc: F32, lin: Array, +zz: U32) -> La: sp.a1c1(ln.at(lin, u.add(zz, 768)), acc, zz) # slots 3 and 11, times 2037 def sp.a1d2(g1: La, +acc: F32, +v192: F32) -> La: match g1: case La{lin2, v704}: La{lin2, f.add(acc, f.mul(f.add(v192, v704), Enc.f32.bits(1157537792)))} def sp.a1d1(g0: La, +acc: F32, +zz: U32) -> La: match g0: case La{lin1, v192}: sp.a1d2(ln.at(lin1, u.add(zz, 704)), acc, v192) def sp.a1d(+acc: F32, lin: Array, +zz: U32) -> La: sp.a1d1(ln.at(lin, u.add(zz, 192)), acc, zz) # slots 10 and 4, times 5153 def sp.a1e2(g1: La, +acc: F32, +v640: F32) -> La: match g1: case La{lin2, v256}: La{lin2, f.add(acc, f.mul(f.sub(v640, v256), Enc.f32.bits(1168181248)))} def sp.a1e1(g0: La, +acc: F32, +zz: U32) -> La: match g0: case La{lin1, v640}: sp.a1e2(ln.at(lin1, u.add(zz, 256)), acc, v640) def sp.a1e(+acc: F32, lin: Array, +zz: U32) -> La: sp.a1e1(ln.at(lin, u.add(zz, 640)), acc, zz) # slots 5 and 9, times 6574 def sp.a1f2(g1: La, +acc: F32, +v320: F32) -> La: match g1: case La{lin2, v576}: La{lin2, f.add(acc, f.mul(f.add(v320, v576), Enc.f32.bits(1171091456)))} def sp.a1f1(g0: La, +acc: F32, +zz: U32) -> La: match g0: case La{lin1, v320}: sp.a1f2(ln.at(lin1, u.add(zz, 576)), acc, v320) def sp.a1f(+acc: F32, lin: Array, +zz: U32) -> La: sp.a1f1(ln.at(lin, u.add(zz, 320)), acc, zz) # slots 8 and 6, times 37489, then slot 7 times 75038 def sp.a1g4(g2: La, +acc: F32, +v512: F32, +v384: F32) -> La: match g2: case La{lin3, v448}: La{lin3, f.add(f.add(acc, f.mul(f.sub(v512, v384), Enc.f32.bits(1192390912))), f.mul(v448, Enc.f32.bits(1200787200)))} def sp.a1g3(g1: La, +acc: F32, +v512: F32, +zz: U32) -> La: match g1: case La{lin2, v384}: sp.a1g4(ln.at(lin2, u.add(zz, 448)), acc, v512, v384) def sp.a1g2(g0: La, +acc: F32, +zz: U32) -> La: match g0: case La{lin1, v512}: sp.a1g3(ln.at(lin1, u.add(zz, 384)), acc, v512, zz) def sp.a1g(+acc: F32, lin: Array, +zz: U32) -> La: sp.a1g2(ln.at(lin, u.add(zz, 512)), acc, zz) def sp.a1go6(g: La, +zz: U32) -> La: match g: case La{lin6, acc6}: sp.a1g(acc6, lin6, zz) def sp.a1go5(g: La, +zz: U32) -> La: match g: case La{lin5, acc5}: sp.a1go6(sp.a1f(acc5, lin5, zz), zz) def sp.a1go4(g: La, +zz: U32) -> La: match g: case La{lin4, acc4}: sp.a1go5(sp.a1e(acc4, lin4, zz), zz) def sp.a1go3(g: La, +zz: U32) -> La: match g: case La{lin3, acc3}: sp.a1go4(sp.a1d(acc3, lin3, zz), zz) def sp.a1go2(g: La, +zz: U32) -> La: match g: case La{lin2, acc2}: sp.a1go3(sp.a1c(acc2, lin2, zz), zz) def sp.a1go1(g: La, +zz: U32) -> La: match g: case La{lin1, acc1}: sp.a1go2(sp.a1b(acc1, lin1, zz), zz) # first grouped window sum def sp.a1(lin: Array, +zz: U32) -> La: sp.a1go1(sp.a1a(lin, zz), zz) # taps 898, 770, 642, 514 def sp.a2a4(g3: La, +v898: F32, +v770: F32, +v642: F32) -> La: match g3: case La{lin4, v514}: La{lin4, f.add(f.add(f.add(f.mul(v898, Enc.f32.bits(1120927744)), f.mul(v770, Enc.f32.bits(1153687552))), f.mul(v642, Enc.f32.bits(1175976960))), f.mul(v514, Enc.f32.bits(1199182592)))} def sp.a2a3(g2: La, +v898: F32, +v770: F32, +zz: U32) -> La: match g2: case La{lin3, v642}: sp.a2a4(ln.at(lin3, u.add(zz, 514)), v898, v770, v642) def sp.a2a2(g1: La, +v898: F32, +zz: U32) -> La: match g1: case La{lin2, v770}: sp.a2a3(ln.at(lin2, u.add(zz, 642)), v898, v770, zz) def sp.a2a1(g0: La, +zz: U32) -> La: match g0: case La{lin1, v898}: sp.a2a2(ln.at(lin1, u.add(zz, 770)), v898, zz) def sp.a2a(lin: Array, +zz: U32) -> La: sp.a2a1(ln.at(lin, u.add(zz, 898)), zz) # taps 386 and 258 def sp.a2b2(g1: La, +acc: F32, +v386: F32) -> La: match g1: case La{lin2, v258}: La{lin2, f.add(f.add(acc, f.mul(v386, Enc.f32.bits(3323714560))), f.mul(v258, Enc.f32.bits(3258187776)))} def sp.a2b1(g0: La, +acc: F32, +zz: U32) -> La: match g0: case La{lin1, v386}: sp.a2b2(ln.at(lin1, u.add(zz, 258)), acc, v386) def sp.a2b(+acc: F32, lin: Array, +zz: U32) -> La: sp.a2b1(ln.at(lin, u.add(zz, 386)), acc, zz) # tap 130 def sp.a2c1(g: La, +acc: F32) -> La: match g: case La{lin1, v130}: La{lin1, f.add(acc, f.mul(v130, Enc.f32.bits(1125253120)))} def sp.a2c(+acc: F32, lin: Array, +zz: U32) -> La: sp.a2c1(ln.at(lin, u.add(zz, 130)), acc) # tap 2 def sp.a2d1(g: La, +acc: F32) -> La: match g: case La{lin1, v2}: La{lin1, f.add(acc, f.mul(v2, Enc.f32.bits(3231711232)))} def sp.a2d(+acc: F32, lin: Array, +zz: U32) -> La: sp.a2d1(ln.at(lin, u.add(zz, 2)), acc) def sp.a2go3(g: La, +zz: U32) -> La: match g: case La{lin3, acc3}: sp.a2d(acc3, lin3, zz) def sp.a2go2(g: La, +zz: U32) -> La: match g: case La{lin2, acc2}: sp.a2go3(sp.a2c(acc2, lin2, zz), zz) def sp.a2go1(g: La, +zz: U32) -> La: match g: case La{lin1, acc1}: sp.a2go2(sp.a2b(acc1, lin1, zz), zz) # second window sum, eight weighted taps def sp.a2(lin: Array, +zz: U32) -> La: sp.a2go1(sp.a2a(lin, zz), zz) # PCM and delay line while pairing synth samples type Pp is Type: Pp{pcm: Array, lin: Array} # one synth pair: two scaled samples def sp.pair2(g2: La, pcm: Array, +pcmi: U32, +nch: U32) -> Pp: match g2: case La{lin2, v2}: Pp{ln.set(pcm, u.add(pcmi, u.mul(16, nch)), mac.sc(v2)), lin2} def sp.pair1(g1: La, pcm: Array, +pcmi: U32, +nch: U32, +zz: U32) -> Pp: match g1: case La{lin1, v1}: sp.pair2(sp.a2(lin1, zz), ln.set(pcm, pcmi, mac.sc(v1)), pcmi, nch) def sp.pair(st: Pp, +pcmi: U32, +nch: U32, +zz: U32) -> Pp: match st: case Pp{pcm, lin}: sp.pair1(sp.a1(lin, zz), pcm, pcmi, nch, zz) # four synth pairs that open a block def blk.pairs(st: Pp, +nch: U32, +pcmi: U32, +z0: U32) -> Pp: sp.pair(sp.pair(sp.pair(sp.pair(st, u.add(pcmi, u.sub(nch, 1)), nch, u.add(z0, 1)), u.add(u.add(pcmi, u.mul(32, nch)), u.sub(nch, 1)), nch, u.add(z0, 65)), pcmi, nch, z0), u.add(pcmi, u.mul(32, nch)), nch, u.add(z0, 64)) # delay line and the PCM written so far type Sb is Type: Sb{lin: Array, pcm: Array} def blk.armed1(got: Pp) -> Sb: match got: case Pp{pcm2, lin2}: Sb{lin2, pcm2} # pairs read the delay line the preamble just wrote def blk.armed(lin: Array, pcm: Array, +nch: U32, +tm: U32) -> Sb: blk.armed1(blk.pairs(Pp{pcm, lin}, nch, u.mul(u.mul(32, nch), tm), u.add(u.mul(tm, 64), 60))) # polyphase state, the granule Array the rows still read, and the window table type Bg is Type: Bg{st: Sb, gr: Array, syn: Array} # preamble done: the pairs read the delay line, the granule and window stay put def blk.arm2(got: Lg, pcm: Array, syn: Array, +nch: U32, +tm: U32) -> Bg: match got: case Lg{lin, gr}: Bg{blk.armed(lin, pcm, nch, tm), gr, syn} # open a block: preamble, then the four pairs def blk.arm(got: Bg, +nch: U32, +tm: U32) -> Bg: match got: case Bg{st, gr, syn}: match st: case Sb{lin, pcm}: blk.arm2(blk.pre(lin, gr, nch, u.add(u.mul(tm, 64), 960), tm), pcm, syn, nch, tm) # a finished row, with the window table the next row still reads type Sr is Type: Sr{st: Sb, syn: Array} def blk.row1(got: Mw, pcm: Array, +nch: U32, +tm: U32, +ii: U32) -> Sr: match got: case Mw{mk, syn}: match mk: case Mk{st, lin2}: Sr{Sb{lin2, mac.put(pcm, st, nch, u.mul(u.mul(32, nch), tm), ii)}, syn} def blk.row3(got: Sr, gr: Array) -> Bg: match got: case Sr{st, syn}: Bg{st, gr, syn} # window the row just written def blk.row( lin: Array, pcm: Array, syn: Array, +nch: U32, +tm: U32, +ii: U32 ) -> Sr: blk.row1(mac.k(8n, Mw{Mk{mac.z(), lin}, syn}, u.add(u.mul(tm, 64), 960), ii, 0), pcm, nch, tm, ii) # window the row, keeping the granule for the rows still to come def blk.step2( got: Lg, pcm: Array, syn: Array, +nch: U32, +tm: U32, +ii: U32 ) -> Bg: match got: case Lg{lin, gr}: blk.row3(blk.row(lin, pcm, syn, nch, tm, ii), gr) # one row of the block, from i = 14 down to 0 def blk.step(got: Bg, +nch: U32, +tm: U32, +ii: U32) -> Bg: match got: case Bg{st, gr, syn}: match st: case Sb{lin, pcm}: blk.step2(blk.i(lin, gr, nch, u.add(u.mul(tm, 64), 960), tm, ii), pcm, syn, nch, tm, ii) # fifteen rows def blk.i.go(hop: Nat, got: Bg, +nch: U32, +tm: U32, +ii: U32) -> Bg: match hop got: case 0n Bg{st, gr, syn}: Bg{st, gr, syn} case 1n+p Bg{st, gr, syn}: blk.i.go(p, blk.step(Bg{st, gr, syn}, nch, tm, ii), nch, tm, u.sub(ii, 1)) # one time block def blk.one(got: Bg, +nch: U32, +tm: U32) -> Bg: blk.i.go(15n, blk.arm(got, nch, tm), nch, tm, 14) # nine time blocks, time = 0, 2, ..., 16 def blk.go(hop: Nat, got: Bg, +nch: U32, +tm: U32) -> Bg: match hop got: case 0n Bg{st, gr, syn}: Bg{st, gr, syn} case 1n+p Bg{st, gr, syn}: blk.go(p, blk.one(Bg{st, gr, syn}, nch, tm), nch, u.add(tm, 2)) # one even delay sample and the odd slot kept from the carried tail. # hop is the pairs still to come after this one. The sample was read at ii. def qmf.ev(hop: Nat, old: List<&2, F32>, got: La, +ii: U32, acc: List<&2, F32>) -> List<&2, F32>: match hop old got: case 0n +_e <> +odd <> _tl La{lin, +cur}: match lin: case _lin: List.reverse(&2, F32, odd <> (cur <> acc)) case 1n+p +_e <> +odd <> tl La{lin, +cur}: qmf.ev(p, tl, ln.at(lin, u.add(ii, 2)), u.add(ii, 2), odd <> (cur <> acc)) case _ _ La{lin, _v}: match lin: case _lin: List.reverse(&2, F32, acc) # stereo tail: 960 delay-Array samples in one pull. The old list is dropped. def qmf.st(+old: List<&2, F32>, lin: Array) -> List<&2, F32>: match old: case _old: ln.pull(959n, ln.at(lin, 1152), 1152, []) # mono keeps odd slots; stereo replaces the whole tail from the delay Array def qmf.pick(mono: Bool, +qmf: List<&2, F32>, lin: Array) -> List<&2, F32>: match mono: case True{}: qmf.ev(479n, qmf, ln.at(lin, 1152), 1152, []) case False{}: qmf.st(qmf, lin) # save the 15-block tail off the delay Array, one pass, no List.set def qmf.save(lin: Array, +qmf: List<&2, F32>, +nch: U32) -> List<&2, F32>: qmf.pick(U32.is_eq(nch, 1), qmf, lin) # PCM, the overlap to carry, and the polyphase delay type Pc is Data: Pc{pcm: List<&2, F32>, ov: List<&2, F32>, qmf: List<&2, F32>} # zero overlap, 288 samples, one channel def syn.zov() -> List<&2, F32>: List.replicate(F32, 288n, 0.0) # zero polyphase delay def syn.zqmf() -> List<&2, F32>: List.replicate(F32, 960n, 0.0) # antialias, IMDCT, and the sign flip for one channel def ch.prep( acc: Array, +ov: List<&2, F32>, +aa: List<&2, F32>, +tw: List<&2, F32>, +win: List<&2, F32>, +base: U32 ) -> Im: im.sign(im.go(32n, Im{aa.run(acc, aa, base), ov}, tw, win, 0, base), base) # append the second channel's overlap. The spectrum Array already holds both. def syn.st2(+ova: List<&2, F32>, bb: Im) -> Im: match bb: case Im{buf, ovb}: Im{buf, List.append(&2, F32, ova, ovb)} # join two channel states. The right channel starts at bin 576. def syn.st( aa: Im, +ovb: List<&2, F32>, +cs: List<&2, F32>, +tw: List<&2, F32>, +win: List<&2, F32> ) -> Im: match aa: case Im{buf, ova}: syn.st2(ova, ch.prep(buf, ovb, cs, tw, win, 576)) # prepare one or two channels def syn.ch( mono: Bool, acc: Array, +ov: List<&2, F32>, +aa: List<&2, F32>, +tw: List<&2, F32>, +win: List<&2, F32> ) -> Im: match mono: case True{}: ch.prep(acc, ov, aa, tw, win, 0) case False{}: syn.st(ch.prep(acc, ls.slice(ov, 0, 288n), aa, tw, win, 0), ls.slice(ov, 288, 288n), aa, tw, win) # tables shared by a granule. 2048 slots cover mono 576 and stereo 1152. def syn.ready(+spec: List<&2, F32>, +ov: List<&2, F32>, +nch: U32) -> Im: syn.ch(U32.is_eq(nch, 1), ln.fill(spec, Array.new(F32, 11n, 0.0), 0), ov, Enc.f.of(Tab.tab.aa()), Enc.f.of(Tab.tab.tw()), Enc.f.of(Tab.tab.winl())) # second channel starts at bin 576 def sy.dct2(mono: Bool, acc: Array, +sec: List<&2, F32>) -> Array: match mono: case True{}: dc.run(18n, acc, sec, 0) case False{}: dc.run(18n, dc.run(18n, acc, sec, 0), sec, 576) # DCT-II of each channel. The spectrum Array is the one IMDCT just wrote. def sy.dct(mono: Bool, acc: Array, +sec: List<&2, F32>) -> Array: sy.dct2(mono, acc, sec) # delay line: the carried tail, then zeros (4096-slot Array starts at 0.0) def sy.lin(+qmf: List<&2, F32>) -> Array: ln.fill(qmf, Array.new(F32, 12n, 0.0), 0) # empty PCM for one granule (2048-slot Array covers mono 576 and stereo 1152) def sy.pcm() -> Array: Array.new(F32, 11n, 0.0) # PCM Array back to a list of length 576*nch def sy.pcm.list(+nch: U32, pcm: Array) -> List<&2, F32>: ln.list(pcm, U32.to_nat(u.sub(u.mul(576, nch), 1))) # the granule Array is finished once the nine blocks have read it def syn.drop(gr: Array, st: Sb, +ov: List<&2, F32>, +qmf: List<&2, F32>, +nch: U32) -> Pc: match gr: case _gr: match st: case Sb{lin, pcm}: Pc{sy.pcm.list(nch, pcm), ov, qmf.save(lin, qmf, nch)} # run the polyphase and keep the overlap from the IMDCT def syn.out(got: Bg, +ov: List<&2, F32>, +qmf: List<&2, F32>, +nch: U32) -> Pc: match got: case Bg{st, gr, syn}: match syn: case _syn: syn.drop(gr, st, ov, qmf, nch) # DCT-II then nine polyphase blocks. The spectrum Array is what the polyphase reads. def syn.pcm(buf: Array, +ov: List<&2, F32>, +qmf: List<&2, F32>, +nch: U32) -> Pc: syn.out(blk.go(9n, Bg{Sb{sy.lin(qmf), sy.pcm()}, sy.dct(U32.is_eq(nch, 1), buf, Enc.f.of(Tab.tab.sec())), sw.tab()}, nch, 0), ov, qmf, nch) # synthesis of a prepared granule def syn.finish(st: Im, +qmf: List<&2, F32>, +nch: U32) -> Pc: match st: case Im{buf, ov}: syn.pcm(buf, ov, qmf, nch) # one granule to PCM. spec is 576 samples per channel. ov is 288 per channel. def syn.gran(+spec: List<&2, F32>, +ov: List<&2, F32>, +qmf: List<&2, F32>, +nch: U32) -> Pc: syn.finish(syn.ready(spec, ov, nch), qmf, nch) # IMDCT only, 32 bands, no antialias. Used to check the unit impulse. def syn.im(+buf: List<&2, F32>, +ov: List<&2, F32>) -> Im: im.go(32n, Im{ln.fill(buf, Array.new(F32, 11n, 0.0), 0), ov}, Enc.f.of(Tab.tab.tw()), Enc.f.of(Tab.tab.winl()), 0, 0) # one PCM sample of a granule, as a binary32 word def syn.nth(st: Pc, +ii: U32) -> U32: match st: case Pc{pcm, _ov, _qmf}: Enc.f32.word(f.at(pcm, ii)) # drop the spectrum Array once sample 0 has been read def syn.word1(got: Dx) -> U32: match got: case Dx{acc, v}: match acc: case _acc: Enc.f32.word(v) # first sample of an IMDCT buffer, as a binary32 word def syn.word(st: Im) -> U32: match st: case Im{buf, _ov}: syn.word1(dx.at(buf, 0)) # PCM word ii of one granule whose spectrum is a unit impulse at bin 0. def syn.probe(+ii: U32) -> U32: syn.nth(syn.gran(f.set(List.replicate(F32, 576n, 0.0), 0, Enc.f32.bits(1065353216)), syn.zov(), syn.zqmf(), 1), ii) # unit impulse at bin 0. The first sample is the float32 word 1022453915. def syn.impulse() -> U32: syn.word(syn.im(f.set(List.replicate(F32, 576n, 0.0), 0, Enc.f32.bits(1065353216)), syn.zov()))