src/mp3_syn.bend source
src/mp3_syn.bend on the hub · documented module
# MPEG-1 Layer III synthesis: antialias, IMDCT-36, sign flip, DCT-II, polyphase.import Baseimport ./mp3_tab.bend as Tabimport ./mp3_enc.bend as Enc# add two binary32 valuesdef f.add(+aa: F32, +bb: F32) -> F32: (aa + bb : F32)# subtract two binary32 valuesdef f.sub(+aa: F32, +bb: F32) -> F32: (aa - bb : F32)# multiply two binary32 valuesdef f.mul(+aa: F32, +bb: F32) -> F32: (aa * bb : F32)# sample at an index, or 0 past the enddef f.at(+xs: List<&2, F32>, +ii: U32) -> F32: Enc.f.nth(xs, ii)# replace one sampledef f.set(xs: List<&2, F32>, +ii: U32, +vv: F32) -> List<&2, F32>: Enc.f.put(xs, ii, vv)# unsigned adddef u.add(+aa: U32, +bb: U32) -> U32: (aa + bb : U32)# unsigned subtractdef u.sub(+aa: U32, +bb: U32) -> U32: (aa - bb : U32)# unsigned multiplydef u.mul(+aa: U32, +bb: U32) -> U32: (aa * bb : U32)# low bitdef u.and(+kk: U32) -> U32: (kk .&. 1 : U32)# 0.5def c.half() -> F32: Enc.f32.bits(1056964608)# 0.93969262def c.c939() -> F32: Enc.f32.bits(1064341426)# 0.76604444def c.c766() -> F32: Enc.f32.bits(1061428093)# 0.17364818def c.c173() -> F32: Enc.f32.bits(1043452116)# 0.86602540def c.c866() -> F32: Enc.f32.bits(1063105495)# 0.98480775def c.c984() -> F32: Enc.f32.bits(1065098332)# 0.34202014def c.c342() -> F32: Enc.f32.bits(1051663684)# 0.64278761def c.c642() -> F32: Enc.f32.bits(1059360187)# 0.70710677def c.c707() -> F32: Enc.f32.bits(1060439283)# 0.198912367def c.c198() -> F32: Enc.f32.bits(1045147567)# 0.382683432def c.c382() -> F32: Enc.f32.bits(1053028117)# 0.50979561def c.c509() -> F32: Enc.f32.bits(1057128951)# 0.54119611def c.c541() -> F32: Enc.f32.bits(1057655764)# 0.60134488def c.c601() -> F32: Enc.f32.bits(1058664893)# 0.89997619def c.c899() -> F32: Enc.f32.bits(1063675095)# 1.30656302def c.c130() -> F32: Enc.f32.bits(1067924853)# 2.56291556def c.c256() -> F32: Enc.f32.bits(1076102863)# 1/32768, the polyphase scaledef c.inv() -> F32: Enc.f32.bits(939524096)# drop a prefixdef ls.drop(hop: Nat, xs: List<&2, F32>) -> List<&2, F32>: match hop xs: case 0n ys: ys case _ Nil{}: [] case 1n+p _hd <> tl: ls.drop(p, tl)# take a prefix, reversed into accdef ls.take(hop: Nat, xs: List<&2, F32>, acc: List<&2, F32>) -> List<&2, F32>: match hop xs: case 0n _: List.reverse(&2, F32, acc) case _ Nil{}: List.reverse(&2, F32, acc) case 1n+p +hd <> tl: ls.take(p, tl, hd <> acc)# a slice of length hop starting at an indexdef ls.slice(+xs: List<&2, F32>, +at: U32, hop: Nat) -> List<&2, F32>: ls.take(hop, ls.drop(U32.to_nat(at), xs), [])# write a short list over a destination starting at an indexdef ls.poke(hop: Nat, dst: List<&2, F32>, src: List<&2, F32>, +at: U32) -> List<&2, F32>: match hop src: case 0n _: dst case _ Nil{}: dst case 1n+p +hd <> tl: ls.poke(p, f.set(dst, at, hd), tl, u.add(at, 1))# nine samples, index 0 firstdef cs.pack( +v0: F32, +v1: F32, +v2: F32, +v3: F32, +v4: F32, +v5: F32, +v6: F32, +v7: F32, +v8: F32) -> List<&2, F32>: v0 <> (v1 <> (v2 <> (v3 <> (v4 <> (v5 <> (v6 <> (v7 <> (v8 <> []))))))))# cosine half of one IMDCT-36 banddef cs.co(+gg: List<&2, F32>) -> List<&2, F32>: cs.pack(F32.neg(f.at(gg, 0)), f.add(f.at(gg, 1), f.at(gg, 2)), F32.neg(f.add(f.at(gg, 3), f.at(gg, 4))), f.add(f.at(gg, 5), f.at(gg, 6)), F32.neg(f.add(f.at(gg, 7), f.at(gg, 8))), f.add(f.at(gg, 9), f.at(gg, 10)), F32.neg(f.add(f.at(gg, 11), f.at(gg, 12))), f.add(f.at(gg, 13), f.at(gg, 14)), F32.neg(f.add(f.at(gg, 15), f.at(gg, 16))))# sine half of one IMDCT-36 banddef cs.si(+gg: List<&2, F32>) -> List<&2, F32>: cs.pack(f.at(gg, 17), f.sub(f.at(gg, 16), f.at(gg, 15)), f.sub(f.at(gg, 13), f.at(gg, 14)), f.sub(f.at(gg, 12), f.at(gg, 11)), f.sub(f.at(gg, 9), f.at(gg, 10)), f.sub(f.at(gg, 8), f.at(gg, 7)), f.sub(f.at(gg, 5), f.at(gg, 6)), f.sub(f.at(gg, 4), f.at(gg, 3)), f.sub(f.at(gg, 1), f.at(gg, 2)))# even-part sum used by the 9-point DCTdef d9.t0(+y0: F32, +y6: F32) -> F32: f.add(y0, f.mul(y6, c.half()))# even-part differencedef d9.s0(+y0: F32, +y6: F32) -> F32: f.sub(y0, y6)# twiddle on bins 4 and 2def d9.t4(+y4: F32, +y2: F32) -> F32: f.mul(f.add(y4, y2), c.c939())# twiddle on bins 8 and 2def d9.t2(+y8: F32, +y2: F32) -> F32: f.mul(f.add(y8, y2), c.c766())# twiddle on bins 4 and 8def d9.s6(+y4: F32, +y8: F32) -> F32: f.mul(f.sub(y4, y8), c.c173())# even residualdef d9.s4(+y4: F32, +y8: F32, +y2: F32) -> F32: f.sub(f.add(y4, y8), y2)# scaled residualdef d9.s2(+s0: F32, +s4: F32) -> F32: f.sub(s0, f.mul(s4, c.half()))# centre bin of the even DCTdef d9.y4(+s4: F32, +s0: F32) -> F32: f.add(s4, s0)# high even outputdef d9.s8c(+t0: F32, +t2: F32, +s6: F32) -> F32: f.add(f.sub(t0, t2), s6)# low even outputdef d9.s0c(+t0: F32, +t4: F32, +t2: F32) -> F32: f.add(f.sub(t0, t4), t2)# mid even outputdef d9.s4c(+t0: F32, +t4: F32, +s6: F32) -> F32: f.sub(f.add(t0, t4), s6)# odd centredef d9.s3(+y3: F32) -> F32: f.mul(y3, c.c866())# odd low twiddledef d9.tb0(+y5: F32, +y1: F32) -> F32: f.mul(f.add(y5, y1), c.c984())# odd high twiddledef d9.tb4(+y5: F32, +y7: F32) -> F32: f.mul(f.sub(y5, y7), c.c342())# odd mid twiddledef d9.tb2(+y1: F32, +y7: F32) -> F32: f.mul(f.add(y1, y7), c.c642())# odd residualdef d9.s1(+y1: F32, +y5: F32, +y7: F32) -> F32: f.mul(f.sub(f.sub(y1, y5), y7), c.c866())# odd combinationdef d9.s5c(+t0: F32, +s3: F32, +t2: F32) -> F32: f.sub(f.sub(t0, s3), t2)# odd combinationdef d9.s7c(+t4: F32, +s3: F32, +t0: F32) -> F32: f.sub(f.sub(t4, s3), t0)# odd combinationdef d9.s3c(+t4: F32, +s3: F32, +t2: F32) -> F32: f.sub(f.add(t4, s3), t2)# even and odd 9-point DCT results, ready to interleavetype D9 is Data: D9{ s2b: F32, y4o: F32, s8c: F32, s0c: F32, s4c: F32, s1b: F32, s5c: F32, s7c: F32, s3c: F32 }# assemble the nine DCT outputs from the even and odd partsdef d9.from( +t0: F32, +s0: F32, +t4: F32, +t2: F32, +s6: F32, +s4: F32, +s3: F32, +u0: F32, +u4: F32, +u2: F32, +s1: F32) -> D9: D9{d9.s2(s0, s4), d9.y4(s4, s0), d9.s8c(t0, t2, s6), d9.s0c(t0, t4, t2), d9.s4c(t0, t4, s6), s1, d9.s5c(u0, s3, u2), d9.s7c(u4, s3, u0), d9.s3c(u4, s3, u2)}# 9-point DCT of one halfdef d9.build( +y0: F32, +y1: F32, +y2: F32, +y3: F32, +y4: F32, +y5: F32, +y6: F32, +y7: F32, +y8: F32) -> D9: d9.from(d9.t0(y0, y6), d9.s0(y0, y6), d9.t4(y4, y2), d9.t2(y8, y2), d9.s6(y4, y8), d9.s4(y4, y8, y2), d9.s3(y3), d9.tb0(y5, y1), d9.tb4(y5, y7), d9.tb2(y1, y7), d9.s1(y1, y5, y7))# nine DCT outputs in bin orderdef d9.list(dd: D9) -> List<&2, F32>: match dd: case D9{+s2b, +y4o, +s8c, +s0c, +s4c, +s1b, +s5c, +s7c, +s3c}: cs.pack(f.sub(s4c, s7c), f.add(s2b, s1b), f.sub(s0c, s3c), f.add(s8c, s5c), y4o, f.sub(s8c, s5c), f.add(s0c, s3c), f.sub(s2b, s1b), f.add(s4c, s7c))# 9-point DCTdef d9.run(+ys: List<&2, F32>) -> List<&2, F32>: d9.list(d9.build(f.at(ys, 0), f.at(ys, 1), f.at(ys, 2), f.at(ys, 3), f.at(ys, 4), f.at(ys, 5), f.at(ys, 6), f.at(ys, 7), f.at(ys, 8)))# negate the odd bins of a sine DCTdef d9.neg(+ys: List<&2, F32>) -> List<&2, F32>: f.set(f.set(f.set(f.set(ys, 1, F32.neg(f.at(ys, 1))), 3, F32.neg(f.at(ys, 3))), 5, F32.neg(f.at(ys, 5))), 7, F32.neg(f.at(ys, 7)))# one windowed band: 18 time samples and 9 new overlap samplestype Wn is Data: Wn{out: List<&2, F32>, nov: List<&2, F32>}# window product and the overlap it savestype Ws is Data: Ws{sm: F32, nv: F32}# aliasing pair for one window indexdef wn.pair(+co: List<&2, F32>, +si: List<&2, F32>, +tw: List<&2, F32>, +ii: U32) -> Ws: Ws{f.add(f.mul(f.at(co, ii), f.at(tw, u.add(ii, 9))), f.mul(f.at(si, ii), f.at(tw, ii))), f.sub(f.mul(f.at(co, ii), f.at(tw, ii)), f.mul(f.at(si, ii), f.at(tw, u.add(ii, 9))))}# write one window index into the banddef wn.put(st: Wn, +ov: List<&2, F32>, +win: List<&2, F32>, +ii: U32, ws: Ws) -> Wn: match st ws: case Wn{out, nov} Ws{+sm, +nv}: Wn{f.set(f.set(out, ii, f.sub(f.mul(f.at(ov, ii), f.at(win, ii)), f.mul(sm, f.at(win, u.add(ii, 9))))), u.sub(17, ii), f.add(f.mul(f.at(ov, ii), f.at(win, u.add(ii, 9))), f.mul(sm, f.at(win, ii)))), f.set(nov, ii, nv)}# one of the nine window stepsdef wn.one( st: Wn, +co: List<&2, F32>, +si: List<&2, F32>, +ov: List<&2, F32>, +tw: List<&2, F32>, +win: List<&2, F32>, +ii: U32) -> Wn: wn.put(st, ov, win, ii, wn.pair(co, si, tw, ii))# window all nine indicesdef wn.go( hop: Nat, st: Wn, +co: List<&2, F32>, +si: List<&2, F32>, +ov: List<&2, F32>, +tw: List<&2, F32>, +win: List<&2, F32>, +ii: U32) -> Wn: match hop: case 0n: st case 1n+p: wn.go(p, wn.one(st, co, si, ov, tw, win, ii), co, si, ov, tw, win, u.add(ii, 1))# IMDCT-36 of one 18-point banddef im.band(+gg: List<&2, F32>, +ov: List<&2, F32>, +tw: List<&2, F32>, +win: List<&2, F32>) -> Wn: wn.go(9n, Wn{List.replicate(F32, 18n, 0.0), List.replicate(F32, 9n, 0.0)}, d9.run(cs.co(gg)), d9.neg(d9.run(cs.si(gg))), ov, tw, win, 0)# one F32 read from the IMDCT/DCT spectrum Arraytype Dx is Type: Dx{acc: Array<F32>, +v: F32}def dx.of(got: Array<F32> & F32) -> Dx: (acc, v) = got Dx{acc, v}def dx.at(acc: Array<F32>, +ii: U32) -> Dx: dx.of(Array.get(F32, acc, ii))# spectrum Array beside the samples just readtype Il is Type: Il{acc: Array<F32>, xs: List<&2, F32>}# pull one sample; hop is remaining after this readdef ix.pull(hop: Nat, got: Dx, +ii: U32, ys: List<&2, F32>) -> Il: match hop got: case 0n Dx{acc, +cur}: Il{acc, List.reverse(&2, F32, cur <> ys)} case 1n+p Dx{acc, +cur}: ix.pull(p, dx.at(acc, u.add(ii, 1)), u.add(ii, 1), cur <> ys)# hop+1 samples starting at `at`. The spectrum Array is handed back.def ix.take(acc: Array<F32>, +at: U32, hop: Nat) -> Il: ix.pull(hop, dx.at(acc, at), at, [])# write a short list into the spectrum Arraydef ix.poke(hop: Nat, acc: Array<F32>, src: List<&2, F32>, +at: U32) -> Array<F32>: match hop src: case 0n _: acc case _ Nil{}: acc case 1n+p +hd <> tl: ix.poke(p, Array.set(F32, acc, at, hd), tl, u.add(at, 1))# 576 lines in the spectrum Array, and 288 overlap samplestype Im is Type: Im{buf: Array<F32>, ov: List<&2, F32>}# write one band back into the spectrum Array and the overlapdef im.store(buf: Array<F32>, +ov: List<&2, F32>, ww: Wn, +jj: U32, +base: U32) -> Im: match ww: case Wn{out, nov}: Im{ix.poke(18n, buf, out, u.add(base, u.mul(jj, 18))), ls.poke(9n, ov, nov, u.mul(jj, 9))}# band samples in hand; window them and storedef im.got( got: Il, +ov: List<&2, F32>, +tw: List<&2, F32>, +win: List<&2, F32>, +jj: U32, +base: U32) -> Im: match got: case Il{buf, gg}: im.store(buf, ov, im.band(gg, ls.slice(ov, u.mul(jj, 9), 9n), tw, win), jj, base)# IMDCT-36 of band jj. Each of the 18 samples is an Array.get.def im.one(st: Im, +tw: List<&2, F32>, +win: List<&2, F32>, +jj: U32, +base: U32) -> Im: match st: case Im{buf, +ov}: im.got(ix.take(buf, u.add(base, u.mul(jj, 18)), 17n), ov, tw, win, jj, base)# IMDCT-36 over 32 bands. base is the channel offset in the spectrum Array.def im.go(hop: Nat, st: Im, +tw: List<&2, F32>, +win: List<&2, F32>, +jj: U32, +base: U32) -> Im: match hop: case 0n: st case 1n+p: im.go(p, im.one(st, tw, win, jj, base), tw, win, u.add(jj, 1), base)# high index of one aliasing pairdef aa.ixh(+base: U32, +bb: U32, +ii: U32) -> U32: u.add(base, u.add(u.mul(u.add(bb, 1), 18), ii))# low index of one aliasing pairdef aa.ixl(+base: U32, +bb: U32, +ii: U32) -> U32: u.add(base, u.add(u.mul(bb, 18), u.sub(17, ii)))# spectrum Array beside the two aliasing samplestype Ia is Type: Ia{acc: Array<F32>, +up: F32, +dp: F32}# low sample, then both indices are in handdef aa.lo(got: Dx, +up: F32) -> Ia: match got: case Dx{acc, dp}: Ia{acc, up, dp}# high sample in hand; read the low sample nextdef aa.rd1(got: Dx, +lo: U32) -> Ia: match got: case Dx{acc, up}: aa.lo(dx.at(acc, lo), up)# read both samples of one aliasing pair. Each read is an Array.get.def aa.rd(acc: Array<F32>, +hi: U32, +lo: U32) -> Ia: aa.rd1(dx.at(acc, hi), lo)# butterfly of one aliasing pair. c0 and c1 are the two cosine rows.def aa.wr( acc: Array<F32>, +up: F32, +dp: F32, +c0: F32, +c1: F32, +hi: U32, +lo: U32) -> Array<F32>: Array.set(F32, Array.set(F32, acc, hi, f.sub(f.mul(c0, up), f.mul(c1, dp))), lo, f.add(f.mul(c1, up), f.mul(c0, dp)))# write the pair once both samples have been readdef aa.use(got: Ia, +c0: F32, +c1: F32, +hi: U32, +lo: U32) -> Array<F32>: match got: case Ia{acc, up, dp}: aa.wr(acc, up, dp, c0, c1, hi, lo)# one butterfly at index ii of band bbdef aa.one(acc: Array<F32>, +aa: List<&2, F32>, +bb: U32, +ii: U32, +base: U32) -> Array<F32>: aa.use(aa.rd(acc, aa.ixh(base, bb, ii), aa.ixl(base, bb, ii)), f.at(aa, ii), f.at(aa, u.add(ii, 8)), aa.ixh(base, bb, ii), aa.ixl(base, bb, ii))# eight butterflies between band bb and the nextdef aa.i(hop: Nat, acc: Array<F32>, +aa: List<&2, F32>, +bb: U32, +ii: U32, +base: U32) -> Array<F32>: match hop: case 0n: acc case 1n+p: aa.i(p, aa.one(acc, aa, bb, ii, base), aa, bb, u.add(ii, 1), base)# antialias from low bands toward high bandsdef aa.b(hop: Nat, acc: Array<F32>, +aa: List<&2, F32>, +bb: U32, +base: U32) -> Array<F32>: match hop: case 0n: acc case 1n+p: aa.b(p, aa.i(8n, acc, aa, bb, 0, base), aa, u.add(bb, 1), base)# 31 long-block aliasing pairs. base is the channel offset in the spectrum Array.def aa.run(acc: Array<F32>, +aa: List<&2, F32>, +base: U32) -> Array<F32>: aa.b(31n, acc, aa, 0, base)# negate the sample just readdef sg.neg(got: Dx, +at: U32) -> Array<F32>: match got: case Dx{acc, v}: Array.set(F32, acc, at, F32.neg(v))# negate one odd sample of an odd banddef sg.i(hop: Nat, acc: Array<F32>, +base: U32, +ii: U32) -> Array<F32>: match hop: case 0n: acc case 1n+p: sg.i(p, sg.neg(dx.at(acc, u.add(base, ii)), u.add(base, ii)), base, u.add(ii, 2))# odd samples of 16 odd bands, starting at sample 18def sg.b(hop: Nat, acc: Array<F32>, +base: U32) -> Array<F32>: match hop: case 0n: acc case 1n+p: sg.b(p, sg.i(9n, acc, base, 1), u.add(base, 36))# frequency inversion after the IMDCT. base is the channel offset.def im.sign(st: Im, +base: U32) -> Im: match st: case Im{buf, ov}: Im{sg.b(16n, buf, u.add(base, 18)), ov}# four DCT-II column resultstype Q4 is Data: Q4{a0: F32, a1: F32, a2: F32, a3: F32}# column sample i of the 32-band vectordef dc.x0(acc: Array<F32>, +kk: U32, +ii: U32) -> Dx: dx.at(acc, u.add(kk, u.mul(ii, 18)))# mirrored columndef dc.x1(acc: Array<F32>, +kk: U32, +ii: U32) -> Dx: dx.at(acc, u.add(kk, u.mul(u.sub(15, ii), 18)))# upper columndef dc.x2(acc: Array<F32>, +kk: U32, +ii: U32) -> Dx: dx.at(acc, u.add(kk, u.mul(u.add(16, ii), 18)))# upper mirrordef dc.x3(acc: Array<F32>, +kk: U32, +ii: U32) -> Dx: dx.at(acc, u.add(kk, u.mul(u.sub(31, ii), 18)))# finish the column butterflydef dc.q2(+t0: F32, +t1: F32, +t2: F32, +t3: F32, +s2: F32) -> Q4: Q4{f.add(t0, t1), f.mul(f.sub(t0, t1), s2), f.add(t3, t2), f.mul(f.sub(t3, t2), s2)}# one DCT-II column from four samples and three secantsdef dc.q(+x0: F32, +x1: F32, +x2: F32, +x3: F32, +s0: F32, +s1: F32, +s2: F32) -> Q4: dc.q2(f.add(x0, x3), f.add(x1, x2), f.mul(f.sub(x1, x2), s0), f.mul(f.sub(x0, x3), s1), s2)# spectrum Array beside one finished columntype Dq is Type: Dq{acc: Array<F32>, q: Q4}# fourth sample, then the column butterflydef dc.rd3(got: Dx, +x0: F32, +x1: F32, +x2: F32, +s0: F32, +s1: F32, +s2: F32) -> Dq: match got: case Dx{acc, x3}: Dq{acc, dc.q(x0, x1, x2, x3, s0, s1, s2)}# third sample and the three secantsdef dc.rd2(got: Dx, +sec: List<&2, F32>, +kk: U32, +ii: U32, +x0: F32, +x1: F32) -> Dq: match got: case Dx{acc, x2}: dc.rd3(dc.x3(acc, kk, ii), x0, x1, x2, f.at(sec, u.mul(ii, 3)), f.at(sec, u.add(u.mul(ii, 3), 1)), f.at(sec, u.add(u.mul(ii, 3), 2)))# second sampledef dc.rd1(got: Dx, +sec: List<&2, F32>, +kk: U32, +ii: U32, +x0: F32) -> Dq: match got: case Dx{acc, x1}: dc.rd2(dc.x2(acc, kk, ii), sec, kk, ii, x0, x1)# read column ii. Each of the four samples is an Array.get.def dc.cell(got: Dx, +sec: List<&2, F32>, +kk: U32, +ii: U32) -> Dq: match got: case Dx{acc, x0}: dc.rd1(dc.x1(acc, kk, ii), sec, kk, ii, x0)# store one column into the 4 by 8 working Arraydef dc.store(xs: Array<F32>, qq: Q4, +ii: U32) -> Array<F32>: match qq: case Q4{a0, a1, a2, a3}: Array.set(F32, Array.set(F32, Array.set(F32, Array.set(F32, xs, ii, a0), u.add(ii, 8), a1), u.add(ii, 16), a2), u.add(ii, 24), a3)# spectrum Array beside the 4 by 8 working rowstype Dr is Type: Dr{acc: Array<F32>, xs: Array<F32>}# write one gathered column and keep both Arraysdef dc.stored(got: Dq, xs: Array<F32>, +ii: U32) -> Dr: match got: case Dq{acc, q}: Dr{acc, dc.store(xs, q, ii)}# gather all eight columns before any write-backdef dc.cols(hop: Nat, got: Dr, +sec: List<&2, F32>, +kk: U32, +ii: U32) -> Dr: match hop got: case 0n Dr{acc, xs}: Dr{acc, xs} case 1n+p Dr{acc, xs}: dc.cols(p, dc.stored(dc.cell(dc.x0(acc, kk, ii), sec, kk, ii), xs, ii), sec, kk, u.add(ii, 1))# eight floats carried through the DCT-II row butterflytype E8 is Data: E8{z0: F32, z1: F32, z2: F32, z3: F32, z4: F32, z5: F32, z6: F32, z7: F32}# finish the first stagedef but.sum2( +xt: F32, +y0: F32, +y7: F32, +y1: F32, +y6: F32, +y2: F32, +y5: F32, +y3: F32) -> E8: E8{f.add(y0, y3), f.add(y1, y2), y6, f.sub(y1, y2), f.sub(y0, y3), y5, y7, xt}# first add/sub stage of a DCT-II rowdef but.sum(+x0: F32, +x1: F32, +x2: F32, +x3: F32, +x4: F32, +x5: F32, +x6: F32, +x7: F32) -> E8: but.sum2(f.sub(x0, x7), f.add(x0, x7), f.sub(x1, x6), f.add(x1, x6), f.sub(x2, x5), f.add(x2, x5), f.sub(x3, x4), f.add(x3, x4))# second stage: the 45-degree rotationsdef but.mid(ee: E8) -> E8: match ee: case E8{+z0, +z1, +z2, +z3, +z4, +z5, +z6, +z7}: E8{f.add(z0, z1), f.mul(f.add(z3, z4), c.c707()), f.mul(f.add(z2, z6), c.c707()), z4, f.mul(f.sub(z0, z1), c.c707()), f.add(z5, z2), f.add(z6, z7), z7}# last rotation updatedef but.rot3( +o0: F32, +w3: F32, +w6: F32, +y4: F32, +o4: F32, +p5: F32, +p7: F32, +xt: F32) -> E8: E8{o0, p7, w3, y4, o4, f.sub(p5, f.mul(p7, c.c198())), f.sub(xt, w6), f.add(xt, w6)}# complete the rotationdef but.rot2( +o0: F32, +w3: F32, +w6: F32, +y4: F32, +o4: F32, +p5: F32, +w7: F32, +xt: F32) -> E8: but.rot3(o0, w3, w6, y4, o4, p5, f.add(w7, f.mul(p5, c.c382())), xt)# third stage: the small-angle rotation, using the updated valuesdef but.rot(ee: E8) -> E8: match ee: case E8{+z0, +z1, +z2, +z3, +z4, +z5, +z6, +z7}: but.rot2(z0, z1, z2, z3, z4, f.sub(z5, f.mul(z6, c.c198())), z6, z7)# eight row outputsdef but.end(ee: E8) -> List<&2, F32>: match ee: case E8{+z0, +z1, +z2, +z3, +z4, +z5, +z6, +z7}: cs.pack(z0, f.mul(f.add(z7, z1), c.c509()), f.mul(f.add(z3, z2), c.c541()), f.mul(f.sub(z6, z5), c.c601()), z4, f.mul(f.add(z6, z5), c.c899()), f.mul(f.sub(z3, z2), c.c130()), f.mul(f.sub(z7, z1), c.c256()), 0.0)# cs.pack takes 9; the row is 8. Drop the padding below.def but.out(ee: E8) -> List<&2, F32>: ls.take(8n, but.end(ee), [])# butterfly the eight samples and write the row backdef but.sum8(acc: Array<F32>, +ys: List<&2, F32>, +base: U32) -> Array<F32>: ix.poke(8n, acc, but.out(but.rot(but.mid(but.sum(f.at(ys, 0), f.at(ys, 1), f.at(ys, 2), f.at(ys, 3), f.at(ys, 4), f.at(ys, 5), f.at(ys, 6), f.at(ys, 7))))), base)# butterfly one row. The eight inputs are one slice of the working Array.def but.use(got: Il, +base: U32) -> Array<F32>: match got: case Il{acc, ys}: but.sum8(acc, ys, base)# butterfly one row of eight and write it backdef but.row(xs: Array<F32>, +base: U32) -> Array<F32>: but.use(ix.take(xs, base, 7n), base)# butterfly all four rowsdef dc.fly(hop: Nat, xs: Array<F32>, +base: U32) -> Array<F32>: match hop: case 0n: xs case 1n+p: dc.fly(p, but.row(xs, base), u.add(base, 8))# seven samples of one scatter column, row-majortype Sv is Type: Sv{xs: Array<F32>, +p0: F32, +p1: F32, +p2: F32, +p3: F32, +p4: F32, +p5: F32, +p6: F32}# row 3, column ii+1def sc.v6(got: Dx, +r0: F32, +r1: F32, +r1n: F32, +r2: F32, +r2n: F32, +r3: F32) -> Sv: match got: case Dx{acc, r3n}: Sv{acc, r0, r1, r1n, r2, r2n, r3, r3n}# row 3, column iidef sc.v5(got: Dx, +ii: U32, +r0: F32, +r1: F32, +r1n: F32, +r2: F32, +r2n: F32) -> Sv: match got: case Dx{acc, r3}: sc.v6(dx.at(acc, u.add(ii, 25)), r0, r1, r1n, r2, r2n, r3)# row 2, column ii+1def sc.v4(got: Dx, +ii: U32, +r0: F32, +r1: F32, +r1n: F32, +r2: F32) -> Sv: match got: case Dx{acc, r2n}: sc.v5(dx.at(acc, u.add(ii, 24)), ii, r0, r1, r1n, r2, r2n)# row 2, column iidef sc.v3(got: Dx, +ii: U32, +r0: F32, +r1: F32, +r1n: F32) -> Sv: match got: case Dx{acc, r2}: sc.v4(dx.at(acc, u.add(ii, 17)), ii, r0, r1, r1n, r2)# row 1, column ii+1def sc.v2(got: Dx, +ii: U32, +r0: F32, +r1: F32) -> Sv: match got: case Dx{acc, r1n}: sc.v3(dx.at(acc, u.add(ii, 16)), ii, r0, r1, r1n)# row 1, column iidef sc.v1(got: Dx, +ii: U32, +r0: F32) -> Sv: match got: case Dx{acc, r1}: sc.v2(dx.at(acc, u.add(ii, 9)), ii, r0, r1)# row 0, column ii, then the six neighbour slotsdef sc.v0(got: Dx, +ii: U32) -> Sv: match got: case Dx{acc, r0}: sc.v1(dx.at(acc, u.add(ii, 8)), ii, r0)# the seven samples of column ii, each an Array.getdef sc.vals(xs: Array<F32>, +ii: U32) -> Sv: sc.v0(dx.at(xs, ii), ii)# spectrum Array beside the rows Arraytype Sc is Type: Sc{acc: Array<F32>, xs: Array<F32>}# four band values from the seven column samplesdef sc.bands( +p0: F32, +p1: F32, +p2: F32, +p3: F32, +p4: F32, +p5: F32, +p6: F32) -> Q4: Q4{p0, f.add(f.add(p3, p5), p6), f.add(p1, p2), f.add(f.add(p4, p5), p6)}# store the four bands and hand both Arrays backdef sc.put4(xs: Array<F32>, acc: Array<F32>, qq: Q4, +yy: U32) -> Sc: match qq: case Q4{a0, a1, a2, a3}: Sc{Array.set(F32, Array.set(F32, Array.set(F32, Array.set(F32, acc, yy, a0), u.add(yy, 18), a1), u.add(yy, 36), a2), u.add(yy, 54), a3), xs}# write one scatter group from the seven samplesdef sc.put(got: Sv, acc: Array<F32>, +yy: U32) -> Sc: match got: case Sv{xs, p0, p1, p2, p3, p4, p5, p6}: sc.put4(xs, acc, sc.bands(p0, p1, p2, p3, p4, p5, p6), yy)# scatter one group of four bands into the spectrum Arraydef sc.one(acc: Array<F32>, xs: Array<F32>, +yy: U32, +ii: U32) -> Sc: sc.put(sc.vals(xs, ii), acc, yy)# seven scatter steps; the eighth is the taildef sc.go(hop: Nat, got: Sc, +yy: U32, +ii: U32) -> Sc: match hop got: case 0n Sc{acc, xs}: Sc{acc, xs} case 1n+p Sc{acc, xs}: sc.go(p, sc.one(acc, xs, yy, ii), u.add(yy, 72), u.add(ii, 1))# row 0, 1, 2 and 3 of column 7type Sf is Type: Sf{xs: Array<F32>, +r0: F32, +r1: F32, +r2: F32, +r3: F32}# fourth tail sample, index 31def sc.e3(got: Dx, +r0: F32, +r1: F32, +r2: F32) -> Sf: match got: case Dx{acc, r3}: Sf{acc, r0, r1, r2, r3}# third tail sample, index 23def sc.e2(got: Dx, +r0: F32, +r1: F32) -> Sf: match got: case Dx{acc, r2}: sc.e3(dx.at(acc, 31), r0, r1, r2)# second tail sample, index 15def sc.e1(got: Dx, +r0: F32) -> Sf: match got: case Dx{acc, r1}: sc.e2(dx.at(acc, 23), r0, r1)# first tail sample, index 7, then the other three rowsdef sc.e0(got: Dx) -> Sf: match got: case Dx{acc, r0}: sc.e1(dx.at(acc, 15), r0)# four tail bands. r3 is both the high sum and the last band.def sc.tset(acc: Array<F32>, +r0: F32, +r1: F32, +r2: F32, +r3: F32, +yy: U32) -> Array<F32>: Array.set(F32, Array.set(F32, Array.set(F32, Array.set(F32, acc, yy, r0), u.add(yy, 18), f.add(r2, r3)), u.add(yy, 36), r1), u.add(yy, 54), r3)# drop the rows Array once the four tail samples have been readdef sc.tend(got: Sf, acc: Array<F32>, +yy: U32) -> Array<F32>: match got: case Sf{xs, r0, r1, r2, r3}: match xs: case _xs: sc.tset(acc, r0, r1, r2, r3, yy)# last scatter group, which has no i+1 neighbour. The rows Array ends here.def sc.tail(got: Sc, +yy: U32) -> Array<F32>: match got: case Sc{acc, xs}: sc.tend(sc.e0(dx.at(xs, 7)), acc, yy)# scatter every band of frequency bin kkdef sc.run(acc: Array<F32>, xs: Array<F32>, +kk: U32) -> Array<F32>: sc.tail(sc.go(7n, Sc{acc, xs}, kk, 0), u.add(kk, 504))# butterfly the gathered rows, then scatter them back into the spectrumdef dc.rows(got: Dr) -> Dr: match got: case Dr{acc, xs}: Dr{acc, dc.fly(4n, xs, 0)}# scatter one frequency bin. The working rows are consumed here.def dc.scat(got: Dr, +kk: U32) -> Array<F32>: match got: case Dr{acc, xs}: sc.run(acc, xs, kk)# one frequency bin of the 32-point DCT-II. The working rows stay an Array.def dc.one(acc: Array<F32>, +sec: List<&2, F32>, +kk: U32) -> Array<F32>: dc.scat(dc.rows(dc.cols(8n, Dr{acc, Array.new(F32, 5n, 0.0)}, sec, kk, 0)), kk)# DCT-II of 18 frequency bins. kk is the first bin's index in the spectrum Array.def dc.run(hop: Nat, acc: Array<F32>, +sec: List<&2, F32>, +kk: U32) -> Array<F32>: match hop: case 0n: acc case 1n+p: dc.run(p, dc.one(acc, sec, kk), sec, u.add(kk, 1))# one F32 read from the polyphase delay Arraytype La is Type: La{lin: Array<F32>, +v: F32}def la.of(got: Array<F32> & F32) -> La: (lin, v) = got La{lin, v}def ln.at(lin: Array<F32>, +ii: U32) -> La: la.of(Array.get(F32, lin, ii))def ln.set(lin: Array<F32>, +ii: U32, +vv: F32) -> Array<F32>: Array.set(F32, lin, ii, vv)def ln.fill(xs: List<&2, F32>, a: Array<F32>, +ii: U32) -> Array<F32>: match xs: case Nil{}: a case +hd <> tl: ln.fill(tl, Array.set(F32, a, ii, hd), u.add(ii, 1))# pull one sample into an accumulator; hop is remaining after this readdef ln.pull(hop: Nat, got: La, +ii: U32, acc: List<&2, F32>) -> List<&2, F32>: match hop got: case 0n La{lin, +cur}: match lin: case _l: List.reverse(&2, F32, cur <> acc) case 1n+p La{lin, +cur}: ln.pull(p, ln.at(lin, u.add(ii, 1)), u.add(ii, 1), cur <> acc)# first hop+1 slots of an Array as a listdef ln.list(a: Array<F32>, hop: Nat) -> List<&2, F32>: ln.pull(hop, ln.at(a, 0), 0, [])# four running polyphase sums, a and btype Mac is Data: Mac{aa: List<&2, F32>, bb: List<&2, F32>}# MAC state plus the delay line handle mac.* only readstype Mk is Type: Mk{st: Mac, lin: Array<F32>}# mode 0 replaces, mode 2 flips the a product, otherwise both adddef mac.av(+mode: U32, +zv: F32, +yv: F32, +w0: F32, +w1: F32) -> F32: match mode: case 2: f.sub(f.mul(yv, w1), f.mul(zv, w0)) case _: f.sub(f.mul(zv, w0), f.mul(yv, w1))# apply one lane of a window loaddef mac.wr( +aa: List<&2, F32>, +bb: List<&2, F32>, +mode: U32, +jj: U32, +bv: F32, +av: F32) -> Mac: match mode: case 0: Mac{f.set(aa, jj, av), f.set(bb, jj, bv)} case _: Mac{f.set(aa, jj, f.add(f.at(aa, jj), av)), f.set(bb, jj, f.add(f.at(bb, jj), bv))}# one of the four lanes, y slotdef mac.jy( st: Mac, got: La, +w0: F32, +w1: F32, +mode: U32, +jj: U32, +zv: F32) -> Mk: match st got: case Mac{aa, bb} La{lin2, yv}: Mk{mac.wr(aa, bb, mode, jj, f.add(f.mul(zv, w1), f.mul(yv, w0)), mac.av(mode, zv, yv, w0, w1)), lin2}# one of the four lanes, z slotdef mac.jz( st: Mac, got: La, +w0: F32, +w1: F32, +vy: U32, +mode: U32, +jj: U32) -> Mk: match got: case La{lin1, zv}: mac.jy(st, ln.at(lin1, u.add(vy, jj)), w0, w1, mode, jj, zv)# one of the four lanesdef mac.j( st: Mac, lin: Array<F32>, +w0: F32, +w1: F32, +vz: U32, +vy: U32, +mode: U32, +jj: U32) -> Mk: mac.jz(st, ln.at(lin, u.add(vz, jj)), w0, w1, vy, mode, jj)# four lanes of one load, advance one lanedef mac.js_next(mk: Mk, +w0: F32, +w1: F32, +vz: U32, +vy: U32, +mode: U32, +jj: U32) -> Mk: match mk: case Mk{st, lin}: mac.j(st, lin, w0, w1, vz, vy, mode, jj)# four lanes of one loaddef mac.js( hop: Nat, mk: Mk, +w0: F32, +w1: F32, +vz: U32, +vy: U32, +mode: U32, +jj: U32) -> Mk: match hop: case 0n: mk case 1n+p: mac.js(p, mac.js_next(mk, w0, w1, vz, vy, mode, jj), w0, w1, vz, vy, mode, u.add(jj, 1))# odd or even loaddef mac.mode3(odd: Bool) -> U32: match odd: case True{}: 2 case False{}: 1# pick the load modedef mac.mode2(zero: Bool, odd: Bool) -> U32: match zero: case True{}: 0 case False{}: mac.mode3(odd)# k = 0 replaces, odd k uses mode 2, other even k uses mode 1def mac.mode(+kk: U32) -> U32: mac.mode2(U32.is_eq(kk, 0), U32.is_eq(u.and(kk), 1))# window-coefficient index for block row ii and load kkdef mac.pos(+ii: U32, +kk: U32) -> U32: u.add(u.mul(u.sub(14, ii), 16), u.mul(kk, 2))# one window coefficient handed back beside the table Array.get just readtype Sw is Type: Sw{syn: Array<F32>, +w: F32}def sw.of(got: Array<F32> & F32) -> Sw: (syn, w) = got Sw{syn, w}def sw.at(syn: Array<F32>, +ii: U32) -> Sw: sw.of(Array.get(F32, syn, ii))# 242 synthesis-window coefficients. 256 slots cover every mac.pos index.def sw.tab() -> Array<F32>: ln.fill(Enc.f.of(Tab.tab.syn()), Array.new(F32, 8n, 0.0), 0)# MAC state plus the window table the next load still readstype Mw is Type: Mw{mk: Mk, syn: Array<F32>}# second coefficient, then the eight-lane loaddef mac.step2(mk: Mk, got: Sw, +w0: F32, +zlin: U32, +ii: U32, +kk: U32) -> Mw: match mk got: case Mk{st, lin} Sw{syn, w1}: Mw{mac.js(4n, Mk{st, lin}, w0, w1, u.sub(u.add(zlin, u.mul(ii, 4)), u.mul(kk, 64)), u.sub(u.add(zlin, u.mul(ii, 4)), u.mul(u.sub(15, kk), 64)), mac.mode(kk), 0), syn}# first coefficient, then its neighbourdef mac.step1(mk: Mk, got: Sw, +pos: U32, +zlin: U32, +ii: U32, +kk: U32) -> Mw: match got: case Sw{syn, w0}: mac.step2(mk, sw.at(syn, u.add(pos, 1)), w0, zlin, ii, kk)# one of the eight loads. Each tap reads the window Array.def mac.step(mk: Mk, syn: Array<F32>, +zlin: U32, +ii: U32, +kk: U32) -> Mw: mac.step1(mk, sw.at(syn, mac.pos(ii, kk)), mac.pos(ii, kk), zlin, ii, kk)# eight loadsdef mac.k(hop: Nat, got: Mw, +zlin: U32, +ii: U32, +kk: U32) -> Mw: match hop got: case 0n Mw{mk, syn}: Mw{mk, syn} case 1n+p Mw{mk, syn}: mac.k(p, mac.step(mk, syn, zlin, ii, kk), zlin, ii, u.add(kk, 1))# scale a polyphase sum into a sampledef mac.sc(+vv: F32) -> F32: f.mul(vv, c.inv())# write the eight samples this row producesdef mac.p8( pcm: Array<F32>, +aa: List<&2, F32>, +bb: List<&2, F32>, +nch: U32, +pcmi: U32, +ii: U32) -> Array<F32>: ln.set(ln.set(ln.set(ln.set(ln.set(ln.set(ln.set(ln.set(pcm, u.add(pcmi, u.add(u.mul(u.sub(15, ii), nch), u.sub(nch, 1))), mac.sc(f.at(aa, 1))), u.add(pcmi, u.add(u.mul(u.add(17, ii), nch), u.sub(nch, 1))), mac.sc(f.at(bb, 1))), u.add(pcmi, u.mul(u.sub(15, ii), nch)), mac.sc(f.at(aa, 0))), u.add(pcmi, u.mul(u.add(17, ii), nch)), mac.sc(f.at(bb, 0))), u.add(pcmi, u.add(u.mul(u.sub(47, ii), nch), u.sub(nch, 1))), mac.sc(f.at(aa, 3))), u.add(pcmi, u.add(u.mul(u.add(49, ii), nch), u.sub(nch, 1))), mac.sc(f.at(bb, 3))), u.add(pcmi, u.mul(u.sub(47, ii), nch)), mac.sc(f.at(aa, 2))), u.add(pcmi, u.mul(u.add(49, ii), nch)), mac.sc(f.at(bb, 2)))# store the rowdef mac.put(pcm: Array<F32>, st: Mac, +nch: U32, +pcmi: U32, +ii: U32) -> Array<F32>: match st: case Mac{+aa, +bb}: mac.p8(pcm, aa, bb, nch, pcmi, ii)# fresh accumulatorsdef mac.z() -> Mac: Mac{List.replicate(F32, 4n, 0.0), List.replicate(F32, 4n, 0.0)}# channel offset: the right channel starts at 576 when there are twodef blk.off(+nch: U32, right: Bool) -> U32: match right: case False{}: 0 case True{}: u.mul(u.sub(nch, 1), 576)# one F32 read from the post-DCT granule Arraytype Ga is Type: Ga{gr: Array<F32>, +v: F32}def ga.of(got: Array<F32> & F32) -> Ga: (gr, v) = got Ga{gr, v}def ga.at(gr: Array<F32>, +ii: U32) -> Ga: ga.of(Array.get(F32, gr, ii))# delay line handed back beside the granule Array a write just readtype Lg is Type: Lg{lin: Array<F32>, gr: Array<F32>}# subband index. right selects the second channel, which starts at 576.def blk.ix(+nch: U32, +band: U32, +tm: U32, right: Bool) -> U32: u.add(u.add(blk.off(nch, right), u.mul(band, 18)), tm)def blk.rd(gr: Array<F32>, +nch: U32, +band: U32, +tm: U32, right: Bool) -> Ga: ga.at(gr, blk.ix(nch, band, tm, right))# fourth sample of a four-lane writedef blk.w4d(g: Ga, lin: Array<F32>, +at: U32, +v0: F32, +v1: F32, +v2: F32) -> Lg: match g: case Ga{gr, v3}: Lg{ln.set(ln.set(ln.set(ln.set(lin, at, v0), u.add(at, 1), v1), u.add(at, 2), v2), u.add(at, 3), v3), gr}# third sample: band 0, this channel's left slotdef blk.w4c(g: Ga, lin: Array<F32>, +nch: U32, +at: U32, +tm: U32, +v0: F32, +v1: F32) -> Lg: match g: case Ga{gr, v2}: blk.w4d(blk.rd(gr, nch, 0, tm, True{}), lin, at, v0, v1, v2)# second sample: the right channel of this banddef blk.w4b(g: Ga, lin: Array<F32>, +nch: U32, +at: U32, +tm: U32, +v0: F32) -> Lg: match g: case Ga{gr, v1}: blk.w4c(blk.rd(gr, nch, 0, tm, False{}), lin, nch, at, tm, v0, v1)# first sample: the left channel of this banddef blk.w4a(g: Ga, lin: Array<F32>, +nch: U32, +at: U32, +band: U32, +tm: U32) -> Lg: match g: case Ga{gr, v0}: blk.w4b(blk.rd(gr, nch, band, tm, True{}), lin, nch, at, tm, v0)# write the four samples at zlin + 4*slotdef blk.w4( lin: Array<F32>, gr: Array<F32>, +nch: U32, +at: U32, +band: U32, +tm: U32) -> Lg: blk.w4a(blk.rd(gr, nch, band, tm, False{}), lin, nch, at, band, tm)# right channel of a two-lane writedef blk.w2b(g: Ga, lin: Array<F32>, +at: U32, +v0: F32) -> Lg: match g: case Ga{gr, v1}: Lg{ln.set(ln.set(lin, at, v0), u.add(at, 1), v1), gr}# left channel of a two-lane writedef blk.w2a(g: Ga, lin: Array<F32>, +nch: U32, +at: U32, +band: U32, +tm: U32) -> Lg: match g: case Ga{gr, v0}: blk.w2b(blk.rd(gr, nch, band, tm, True{}), lin, at, v0)# the second write of a row uses band (1+ii), not band 0def blk.w2( lin: Array<F32>, gr: Array<F32>, +nch: U32, +at: U32, +band: U32, +tm: U32) -> Lg: blk.w2a(blk.rd(gr, nch, band, tm, False{}), lin, nch, at, band, tm)# second preamble write: band 16 at the next timedef blk.pre2(got: Lg, +nch: U32, +zlin: U32, +tm: U32) -> Lg: match got: case Lg{lin, gr}: blk.w4(lin, gr, nch, u.add(zlin, 124), 16, u.add(tm, 1))# preamble: bands 16 and 0 at the current and next timedef blk.pre( lin: Array<F32>, gr: Array<F32>, +nch: U32, +zlin: U32, +tm: U32) -> Lg: blk.pre2(blk.w4(lin, gr, nch, u.add(zlin, 60), 16, tm), nch, zlin, tm)# index of the history write, which steps back 16 slotsdef blk.hist(+zlin: U32, +ii: U32) -> U32: u.add(u.sub(zlin, u.mul(u.sub(16, ii), 4)), 2)# fourth lane: this band at the next time, right channeldef blk.wsd(g: Ga, lin: Array<F32>, +at: U32, +v0: F32, +v1: F32, +v2: F32) -> Lg: match g: case Ga{gr, v3}: Lg{ln.set(ln.set(ln.set(ln.set(lin, at, v0), u.add(at, 1), v1), u.add(at, 2), v2), u.add(at, 3), v3), gr}# third lane: this band at the next time, left channeldef blk.wsc( g: Ga, lin: Array<F32>, +nch: U32, +at: U32, +band: U32, +tm: U32, +v0: F32, +v1: F32) -> Lg: match g: case Ga{gr, v2}: blk.wsd(blk.rd(gr, nch, band, u.add(tm, 1), True{}), lin, at, v0, v1, v2)# second lane: this band at this time, right channeldef blk.wsb(g: Ga, lin: Array<F32>, +nch: U32, +at: U32, +band: U32, +tm: U32, +v0: F32) -> Lg: match g: case Ga{gr, v1}: blk.wsc(blk.rd(gr, nch, band, u.add(tm, 1), False{}), lin, nch, at, band, tm, v0, v1)# first lane: this band at this time, left channeldef blk.wsa(g: Ga, lin: Array<F32>, +nch: U32, +at: U32, +band: U32, +tm: U32) -> Lg: match g: case Ga{gr, v0}: blk.wsb(blk.rd(gr, nch, band, tm, True{}), lin, nch, at, band, tm, v0)# same band at this time and the next, four lanesdef blk.ws( lin: Array<F32>, gr: Array<F32>, +nch: U32, +at: U32, +band: U32, +tm: U32) -> Lg: blk.wsa(blk.rd(gr, nch, band, tm, False{}), lin, nch, at, band, tm)# current-slot writes for row iidef blk.i1( lin: Array<F32>, gr: Array<F32>, +nch: U32, +zlin: U32, +tm: U32, +ii: U32) -> Lg: blk.ws(lin, gr, nch, u.add(zlin, u.mul(ii, 4)), u.sub(31, ii), tm)# history write, after the next-slot write has returned the granuledef blk.i2b(got: Lg, +nch: U32, +zlin: U32, +tm: U32, +ii: U32) -> Lg: match got: case Lg{lin, gr}: blk.w2(lin, gr, nch, blk.hist(zlin, ii), u.add(1, ii), tm)# next-slot and history writes for row iidef blk.i2( lin: Array<F32>, gr: Array<F32>, +nch: U32, +zlin: U32, +tm: U32, +ii: U32) -> Lg: blk.i2b(blk.w2(lin, gr, nch, u.add(u.add(zlin, u.mul(ii, 4)), 64), u.add(1, ii), u.add(tm, 1)), nch, zlin, tm, ii)# history writes, after the current-slot write has returned the granuledef blk.ib(got: Lg, +nch: U32, +zlin: U32, +tm: U32, +ii: U32) -> Lg: match got: case Lg{lin, gr}: blk.i2(lin, gr, nch, zlin, tm, ii)# both writes for row iidef blk.i( lin: Array<F32>, gr: Array<F32>, +nch: U32, +zlin: U32, +tm: U32, +ii: U32) -> Lg: blk.ib(blk.i1(lin, gr, nch, zlin, tm, ii), nch, zlin, tm, ii)# (z+896) - z, times 29def sp.a1a3(g1: La, +v896: F32) -> La: match g1: case La{lin2, vz}: La{lin2, f.mul(f.sub(v896, vz), Enc.f32.bits(1105723392))}def sp.a1a2(g0: La, +zz: U32) -> La: match g0: case La{lin1, v896}: sp.a1a3(ln.at(lin1, zz), v896)def sp.a1a(lin: Array<F32>, +zz: U32) -> La: sp.a1a2(ln.at(lin, u.add(zz, 896)), zz)# slots 1 and 13, times 213def sp.a1b2(g1: La, +acc: F32, +v64: F32) -> La: match g1: case La{lin2, v832}: La{lin2, f.add(acc, f.mul(f.add(v64, v832), Enc.f32.bits(1129644032)))}def sp.a1b1(g0: La, +acc: F32, +zz: U32) -> La: match g0: case La{lin1, v64}: sp.a1b2(ln.at(lin1, u.add(zz, 832)), acc, v64)def sp.a1b(+acc: F32, lin: Array<F32>, +zz: U32) -> La: sp.a1b1(ln.at(lin, u.add(zz, 64)), acc, zz)# slots 12 and 2, times 459def sp.a1c2(g1: La, +acc: F32, +v768: F32) -> La: match g1: case La{lin2, v128}: La{lin2, f.add(acc, f.mul(f.sub(v768, v128), Enc.f32.bits(1139113984)))}def sp.a1c1(g0: La, +acc: F32, +zz: U32) -> La: match g0: case La{lin1, v768}: sp.a1c2(ln.at(lin1, u.add(zz, 128)), acc, v768)def sp.a1c(+acc: F32, lin: Array<F32>, +zz: U32) -> La: sp.a1c1(ln.at(lin, u.add(zz, 768)), acc, zz)# slots 3 and 11, times 2037def sp.a1d2(g1: La, +acc: F32, +v192: F32) -> La: match g1: case La{lin2, v704}: La{lin2, f.add(acc, f.mul(f.add(v192, v704), Enc.f32.bits(1157537792)))}def sp.a1d1(g0: La, +acc: F32, +zz: U32) -> La: match g0: case La{lin1, v192}: sp.a1d2(ln.at(lin1, u.add(zz, 704)), acc, v192)def sp.a1d(+acc: F32, lin: Array<F32>, +zz: U32) -> La: sp.a1d1(ln.at(lin, u.add(zz, 192)), acc, zz)# slots 10 and 4, times 5153def sp.a1e2(g1: La, +acc: F32, +v640: F32) -> La: match g1: case La{lin2, v256}: La{lin2, f.add(acc, f.mul(f.sub(v640, v256), Enc.f32.bits(1168181248)))}def sp.a1e1(g0: La, +acc: F32, +zz: U32) -> La: match g0: case La{lin1, v640}: sp.a1e2(ln.at(lin1, u.add(zz, 256)), acc, v640)def sp.a1e(+acc: F32, lin: Array<F32>, +zz: U32) -> La: sp.a1e1(ln.at(lin, u.add(zz, 640)), acc, zz)# slots 5 and 9, times 6574def sp.a1f2(g1: La, +acc: F32, +v320: F32) -> La: match g1: case La{lin2, v576}: La{lin2, f.add(acc, f.mul(f.add(v320, v576), Enc.f32.bits(1171091456)))}def sp.a1f1(g0: La, +acc: F32, +zz: U32) -> La: match g0: case La{lin1, v320}: sp.a1f2(ln.at(lin1, u.add(zz, 576)), acc, v320)def sp.a1f(+acc: F32, lin: Array<F32>, +zz: U32) -> La: sp.a1f1(ln.at(lin, u.add(zz, 320)), acc, zz)# slots 8 and 6, times 37489, then slot 7 times 75038def sp.a1g4(g2: La, +acc: F32, +v512: F32, +v384: F32) -> La: match g2: case La{lin3, v448}: La{lin3, f.add(f.add(acc, f.mul(f.sub(v512, v384), Enc.f32.bits(1192390912))), f.mul(v448, Enc.f32.bits(1200787200)))}def sp.a1g3(g1: La, +acc: F32, +v512: F32, +zz: U32) -> La: match g1: case La{lin2, v384}: sp.a1g4(ln.at(lin2, u.add(zz, 448)), acc, v512, v384)def sp.a1g2(g0: La, +acc: F32, +zz: U32) -> La: match g0: case La{lin1, v512}: sp.a1g3(ln.at(lin1, u.add(zz, 384)), acc, v512, zz)def sp.a1g(+acc: F32, lin: Array<F32>, +zz: U32) -> La: sp.a1g2(ln.at(lin, u.add(zz, 512)), acc, zz)def sp.a1go6(g: La, +zz: U32) -> La: match g: case La{lin6, acc6}: sp.a1g(acc6, lin6, zz)def sp.a1go5(g: La, +zz: U32) -> La: match g: case La{lin5, acc5}: sp.a1go6(sp.a1f(acc5, lin5, zz), zz)def sp.a1go4(g: La, +zz: U32) -> La: match g: case La{lin4, acc4}: sp.a1go5(sp.a1e(acc4, lin4, zz), zz)def sp.a1go3(g: La, +zz: U32) -> La: match g: case La{lin3, acc3}: sp.a1go4(sp.a1d(acc3, lin3, zz), zz)def sp.a1go2(g: La, +zz: U32) -> La: match g: case La{lin2, acc2}: sp.a1go3(sp.a1c(acc2, lin2, zz), zz)def sp.a1go1(g: La, +zz: U32) -> La: match g: case La{lin1, acc1}: sp.a1go2(sp.a1b(acc1, lin1, zz), zz)# first grouped window sumdef sp.a1(lin: Array<F32>, +zz: U32) -> La: sp.a1go1(sp.a1a(lin, zz), zz)# taps 898, 770, 642, 514def sp.a2a4(g3: La, +v898: F32, +v770: F32, +v642: F32) -> La: match g3: case La{lin4, v514}: La{lin4, f.add(f.add(f.add(f.mul(v898, Enc.f32.bits(1120927744)), f.mul(v770, Enc.f32.bits(1153687552))), f.mul(v642, Enc.f32.bits(1175976960))), f.mul(v514, Enc.f32.bits(1199182592)))}def sp.a2a3(g2: La, +v898: F32, +v770: F32, +zz: U32) -> La: match g2: case La{lin3, v642}: sp.a2a4(ln.at(lin3, u.add(zz, 514)), v898, v770, v642)def sp.a2a2(g1: La, +v898: F32, +zz: U32) -> La: match g1: case La{lin2, v770}: sp.a2a3(ln.at(lin2, u.add(zz, 642)), v898, v770, zz)def sp.a2a1(g0: La, +zz: U32) -> La: match g0: case La{lin1, v898}: sp.a2a2(ln.at(lin1, u.add(zz, 770)), v898, zz)def sp.a2a(lin: Array<F32>, +zz: U32) -> La: sp.a2a1(ln.at(lin, u.add(zz, 898)), zz)# taps 386 and 258def sp.a2b2(g1: La, +acc: F32, +v386: F32) -> La: match g1: case La{lin2, v258}: La{lin2, f.add(f.add(acc, f.mul(v386, Enc.f32.bits(3323714560))), f.mul(v258, Enc.f32.bits(3258187776)))}def sp.a2b1(g0: La, +acc: F32, +zz: U32) -> La: match g0: case La{lin1, v386}: sp.a2b2(ln.at(lin1, u.add(zz, 258)), acc, v386)def sp.a2b(+acc: F32, lin: Array<F32>, +zz: U32) -> La: sp.a2b1(ln.at(lin, u.add(zz, 386)), acc, zz)# tap 130def sp.a2c1(g: La, +acc: F32) -> La: match g: case La{lin1, v130}: La{lin1, f.add(acc, f.mul(v130, Enc.f32.bits(1125253120)))}def sp.a2c(+acc: F32, lin: Array<F32>, +zz: U32) -> La: sp.a2c1(ln.at(lin, u.add(zz, 130)), acc)# tap 2def sp.a2d1(g: La, +acc: F32) -> La: match g: case La{lin1, v2}: La{lin1, f.add(acc, f.mul(v2, Enc.f32.bits(3231711232)))}def sp.a2d(+acc: F32, lin: Array<F32>, +zz: U32) -> La: sp.a2d1(ln.at(lin, u.add(zz, 2)), acc)def sp.a2go3(g: La, +zz: U32) -> La: match g: case La{lin3, acc3}: sp.a2d(acc3, lin3, zz)def sp.a2go2(g: La, +zz: U32) -> La: match g: case La{lin2, acc2}: sp.a2go3(sp.a2c(acc2, lin2, zz), zz)def sp.a2go1(g: La, +zz: U32) -> La: match g: case La{lin1, acc1}: sp.a2go2(sp.a2b(acc1, lin1, zz), zz)# second window sum, eight weighted tapsdef sp.a2(lin: Array<F32>, +zz: U32) -> La: sp.a2go1(sp.a2a(lin, zz), zz)# PCM and delay line while pairing synth samplestype Pp is Type: Pp{pcm: Array<F32>, lin: Array<F32>}# one synth pair: two scaled samplesdef sp.pair2(g2: La, pcm: Array<F32>, +pcmi: U32, +nch: U32) -> Pp: match g2: case La{lin2, v2}: Pp{ln.set(pcm, u.add(pcmi, u.mul(16, nch)), mac.sc(v2)), lin2}def sp.pair1(g1: La, pcm: Array<F32>, +pcmi: U32, +nch: U32, +zz: U32) -> Pp: match g1: case La{lin1, v1}: sp.pair2(sp.a2(lin1, zz), ln.set(pcm, pcmi, mac.sc(v1)), pcmi, nch)def sp.pair(st: Pp, +pcmi: U32, +nch: U32, +zz: U32) -> Pp: match st: case Pp{pcm, lin}: sp.pair1(sp.a1(lin, zz), pcm, pcmi, nch, zz)# four synth pairs that open a blockdef blk.pairs(st: Pp, +nch: U32, +pcmi: U32, +z0: U32) -> Pp: sp.pair(sp.pair(sp.pair(sp.pair(st, u.add(pcmi, u.sub(nch, 1)), nch, u.add(z0, 1)), u.add(u.add(pcmi, u.mul(32, nch)), u.sub(nch, 1)), nch, u.add(z0, 65)), pcmi, nch, z0), u.add(pcmi, u.mul(32, nch)), nch, u.add(z0, 64))# delay line and the PCM written so fartype Sb is Type: Sb{lin: Array<F32>, pcm: Array<F32>}def blk.armed1(got: Pp) -> Sb: match got: case Pp{pcm2, lin2}: Sb{lin2, pcm2}# pairs read the delay line the preamble just wrotedef blk.armed(lin: Array<F32>, pcm: Array<F32>, +nch: U32, +tm: U32) -> Sb: blk.armed1(blk.pairs(Pp{pcm, lin}, nch, u.mul(u.mul(32, nch), tm), u.add(u.mul(tm, 64), 60)))# polyphase state, the granule Array the rows still read, and the window tabletype Bg is Type: Bg{st: Sb, gr: Array<F32>, syn: Array<F32>}# preamble done: the pairs read the delay line, the granule and window stay putdef blk.arm2(got: Lg, pcm: Array<F32>, syn: Array<F32>, +nch: U32, +tm: U32) -> Bg: match got: case Lg{lin, gr}: Bg{blk.armed(lin, pcm, nch, tm), gr, syn}# open a block: preamble, then the four pairsdef blk.arm(got: Bg, +nch: U32, +tm: U32) -> Bg: match got: case Bg{st, gr, syn}: match st: case Sb{lin, pcm}: blk.arm2(blk.pre(lin, gr, nch, u.add(u.mul(tm, 64), 960), tm), pcm, syn, nch, tm)# a finished row, with the window table the next row still readstype Sr is Type: Sr{st: Sb, syn: Array<F32>}def blk.row1(got: Mw, pcm: Array<F32>, +nch: U32, +tm: U32, +ii: U32) -> Sr: match got: case Mw{mk, syn}: match mk: case Mk{st, lin2}: Sr{Sb{lin2, mac.put(pcm, st, nch, u.mul(u.mul(32, nch), tm), ii)}, syn}def blk.row3(got: Sr, gr: Array<F32>) -> Bg: match got: case Sr{st, syn}: Bg{st, gr, syn}# window the row just writtendef blk.row( lin: Array<F32>, pcm: Array<F32>, syn: Array<F32>, +nch: U32, +tm: U32, +ii: U32) -> Sr: blk.row1(mac.k(8n, Mw{Mk{mac.z(), lin}, syn}, u.add(u.mul(tm, 64), 960), ii, 0), pcm, nch, tm, ii)# window the row, keeping the granule for the rows still to comedef blk.step2( got: Lg, pcm: Array<F32>, syn: Array<F32>, +nch: U32, +tm: U32, +ii: U32) -> Bg: match got: case Lg{lin, gr}: blk.row3(blk.row(lin, pcm, syn, nch, tm, ii), gr)# one row of the block, from i = 14 down to 0def blk.step(got: Bg, +nch: U32, +tm: U32, +ii: U32) -> Bg: match got: case Bg{st, gr, syn}: match st: case Sb{lin, pcm}: blk.step2(blk.i(lin, gr, nch, u.add(u.mul(tm, 64), 960), tm, ii), pcm, syn, nch, tm, ii)# fifteen rowsdef blk.i.go(hop: Nat, got: Bg, +nch: U32, +tm: U32, +ii: U32) -> Bg: match hop got: case 0n Bg{st, gr, syn}: Bg{st, gr, syn} case 1n+p Bg{st, gr, syn}: blk.i.go(p, blk.step(Bg{st, gr, syn}, nch, tm, ii), nch, tm, u.sub(ii, 1))# one time blockdef blk.one(got: Bg, +nch: U32, +tm: U32) -> Bg: blk.i.go(15n, blk.arm(got, nch, tm), nch, tm, 14)# nine time blocks, time = 0, 2, ..., 16def blk.go(hop: Nat, got: Bg, +nch: U32, +tm: U32) -> Bg: match hop got: case 0n Bg{st, gr, syn}: Bg{st, gr, syn} case 1n+p Bg{st, gr, syn}: blk.go(p, blk.one(Bg{st, gr, syn}, nch, tm), nch, u.add(tm, 2))# one even delay sample and the odd slot kept from the carried tail.# hop is the pairs still to come after this one. The sample was read at ii.def qmf.ev(hop: Nat, old: List<&2, F32>, got: La, +ii: U32, acc: List<&2, F32>) -> List<&2, F32>: match hop old got: case 0n +_e <> +odd <> _tl La{lin, +cur}: match lin: case _lin: List.reverse(&2, F32, odd <> (cur <> acc)) case 1n+p +_e <> +odd <> tl La{lin, +cur}: qmf.ev(p, tl, ln.at(lin, u.add(ii, 2)), u.add(ii, 2), odd <> (cur <> acc)) case _ _ La{lin, _v}: match lin: case _lin: List.reverse(&2, F32, acc)# stereo tail: 960 delay-Array samples in one pull. The old list is dropped.def qmf.st(+old: List<&2, F32>, lin: Array<F32>) -> List<&2, F32>: match old: case _old: ln.pull(959n, ln.at(lin, 1152), 1152, [])# mono keeps odd slots; stereo replaces the whole tail from the delay Arraydef qmf.pick(mono: Bool, +qmf: List<&2, F32>, lin: Array<F32>) -> List<&2, F32>: match mono: case True{}: qmf.ev(479n, qmf, ln.at(lin, 1152), 1152, []) case False{}: qmf.st(qmf, lin)# save the 15-block tail off the delay Array, one pass, no List.setdef qmf.save(lin: Array<F32>, +qmf: List<&2, F32>, +nch: U32) -> List<&2, F32>: qmf.pick(U32.is_eq(nch, 1), qmf, lin)# PCM, the overlap to carry, and the polyphase delaytype Pc is Data: Pc{pcm: List<&2, F32>, ov: List<&2, F32>, qmf: List<&2, F32>}# zero overlap, 288 samples, one channeldef syn.zov() -> List<&2, F32>: List.replicate(F32, 288n, 0.0)# zero polyphase delaydef syn.zqmf() -> List<&2, F32>: List.replicate(F32, 960n, 0.0)# antialias, IMDCT, and the sign flip for one channeldef ch.prep( acc: Array<F32>, +ov: List<&2, F32>, +aa: List<&2, F32>, +tw: List<&2, F32>, +win: List<&2, F32>, +base: U32) -> Im: im.sign(im.go(32n, Im{aa.run(acc, aa, base), ov}, tw, win, 0, base), base)# append the second channel's overlap. The spectrum Array already holds both.def syn.st2(+ova: List<&2, F32>, bb: Im) -> Im: match bb: case Im{buf, ovb}: Im{buf, List.append(&2, F32, ova, ovb)}# join two channel states. The right channel starts at bin 576.def syn.st( aa: Im, +ovb: List<&2, F32>, +cs: List<&2, F32>, +tw: List<&2, F32>, +win: List<&2, F32>) -> Im: match aa: case Im{buf, ova}: syn.st2(ova, ch.prep(buf, ovb, cs, tw, win, 576))# prepare one or two channelsdef syn.ch( mono: Bool, acc: Array<F32>, +ov: List<&2, F32>, +aa: List<&2, F32>, +tw: List<&2, F32>, +win: List<&2, F32>) -> Im: match mono: case True{}: ch.prep(acc, ov, aa, tw, win, 0) case False{}: syn.st(ch.prep(acc, ls.slice(ov, 0, 288n), aa, tw, win, 0), ls.slice(ov, 288, 288n), aa, tw, win)# tables shared by a granule. 2048 slots cover mono 576 and stereo 1152.def syn.ready(+spec: List<&2, F32>, +ov: List<&2, F32>, +nch: U32) -> Im: syn.ch(U32.is_eq(nch, 1), ln.fill(spec, Array.new(F32, 11n, 0.0), 0), ov, Enc.f.of(Tab.tab.aa()), Enc.f.of(Tab.tab.tw()), Enc.f.of(Tab.tab.winl()))# second channel starts at bin 576def sy.dct2(mono: Bool, acc: Array<F32>, +sec: List<&2, F32>) -> Array<F32>: match mono: case True{}: dc.run(18n, acc, sec, 0) case False{}: dc.run(18n, dc.run(18n, acc, sec, 0), sec, 576)# DCT-II of each channel. The spectrum Array is the one IMDCT just wrote.def sy.dct(mono: Bool, acc: Array<F32>, +sec: List<&2, F32>) -> Array<F32>: sy.dct2(mono, acc, sec)# delay line: the carried tail, then zeros (4096-slot Array starts at 0.0)def sy.lin(+qmf: List<&2, F32>) -> Array<F32>: ln.fill(qmf, Array.new(F32, 12n, 0.0), 0)# empty PCM for one granule (2048-slot Array covers mono 576 and stereo 1152)def sy.pcm() -> Array<F32>: Array.new(F32, 11n, 0.0)# PCM Array back to a list of length 576*nchdef sy.pcm.list(+nch: U32, pcm: Array<F32>) -> List<&2, F32>: ln.list(pcm, U32.to_nat(u.sub(u.mul(576, nch), 1)))# the granule Array is finished once the nine blocks have read itdef syn.drop(gr: Array<F32>, st: Sb, +ov: List<&2, F32>, +qmf: List<&2, F32>, +nch: U32) -> Pc: match gr: case _gr: match st: case Sb{lin, pcm}: Pc{sy.pcm.list(nch, pcm), ov, qmf.save(lin, qmf, nch)}# run the polyphase and keep the overlap from the IMDCTdef syn.out(got: Bg, +ov: List<&2, F32>, +qmf: List<&2, F32>, +nch: U32) -> Pc: match got: case Bg{st, gr, syn}: match syn: case _syn: syn.drop(gr, st, ov, qmf, nch)# DCT-II then nine polyphase blocks. The spectrum Array is what the polyphase reads.def syn.pcm(buf: Array<F32>, +ov: List<&2, F32>, +qmf: List<&2, F32>, +nch: U32) -> Pc: syn.out(blk.go(9n, Bg{Sb{sy.lin(qmf), sy.pcm()}, sy.dct(U32.is_eq(nch, 1), buf, Enc.f.of(Tab.tab.sec())), sw.tab()}, nch, 0), ov, qmf, nch)# synthesis of a prepared granuledef syn.finish(st: Im, +qmf: List<&2, F32>, +nch: U32) -> Pc: match st: case Im{buf, ov}: syn.pcm(buf, ov, qmf, nch)# one granule to PCM. spec is 576 samples per channel. ov is 288 per channel.def syn.gran(+spec: List<&2, F32>, +ov: List<&2, F32>, +qmf: List<&2, F32>, +nch: U32) -> Pc: syn.finish(syn.ready(spec, ov, nch), qmf, nch)# IMDCT only, 32 bands, no antialias. Used to check the unit impulse.def syn.im(+buf: List<&2, F32>, +ov: List<&2, F32>) -> Im: im.go(32n, Im{ln.fill(buf, Array.new(F32, 11n, 0.0), 0), ov}, Enc.f.of(Tab.tab.tw()), Enc.f.of(Tab.tab.winl()), 0, 0)# one PCM sample of a granule, as a binary32 worddef syn.nth(st: Pc, +ii: U32) -> U32: match st: case Pc{pcm, _ov, _qmf}: Enc.f32.word(f.at(pcm, ii))# drop the spectrum Array once sample 0 has been readdef syn.word1(got: Dx) -> U32: match got: case Dx{acc, v}: match acc: case _acc: Enc.f32.word(v)# first sample of an IMDCT buffer, as a binary32 worddef syn.word(st: Im) -> U32: match st: case Im{buf, _ov}: syn.word1(dx.at(buf, 0))# PCM word ii of one granule whose spectrum is a unit impulse at bin 0.def syn.probe(+ii: U32) -> U32: syn.nth(syn.gran(f.set(List.replicate(F32, 576n, 0.0), 0, Enc.f32.bits(1065353216)), syn.zov(), syn.zqmf(), 1), ii)# unit impulse at bin 0. The first sample is the float32 word 1022453915.def syn.impulse() -> U32: syn.word(syn.im(f.set(List.replicate(F32, 576n, 0.0), 0, Enc.f32.bits(1065353216)), syn.zov()))