encoding.bend source
encoding.bend on the hub · documented module
# Hex of each char as two nibbles.import Base# Hex of each char as two nibbles. ASCII-oriented.def nibble(+n: U32) -> U32: Bool.pick(U32, U32.is_le(n, 9), (48 + n : U32), (87 + n : U32))@unsafedef hi.go(+n: U32, +acc: U32, small: Bool) -> U32: match small: case True{}: acc case False{}: hi.go((n - 16 : U32), (acc + 1 : U32), U32.is_le((n - 16 : U32), 15))def hi(+n: U32) -> U32: hi.go(n, 0, U32.is_le(n, 15))def lo(+n: U32) -> U32: (n - (hi(n) * 16 : U32) : U32)def encode.go(s: String, acc: String) -> String: match s: case SNil{}: String.reverse(acc) case SCon{Chr{+c}, t}: encode.go(t, SCon{Chr{nibble(lo(c))}, SCon{Chr{nibble(hi(c))}, acc}})def encode(s: String) -> String: encode.go(s, SNil{})def unhex.h(c: U32, hex: Bool) -> Maybe<&2, U32>: match hex: case True{}: Some{(c - 87 : U32)} case False{}: None{}def unhex.d(+c: U32, dec: Bool, hex: Bool) -> Maybe<&2, U32>: match dec: case True{}: Some{(c - 48 : U32)} case False{}: unhex.h(c, hex)def unhex(+c: U32) -> Maybe<&2, U32>: unhex.d(c, Bool.and(U32.is_le(48, c), U32.is_le(c, 57)), Bool.and(U32.is_le(97, c), U32.is_le(c, 102)))def decode.cons(byte: U32, rest: Maybe<&2, String>) -> Maybe<&2, String>: match rest: case None{}: None{} case Some{s}: Some{SCon{Chr{byte}, s}}def decode.join2(x: U32, b: Maybe<&2, U32>, rest: Maybe<&2, String>) -> Maybe<&2, String>: match b: case None{}: None{} case Some{y}: decode.cons((x * 16 + y : U32), rest)def decode.join(a: Maybe<&2, U32>, b: Maybe<&2, U32>, rest: Maybe<&2, String>) -> Maybe<&2, String>: match a: case None{}: None{} case Some{x}: decode.join2(x, b, rest)@unsafedef decode.go(s: String) -> Maybe<&2, String>: match s: case SNil{}: Some{SNil{}} case SCon{Chr{+c}, t}: match t: case SNil{}: None{} case SCon{Chr{+d}, r}: decode.join(unhex(c), unhex(d), decode.go(r))def decode(s: String) -> Maybe<&2, String>: decode.go(s)# UTF-8 (RFC 3629). Byte strings hold one Char per octet (0..255).def utf8.cont(c: U32, n: Nat) -> U32: (128 + U32.and(U32.shrn(c, n), 63) : U32)def utf8.push4(+c: U32, acc: String) -> String: SCon{Chr{utf8.cont(c, 0n)}, SCon{Chr{utf8.cont(c, 6n)}, SCon{Chr{utf8.cont(c, 12n)}, SCon{Chr{(240 + U32.shrn(c, 18n) : U32)}, acc}}}}def utf8.push3(+c: U32, acc: String, small: Bool) -> String: match small: case True{}: SCon{Chr{utf8.cont(c, 0n)}, SCon{Chr{utf8.cont(c, 6n)}, SCon{Chr{(224 + U32.shrn(c, 12n) : U32)}, acc}}} case False{}: utf8.push4(c, acc)def utf8.push2(+c: U32, acc: String, small: Bool) -> String: match small: case True{}: SCon{Chr{utf8.cont(c, 0n)}, SCon{Chr{(192 + U32.shrn(c, 6n) : U32)}, acc}} case False{}: utf8.push3(c, acc, U32.is_lt(c, 65536))def utf8.push(+c: U32, acc: String, ascii: Bool) -> String: match ascii: case True{}: SCon{Chr{c}, acc} case False{}: utf8.push2(c, acc, U32.is_lt(c, 2048))def utf8.encode.go(s: String, acc: String) -> String: match s: case SNil{}: String.reverse(acc) case SCon{Chr{+c}, t}: utf8.encode.go(t, utf8.push(c, acc, U32.is_lt(c, 128)))# Text to octets.def utf8.encode(s: String) -> String: utf8.encode.go(s, SNil{})# Decoder state: continuation bytes still needed, the allowed range of the# next one, the code point so far, and the output reversed.type Utf8 is Data: Utf8{need: U32, lo: U32, hi: U32, cp: U32, out: String}def utf8.bad(out: String) -> String: SCon{Chr{65533}, out}def utf8.lead.multi(+b: U32, out: String) -> Utf8: +need = Bool.pick(U32, U32.is_lt(b, 224), 1, Bool.pick(U32, U32.is_lt(b, 240), 2, 3)) +lo = Bool.pick(U32, U32.is_eq(b, 224), 160, Bool.pick(U32, U32.is_eq(b, 240), 144, 128)) +hi = Bool.pick(U32, U32.is_eq(b, 237), 159, Bool.pick(U32, U32.is_eq(b, 244), 143, 191)) Utf8{need, lo, hi, U32.and(b, U32.shrn(63, U32.to_nat(need))), out}def utf8.lead.high(+b: U32, out: String, bad: Bool) -> Utf8: match bad: case True{}: Utf8{0, 128, 191, 0, utf8.bad(out)} case False{}: utf8.lead.multi(b, out)# WHATWG: a byte that cannot start a sequence decodes to U+FFFD.def utf8.lead(+b: U32, out: String, ascii: Bool) -> Utf8: match ascii: case True{}: Utf8{0, 128, 191, 0, SCon{Chr{b}, out}} case False{}: utf8.lead.high(b, out, Bool.or(U32.is_lt(b, 194), U32.is_lt(244, b)))def utf8.more(+need: U32, cp: U32, out: String, done: Bool) -> Utf8: match done: case True{}: Utf8{0, 128, 191, 0, SCon{Chr{cp}, out}} case False{}: Utf8{need, 128, 191, cp, out}# WHATWG: a byte that breaks a sequence yields U+FFFD and is read again as a lead.def utf8.cont.in(+need: U32, cp: U32, +b: U32, out: String, ok: Bool) -> Utf8: match ok: case True{}: utf8.more((need - 1 : U32), (cp * 64 + U32.and(b, 63) : U32), out, U32.is_eq(need, 1)) case False{}: utf8.lead(b, utf8.bad(out), U32.is_lt(b, 128))def utf8.step.if(+need: U32, +lo: U32, +hi: U32, cp: U32, +b: U32, out: String, idle: Bool) -> Utf8: match idle: case True{}: utf8.lead(b, out, U32.is_lt(b, 128)) case False{}: utf8.cont.in(need, cp, b, out, Bool.and(U32.is_le(lo, b), U32.is_le(b, hi)))def utf8.step.st(st: Utf8, +b: U32) -> Utf8: Utf8{+need, +lo, +hi, cp, out} = st utf8.step.if(need, lo, hi, cp, b, out, U32.is_eq(need, 0))def utf8.finish(st: Utf8) -> String: Utf8{+need, lo, hi, cp, +out} = st String.reverse(Bool.pick(String, U32.is_eq(need, 0), out, utf8.bad(out)))def utf8.decode.go(s: String, st: Utf8) -> String: match s: case SNil{}: utf8.finish(st) case SCon{Chr{b}, t}: utf8.decode.go(t, utf8.step.st(st, b))# Octets to text, as WHATWG decodes UTF-8: malformed input becomes U+FFFD.def utf8.decode(s: String) -> String: utf8.decode.go(s, Utf8{0, 128, 191, 0, SNil{}})