# Intake normalize — strip bidi/ZWSP/BOM, fullwidth punct, dash variants, # abbr periods before space (Rom. 8 → Rom 8), collapse space import Base def char1(+c: Char) -> String: String.from_list(c <> Nil{}) def u32_in(+x: U32, +lo: U32, +hi: U32) -> Bool: Bool.and(U32.is_le(lo, x), U32.is_le(x, hi)) # Invisible formatting: U+061C, U+200B-200D, U+200E-200F, U+202A-202E, U+2066-2069, U+FEFF def is_invisible_code(+code: U32) -> Bool: Bool.or( U32.is_eq(code, 1564), Bool.or( u32_in(code, 8203, 8207), Bool.or( u32_in(code, 8234, 8238), Bool.or(u32_in(code, 8294, 8297), U32.is_eq(code, 65279)) ) ) ) def is_invisible(+c: Char) -> Bool: match c: case Chr{+code}: is_invisible_code(code) # Fullwidth :./ → :./ # Cannot nest match on computed — use phase type FwPh is Data: FwCheck{code: U32} FwColon{yes: Bool, code: U32} FwDot{yes: Bool, code: U32} FwSlash{yes: Bool, code: U32} def map_fw_go(fuel: Nat, ph: FwPh) -> Char: match fuel: case 0n: Chr{63} case 1n+p: match ph: case FwCheck{+code}: map_fw_go(p, FwColon{U32.is_eq(code, 65306), code}) case FwColon{yes, +code}: match yes: case True{}: Chr{58} case False{}: map_fw_go(p, FwDot{U32.is_eq(code, 65294), code}) case FwDot{yes, +code}: match yes: case True{}: Chr{46} case False{}: map_fw_go(p, FwSlash{U32.is_eq(code, 65295), code}) case FwSlash{yes, +code}: match yes: case True{}: Chr{47} case False{}: Chr{code} def map_fullwidth(+c: Char) -> Char: match c: case Chr{+code}: map_fw_go(4n, FwCheck{code}) # Dash variants → ASCII hyphen-minus (45) # ‐‑‒–—―−﹘﹣- = 8208-8213, 8722, 65112, 65123, 65293 def is_dash_code(+code: U32) -> Bool: Bool.or( u32_in(code, 8208, 8213), Bool.or( U32.is_eq(code, 8722), Bool.or( U32.is_eq(code, 65112), Bool.or(U32.is_eq(code, 65123), U32.is_eq(code, 65293)) ) ) ) def map_dash_go(+c: Char, is_d: Bool) -> Char: match is_d: case True{}: Chr{45} case False{}: map_fullwidth(c) def map_dash(+c: Char) -> Char: match c: case Chr{+code}: map_dash_go(c, is_dash_code(code)) type NmSt is Data: NmCheck{} NmDecide{skip: Bool, space: Bool, prev_space: Bool} NmEmitSpace{} NmMaybeDot{is_dot: Bool} NmDotPeek{} NmDotDecide{next_space: Bool} NmEmitChar{} def normalize_go(fuel: Nat, st: NmSt, rest: String, acc: String, +prev_space: Bool) -> String: match fuel: case 0n: acc case 1n+p: match st: case NmCheck{}: match rest: case SNil{}: acc case SCon{+h, +t}: normalize_go( p, NmDecide{is_invisible(h), Char.is_space(h), prev_space}, rest, acc, prev_space ) case NmDecide{skip, space, +ps}: match skip: case True{}: match rest: case SNil{}: acc case SCon{h, t}: normalize_go(p, NmCheck{}, t, acc, ps) case False{}: match space: case True{}: match ps: case True{}: match rest: case SNil{}: acc case SCon{h, t}: normalize_go(p, NmCheck{}, t, acc, True{}) case False{}: normalize_go(p, NmEmitSpace{}, rest, acc, True{}) case False{}: match rest: case SNil{}: acc case SCon{+h, t}: normalize_go(p, NmMaybeDot{Char.is_eq(h, '.')}, rest, acc, ps) case NmEmitSpace{}: match rest: case SNil{}: acc case SCon{h, t}: normalize_go(p, NmCheck{}, t, String.append(acc, " "), True{}) case NmMaybeDot{is_dot}: match is_dot: case True{}: normalize_go(p, NmDotPeek{}, rest, acc, prev_space) case False{}: normalize_go(p, NmEmitChar{}, rest, acc, False{}) case NmDotPeek{}: match rest: case SNil{}: normalize_go(p, NmEmitChar{}, rest, acc, False{}) case SCon{h, t}: match t: case SNil{}: normalize_go(p, NmEmitChar{}, rest, acc, False{}) case SCon{+h2, t2}: normalize_go(p, NmDotDecide{Char.is_space(h2)}, rest, acc, prev_space) case NmDotDecide{next_space}: match next_space: case True{}: match rest: case SNil{}: acc case SCon{h, t}: normalize_go(p, NmCheck{}, t, acc, prev_space) case False{}: normalize_go(p, NmEmitChar{}, rest, acc, False{}) case NmEmitChar{}: match rest: case SNil{}: acc case SCon{+h, +t}: normalize_go(p, NmCheck{}, t, String.append(acc, char1(map_dash(h))), False{}) def normalize_intake(+input: String) -> String: String.trim( normalize_go( Nat.add(Nat.mul(6n, String.length(input)), 8n), NmCheck{}, input, "", False{} ) )