import Base import ./geom.bend as G import ./protein.bend as P # --- PDB reader v0: ATOM records -> Complex. --- # Fixed columns; ATOM lines before the first ENDMDL (HETATM and the rest are # dropped); consecutive records with equal (chain, resSeq) merge into one # Residue, consecutive same-chain residues into one Chain. Insertion codes # are ignored, so insertion variants of one residue merge. # Element symbols map to Z (unknown -> 0); a blank element field falls back # to the atom name's first letter (so CA-the-carbon needs its element field; # without one it reads as calcium). # Coordinates need digits on both sides of the point ("12." and ".5" fail); # mantissas must fit U32 (about 9 digits). A line that fails any field is # skipped and counted: parse_pdb answers the skip count beside the Complex. # # Structure note: Bend checks top-to-bottom with no forward references, and # branching on a computed Bool needs a callee, so every helper below only # calls defs above it. Grouping is branchless: both next-states are built # as data and picked (Bool.pick), with a single self-call on the tail. # Fixed-column slice: 0-based start, length. def pdb_slice(s: String, start: Nat, len: Nat) -> String: String.take(String.drop(s, start), len) def is_atom_line(line: String) -> Bool: String.starts_with(line, "ATOM ") # Tail-recursive lines(): single self-call in tail position, so long files # do not grow the machine stack (String.split recurses per character). def tlines_go(s: String, cur: String, acc: List<&2, String>) -> List<&2, String>: match s: case SNil{}: List.reverse(&2, String, String.reverse(cur) <> acc) case SCon{'\n', t}: tlines_go(t, SNil{}, String.reverse(cur) <> acc) case SCon{h, t}: tlines_go(t, SCon{h, cur}, acc) def tlines(s: String) -> List<&2, String>: tlines_go(s, SNil{}, Nil{}) # ATOM lines up to (excluding) the first ENDMDL. Decided by matching the # first six characters literally, so no computed Bool is ever matched. # Tail-recursive over lines for the same stack reason as tlines. def atom_lines_go(lines: List<&2, String>, acc: List<&2, String>) -> List<&2, String>: match lines: case Nil{}: List.reverse(&2, String, acc) case h <> t: match h: case SCon{'E', SCon{'N', SCon{'D', SCon{'M', SCon{'D', SCon{'L', _}}}}}}: List.reverse(&2, String, acc) case SCon{'A', SCon{'T', SCon{'O', SCon{'M', SCon{' ', SCon{' ', _}}}}}}: atom_lines_go(t, h <> acc) case _: atom_lines_go(t, acc) def atom_lines(lines: List<&2, String>) -> List<&2, String>: atom_lines_go(lines, Nil{}) def is_conect_line(line: String) -> Bool: String.starts_with(line, "CONECT") # CONECT lines anywhere in the file (they follow ENDMDL in practice). def conect_lines(lines: List<&2, String>) -> List<&2, String>: match lines: case Nil{}: Nil{} case h <> t: match h: case SCon{'C', SCon{'O', SCon{'N', SCon{'E', SCon{'C', SCon{'T', _}}}}}}: h <> conect_lines(t) case _: conect_lines(t) # Powers of ten for the fractional scale. def pow10(n: Nat) -> U32: match n: case 0n: 1 case 1n+p: U32.mul(10, pow10(p)) def f32_apply_sign(neg: Bool, v: F32) -> F32: match neg: case True{}: (0.0 - v : F32) case False{}: v def f32_combine(neg: Bool, iv: U32, fv: U32, sc: U32) -> F32: f32_apply_sign(neg, (U32.to_f32(iv) + (U32.to_f32(fv) / U32.to_f32(sc) : F32) : F32)) def parse_f32_frac(neg: Bool, iv: U32, f: Maybe<&2, U32>, sc: U32) -> Maybe<&2, F32>: match f: case None{}: None{} case Some{fv}: Some{f32_combine(neg, iv, fv, sc)} def parse_f32_num(neg: Bool, i: Maybe<&2, U32>, fp: String, flen: Nat) -> Maybe<&2, F32>: match i: case None{}: None{} case Some{iv}: parse_f32_frac(neg, iv, U32.read(fp), pow10(flen)) def parse_f32_int(neg: Bool, ip: String, +fp: String) -> Maybe<&2, F32>: parse_f32_num(neg, U32.read(ip), fp, String.length(fp)) def parse_f32_split(neg: Bool, parts: List<&2, String>) -> Maybe<&2, F32>: match parts: case Nil{}: None{} case ip <> Nil{}: parse_f32_int(neg, ip, "0") case ip <> fp <> Nil{}: parse_f32_int(neg, ip, fp) case ip <> fp <> rest: None{} def parse_f32_stripped(p: Bool & String) -> Maybe<&2, F32>: match p: case (neg, rest): parse_f32_split(neg, String.split(rest, '.')) def strip_sign_kept(plus: Bool, h: Char, t: String) -> Bool & String: match plus: case True{}: (False{}, t) case False{}: (False{}, SCon{h, t}) def strip_sign_plus(+h: Char, t: String) -> Bool & String: strip_sign_kept(Char.is_eq(h, '+'), h, t) def strip_sign_if(dash: Bool, h: Char, t: String) -> Bool & String: match dash: case True{}: (True{}, t) case False{}: strip_sign_plus(h, t) def strip_sign_go(+h: Char, t: String) -> Bool & String: strip_sign_if(Char.is_eq(h, '-'), h, t) def strip_sign(s: String) -> Bool & String: match s: case SNil{}: (False{}, SNil{}) case SCon{h, t}: strip_sign_go(h, t) # Decimal floats: optional sign, digits, optional '.', digits. def parse_f32(s: String) -> Maybe<&2, F32>: parse_f32_stripped(strip_sign(String.trim(s))) # Element symbol -> atomic number; unknown -> 0. def elem_1(h: Char) -> U32: match h: case 'H': 1 case 'B': 5 case 'C': 6 case 'N': 7 case 'O': 8 case 'F': 9 case 'P': 15 case 'S': 16 case 'K': 19 case 'V': 23 case 'I': 53 case _: 0 def elem_2(h: Char, h2: Char) -> U32: match h h2: case 'F' 'E': 26 case 'Z' 'N': 30 case 'C' 'A': 20 case 'M' 'G': 12 case 'M' 'N': 25 case 'N' 'A': 11 case 'C' 'L': 17 case 'C' 'U': 29 case _ _: 0 def elem_no_1(h: Char, t: String) -> U32: match t: case SNil{}: elem_1(h) case SCon{h2, t2}: elem_2(h, h2) def elem_no(s: String) -> U32: match s: case SNil{}: 0 case SCon{h, t}: elem_no_1(h, t) def elem_from_name(n: String) -> U32: match n: case SNil{}: 0 case SCon{h, t}: elem_1(Char.to_upper(h)) def elem_of_go(e: String, n: String) -> U32: match e: case SNil{}: elem_from_name(n) case SCon{h, t}: elem_no(String.to_upper(e)) def elem_of(raw_elem: String, raw_name: String) -> U32: elem_of_go(String.trim(raw_elem), String.trim(raw_name)) def chain_u32(s: String) -> U32: match s: case SNil{}: 32 case SCon{h, t}: Char.to_u32(h) # One ATOM line -> (chain, seq, atom). None{} when any field fails. # Threaded bottom-up: each helper matches one Maybe and passes the rest on. def parse_atom_z(serial: U32, ch: U32, seq: Nat, x: F32, y: F32, mz: Maybe<&2, F32>, elem_s: String, name_s: String) -> Maybe<&1, U32 & Nat & P.Atom>: match mz: case None{}: None{} case Some{z}: Some{(ch, seq, P.Atom{serial, elem_of(elem_s, name_s), G.V3{x, y, z}})} def parse_atom_y(serial: U32, ch: U32, seq: Nat, x: F32, my: Maybe<&2, F32>, zs: String, elem_s: String, name_s: String) -> Maybe<&1, U32 & Nat & P.Atom>: match my: case None{}: None{} case Some{y}: parse_atom_z(serial, ch, seq, x, y, parse_f32(String.trim(zs)), elem_s, name_s) def parse_atom_x(serial: U32, ch: U32, seq: Nat, mx: Maybe<&2, F32>, ys: String, zs: String, elem_s: String, name_s: String) -> Maybe<&1, U32 & Nat & P.Atom>: match mx: case None{}: None{} case Some{x}: parse_atom_y(serial, ch, seq, x, parse_f32(String.trim(ys)), zs, elem_s, name_s) def parse_atom_seq(serial: U32, ch: U32, ms: Maybe<&2, Nat>, xs: String, ys: String, zs: String, elem_s: String, name_s: String) -> Maybe<&1, U32 & Nat & P.Atom>: match ms: case None{}: None{} case Some{seq}: parse_atom_x(serial, ch, seq, parse_f32(String.trim(xs)), ys, zs, elem_s, name_s) def parse_atom_chain(serial: U32, chain_s: String, seq_s: String, xs: String, ys: String, zs: String, elem_s: String, name_s: String) -> Maybe<&1, U32 & Nat & P.Atom>: parse_atom_seq(serial, chain_u32(chain_s), Nat.read(String.trim(seq_s)), xs, ys, zs, elem_s, name_s) def parse_atom_serial(ms: Maybe<&2, U32>, chain_s: String, seq_s: String, xs: String, ys: String, zs: String, elem_s: String, name_s: String) -> Maybe<&1, U32 & Nat & P.Atom>: match ms: case None{}: None{} case Some{serial}: parse_atom_chain(serial, chain_s, seq_s, xs, ys, zs, elem_s, name_s) def parse_atom_fields(serial_s: String, chain_s: String, seq_s: String, xs: String, ys: String, zs: String, elem_s: String, name_s: String) -> Maybe<&1, U32 & Nat & P.Atom>: parse_atom_serial(U32.read(String.trim(serial_s)), chain_s, seq_s, xs, ys, zs, elem_s, name_s) def parse_atom_line(+line: String) -> Maybe<&1, U32 & Nat & P.Atom>: parse_atom_fields(pdb_slice(line, 6n, 5n), pdb_slice(line, 21n, 1n), pdb_slice(line, 22n, 4n), pdb_slice(line, 30n, 8n), pdb_slice(line, 38n, 8n), pdb_slice(line, 46n, 8n), pdb_slice(line, 76n, 2n), pdb_slice(line, 12n, 4n)) # One CONECT line -> [(s0, si)] bonds for each partner present. def conect_p4(s0: U32, m4: Maybe<&2, U32>) -> List<&1, U32 & U32>: match m4: case None{}: Nil{} case Some{s4}: [(s0, s4)] def conect_p3(+s0: U32, m3: Maybe<&2, U32>, m4: Maybe<&2, U32>) -> List<&1, U32 & U32>: match m3: case None{}: conect_p4(s0, m4) case Some{s3}: (s0, s3) <> conect_p4(s0, m4) def conect_p2(+s0: U32, m2: Maybe<&2, U32>, m3: Maybe<&2, U32>, m4: Maybe<&2, U32>) -> List<&1, U32 & U32>: match m2: case None{}: conect_p3(s0, m3, m4) case Some{s2}: (s0, s2) <> conect_p3(s0, m3, m4) def conect_p1(+s0: U32, m1: Maybe<&2, U32>, m2: Maybe<&2, U32>, m3: Maybe<&2, U32>, m4: Maybe<&2, U32>) -> List<&1, U32 & U32>: match m1: case None{}: conect_p2(s0, m2, m3, m4) case Some{s1}: (s0, s1) <> conect_p2(s0, m2, m3, m4) def conect_assemble(m0: Maybe<&2, U32>, m1: Maybe<&2, U32>, m2: Maybe<&2, U32>, m3: Maybe<&2, U32>, m4: Maybe<&2, U32>) -> List<&1, U32 & U32>: match m0: case None{}: Nil{} case Some{s0}: conect_p1(s0, m1, m2, m3, m4) def parse_conect_line(+line: String) -> List<&1, U32 & U32>: conect_assemble(U32.read(String.trim(pdb_slice(line, 6n, 5n))), U32.read(String.trim(pdb_slice(line, 11n, 5n))), U32.read(String.trim(pdb_slice(line, 16n, 5n))), U32.read(String.trim(pdb_slice(line, 21n, 5n))), U32.read(String.trim(pdb_slice(line, 26n, 5n)))) # Collect parsed atoms plus the skipped-line count. Tail-recursive: the # extend step builds both outcomes as data (no match, no cycle) and the # single self-call runs on the tail. def collect_extend(m: Maybe<&1, U32 & Nat & P.Atom>, st: List<&1, U32 & Nat & P.Atom> & Nat) -> List<&1, U32 & Nat & P.Atom> & Nat: match m: case None{}: match st: case (as, n): (as, 1n+n) case Some{x}: match st: case (as, n): (x <> as, n) def collect_finish(st: List<&1, U32 & Nat & P.Atom> & Nat) -> List<&1, U32 & Nat & P.Atom> & Nat: match st: case (as, n): (List.reverse(&1, U32 & Nat & P.Atom, as), n) def collect_go(lines: List<&2, String>, st: List<&1, U32 & Nat & P.Atom> & Nat) -> List<&1, U32 & Nat & P.Atom> & Nat: match lines: case Nil{}: collect_finish(st) case h <> t: collect_go(t, collect_extend(parse_atom_line(h), st)) def collect_atoms(lines: List<&2, String>) -> List<&1, U32 & Nat & P.Atom> & Nat: collect_go(lines, (Nil{}, 0n)) def append_conect(a: List<&1, U32 & U32>, b: List<&1, U32 & U32>) -> List<&1, U32 & U32>: match a: case Nil{}: b case h <> t: h <> append_conect(t, b) def conect_pairs(lines: List<&2, String>) -> List<&1, U32 & U32>: match lines: case Nil{}: Nil{} case h <> t: append_conect(parse_conect_line(h), conect_pairs(t)) # Pass 1: consecutive equal (chain, seq) merge. Chains and residues run in # parallel Data lists: a pair accumulator would not be reusable (+ needs # Data), so the two stay aligned and are zipped after (zip_chains). # Branchless: both next-states are built as data and picked, with one # self-call on the tail, so no helper cycle is needed. def group_res_go(items: List<&1, U32 & Nat & P.Atom>, +cur_c: U32, +cur_s: Nat, +cur_as: List<&2, P.Atom>, +acc_c: List<&2, U32>, +acc_r: List<&2, P.Residue>) -> List<&2, U32> & List<&2, P.Residue>: match items: case Nil{}: (List.reverse(&2, U32, cur_c <> acc_c), List.reverse(&2, P.Residue, P.Residue{cur_s, List.reverse(&2, P.Atom, cur_as)} <> acc_r)) case (c, s, a) <> t: +c2 = c +s2 = s +same = Bool.and(U32.is_eq(c2, cur_c), Nat.is_eq(s2, cur_s)) group_res_go(t, Bool.pick(U32, same, cur_c, c2), Bool.pick(Nat, same, cur_s, s2), a <> Bool.pick(List<&2, P.Atom>, same, cur_as, Nil{}), Bool.pick(List<&2, U32>, same, acc_c, cur_c <> acc_c), Bool.pick(List<&2, P.Residue>, same, acc_r, P.Residue{cur_s, List.reverse(&2, P.Atom, cur_as)} <> acc_r)) def group_residues(items: List<&1, U32 & Nat & P.Atom>) -> List<&2, U32> & List<&2, P.Residue>: match items: case Nil{}: (Nil{}, Nil{}) case (c, s, a) <> t: group_res_go(t, c, s, [a], Nil{}, Nil{}) # Zip the parallel chain/residue lists back into pairs (aligned by # construction; extra elements on either side are dropped). def zip_chains(cs: List<&2, U32>, rs: List<&2, P.Residue>) -> List<&1, U32 & P.Residue>: match cs rs: case Nil{} Nil{}: Nil{} case Nil{} _ <> _: Nil{} case _ <> _ Nil{}: Nil{} case c <> ct r <> rt: (c, r) <> zip_chains(ct, rt) # Pass 2: consecutive equal chains merge into Chain. def group_chain_go(items: List<&1, U32 & P.Residue>, +cur_c: U32, +cur_rs: List<&2, P.Residue>, +acc: List<&2, P.Chain>) -> List<&2, P.Chain>: match items: case Nil{}: List.reverse(&2, P.Chain, P.Chain{cur_c, List.reverse(&2, P.Residue, cur_rs)} <> acc) case (c, r) <> t: +c2 = c +same = U32.is_eq(c2, cur_c) group_chain_go(t, Bool.pick(U32, same, cur_c, c2), r <> Bool.pick(List<&2, P.Residue>, same, cur_rs, Nil{}), Bool.pick(List<&2, P.Chain>, same, acc, P.Chain{cur_c, List.reverse(&2, P.Residue, cur_rs)} <> acc)) def group_chains_zip(items: List<&1, U32 & P.Residue>) -> List<&2, P.Chain>: match items: case Nil{}: Nil{} case (c, r) <> t: group_chain_go(t, c, [r], Nil{}) def group_chains(cs: List<&2, U32>, rs: List<&2, P.Residue>) -> List<&2, P.Chain>: group_chains_zip(zip_chains(cs, rs)) # Whole file -> (Complex, skipped lines). def parse_pdb_group(g: List<&2, U32> & List<&2, P.Residue>, n: Nat) -> P.Complex & Nat: match g: case (cs, rs): (P.Complex{group_chains(cs, rs)}, n) def parse_pdb_collect(p: List<&1, U32 & Nat & P.Atom> & Nat) -> P.Complex & Nat: match p: case (items, n): parse_pdb_group(group_residues(items), n) def parse_pdb(text: String) -> P.Complex & Nat: parse_pdb_collect(collect_atoms(atom_lines(tlines(text)))) def complex_of(p: P.Complex & Nat) -> P.Complex: match p: case (cx, n): cx def skipped_of(p: P.Complex & Nat) -> Nat: match p: case (cx, n): n def first_elem_go(xs: List<&2, P.Atom>) -> Maybe<&2, U32>: match xs: case Nil{}: None{} case h <> t: Some{P.Atom.elem(h)} def first_elem(cx: P.Complex) -> Maybe<&2, U32>: first_elem_go(P.Complex.flatten(cx)) # Two-line ALA snippet for laws and smoke tests. def demo_text() -> String: "ATOM 1 N ALA A 1 11.104 13.207 2.100 1.00 13.79 N \nATOM 2 CA ALA A 1 12.104 13.207 2.100 1.00 10.80 C " # A bad serial ("X") so the second line skips. def demo_bad_text() -> String: "ATOM 1 N ALA A 1 11.104 13.207 2.100 1.00 13.79 N \nATOM X CA ALA A 1 12.104 13.207 2.100 1.00 10.80 C " def finish_read(fr: File & Result<&1, &1, U32 & String, String>) -> IO(P.Complex & Nat): match fr: case (fh2, res): do IO