← Files Biological Sequence & Alignment ViewerARCHIVED FILE

src/msa/parser.test.ts

17.6 KB · Sep 30, 2026 · 23:01 UTC

↓ Download file

import { describe, expect, it } from "vitest";

import {
  inferCdsContext,
  isNonsynonymousCodonDifference,
  translateStandardCodon,
} from "./codon";
import {
  computeNucleotideConsensus,
  computeProteinRepresentative,
} from "./consensus";
import {
  computeColumnSummaries,
  getMeanIdentity,
  getMeanNormalizedConservation,
} from "./conservation";
import { buildUngappedProjection, getUngappedPosition } from "./coordinate-map";
import { parseMsa } from "./parser";
import {
  classifyResidue,
  complementNucleotideSymbol,
  countUngappedResidues,
  reverseComplement,
} from "./residue-alphabet";
import { parseRnaPairs } from "./rna-structure";
import { searchMsaMotif } from "./search";
import type { MsaDocument } from "./types";

function expectDocument(contents: string, path: string): MsaDocument {
  const result = parseMsa(contents, path);
  expect(result.status).toBe("success");
  if (result.status !== "success") {
    throw new Error(result.message);
  }
  return result.document;
}

describe("parseMsa", () => {
  it("parses aligned FASTA DNA and preserves aligned rows", () => {
    const document = expectDocument(
      [">dna-a", "ACGT-R", ">dna-b", "ACGT-A"].join("\n"),
      "/tmp/example.afa",
    );
    expect(document.format).toBe("aligned-fasta");
    expect(document.molecule.moleculeType).toBe("dna");
    expect(document.rows).toHaveLength(2);
    expect(document.alignedLength).toBe(6);
    expect(document.rawSummary.gapFraction).toBeGreaterThan(0);
    expect(document.cdsContext.applicability).toBe("unknown");
  });

  it("parses CLUSTAL and retains consensus annotations", () => {
    const document = expectDocument(
      ["CLUSTAL W", "", "seq1    AC-G", "seq2    ACAG", "        ** *"].join(
        "\n",
      ),
      "/tmp/example.aln",
    );
    expect(document.format).toBe("clustal");
    expect(document.annotations[0]).toMatchObject({
      kind: "conservation",
      values: "** *",
    });
  });

  it("preserves leading blank CLUSTAL annotation columns", () => {
    const document = expectDocument(
      ["CLUSTAL W", "", "seq1    AC-G", "seq2    ACAG", "         * *"].join(
        "\n",
      ),
      "/tmp/leading-consensus.aln",
    );
    expect(document.annotations[0]?.values).toBe(" * *");
  });

  it("pads a stripped blank CLUSTAL consensus block when another block has consensus", () => {
    const document = expectDocument(
      [
        "CLUSTAL W",
        "",
        "seq1    AC",
        "seq2    AC",
        "",
        "seq1    GT",
        "seq2    GT",
        "        **",
      ].join("\n"),
      "/tmp/blank-consensus-block.aln",
    );

    expect(document.rows.map(({ alignedSequence }) => alignedSequence)).toEqual([
      "ACGT",
      "ACGT",
    ]);
    expect(document.annotations[0]).toMatchObject({
      kind: "conservation",
      values: "  **",
    });
  });

  it("treats FASTA-record `.aln` exports as aligned FASTA rather than forcing CLUSTAL parsing", () => {
    const document = expectDocument(
      [">alpha", "AC-G", ">beta", "ACAG"].join("\n"),
      "/tmp/exported-alignment.aln",
    );
    expect(document.format).toBe("aligned-fasta");
    expect(document.rows).toHaveLength(2);
    expect(document.alignedLength).toBe(4);
  });

  it("parses Stockholm RNA annotations into structure data", () => {
    const document = expectDocument(
      [
        "# STOCKHOLM 1.0",
        "rna1 AC-G",
        "rna2 AU-G",
        "#=GC SS_cons <<>>",
        "#=GC RF xxxx",
        "//",
      ].join("\n"),
      "/tmp/rfam.sto",
    );
    expect(document.molecule.moleculeType).toBe("rna");
    expect(document.rnaStructure?.pairs).toHaveLength(2);
    expect(document.rnaStructure?.referenceTrack).toBe("xxxx");
    expect(document.rawSummary.structureTrackCount).toBe(2);
  });

  it("retains Stockholm GF, GS, and GR metadata as inspectable document, row, and annotation data", () => {
    const document = expectDocument(
      [
        "# STOCKHOLM 1.0",
        "#=GF ID RF00001",
        "#=GS rna1 DE Example sequence",
        "rna1 AC",
        "rna2 AU",
        "#=GR rna1 SS <<",
        "#=GC SS_cons <>",
        "//",
      ].join("\n"),
      "/tmp/stockholm-metadata.sto",
    );
    expect(document.formatMetadata).toMatchObject({ "GF:ID": "RF00001" });
    expect(document.rows[0]?.metadata).toMatchObject({
      "GS:DE": "Example sequence",
    });
    expect(
      document.annotations.some(
        (track) =>
          track.id === "stockholm-gr-rna1-SS" && track.label === "rna1 SS",
      ),
    ).toBe(true);
  });

  it("parses A3M lowercase insertions without flattening them", () => {
    const document = expectDocument(
      [">query", "AC-DE", ">hit", "ACaa-DE"].join("\n"),
      "/tmp/profile.a3m",
    );
    expect(document.format).toBe("a3m");
    expect(document.molecule.moleculeType).toBe("protein");
    expect(document.rows.map((row) => row.alignedSequence)).toEqual([
      "AC-DE",
      "AC-DE",
    ]);
    expect(document.insertions).toEqual([
      {
        afterAlignmentColumn: 1,
        residues: "aa",
        rowId: "hit",
        sourceKind: "a2m-a3m-lowercase",
        sourceRowIndex: 1,
      },
    ]);
  });

  it("parses aligned A2M insertion columns without turning periods into deletions", () => {
    const document = expectDocument(
      [">query", "AC.-DEFGH", ">hit", "ACaD-E-GH"].join("\n"),
      "/tmp/profile.a2m",
    );
    expect(document.format).toBe("a2m");
    expect(document.rows.map((row) => row.alignedSequence)).toEqual([
      "AC-DEFGH",
      "ACD-E-GH",
    ]);
    expect(document.insertions).toEqual([
      {
        afterAlignmentColumn: 1,
        residues: "a",
        rowId: "hit",
        sourceKind: "a2m-a3m-lowercase",
        sourceRowIndex: 1,
      },
    ]);
  });

  it("rejects malformed A2M columns and A3M insertion-gap periods", () => {
    expect(
      parseMsa([">query", "AC.DE", ">hit", "AC-DE"].join("\n"), "/tmp/bad.a2m"),
    ).toMatchObject({
      message: expect.stringContaining("mixes insertion symbols"),
      status: "error",
    });
    expect(
      parseMsa([">query", "AC.DE", ">hit", "ACaDE"].join("\n"), "/tmp/bad.a3m"),
    ).toMatchObject({
      message: expect.stringContaining("A3M omits insertion-gap columns"),
      status: "error",
    });
  });

  it("rejects empty FASTA rows and content before the first FASTA header", () => {
    expect(parseMsa(">empty\n", "/tmp/empty.afa")).toMatchObject({
      status: "error",
      warnings: expect.arrayContaining([
        expect.objectContaining({ code: "fasta-record-empty" }),
      ]),
    });
    expect(parseMsa("ACGT\n>row\nACGT", "/tmp/prefix.afa")).toMatchObject({
      status: "error",
      warnings: expect.arrayContaining([
        expect.objectContaining({ code: "fasta-content-before-header" }),
      ]),
    });
  });

  it("parses MSF, PHYLIP, NEXUS, and PIR compatibility formats", () => {
    const msf = expectDocument(
      [
        "PileUp",
        " Name: a Len: 4",
        " MSF: 4 Type: N",
        "//",
        "a AC-G",
        "b ACAG",
      ].join("\n"),
      "/tmp/example.msf",
    );
    expect(msf.format).toBe("msf");
    expect(msf.rows).toHaveLength(2);

    const phylip = expectDocument(
      ["2 4", "alpha AC-G", "beta ACAG"].join("\n"),
      "/tmp/example.phy",
    );
    expect(phylip.format).toBe("phylip");

    const nexus = expectDocument(
      [
        "#NEXUS",
        "BEGIN DATA;",
        "MATRIX",
        "'alpha one' AC-G",
        "beta ACAG",
        ";",
        "END;",
      ].join("\n"),
      "/tmp/example.nex",
    );
    expect(nexus.format).toBe("nexus");
    expect(nexus.rows[0]?.label).toBe("alpha one");

    const pir = expectDocument(
      [">P1;alpha", "Alpha", "AC-G*", ">P1;beta", "Beta", "ACAG*"].join("\n"),
      "/tmp/example.pir",
    );
    expect(pir.format).toBe("pir");
    expect(pir.rows[1]?.description).toBe("Beta");
  });

  it("handles repeated-label PHYLIP continuations and interleaved NEXUS continuation blocks", () => {
    const phylip = expectDocument(
      ["2 6", "alpha ACG", "beta ACG", "alpha TTT", "beta GGG"].join("\n"),
      "/tmp/repeated.phy",
    );
    expect(phylip.rows.map((row) => row.alignedSequence)).toEqual([
      "ACGTTT",
      "ACGGGG",
    ]);
    expect(phylip.formatMetadata).toMatchObject({ columns: "6", rows: "2" });

    const nexus = expectDocument(
      [
        "#NEXUS",
        "BEGIN DATA;",
        "FORMAT DATATYPE=DNA MISSING=? GAP=- INTERLEAVE=YES;",
        "MATRIX",
        "alpha AC",
        "beta A-",
        "GT",
        "G?",
        ";",
        "END;",
      ].join("\n"),
      "/tmp/interleaved.nex",
    );
    expect(nexus.rows.map((row) => row.alignedSequence)).toEqual([
      "ACGT",
      "A-G?",
    ]);
    expect(nexus.formatMetadata).toMatchObject({
      gap: "-",
      interleaved: "true",
      missing: "?",
    });
  });

  it("parses sequential PHYLIP and fixed-width names containing spaces", () => {
    const document = expectDocument(
      [
        "2 12",
        "H. SapiensACGTAC",
        "300 GTACGT",
        "Salmo gairACGTAC",
        "300 AAAAAA",
      ].join("\n"),
      "/tmp/sequential.phy",
    );

    expect(document.formatMetadata).toMatchObject({
      layout: "sequential",
      nameMode: "fixed",
    });
    expect(document.rows.map(({ label }) => label)).toEqual([
      "H. Sapiens",
      "Salmo gair",
    ]);
    expect(document.rows.map(({ alignedSequence }) => alignedSequence)).toEqual(
      ["ACGTACGTACGT", "ACGTACAAAAAA"],
    );
  });

  it("rejects declared PHYLIP, Stockholm, and NEXUS dimension mismatches", () => {
    expect(
      parseMsa("2 5\nalpha ACGT\nbeta ACGT", "/tmp/bad.phy"),
    ).toMatchObject({
      status: "error",
      warnings: expect.arrayContaining([
        expect.objectContaining({ code: "phylip-column-count-mismatch" }),
      ]),
    });
    expect(
      parseMsa(
        "# STOCKHOLM 1.0\na ACGT\nb ACGT\n#=GR a SS <<<\n//",
        "/tmp/bad.sto",
      ),
    ).toMatchObject({
      status: "error",
      warnings: expect.arrayContaining([
        expect.objectContaining({ code: "stockholm-gr-width-mismatch" }),
      ]),
    });
    expect(
      parseMsa(
        "#NEXUS\nBEGIN DATA;\nDIMENSIONS NTAX=2 NCHAR=5;\nMATRIX\na ACGT\nb ACGT\n;\nEND;",
        "/tmp/bad.nex",
      ),
    ).toMatchObject({
      status: "error",
      warnings: expect.arrayContaining([
        expect.objectContaining({ code: "nexus-nchar-mismatch" }),
      ]),
    });
  });

  it("pads omitted MSF rows by block and reports recoverable header mismatches", () => {
    const result = parseMsa(
      [
        "!!AA_MULTIPLE_ALIGNMENT",
        " MSF: 2 Type: P Check: 0 ..",
        " Name: full Len: 6 Check: 0 Weight: 1.00",
        " Name: short Len: 2 Check: 0 Weight: 1.00",
        "//",
        "full ABCD",
        "short AB",
        "",
        "full EF",
      ].join("\n"),
      "/tmp/recoverable.msf",
    );

    expect(result).toMatchObject({
      document: {
        alignedLength: 6,
        rows: [
          { alignedSequence: "ABCDEF", id: "full" },
          { alignedSequence: "AB----", id: "short" },
        ],
        warnings: expect.arrayContaining([
          expect.objectContaining({ code: "msf-width-mismatch" }),
        ]),
      },
      status: "success",
    });
  });

  it("rejects an MSF declaration with no sequence data", () => {
    expect(
      parseMsa(
        "PileUp\n MSF: 4 Type: N\n Name: a Len: 4\n Name: b Len: 4\n//\na ACGT",
        "/tmp/missing-row.msf",
      ),
    ).toMatchObject({
      status: "error",
      warnings: expect.arrayContaining([
        expect.objectContaining({ code: "msf-declared-row-missing" }),
      ]),
    });
  });

  it("handles multiline NEXUS comments, dimensions, and MATCHCHAR", () => {
    const document = expectDocument(
      [
        "#NEXUS",
        "[comment begins",
        "and ends here]",
        "BEGIN DATA;",
        "DIMENSIONS NTAX=2 NCHAR=4;",
        "FORMAT DATATYPE=DNA GAP=- MISSING=? MATCHCHAR=.;",
        "MATRIX",
        "reference ACGT",
        "sample .C.T",
        ";",
        "END;",
      ].join("\n"),
      "/tmp/matchchar.nex",
    );
    expect(document.rows.map(({ alignedSequence }) => alignedSequence)).toEqual(
      ["ACGT", "ACGT"],
    );
  });

  it("does not treat a semicolon inside a quoted NEXUS taxon as the matrix terminator", () => {
    const document = expectDocument(
      [
        "#NEXUS",
        "BEGIN DATA;",
        "DIMENSIONS NTAX=2 NCHAR=4;",
        "MATRIX",
        "'alpha;one' ACGT",
        "beta AC-T",
        ";",
        "END;",
      ].join("\n"),
      "/tmp/quoted-semicolon.nex",
    );

    expect(document.rows.map(({ label }) => label)).toEqual([
      "alpha;one",
      "beta",
    ]);
  });

  it("rejects truncated containers and missing format headers", () => {
    expect(
      parseMsa("#NEXUS\nBEGIN DATA;\nMATRIX\na ACGT", "/tmp/truncated.nex"),
    ).toMatchObject({
      status: "error",
      warnings: expect.arrayContaining([
        expect.objectContaining({ code: "nexus-matrix-terminator-missing" }),
      ]),
    });
    expect(
      parseMsa("#NEXUS\n[unclosed\nMATRIX\na ACGT\n;", "/tmp/comment.nex"),
    ).toMatchObject({
      status: "error",
      warnings: expect.arrayContaining([
        expect.objectContaining({ code: "nexus-comment-unterminated" }),
      ]),
    });
    expect(parseMsa("a ACGT\nb ACGT", "/tmp/headerless.aln")).toMatchObject({
      status: "error",
      warnings: expect.arrayContaining([
        expect.objectContaining({ code: "clustal-header-missing" }),
      ]),
    });
    expect(
      parseMsa(">P1;a\nA\nACGT\n>P1;b\nB\nACGT*", "/tmp/truncated.pir"),
    ).toMatchObject({
      status: "error",
      warnings: expect.arrayContaining([
        expect.objectContaining({ code: "pir-terminator-missing" }),
      ]),
    });
  });

  it("disambiguates duplicate source labels with unique internal row IDs", () => {
    const document = expectDocument(
      [">dup", "AA", ">dup", "AG"].join("\n"),
      "/tmp/duplicates.afa",
    );
    expect(document.rows.map((row) => row.id)).toEqual(["dup__1", "dup__2"]);
    expect(document.rows.map((row) => row.label)).toEqual(["dup", "dup"]);
    expect(document.rows[1]).toMatchObject({
      duplicateSourceLabelCount: 2,
      duplicateSourceLabelIndex: 2,
      sourceId: "dup",
    });
    expect(
      document.warnings.some(
        (warning) => warning.code === "duplicate-sequence-id",
      ),
    ).toBe(true);
  });

  it("returns fatal errors for unknown containers, empty alignments, and width mismatches", () => {
    expect(parseMsa("plain text", "/tmp/example.txt")).toMatchObject({
      status: "error",
    });
    expect(parseMsa("#NEXUS\nBEGIN DATA;", "/tmp/empty.nex")).toMatchObject({
      status: "error",
    });
    expect(
      parseMsa([">a", "ACG", ">b", "ACGT"].join("\n"), "/tmp/bad.afa"),
    ).toMatchObject({
      message: expect.stringContaining("common display width"),
      status: "error",
    });
  });

  it("carries low-confidence warnings for mixed or ambiguous nucleic-acid symbols", () => {
    const document = expectDocument(
      [">a", "ACGT", ">b", "ACGU"].join("\n"),
      "/tmp/mixed.afa",
    );
    expect(document.molecule.moleculeType).toBe("nucleic-acid-ambiguous");
    expect(
      document.warnings.some(
        (warning) => warning.code === "molecule-inference-warning",
      ),
    ).toBe(true);
  });
});

describe("MSA biological helpers", () => {
  it("computes IUPAC consensus and protein representative residues", () => {
    const dna = expectDocument([">a", "A", ">b", "G"].join("\n"), "/tmp/a.afa");
    expect(computeNucleotideConsensus(dna.rows, "dna")).toBe("R");

    const protein = expectDocument(
      [">p1", "MQ", ">p2", "ME", ">p3", "MQ"].join("\n"),
      "/tmp/p.faa",
    );
    expect(computeProteinRepresentative(protein.rows)).toBe("MQ");
  });

  it("searches nucleic acids in forward and reverse-complement orientations", () => {
    const document = expectDocument(
      [">a", "AAGT", ">b", "ACGT"].join("\n"),
      "/tmp/search.afa",
    );
    const hits = searchMsaMotif(document, "ACT");
    expect(hits.some((hit) => hit.orientation === "reverse-complement")).toBe(
      true,
    );
    expect(reverseComplement("ACT")).toBe("AGT");
    expect(complementNucleotideSymbol("R")).toBe("Y");
  });

  it("builds coordinate, conservation, codon, and RNA-pair summaries", () => {
    const projection = buildUngappedProjection("A-CG");
    expect(projection).toEqual({
      alignmentColumns: [0, 2, 3],
      sequence: "ACG",
    });
    expect(getUngappedPosition("A-CG", 2)).toBe(2);
    expect(getUngappedPosition("A-CG", -1)).toBeNull();

    const document = expectDocument(
      [">a", "ATGAAA", ">b", "ATGGAA"].join("\n"),
      "/tmp/cds.afa",
    );
    const summaries = computeColumnSummaries(document.rows, "dna");
    expect(getMeanIdentity(summaries)).toBeGreaterThan(0.8);
    expect(getMeanNormalizedConservation(summaries)).toBeGreaterThan(0.6);
    expect(
      inferCdsContext({
        alignedLength: document.alignedLength,
        moleculeType: "dna",
        rows: document.rows,
      }).applicability,
    ).toBe("eligible");
    expect(translateStandardCodon("AUG")).toBe("M");
    expect(
      isNonsynonymousCodonDifference({
        anchorCodon: "AAA",
        candidateCodon: "GAA",
      }),
    ).toBe(true);
    expect(parseRnaPairs("<A.a>")).toMatchObject({
      pairs: expect.any(Array),
    });
  });

  it("classifies standards-aware residues and gap accounting", () => {
    expect(classifyResidue("N", "dna")).toBe("ambiguous-nucleotide");
    expect(classifyResidue("J", "protein")).toBe("ambiguous-amino-acid");
    expect(classifyResidue("O", "protein")).toBe("special-amino-acid");
    expect(classifyResidue("*", "protein")).toBe("termination");
    expect(countUngappedResidues("A-C_")).toBe(2);
  });
});

SHA-256: 04b51fc9c11460f69ada69fb8169ccbd5ef7a8eb8beb8f6a37575f1a46bab56f