← Files Biological Sequence & Alignment ViewerARCHIVED FILE

src/indexed-sequence-parser.test.ts

12.6 KB · Sep 30, 2026 · 23:01 UTC

↓ Download file

import { describe, expect, it, vi } from "vitest";

import {
  buildIndexedPreviewContents,
  parseIndexedSequenceEnvelope,
  serializeIndexedSequenceEnvelope,
} from "./indexed-sequence-envelope";
import { parseIndexedSequenceLines } from "./indexed-sequence-parser";

// Public Biopython c9489604d1d9607602ca9199a3852c1219ed330f,
// Tests/Quality/error_diff_ids.fastq; intentionally malformed negative control.
// SHA-256: fb28be12cda772adc2ca0d29c7b1159b439472ad8383406b2be512249bda347e.
const publicMismatchedFastq = [
  "@SLXA-B3_649_FC8437_R1_1_1_610_79",
  "GATGTGCAATACCTTTGTAGAGGAA",
  "+SLXA-B3_649_FC8437_R1_1_1_610_79",
  "YYYYYYYYYYYYYYYYYYWYWYYSU",
  "@SLXA-B3_649_FC8437_R1_1_1_397_389",
  "GGTTTGAGAAAGAGAAATGAGATAA",
  "+SLXA-B3_649_FC8437_R1_1_1_397_389",
  "YYYYYYYYYWYYYYWWYYYWYWYWW",
  "@SLXA-B3_649_FC8437_R1_1_1_850_123",
  "GAGGGTGTTGATCATGATGATGGCG",
  "+SLXA-B3_649_FC8437_R1_1_1_850_124",
  "YYYYYYYYYYYYYWYYWYYSYYYSY",
  "@SLXA-B3_649_FC8437_R1_1_1_362_549",
  "GGAAACAAAGTTTTTCTCAACATAG",
  "+SLXA-B3_649_FC8437_R1_1_1_362_549",
  "YYYYYYYYYYYYYYYYYYWWWWYWY",
  "@SLXA-B3_649_FC8437_R1_1_1_183_714",
  "GTATTATTTAATGGCATACACTCAA",
  "+SLXA-B3_649_FC8437_R1_1_1_183_714",
  "YYYYYYYYYYWYYYYWYWWUWWWQQ",
  "",
].join("\n");

// Public Biopython c9489604d1d9607602ca9199a3852c1219ed330f,
// Tests/Quality/example.fastq.
// SHA-256: 10bc5b39327a363b0019193c9823bc424a6d5706197688fdbdd45023a1481a0c.
const publicValidFastq = [
  "@EAS54_6_R1_2_1_413_324",
  "CCCTTCTTGTCTTCAGCGTTTCTCC",
  "+",
  ";;3;;;;;;;;;;;;7;;;;;;;88",
  "@EAS54_6_R1_2_1_540_792",
  "TTGGCAGGCCAAGGCCGATGGATCA",
  "+",
  ";;;;;;;;;;;7;;;;;-;;;3;83",
  "@EAS54_6_R1_2_1_443_348",
  "GTTGCTTCTGGCGTGGGTGGGGGGG",
  "+",
  ";;;;;;;;;;;9;7;;.7;393333",
  "",
].join("\n");

describe("streaming indexed sequence parser", () => {
  it("indexes FASTA incrementally with a bounded materialized cache", async () => {
    const progress = vi.fn();
    const envelope = await parseIndexedSequenceLines({
      compressed: true,
      fileName: "large.fasta.gz",
      format: "fasta",
      lines: asyncLines([">a first", "ACGT", ">b", "TTTT"]),
      onProgress: progress,
      sourceBytes: 42,
    });

    expect(envelope.index).toMatchObject({
      complete: true,
      compressed: true,
      indexedRecordCount: 2,
      materializedBases: 8,
      materializedRecordCount: 2,
      sourceBytes: 42,
      sourceVersion: "size-42",
    });
    expect(envelope.document.recordInventory).toEqual({
      materializedCount: 2,
      totalCount: 2,
      truncated: false,
    });
    expect(progress).toHaveBeenCalledWith(
      expect.objectContaining({ recordCount: 2 }),
    );

    const serialized = serializeIndexedSequenceEnvelope(envelope);
    const parsed = parseIndexedSequenceEnvelope(serialized);
    expect(parsed?.document.records.map(({ id }) => id)).toEqual(["a", "b"]);
    expect(buildIndexedPreviewContents(parsed!.document)).toContain(
      ">a first\nACGT",
    );
  });

  it("excludes genuine RF00360 secondary structure from streamed RNA residues", async () => {
    // R2DT a1ca674f245e4dc13838e346b8e03295c2130d0c,
    // data/rfam/RF00360/RF00360-traveler.fasta.
    const rna =
      "NNUNGCRGUGAYGACUYGGNRANAUUCAAGCUCAACAGACCRNANYRYAGNNYUUUYUYNNNNNNNNYYNRNNGGAUYGNUUUGNNNRNNNNGAUNNYYCCGCUGANCYGAGCNRNN";
    const structure =
      "((((((.........................................................................................................))))))";
    const envelope = await parseIndexedSequenceLines({
      compressed: true,
      fileName: "RF00360-traveler.fasta.gz",
      format: "fasta",
      lines: asyncLines([">RF00360", rna, structure]),
      sourceBytes: 245,
    });

    expect(envelope.document.records[0]).toMatchObject({
      length: 117,
      molecule: "rna",
      sequence: rna,
      sourceLabel: "RF00360",
    });
    expect(envelope.document.classification.molecule).toBe("rna");
    expect(envelope.index).toMatchObject({
      decodedBytes: 245,
      indexedRecordCount: 1,
      materializedBases: 117,
    });
  });

  it("keeps gap-only FASTA rows while rejecting structure-only empty records", async () => {
    const aligned = await parseIndexedSequenceLines({
      compressed: false,
      fileName: "gaps.fasta",
      format: "fasta",
      lines: asyncLines([">gaps", "..--", ">rna", "ACGU"]),
      sourceBytes: 23,
    });

    expect(aligned.document.records.map(({ sequence }) => sequence)).toEqual([
      "..--",
      "ACGU",
    ]);

    await expect(
      parseIndexedSequenceLines({
        compressed: false,
        fileName: "structure-only.fasta",
        format: "fasta",
        lines: asyncLines([">empty", "((..))", ">valid", "ACGU"]),
        sourceBytes: 28,
      }),
    ).rejects.toThrow("does not contain a sequence");
  });

  it("does not treat structural punctuation in FASTQ quality as FASTA annotation", async () => {
    const envelope = await parseIndexedSequenceLines({
      compressed: false,
      fileName: "quality.fastq",
      format: "fastq",
      lines: asyncLines(["@read", "ACGU", "+", "(())"]),
      sourceBytes: 18,
    });

    expect(envelope.document.records[0]?.quality?.ascii).toBe("(())");
    expect(envelope.document.fastqSummary?.readCount).toBe(1);
  });

  it("rejects the public Biopython FASTQ whose third repeated identifier differs", async () => {
    await expect(
      parseIndexedSequenceLines({
        compressed: true,
        fileName: "error_diff_ids.fastq.gz",
        format: "fastq",
        lines: asyncLines(publicMismatchedFastq.trimEnd().split("\n")),
        sourceBytes: publicMismatchedFastq.length,
      }),
    ).rejects.toThrow(
      "FASTQ + label 'SLXA-B3_649_FC8437_R1_1_1_850_124' does not match header 'SLXA-B3_649_FC8437_R1_1_1_850_123'",
    );
  });

  it("rejects a mismatched repeated FASTQ title before consuming its quality", async () => {
    async function* mismatchedLines() {
      yield "@read first title";
      yield "ACGT";
      yield "+read second title";
      throw new Error("quality must not be consumed after a mismatched title");
    }

    await expect(
      parseIndexedSequenceLines({
        compressed: false,
        fileName: "different-title.fastq",
        format: "fastq",
        lines: mismatchedLines(),
        sourceBytes: 48,
      }),
    ).rejects.toThrow(
      "FASTQ + label 'read second title' does not match header 'read first title'",
    );
  });

  it("preserves valid public Biopython FASTQ bare separators and punctuation qualities", async () => {
    const envelope = await parseIndexedSequenceLines({
      compressed: true,
      fileName: "example.fastq.gz",
      format: "fastq",
      lines: asyncLines(publicValidFastq.trimEnd().split("\n")),
      sourceBytes: publicValidFastq.length,
    });

    expect(envelope.document.fastqSummary?.readCount).toBe(3);
    expect(
      envelope.document.records.map(({ quality }) => quality?.ascii),
    ).toEqual([
      ";;3;;;;;;;;;;;;7;;;;;;;88",
      ";;;;;;;;;;;7;;;;;-;;;3;83",
      ";;;;;;;;;;;9;7;;.7;393333",
    ]);
  });

  it("matches repeated FASTQ full titles while ignoring trailing whitespace", async () => {
    const envelope = await parseIndexedSequenceLines({
      compressed: false,
      fileName: "titled.fastq",
      format: "fastq",
      lines: asyncLines([
        "@read sample description  ",
        "AC",
        "GT",
        "+read sample description\t",
        "()",
        "[]",
        "@bare",
        "AC",
        "+ \t",
        "{}",
        "@  leading title  ",
        "AC",
        "+  leading title\t",
        "<>",
      ]),
      sourceBytes: 80,
    });

    expect(
      envelope.document.records.map(({ quality }) => quality?.ascii),
    ).toEqual(["()[]", "{}", "<>"]);
  });

  it("rejects repeated FASTQ titles with additional leading whitespace", async () => {
    await expect(
      parseIndexedSequenceLines({
        compressed: false,
        fileName: "different-leading-space.fastq",
        format: "fastq",
        lines: asyncLines(["@read title", "ACGT", "+ read title", "IIII"]),
        sourceBytes: 36,
      }),
    ).rejects.toThrow(
      "FASTQ + label ' read title' does not match header 'read title'",
    );
  });

  it("streams multiline FASTQ while summarizing all reads and retaining quality", async () => {
    const envelope = await parseIndexedSequenceLines({
      compressed: false,
      fileName: "reads.fastq",
      format: "fastq",
      lines: asyncLines([
        "@read1 sample",
        "AC",
        "GT",
        "+",
        "II",
        "II",
        "@read2",
        "NNNN",
        "+",
        "!!!!",
      ]),
      sourceBytes: 64,
    });

    expect(envelope.document.fastqSummary).toMatchObject({
      readCount: 2,
      totalBases: 8,
    });
    expect(envelope.document.records[0]?.quality?.phred).toEqual([
      40, 40, 40, 40,
    ]);
    expect(buildIndexedPreviewContents(envelope.document)).toContain(
      "@read2\nNNNN\n+\n!!!!",
    );
  });

  it("preserves a host-provided source identity in the indexed envelope", async () => {
    const envelope = await parseIndexedSequenceLines({
      compressed: false,
      fileName: "identity.fasta",
      format: "fasta",
      lines: asyncLines([">record", "ACGT"]),
      sourceBytes: 13,
      sourceVersion: "sha256:source-v2",
    });

    expect(envelope.index.sourceVersion).toBe("sha256:source-v2");
  });

  it("rejects fatal streaming diagnostics instead of mounting partial data", async () => {
    await expect(
      parseIndexedSequenceLines({
        compressed: false,
        fileName: "broken.fastq",
        format: "fastq",
        lines: asyncLines(["@read", "ACGT", "+", "II"]),
        sourceBytes: 18,
      }),
    ).rejects.toThrow("quality characters");
    await expect(
      parseIndexedSequenceLines({
        compressed: false,
        fileName: "broken.fasta",
        format: "fasta",
        lines: asyncLines(["ACGT"]),
        sourceBytes: 5,
      }),
    ).rejects.toThrow("before the first header");
  });

  it("honors cancellation between streamed lines", async () => {
    const controller = new AbortController();
    async function* cancellingLines() {
      yield ">a";
      controller.abort();
      yield "ACGT";
    }
    await expect(
      parseIndexedSequenceLines({
        compressed: false,
        fileName: "cancel.fasta",
        format: "fasta",
        lines: cancellingLines(),
        signal: controller.signal,
        sourceBytes: 10,
      }),
    ).rejects.toThrow("cancelled");
  });

  it("enforces a decoded byte budget inside the streaming parser", async () => {
    await expect(
      parseIndexedSequenceLines({
        compressed: true,
        fileName: "expanded.fasta.gz",
        format: "fasta",
        lines: asyncLines([">expanded", "A".repeat(128)]),
        maxDecodedBytes: 32,
        sourceBytes: 24,
      }),
    ).rejects.toThrow("32-byte decoded input budget");
  });

  it("rejects tampered envelope cache metadata", async () => {
    const envelope = await parseIndexedSequenceLines({
      compressed: false,
      fileName: "demo.fasta",
      format: "fasta",
      lines: asyncLines([">a", "ACGT"]),
      sourceBytes: 8,
    });
    const serialized = serializeIndexedSequenceEnvelope(envelope).replace(
      '"materializedBases":4',
      '"materializedBases":9',
    );
    expect(() => parseIndexedSequenceEnvelope(serialized)).toThrow(
      "metadata is inconsistent",
    );
  });

  it("keeps a generated one-million-read FASTQ bounded while summarizing every read", async () => {
    const readCount = 1_000_000;
    const startedAt = performance.now();
    async function* generatedFastq() {
      for (let index = 0; index < readCount; index += 1) {
        yield `@read-${index + 1}`;
        yield "ACGT";
        yield "+";
        yield "IIII";
      }
    }
    const envelope = await parseIndexedSequenceLines({
      compressed: true,
      fileName: "generated.fastq.gz",
      format: "fastq",
      lines: generatedFastq(),
      sourceBytes: 24_000_000,
    });
    const elapsedMs = performance.now() - startedAt;

    expect(envelope.document.fastqSummary).toMatchObject({
      readCount,
      totalBases: readCount * 4,
    });
    expect(envelope.document.records.length).toBeLessThan(readCount);
    expect(envelope.document.records.length).toBeLessThanOrEqual(5_000);
    expect(envelope.index.materializedBases).toBeLessThanOrEqual(1_000_000);
    expect(elapsedMs).toBeLessThan(30_000);
    expect(envelope.document.recordInventory).toMatchObject({
      totalCount: readCount,
      truncated: true,
    });
    expect(serializeIndexedSequenceEnvelope(envelope).length).toBeLessThan(
      4 * 1_024 * 1_024,
    );
  }, 60_000);
});

async function* asyncLines(lines: Array<string>) {
  for (const line of lines) yield line;
}

SHA-256: 055270741b6dc4d83fb49b662bde08e46c756b9c16dd92655116eb3c256d4550