← Files Biological Sequence & Alignment ViewerARCHIVED FILE
src/sequence/parser.test.ts
25.8 KB · Sep 30, 2026 · 23:01 UTC
import { describe, expect, it } from "vitest";
import { parseFastqDocument } from "./formats/fastq";
import { parseSequenceDocument, parseSequenceDocumentResult } from "./parser";
// Public Biopython c9489604d1d9607602ca9199a3852c1219ed330f,
// Tests/NBRF/DMB_prot.pir, complete HLA:HLA00490 and HLA:HLA00492 records.
const publicHlaPirProteinRecords = [
">P1;HLA:HLA00490",
"HLA:HLA00490 DMB*0102, 94 bases, 73D5CC44 checksum.",
" PPSVQVAKTT PFNTREPVML ACYVWGFYPA EVTITWRKNG KLVMPHSSEH",
" KTAQPNGDWT YQTLSHLALT PSYGDTYTCV VEHIGAPEPI LRDW*",
">P1;HLA:HLA00492",
"HLA:HLA00492 DMB*0104, 80 bases, 453718BE checksum.",
" KTTPFNTREP VMLACYVWGF YPAEVTITWR KNGKLVMPHS SVHKTAQPNG",
" DWTYQTLSHL ALTPSYGDTY TCVVEHTGAP*",
].join("\n");
// Public Biopython c9489604d1d9607602ca9199a3852c1219ed330f,
// Tests/Quality/error_diff_ids.fastq; intentionally malformed negative control.
// SHA-256: fb28be12cda772adc2ca0d29c7b1159b439472ad8383406b2be512249bda347e.
const publicMismatchedFastq = [
"@SLXA-B3_649_FC8437_R1_1_1_610_79",
"GATGTGCAATACCTTTGTAGAGGAA",
"+SLXA-B3_649_FC8437_R1_1_1_610_79",
"YYYYYYYYYYYYYYYYYYWYWYYSU",
"@SLXA-B3_649_FC8437_R1_1_1_397_389",
"GGTTTGAGAAAGAGAAATGAGATAA",
"+SLXA-B3_649_FC8437_R1_1_1_397_389",
"YYYYYYYYYWYYYYWWYYYWYWYWW",
"@SLXA-B3_649_FC8437_R1_1_1_850_123",
"GAGGGTGTTGATCATGATGATGGCG",
"+SLXA-B3_649_FC8437_R1_1_1_850_124",
"YYYYYYYYYYYYYWYYWYYSYYYSY",
"@SLXA-B3_649_FC8437_R1_1_1_362_549",
"GGAAACAAAGTTTTTCTCAACATAG",
"+SLXA-B3_649_FC8437_R1_1_1_362_549",
"YYYYYYYYYYYYYYYYYYWWWWYWY",
"@SLXA-B3_649_FC8437_R1_1_1_183_714",
"GTATTATTTAATGGCATACACTCAA",
"+SLXA-B3_649_FC8437_R1_1_1_183_714",
"YYYYYYYYYYWYYYYWYWWUWWWQQ",
"",
].join("\n");
// Public Biopython c9489604d1d9607602ca9199a3852c1219ed330f,
// Tests/Quality/example.fastq.
// SHA-256: 10bc5b39327a363b0019193c9823bc424a6d5706197688fdbdd45023a1481a0c.
const publicValidFastq = [
"@EAS54_6_R1_2_1_413_324",
"CCCTTCTTGTCTTCAGCGTTTCTCC",
"+",
";;3;;;;;;;;;;;;7;;;;;;;88",
"@EAS54_6_R1_2_1_540_792",
"TTGGCAGGCCAAGGCCGATGGATCA",
"+",
";;;;;;;;;;;7;;;;;-;;;3;83",
"@EAS54_6_R1_2_1_443_348",
"GTTGCTTCTGGCGTGGGTGGGGGGG",
"+",
";;;;;;;;;;;9;7;;.7;393333",
"",
].join("\n");
describe("parseSequenceDocument", () => {
it("parses FASTA sequences and flags likely aligned FASTA", () => {
const document = parseSequenceDocument({
contents: ">a\nAC-GT\n>b\nACTGT\n",
fileName: "family.fasta",
});
expect(document.format).toBe("fasta");
expect(document.records).toHaveLength(2);
expect(document.warnings).toEqual(
expect.arrayContaining([expect.objectContaining({ code: "likely-msa" })]),
);
});
it("parses complete public HLA PIR records without ingesting descriptions or terminators", () => {
const document = parseSequenceDocument({
contents: publicHlaPirProteinRecords,
fileName: "DMB_prot.pir",
});
expect(document.kind).toBe("sequence-collection");
expect(document.classification.suggestedViewer).toBe("sequence");
expect(document.records).toEqual([
expect.objectContaining({
description: "HLA:HLA00490 DMB*0102, 94 bases, 73D5CC44 checksum.",
length: 94,
molecule: "protein",
sourceLabel: "HLA:HLA00490",
}),
expect.objectContaining({
description: "HLA:HLA00492 DMB*0104, 80 bases, 453718BE checksum.",
length: 80,
molecule: "protein",
sourceLabel: "HLA:HLA00492",
}),
]);
expect(document.records.every(({ sequence }) => !sequence.includes("*"))).toBe(
true,
);
});
it("preserves fatal diagnostics for PIR records missing their terminators", () => {
expect(
parseSequenceDocumentResult({
contents: ">P1;HLA:HLA00490\nHLA:HLA00490 DMB*0102\nPPSVQVAKTT\n",
fileName: "truncated.pir",
}),
).toMatchObject({
diagnostics: expect.arrayContaining([
expect.objectContaining({
code: "pir-terminator-missing",
severity: "error",
}),
]),
status: "error",
});
});
it("excludes the real RF00360 dot-bracket annotation from its 117 RNA residues", () => {
// R2DT a1ca674f245e4dc13838e346b8e03295c2130d0c,
// data/rfam/RF00360/RF00360-traveler.fasta.
const rna =
"NNUNGCRGUGAYGACUYGGNRANAUUCAAGCUCAACAGACCRNANYRYAGNNYUUUYUYNNNNNNNNYYNRNNGGAUYGNUUUGNNNRNNNNGAUNNYYCCGCUGANCYGAGCNRNN";
const structure =
"((((((.........................................................................................................))))))";
const document = parseSequenceDocument({
contents: [">RF00360", rna, structure].join("\n"),
fileName: "RF00360-traveler.fasta",
});
expect(document.records[0]).toEqual(
expect.objectContaining({
length: 117,
molecule: "rna",
sequence: rna,
sourceLabel: "RF00360",
}),
);
});
it("does not turn a structure-only FASTA record into a biological sequence", () => {
expect(
parseSequenceDocumentResult({
contents: ">empty\n((..))\n>valid\nACGU\n",
fileName: "structure-only.fasta",
}),
).toMatchObject({
diagnostics: expect.arrayContaining([
expect.objectContaining({ code: "fasta-record-empty" }),
]),
status: "error",
});
});
it("preserves malformed empty FASTA headers while ignoring structure annotations", () => {
expect(
parseSequenceDocumentResult({
contents: ">\nACGU\n",
fileName: "empty-header.fasta",
}),
).toMatchObject({ status: "error" });
});
it("parses GenBank metadata, sequence, and CDS translation", () => {
const document = parseSequenceDocument({
contents: `LOCUS DEMO 12 bp DNA linear
DEFINITION demo sequence.
ACCESSION DEMO1
VERSION DEMO1.1
SOURCE synthetic
FEATURES Location/Qualifiers
gene 1..12
/gene="foo"
CDS 1..12
/product="Foo protein"
/translation="MKTW"
ORIGIN
1 atgaaaacgtgg
//`,
fileName: "demo.gb",
});
expect(document.format).toBe("genbank");
expect(document.records[0]).toEqual(
expect.objectContaining({
sourceLabel: "DEMO1",
sequence: "ATGAAAACGTGG",
}),
);
expect(document.records[0]?.features).toEqual(
expect.arrayContaining([
expect.objectContaining({ type: "gene" }),
expect.objectContaining({ translation: "MKTW", type: "CDS" }),
]),
);
});
it("labels the public Arabidopsis cor6.6 mRNA as RNA without rewriting source thymine", () => {
// Biopython c9489604d1d9607602ca9199a3852c1219ed330f,
// Tests/GenBank/cor6_6.gb, accession X55053.1.
const document = parseSequenceDocument({
contents: `LOCUS ATCOR66M 513 bp mRNA PLN 02-MAR-1992
DEFINITION A.thaliana cor6.6 mRNA.
ACCESSION X55053
VERSION X55053.1 GI:16229
ORIGIN
1 aacaaaacac acatcaaaaa cgattttaca agaaaaaaat atctgaaaaa tgtcagagac
61 caacaagaat gccttccaag ccggtcaggc cgctggcaaa gctgaggaga agagcaatgt
121 tctgctggac aaggccaagg atgctgctgc tgcagctgga gcttccgcgc aacaggcggg
181 aaagagtata tcggatgcgg cagtgggagg tgttaacttc gtgaaggaca agaccggcct
241 gaacaagtag cgatccgagt caactttggg agttataatt tcccttttct aattaattgt
301 tgggattttc aaataaaatt tgggagtcat aattgattct cgtactcatc gtacttgttg
361 ttgtttttag tgttgtaatg ttttaatgtt tcttctccct ttagatgtac tacgtttgga
421 actttaagtt taatcaacaa aatctagttt aagttctaaa aaaaaaaaaa aaaaaaaaaa
481 aaaaaaaaaa aaaaaaaaaa aaaaaaaaaa aaa
//`,
fileName: "cor6_6.gb",
});
expect(document.classification.molecule).toBe("rna");
expect(document.records[0]).toMatchObject({
length: 513,
molecule: "rna",
sourceLabel: "X55053",
});
expect(document.records[0]?.sequence).toContain("T");
expect(document.records[0]?.sequence).not.toContain("U");
});
it("preserves distinct RNA and DNA molecule declarations across GenBank records", () => {
const document = parseSequenceDocument({
contents: `LOCUS ATCOR66M 513 bp mRNA PLN 02-MAR-1992
ACCESSION X55053
ORIGIN
1 aacaaaacac acatcaaaaa cgattttaca
//
LOCUS ARU237582 206 bp DNA PLN 24-MAR-1999
ACCESSION AJ237582
FEATURES Location/Qualifiers
mRNA 1..10
ORIGIN
1 ggacaaggcc aaggatgctg ctgctgcagc
//`,
fileName: "cor6_6.gb",
});
expect(document.classification.molecule).toBe("nucleic-acid-ambiguous");
expect(document.records.map(({ molecule }) => molecule)).toEqual([
"rna",
"dna",
]);
});
it("preserves wrapped GenBank metadata, wrapped qualifiers, and multiple records", () => {
const document = parseSequenceDocument({
contents: `LOCUS DEMO1 12 bp DNA linear
DEFINITION demo sequence with a wrapped
description.
ACCESSION DEMO1
FEATURES Location/Qualifiers
CDS 1..12
/product="Demo
protein"
/translation="MKT
W"
ORIGIN
1 atgaaaacgtgg
//
LOCUS DEMO2 6 bp DNA linear
DEFINITION second record.
ACCESSION DEMO2
ORIGIN
1 atgaaa
//`,
fileName: "demo.gbff",
});
expect(document.kind).toBe("sequence-collection");
expect(document.records.map((record) => record.sourceLabel)).toEqual([
"DEMO1",
"DEMO2",
]);
expect(document.records[0]).toEqual(
expect.objectContaining({
description: "demo sequence with a wrapped description.",
}),
);
expect(document.records[0]?.features[0]).toEqual(
expect.objectContaining({
label: "Demo protein",
translation: "MKTW",
}),
);
});
it("parses EMBL sequence records", () => {
const document = parseSequenceDocument({
contents: `ID DEMO; SV 1; linear; genomic DNA; STD; UNC; 12 BP.
AC DEMO1;
DE demo EMBL sequence
OS synthetic construct
FT CDS complement(1..12)
FT /product="Demo protein"
FT /translation="MKTW"
SQ Sequence 12 BP; 3 A; 3 C; 3 G; 3 T; 0 other;
atgaaaacgtgg 12
//`,
fileName: "demo.embl",
});
expect(document.format).toBe("embl");
expect(document.records[0]?.sequence).toBe("ATGAAAACGTGG");
expect(document.records[0]?.features[0]).toEqual(
expect.objectContaining({
label: "Demo protein",
strand: "-",
translation: "MKTW",
translationTrackReliable: true,
type: "CDS",
}),
);
});
it("preserves wrapped EMBL qualifiers and multiple entries", () => {
const document = parseSequenceDocument({
contents: `ID DEMO1; SV 1; linear; genomic DNA; STD; UNC; 12 BP.
AC DEMO1;
DE demo EMBL
DE sequence
FT CDS 1..12
FT /product="Demo
FT protein"
FT /translation="MKT
FT W"
SQ Sequence 12 BP; 3 A; 3 C; 3 G; 3 T; 0 other;
atgaaaacgtgg 12
//
ID DEMO2; SV 1; linear; genomic DNA; STD; UNC; 6 BP.
AC DEMO2;
DE second EMBL sequence
SQ Sequence 6 BP; 3 A; 0 C; 1 G; 2 T; 0 other;
atgaaa 6
//`,
fileName: "demo.embl",
});
expect(document.kind).toBe("sequence-collection");
expect(document.records.map((record) => record.sourceLabel)).toEqual([
"DEMO1",
"DEMO2",
]);
expect(document.records[0]).toEqual(
expect.objectContaining({
description: "demo EMBL sequence",
}),
);
expect(document.records[0]?.features[0]).toEqual(
expect.objectContaining({
label: "Demo protein",
translation: "MKTW",
}),
);
});
it("maps exact compound GenBank CDS translations across exon boundaries", () => {
const document = parseSequenceDocument({
contents: `LOCUS DEMO 12 bp DNA linear
DEFINITION demo sequence.
ACCESSION DEMO1
FEATURES Location/Qualifiers
CDS join(1..3,7..9)
/product="Joined protein"
/translation="MK"
ORIGIN
1 atgaaaacgtgg
//`,
fileName: "joined.gb",
});
expect(document.records[0]?.features[0]).toEqual(
expect.objectContaining({
end: 9,
segments: [
expect.objectContaining({ end: 3, start: 1 }),
expect.objectContaining({ end: 9, start: 7 }),
],
sourceLocation: "join(1..3,7..9)",
start: 1,
translation: "MK",
translationCoordinateMap: [
expect.objectContaining({
aminoAcidIndex: 1,
codonCoordinates: [1, 2, 3],
}),
expect.objectContaining({
aminoAcidIndex: 2,
codonCoordinates: [7, 8, 9],
}),
],
translationTrackReliable: true,
type: "CDS",
}),
);
expect(document.warnings).not.toEqual(
expect.arrayContaining([
expect.objectContaining({ code: "ambiguous-genbank-location" }),
]),
);
});
it("parses wrapped feature locations and honors codon_start", () => {
const document = parseSequenceDocument({
contents: `LOCUS WRAPPED 10 bp DNA linear
ACCESSION WRAPPED1
FEATURES Location/Qualifiers
CDS join(1..4,
5..10)
/codon_start=2
/product="wrapped CDS"
ORIGIN
1 aatgaaatag
//`,
fileName: "wrapped.gb",
});
expect(document.records[0]?.features[0]).toMatchObject({
codonStart: 2,
sourceLocation: "join(1..4,5..10)",
translation: "MK",
translationCoordinateMap: [
{
aminoAcidIndex: 1,
codonCoordinates: [2, 3, 4],
displayCoordinate: 2,
},
{
aminoAcidIndex: 2,
codonCoordinates: [5, 6, 7],
displayCoordinate: 5,
},
],
translationSource: "computed",
});
});
it("maps reverse-strand joins in biological order and honors transl_table", () => {
const reverse = parseSequenceDocument({
contents: `LOCUS REVERSE 12 bp DNA linear
ACCESSION REVERSE1
FEATURES Location/Qualifiers
CDS complement(join(1..3,10..12))
ORIGIN
1 aaacccgggcat
//`,
fileName: "reverse.gb",
});
expect(reverse.records[0]?.features[0]).toMatchObject({
strand: "-",
translation: "MF",
translationCoordinateMap: [
expect.objectContaining({ codonCoordinates: [12, 11, 10] }),
expect.objectContaining({ codonCoordinates: [3, 2, 1] }),
],
});
const mitochondrial = parseSequenceDocument({
contents: `LOCUS MITO 9 bp DNA linear
ACCESSION MITO1
FEATURES Location/Qualifiers
CDS 1..9
/transl_table=2
ORIGIN
1 atatgaaga
//`,
fileName: "mito.gb",
});
expect(mitochondrial.records[0]?.features[0]).toMatchObject({
geneticCodeId: 2,
translation: "MW",
translationSource: "computed",
});
const bacterial = parseSequenceDocument({
contents: `LOCUS BACTERIAL 6 bp DNA linear
ACCESSION BACTERIAL1
FEATURES Location/Qualifiers
CDS 1..6
/transl_table=11
ORIGIN
1 gtggtg
//`,
fileName: "bacterial.gb",
});
expect(bacterial.records[0]?.features[0]).toMatchObject({
geneticCodeId: 11,
translation: "MV",
translationSource: "computed",
});
});
it("preserves and computes exact CDS mappings across supported NCBI genetic codes", () => {
const sourceTranslated = parseSequenceDocument({
contents: `LOCUS MITO 9 bp DNA linear
ACCESSION MITO5
FEATURES Location/Qualifiers
CDS 1..9
/transl_table=5
/translation="MKW"
ORIGIN
1 atgaaatgg
//`,
fileName: "mito-source.gb",
});
expect(sourceTranslated.records[0]?.features[0]).toMatchObject({
geneticCodeId: 5,
translation: "MKW",
translationCoordinateMap: [
expect.objectContaining({ codonCoordinates: [1, 2, 3] }),
expect.objectContaining({ codonCoordinates: [4, 5, 6] }),
expect.objectContaining({ codonCoordinates: [7, 8, 9] }),
],
translationMappingUnavailableReason: undefined,
translationSource: "qualifier",
translationTrackReliable: true,
});
expect(sourceTranslated.warnings).toEqual([]);
const computedOnly = parseSequenceDocument({
contents: `LOCUS MITO 9 bp DNA linear
ACCESSION MITO5
FEATURES Location/Qualifiers
CDS 1..9
/transl_table=5
ORIGIN
1 atgaaatgg
//`,
fileName: "mito-computed.gb",
});
expect(computedOnly.records[0]?.features[0]).toMatchObject({
geneticCodeId: 5,
translation: "MKW",
translationCoordinateMap: [
expect.objectContaining({ codonCoordinates: [1, 2, 3] }),
expect.objectContaining({ codonCoordinates: [4, 5, 6] }),
expect.objectContaining({ codonCoordinates: [7, 8, 9] }),
],
translationSource: "computed",
translationTrackReliable: true,
});
});
it("reports why remote or ordered CDS locations cannot be mapped", () => {
const document = parseSequenceDocument({
contents: `LOCUS REMOTE 9 bp DNA linear
ACCESSION REMOTE1
FEATURES Location/Qualifiers
CDS order(1..3,OTHER.1:4..6)
/translation="MK"
ORIGIN
1 atgaaataa
//`,
fileName: "remote.gb",
});
expect(document.records[0]?.features[0]).toMatchObject({
translationCoordinateMap: undefined,
translationMappingUnavailableReason: "remote segment",
translationTrackReliable: false,
});
expect(document.warnings).toEqual(
expect.arrayContaining([
expect.objectContaining({ code: "ambiguous-genbank-location" }),
]),
);
});
it("parses FASTQ quality strings", () => {
const document = parseSequenceDocument({
contents: "@read1\nACGT\n+\nIIII\n",
fileName: "reads.fastq",
});
expect(document.format).toBe("fastq");
expect(document.records[0]?.quality?.phred).toEqual([40, 40, 40, 40]);
expect(document.fastqSummary).toEqual(
expect.objectContaining({
gcFraction: 0.5,
meanQuality: 40,
q30Fraction: 1,
readCount: 1,
totalBases: 4,
}),
);
});
it("rejects the public Biopython FASTQ whose third repeated identifier differs", () => {
const document = parseFastqDocument({
contents: publicMismatchedFastq,
fileName: "error_diff_ids.fastq",
});
expect(document.fastqSummary?.readCount).toBe(2);
expect(document.records).toHaveLength(2);
expect(document.warnings).toEqual(
expect.arrayContaining([
expect.objectContaining({
code: "fastq-separator-label-mismatch",
line: 11,
severity: "error",
}),
]),
);
const result = parseSequenceDocumentResult({
contents: publicMismatchedFastq,
fileName: "error_diff_ids.fastq",
});
expect(result).toMatchObject({
message: expect.stringContaining("850_124"),
status: "error",
});
expect(result).not.toHaveProperty("document");
});
it("preserves valid public Biopython FASTQ bare separators and punctuation qualities", () => {
const document = parseSequenceDocument({
contents: publicValidFastq,
fileName: "example.fastq",
});
expect(document.fastqSummary?.readCount).toBe(3);
expect(document.records.map(({ quality }) => quality?.ascii)).toEqual([
";;3;;;;;;;;;;;;7;;;;;;;88",
";;;;;;;;;;;7;;;;;-;;;3;83",
";;;;;;;;;;;9;7;;.7;393333",
]);
});
it("matches repeated FASTQ full titles while ignoring trailing whitespace", () => {
const document = parseSequenceDocument({
contents: [
"@read sample description ",
"AC",
"GT",
"+read sample description\t",
"()",
"[]",
"@bare",
"AC",
"+ \t",
"{}",
"@ leading title ",
"AC",
"+ leading title\t",
"<>",
].join("\n"),
fileName: "titled.fastq",
});
expect(document.records.map(({ quality }) => quality?.ascii)).toEqual([
"()[]",
"{}",
"<>",
]);
});
it("rejects repeated FASTQ titles with additional leading whitespace", () => {
expect(
parseSequenceDocumentResult({
contents: "@read title\nACGT\n+ read title\nIIII\n",
fileName: "different-leading-space.fastq",
}),
).toMatchObject({
diagnostics: expect.arrayContaining([
expect.objectContaining({
code: "fastq-separator-label-mismatch",
message:
"FASTQ + label ' read title' does not match header 'read title'.",
severity: "error",
}),
]),
status: "error",
});
});
it("rejects repeated FASTQ titles that differ after a shared first identifier", () => {
expect(
parseSequenceDocumentResult({
contents: "@read first title\nACGT\n+read second title\nIIII\n",
fileName: "different-title.fastq",
}),
).toMatchObject({
diagnostics: expect.arrayContaining([
expect.objectContaining({
code: "fastq-separator-label-mismatch",
severity: "error",
}),
]),
status: "error",
});
});
it("aggregates multi-read FASTQ length, composition, and quality metrics", () => {
const document = parseSequenceDocument({
contents: ["@read1", "ACGN", "+", "I5+!", "@read2", "GG", "+", "??"].join(
"\n",
),
fileName: "reads.fastq",
});
expect(document.fastqSummary).toEqual({
gcFraction: 4 / 6,
meanQuality: 130 / 6,
meanReadLength: 3,
nFraction: 1 / 6,
q20Fraction: 4 / 6,
q30Fraction: 3 / 6,
qualityEncoding: "phred+33-assumed",
readCount: 2,
readLengthMax: 4,
readLengthMin: 2,
totalBases: 6,
});
});
it("parses wrapped FASTQ sequence and quality lines as one logical record", () => {
const document = parseSequenceDocument({
contents: "@wrapped\nACGT\nNN\n+wrapped\nIIII\n!!\n",
fileName: "wrapped.fastq",
});
expect(document.records[0]).toMatchObject({
sequence: "ACGTNN",
sourceLabel: "wrapped",
});
expect(document.records[0]?.quality?.phred).toEqual([40, 40, 40, 40, 0, 0]);
expect(document.recordInventory).toEqual({
materializedCount: 1,
totalCount: 1,
truncated: false,
});
});
it("streams mixed CRLF and CR FASTQ line endings without materializing a line array", () => {
const document = parseSequenceDocument({
contents: "@mixed\r\nAC\rGT\r\n+\r\nII\rII",
fileName: "mixed.fastq",
});
expect(document.records[0]).toMatchObject({
quality: { ascii: "IIII", phred: [40, 40, 40, 40] },
sequence: "ACGT",
sourceLabel: "mixed",
});
});
it("reports truncated FASTQ as an explicit parse error", async () => {
const { parseSequenceDocumentResult } = await import("./parser");
const result = parseSequenceDocumentResult({
contents: "@truncated\nACGT\n+\nII\n",
fileName: "truncated.fastq",
});
expect(result).toMatchObject({
diagnostics: [
expect.objectContaining({
code: "truncated-fastq-quality",
severity: "error",
}),
],
status: "error",
});
});
it("summarizes every FASTQ read while bounding retained interactive records", () => {
const readCount = 5_001;
const document = parseSequenceDocument({
contents: Array.from(
{ length: readCount },
(_, index) => `@read-${index + 1}\nACGT\n+\nIIII`,
).join("\n"),
fileName: "many.fastq",
});
expect(document.fastqSummary).toMatchObject({
readCount,
totalBases: readCount * 4,
});
expect(document.records).toHaveLength(5_000);
expect(document.recordInventory).toEqual({
materializedCount: 5_000,
totalCount: readCount,
truncated: true,
});
});
it("does not allocate an unbounded decoded-quality array for an oversized read", () => {
const sequence = "A".repeat(1_000_001);
const document = parseSequenceDocument({
contents: `@long-read\n${sequence}\n+\n${"I".repeat(sequence.length)}\n`,
fileName: "long.fastq",
});
expect(document.fastqSummary).toMatchObject({
readCount: 1,
totalBases: sequence.length,
});
expect(document.records[0]?.sequence).toBe(sequence);
expect(document.records[0]?.quality).toBeUndefined();
expect(document.warnings).toEqual(
expect.arrayContaining([
expect.objectContaining({
code: "fastq-inspection-retention-limit",
severity: "info",
}),
]),
);
});
it("recognizes featureless GenBank and EMBL records", () => {
expect(
parseSequenceDocument({
contents:
"LOCUS NOFEATURE 4 bp DNA linear\nORIGIN\n 1 acgt\n//\n",
fileName: "featureless.gb",
}).records[0]?.sequence,
).toBe("ACGT");
expect(
parseSequenceDocument({
contents:
"ID NOFEATURE; SV 1; linear; DNA; STD; UNC; 4 BP.\nSQ Sequence 4 BP;\n acgt 4\n//\n",
fileName: "featureless.embl",
}).records[0]?.sequence,
).toBe("ACGT");
});
it("rejects an otherwise valid FASTA collection that contains an empty record", () => {
expect(
parseSequenceDocumentResult({
contents: ">empty\n>valid\nACGT\n",
fileName: "mixed.fasta",
}),
).toMatchObject({
diagnostics: expect.arrayContaining([
expect.objectContaining({ code: "fasta-record-empty" }),
]),
status: "error",
});
});
});
SHA-256: 1d22586bc856214ee269835c64dabf129f9203d34f3a2a5b8b645d9346cb5e8d