← Files Biological Sequence & Alignment ViewerARCHIVED FILE
src/workbench-exports.test.ts
38.7 KB · Sep 30, 2026 · 23:01 UTC
import { describe, expect, it } from "vitest";
import { parseMsa } from "./msa/parser";
import { parseSequenceDocument } from "./sequence/parser";
import { parseAnnotationTrack, parseSequenceTrack } from "./sequence/tracks";
import {
exportAlignmentWorkbench,
exportSequenceWorkbench,
} from "./workbench-exports";
describe("round-trippable workbench exports", () => {
it("omits unmapped source traces from subsequence exports and preserves the full source", () => {
// Synthetic trace control, not an experimental chromatogram.
const document = parseSequenceDocument({
contents: ">QA-SYNTHETIC-TRACE\nATGC\n",
fileName: "synthetic-trace.fasta",
});
const record = document.records[0]!;
record.chromatogram = {
channels: {
A: [1, 2, 3, 4],
C: [2, 3, 4, 5],
G: [3, 4, 5, 6],
T: [4, 5, 6, 7],
},
format: "abif",
peakLocations: [0, 1, 2, 3],
quality: [20, 30, 40, 50],
qualityEncoding: "phred",
sampleCount: 4,
};
const source = structuredClone(document);
const selected = JSON.parse(
exportSequenceWorkbench({
document,
format: "json",
recordId: record.id,
scope: "selection",
selection: { end: 3, recordId: record.id, start: 2 },
}).content,
) as { document: typeof document };
expect(selected.document.records[0]?.sequence).toBe("TG");
expect(selected.document.records[0]?.chromatogram).toBeUndefined();
expect(selected.document.records[0]?.quality).toBeUndefined();
expect(selected.document.warnings).toContainEqual(
expect.objectContaining({ code: "source-trace-not-carried" }),
);
const full = JSON.parse(
exportSequenceWorkbench({
document,
format: "json",
recordId: record.id,
scope: "all",
}).content,
) as { document: typeof document };
expect(full.document.records[0]?.chromatogram).toEqual(record.chromatogram);
expect(document).toEqual(source);
expect(() =>
exportSequenceWorkbench({
document,
format: "fastq",
recordId: record.id,
scope: "selection",
selection: { end: 3, recordId: record.id, start: 2 },
}),
).toThrow();
});
it.each(["genbank", "embl"] as const)(
"preserves compound feature segments and qualifiers in %s",
(format) => {
const document = annotatedDocument();
const record = document.records[0];
const exported = exportSequenceWorkbench({
document,
format,
recordId: record?.id ?? "",
scope: "all",
});
const reparsed = parseSequenceDocument({
contents: exported.content,
fileName: exported.name,
});
expect(reparsed.records[0]?.sequence).toBe(record?.sequence);
expect(reparsed.records[0]?.features[0]).toMatchObject({
qualifiers: expect.objectContaining({ note: "important" }),
segments: [
expect.objectContaining({ start: 1, end: 3 }),
expect.objectContaining({ start: 7, end: 9 }),
],
strand: "-",
});
},
);
it("exports selected sequence and exact one-based annotations", () => {
const document = annotatedDocument();
const record = document.records[0];
const selection = { end: 8, recordId: record?.id ?? "", start: 4 };
const fasta = exportSequenceWorkbench({
document,
format: "fasta",
recordId: record?.id ?? "",
scope: "selection",
selection,
});
const gff = exportSequenceWorkbench({
document,
format: "gff3",
recordId: record?.id ?? "",
scope: "all",
});
const bed = exportSequenceWorkbench({
document,
format: "bed",
recordId: record?.id ?? "",
scope: "all",
});
expect(fasta.content).toContain("demo:4-8\nTACGT");
expect(gff.content).toContain("demo\tsequence-viewer\tmisc_feature\t1\t3");
expect(bed.content).toContain("demo\t0\t3");
});
it("exports standards-compliant GTF with exact compound coordinates", () => {
const document = annotatedDocument();
const exported = exportSequenceWorkbench({
document,
format: "gtf",
recordId: document.records[0]?.id ?? "",
scope: "all",
});
const features = parseAnnotationTrack(exported.content, "gtf");
expect(exported.mediaType).toBe("text/x-gtf");
expect(exported.name).toMatch(/\.gtf$/u);
expect(features).toHaveLength(2);
expect(features).toEqual(
expect.arrayContaining([
expect.objectContaining({
end: 3,
reference: "demo",
start: 1,
strand: "-",
}),
expect.objectContaining({
end: 9,
reference: "demo",
start: 7,
strand: "-",
}),
]),
);
expect(exported.content).toContain('gene_id "');
expect(exported.content).toContain('transcript_id "');
});
it.each(["sequence", "alignment"] as const)(
"writes a valid, self-contained %s PDF with accurate cross-references",
(kind) => {
const sequence = annotatedDocument();
const alignment = alignmentDocument();
const exported =
kind === "sequence"
? exportSequenceWorkbench({
document: sequence,
format: "pdf",
recordId: sequence.records[0]?.id ?? "",
scope: "all",
})
: exportAlignmentWorkbench({
document: alignment,
format: "pdf",
scope: "all",
visibleRows: alignment.rows,
});
const crossReference = exported.content.match(
/startxref\n(\d+)\n%%EOF\n$/u,
);
expect(exported.mediaType).toBe("application/pdf");
expect(exported.name).toMatch(/\.pdf$/u);
expect(exported.content).toMatch(/^%PDF-1\.4\n/u);
expect(crossReference).not.toBeNull();
expect(exported.content.slice(Number(crossReference?.[1]))).toMatch(
/^xref\n/u,
);
},
);
it("exports loaded variants as valid VCF rows", () => {
const document = annotatedDocument();
const exported = exportSequenceWorkbench({
document,
format: "vcf",
recordId: document.records[0]?.id ?? "",
scope: "all",
tracks: [
{
format: "vcf",
id: "variants",
kind: "variants",
mapping: {
matchedReference: "demo",
requestedReference: "demo",
status: "matched",
unmatchedReferences: [],
},
name: "variants.vcf",
source: { displayName: "variants.vcf" },
summary: {
itemCount: 1,
references: ["demo"],
truncated: false,
},
variants: [
{
alternateAlleles: ["G"],
filters: [],
id: "v1",
info: { DP: "10" },
position: 4,
quality: 50,
reference: "demo",
referenceAllele: "T",
samples: {},
},
],
},
],
});
expect(exported.content).toContain("demo\t4\tv1\tT\tG\t50\tPASS\tDP=10");
});
it("limits selected and visible VCF exports to their exact active reference and interval", () => {
const document = parseSequenceDocument({
contents: ">chr1\nAAAAAAAAAAAA\n",
fileName: "reference.fasta",
});
const record = document.records[0];
if (record == null) throw new Error("Expected chromosome 1.");
const track = parseSequenceTrack({
content: [
"##fileformat=VCFv4.3",
"#CHROM\tPOS\tID\tREF\tALT\tQUAL\tFILTER\tINFO",
"1\t2\tselected\tA\tG\t50\tPASS\t.",
"1\t8\tunselected\tA\tT\t50\tPASS\t.",
"chr2\t3\tother-chromosome\tA\tC\t50\tPASS\t.",
].join("\n"),
displayName: "variants.vcf",
format: "vcf",
id: "variants",
requestedReference: record.sourceLabel,
});
const selected = exportSequenceWorkbench({
document,
format: "vcf",
recordId: record.id,
scope: "selection",
selection: { end: 4, recordId: record.id, start: 1 },
tracks: [track],
});
const visible = exportSequenceWorkbench({
document,
format: "vcf",
recordId: record.id,
scope: "visible",
tracks: [track],
});
const all = exportSequenceWorkbench({
document,
format: "vcf",
recordId: record.id,
scope: "all",
tracks: [track],
});
expect(selected.content).toContain("\tselected\t");
expect(selected.content).not.toContain("\tunselected\t");
expect(selected.content).not.toContain("\tother-chromosome\t");
expect(visible.content).toContain("\tselected\t");
expect(visible.content).toContain("\tunselected\t");
expect(visible.content).not.toContain("\tother-chromosome\t");
expect(all.content).toContain("\tother-chromosome\t");
});
it("preserves VCF metadata, FORMAT, all 100 genotypes, and unassessed FILTER values", () => {
const document = parseSequenceDocument({
contents: ">1\nAAAAAAAAAAAA\n",
fileName: "reference.fasta",
});
const record = document.records[0];
if (record == null) throw new Error("Expected chromosome 1.");
const sampleNames = Array.from(
{ length: 100 },
(_, index) => `HG${String(index + 96).padStart(5, "0")}`,
);
const sampleValues = sampleNames.map(
(_, index) => `${index % 2}|${(index + 1) % 2}:0.200:-0.18,-0.47,-2.42`,
);
const metadata = [
"##fileformat=VCFv4.3",
'##FORMAT=<ID=GT,Number=1,Type=String,Description="Genotype">',
'##FORMAT=<ID=DS,Number=1,Type=Float,Description="Dosage">',
'##FILTER=<ID=q10,Description="Quality below 10">',
...Array.from(
{ length: 42 },
(_, index) =>
`##INFO=<ID=I${index},Number=1,Type=String,Description="Annotation ${index}">`,
),
];
const originalRow = [
"1",
"2",
".",
"A",
"G",
"10.500",
".",
"I0=retained;I1=exact",
"GT:DS:GL",
...sampleValues,
].join("\t");
const content = [
...metadata,
[
"#CHROM",
"POS",
"ID",
"REF",
"ALT",
"QUAL",
"FILTER",
"INFO",
"FORMAT",
...sampleNames,
].join("\t"),
originalRow,
].join("\n");
const track = parseSequenceTrack({
content,
displayName: "one-hundred-samples.vcf",
format: "vcf",
id: "variants",
requestedReference: record.sourceLabel,
});
const exported = exportSequenceWorkbench({
document,
format: "vcf",
recordId: record.id,
scope: "all",
tracks: [track],
});
const lines = exported.content.trimEnd().split("\n");
const exportedHeader = lines.find((line) => line.startsWith("#CHROM"));
expect(lines.filter((line) => line.startsWith("##"))).toEqual(metadata);
expect(exportedHeader?.split("\t")).toHaveLength(109);
expect(exportedHeader?.split("\t").slice(9)).toEqual(sampleNames);
expect(lines.at(-1)).toBe(originalRow);
expect(lines.at(-1)?.split("\t")[6]).toBe(".");
expect(lines.at(-1)?.split("\t").slice(9)).toEqual(sampleValues);
});
it("excludes unrelated sample identities and metadata from selection-scoped VCF exports", () => {
const document = parseSequenceDocument({
contents: ">chr1\nAAAAAAAAAAAA\n",
fileName: "reference.fasta",
});
const record = document.records[0];
if (record == null) throw new Error("Expected chromosome 1.");
const selectedTrack = parseSequenceTrack({
content: [
"##fileformat=VCFv4.3",
"##contig=<ID=1,length=12>",
"##contig=<ID=2,length=12>",
'##INFO=<ID=USED,Number=1,Type=String,Description="Used">',
'##INFO=<ID=UNUSED,Number=1,Type=String,Description="Not selected">',
'##FORMAT=<ID=GT,Number=1,Type=String,Description="Genotype">',
'##FORMAT=<ID=DS,Number=1,Type=Float,Description="Unused dosage">',
"#CHROM\tPOS\tID\tREF\tALT\tQUAL\tFILTER\tINFO\tFORMAT\tSELECTED_SAMPLE",
"1\t2\tselected\tA\tG\t50\tPASS\tUSED=yes\tGT\t0|1",
"2\t2\toff-target\tA\tT\t50\tPASS\tUNUSED=no\tGT:DS\t0|0:0.1",
].join("\n"),
displayName: "selected.vcf",
format: "vcf",
id: "selected",
requestedReference: record.sourceLabel,
});
const unrelatedTrack = parseSequenceTrack({
content: [
"##fileformat=VCFv4.3",
"##source=unrelated-private-study",
"#CHROM\tPOS\tID\tREF\tALT\tQUAL\tFILTER\tINFO\tFORMAT\tUNRELATED_PRIVATE_SAMPLE",
"1\t9\tunrelated\tA\tC\t50\tPASS\t.\tGT\t1|1",
].join("\n"),
displayName: "unrelated.vcf",
format: "vcf",
id: "unrelated",
requestedReference: record.sourceLabel,
});
const exported = exportSequenceWorkbench({
document,
format: "vcf",
recordId: record.id,
scope: "selection",
selection: { end: 2, recordId: record.id, start: 2 },
tracks: [selectedTrack, unrelatedTrack],
});
expect(exported.content).toContain("SELECTED_SAMPLE");
expect(exported.content).toContain("##contig=<ID=1");
expect(exported.content).toContain("##INFO=<ID=USED");
expect(exported.content).toContain("##FORMAT=<ID=GT");
expect(exported.content).not.toContain("UNRELATED_PRIVATE_SAMPLE");
expect(exported.content).not.toContain("unrelated-private-study");
expect(exported.content).not.toContain("##contig=<ID=2");
expect(exported.content).not.toContain("##INFO=<ID=UNUSED");
expect(exported.content).not.toContain("##FORMAT=<ID=DS");
});
it("preserves all three authentic public FASTQ reads when exporting every record", () => {
// Immutable public Biopython Tests/Quality/example.fastq, SHA-256
// 10bc5b39327a363b0019193c9823bc424a6d5706197688fdbdd45023a1481a0c.
const contents = [
"@EAS54_6_R1_2_1_413_324",
"CCCTTCTTGTCTTCAGCGTTTCTCC",
";;3;;;;;;;;;;;;7;;;;;;;88",
];
const source = [
contents[0],
contents[1],
"+",
contents[2],
"@EAS54_6_R1_2_1_540_792",
"TTGGCAGGCCAAGGCCGATGGATCA",
"+",
";;;;;;;;;;;7;;;;;-;;;3;83",
"@EAS54_6_R1_2_1_443_348",
"GTTGCTTCTGGCGTGGGTGGGGGGG",
"+",
";;;;;;;;;;;9;7;;.7;393333",
"",
].join("\n");
const document = parseSequenceDocument({
contents: source,
fileName: "example.fastq",
});
const selectedRecord = document.records[1];
if (selectedRecord == null) throw new Error("Expected a second FASTQ read.");
const all = exportSequenceWorkbench({
document,
format: "fastq",
recordId: selectedRecord.id,
scope: "all",
});
const visible = exportSequenceWorkbench({
document,
format: "fastq",
recordId: selectedRecord.id,
scope: "visible",
});
const reparsed = parseSequenceDocument({
contents: all.content,
fileName: "roundtrip.fastq",
});
expect(reparsed.records).toHaveLength(3);
expect(reparsed.records.map(({ quality }) => quality?.ascii)).toEqual(
document.records.map(({ quality }) => quality?.ascii),
);
expect(visible.content).toContain(`@${selectedRecord.sourceLabel}`);
expect(visible.content.match(/^@/gmu)).toHaveLength(1);
});
it.each(["genbank", "embl"] as const)(
"round-trips every sequence record in an all-scope %s export",
(format) => {
const document = parseSequenceDocument({
contents: ">first\nACGTACGT\n>second\nAACC\n>third\nGGTTAA\n",
fileName: "records.fasta",
});
const selectedRecord = document.records[1];
if (selectedRecord == null) throw new Error("Expected multiple records.");
const all = exportSequenceWorkbench({
document,
format,
recordId: selectedRecord.id,
scope: "all",
});
const visible = exportSequenceWorkbench({
document,
format,
recordId: selectedRecord.id,
scope: "visible",
});
const reparsed = parseSequenceDocument({
contents: all.content,
fileName: all.name,
});
expect(reparsed.records.map(({ sourceLabel }) => sourceLabel)).toEqual([
"first",
"second",
"third",
]);
expect(
parseSequenceDocument({ contents: visible.content, fileName: visible.name })
.records,
).toHaveLength(1);
},
);
it("exports RNA as RNA and protein EMBL records in amino-acid units", () => {
const rna = parseSequenceDocument({
contents: ">URS0000D6941A\nAUGCUUAGCAUUGCAU\n",
fileName: "accessioned-rna.fasta",
});
const protein = parseSequenceDocument({
contents: ">A00022\nMQWPEPTIDEFKWY\n",
fileName: "protein.faa",
});
const rnaRecord = rna.records[0];
const proteinRecord = protein.records[0];
if (rnaRecord == null || proteinRecord == null) {
throw new Error("Expected RNA and protein records.");
}
const rnaGenBank = exportSequenceWorkbench({
document: rna,
format: "genbank",
recordId: rnaRecord.id,
scope: "all",
});
const rnaEmbl = exportSequenceWorkbench({
document: rna,
format: "embl",
recordId: rnaRecord.id,
scope: "all",
});
const proteinEmbl = exportSequenceWorkbench({
document: protein,
format: "embl",
recordId: proteinRecord.id,
scope: "all",
});
expect(rnaGenBank.content).toMatch(/^LOCUS\s+URS0000D6941A\s+16 bp RNA/mu);
expect(rnaEmbl.content).toContain("; RNA; UNC; 16 BP.");
expect(proteinEmbl.content).toContain("; PROTEIN; UNC; 14 AA.");
expect(proteinEmbl.content).toContain("SQ Sequence 14 AA;");
expect(proteinEmbl.content).not.toContain("14 BP");
});
it("computes strand-aware compound CDS phases and preserves codon_start", () => {
const document = parseSequenceDocument({
contents: `LOCUS FRAMES 40 bp DNA linear
ACCESSION FRAMES
FEATURES Location/Qualifiers
CDS join(1..4,11..17,21..26)
/gene="forward"
/codon_start=1
CDS complement(join(2..5,12..17,22..29))
/gene="reverse"
/codon_start=2
misc_feature 30..35
ORIGIN
1 aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa
//`,
fileName: "compound-cds.gb",
});
const record = document.records[0];
if (record == null) throw new Error("Expected coding features.");
const exported = exportSequenceWorkbench({
document,
format: "gtf",
recordId: record.id,
scope: "all",
});
const rows = exported.content
.trimEnd()
.split("\n")
.map((line) => line.split("\t"));
expect(
rows
.filter((fields) => fields[8]?.includes('gene_id "forward"'))
.map((fields) => [fields[3], fields[4], fields[7]]),
).toEqual([
["1", "4", "0"],
["11", "17", "2"],
["21", "26", "1"],
]);
expect(
rows
.filter((fields) => fields[8]?.includes('gene_id "reverse"'))
.map((fields) => [fields[3], fields[4], fields[7]]),
).toEqual([
["22", "29", "1"],
["12", "17", "2"],
["2", "5", "2"],
]);
expect(rows.find((fields) => fields[2] === "misc_feature")?.[7]).toBe(".");
});
it("preserves Stockholm GF, repeated GS, GR, GC, and RNA secondary structure", () => {
const document = annotatedStockholmDocument();
const exported = exportAlignmentWorkbench({
document,
format: "stockholm",
scope: "all",
visibleRows: document.rows,
});
const reparsed = parseMsa(exported.content, exported.name);
expect(exported.content).toContain("#=GF AC RF04178");
expect(exported.content.match(/^#=GF AU /gmu)).toHaveLength(2);
expect(exported.content.match(/^#=GS AE015928\.1 DR /gmu)).toHaveLength(2);
expect(exported.content).toContain("#=GR AE015928.1 PP 999887766555");
expect(exported.content).toContain("#=GC SS_cons <<<<<..>>>>>");
expect(reparsed.status).toBe("success");
if (reparsed.status !== "success") throw new Error(reparsed.message);
expect(reparsed.document.formatMetadata).toEqual(document.formatMetadata);
expect(reparsed.document.rows[0]?.metadata).toEqual(document.rows[0]?.metadata);
expect(reparsed.document.annotations).toHaveLength(document.annotations.length);
expect(reparsed.document.rnaStructure?.pairs).toHaveLength(5);
});
it("limits Stockholm row annotations and slices every selected annotation column", () => {
const document = annotatedStockholmDocument();
const selectedRow = document.rows[1];
if (selectedRow == null) throw new Error("Expected a selected RNA row.");
const exported = exportAlignmentWorkbench({
document,
format: "stockholm",
scope: "selection",
selectedColumns: { end: 10, start: 2 },
selectedRows: [selectedRow.id],
visibleRows: document.rows,
});
const reparsed = parseMsa(exported.content, exported.name);
expect(exported.content).not.toContain("AE015928.1");
expect(exported.content).toContain("#=GS CP000139.1 DE Second public RNA");
expect(exported.content).toContain("#=GR CP000139.1 PP 11122233");
expect(exported.content).toContain("#=GC SS_cons <<<..>>>");
expect(exported.content).toContain("#=GC RF AAGUAAaa");
expect(reparsed.status).toBe("success");
if (reparsed.status !== "success") throw new Error(reparsed.message);
expect(reparsed.document.rows).toHaveLength(1);
expect(reparsed.document.alignedLength).toBe(8);
expect(reparsed.document.annotations.every(({ values }) => values.length === 8)).toBe(
true,
);
});
it.each(["csv", "tsv"] as const)(
"neutralizes spreadsheet formulas in %s feature export fields",
(format) => {
const original = annotatedDocument();
const record = original.records[0];
if (record == null) throw new Error("Expected a sequence record.");
const document = {
...original,
records: [
{
...record,
features: [
{
...record.features[0]!,
id: "@SUM(1,2)",
label: "=HYPERLINK(\"https://invalid.example\")",
type: "+cmd",
},
],
sourceLabel: "\uFEFF-2+3",
},
],
};
const exported = exportSequenceWorkbench({
document,
format,
recordId: record.id,
scope: "all",
});
expect(exported.content).toContain("'\uFEFF-2+3");
expect(exported.content).toContain("'@SUM(1,2)");
expect(exported.content).toContain("'+cmd");
expect(exported.content).toContain("'=HYPERLINK(");
expect(exported.content).not.toMatch(/(?:^|\t),?=HYPERLINK/mu);
},
);
it("neutralizes spreadsheet formulas in alignment TSV row IDs, labels, and residues", () => {
const original = alignmentDocument();
const document = {
...original,
rows: [
{
...original.rows[0]!,
alignedSequence: "-ACG",
id: "\t=CMD()",
label: "\uFF1D2+2",
},
],
};
const exported = exportAlignmentWorkbench({
document,
format: "tsv",
scope: "all",
visibleRows: document.rows,
});
expect(exported.content).toContain("' =CMD()");
expect(exported.content).toContain("'\uFF1D2+2");
expect(exported.content).toContain("'-ACG");
});
it("exports alignment scopes, SVG, JSON, and Newick", () => {
const document = alignmentDocument();
const visibleRows = document.rows.slice(0, 1);
const fasta = exportAlignmentWorkbench({
document,
format: "aligned-fasta",
scope: "selection",
selectedColumns: { end: 2, start: 0 },
visibleRows,
});
const svg = exportAlignmentWorkbench({
document,
format: "svg",
scope: "visible",
visibleRows,
});
const newick = exportAlignmentWorkbench({
document,
format: "newick",
newick: "('a':0.1,'b':0.1)",
scope: "all",
visibleRows,
});
const reparsed = parseMsa(fasta.content, fasta.name);
expect(reparsed.status).toBe("success");
expect(fasta.content).toContain("AC");
expect(svg.content).toContain("<svg");
expect(newick.content).toBe("('a':0.1,'b':0.1);\n");
});
it("exports selected alignment rows instead of every visible row", () => {
const document = alignmentDocument();
const selectedRow = document.rows[1];
if (selectedRow == null) throw new Error("Expected a selected row.");
const fasta = exportAlignmentWorkbench({
document,
format: "aligned-fasta",
scope: "selection",
selectedRows: [selectedRow.id],
visibleRows: document.rows,
});
expect(fasta.content).toContain(`>${selectedRow.label}\n`);
expect(fasta.content).not.toContain(`>${document.rows[0]?.label}\n`);
});
it.each([
["clustal", "text/x-clustal", /^CLUSTAL W/u, ".aln"],
["stockholm", "text/x-stockholm", /^# STOCKHOLM 1\.0/u, ".sto"],
] as const)(
"round-trips exact selected alignment rows through %s",
(format, mediaType, header, extension) => {
const document = alignmentDocument();
const exported = exportAlignmentWorkbench({
document,
format,
scope: "all",
visibleRows: document.rows,
});
const reparsed = parseMsa(exported.content, exported.name);
expect(exported.mediaType).toBe(mediaType);
expect(exported.name.endsWith(extension)).toBe(true);
expect(exported.content).toMatch(header);
expect(reparsed.status).toBe("success");
if (reparsed.status !== "success") throw new Error(reparsed.message);
expect(
reparsed.document.rows.map(({ alignedSequence }) => alignedSequence),
).toEqual(document.rows.map(({ alignedSequence }) => alignedSequence));
},
);
it("round-trips A3M match columns and lowercase insertion provenance", () => {
const parsed = parseMsa(">alpha\nACgtGT\n>beta\nA-GT\n", "profile.a3m");
if (parsed.status !== "success") throw new Error(parsed.message);
const exported = exportAlignmentWorkbench({
document: parsed.document,
format: "a3m",
scope: "all",
visibleRows: parsed.document.rows,
});
const reparsed = parseMsa(exported.content, exported.name);
expect(exported.mediaType).toBe("text/x-a3m");
expect(exported.name).toMatch(/\.a3m$/u);
expect(exported.content).toContain("ACgtGT");
expect(reparsed.status).toBe("success");
if (reparsed.status !== "success") throw new Error(reparsed.message);
expect(
reparsed.document.rows.map(({ alignedSequence }) => alignedSequence),
).toEqual(
parsed.document.rows.map(({ alignedSequence }) => alignedSequence),
);
expect(reparsed.document.insertions).toEqual(
expect.arrayContaining([
expect.objectContaining({
afterAlignmentColumn: 1,
residues: "gt",
}),
]),
);
});
it("restricts selected JSON to one public VCF variant while retaining all 100 authorized genotypes", () => {
// HG00096 and chr1:10583 rs58108140 are from the immutable public
// hts-specs VCFv4.3 100-sample fixture; 27 rows exercise the same shape.
const original = parseSequenceDocument({
contents: `>1\n${"A".repeat(10_582)}G${"A".repeat(100)}\n>OFF_RECORD_ACCESSION\nCCCCCCCCCCCC\n`,
fileName: "public-reference.fa",
});
const active = original.records[0];
if (active == null) throw new Error("Expected a public chr1 record.");
const document = {
...original,
records: [
{
...active,
features: [
{
end: 10_590,
id: "selected-cds",
qualifiers: { translation: "OFF_SELECTION_PRIVATE_PROTEIN" },
start: 10_580,
strand: "+" as const,
translation: "OFF_SELECTION_PRIVATE_PROTEIN",
type: "CDS",
},
],
},
...original.records.slice(1),
],
recordInventory: {
materializedCount: 2,
totalCount: 2,
truncated: false,
},
};
const sampleNames = [
"HG00096",
...Array.from({ length: 99 }, (_, index) => `HG${String(index + 97).padStart(5, "0")}`),
];
const selectedGenotypes = sampleNames.map((_, index) =>
index === 0 ? "0|1:0.48" : "0|0:0.01",
);
const rows = Array.from({ length: 27 }, (_, index) => {
const reference = index === 26 ? "<1>" : "1";
const position = index === 0 || index === 26 ? 10_583 : 10_583 + index;
const id =
index === 0
? "rs58108140"
: index === 1
? "rs189107123"
: `rs1403379${String(index).padStart(2, "0")}`;
const genotypes =
index === 0
? selectedGenotypes
: sampleNames.map(() => "1|1:OFF_RANGE_PRIVATE_GENOTYPE");
return [reference, position, id, "G", "A", ".", ".", "AC=1", "GT:DS", ...genotypes].join("\t");
});
const track = parseSequenceTrack({
content: [
"##fileformat=VCFv4.3",
"##contig=<ID=1,length=10800>",
"##contig=<ID=<1>,length=10800>",
"##INFO=<ID=AC,Number=A,Type=Integer,Description=Allele count>",
"##INFO=<ID=OFF_RANGE_PRIVATE_INFO,Number=1,Type=String,Description=Private>",
"##FORMAT=<ID=GT,Number=1,Type=String,Description=Genotype>",
"##FORMAT=<ID=DS,Number=1,Type=Float,Description=Dosage>",
["#CHROM", "POS", "ID", "REF", "ALT", "QUAL", "FILTER", "INFO", "FORMAT", ...sampleNames].join("\t"),
...rows,
].join("\n"),
displayName: "100-public-samples.vcf",
format: "vcf",
id: "public-100-samples",
requestedReference: active.sourceLabel,
});
const unrelated = parseSequenceTrack({
content: [
"##fileformat=VCFv4.3",
"##source=UNRELATED_PRIVATE_STUDY",
"#CHROM\tPOS\tID\tREF\tALT\tQUAL\tFILTER\tINFO\tFORMAT\tUNRELATED_PRIVATE_SAMPLE",
"1\t10584\tunrelated-private-variant\tA\tC\t.\t.\t.\tGT\t1|1",
].join("\n"),
displayName: "unrelated.vcf",
format: "vcf",
id: "unrelated-private-track",
requestedReference: active.sourceLabel,
});
const selected = exportSequenceWorkbench({
document,
format: "json",
recordId: active.id,
scope: "selection",
selection: { end: 10_583, recordId: active.id, start: 10_583 },
tracks: [track, unrelated],
});
const payload = JSON.parse(selected.content) as {
document: typeof document;
tracks: Array<typeof track>;
};
expect(payload.document.records).toHaveLength(1);
expect(payload.document.records[0]).toMatchObject({
length: 1,
sequence: "G",
sourceLabel: "1",
features: [{ end: 1, qualifiers: {}, start: 1 }],
});
expect(payload.document.recordInventory).toEqual({
materializedCount: 1,
totalCount: 1,
truncated: false,
});
expect(payload.tracks).toHaveLength(1);
expect(payload.tracks[0]?.mapping.unmatchedReferences).toEqual([]);
expect(payload.tracks[0]?.summary).toMatchObject({
itemCount: 1,
materializedItemCount: 1,
references: ["1"],
});
expect(payload.tracks[0]?.variants).toHaveLength(1);
expect(payload.tracks[0]?.variants?.[0]).toMatchObject({
id: "rs58108140",
position: 10_583,
rawFilter: ".",
reference: "1",
samples: { HG00096: "0|1:0.48" },
});
expect(payload.tracks[0]?.variants?.[0]?.sampleValues).toHaveLength(100);
expect(payload.tracks[0]?.vcfHeader?.sampleNames).toHaveLength(100);
for (const forbidden of [
"OFF_RECORD_ACCESSION",
"OFF_SELECTION_PRIVATE_PROTEIN",
"OFF_RANGE_PRIVATE_GENOTYPE",
"OFF_RANGE_PRIVATE_INFO",
"UNRELATED_PRIVATE_STUDY",
"UNRELATED_PRIVATE_SAMPLE",
"rs189107123",
"rs140337953",
"<1>",
]) {
expect(selected.content).not.toContain(forbidden);
}
const all = JSON.parse(
exportSequenceWorkbench({
document,
format: "json",
recordId: active.id,
scope: "all",
tracks: [track, unrelated],
}).content,
) as { document: typeof document; tracks: Array<typeof track> };
expect(all.document).toEqual(document);
expect(all.tracks[0]?.variants).toHaveLength(27);
expect(all.tracks[1]?.vcfHeader?.sampleNames).toEqual([
"UNRELATED_PRIVATE_SAMPLE",
]);
});
it.each(["fastq", "genbank", "embl", "pdf", "svg"] as const)(
"never exposes unselected sequence or quality in selected %s exports",
(format) => {
const document = parseSequenceDocument({
contents: "@selected-read\nACGTACGT\n+\n!\"#$%&'(\n@private-read\nTTTTAAAA\n+\nIIIIIIII\n",
fileName: "selection.fastq",
});
const record = document.records[0];
if (record == null) throw new Error("Expected a public FASTQ record.");
const exported = exportSequenceWorkbench({
document,
format,
recordId: record.id,
scope: "selection",
selection: { end: 4, recordId: record.id, start: 3 },
});
expect(exported.content).not.toContain("ACGTACGT");
expect(exported.content).not.toContain("TTTTAAAA");
expect(exported.content).not.toContain("private-read");
expect(exported.content).not.toContain("!\"#$%&'(");
if (format === "fastq") {
expect(exported.content).toContain("\nGT\n+\n#$");
}
if (format === "genbank" || format === "embl") {
expect(
parseSequenceDocument({
contents: exported.content,
fileName: exported.name,
}).records[0]?.sequence,
).toBe("GT");
}
if (format === "pdf") expect(exported.content).toContain("Sequence: GT");
if (format === "svg") expect(exported.content).toContain("2 residues");
},
);
it("fails closed when a selected sequence export has no active-record selection", () => {
const document = annotatedDocument();
expect(() =>
exportSequenceWorkbench({
document,
format: "json",
recordId: document.records[0]?.id ?? "",
scope: "selection",
}),
).toThrow(/selection on the active record/u);
});
it("projects selected Stockholm JSON rows, annotations, insertions, and RNA pairs", () => {
const original = annotatedStockholmDocument();
const hiddenRow = original.rows[0];
const selectedRow = original.rows[1];
if (hiddenRow == null || selectedRow == null) {
throw new Error("Expected two public Rfam RNA rows.");
}
const document = {
...original,
insertions: [
{
afterAlignmentColumn: 4,
residues: "SELECTED_INSERTION",
rowId: selectedRow.id,
sourceKind: "other" as const,
sourceRowIndex: 1,
},
{
afterAlignmentColumn: 4,
residues: "HIDDEN_ROW_PRIVATE_INSERTION",
rowId: hiddenRow.id,
sourceKind: "other" as const,
},
{
afterAlignmentColumn: 11,
residues: "OFF_COLUMN_PRIVATE_INSERTION",
rowId: selectedRow.id,
sourceKind: "other" as const,
},
],
};
const exported = exportAlignmentWorkbench({
document,
format: "json",
scope: "selection",
selectedColumns: { end: 10, start: 2 },
selectedRows: [selectedRow.id],
visibleRows: document.rows,
});
const payload = JSON.parse(exported.content) as {
document: typeof document;
};
expect(payload.document.rows).toHaveLength(1);
expect(payload.document.rows[0]).toMatchObject({
alignedSequence: "AAGUAAAA",
id: selectedRow.id,
ungappedLength: 8,
});
expect(payload.document.alignedLength).toBe(8);
expect(payload.document.rawSummary).toMatchObject({
maxLabelLength: selectedRow.label.length,
sequenceCount: 1,
visibleSequenceCount: 1,
});
expect(payload.document.insertions).toEqual([
expect.objectContaining({
afterAlignmentColumn: 2,
residues: "SELECTED_INSERTION",
sourceRowIndex: 0,
}),
]);
expect(payload.document.annotations.every(({ values }) => values.length === 8)).toBe(true);
expect(payload.document.rnaStructure?.rawStructure).toBe("<<<..>>>");
expect(payload.document.rnaStructure?.referenceTrack).toBe("AAGUAAaa");
expect(payload.document.rnaStructure?.pairs).toEqual(
expect.arrayContaining([
expect.objectContaining({ leftColumn: 0, rightColumn: 7 }),
expect.objectContaining({ leftColumn: 1, rightColumn: 6 }),
expect.objectContaining({ leftColumn: 2, rightColumn: 5 }),
]),
);
expect(exported.content).not.toContain(hiddenRow.id);
expect(exported.content).not.toContain("HIDDEN_ROW_PRIVATE_INSERTION");
expect(exported.content).not.toContain("OFF_COLUMN_PRIVATE_INSERTION");
expect(exported.content).not.toContain("999887766555");
expect(exported.content).not.toContain("<<<<<..>>>>>");
const all = JSON.parse(
exportAlignmentWorkbench({
document,
format: "json",
scope: "all",
visibleRows: document.rows,
}).content,
) as { document: typeof document };
expect(all.document).toEqual(document);
});
it.each(["selection", "visible"] as const)(
"rejects %s Newick exports that would disclose unselected guide-tree labels",
(scope) => {
const document = alignmentDocument();
expect(() =>
exportAlignmentWorkbench({
document,
format: "newick",
newick: "('a':0.1,'PRIVATE_HIDDEN_SAMPLE':0.1)",
scope,
selectedRows: [document.rows[0]?.id ?? ""],
visibleRows: document.rows.slice(0, 1),
}),
).toThrow(/requires all scope/u);
},
);
});
function annotatedDocument() {
return parseSequenceDocument({
contents: `LOCUS demo 12 bp DNA circular
ACCESSION demo
FEATURES Location/Qualifiers
misc_feature complement(join(1..3,7..9))
/note="important"
ORIGIN
1 acgtacgtacgt
//`,
fileName: "demo.gb",
});
}
function alignmentDocument() {
const result = parseMsa(">a\nACGT\n>b\nA-GT\n", "demo.aln-fasta");
if (result.status !== "success") throw new Error(result.message);
return result.document;
}
function annotatedStockholmDocument() {
const result = parseMsa(
[
"# STOCKHOLM 1.0",
"#=GF AC RF04178",
"#=GF AU Prezza, G",
"#=GF AU Ryan, D",
"#=GS AE015928.1 DE First public RNA",
"#=GS AE015928.1 DR PDB; FIRST;",
"#=GS AE015928.1 DR PDB; SECOND;",
"#=GS CP000139.1 DE Second public RNA",
"AE015928.1 GUAAGUAAAAGU",
"CP000139.1 GUAAGUAAAAGU",
"#=GR AE015928.1 PP 999887766555",
"#=GR CP000139.1 PP 001112223344",
"#=GC SS_cons <<<<<..>>>>>",
"#=GC RF guAAGUAAaaGU",
"//",
"",
].join("\n"),
"rfam-rna-selection.sto",
);
if (result.status !== "success") throw new Error(result.message);
return result.document;
}
SHA-256: 450e4bdbb38afbd41499e1d98ba4cc5346d7550aa30b54c1954657b0a72f6bd3