← Files Biological Sequence & Alignment ViewerARCHIVED FILE
src/sequence/formats/snapgene.test.ts
19.5 KB · Sep 30, 2026 · 23:01 UTC
import { describe, expect, it } from "vitest";
import { SEQUENCE_VIEWER_LIMITS } from "../../runtime-contract";
import { parseSnapGeneRecord } from "./snapgene";
const encoder = new TextEncoder();
function concatenate(...pieces: Array<Uint8Array>): Uint8Array {
const result = new Uint8Array(
pieces.reduce((sum, piece) => sum + piece.length, 0),
);
let offset = 0;
for (const piece of pieces) {
result.set(piece, offset);
offset += piece.length;
}
return result;
}
function packet(type: number, payload: string | Uint8Array): Uint8Array {
const body = typeof payload === "string" ? encoder.encode(payload) : payload;
const header = new Uint8Array(5);
header[0] = type;
new DataView(header.buffer).setUint32(1, body.length);
return concatenate(header, body);
}
function cookie(
exportVersion = 15,
importVersion = 19,
sequenceType = 1,
): Uint8Array {
const body = new Uint8Array(14);
body.set(encoder.encode("SnapGene"));
const view = new DataView(body.buffer);
view.setUint16(8, sequenceType);
view.setUint16(10, exportVersion);
view.setUint16(12, importVersion);
return packet(9, body);
}
function sequencePacket(
sequence = "ATGAAACCCGGG",
circular = false,
): Uint8Array {
return packet(
0,
concatenate(Uint8Array.of(circular ? 1 : 0), encoder.encode(sequence)),
);
}
function document(...packets: Array<Uint8Array>): Uint8Array {
return concatenate(cookie(), sequencePacket(), ...packets);
}
describe("native SnapGene import", () => {
it("reads genuine TLV headers from a nonzero byte offset without disclosing a path", () => {
const bytes = concatenate(
Uint8Array.of(255, 254),
document(),
Uint8Array.of(253),
);
const result = parseSnapGeneRecord({
bytes: bytes.subarray(2, bytes.length - 1),
fileName: "/private/lab/project/vector.dna",
});
expect(result.record).toMatchObject({
id: "vector",
sourceLabel: "vector",
length: 12,
sequence: "ATGAAACCCGGG",
molecule: "dna",
topology: "linear",
features: [],
metadata: { snapgeneExportVersion: "15", snapgeneImportVersion: "19" },
});
expect(result.warnings).toEqual([]);
expect(JSON.stringify(result)).not.toContain("/private/lab");
});
it.each([
[13, 11],
[13, 12],
[14, 16],
[15, 19],
[15, 20],
])(
"accepts reference-tested header revision %i:%i",
(exportVersion, importVersion) => {
expect(
parseSnapGeneRecord({
bytes: concatenate(
cookie(exportVersion, importVersion),
sequencePacket("acgtryswkmbdhvn", true),
),
}).record,
).toMatchObject({ sequence: "ACGTRYSWKMBDHVN", topology: "circular" });
},
);
it("imports annotations before DNA, compound negative-strand CDS, notes, and qualifiers", () => {
const bytes = concatenate(
cookie(),
packet(
10,
`<?xml version="1.0" encoding="UTF-8"?>
<Features nextValidID="1"><Feature recentID="0" name="enzyme & tag" type="CDS" directionality="2">
<Segment range="1-3" name="second exon" color="#aabbcc"/>
<Segment range="7-12" name="first exon"/>
<Q name="codon_start"><V int="1"/></Q><Q name="transl_table"><V int="1"/></Q>
<Q name="note"><V text="first"/><V predef="second"/></Q>
</Feature></Features>`,
),
packet(
6,
`<Notes><AccessionNumber>TEST01</AccessionNumber><Description><html><body>Demo α</body></html></Description><Created UTC="12:00">2026.8.21</Created></Notes>`,
),
sequencePacket(),
);
const { record, warnings } = parseSnapGeneRecord({ bytes });
expect(record.description).toBe("Demo α");
expect(record.metadata).toMatchObject({
"snapgene.AccessionNumber": "TEST01",
"snapgene.Created.UTC": "12:00",
});
expect(record.features[0]).toMatchObject({
type: "CDS",
label: "enzyme & tag",
start: 1,
end: 12,
strand: "-",
segments: [
{ start: 7, end: 12 },
{ start: 1, end: 3 },
],
sourceLocation: "complement(join(1..3,7..12))",
translation: "PGH",
translationTrackReliable: true,
qualifiers: {
note: ["first", "second"],
snapgene_segment_color: "1-3:#aabbcc",
snapgene_segment_name: ["1-3:second exon", "7-12:first exon"],
},
});
expect(
record.features[0]?.translationCoordinateMap?.map(
(codon) => codon.codonCoordinates,
),
).toEqual([
[12, 11, 10],
[9, 8, 7],
[3, 2, 1],
]);
expect(
warnings.some((warning) => warning.code === "snapgene-rich-text"),
).toBe(true);
});
it("preserves forward and reverse origin-spanning feature paths without flattening them", () => {
const bytes = concatenate(
cookie(),
sequencePacket("ATGAAACCCGGG", true),
packet(
10,
`<Features><Feature name="forward" type="CDS" directionality="1"><Segment range="10-3"/></Feature><Feature name="reverse" type="CDS" directionality="2"><Segment range="10-3"/></Feature></Features>`,
),
);
const { record } = parseSnapGeneRecord({ bytes });
expect(record.features[0]).toMatchObject({
segments: [
{ start: 10, end: 12 },
{ start: 1, end: 3 },
],
translation: "GM",
sourceLocation: "join(10..12,1..3)",
});
expect(record.features[1]).toMatchObject({
segments: [
{ start: 1, end: 3 },
{ start: 10, end: 12 },
],
translation: "HP",
sourceLocation: "complement(join(10..12,1..3))",
});
});
it("treats equal range endpoints as one base and supports full circular paths", () => {
const bytes = concatenate(
cookie(),
sequencePacket("ACGT", true),
packet(
10,
`<Features><Feature><Segment range="2-2"/></Feature><Feature><Segment range="4-3"/></Feature></Features>`,
),
);
expect(
parseSnapGeneRecord({ bytes }).record.features.map(
(feature) => feature.segments,
),
).toEqual([
[{ start: 2, end: 2 }],
[
{ start: 4, end: 4 },
{ start: 1, end: 3 },
],
]);
});
it("imports primer sites using their zero-based inclusive coordinates and filters hidden duplicates", () => {
const bytes = document(
packet(
5,
`<Primers><HybridizationParams minContinuousMatchLen="3" minMeltingTemperature="40"/>
<Primer name="Forward" sequence="AAATGAAA" description="a primer">
<BindingSite location="0-5" boundStrand="0" annealedBases="ATGAAA" meltingTemperature="55"><Component bases="AA"/></BindingSite>
<BindingSite location="0-5" boundStrand="0" simplified="1" annealedBases="ATGAAA" meltingTemperature="55"/>
<BindingSite location="6-7" boundStrand="1" annealedBases="CC" meltingTemperature="12"/>
</Primer><Primer name="Reverse" sequence="CCC"><BindingSite location="9-11" boundStrand="1" annealedBases="CCC" meltingTemperature="50"/></Primer>
</Primers>`,
),
);
const { record, warnings } = parseSnapGeneRecord({ bytes });
expect(record.features).toHaveLength(2);
expect(record.features[0]).toMatchObject({
type: "primer_bind",
label: "Forward",
start: 1,
end: 6,
strand: "+",
qualifiers: {
snapgene_sequence: "AAATGAAA",
snapgene_primer_location: "0-5",
},
});
expect(record.features[1]).toMatchObject({
label: "Reverse",
start: 10,
end: 12,
strand: "-",
});
expect(warnings.map((warning) => warning.code)).toEqual(
expect.arrayContaining([
"snapgene-simplified-primer-sites",
"snapgene-hidden-primer-sites",
"unsupported-snapgene-xml-property",
]),
);
});
it("imports origin-spanning primers and warns about missing filter measurements", () => {
const bytes = concatenate(
cookie(),
sequencePacket("ACGT", true),
packet(
5,
`<Primers><HybridizationParams minContinuousMatchLen="2" minMeltingTemperature="30"/><Primer name="wrap"><BindingSite location="3-0" boundStrand="0"/></Primer><Primer name="unbound" sequence="AAA"/></Primers>`,
),
);
const { record, warnings } = parseSnapGeneRecord({ bytes });
expect(record.features[0]?.segments).toEqual([
{ start: 4, end: 4 },
{ start: 1, end: 1 },
]);
expect(warnings.map((warning) => warning.code)).toEqual(
expect.arrayContaining([
"snapgene-primer-filter-unknown",
"snapgene-unbound-primer",
]),
);
});
it("only suppresses a simplified primer site when a visible detailed counterpart exists", () => {
const { record } = parseSnapGeneRecord({
bytes: document(
packet(
5,
`<Primers><HybridizationParams minMeltingTemperature="40"/><Primer name="detailed hidden"><BindingSite location="0-2" simplified="1" meltingTemperature="55"/><BindingSite location="0-2" meltingTemperature="12"/></Primer><Primer name="detailed visible"><BindingSite location="9-11" simplified="1" meltingTemperature="55"/><BindingSite location="9-11" meltingTemperature="55"/></Primer></Primers>`,
),
),
});
expect(
record.features.map(({ label, start, end }) => ({ label, start, end })),
).toEqual([
{ label: "detailed hidden", start: 1, end: 3 },
{ label: "detailed visible", start: 10, end: 12 },
]);
});
it("normalizes qualifier linebreaks without dropping their text", () => {
const { record } = parseSnapGeneRecord({
bytes: document(
packet(
10,
`<Features><Feature><Segment range="1-3"/><Q name="note"><V text="line one line two line three"/></Q></Feature></Features>`,
),
),
});
expect(record.features[0]?.qualifiers.note).toBe(
"line one line two line three",
);
});
it("retains incomplete literal markup without repeatedly scanning an untrusted string", () => {
const literal = "<a".repeat(20_000);
const { record } = parseSnapGeneRecord({
bytes: document(
packet(6, `<Notes><Description>${literal}</Description></Notes>`),
),
});
expect(record.description).toBe("<a".repeat(20_000));
});
it("keeps unknown directional or segment semantics explicit without inventing translation", () => {
const { record, warnings } = parseSnapGeneRecord({
bytes: document(
packet(
10,
`<Features><Feature directionality="3" type="CDS"><Segment range="1-3"/><Q name="translation"><V text="M"/></Q></Feature><Feature directionality="1" type="CDS"><Segment range="1-3" type="run-on-translation"/><Segment range="4-6" type="gap"/></Feature></Features>`,
),
),
});
expect(record.features[0]).toMatchObject({
strand: "?",
translation: "M",
translationTrackReliable: false,
});
expect(record.features[1]).toMatchObject({
segments: [{ start: 1, end: 3 }],
translationTrackReliable: false,
qualifiers: { snapgene_gap_range: "4-6" },
});
expect(record.features[1]?.translation).toBeUndefined();
expect(warnings.map((warning) => warning.code)).toEqual(
expect.arrayContaining([
"unsupported-snapgene-directionality",
"unsupported-snapgene-segment",
"snapgene-gap-segments",
]),
);
});
it("reports unsupported packet and XML properties rather than silently discarding them", () => {
const { warnings } = parseSnapGeneRecord({
bytes: document(
packet(0xfe, Uint8Array.of(1, 2, 3)),
packet(0xfe, Uint8Array.of(4)),
packet(
10,
`<Features display="new"><Unknown/><Feature name="feature" hidden="1"><Segment range="1-3"/><Q name="note"><V unknown="value"/></Q></Feature></Features>`,
),
),
});
expect(
warnings.filter(
(warning) => warning.code === "unsupported-snapgene-packet",
),
).toHaveLength(1);
expect(warnings.map((warning) => warning.code)).toEqual(
expect.arrayContaining([
"unsupported-snapgene-packet",
"unsupported-snapgene-xml-property",
"unsupported-snapgene-qualifier",
]),
);
});
it("treats prototype-shaped qualifier names as ordinary data", () => {
const { record } = parseSnapGeneRecord({
bytes: document(
packet(
10,
`<Features><Feature><Segment range="1-3"/><Q name="__proto__"><V text="literal"/></Q><Q name="constructor"><V text="also literal"/></Q></Feature></Features>`,
),
),
});
expect(
Object.getOwnPropertyDescriptor(
record.features[0]?.qualifiers,
"__proto__",
)?.value,
).toBe("literal");
expect(record.features[0]?.qualifiers.constructor).toBe("also literal");
expect(Object.getPrototypeOf(record.features[0]?.qualifiers)).toBe(
Object.prototype,
);
});
it("supports comments, CDATA, numeric entities, and quoted greater-than signs without a DOM", () => {
const { record } = parseSnapGeneRecord({
bytes: document(
packet(
6,
`<?xml version='1.0'?><!--source--><Notes><Description><![CDATA[ATG & text]]></Description><Comments>one "two" β</Comments></Notes>`,
),
packet(
10,
`<Features><Feature name="A > B"><Segment range='1-3'/></Feature></Features>`,
),
),
});
expect(record.description).toBe("ATG & text");
expect(record.metadata["snapgene.Comments"]).toBe('one "two" β');
expect(record.features[0]?.label).toBe("A > B");
});
it("bounds computed CDS coordinate-map materialization while preserving imported translations", () => {
const sequence = "ATG".repeat(166_667);
const feature = `<Feature directionality="1" type="CDS"><Segment range="1-${sequence.length}"/><Q name="translation"><V text="M"/></Q></Feature>`;
const { record, warnings } = parseSnapGeneRecord({
bytes: concatenate(
cookie(),
sequencePacket(sequence),
packet(10, `<Features>${feature}${feature}</Features>`),
),
});
expect(record.features[0]?.translationTrackReliable).toBe(true);
expect(record.features[1]).toMatchObject({
translation: "M",
translationTrackReliable: false,
translationMappingUnavailableReason:
"bounded SnapGene import translation budget",
});
expect(
warnings.some(
(warning) => warning.code === "snapgene-translation-budget",
),
).toBe(true);
});
});
describe("SnapGene untrusted-input validation", () => {
it.each([
new Uint8Array(),
Uint8Array.of(9),
packet(9, "SnapGene"),
packet(8, "not a cookie"),
concatenate(cookie(99, 19), sequencePacket()),
concatenate(cookie(15, 99), sequencePacket()),
concatenate(cookie(15, 19, 2), sequencePacket()),
concatenate(cookie(), cookie(), sequencePacket()),
concatenate(cookie(), sequencePacket(), sequencePacket()),
cookie(),
concatenate(cookie(), packet(0, Uint8Array.of(1))),
concatenate(cookie(), sequencePacket("ACGTX")),
concatenate(cookie(), sequencePacket("ACGTſ")),
concatenate(cookie(), packet(0, Uint8Array.of(0, 0xff))),
])(
"rejects invalid cookies, revisions, sequence types, duplicate or invalid DNA packets (%#)",
(bytes) => {
expect(() => parseSnapGeneRecord({ bytes })).toThrow();
},
);
it("validates the cookie magic even when length and type match", () => {
const bytes = document();
bytes[5] = "s".charCodeAt(0);
expect(() => parseSnapGeneRecord({ bytes })).toThrow(/cookie/);
});
it("rejects truncated trailing headers and unsigned packet lengths beyond EOF", () => {
expect(() =>
parseSnapGeneRecord({
bytes: concatenate(document(), Uint8Array.of(10, 0, 0)),
}),
).toThrow(/Truncated/);
expect(() =>
parseSnapGeneRecord({
bytes: concatenate(document(), Uint8Array.of(10, 255, 255, 255, 255)),
}),
).toThrow(/beyond/);
const bytes = document(packet(6, "<Notes/>"));
expect(() =>
parseSnapGeneRecord({ bytes: bytes.subarray(0, bytes.length - 1) }),
).toThrow(/beyond/);
});
it.each([
"<Features>",
"<Features><Feature></Features>",
"<Features/><Features/>",
"<Features x='1' x='2'/>",
"<Features x='1'y='2'/>",
"<Features x='a<b'/>",
"<Features>&unknown;</Features>",
"<Features>�</Features>",
"<Features>�</Features>",
"<Features>�</Features>",
"<Features>&</Features>",
"<!DOCTYPE Features [<!ENTITY secret SYSTEM 'file:///private/secret'>]><Features>&secret;</Features>",
"<?fetch remote?><Features/>",
"<Features>\u0000</Features>",
"<Features/><!--broken--comment-->",
"<Features xmlns:test='remote'/>",
"<Features><Feature><Segment range='1-3'/><Q name='n'><V int='1e10'/></Q></Feature></Features>",
])(
"rejects malformed or unsupported XML without external resolution (%#)",
(xml) => {
expect(() =>
parseSnapGeneRecord({ bytes: document(packet(10, xml)) }),
).toThrow();
},
);
it.each([
"0-3",
"1-13",
"4-2",
"2",
"1-2-3",
"1.5-3",
"999999999999999999999-3",
])("rejects invalid or out-of-bounds feature locations %s", (range) => {
expect(() =>
parseSnapGeneRecord({
bytes: document(
packet(
10,
`<Features><Feature><Segment range="${range}"/></Feature></Features>`,
),
),
}),
).toThrow();
});
it("rejects malformed feature/primer structures and mislabeled XML packets", () => {
for (const bytes of [
document(packet(10, "<Notes/>")),
document(packet(10, "<Features><Feature/></Features>")),
document(
packet(
10,
"<Features><Feature><Segment range='1-3'/><Q><V text='a'/></Q></Feature></Features>",
),
),
document(
packet(
5,
"<Primers><Primer><BindingSite location='0-12'/></Primer></Primers>",
),
),
document(
packet(
5,
"<Primers><Primer><BindingSite location='0-1' boundStrand='3'/></Primer></Primers>",
),
),
document(
packet(
5,
"<Primers><HybridizationParams/><HybridizationParams/></Primers>",
),
),
document(packet(6, "<Notes/>"), packet(6, "<Notes/>")),
]) {
expect(() => parseSnapGeneRecord({ bytes })).toThrow();
}
});
it.each([
"<Features\u00a0value='x'/>",
"<!--before--><?xml version='1.0'?><Features/>",
"<Features><Feature type='CDS'><Segment range='1-3'/><Q name='codon_start'><V int='0'/></Q></Feature></Features>",
"<Features><Feature type='CDS'><Segment range='1-3'/><Q name='transl_table'><V text='not a code'/></Q></Feature></Features>",
])(
"rejects malformed XML declaration/whitespace or invalid CDS qualifiers (%#)",
(xml) => {
expect(() =>
parseSnapGeneRecord({ bytes: document(packet(10, xml)) }),
).toThrow();
},
);
it("rejects invalid UTF-8 annotations", () => {
expect(() =>
parseSnapGeneRecord({ bytes: document(packet(10, Uint8Array.of(0xff))) }),
).toThrow();
});
it("enforces the input and XML nesting budgets before materializing an unbounded document", () => {
expect(() =>
parseSnapGeneRecord({
bytes: new Uint8Array(
SEQUENCE_VIEWER_LIMITS.session.maxArtifactBytes + 1,
),
}),
).toThrow(/byte limit/);
const nested = `<Features>${"<x>".repeat(32)}${"</x>".repeat(32)}</Features>`;
expect(() =>
parseSnapGeneRecord({ bytes: document(packet(10, nested)) }),
).toThrow(/nesting limit/);
});
});
SHA-256: 34475b333637ec28d6a2991084e9efd3ee449a54a1aea5d2175e7f8acdfb355a