seqcolyte / spec /technologies /cel_seq2.json
seqmachines's picture
Deploy Seqcolyte Studio
bd7cf06 verified
Raw
History Blame Contribute Delete
31.9 kB
{
"schema_version": "seqcolyte.spec.v1",
"spec_id": "cel_seq2",
"assay": "CEL-Seq2",
"chemistry_version": "",
"platform": "illumina",
"platform_params": {
"read_type": "short"
},
"source_docs": [
{
"doc_id": "CEL-seq2_detailed_protocol.docx",
"title": "CEL-Seq2 detailed protocol (Additional file 4)",
"url": "https://doi.org/10.1186/s13059-016-0938-8",
"path": "/Users/seqmachines/playground/protocols-test/protocols/cel_seq2/CEL-seq2_detailed_protocol.docx",
"retrieved_date": null
},
{
"doc_id": "CEL-seq2_supp.pdf",
"title": "CEL-seq2_supp.pdf",
"url": null,
"path": "/Users/seqmachines/playground/protocols-test/protocols/cel_seq2/CEL-seq2_supp.pdf",
"retrieved_date": null
},
{
"doc_id": "CEL-seq2.pdf",
"title": "CEL-seq2.pdf",
"url": null,
"path": "/Users/seqmachines/playground/protocols-test/protocols/cel_seq2/CEL-seq2.pdf",
"retrieved_date": null
}
],
"oligos": [
{
"oligo_id": "oligo_cel_seq2_rt_primer",
"name": "CEL-Seq2 primer (barcoded RT primer)",
"aliases": [],
"role": "Barcoded reverse-transcription / capture primer with T7 promoter, Illumina 5' adapter, UMI, cell barcode and anchored poly(dT)",
"kind": "assembled",
"sequence": "GCCGGTAATACGACTCACTATAGGGAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV",
"direction": "5_to_3",
"components": [
{
"name": "T7 promoter",
"sequence": "GCCGGTAATACGACTCACTATAGGG",
"role": "T7 RNA polymerase promoter (transcription start at the terminal GGG); GCCGG 5' leader"
},
{
"name": "Partial Illumina 5' adapter (RA5)",
"sequence": "AGTTCTACAGTCCGACGATC",
"role": "Shortened Illumina small-RNA 5' adapter; becomes the Read 1 primer landing site"
},
{
"name": "UMI",
"sequence": "[UMI:6]",
"role": "6-nt unique molecular identifier (NNNNNN), upstream of the barcode"
},
{
"name": "Cell barcode",
"sequence": "[CELL_BARCODE:6]",
"role": "6-nt sample/cell barcode (Hamming distance >=2); one per well"
},
{
"name": "Poly(dT)",
"sequence": "TTTTTTTTTTTTTTTTTTTTTTTT",
"role": "24-nt poly(dT) to prime poly(A) mRNA"
},
{
"name": "Anchor (V)",
"sequence": "V",
"role": "3' anchor base (A/C/G) that seats the primer at the poly(A) junction"
}
],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "82-nt primer (shortened from CEL-Seq's 92 nt). 96 barcoded variants are listed in the detailed protocol; e.g. barcode 1 = AGACTC, 4 = AGCTTC, 5 = CATGAG, 46 = TGCAGA. Recommended 10-primer pool: barcodes 1,4,5,9,10,23,25,26,31,46. 168 unique 6-nt barcodes were designed (GC 33-67%, last base != T)."
},
{
"oligo_id": "oligo_cel_seq2_barcode_primer_1s",
"name": "CEL-Seq2 primer 1s (barcode AGACTC)",
"aliases": [],
"role": "Representative fully-specified barcoded RT primer (well 1)",
"kind": "single",
"sequence": "GCCGGTAATACGACTCACTATAGGGAGTTCTACAGTCCGACGATCNNNNNNAGACTCTTTTTTTTTTTTTTTTTTTTTTTTV",
"direction": "5_to_3",
"components": [],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "NNNNNN = 6-nt UMI; AGACTC = 6-nt cell barcode for well 1. Concrete example of the CEL-Seq2 primer series (1s-96s) transcribed verbatim from the detailed protocol table."
},
{
"oligo_id": "oligo_cel_seq_original_primer",
"name": "CEL-Seq primer (original, 8-nt barcode)",
"aliases": [],
"role": "Original CEL-Seq barcoded RT primer (no UMI)",
"kind": "assembled",
"sequence": "CGATTGAGGCCGGTAATACGACTCACTATAGGGGTTCAGAGTTCTACAGTCCGACGATC[CELL_BARCODE:8]TTTTTTTTTTTTTTTTTTTTTTTTV",
"direction": "5_to_3",
"components": [
{
"name": "5' leader + T7 promoter",
"sequence": "CGATTGAGGCCGGTAATACGACTCACTATAGGG",
"role": "T7 promoter with extended 5' leader"
},
{
"name": "Illumina 5' adapter (RA5)",
"sequence": "GTTCAGAGTTCTACAGTCCGACGATC",
"role": "Full-length Illumina small-RNA 5' adapter"
},
{
"name": "Cell barcode",
"sequence": "[CELL_BARCODE:8]",
"role": "8-nt cell barcode (as previously published)"
},
{
"name": "Poly(dT)",
"sequence": "TTTTTTTTTTTTTTTTTTTTTTTT",
"role": "24-nt poly(dT)"
},
{
"name": "Anchor (V)",
"sequence": "V",
"role": "3' anchor base"
}
],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "Original CEL-Seq design (Table S2); 92 nt, longer T7 promoter and 5' adapter, no UMI."
},
{
"oligo_id": "oligo_cel_seq_umi_primer",
"name": "CEL-Seq + UMI primer (5-nt UMI, 6-nt barcode)",
"aliases": [],
"role": "Intermediate CEL-Seq primer with 5-nt UMI",
"kind": "assembled",
"sequence": "CGATTGAGGCCGGTAATACGACTCACTATAGGGGTTCAGAGTTCTACAGTCCGACGATC[UMI:5][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV",
"direction": "5_to_3",
"components": [
{
"name": "5' leader + T7 promoter",
"sequence": "CGATTGAGGCCGGTAATACGACTCACTATAGGG",
"role": "T7 promoter with extended 5' leader"
},
{
"name": "Illumina 5' adapter (RA5)",
"sequence": "GTTCAGAGTTCTACAGTCCGACGATC",
"role": "Full-length Illumina small-RNA 5' adapter"
},
{
"name": "UMI",
"sequence": "[UMI:5]",
"role": "5-nt UMI (NNNNN)"
},
{
"name": "Cell barcode",
"sequence": "[CELL_BARCODE:6]",
"role": "6-nt cell barcode"
},
{
"name": "Poly(dT)",
"sequence": "TTTTTTTTTTTTTTTTTTTTTTTT",
"role": "24-nt poly(dT)"
},
{
"name": "Anchor (V)",
"sequence": "V",
"role": "3' anchor base"
}
],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "Table S2; the 'CEL-Seq performed as previously described with a 5-base UMI and 6-base barcode' variant used for comparison."
},
{
"oligo_id": "oligo_library_rt_primer",
"name": "Library RT primer (randomhexRT)",
"aliases": [],
"role": "Random-hexamer RT primer with 5'-tail Illumina 3' adapter; converts fragmented aRNA back to cDNA",
"kind": "assembled",
"sequence": "GCCTTGGCACCCGAGAATTCCANNNNNN",
"direction": "5_to_3",
"components": [
{
"name": "Illumina 3' adapter (RA3, rev-comp orientation)",
"sequence": "GCCTTGGCACCCGAGAATTCCA",
"role": "5'-tail matching Illumina small-RNA 3' adapter (rev-comp of TGGAATTCTCGGGTGCCAAGGC)"
},
{
"name": "Random hexamer",
"sequence": "NNNNNN",
"role": "Random-priming 3' hexamer that anneals across fragmented aRNA"
}
],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "CEL-Seq2 change 5: inserts the Illumina 3' adapter at the RT step via a random hexamer, eliminating the ligation step of the original CEL-Seq. Transcribed verbatim from the protocol/Table S2."
},
{
"oligo_id": "oligo_rna_pcr_primer_rp1",
"name": "RNA PCR Primer (RP1)",
"aliases": [],
"role": "Forward library PCR primer; adds P5 and completes the Illumina 5' adapter",
"kind": "assembled",
"sequence": "AATGATACGGCGACCACCGAGATCTACACGTTCAGAGTTCTACAGTCCGACGATC",
"direction": "5_to_3",
"components": [
{
"name": "P5",
"sequence": "AATGATACGGCGACCACCGAGATCTACAC",
"role": "Illumina P5 flow-cell adapter"
},
{
"name": "Illumina 5' adapter (RA5)",
"sequence": "GTTCAGAGTTCTACAGTCCGACGATC",
"role": "Full 5' small-RNA adapter; 3' end anneals to the AGTTCTACAGTCCGACGATC tag in the fragment"
}
],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "Standard Illumina TruSeq Small-RNA RP1; the documents state 'sequences available from Illumina' and are not printed. Sequence assembled from the verified P5 constant plus the in-document 5' adapter tag."
},
{
"oligo_id": "oligo_rna_pcr_index_primer_rpi",
"name": "RNA PCR Index Primer (RPIX)",
"aliases": [],
"role": "Reverse indexed library PCR primer; adds i7 sample index and P7 via the Illumina 3' adapter",
"kind": "assembled",
"sequence": "CAAGCAGAAGACGGCATACGAGAT[SAMPLE_INDEX:6]GTGACTGGAGTTCCTTGGCACCCGAGAATTCCA",
"direction": "5_to_3",
"components": [
{
"name": "P7",
"sequence": "CAAGCAGAAGACGGCATACGAGAT",
"role": "Illumina P7 flow-cell adapter"
},
{
"name": "i7 sample index",
"sequence": "[SAMPLE_INDEX:6]",
"role": "6-nt sample index (read in the 7-cycle index read)"
},
{
"name": "Illumina 3' adapter (RA3, rev-comp)",
"sequence": "GTGACTGGAGTTCCTTGGCACCCGAGAATTCCA",
"role": "3' portion whose 3' end anneals to the RA3 tag added by the library RT primer"
}
],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "Standard Illumina TruSeq Small-RNA RPIX; not printed in the docs ('sequences from Illumina kit'). Assembled from the verified P7 constant, a 6-nt index, and the in-document 3' adapter. A uniquely indexed RPIX is added per pooled library."
},
{
"oligo_id": "oligo_illumina_ra5_adapter",
"name": "Illumina 5' small-RNA adapter (RA5)",
"aliases": [],
"role": "5' adapter / Read 1 primer landing region of the final library",
"kind": "single",
"sequence": "GTTCAGAGTTCTACAGTCCGACGATC",
"direction": "5_to_3",
"components": [],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "The shortened form AGTTCTACAGTCCGACGATC is carried on the CEL-Seq2 primer; full length restored by RP1 during PCR."
},
{
"oligo_id": "oligo_illumina_ra3_adapter",
"name": "Illumina 3' small-RNA adapter (RA3)",
"aliases": [],
"role": "3' adapter / Read 2 primer landing region of the final library",
"kind": "single",
"sequence": "TGGAATTCTCGGGTGCCAAGGC",
"direction": "5_to_3",
"components": [],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "Reverse complement of the library RT primer 5'-tail (GCCTTGGCACCCGAGAATTCCA)."
},
{
"oligo_id": "oligo_illumina_p5_adapter",
"name": "Illumina P5 adapter",
"aliases": [],
"role": "P5 flow-cell binding sequence",
"kind": "single",
"sequence": "AATGATACGGCGACCACCGAGATCTACAC",
"direction": "5_to_3",
"components": [],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "Standard Illumina constant; contributed by RP1."
},
{
"oligo_id": "oligo_illumina_p7_adapter",
"name": "Illumina P7 adapter",
"aliases": [],
"role": "P7 flow-cell binding sequence",
"kind": "single",
"sequence": "CAAGCAGAAGACGGCATACGAGAT",
"direction": "5_to_3",
"components": [],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "Standard Illumina constant; contributed by RPIX. Appears at the 3' end of the top strand as its reverse complement (ATCTCGTATGCCGTCTTCTGCTTG)."
}
],
"final_library": {
"source_label": "CEL-Seq2 final Illumina small-RNA library (5'->3' top strand)",
"annotated_library_sequence": "AATGATACGGCGACCACCGAGATCTACACGTTCAGAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV[CDNA]TGGAATTCTCGGGTGCCAAGGAACTCCAGTCAC[SAMPLE_INDEX:6]ATCTCGTATGCCGTCTTCTGCTTG",
"library_sequence": "AATGATACGGCGACCACCGAGATCTACACGTTCAGAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV[CDNA]TGGAATTCTCGGGTGCCAAGGAACTCCAGTCAC[SAMPLE_INDEX:6]ATCTCGTATGCCGTCTTCTGCTTG",
"strands": [
{
"direction": "5_to_3",
"source_html": "AATGATACGGCGACCACCGAGATCTACACGTTCAGAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV[CDNA]TGGAATTCTCGGGTGCCAAGGAACTCCAGTCAC[SAMPLE_INDEX:6]ATCTCGTATGCCGTCTTCTGCTTG",
"source_sequence": "AATGATACGGCGACCACCGAGATCTACACGTTCAGAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV[CDNA]TGGAATTCTCGGGTGCCAAGGAACTCCAGTCAC[SAMPLE_INDEX:6]ATCTCGTATGCCGTCTTCTGCTTG"
}
],
"annotation_lines": [
"AATGATACGGCGACCACCGAGATCTACAC = P5",
"GTTCAGAGTTCTACAGTCCGACGATC = Illumina 5' small-RNA adapter (RA5 / Read 1 primer site)",
"[UMI:6] = UMI",
"[CELL_BARCODE:6] = Cell barcode",
"TTTTTTTTTTTTTTTTTTTTTTTT = poly(dT)",
"V = anchor base",
"[CDNA] = cDNA insert",
"TGGAATTCTCGGGTGCCAAGGAACTCCAGTCAC = Illumina 3' small-RNA adapter (RA3 / Read 2 primer site)",
"[SAMPLE_INDEX:6] = i7 sample index (reverse complement)",
"ATCTCGTATGCCGTCTTCTGCTTG = reverse complement of P7"
],
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "CEL-Seq2 final Illumina small-RNA library (5'->3' top strand)",
"method": "claude_llm_extraction"
}
]
},
"read_structure": {
"reads": [
{
"read": "R1",
"primer": "Illumina small-RNA Read 1 sequencing primer",
"template": "bottom",
"cycles": 15,
"segments": [
{
"name": "UMI",
"type": "umi",
"order": 0,
"scored": true,
"provenance": null,
"whitelist_ref": null,
"constant_ref": null,
"notes": null,
"length": 6
},
{
"name": "Cell barcode",
"type": "barcode",
"order": 1,
"scored": true,
"provenance": null,
"whitelist_ref": null,
"constant_ref": null,
"notes": null,
"length": 6
},
{
"name": "poly(dT)",
"type": "constant",
"order": 2,
"scored": false,
"provenance": null,
"whitelist_ref": null,
"constant_ref": null,
"notes": null,
"length": 3
}
]
},
{
"read": "I1",
"primer": "Illumina index sequencing primer",
"template": "top",
"cycles": 7,
"segments": [
{
"name": "i7 sample index",
"type": "index",
"order": 0,
"scored": false,
"provenance": null,
"whitelist_ref": null,
"constant_ref": null,
"notes": null,
"length": 6
}
]
},
{
"read": "R2",
"primer": "Illumina small-RNA Read 2 sequencing primer",
"template": "top",
"cycles": 36,
"segments": [
{
"name": "cDNA insert",
"type": "insert",
"order": 0,
"scored": true,
"provenance": null,
"whitelist_ref": null,
"constant_ref": null,
"notes": null
}
]
}
]
},
"library_generation": [
{
"step": 1,
"title": "Anneal barcoded CEL-Seq2 primer & reverse transcription",
"summary": "The barcoded poly(dT) CEL-Seq2 primer captures a single cell's poly(A) mRNA and SuperScript II reverse-transcribes first-strand cDNA.",
"note": "Each cell/well gets a unique 6-nt barcode; the primer also carries a T7 promoter, the Illumina 5' adapter and a 6-nt UMI. SuperScript II extends from the anchored poly(dT) into the transcript body.",
"product": "5'- GCCGGTAATACGACTCACTATAGGGAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV[CDNA]------->\n 3'- AAAAAAAAAAAAAAAAAAAAAAAA...(mRNA)...-5'"
},
{
"step": 2,
"title": "Second-strand synthesis (dsDNA with T7 promoter)",
"summary": "RNase H, E. coli DNA Pol I and DNA ligase convert the RNA:cDNA hybrid into double-stranded DNA carrying an intact T7 promoter.",
"note": "SuperScript II Double-Stranded cDNA Synthesis Kit reagents. Barcoded samples are then pooled and bead-purified (AMPure XP) before amplification.",
"product": "5'- GCCGGTAATACGACTCACTATAGGGAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV[CDNA] -3'\n3'- CGGCCATTATGCTGAGTGATATCCCTCAAGATGTCAGGCTGCTAG[UMI:6][CELL_BARCODE:6]AAAAAAAAAAAAAAAAAAAAAAAAB[CDNA] -5'"
},
{
"step": 3,
"title": "In vitro transcription (T7 linear amplification)",
"summary": "T7 RNA polymerase transcribes from the promoter, linearly amplifying each molecule into many antisense aRNA copies.",
"note": "IVT (MEGAscript/SuperScript II kit) avoids exponential PCR bias; transcription starts at the terminal GGG so the T7 promoter itself is not copied into the aRNA.",
"product": "5'- GGGAGUUCUACAGUCCGACGAUC[UMI:6][CELL_BARCODE:6]UUUUUUUUUUUUUUUUUUUUUUUUV[CDNA] -3' (aRNA, antisense to mRNA)"
},
{
"step": 4,
"title": "aRNA fragmentation",
"summary": "The amplified RNA is chemically fragmented (Mg2+, heat) to ~200-500 nt.",
"note": "Fragmentation is stopped with EDTA and the aRNA is bead-purified (RNAClean XP). Only fragments retaining the 5' adapter + UMI + barcode end carry the demultiplexing information.",
"product": "5'- GGGAGUUCUACAGUCCGACGAUC[UMI:6][CELL_BARCODE:6]UUUU...V[CDNA frag] -3' + internal [CDNA] aRNA fragments"
},
{
"step": 5,
"title": "RT of aRNA with random-hexamer / 3'-adapter primer",
"summary": "A random hexamer bearing a 5'-tail Illumina 3' adapter reverse-transcribes the fragmented aRNA into cDNA, appending the 3' adapter (ligation-free).",
"note": "SuperScript II RT. This CEL-Seq2 change replaces the inefficient adapter ligation of the original CEL-Seq, improving read mapping (93.8% vs 60.9%).",
"product": "5'- ...GAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV[CDNA]TGGAATTCTCGGGTGCCAAGGC -3'\n (Illumina 3' adapter from randomhexRT)"
},
{
"step": 6,
"title": "Library PCR (RP1 + RPIX) \u2014 final indexed library",
"summary": "Phusion PCR with RP1 (adds P5 + full 5' adapter) and a uniquely indexed RPIX (adds i7 index + P7) produces the sequenceable Illumina small-RNA library.",
"note": "11-15 cycles. Double AMPure XP cleanup; expected 200-400 bp peak. Handled as an Illumina Small-RNA library on HiSeq 2500 rapid mode.",
"product": "5'- AATGATACGGCGACCACCGAGATCTACACGTTCAGAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV[CDNA]TGGAATTCTCGGGTGCCAAGGAACTCCAGTCAC[SAMPLE_INDEX:6]ATCTCGTATGCCGTCTTCTGCTTG -3'"
}
],
"library_sequencing": [
{
"read": "R1",
"primer": "Illumina small-RNA Read 1 sequencing primer (= RA5 sense; anneals to the bottom strand, extends 5'->3')",
"template": "bottom",
"cycles": 15,
"note": "Reads the 6-nt UMI then the 6-nt cell barcode (UMI is 5' of the barcode in the RT primer), followed by ~3 poly(dT) bases. The barcode demultiplexes wells; the UMI is carried onto the R2 molecule for counting.",
"diagram": "5'- GTTCAGAGTTCTACAGTCCGACGATC-----------------> Read 1 primer (= RA5)\n5'- GTTCAGAGTTCTACAGTCCGACGATCNNNNNNNNNNNNTTTTTTTTTT...V[cDNA] -3'\n3'- CAAGTCTCAAGATGTCAGGCTGCTAGNNNNNNNNNNNNAAAAAAAAAA...B[cDNA] -5'\n ^^^^^^^^^^^^\n UMI(6) + cell barcode(6) (read 5'->3')"
},
{
"read": "I1",
"primer": "Illumina i7 index sequencing primer (anneals over the P7 region of the top strand, extends 5'->3' toward the insert)",
"template": "top",
"cycles": 7,
"note": "Reads the 6-nt i7 sample index (7 cycles run). The index is stored as its reverse complement on the top strand, so the read reproduces the designed index sequence.",
"diagram": " <------ i7 index read (7 cycles)\n 3'-TAGAGCATACGGCAGAAGACGAAC-5' i7 index primer\n5'- ...GGAACTCCAGTCACNNNNNNATCTCGTATGCCGTCTTCTGCTTG -3'\n3'- ...CCTTGAGGTCAGTGNNNNNNTAGAGCATACGGCAGAAGACGAAC -5'\n ^^^^^^ i7 index (6 nt; stored as rev-comp on top strand)"
},
{
"read": "R2",
"primer": "Illumina small-RNA Read 2 sequencing primer (= RA3 antisense; anneals to the RA3 region of the top strand, extends 5'->3' into the cDNA)",
"template": "top",
"cycles": 36,
"note": "Reads 36 bases of the 3'-biased cDNA insert (used for genome mapping); paired with the R1 UMI/barcode.",
"diagram": " <--------------------------- read (36 nt, into cDNA)\n 3'-ACCTTAAGAGCCCACGGTTCCTTGAGGTCAGTG-5' Read 2 primer\n5'- ...[cDNA]TGGAATTCTCGGGTGCCAAGGAACTCCAGTCAC[i7]ATCTCGTATGCCGTCTTCTGCTTG -3'\n3'- ...[cDNA]ACCTTAAGAGCCCACGGTTCCTTGAGGTCAGTG[i7]TAGAGCATACGGCAGAAGACGAAC -5'"
}
],
"whitelists": {},
"build": {
"builder_version": "llm-generic-1.0",
"deterministic": false,
"source_html_sha256": null,
"extraction_method": "claude_llm_generic",
"model": "claude-opus-4-8"
},
"title": "CEL-Seq2",
"description": "CEL-Seq2 is a 3'-end-tag, early-barcoding single-cell RNA-seq method that linearly amplifies transcripts by in vitro transcription (IVT). A barcoded RT primer carrying a T7 promoter, a shortened Illumina 5' (small-RNA) adapter, a 6-nt UMI and a 6-nt cell barcode captures poly(A) mRNA at the reverse-transcription step. After second-strand synthesis and T7 IVT (linear amplification), the amplified antisense RNA (aRNA) is fragmented, reverse-transcribed with a random hexamer bearing the Illumina 3' adapter, and PCR-amplified with the Illumina small-RNA RP1/RPI primers into a sequenceable library. Read 1 reads the UMI + cell barcode, Read 2 reads the 3'-biased cDNA, enabling accurate molecule counting across many multiplexed cells.",
"reference": {
"kind": "paper",
"label": "CEL-Seq2 detailed protocol (Additional file 4)",
"path": null,
"url": "https://doi.org/10.1186/s13059-016-0938-8",
"doi": "10.1186/s13059-016-0938-8"
},
"publication": {
"year": 2016,
"original_publication": {
"title": "CEL-Seq2: sensitive highly-multiplexed single-cell RNA-Seq",
"journal": "Genome Biology",
"doi": "10.1186/s13059-016-0938-8",
"url": "https://doi.org/10.1186/s13059-016-0938-8"
},
"authors": [
{
"name": "Tamar Hashimshony",
"affiliation": "Department of Biology, Technion - Israel Institute of Technology, Haifa, Israel"
},
{
"name": "Naftalie Senderovich",
"affiliation": "Department of Biology, Technion - Israel Institute of Technology, Haifa, Israel"
},
{
"name": "Gal Avital",
"affiliation": "Department of Biology, Technion - Israel Institute of Technology, Haifa, Israel"
},
{
"name": "Agnes Klochendler",
"affiliation": "Department of Developmental Biology and Cancer Research, The Hebrew University-Hadassah Medical School, Jerusalem, Israel"
},
{
"name": "Yaron de Leeuw",
"affiliation": "Department of Biology, Technion - Israel Institute of Technology, Haifa, Israel"
},
{
"name": "Leon Anavy",
"affiliation": "Department of Biology, Technion - Israel Institute of Technology, Haifa, Israel"
},
{
"name": "Dave Gennert",
"affiliation": "Klarman Cell Observatory, Broad Institute of Harvard and MIT; Department of Biology, MIT; HHMI, Massachusetts Institute of Technology, Cambridge, MA, USA"
},
{
"name": "Shuqiang Li",
"affiliation": "Fluidigm Corporation, South San Francisco, CA, USA"
},
{
"name": "Kenneth J. Livak",
"affiliation": "Fluidigm Corporation, South San Francisco, CA, USA"
},
{
"name": "Orit Rozenblatt-Rosen",
"affiliation": "Klarman Cell Observatory, Broad Institute of Harvard and MIT; Department of Biology, MIT; HHMI, Massachusetts Institute of Technology, Cambridge, MA, USA"
},
{
"name": "Yuval Dor",
"affiliation": "Department of Developmental Biology and Cancer Research, The Hebrew University-Hadassah Medical School, Jerusalem, Israel"
},
{
"name": "Aviv Regev",
"affiliation": "Klarman Cell Observatory, Broad Institute of Harvard and MIT; Department of Biology, MIT; HHMI, Massachusetts Institute of Technology, Cambridge, MA, USA"
},
{
"name": "Itai Yanai",
"corresponding": true,
"email": "yanai@technion.ac.il",
"affiliation": "Department of Biology, Technion - Israel Institute of Technology, Haifa, Israel"
}
],
"throughput": {
"summary": "Highly-multiplexed single cells: 24 fibroblasts (CEL-Seq), 20 (CEL-Seq2), 72 captured on the Fluidigm C1, and dendritic cells in a 384-well plate; 96 barcoded primers per run (168 6-nt barcodes designed).",
"cells": "Up to 96 cells per plate / C1 run (168 unique barcodes available)"
},
"statistical_model": "Binomial statistics to convert UMI counts into transcript counts (UMI-based molecule counting; efficiency estimated by linear fit on ERCC spike-in log-log plots)",
"other": [
{
"label": "RT/detection efficiency",
"value": "19.7% manual, 22% on Fluidigm C1, vs 5.8% for original CEL-Seq (ERCC spike-in estimate)"
},
{
"label": "Primer length",
"value": "Shortened from 92 nt (CEL-Seq) to 82 nt while adding a 6-nt UMI"
},
{
"label": "Ligation-free prep",
"value": "Illumina adapter inserted at RT via random hexamer; raised barcoded-read mapping from 60.9% to 93.8%"
},
{
"label": "GEO accession",
"value": "GSE78779"
},
{
"label": "Pipeline",
"value": "https://github.com/yanailab/CEL-Seq-pipeline (GPLv3); demultiplex R1 barcode/UMI, Bowtie2 map, UMI-aware htseq-count"
}
]
},
"modality": "RNA",
"method_type": "plate-based",
"data_processing": {
"summary": "Read 1 is used to demultiplex the per-well cell barcode and extract the UMI; Read 2 (3'-biased cDNA) is mapped to the genome and reads are assigned to genes, then collapsed by UMI to give a per-cell molecule (UMI) count matrix.",
"stages": [
{
"id": "demux",
"label": "Barcode & UMI processing"
},
{
"id": "align",
"label": "Alignment"
},
{
"id": "quant",
"label": "Quantification"
}
],
"nodes": [
{
"id": "demux_bc",
"label": "Demultiplex by cell barcode",
"tool": "CEL-Seq-pipeline",
"stage": "demux",
"scope": "per_cell",
"terminal": false,
"viz_only": false
},
{
"id": "extract_umi",
"label": "Extract UMI",
"tool": "CEL-Seq-pipeline",
"stage": "demux",
"scope": "per_cell",
"terminal": false,
"viz_only": false
},
{
"id": "map_cdna",
"label": "Map cDNA to genome",
"tool": "Bowtie2",
"stage": "align",
"scope": "per_cell",
"terminal": false,
"viz_only": false
},
{
"id": "assign_genes",
"label": "Assign reads to genes",
"tool": "htseq-count",
"stage": "quant",
"scope": "per_cell",
"terminal": false,
"viz_only": false
},
{
"id": "collapse_umi",
"label": "Collapse duplicate UMIs",
"tool": "CEL-Seq-pipeline",
"stage": "quant",
"scope": "per_cell",
"terminal": false,
"viz_only": false
},
{
"id": "count_matrix",
"label": "Build UMI count matrix",
"tool": "CEL-Seq-pipeline",
"stage": "quant",
"scope": "bulk",
"terminal": true,
"viz_only": false
}
],
"edges": [
{
"from": "demux_bc",
"to": "map_cdna",
"kind": "sequential"
},
{
"from": "map_cdna",
"to": "assign_genes",
"kind": "sequential"
},
{
"from": "assign_genes",
"to": "collapse_umi",
"kind": "sequential"
},
{
"from": "extract_umi",
"to": "collapse_umi",
"kind": "sequential"
},
{
"from": "collapse_umi",
"to": "count_matrix",
"kind": "fan_in"
}
],
"statistical_model": "Binomial (UMI-based molecule counting; observed UMIs converted to transcript counts, efficiency estimated from ERCC spike-ins)"
}
}