Spaces:
Paused
Paused
| { | |
| "schema_version": "seqcolyte.spec.v1", | |
| "spec_id": "cel_seq2", | |
| "assay": "CEL-Seq2", | |
| "chemistry_version": "", | |
| "platform": "illumina", | |
| "platform_params": { | |
| "read_type": "short" | |
| }, | |
| "source_docs": [ | |
| { | |
| "doc_id": "CEL-seq2_detailed_protocol.docx", | |
| "title": "CEL-Seq2 detailed protocol (Additional file 4)", | |
| "url": "https://doi.org/10.1186/s13059-016-0938-8", | |
| "path": "/Users/seqmachines/playground/protocols-test/protocols/cel_seq2/CEL-seq2_detailed_protocol.docx", | |
| "retrieved_date": null | |
| }, | |
| { | |
| "doc_id": "CEL-seq2_supp.pdf", | |
| "title": "CEL-seq2_supp.pdf", | |
| "url": null, | |
| "path": "/Users/seqmachines/playground/protocols-test/protocols/cel_seq2/CEL-seq2_supp.pdf", | |
| "retrieved_date": null | |
| }, | |
| { | |
| "doc_id": "CEL-seq2.pdf", | |
| "title": "CEL-seq2.pdf", | |
| "url": null, | |
| "path": "/Users/seqmachines/playground/protocols-test/protocols/cel_seq2/CEL-seq2.pdf", | |
| "retrieved_date": null | |
| } | |
| ], | |
| "oligos": [ | |
| { | |
| "oligo_id": "oligo_cel_seq2_rt_primer", | |
| "name": "CEL-Seq2 primer (barcoded RT primer)", | |
| "aliases": [], | |
| "role": "Barcoded reverse-transcription / capture primer with T7 promoter, Illumina 5' adapter, UMI, cell barcode and anchored poly(dT)", | |
| "kind": "assembled", | |
| "sequence": "GCCGGTAATACGACTCACTATAGGGAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV", | |
| "direction": "5_to_3", | |
| "components": [ | |
| { | |
| "name": "T7 promoter", | |
| "sequence": "GCCGGTAATACGACTCACTATAGGG", | |
| "role": "T7 RNA polymerase promoter (transcription start at the terminal GGG); GCCGG 5' leader" | |
| }, | |
| { | |
| "name": "Partial Illumina 5' adapter (RA5)", | |
| "sequence": "AGTTCTACAGTCCGACGATC", | |
| "role": "Shortened Illumina small-RNA 5' adapter; becomes the Read 1 primer landing site" | |
| }, | |
| { | |
| "name": "UMI", | |
| "sequence": "[UMI:6]", | |
| "role": "6-nt unique molecular identifier (NNNNNN), upstream of the barcode" | |
| }, | |
| { | |
| "name": "Cell barcode", | |
| "sequence": "[CELL_BARCODE:6]", | |
| "role": "6-nt sample/cell barcode (Hamming distance >=2); one per well" | |
| }, | |
| { | |
| "name": "Poly(dT)", | |
| "sequence": "TTTTTTTTTTTTTTTTTTTTTTTT", | |
| "role": "24-nt poly(dT) to prime poly(A) mRNA" | |
| }, | |
| { | |
| "name": "Anchor (V)", | |
| "sequence": "V", | |
| "role": "3' anchor base (A/C/G) that seats the primer at the poly(A) junction" | |
| } | |
| ], | |
| "provenance": "document", | |
| "derivation": null, | |
| "sequence_source": "llm_extracted_from_docs", | |
| "evidence": [ | |
| { | |
| "source_doc": "protocol_docs", | |
| "locator": "oligo / final library", | |
| "method": "claude_llm_extraction" | |
| } | |
| ], | |
| "notes": "82-nt primer (shortened from CEL-Seq's 92 nt). 96 barcoded variants are listed in the detailed protocol; e.g. barcode 1 = AGACTC, 4 = AGCTTC, 5 = CATGAG, 46 = TGCAGA. Recommended 10-primer pool: barcodes 1,4,5,9,10,23,25,26,31,46. 168 unique 6-nt barcodes were designed (GC 33-67%, last base != T)." | |
| }, | |
| { | |
| "oligo_id": "oligo_cel_seq2_barcode_primer_1s", | |
| "name": "CEL-Seq2 primer 1s (barcode AGACTC)", | |
| "aliases": [], | |
| "role": "Representative fully-specified barcoded RT primer (well 1)", | |
| "kind": "single", | |
| "sequence": "GCCGGTAATACGACTCACTATAGGGAGTTCTACAGTCCGACGATCNNNNNNAGACTCTTTTTTTTTTTTTTTTTTTTTTTTV", | |
| "direction": "5_to_3", | |
| "components": [], | |
| "provenance": "document", | |
| "derivation": null, | |
| "sequence_source": "llm_extracted_from_docs", | |
| "evidence": [ | |
| { | |
| "source_doc": "protocol_docs", | |
| "locator": "oligo / final library", | |
| "method": "claude_llm_extraction" | |
| } | |
| ], | |
| "notes": "NNNNNN = 6-nt UMI; AGACTC = 6-nt cell barcode for well 1. Concrete example of the CEL-Seq2 primer series (1s-96s) transcribed verbatim from the detailed protocol table." | |
| }, | |
| { | |
| "oligo_id": "oligo_cel_seq_original_primer", | |
| "name": "CEL-Seq primer (original, 8-nt barcode)", | |
| "aliases": [], | |
| "role": "Original CEL-Seq barcoded RT primer (no UMI)", | |
| "kind": "assembled", | |
| "sequence": "CGATTGAGGCCGGTAATACGACTCACTATAGGGGTTCAGAGTTCTACAGTCCGACGATC[CELL_BARCODE:8]TTTTTTTTTTTTTTTTTTTTTTTTV", | |
| "direction": "5_to_3", | |
| "components": [ | |
| { | |
| "name": "5' leader + T7 promoter", | |
| "sequence": "CGATTGAGGCCGGTAATACGACTCACTATAGGG", | |
| "role": "T7 promoter with extended 5' leader" | |
| }, | |
| { | |
| "name": "Illumina 5' adapter (RA5)", | |
| "sequence": "GTTCAGAGTTCTACAGTCCGACGATC", | |
| "role": "Full-length Illumina small-RNA 5' adapter" | |
| }, | |
| { | |
| "name": "Cell barcode", | |
| "sequence": "[CELL_BARCODE:8]", | |
| "role": "8-nt cell barcode (as previously published)" | |
| }, | |
| { | |
| "name": "Poly(dT)", | |
| "sequence": "TTTTTTTTTTTTTTTTTTTTTTTT", | |
| "role": "24-nt poly(dT)" | |
| }, | |
| { | |
| "name": "Anchor (V)", | |
| "sequence": "V", | |
| "role": "3' anchor base" | |
| } | |
| ], | |
| "provenance": "document", | |
| "derivation": null, | |
| "sequence_source": "llm_extracted_from_docs", | |
| "evidence": [ | |
| { | |
| "source_doc": "protocol_docs", | |
| "locator": "oligo / final library", | |
| "method": "claude_llm_extraction" | |
| } | |
| ], | |
| "notes": "Original CEL-Seq design (Table S2); 92 nt, longer T7 promoter and 5' adapter, no UMI." | |
| }, | |
| { | |
| "oligo_id": "oligo_cel_seq_umi_primer", | |
| "name": "CEL-Seq + UMI primer (5-nt UMI, 6-nt barcode)", | |
| "aliases": [], | |
| "role": "Intermediate CEL-Seq primer with 5-nt UMI", | |
| "kind": "assembled", | |
| "sequence": "CGATTGAGGCCGGTAATACGACTCACTATAGGGGTTCAGAGTTCTACAGTCCGACGATC[UMI:5][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV", | |
| "direction": "5_to_3", | |
| "components": [ | |
| { | |
| "name": "5' leader + T7 promoter", | |
| "sequence": "CGATTGAGGCCGGTAATACGACTCACTATAGGG", | |
| "role": "T7 promoter with extended 5' leader" | |
| }, | |
| { | |
| "name": "Illumina 5' adapter (RA5)", | |
| "sequence": "GTTCAGAGTTCTACAGTCCGACGATC", | |
| "role": "Full-length Illumina small-RNA 5' adapter" | |
| }, | |
| { | |
| "name": "UMI", | |
| "sequence": "[UMI:5]", | |
| "role": "5-nt UMI (NNNNN)" | |
| }, | |
| { | |
| "name": "Cell barcode", | |
| "sequence": "[CELL_BARCODE:6]", | |
| "role": "6-nt cell barcode" | |
| }, | |
| { | |
| "name": "Poly(dT)", | |
| "sequence": "TTTTTTTTTTTTTTTTTTTTTTTT", | |
| "role": "24-nt poly(dT)" | |
| }, | |
| { | |
| "name": "Anchor (V)", | |
| "sequence": "V", | |
| "role": "3' anchor base" | |
| } | |
| ], | |
| "provenance": "document", | |
| "derivation": null, | |
| "sequence_source": "llm_extracted_from_docs", | |
| "evidence": [ | |
| { | |
| "source_doc": "protocol_docs", | |
| "locator": "oligo / final library", | |
| "method": "claude_llm_extraction" | |
| } | |
| ], | |
| "notes": "Table S2; the 'CEL-Seq performed as previously described with a 5-base UMI and 6-base barcode' variant used for comparison." | |
| }, | |
| { | |
| "oligo_id": "oligo_library_rt_primer", | |
| "name": "Library RT primer (randomhexRT)", | |
| "aliases": [], | |
| "role": "Random-hexamer RT primer with 5'-tail Illumina 3' adapter; converts fragmented aRNA back to cDNA", | |
| "kind": "assembled", | |
| "sequence": "GCCTTGGCACCCGAGAATTCCANNNNNN", | |
| "direction": "5_to_3", | |
| "components": [ | |
| { | |
| "name": "Illumina 3' adapter (RA3, rev-comp orientation)", | |
| "sequence": "GCCTTGGCACCCGAGAATTCCA", | |
| "role": "5'-tail matching Illumina small-RNA 3' adapter (rev-comp of TGGAATTCTCGGGTGCCAAGGC)" | |
| }, | |
| { | |
| "name": "Random hexamer", | |
| "sequence": "NNNNNN", | |
| "role": "Random-priming 3' hexamer that anneals across fragmented aRNA" | |
| } | |
| ], | |
| "provenance": "document", | |
| "derivation": null, | |
| "sequence_source": "llm_extracted_from_docs", | |
| "evidence": [ | |
| { | |
| "source_doc": "protocol_docs", | |
| "locator": "oligo / final library", | |
| "method": "claude_llm_extraction" | |
| } | |
| ], | |
| "notes": "CEL-Seq2 change 5: inserts the Illumina 3' adapter at the RT step via a random hexamer, eliminating the ligation step of the original CEL-Seq. Transcribed verbatim from the protocol/Table S2." | |
| }, | |
| { | |
| "oligo_id": "oligo_rna_pcr_primer_rp1", | |
| "name": "RNA PCR Primer (RP1)", | |
| "aliases": [], | |
| "role": "Forward library PCR primer; adds P5 and completes the Illumina 5' adapter", | |
| "kind": "assembled", | |
| "sequence": "AATGATACGGCGACCACCGAGATCTACACGTTCAGAGTTCTACAGTCCGACGATC", | |
| "direction": "5_to_3", | |
| "components": [ | |
| { | |
| "name": "P5", | |
| "sequence": "AATGATACGGCGACCACCGAGATCTACAC", | |
| "role": "Illumina P5 flow-cell adapter" | |
| }, | |
| { | |
| "name": "Illumina 5' adapter (RA5)", | |
| "sequence": "GTTCAGAGTTCTACAGTCCGACGATC", | |
| "role": "Full 5' small-RNA adapter; 3' end anneals to the AGTTCTACAGTCCGACGATC tag in the fragment" | |
| } | |
| ], | |
| "provenance": "document", | |
| "derivation": null, | |
| "sequence_source": "llm_extracted_from_docs", | |
| "evidence": [ | |
| { | |
| "source_doc": "protocol_docs", | |
| "locator": "oligo / final library", | |
| "method": "claude_llm_extraction" | |
| } | |
| ], | |
| "notes": "Standard Illumina TruSeq Small-RNA RP1; the documents state 'sequences available from Illumina' and are not printed. Sequence assembled from the verified P5 constant plus the in-document 5' adapter tag." | |
| }, | |
| { | |
| "oligo_id": "oligo_rna_pcr_index_primer_rpi", | |
| "name": "RNA PCR Index Primer (RPIX)", | |
| "aliases": [], | |
| "role": "Reverse indexed library PCR primer; adds i7 sample index and P7 via the Illumina 3' adapter", | |
| "kind": "assembled", | |
| "sequence": "CAAGCAGAAGACGGCATACGAGAT[SAMPLE_INDEX:6]GTGACTGGAGTTCCTTGGCACCCGAGAATTCCA", | |
| "direction": "5_to_3", | |
| "components": [ | |
| { | |
| "name": "P7", | |
| "sequence": "CAAGCAGAAGACGGCATACGAGAT", | |
| "role": "Illumina P7 flow-cell adapter" | |
| }, | |
| { | |
| "name": "i7 sample index", | |
| "sequence": "[SAMPLE_INDEX:6]", | |
| "role": "6-nt sample index (read in the 7-cycle index read)" | |
| }, | |
| { | |
| "name": "Illumina 3' adapter (RA3, rev-comp)", | |
| "sequence": "GTGACTGGAGTTCCTTGGCACCCGAGAATTCCA", | |
| "role": "3' portion whose 3' end anneals to the RA3 tag added by the library RT primer" | |
| } | |
| ], | |
| "provenance": "document", | |
| "derivation": null, | |
| "sequence_source": "llm_extracted_from_docs", | |
| "evidence": [ | |
| { | |
| "source_doc": "protocol_docs", | |
| "locator": "oligo / final library", | |
| "method": "claude_llm_extraction" | |
| } | |
| ], | |
| "notes": "Standard Illumina TruSeq Small-RNA RPIX; not printed in the docs ('sequences from Illumina kit'). Assembled from the verified P7 constant, a 6-nt index, and the in-document 3' adapter. A uniquely indexed RPIX is added per pooled library." | |
| }, | |
| { | |
| "oligo_id": "oligo_illumina_ra5_adapter", | |
| "name": "Illumina 5' small-RNA adapter (RA5)", | |
| "aliases": [], | |
| "role": "5' adapter / Read 1 primer landing region of the final library", | |
| "kind": "single", | |
| "sequence": "GTTCAGAGTTCTACAGTCCGACGATC", | |
| "direction": "5_to_3", | |
| "components": [], | |
| "provenance": "document", | |
| "derivation": null, | |
| "sequence_source": "llm_extracted_from_docs", | |
| "evidence": [ | |
| { | |
| "source_doc": "protocol_docs", | |
| "locator": "oligo / final library", | |
| "method": "claude_llm_extraction" | |
| } | |
| ], | |
| "notes": "The shortened form AGTTCTACAGTCCGACGATC is carried on the CEL-Seq2 primer; full length restored by RP1 during PCR." | |
| }, | |
| { | |
| "oligo_id": "oligo_illumina_ra3_adapter", | |
| "name": "Illumina 3' small-RNA adapter (RA3)", | |
| "aliases": [], | |
| "role": "3' adapter / Read 2 primer landing region of the final library", | |
| "kind": "single", | |
| "sequence": "TGGAATTCTCGGGTGCCAAGGC", | |
| "direction": "5_to_3", | |
| "components": [], | |
| "provenance": "document", | |
| "derivation": null, | |
| "sequence_source": "llm_extracted_from_docs", | |
| "evidence": [ | |
| { | |
| "source_doc": "protocol_docs", | |
| "locator": "oligo / final library", | |
| "method": "claude_llm_extraction" | |
| } | |
| ], | |
| "notes": "Reverse complement of the library RT primer 5'-tail (GCCTTGGCACCCGAGAATTCCA)." | |
| }, | |
| { | |
| "oligo_id": "oligo_illumina_p5_adapter", | |
| "name": "Illumina P5 adapter", | |
| "aliases": [], | |
| "role": "P5 flow-cell binding sequence", | |
| "kind": "single", | |
| "sequence": "AATGATACGGCGACCACCGAGATCTACAC", | |
| "direction": "5_to_3", | |
| "components": [], | |
| "provenance": "document", | |
| "derivation": null, | |
| "sequence_source": "llm_extracted_from_docs", | |
| "evidence": [ | |
| { | |
| "source_doc": "protocol_docs", | |
| "locator": "oligo / final library", | |
| "method": "claude_llm_extraction" | |
| } | |
| ], | |
| "notes": "Standard Illumina constant; contributed by RP1." | |
| }, | |
| { | |
| "oligo_id": "oligo_illumina_p7_adapter", | |
| "name": "Illumina P7 adapter", | |
| "aliases": [], | |
| "role": "P7 flow-cell binding sequence", | |
| "kind": "single", | |
| "sequence": "CAAGCAGAAGACGGCATACGAGAT", | |
| "direction": "5_to_3", | |
| "components": [], | |
| "provenance": "document", | |
| "derivation": null, | |
| "sequence_source": "llm_extracted_from_docs", | |
| "evidence": [ | |
| { | |
| "source_doc": "protocol_docs", | |
| "locator": "oligo / final library", | |
| "method": "claude_llm_extraction" | |
| } | |
| ], | |
| "notes": "Standard Illumina constant; contributed by RPIX. Appears at the 3' end of the top strand as its reverse complement (ATCTCGTATGCCGTCTTCTGCTTG)." | |
| } | |
| ], | |
| "final_library": { | |
| "source_label": "CEL-Seq2 final Illumina small-RNA library (5'->3' top strand)", | |
| "annotated_library_sequence": "AATGATACGGCGACCACCGAGATCTACACGTTCAGAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV[CDNA]TGGAATTCTCGGGTGCCAAGGAACTCCAGTCAC[SAMPLE_INDEX:6]ATCTCGTATGCCGTCTTCTGCTTG", | |
| "library_sequence": "AATGATACGGCGACCACCGAGATCTACACGTTCAGAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV[CDNA]TGGAATTCTCGGGTGCCAAGGAACTCCAGTCAC[SAMPLE_INDEX:6]ATCTCGTATGCCGTCTTCTGCTTG", | |
| "strands": [ | |
| { | |
| "direction": "5_to_3", | |
| "source_html": "AATGATACGGCGACCACCGAGATCTACACGTTCAGAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV[CDNA]TGGAATTCTCGGGTGCCAAGGAACTCCAGTCAC[SAMPLE_INDEX:6]ATCTCGTATGCCGTCTTCTGCTTG", | |
| "source_sequence": "AATGATACGGCGACCACCGAGATCTACACGTTCAGAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV[CDNA]TGGAATTCTCGGGTGCCAAGGAACTCCAGTCAC[SAMPLE_INDEX:6]ATCTCGTATGCCGTCTTCTGCTTG" | |
| } | |
| ], | |
| "annotation_lines": [ | |
| "AATGATACGGCGACCACCGAGATCTACAC = P5", | |
| "GTTCAGAGTTCTACAGTCCGACGATC = Illumina 5' small-RNA adapter (RA5 / Read 1 primer site)", | |
| "[UMI:6] = UMI", | |
| "[CELL_BARCODE:6] = Cell barcode", | |
| "TTTTTTTTTTTTTTTTTTTTTTTT = poly(dT)", | |
| "V = anchor base", | |
| "[CDNA] = cDNA insert", | |
| "TGGAATTCTCGGGTGCCAAGGAACTCCAGTCAC = Illumina 3' small-RNA adapter (RA3 / Read 2 primer site)", | |
| "[SAMPLE_INDEX:6] = i7 sample index (reverse complement)", | |
| "ATCTCGTATGCCGTCTTCTGCTTG = reverse complement of P7" | |
| ], | |
| "evidence": [ | |
| { | |
| "source_doc": "protocol_docs", | |
| "locator": "CEL-Seq2 final Illumina small-RNA library (5'->3' top strand)", | |
| "method": "claude_llm_extraction" | |
| } | |
| ] | |
| }, | |
| "read_structure": { | |
| "reads": [ | |
| { | |
| "read": "R1", | |
| "primer": "Illumina small-RNA Read 1 sequencing primer", | |
| "template": "bottom", | |
| "cycles": 15, | |
| "segments": [ | |
| { | |
| "name": "UMI", | |
| "type": "umi", | |
| "order": 0, | |
| "scored": true, | |
| "provenance": null, | |
| "whitelist_ref": null, | |
| "constant_ref": null, | |
| "notes": null, | |
| "length": 6 | |
| }, | |
| { | |
| "name": "Cell barcode", | |
| "type": "barcode", | |
| "order": 1, | |
| "scored": true, | |
| "provenance": null, | |
| "whitelist_ref": null, | |
| "constant_ref": null, | |
| "notes": null, | |
| "length": 6 | |
| }, | |
| { | |
| "name": "poly(dT)", | |
| "type": "constant", | |
| "order": 2, | |
| "scored": false, | |
| "provenance": null, | |
| "whitelist_ref": null, | |
| "constant_ref": null, | |
| "notes": null, | |
| "length": 3 | |
| } | |
| ] | |
| }, | |
| { | |
| "read": "I1", | |
| "primer": "Illumina index sequencing primer", | |
| "template": "top", | |
| "cycles": 7, | |
| "segments": [ | |
| { | |
| "name": "i7 sample index", | |
| "type": "index", | |
| "order": 0, | |
| "scored": false, | |
| "provenance": null, | |
| "whitelist_ref": null, | |
| "constant_ref": null, | |
| "notes": null, | |
| "length": 6 | |
| } | |
| ] | |
| }, | |
| { | |
| "read": "R2", | |
| "primer": "Illumina small-RNA Read 2 sequencing primer", | |
| "template": "top", | |
| "cycles": 36, | |
| "segments": [ | |
| { | |
| "name": "cDNA insert", | |
| "type": "insert", | |
| "order": 0, | |
| "scored": true, | |
| "provenance": null, | |
| "whitelist_ref": null, | |
| "constant_ref": null, | |
| "notes": null | |
| } | |
| ] | |
| } | |
| ] | |
| }, | |
| "library_generation": [ | |
| { | |
| "step": 1, | |
| "title": "Anneal barcoded CEL-Seq2 primer & reverse transcription", | |
| "summary": "The barcoded poly(dT) CEL-Seq2 primer captures a single cell's poly(A) mRNA and SuperScript II reverse-transcribes first-strand cDNA.", | |
| "note": "Each cell/well gets a unique 6-nt barcode; the primer also carries a T7 promoter, the Illumina 5' adapter and a 6-nt UMI. SuperScript II extends from the anchored poly(dT) into the transcript body.", | |
| "product": "5'- GCCGGTAATACGACTCACTATAGGGAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV[CDNA]------->\n 3'- AAAAAAAAAAAAAAAAAAAAAAAA...(mRNA)...-5'" | |
| }, | |
| { | |
| "step": 2, | |
| "title": "Second-strand synthesis (dsDNA with T7 promoter)", | |
| "summary": "RNase H, E. coli DNA Pol I and DNA ligase convert the RNA:cDNA hybrid into double-stranded DNA carrying an intact T7 promoter.", | |
| "note": "SuperScript II Double-Stranded cDNA Synthesis Kit reagents. Barcoded samples are then pooled and bead-purified (AMPure XP) before amplification.", | |
| "product": "5'- GCCGGTAATACGACTCACTATAGGGAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV[CDNA] -3'\n3'- CGGCCATTATGCTGAGTGATATCCCTCAAGATGTCAGGCTGCTAG[UMI:6][CELL_BARCODE:6]AAAAAAAAAAAAAAAAAAAAAAAAB[CDNA] -5'" | |
| }, | |
| { | |
| "step": 3, | |
| "title": "In vitro transcription (T7 linear amplification)", | |
| "summary": "T7 RNA polymerase transcribes from the promoter, linearly amplifying each molecule into many antisense aRNA copies.", | |
| "note": "IVT (MEGAscript/SuperScript II kit) avoids exponential PCR bias; transcription starts at the terminal GGG so the T7 promoter itself is not copied into the aRNA.", | |
| "product": "5'- GGGAGUUCUACAGUCCGACGAUC[UMI:6][CELL_BARCODE:6]UUUUUUUUUUUUUUUUUUUUUUUUV[CDNA] -3' (aRNA, antisense to mRNA)" | |
| }, | |
| { | |
| "step": 4, | |
| "title": "aRNA fragmentation", | |
| "summary": "The amplified RNA is chemically fragmented (Mg2+, heat) to ~200-500 nt.", | |
| "note": "Fragmentation is stopped with EDTA and the aRNA is bead-purified (RNAClean XP). Only fragments retaining the 5' adapter + UMI + barcode end carry the demultiplexing information.", | |
| "product": "5'- GGGAGUUCUACAGUCCGACGAUC[UMI:6][CELL_BARCODE:6]UUUU...V[CDNA frag] -3' + internal [CDNA] aRNA fragments" | |
| }, | |
| { | |
| "step": 5, | |
| "title": "RT of aRNA with random-hexamer / 3'-adapter primer", | |
| "summary": "A random hexamer bearing a 5'-tail Illumina 3' adapter reverse-transcribes the fragmented aRNA into cDNA, appending the 3' adapter (ligation-free).", | |
| "note": "SuperScript II RT. This CEL-Seq2 change replaces the inefficient adapter ligation of the original CEL-Seq, improving read mapping (93.8% vs 60.9%).", | |
| "product": "5'- ...GAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV[CDNA]TGGAATTCTCGGGTGCCAAGGC -3'\n (Illumina 3' adapter from randomhexRT)" | |
| }, | |
| { | |
| "step": 6, | |
| "title": "Library PCR (RP1 + RPIX) \u2014 final indexed library", | |
| "summary": "Phusion PCR with RP1 (adds P5 + full 5' adapter) and a uniquely indexed RPIX (adds i7 index + P7) produces the sequenceable Illumina small-RNA library.", | |
| "note": "11-15 cycles. Double AMPure XP cleanup; expected 200-400 bp peak. Handled as an Illumina Small-RNA library on HiSeq 2500 rapid mode.", | |
| "product": "5'- AATGATACGGCGACCACCGAGATCTACACGTTCAGAGTTCTACAGTCCGACGATC[UMI:6][CELL_BARCODE:6]TTTTTTTTTTTTTTTTTTTTTTTTV[CDNA]TGGAATTCTCGGGTGCCAAGGAACTCCAGTCAC[SAMPLE_INDEX:6]ATCTCGTATGCCGTCTTCTGCTTG -3'" | |
| } | |
| ], | |
| "library_sequencing": [ | |
| { | |
| "read": "R1", | |
| "primer": "Illumina small-RNA Read 1 sequencing primer (= RA5 sense; anneals to the bottom strand, extends 5'->3')", | |
| "template": "bottom", | |
| "cycles": 15, | |
| "note": "Reads the 6-nt UMI then the 6-nt cell barcode (UMI is 5' of the barcode in the RT primer), followed by ~3 poly(dT) bases. The barcode demultiplexes wells; the UMI is carried onto the R2 molecule for counting.", | |
| "diagram": "5'- GTTCAGAGTTCTACAGTCCGACGATC-----------------> Read 1 primer (= RA5)\n5'- GTTCAGAGTTCTACAGTCCGACGATCNNNNNNNNNNNNTTTTTTTTTT...V[cDNA] -3'\n3'- CAAGTCTCAAGATGTCAGGCTGCTAGNNNNNNNNNNNNAAAAAAAAAA...B[cDNA] -5'\n ^^^^^^^^^^^^\n UMI(6) + cell barcode(6) (read 5'->3')" | |
| }, | |
| { | |
| "read": "I1", | |
| "primer": "Illumina i7 index sequencing primer (anneals over the P7 region of the top strand, extends 5'->3' toward the insert)", | |
| "template": "top", | |
| "cycles": 7, | |
| "note": "Reads the 6-nt i7 sample index (7 cycles run). The index is stored as its reverse complement on the top strand, so the read reproduces the designed index sequence.", | |
| "diagram": " <------ i7 index read (7 cycles)\n 3'-TAGAGCATACGGCAGAAGACGAAC-5' i7 index primer\n5'- ...GGAACTCCAGTCACNNNNNNATCTCGTATGCCGTCTTCTGCTTG -3'\n3'- ...CCTTGAGGTCAGTGNNNNNNTAGAGCATACGGCAGAAGACGAAC -5'\n ^^^^^^ i7 index (6 nt; stored as rev-comp on top strand)" | |
| }, | |
| { | |
| "read": "R2", | |
| "primer": "Illumina small-RNA Read 2 sequencing primer (= RA3 antisense; anneals to the RA3 region of the top strand, extends 5'->3' into the cDNA)", | |
| "template": "top", | |
| "cycles": 36, | |
| "note": "Reads 36 bases of the 3'-biased cDNA insert (used for genome mapping); paired with the R1 UMI/barcode.", | |
| "diagram": " <--------------------------- read (36 nt, into cDNA)\n 3'-ACCTTAAGAGCCCACGGTTCCTTGAGGTCAGTG-5' Read 2 primer\n5'- ...[cDNA]TGGAATTCTCGGGTGCCAAGGAACTCCAGTCAC[i7]ATCTCGTATGCCGTCTTCTGCTTG -3'\n3'- ...[cDNA]ACCTTAAGAGCCCACGGTTCCTTGAGGTCAGTG[i7]TAGAGCATACGGCAGAAGACGAAC -5'" | |
| } | |
| ], | |
| "whitelists": {}, | |
| "build": { | |
| "builder_version": "llm-generic-1.0", | |
| "deterministic": false, | |
| "source_html_sha256": null, | |
| "extraction_method": "claude_llm_generic", | |
| "model": "claude-opus-4-8" | |
| }, | |
| "title": "CEL-Seq2", | |
| "description": "CEL-Seq2 is a 3'-end-tag, early-barcoding single-cell RNA-seq method that linearly amplifies transcripts by in vitro transcription (IVT). A barcoded RT primer carrying a T7 promoter, a shortened Illumina 5' (small-RNA) adapter, a 6-nt UMI and a 6-nt cell barcode captures poly(A) mRNA at the reverse-transcription step. After second-strand synthesis and T7 IVT (linear amplification), the amplified antisense RNA (aRNA) is fragmented, reverse-transcribed with a random hexamer bearing the Illumina 3' adapter, and PCR-amplified with the Illumina small-RNA RP1/RPI primers into a sequenceable library. Read 1 reads the UMI + cell barcode, Read 2 reads the 3'-biased cDNA, enabling accurate molecule counting across many multiplexed cells.", | |
| "reference": { | |
| "kind": "paper", | |
| "label": "CEL-Seq2 detailed protocol (Additional file 4)", | |
| "path": null, | |
| "url": "https://doi.org/10.1186/s13059-016-0938-8", | |
| "doi": "10.1186/s13059-016-0938-8" | |
| }, | |
| "publication": { | |
| "year": 2016, | |
| "original_publication": { | |
| "title": "CEL-Seq2: sensitive highly-multiplexed single-cell RNA-Seq", | |
| "journal": "Genome Biology", | |
| "doi": "10.1186/s13059-016-0938-8", | |
| "url": "https://doi.org/10.1186/s13059-016-0938-8" | |
| }, | |
| "authors": [ | |
| { | |
| "name": "Tamar Hashimshony", | |
| "affiliation": "Department of Biology, Technion - Israel Institute of Technology, Haifa, Israel" | |
| }, | |
| { | |
| "name": "Naftalie Senderovich", | |
| "affiliation": "Department of Biology, Technion - Israel Institute of Technology, Haifa, Israel" | |
| }, | |
| { | |
| "name": "Gal Avital", | |
| "affiliation": "Department of Biology, Technion - Israel Institute of Technology, Haifa, Israel" | |
| }, | |
| { | |
| "name": "Agnes Klochendler", | |
| "affiliation": "Department of Developmental Biology and Cancer Research, The Hebrew University-Hadassah Medical School, Jerusalem, Israel" | |
| }, | |
| { | |
| "name": "Yaron de Leeuw", | |
| "affiliation": "Department of Biology, Technion - Israel Institute of Technology, Haifa, Israel" | |
| }, | |
| { | |
| "name": "Leon Anavy", | |
| "affiliation": "Department of Biology, Technion - Israel Institute of Technology, Haifa, Israel" | |
| }, | |
| { | |
| "name": "Dave Gennert", | |
| "affiliation": "Klarman Cell Observatory, Broad Institute of Harvard and MIT; Department of Biology, MIT; HHMI, Massachusetts Institute of Technology, Cambridge, MA, USA" | |
| }, | |
| { | |
| "name": "Shuqiang Li", | |
| "affiliation": "Fluidigm Corporation, South San Francisco, CA, USA" | |
| }, | |
| { | |
| "name": "Kenneth J. Livak", | |
| "affiliation": "Fluidigm Corporation, South San Francisco, CA, USA" | |
| }, | |
| { | |
| "name": "Orit Rozenblatt-Rosen", | |
| "affiliation": "Klarman Cell Observatory, Broad Institute of Harvard and MIT; Department of Biology, MIT; HHMI, Massachusetts Institute of Technology, Cambridge, MA, USA" | |
| }, | |
| { | |
| "name": "Yuval Dor", | |
| "affiliation": "Department of Developmental Biology and Cancer Research, The Hebrew University-Hadassah Medical School, Jerusalem, Israel" | |
| }, | |
| { | |
| "name": "Aviv Regev", | |
| "affiliation": "Klarman Cell Observatory, Broad Institute of Harvard and MIT; Department of Biology, MIT; HHMI, Massachusetts Institute of Technology, Cambridge, MA, USA" | |
| }, | |
| { | |
| "name": "Itai Yanai", | |
| "corresponding": true, | |
| "email": "yanai@technion.ac.il", | |
| "affiliation": "Department of Biology, Technion - Israel Institute of Technology, Haifa, Israel" | |
| } | |
| ], | |
| "throughput": { | |
| "summary": "Highly-multiplexed single cells: 24 fibroblasts (CEL-Seq), 20 (CEL-Seq2), 72 captured on the Fluidigm C1, and dendritic cells in a 384-well plate; 96 barcoded primers per run (168 6-nt barcodes designed).", | |
| "cells": "Up to 96 cells per plate / C1 run (168 unique barcodes available)" | |
| }, | |
| "statistical_model": "Binomial statistics to convert UMI counts into transcript counts (UMI-based molecule counting; efficiency estimated by linear fit on ERCC spike-in log-log plots)", | |
| "other": [ | |
| { | |
| "label": "RT/detection efficiency", | |
| "value": "19.7% manual, 22% on Fluidigm C1, vs 5.8% for original CEL-Seq (ERCC spike-in estimate)" | |
| }, | |
| { | |
| "label": "Primer length", | |
| "value": "Shortened from 92 nt (CEL-Seq) to 82 nt while adding a 6-nt UMI" | |
| }, | |
| { | |
| "label": "Ligation-free prep", | |
| "value": "Illumina adapter inserted at RT via random hexamer; raised barcoded-read mapping from 60.9% to 93.8%" | |
| }, | |
| { | |
| "label": "GEO accession", | |
| "value": "GSE78779" | |
| }, | |
| { | |
| "label": "Pipeline", | |
| "value": "https://github.com/yanailab/CEL-Seq-pipeline (GPLv3); demultiplex R1 barcode/UMI, Bowtie2 map, UMI-aware htseq-count" | |
| } | |
| ] | |
| }, | |
| "modality": "RNA", | |
| "method_type": "plate-based", | |
| "data_processing": { | |
| "summary": "Read 1 is used to demultiplex the per-well cell barcode and extract the UMI; Read 2 (3'-biased cDNA) is mapped to the genome and reads are assigned to genes, then collapsed by UMI to give a per-cell molecule (UMI) count matrix.", | |
| "stages": [ | |
| { | |
| "id": "demux", | |
| "label": "Barcode & UMI processing" | |
| }, | |
| { | |
| "id": "align", | |
| "label": "Alignment" | |
| }, | |
| { | |
| "id": "quant", | |
| "label": "Quantification" | |
| } | |
| ], | |
| "nodes": [ | |
| { | |
| "id": "demux_bc", | |
| "label": "Demultiplex by cell barcode", | |
| "tool": "CEL-Seq-pipeline", | |
| "stage": "demux", | |
| "scope": "per_cell", | |
| "terminal": false, | |
| "viz_only": false | |
| }, | |
| { | |
| "id": "extract_umi", | |
| "label": "Extract UMI", | |
| "tool": "CEL-Seq-pipeline", | |
| "stage": "demux", | |
| "scope": "per_cell", | |
| "terminal": false, | |
| "viz_only": false | |
| }, | |
| { | |
| "id": "map_cdna", | |
| "label": "Map cDNA to genome", | |
| "tool": "Bowtie2", | |
| "stage": "align", | |
| "scope": "per_cell", | |
| "terminal": false, | |
| "viz_only": false | |
| }, | |
| { | |
| "id": "assign_genes", | |
| "label": "Assign reads to genes", | |
| "tool": "htseq-count", | |
| "stage": "quant", | |
| "scope": "per_cell", | |
| "terminal": false, | |
| "viz_only": false | |
| }, | |
| { | |
| "id": "collapse_umi", | |
| "label": "Collapse duplicate UMIs", | |
| "tool": "CEL-Seq-pipeline", | |
| "stage": "quant", | |
| "scope": "per_cell", | |
| "terminal": false, | |
| "viz_only": false | |
| }, | |
| { | |
| "id": "count_matrix", | |
| "label": "Build UMI count matrix", | |
| "tool": "CEL-Seq-pipeline", | |
| "stage": "quant", | |
| "scope": "bulk", | |
| "terminal": true, | |
| "viz_only": false | |
| } | |
| ], | |
| "edges": [ | |
| { | |
| "from": "demux_bc", | |
| "to": "map_cdna", | |
| "kind": "sequential" | |
| }, | |
| { | |
| "from": "map_cdna", | |
| "to": "assign_genes", | |
| "kind": "sequential" | |
| }, | |
| { | |
| "from": "assign_genes", | |
| "to": "collapse_umi", | |
| "kind": "sequential" | |
| }, | |
| { | |
| "from": "extract_umi", | |
| "to": "collapse_umi", | |
| "kind": "sequential" | |
| }, | |
| { | |
| "from": "collapse_umi", | |
| "to": "count_matrix", | |
| "kind": "fan_in" | |
| } | |
| ], | |
| "statistical_model": "Binomial (UMI-based molecule counting; observed UMIs converted to transcript counts, efficiency estimated from ERCC spike-ins)" | |
| } | |
| } | |