seqcolyte / spec /technologies /dr_seq.json
seqmachines's picture
Deploy Seqcolyte Studio
8ffc9f7 verified
Raw
History Blame Contribute Delete
31.4 kB
{
"schema_version": "seqcolyte.spec.v1",
"spec_id": "dr_seq",
"assay": "DR-Seq",
"chemistry_version": "",
"platform": "illumina",
"platform_params": {
"read_type": "short"
},
"source_docs": [
{
"doc_id": "supplementary_NIHMS61543-supplement-1.pdf",
"title": "Single-cell analysis of genome and transcriptome diversity in humans",
"url": "https://doi.org/10.1038/nbt.3129",
"path": "/Users/seqmachines/playground/protocols-test/protocols/dr_seq/supplementary_NIHMS61543-supplement-1.pdf",
"retrieved_date": null
},
{
"doc_id": "DR-seq.pdf",
"title": "DR-seq.pdf",
"url": null,
"path": "/Users/seqmachines/playground/protocols-test/protocols/dr_seq/DR-seq.pdf",
"retrieved_date": null
}
],
"oligos": [
{
"oligo_id": "oligo_ad1x_rt_primer",
"name": "Adaptor-1x (Ad-1x) barcoded RT primer",
"aliases": [],
"role": "Reverse-transcription primer: barcoded poly-T primer with a 5' Illumina adaptor and a T7 promoter overhang; primes first-strand cDNA from mRNA poly(A) tails (CEL-Seq primer, ref 13)",
"kind": "assembled",
"sequence": "[ILLUMINA_ADAPTOR][T7_PROMOTER][CELL_BARCODE](T)n",
"direction": "5_to_3",
"components": [
{
"name": "5' Illumina adaptor",
"sequence": "[ILLUMINA_ADAPTOR]",
"role": "5' Illumina adaptor overhang (exact sequence not printed in document; per CEL-Seq ref 13)"
},
{
"name": "T7 promoter",
"sequence": "[T7_PROMOTER]",
"role": "T7 RNA polymerase promoter enabling in vitro transcription of cDNA into aRNA"
},
{
"name": "Cell barcode",
"sequence": "[CELL_BARCODE]",
"role": "Cell-specific barcode identifying the cell of origin (length not stated in document)"
},
{
"name": "Poly(dT)",
"sequence": "(T)n",
"role": "Poly-T that anneals to mRNA poly(A) tail"
}
],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "Exact nucleotide sequence not provided in the DR-Seq documents; described structurally per the referenced CEL-Seq primer (Hashimshony et al., Cell Reports 2012, ref 13). DR-Seq uses length-based identifiers instead of a random UMI."
},
{
"oligo_id": "oligo_ad2_quasilinear_primer",
"name": "Adaptor-2 (Ad-2) quasilinear amplification primer",
"aliases": [],
"role": "MALBAC-style random-priming amplification primer: 27-nt defined 5' sequence followed by 8 random nucleotides; randomly primes gDNA and single-stranded cDNA during quasilinear whole-genome amplification",
"kind": "assembled",
"sequence": "GTGAGTGATGGTTGAGGTAGTGTGGAGNNNNNNNN",
"direction": "5_to_3",
"components": [
{
"name": "Ad-2 common 27-nt sequence",
"sequence": "GTGAGTGATGGTTGAGGTAGTGTGGAG",
"role": "Defined 27-nt 5' common sequence (identical to primer P2; MALBAC common sequence, ref 7)"
},
{
"name": "Random priming octamer",
"sequence": "NNNNNNNN",
"role": "8 random nucleotides for random priming; 3' end preferentially ends in GGG or TTT"
}
],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "Paper states 'defined 27-nt sequence at the 5' end followed by eight random nucleotides (Ad-2)'; the 27-nt segment matches primer P2 (GTGAGTGATGGTTGAGGTAGTGTGGAG). Supplementary Note: 3' end has sequence GGG or TTT."
},
{
"oligo_id": "oligo_p1_second_strand_primer",
"name": "P1 second-strand synthesis primer",
"aliases": [],
"role": "Primes mRNA-specific second-strand synthesis of the quasilinear-amplified cDNA (single PCR cycle) to regenerate the T7 promoter for IVT",
"kind": "single",
"sequence": "CGATTGAGGCCGGTAATAC",
"direction": "5_to_3",
"components": [],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "5'-CGATTGAGGCCGGTAATAC-3' (Online Methods)."
},
{
"oligo_id": "oligo_p2_gdna_pcr_primer",
"name": "P2 gDNA amplification primer",
"aliases": [],
"role": "PCR primer that amplifies the gDNA-half quasilinear amplification product (21 cycles); anneals to the Ad-2 common sequence",
"kind": "single",
"sequence": "GTGAGTGATGGTTGAGGTAGTGTGGAG",
"direction": "5_to_3",
"components": [],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "5'-GTGAGTGATGGTTGAGGTAGTGTGGAG-3', 27 nt; identical to the Ad-2 common region."
},
{
"oligo_id": "oligo_p3_biotin_adaptor_removal_primer",
"name": "P3 biotinylated adaptor-removal primer",
"aliases": [],
"role": "5'-biotinylated PCR primer used to replace/remove adaptor Ad-2 from the gDNA PCR product before Illumina library prep; biotinylated ends are captured on streptavidin beads and discarded",
"kind": "single",
"sequence": "GTGAGCTGGAGTTGAGGTAGTGTGGAG",
"direction": "5_to_3",
"components": [
{
"name": "5' Biotin",
"sequence": "/5Biosg/",
"role": "5' biotin modification for streptavidin capture"
}
],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "5'-[biotin]-GTGAGCTGGAGTTGAGGTAGTGTGGAG-3' (Online Methods)."
},
{
"oligo_id": "oligo_3prime_illumina_adaptor",
"name": "3' Illumina adaptor (aRNA ligation)",
"aliases": [],
"role": "3' Illumina adaptor ligated to the IVT-generated aRNA in the mRNA branch prior to RT and PCR (CEL-Seq library prep, ref 13)",
"kind": "single",
"sequence": null,
"direction": "5_to_3",
"components": [],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "Sequence not printed in the DR-Seq documents; library preparation performed as in CEL-Seq (ref 13), which uses Illumina TruSeq small-RNA adaptors."
},
{
"oligo_id": "oligo_illumina_index_primers",
"name": "Illumina index / library primers",
"aliases": [],
"role": "Cell-specific index primers introducing sample indices and Illumina P5/P7 flanks during final library preparation (NEBNext Ultra DNA Library Prep Kit for the gDNA library; CEL-Seq index primers for the mRNA library)",
"kind": "single",
"sequence": null,
"direction": "5_to_3",
"components": [],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "Exact index/adaptor sequences not printed in the DR-Seq documents."
},
{
"oligo_id": "oligo_illumina_p5",
"name": "Illumina P5 adapter",
"aliases": [],
"role": "5' Illumina flow-cell adapter added during Illumina library preparation",
"kind": "single",
"sequence": "AATGATACGGCGACCACCGAGATCTACAC",
"direction": "5_to_3",
"components": [],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "Standard Illumina P5; introduced during Illumina library prep (not explicitly sequenced in the DR-Seq documents)."
},
{
"oligo_id": "oligo_illumina_p7",
"name": "Illumina P7 adapter",
"aliases": [],
"role": "3' Illumina flow-cell adapter added during Illumina library preparation",
"kind": "single",
"sequence": "CAAGCAGAAGACGGCATACGAGAT",
"direction": "5_to_3",
"components": [],
"provenance": "document",
"derivation": null,
"sequence_source": "llm_extracted_from_docs",
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "oligo / final library",
"method": "claude_llm_extraction"
}
],
"notes": "Standard Illumina P7 (final-library bottom-strand end is revcomp ATCTCGTATGCCGTCTTCTGCTTG); introduced during Illumina library prep (not explicitly sequenced in the DR-Seq documents)."
}
],
"final_library": {
"source_label": "DR-Seq mRNA (transcriptome) Illumina library \u2014 CEL-Seq-style; exact internal adaptor/barcode sequences are deferred to CEL-Seq (ref 13) and not printed in these documents",
"annotated_library_sequence": "AATGATACGGCGACCACCGAGATCTACAC + [Illumina Read 1 adaptor] + [CELL_BARCODE] + [CDNA] + [3' Illumina / Read 2 adaptor] + [SAMPLE_INDEX] + ATCTCGTATGCCGTCTTCTGCTTG",
"library_sequence": "AATGATACGGCGACCACCGAGATCTACAC[CELL_BARCODE][CDNA][SAMPLE_INDEX]ATCTCGTATGCCGTCTTCTGCTTG",
"strands": [
{
"direction": "5_to_3",
"source_html": "AATGATACGGCGACCACCGAGATCTACAC[CELL_BARCODE][CDNA][SAMPLE_INDEX]ATCTCGTATGCCGTCTTCTGCTTG",
"source_sequence": "AATGATACGGCGACCACCGAGATCTACAC[CELL_BARCODE][CDNA][SAMPLE_INDEX]ATCTCGTATGCCGTCTTCTGCTTG"
}
],
"annotation_lines": [
"AATGATACGGCGACCACCGAGATCTACAC = P5",
"[Illumina Read 1 adaptor] = TruSeq Read 1 primer region (sequence per CEL-Seq ref 13; not printed)",
"[CELL_BARCODE] = Cell barcode (from Ad-1x; length not stated in document)",
"[CDNA] = cDNA insert (3' end of transcript; length-based identifier derived from Ad-2 priming position, no random UMI)",
"[3' Illumina / Read 2 adaptor] = TruSeq Read 2 adaptor region (ligated 3' Illumina adaptor; sequence not printed)",
"[SAMPLE_INDEX] = i7 sample index (per-cell index primer)",
"ATCTCGTATGCCGTCTTCTGCTTG = reverse complement of P7"
],
"evidence": [
{
"source_doc": "protocol_docs",
"locator": "DR-Seq mRNA (transcriptome) Illumina library \u2014 CEL-Seq-style; exact internal adaptor/barcode sequences are deferred to CEL-Seq (ref 13) and not printed in these documents",
"method": "claude_llm_extraction"
}
]
},
"read_structure": {
"reads": [
{
"read": "R1",
"primer": "Illumina Read 1 sequencing primer",
"template": "top",
"cycles": 100,
"segments": [
{
"name": "Cell barcode",
"type": "barcode",
"order": 0,
"scored": true,
"provenance": null,
"whitelist_ref": null,
"constant_ref": null,
"notes": null
}
]
},
{
"read": "R2",
"primer": "Illumina Read 2 sequencing primer",
"template": "bottom",
"cycles": 100,
"segments": [
{
"name": "cDNA (transcript 3' end)",
"type": "insert",
"order": 0,
"scored": true,
"provenance": null,
"whitelist_ref": null,
"constant_ref": null,
"notes": null
}
]
}
]
},
"library_generation": [
{
"step": 1,
"title": "Single-cell lysis & reverse transcription with barcoded Ad-1x",
"summary": "A hand-picked single cell is lysed and its mRNA reverse-transcribed with the barcoded poly-T primer Ad-1x, leaving intact gDNA alongside single-stranded cDNA.",
"note": "Arrayscript reverse transcriptase extends from the Ad-1x poly(T) annealed to mRNA poly(A); the cDNA acquires a 5' cell barcode, a 5' Illumina adaptor and a T7 promoter overhang. ERCC spike-ins are added at lysis.",
"product": "gDNA (unmodified double-stranded genome):\n5'- ...genomic DNA... -3'\n3'- ...genomic DNA... -5'\n\nFirst-strand cDNA (from mRNA):\n5'- [ILLUMINA_ADAPTOR][T7_PROMOTER][CELL_BARCODE](T)n[CDNA] -3'\n 3'- (A)n...mRNA...cap -5'"
},
{
"step": 2,
"title": "Quasilinear whole-genome amplification with Ad-2 (7 rounds)",
"summary": "gDNA and single-stranded cDNA are co-amplified by 7 cycles of MALBAC-style quasilinear amplification using the random-priming adaptor Ad-2.",
"note": "Bst large fragment + Pyrophage 3173 exo- randomly prime both templates with Ad-2 (27-nt common + 8N). Most short (0.5-2.5 kb) amplicons carry Ad-2 at both ends; a minority of cDNA-derived amplicons carry Ad-2 at one end and Ad-1x at the other. The Ad-2 priming position gives each cDNA molecule a length-based identifier.",
"product": "Majority amplicon (gDNA- or cDNA-derived, Ad-2 both ends):\n5'- GTGAGTGATGGTTGAGGTAGTGTGGAGNNN[insert]NNNCTCCACACTACCTCAACCATCACTCAC -3'\n\nMinority cDNA-derived amplicon (Ad-2 one end, Ad-1x other):\n5'- GTGAGTGATGGTTGAGGTAGTGTGGAGNNN[CDNA](T)n[CELL_BARCODE][T7_PROMOTER][ILLUMINA_ADAPTOR] -3'"
},
{
"step": 3,
"title": "Split sample into gDNA and mRNA halves",
"summary": "After 7 rounds of quasilinear amplification the reaction is divided in two, processed separately for gDNA and mRNA sequencing.",
"note": "No physical separation of nucleic acids occurred before amplification; the split simply routes half toward genomic-DNA library prep and half toward transcriptome (IVT) library prep.",
"product": "Half A -> gDNA branch (steps 4-7)\nHalf B -> mRNA branch (steps 8-11)"
},
{
"step": 4,
"title": "gDNA branch: PCR amplification with P2 (21 cycles)",
"summary": "The gDNA half is PCR-amplified for 21 cycles with primer P2 against the Ad-2 common sequence.",
"note": "Deep VentR (exo-) exponentially amplifies the Ad-2-flanked amplicons.",
"product": "5'- GTGAGTGATGGTTGAGGTAGTGTGGAGNNN[genomic insert]NNNCTCCACACTACCTCAACCATCACTCAC -3'\n3'- CACTCACTACCAACTCCATCACACCTCNNN[genomic insert]NNNGAGGTGTGATGGAGTTGGTAGTGAGTG -5'"
},
{
"step": 5,
"title": "gDNA branch: Ad-2 removal with biotinylated P3 PCR",
"summary": "A short PCR with the 5'-biotinylated primer P3 replaces adaptor Ad-2 on the gDNA products.",
"note": "P3 (5' biotin) primes off the Ad-2 common region; the biotin tag marks the adaptor-bearing ends for later removal.",
"product": "5'- [biotin]GTGAGCTGGAGTTGAGGTAGTGTGGAGNNN[genomic insert]NNNCTCCACACTACCTCAACCTCCAGCTCAC[biotin] -3'"
},
{
"step": 6,
"title": "gDNA branch: sonication & streptavidin removal of adaptor ends",
"summary": "Products are sheared to ~300 bp and biotinylated adaptor fragments removed on streptavidin beads, keeping internal genomic fragments.",
"note": "Sonication (Biorupter) to ~300 bp; Dynabeads MyOne Streptavidin C1 capture biotin-tagged Ad-2/P3 ends, and the non-biotinylated supernatant (pure genomic inserts) is retained.",
"product": "5'- [genomic insert ~300 bp] -3'\n3'- [genomic insert ~300 bp] -5'"
},
{
"step": 7,
"title": "gDNA branch: Illumina library preparation (NEBNext Ultra)",
"summary": "Sheared genomic fragments are converted into a cell-indexed Illumina library with the NEBNext Ultra DNA Library Prep Kit.",
"note": "End-repair, A-tailing, adapter ligation and indexing add P5, a sample index and P7 (standard TruSeq/NEBNext chemistry; exact sequences not printed).",
"product": "5'- AATGATACGGCGACCACCGAGATCTACAC[Illumina R1 adaptor][CDNA/genomic insert][Illumina R2 adaptor][SAMPLE_INDEX]ATCTCGTATGCCGTCTTCTGCTTG -3'"
},
{
"step": 8,
"title": "mRNA branch: second-strand synthesis with P1",
"summary": "The cDNA half undergoes a single PCR cycle of mRNA-specific second-strand synthesis with primer P1.",
"note": "P1 (5'-CGATTGAGGCCGGTAATAC-3') generates double-stranded cDNA and restores the intact T7 promoter needed for IVT; only cDNA-derived molecules acquire this.",
"product": "5'- [ILLUMINA_ADAPTOR][T7_PROMOTER][CELL_BARCODE](T)n[CDNA] -3'\n3'- [ILLUMINA_ADAPTOR'][T7_PROMOTER'][CELL_BARCODE'](A)n[CDNA'] -5'"
},
{
"step": 9,
"title": "mRNA branch: in vitro transcription (IVT) to aRNA",
"summary": "T7 in vitro transcription linearly amplifies the double-stranded cDNA into antisense aRNA, produced only from cDNA (not gDNA).",
"note": "13 h T7 IVT (MessageAmp II) makes many antisense aRNA copies per cDNA; gDNA lacks a T7 promoter and is not transcribed, enriching the transcriptome fraction.",
"product": "aRNA (antisense, no poly-A):\n3'- [ILLUMINA_ADAPTOR][CELL_BARCODE](U)n[CDNA-antisense] -5'"
},
{
"step": 10,
"title": "mRNA branch: 3' Illumina adaptor ligation to aRNA",
"summary": "A 3' Illumina adaptor is ligated onto the aRNA to provide the second priming site.",
"note": "CEL-Seq-style ligation (ref 13) of the TruSeq small-RNA 3' adaptor to the aRNA 3' end.",
"product": "5'- [3' Illumina adaptor][CDNA-antisense](U)n[CELL_BARCODE][ILLUMINA_ADAPTOR] -3' (aRNA)"
},
{
"step": 11,
"title": "mRNA branch: RT, PCR & indexed Illumina library",
"summary": "The adaptor-ligated aRNA is reverse-transcribed and PCR-amplified into a cell-indexed Illumina mRNA library (CEL-Seq prep).",
"note": "RT primes off the ligated 3' adaptor; PCR with indexed Illumina primers adds P5/P7 and a sample index, yielding the sequenceable transcriptome library.",
"product": "5'- AATGATACGGCGACCACCGAGATCTACAC[Illumina R1 adaptor][CELL_BARCODE][CDNA][3' Illumina/R2 adaptor][SAMPLE_INDEX]ATCTCGTATGCCGTCTTCTGCTTG -3'"
}
],
"library_sequencing": [
{
"read": "Read 1 (cell barcode; left mate)",
"primer": "Illumina Read 1 sequencing primer (TruSeq Read 1 region of Ad-1x; sequence per CEL-Seq ref 13, not printed)",
"template": "top",
"cycles": 100,
"note": "The left mate reads the cell-specific barcode carried by Ad-1x, identifying the cell of origin. The Read 1 primer anneals in the TruSeq Read 1 region and extends rightward through the barcode. 100 bp paired-end on Illumina HiSeq 2500. Internal adaptor/barcode segments are placeholders because their sequences are deferred to CEL-Seq and not printed in the DR-Seq documents; only the P5/P7 flanks are known and shown base-paired.",
"diagram": " Illumina Read 1 sequencing primer\n 5'-[TruSeq Read 1]-------------------------------------------> (reads barcode)\n5'- AATGATACGGCGACCACCGAGATCTACAC[TruSeq Read 1 ][CELL_BARCODE ][cDNA insert ]...[i7] ATCTCGTATGCCGTCTTCTGCTTG -3'\n3'- TTACTATGCCGCTGGTGGCTCTAGATGTG[TruSeq Read 1'][CELL_BARCODE'][cDNA insert']...[i7']TAGAGCATACGGCAGAAGACGAAC -5'"
},
{
"read": "Read 2 (cDNA / transcript 3' end; right mate)",
"primer": "Illumina Read 2 sequencing primer (3' TruSeq adaptor ligated to aRNA; sequence per CEL-Seq ref 13, not printed)",
"template": "bottom",
"cycles": 100,
"note": "The right mate reads the 3' end of the transcript. The Read 2 primer anneals in the TruSeq Read 2 region and extends leftward across the cDNA insert; residual Ad-2 sequence is trimmed computationally and the first mapped coordinate becomes the length-based identifier used to collapse PCR duplicates. 100 bp paired-end.",
"diagram": "5'- AATGATACGGCGACCACCGAGATCTACAC[TruSeq Read 1 ][CELL_BARCODE ][cDNA insert ][TruSeq Read 2 ]...[i7] ATCTCGTATGCCGTCTTCTGCTTG -3'\n3'- TTACTATGCCGCTGGTGGCTCTAGATGTG[TruSeq Read 1'][CELL_BARCODE'][cDNA insert'][TruSeq Read 2']...[i7']TAGAGCATACGGCAGAAGACGAAC -5'\n(reads transcript) <-------------------------------------[TruSeq Read 2]-5'"
}
],
"whitelists": {},
"build": {
"builder_version": "llm-generic-1.0",
"deterministic": false,
"source_html_sha256": null,
"extraction_method": "claude_llm_generic",
"model": "claude-opus-4-8"
},
"title": "DR-Seq",
"description": "DR-Seq (gDNA-mRNA sequencing) simultaneously quantifies the genome and transcriptome of the same single cell without physically separating nucleic acids before amplification. A single hand-picked cell is lysed and its mRNA reverse-transcribed with a barcoded poly-T primer (Ad-1x) carrying a 5' Illumina adaptor and T7 promoter; gDNA and single-stranded cDNA are then co-amplified by MALBAC-style quasilinear whole-genome amplification with a random-priming adaptor (Ad-2). The sample is split: one half is PCR-amplified and made into a gDNA Illumina library for copy-number/SNV calling, the other is IVT-amplified (aRNA is produced only from cDNA) and made into a CEL-Seq-style mRNA library. Unique Ad-2 random-priming positions (\"length-based identifiers\") replace UMIs to remove PCR duplicates and count original cDNA molecules.",
"reference": {
"kind": "paper",
"label": "Single-cell analysis of genome and transcriptome diversity in humans",
"path": null,
"url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC4374170/",
"doi": "10.1038/nbt.3129"
},
"publication": {
"year": 2015,
"original_publication": {
"title": "Integrated genome and transcriptome sequencing of the same cell",
"journal": "Nature Biotechnology",
"doi": "10.1038/nbt.3129",
"url": "https://doi.org/10.1038/nbt.3129"
},
"authors": [
{
"name": "Siddharth S Dey",
"affiliation": "Hubrecht Institute-KNAW (Royal Netherlands Academy of Arts and Sciences), Utrecht, the Netherlands; University Medical Center Utrecht, Cancer Genomics Netherlands, Utrecht, the Netherlands"
},
{
"name": "Lennart Kester",
"affiliation": "Hubrecht Institute-KNAW, Utrecht, the Netherlands; University Medical Center Utrecht, Cancer Genomics Netherlands, Utrecht, the Netherlands"
},
{
"name": "Bastiaan Spanjaard",
"affiliation": "Hubrecht Institute-KNAW, Utrecht, the Netherlands; University Medical Center Utrecht, Utrecht, the Netherlands"
},
{
"name": "Magda Bienko",
"affiliation": "Hubrecht Institute-KNAW, Utrecht, the Netherlands; University Medical Center Utrecht, Utrecht, the Netherlands; present address: Science for Life Laboratory, Karolinska Institute, Stockholm, Sweden"
},
{
"name": "Alexander van Oudenaarden",
"corresponding": true,
"email": "a.vanoudenaarden@hubrecht.eu",
"affiliation": "Hubrecht Institute-KNAW, Utrecht, the Netherlands; University Medical Center Utrecht, Cancer Genomics Netherlands, Utrecht, the Netherlands"
}
],
"throughput": {
"summary": "Low-throughput, plate/tube-based hand-picked single cells; ~70% single-cell amplification success (21/30 SK-BR-3, 13/18 E14). E14: mRNA from 13 cells, gDNA from 3 of them; SK-BR-3: mRNA from 21 cells, gDNA from 7. ~10,674 (E14) and 12,205 (SK-BR-3) genes detected; single-cell gDNA sequenced at 0.6-2.5x depth.",
"cells": "13 E14 + 21 SK-BR-3 single cells (integrated gDNA+mRNA from the same cell)",
"rna": "3' transcript counting via length-based identifiers; 66/92 ERCC spike-in species detected",
"dna": "Whole-genome copy-number and SNV calling at 0.6-2.5x single-cell depth"
},
"other": [
{
"label": "GEO accession",
"value": "GSE62952"
},
{
"label": "Quasilinear amplification",
"value": "7 rounds of MALBAC-style quasilinear whole-genome amplification (Bst large fragment + Pyrophage 3173 exo-)"
},
{
"label": "Amplicon size",
"value": "Short 0.5-2.5 kb amplicons; final DNA library ~300 bp"
},
{
"label": "Sequencer",
"value": "Illumina HiSeq 2500; cDNA libraries 100 bp paired-end, gDNA/CEL-Seq libraries 50 or 100 bp paired-end"
},
{
"label": "Cell lines",
"value": "Mouse E14 embryonic stem cells (mm10) and human SK-BR-3 breast cancer cells (hg19)"
}
]
},
"modality": "DNA + RNA",
"method_type": "manual/tube",
"data_processing": {
"summary": "Two computational branches. mRNA: the left mate is demultiplexed by cell barcode, the right mate has the Ad-2 adaptor trimmed and is aligned to the transcriptome with BWA, and the first genomic coordinate of the right mate serves as a 'length-based identifier' to collapse PCR duplicates before RPM transcript counting. gDNA: reads are aligned to a coding-region-masked genome with BWA, PCR-duplicated, variable-binned, coverage-corrected and GC-corrected, segmented by circular binary segmentation (CBS), then calibrated against bulk to call integer copy numbers; SNVs are called with GATK.",
"stages": [
{
"id": "s_prep",
"label": "Read preprocessing"
},
{
"id": "s_align",
"label": "Alignment"
},
{
"id": "s_dedup",
"label": "Deduplication"
},
{
"id": "s_expr",
"label": "Expression quantification"
},
{
"id": "s_cnv",
"label": "Copy-number profiling"
},
{
"id": "s_snv",
"label": "Variant calling"
}
],
"nodes": [
{
"id": "reads",
"label": "Split gDNA and mRNA libraries",
"tool": "",
"stage": "s_prep",
"scope": "per_cell",
"terminal": false,
"viz_only": false
},
{
"id": "demux",
"label": "Demultiplex cell barcode",
"tool": "",
"stage": "s_prep",
"scope": "per_cell",
"terminal": false,
"viz_only": false
},
{
"id": "trim",
"label": "Trim Ad-2 adaptor",
"tool": "",
"stage": "s_prep",
"scope": "per_cell",
"terminal": false,
"viz_only": false
},
{
"id": "txome_align",
"label": "Align to transcriptome",
"tool": "BWA",
"stage": "s_align",
"scope": "per_cell",
"terminal": false,
"viz_only": false
},
{
"id": "collapse",
"label": "Collapse duplicate reads",
"tool": "",
"stage": "s_dedup",
"scope": "per_cell",
"terminal": false,
"viz_only": false
},
{
"id": "rpm",
"label": "Build RPM count matrix",
"tool": "",
"stage": "s_expr",
"scope": "per_cell",
"terminal": true,
"viz_only": false
},
{
"id": "gdna_align",
"label": "Align to masked genome",
"tool": "BWA",
"stage": "s_align",
"scope": "per_cell",
"terminal": false,
"viz_only": false
},
{
"id": "dedup_g",
"label": "Remove PCR duplicates",
"tool": "",
"stage": "s_dedup",
"scope": "per_cell",
"terminal": false,
"viz_only": false
},
{
"id": "bin",
"label": "Bin into variable-width bins",
"tool": "",
"stage": "s_cnv",
"scope": "per_cell",
"terminal": false,
"viz_only": false
},
{
"id": "count",
"label": "Count reads per bin",
"tool": "",
"stage": "s_cnv",
"scope": "per_cell",
"terminal": false,
"viz_only": false
},
{
"id": "gc",
"label": "Correct GC bias",
"tool": "",
"stage": "s_cnv",
"scope": "per_cell",
"terminal": false,
"viz_only": false
},
{
"id": "cbs",
"label": "Segment copy-number profile",
"tool": "circular binary segmentation (CBS)",
"stage": "s_cnv",
"scope": "per_cell",
"terminal": false,
"viz_only": false
},
{
"id": "cnv",
"label": "Call integer copy number",
"tool": "",
"stage": "s_cnv",
"scope": "per_cell",
"terminal": true,
"viz_only": false
},
{
"id": "hc",
"label": "Call variants",
"tool": "GATK HaplotypeCaller",
"stage": "s_snv",
"scope": "per_cell",
"terminal": false,
"viz_only": false
},
{
"id": "vf",
"label": "Filter variants",
"tool": "GATK VariantFiltration",
"stage": "s_snv",
"scope": "per_cell",
"terminal": true,
"viz_only": false
}
],
"edges": [
{
"from": "reads",
"to": "demux",
"kind": "branch"
},
{
"from": "reads",
"to": "trim",
"kind": "branch"
},
{
"from": "reads",
"to": "gdna_align",
"kind": "branch"
},
{
"from": "trim",
"to": "txome_align",
"kind": "sequential"
},
{
"from": "txome_align",
"to": "collapse",
"kind": "sequential"
},
{
"from": "demux",
"to": "collapse",
"kind": "sequential"
},
{
"from": "collapse",
"to": "rpm",
"kind": "sequential"
},
{
"from": "gdna_align",
"to": "dedup_g",
"kind": "sequential"
},
{
"from": "dedup_g",
"to": "bin",
"kind": "branch"
},
{
"from": "dedup_g",
"to": "hc",
"kind": "branch"
},
{
"from": "bin",
"to": "count",
"kind": "sequential"
},
{
"from": "count",
"to": "gc",
"kind": "sequential"
},
{
"from": "gc",
"to": "cbs",
"kind": "sequential"
},
{
"from": "cbs",
"to": "cnv",
"kind": "sequential"
},
{
"from": "hc",
"to": "vf",
"kind": "sequential"
}
],
"statistical_model": "Circular binary segmentation (CBS) for copy-number breakpoint detection; median segment counts calibrated against a bulk reference to call integer copy numbers."
}
}