1<!DOCTYPE html> 2<html class="writer-html5" lang="en" > 3<head> 4 <meta charset="utf-8" /><meta name="generator" content="Docutils 0.18.1: http://docutils.sourceforge.net/" /> 5 6 <meta name="viewport" content="width=device-width, initial-scale=1.0" /> 7 <title>Source Data — refine.bio documentation</title> 8 <link rel="stylesheet" href="_static/pygments.css" type="text/css" /> 9 <link rel="stylesheet" href="_static/css/theme.css" type="text/css" /> 10 <link rel="stylesheet" href="_static/styles.css" type="text/css" /> 11 <link rel="shortcut icon" href="_static/favicon.ico"/> 12 <!--[if lt IE 9]> 13
13<script src="_static/js/html5shiv.min.js"></script>
13 14 <![endif]--> 15 16
16<script src="_static/jquery.js"></script>
16 17
17<script src="_static/_sphinx_javascript_frameworks_compat.js"></script>
17 18
18<script data-url_root="./" id="documentation_options" src="_static/documentation_options.js"></script>
18 19
19<script src="_static/doctools.js"></script>
19 20
20<script src="_static/sphinx_highlight.js"></script>
20 21 22
22<script src="_static/js/theme.js"></script>
22 23 <link rel="index" title="Index" href="genindex.html" /> 24 <link rel="search" title="Search" href="search.html" /> 25 <link rel="next" title="Getting Started with a refine.bio dataset" href="getting_started.html" /> 26 <link rel="prev" title="refine.bio Documentation" href="index.html" /> 27<!-- Hotjar Tracking Code for refine.bio docs -->
28<script>
vendor: 125 bytes, lines 28-31
28 29 (function(h,o,t,j,a,r){ 30 h.hj=h.hj||function(){(h.hj.q=h.hj.q||[]).push(arguments)}; 31 h._hjSettings={hjid:
313564747
vendor: 258 bytes, lines 31-36
31,hjsv:6}; 32 a=o.getElementsByTagName('head')[0]; 33 r=o.createElement('script');r.async=1; 34 r.src=t+h._hjSettings.hjid+j+h._hjSettings.hjsv; 35 a.appendChild(r); 36 })(window,document,'https://static.hotjar.com/c/hotjar-','.js?sv=');
37</script>
37 38 39 40<!-- RTD Extra Head --> 41 42 43
44<script type="application/json" id="READTHEDOCS_DATA">{"ad_free": false, "api_host": "https://readthedocs.org", "builder": "sphinx", "canonical_url": null, "docroot": "/docs/", "features": {"docsearch_disabled": false}, "global_analytics_code": "UA-17997319-1", "language": "en", "page": "main_text", "programming_language": "words", "project": "refinebio-docs", "proxied_api_host": "/_", "source_suffix": ".md", "subprojects": {}, "theme": "sphinx_rtd_theme", "user_analytics_code": "", "version": "latest"}</script>
44 45 46<!-- 47Using this variable directly instead of using `JSON.parse` is deprecated. 48The READTHEDOCS_DATA global variable will be removed in the future. 49-->
50<script type="text/javascript"> 51READTHEDOCS_DATA = JSON.parse(document.getElementById('READTHEDOCS_DATA').innerHTML); 52</script>
52 53 54 55 56<!-- end RTD <extrahead> -->
57<script async type="text/javascript" src="/_/static/javascript/readthedocs-addons.js"></script>
57<meta name="readthedocs-project-slug" content="refinebio-docs" /><meta name="readthedocs-version-slug" content="latest" /><meta name="readthedocs-resolver-filename" content="/main_text.html" /><meta name="readthedocs-http-status" content="200" /></head> 58 59<body class="wy-body-for-nav"> 60 <div class="wy-grid-for-nav"> 61 <nav data-toggle="wy-nav-shift" class="wy-nav-side"> 62 <div class="wy-side-scroll"> 63 <div class="wy-side-nav-search" style="background: #003595" > 64 <a href="https://www.refine.bio/" class="wy-side-nav-search__back">Back to refine.bio</a> 65 66 67 68 69 <a href="index.html" class="icon icon-home"> 70 refine.bio 71 </a> 72 <div class="version"> 73 latest 74 </div> 75<div role="search"> 76 <form id="rtd-search-form" class="wy-form" action="search.html" method="get"> 77 <input type="text" name="q" placeholder="Search docs" aria-label="Search docs" /> 78 <input type="hidden" name="check_keywords" value="yes" /> 79 <input type="hidden" name="area" value="default" /> 80 </form> 81</div> 82 83 </div><div class="wy-menu wy-menu-vertical" data-spy="affix" role="navigation" aria-label="Navigation menu"> 84 <ul class="current"> 85<li class="toctree-l1 current"><a class="current reference internal" href="#">Source Data</a><ul> 86<li class="toctree-l2"><a class="reference internal" href="#types-of-data">Types of Data</a></li> 87<li class="toctree-l2"><a class="reference internal" href="#sources">Sources</a></li> 88<li class="toctree-l2"><a class="reference internal" href="#metadata">Metadata</a><ul> 89<li class="toctree-l3"><a class="reference internal" href="#refine-bio-harmonized-metadata">refine.bio-harmonized Metadata</a></li> 90<li class="toctree-l3"><a class="reference internal" href="#submitter-supplied-metadata">Submitter Supplied Metadata</a></li> 91</ul> 92</li> 93</ul> 94</li> 95<li class="toctree-l1"><a class="reference internal" href="#processing-information">Processing Information</a><ul> 96<li class="toctree-l2"><a class="reference internal" href="#refine-bio-processed">refine.bio processed </a><ul> 97<li class="toctree-l3"><a class="reference internal" href="#microarray-pipelines">Microarray pipelines</a><ul> 98<li class="toctree-l4"><a class="reference internal" href="#affymetrix">Affymetrix</a></li> 99<li class="toctree-l4"><a class="reference internal" href="#illumina-beadarrays">Illumina BeadArrays</a></li> 100</ul> 101</li> 102<li class="toctree-l3"><a class="reference internal" href="#rna-seq-pipelines">RNA-seq pipelines</a><ul> 103<li class="toctree-l4"><a class="reference internal" href="#salmon">Salmon</a></li> 104<li class="toctree-l4"><a class="reference internal" href="#tximport">tximport</a></li> 105</ul> 106</li> 107</ul> 108</li> 109<li class="toctree-l2"><a class="reference internal" href="#submitter-processed">Submitter processed </a><ul> 110<li class="toctree-l3"><a class="reference internal" href="#affymetrix-identifier-conversion">Affymetrix identifier conversion</a></li> 111<li class="toctree-l3"><a class="reference internal" href="#illumina-identifier-conversion">Illumina identifier conversion</a></li> 112</ul> 113</li> 114<li class="toctree-l2"><a class="reference internal" href="#aggregations">Aggregations</a><ul> 115<li class="toctree-l3"><a class="reference internal" href="#limitations-of-gene-identifiers-when-combining-across-platforms">Limitations of gene identifiers when combining across platforms</a></li> 116</ul> 117</li> 118<li class="toctree-l2"><a class="reference internal" href="#transformations">Transformations</a><ul> 119<li class="toctree-l3"><a class="reference internal" href="#quantile-normalization">Quantile normalization</a><ul> 120<li class="toctree-l4"><a class="reference internal" href="#reference-distribution">Reference distribution</a></li> 121<li class="toctree-l4"><a class="reference internal" href="#quantile-normalizing-samples-for-delivery">Quantile normalizing samples for delivery</a></li> 122<li class="toctree-l4"><a class="reference internal" href="#limitations-of-quantile-normalization-across-platforms-with-many-zeroes">Limitations of quantile normalization across platforms with many zeroe
122s</a></li> 123<li class="toctree-l4"><a class="reference internal" href="#skipping-quantile-normalization-for-rna-seq-experiments">Skipping quantile normalization for RNA-seq experiments</a></li> 124</ul> 125</li> 126<li class="toctree-l3"><a class="reference internal" href="#gene-transformations">Gene transformations</a></li> 127</ul> 128</li> 129</ul> 130</li> 131<li class="toctree-l1"><a class="reference internal" href="#downloadable-files">Downloadable Files</a><ul> 132<li class="toctree-l2"><a class="reference internal" href="#the-download-folder-structure-for-data-aggregated-by-experiment">The download folder structure for data aggregated by experiment:</a></li> 133<li class="toctree-l2"><a class="reference internal" href="#the-download-folder-structure-for-data-aggregated-by-species">The download folder structure for data aggregated by species:</a></li> 134<li class="toctree-l2"><a class="reference internal" href="#gene-expression-matrix">Gene Expression Matrix</a></li> 135<li class="toctree-l2"><a class="reference internal" href="#sample-metadata">Sample Metadata</a><ul> 136<li class="toctree-l3"><a class="reference internal" href="#tsv-files">TSV files</a></li> 137<li class="toctree-l3"><a class="reference internal" href="#json-files">JSON files</a></li> 138</ul> 139</li> 140<li class="toctree-l2"><a class="reference internal" href="#experiment-metadata">Experiment Metadata</a></li> 141</ul> 142</li> 143<li class="toctree-l1"><a class="reference internal" href="#refine-bio-compendia">refine.bio Compendia</a><ul> 144<li class="toctree-l2"><a class="reference internal" href="#normalized-compendia">Normalized compendia</a><ul> 145<li class="toctree-l3"><a class="reference internal" href="#collapsing-by-genus">Collapsing by genus</a></li> 146<li class="toctree-l3"><a class="reference internal" href="#normalized-compendium-download-folder">Normalized Compendium Download Folder</a></li> 147</ul> 148</li> 149<li class="toctree-l2"><a class="reference internal" href="#rna-seq-sample-compendia">RNA-Seq Sample Compendia</a><ul> 150<li class="toctree-l3"><a class="reference internal" href="#rna-seq-sample-compendium-download-folder">RNA-Seq Sample Compendium Download Folder</a></li> 151</ul> 152</li> 153</ul> 154</li> 155<li class="toctree-l1"><a class="reference internal" href="#api">API</a></li> 156<li class="toctree-l1"><a class="reference internal" href="#downstream-analysis-with-refine-bio-examples">Downstream Analysis with refine.bio Examples</a></li> 157<li class="toctree-l1"><a class="reference internal" href="getting_started.html">Getting Started with a refine.bio dataset</a><ul> 158<li class="toctree-l2"><a class="reference internal" href="getting_started.html#structure">
158Structure</a></li> 159<li class="toctree-l2"><a class="reference internal" href="getting_started.html#contents">Contents</a></li> 160<li class="toctree-l2"><a class="reference internal" href="getting_started.html#usage">Usage</a><ul> 161<li class="toctree-l3"><a class="reference internal" href="getting_started.html#reading-tsv-files">Reading TSV Files</a></li> 162<li class="toctree-l3"><a class="reference internal" href="getting_started.html#reading-json-files">Reading JSON Files</a></li> 163</ul> 164</li> 165</ul> 166</li> 167<li class="toctree-l1"><a class="reference internal" href="getting_started.html#getting-started-with-normalized-compendia">Getting Started with Normalized Compendia</a><ul> 168<li class="toctree-l2"><a class="reference internal" href="getting_started.html#id1">Structure</a></li> 169<li class="toctree-l2"><a class="reference internal" href="getting_started.html#id2">Contents</a></li> 170<li class="toctree-l2"><a class="reference internal" href="getting_started.html#notes-and-observations">Notes and observations</a><ul> 171<li class="toctree-l3"><a class="reference internal" href="getting_started.html#methods-evaluation-and-exploratory-data-analysis">Methods evaluation and exploratory data analysis</a></li> 172</ul> 173</li> 174<li class="toctree-l2"><a class="reference internal" href="getting_started.html#id3">Usage</a><ul> 175<li class="toctree-l3"><a class="reference internal" href="getting_started.html#id4">Reading TSV Files</a></li> 176<li class="toctree-l3"><a class="reference internal" href="getting_started.html#id5">Reading JSON Files</a></li> 177</ul> 178</li> 179</ul> 180</li> 181<li class="toctree-l1"><a class="reference internal" href="faq.html">FAQ</a><ul> 182<li class="toctree-l2"><a class="reference internal" href="faq.html#what-is-the-difference-between-refine-bio-processed-and-submitter-processed-datasets">What is the difference between refine.bio-processed and submitter-processed datasets?</a></li> 183<li class="toctree-l2"><a class="reference internal" href="faq.html#how-do-you-process-the-data">How do you process the data?</a></li> 184<li class="toctree-l2"><a class="reference internal" href="faq.html#what-type-of-data-does-refine-bio-support">What type of data does refine.bio support?</a></li> 185<li class="toctree-l2"><a class="reference internal" href="faq.html#what-does-corrected-metadata-mean">What does âcorrectedâ metadata mean?</a></li> 186<li class="toctree-l2"><a class="reference internal" href="faq.html#why-do-the-values-differ-a-little-bit-if-i-download-different-datasets">Why do the values differ a little bit if I download different datasets?</a></li> 187<li class="toctree-l2"><a class="reference internal" href="faq.html#why-do-i-get-a-limited-number-of-genes-back-when-i-aggregate-samples-from-different-experiments">
187Why do I get a limited number of genes back when I aggregate samples from different experiments?</a></li> 188<li class="toctree-l2"><a class="reference internal" href="faq.html#why-can-t-i-add-certain-samples-to-my-dataset">Why canât I add certain samples to my dataset?</a></li> 189<li class="toctree-l2"><a class="reference internal" href="faq.html#why-do-the-genes-included-in-rna-seq-experiments-change-between-experiments-from-the-same-organism">Why do the genes included in RNA-seq experiments change between experiments from the same organism?</a></li> 190<li class="toctree-l2"><a class="reference internal" href="faq.html#how-can-i-find-out-what-genome-build-and-release-were-used-to-process-rna-seq-data">How can I find out what genome build and release were used to process RNA-seq data?</a></li> 191<li class="toctree-l2"><a class="reference internal" href="faq.html#how-can-i-find-out-what-versions-of-software-packages-were-used-to-process-the-data">How can I find out what versions of software/packages were used to process the data?</a></li> 192<li class="toctree-l2"><a class="reference internal" href="faq.html#are-refine-bio-datasets-i-download-batch-corrected">Are refine.bio datasets I download batch corrected?</a></li> 193<li class="toctree-l2"><a class="reference internal" href="faq.html#why-are-the-expression-values-different-if-i-regenerate-a-dataset">Why are the expression values different if I regenerate a dataset?</a></li> 194<li class="toctree-l2"><a class="reference internal" href="faq.html#what-does-it-mean-to-skip-quantile-normalization-for-rna-seq-samples">What does it mean to skip quantile normalization for RNA-seq samples?</a></li> 195<li class="toctree-l2"><a class="reference internal" href="faq.html#how-do-i-cite-refine-bio">How do I cite refine.bio?</a></li> 196</ul> 197</li> 198<li class="toctree-l1"><a class="reference internal" href="license.html">License</a></li> 199</ul> 200 201 </div> 202 </div> 203 </nav> 204 205 <section data-toggle="wy-nav-shift" class="wy-nav-content-wrap"><nav class="wy-nav-top" aria-label="Mobile navigation menu" style="background: #003595" > 206 <i data-toggle="wy-nav-top" class="fa fa-bars"></i> 207 <a href="index.html">refine.bio</a> 208 </nav> 209 210 <div class="wy-nav-content"> 211 <div class="rst-content"> 212 <div role="navigation" aria-label="Page navigation"> 213 <ul class="wy-breadcrumbs"> 214 <li><a href="index.html" class="icon icon-home" aria-label="Home"></a></li> 215 <li class="breadcrumb-item active">Source Data</li> 216 <li class="wy-breadcrumbs-aside"> 217 <a href="https://github.com/AlexsLemonade/refinebio-docs/blob/main/docs/main_text.md" class="fa fa-github"> Edit on GitHub</a> 218 </li> 219 </ul> 220 <hr/> 221</div> 222 <div role="main" class="document" itemscope="itemscope" itemtype="http://schema.org/Article"> 223 <div itemprop="articleBody"> 224 225 <section id="source-data"> 226<h1>Source Data<a class="headerlink" href="#source-data" title="Permalink to this heading">ï</a></h1> 227<section id="types-of-data"> 228<h2>Types of Data<a class="headerlink" href="#types-of-data" title="Permalink to this heading">ï</a></h2> 229<p>The current version of <a href ="https://www.refine.bio" target = "blank">refine.bio </a>is designed to process gene expression data. 230This includes both microarray data and RNA-seq data. 231We normalize data to Ensembl gene identifiers and provide abundance estimates.</p> 232<p>More precisely, we support microarray platforms based on their GEO or ArrayExpress accessions. 233We currently support Affymetrix microarrays and Illumina BeadArrays, and we are continuing to evaluate and add support for more platforms. 234This <a href ="https://github.com/AlexsLemonade/refinebio/blob/dev/config/supported_microarray_platforms.csv" target = "blank">table </a> contains the microarray platforms that we support. 235We process a subset of Affymetrix platforms using the <a href = "http://brainarray.mbni.med.umich.edu/Brainarray/Database/CustomCDF/CDF_download.asp" target = "blank">BrainArray Custom CDFs</a>, which are denoted by a <code class="docutils literal notranslate"><span class="pre">y</span></code> in the <code class="docutils literal notranslate"><span class="pre">is_brainarray</span></code> column. 236We also support RNA-seq experiments performed on <a href ="https://github.com/AlexsLemonade/refinebio/blob/dev/config/supported_rnaseq_platforms.txt" target = "blank">these</a> short-read platforms. 237For more information on how data are processed, see the <a class="reference internal" href="#processing-information"><span class="xref myst">Processing Information</span></a> section of this document. 238If there is a platform that you would like to see processed, please <a href ="https://github.com/AlexsLemonade/refinebio/issues" target = "blank">file an issue on GitHub</a>. 239If you would prefer to report issues via e-mail, you can also email <a class="reference external" href="mailto:requests%40ccdatalab.org">requests<span>@</span>
239ccdatalab<span>.</span>org</a>.</p> 240</section> 241<section id="sources"> 242<h2>Sources<a class="headerlink" href="#sources" title="Permalink to this heading">ï</a></h2> 243<p>We download gene expression information and metadata from <a href = "https://www.ebi.ac.uk/arrayexpress/" target = "blank">EBIâs ArrayExpress</a>, <a href = "https://www.ncbi.nlm.nih.gov/geo/" target = "blank">NCBIâs Gene Expression Omnibus (GEO)</a>, and <a href = "https://www.ncbi.nlm.nih.gov/sra" target = "blank">NCBIâs Sequence Read Archive (SRA)</a>. 244NCBIâs SRA also contains experiments from EBIâs ENA (<a href = "https://www.ncbi.nlm.nih.gov/sra/?term=ERP000447" target = "blank">example</a>) and the DDBJ (<a href = "https://www.ncbi.nlm.nih.gov/sra/?term=DRP000017" target = "blank">example</a>).</p> 245<p><img alt="sources" src="https://user-images.githubusercontent.com/15315514/44533218-08b95f80-a6c3-11e8-9eb8-827086fd2d99.png" /></p> 246</section> 247<section id="metadata"> 248<h2>Metadata<a class="headerlink" href="#metadata" title="Permalink to this heading">ï</a></h2> 249<p>We provide metadata that we obtain from the source repositories. 250We also attempt to, where possible, perform some harmonization of metadata to improve the discoverability of useful data. 251Note that we do not yet obtain sample metadata from the <a href = "https://www.ncbi.nlm.nih.gov/biosample/" target = "blank">BioSample</a> database, so the metadata available for RNA-seq samples is limited.</p> 252<section id="refine-bio-harmonized-metadata"> 253<h3>refine.bio-harmonized Metadata<a class="headerlink" href="#refine-bio-harmonized-metadata" title="Permalink to this heading">ï</a></h3> 254<p><em>The documentation in this section reflects data that has been processed via refine.bio as of version <code class="docutils literal notranslate"><span class="pre">v1.45.0</span></code>.</em> 255<em>See the documentation sidebar for the current version of refine.bio.</em> 256<em>The <code class="docutils literal notranslate"><span class="pre">refinebio_processor_version</span></code> field in the downloaded metadata file captures the refine.bio version when a sample was processed.</em></p> 257<p>Scientists who upload results donât always use the same names for related values. 258This makes it challenging to search across datasets. 259We have implemented some processes to smooth out some of these issues.</p> 260<p><img alt="harmonized-metadata" src="https://user-images.githubusercontent.com/15315514/44549202-5eefc800-a6ee-11e8-8a7b-57826f0153f2.png" /></p> 261<p>To aid in searches and for general convenience, we combine certain fields based on similar keys to produce lightly harmonized metadata. 262For example, <code class="docutils literal notranslate"><span class="pre">treatment</span></code>, <code class="docutils literal notranslate"><span class="pre">treatment</span> <span class="pre">group</span></code>, <code class="docutils literal notranslate"><span class="pre">treatment</span> <span class="pre">protocol</span></code>, <code class="docutils literal notranslate"><span class="pre">drug</span> <span class="pre">treatment</span></code>, and <code class="docutils literal notranslate"><span class="pre">clinical</span> <span class="pre">treatment</span></code> fields get collapsed down to <code class="docutils literal notranslate"><span class="pre">treatment</span></code>. 263The fields that we currently collapse to includes <code class="docutils literal notranslate"><span class="pre">specimen</span> <span class="pre">part</span></code>, <code class="docutils literal notranslate"><span class="pre">genetic</span> <span class="pre">information</span></code>, <code class="docutils literal notranslate"><span class="pre">disease</span></code>, <code class="docutils literal notranslate"><span class="pre">disease</span> <span class="pre">stage</span></code>, <code class="docutils literal notranslate"><span class="pre">treatment</span></code>, <code class="docutils literal notranslate"><span class="pre">race</span></code>, <code class="docutils literal notranslate"><span class="pre">subject</span></code>, <code class="docutils literal notranslate"><span class="pre">compound</span></code>, <code class="docutils literal notranslate"><span class="pre">cell_line</span></code>, and <code class="docutils literal notranslate"><span class="pre">time</span></code>.</p> 264<p>See the table below for the mappings between the keys from source data and the harmonized keys. 265In addition to the source data keys explicitly listed in the table, we check for variants in the metadata from the source repositories, e.g., the source keys <code class="docutils literal notranslate"><span class="pre">age</span></code>, <code class="docutils literal notranslate"><span class="pre">characteristic</span> <span class="pre">[age]</span></code>, and <code class="docutils literal notranslate"><span class="pre">characteristic_age</span></code> would all map to the harmonized key <code class="docutils literal notranslate"><span class="pre">age</span></code>.</p> 266<table class="docutils align-default"> 267<thead> 268<tr class="row-odd"><th class="head text-center"><p>Harmonized key</p></th> 269<th class="head"><p>Keys from data sources</p></th> 270</tr> 271</thead> 272<tbody> 273<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">specimen</span> <span class="pre">part</span></code></p></td> 274<td><p><code class="docutils literal notranslate"><span class="pre">organism</span> <span class="pre">part</span></code>, <code class="docutils literal notranslate"><span class="pre">cell</span> <span class="pre">type</span></code>, <code class="docutils literal notranslate"><span class="pre">tissue</span></code>, <code class="docutils literal notranslate"><span class="pre">tissue</span> <span class="pre">type</span></code>, <code class="docutils literal notranslate"><span class="pre">tissue</span> <span class="pre">source</span></code>, <code class="docutils literal notranslate"><span class="pre">tissue</span> <span class="pre">origin</span></code>, <code class="docutils literal notranslate"><span class="pre">source</span> <span class="pre">tissue</span></code>, <code class="docutils literal notranslate"><span class="pre">tissue</span> <span class="pre">subtype</span></code>, <code class="docutils literal notranslate"><span class="pre">tissue/cell</span> <span class="pre">type</span></code>, <code class="docutils literal notranslate"><span class="pre">tissue</span> <span class="pre">region</span></code>, <code class="docutils literal notranslate"><span class="pre">tissue</span> <span class="pre">compartment</span></code>, <code class="docutils literal notranslate"><span class="pre">tissues</span></code>, <code class="docutils literal notranslate"><span class="pre">tissue</span> <span class="pre">
274of</span> <span class="pre">origin</span></code>, <code class="docutils literal notranslate"><span class="pre">tissue-type</span></code>, <code class="docutils literal notranslate"><span class="pre">tissue</span> <span class="pre">harvested</span></code>, <code class="docutils literal notranslate"><span class="pre">cell/tissue</span> <span class="pre">type</span></code>, <code class="docutils literal notranslate"><span class="pre">tissue</span> <span class="pre">subregion</span></code>, <code class="docutils literal notranslate"><span class="pre">organ</span></code>, <code class="docutils literal notranslate"><span class="pre">cell_type</span></code>, <code class="docutils literal notranslate"><span class="pre">organismpart</span></code>, <code class="docutils literal notranslate"><span class="pre">isolation</span> <span class="pre">source</span></code>, <code class="docutils literal notranslate"><span class="pre">tissue</span> <span class="pre">sampled</span></code>, <code class="docutils literal notranslate"><span class="pre">cell</span> <span class="pre">description</span></code></p></td> 275</tr> 276<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">genetic</span> <span class="pre">information</span></code></p></td> 277<td><p><code class="docutils literal notranslate"><span class="pre">strain/background</span></code>, <code class="docutils literal notranslate"><span class="pre">strain</span></code>, <code class="docutils literal notranslate"><span class="pre">strain</span> <span class="pre">or</span> <span class="pre">line</span></code>, <code class="docutils literal notranslate"><span class="pre">background</span> <span class="pre">strain</span></code>, <code class="docutils literal notranslate"><span class="pre">genotype</span></code>, <code class="docutils literal notranslate"><span class="pre">genetic</span> <span class="pre">background</span></code>, <code class="docutils literal notranslate"><span class="pre">genotype/variation</span></code>, <code class="docutils literal notranslate"><span class="pre">ecotype</span></code>, <code class="docutils literal notranslate"><span class="pre">cultivar</span></code>, <code class="docutils literal notranslate"><span class="pre">strain/genotype</span></code>, <code class="docutils literal notranslate"><span class="pre">strain</span> <span class="pre">background</span></code></p></td> 278</tr> 279<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">disease</span></code></p></td> 280<td><p><code class="docutils literal notranslate"><span class="pre">disease</span> </code>, <code class="docutils literal notranslate"><span class="pre">disease</span> <span class="pre">state</span> </code>, <code class="docutils literal notranslate"><span class="pre">disease</span> <span class="pre">status</span> </code>, <code class="docutils literal notranslate"><span class="pre">diagnosis</span> </code>, <code class="docutils literal notranslate"><span class="pre">disease</span> </code>, <code class="docutils literal notranslate"><span class="pre">infection</span> <span class="pre">with</span> </code>, <code class="docutils literal notranslate"><span class="pre">sample</span> <span class="pre">type</span> </code></p></td> 281</tr> 282<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">disease</span> <span class="pre">stage</span></code></p></td> 283<td><p><code class="docutils literal notranslate"><span class="pre">disease</span> <span class="pre">state</span> </code>, <code class="docutils literal notranslate"><span class="pre">disease</span> <span class="pre">staging</span> </code>, <code class="docutils literal notranslate"><span class="pre">disease</span> <span class="pre">stage</span> </code>, <code class="docutils literal notranslate"><span class="pre">grade</span> </code>, <code class="docutils literal notranslate"><span class="pre">tumor</span> <span class="pre">grade</span> </code>, <code class="docutils literal notranslate"><span class="pre">who</span> <span class="pre">grade</span> </code>, <code class="docutils literal notranslate"><span class="pre">histological</span> <span class="pre">grade</span> </code>, <code class="docutils literal notranslate"><span class="pre">tumor</span> <span class="pre">grading</span> </code>, <code class="docutils literal notranslate"><span class="pre">disease</span> <span class="pre">outcome</span> </code>, <code class="docutils literal notranslate"><span class="pre">subject</span> <span class="pre">status</span> </code></p></td> 284</tr> 285<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">treatment</span></code></p></td> 286<td><p><code class="docutils literal notranslate"><span class="pre">treatment</span></code>, <code class="docutils literal notranslate"><span class="pre">treatment</span> <span class="pre">group</span></code>, <code class="docutils literal notranslate"><span class="pre">treatment</span> <span class="pre">protocol</span></code>, <code class="docutils literal notranslate"><span class="pre">drug</span> <span class="pre">treatment</span></code>, <code class="docutils literal notranslate"><span class="pre">clinical</span> <span class="pre">treatment</span></code></p></td> 287</tr> 288<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">race</span></code></p></td> 289<td><p><code class="docutils literal notranslate"><span class="pre">race</span></code>, <code class="docutils literal notranslate"><span class="pre">ethnicity</span></code>, <code class="docutils literal notranslate"><span class="pre">race/ethnicity</span></code></p></td> 290</tr> 291<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">subject</span></code></p></td> 292<td><p><code class="docutils literal notranslate"><span class="pre">subject</span> </code>, <code class="docutils literal notranslate"><span class="pre">subject</span> <span class="pre">id</span> </code>, <code class="docutils literal notranslate"><span class="pre">subject/sample</span> <span class="pre">source</span> <span class="pre">id</span> </code>, <code class="docutils literal notranslate"><span class="pre">subject</span> <span class="pre">identifier</span> </code>, <code class="docutils literal notranslate"><span class="pre">human</span> <span class="pre">subject</span> <span class="pre">anonymized</span> <span class="pre">id</span> </code>, <code class="docutils literal notranslate"><span class="pre">
292individual</span> </code>, <code class="docutils literal notranslate"><span class="pre">individual</span> <span class="pre">identifier</span> </code>, <code class="docutils literal notranslate"><span class="pre">individual</span> <span class="pre">id</span> </code>, <code class="docutils literal notranslate"><span class="pre">patient</span> </code>, <code class="docutils literal notranslate"><span class="pre">patient</span> <span class="pre">id</span> </code>, <code class="docutils literal notranslate"><span class="pre">patient</span> <span class="pre">identifier</span> </code>, <code class="docutils literal notranslate"><span class="pre">patient</span> <span class="pre">number</span> </code>, <code class="docutils literal notranslate"><span class="pre">patient</span> <span class="pre">no</span> </code>, <code class="docutils literal notranslate"><span class="pre">donor</span> <span class="pre">id</span> </code>, <code class="docutils literal notranslate"><span class="pre">donor</span> </code>, <code class="docutils literal notranslate"><span class="pre">sample_source_name</span> </code></p></td> 293</tr> 294<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">compound</span></code></p></td> 295<td><p><code class="docutils literal notranslate"><span class="pre">compound</span></code>, <code class="docutils literal notranslate"><span class="pre">compound1</span></code>, <code class="docutils literal notranslate"><span class="pre">compound2</span></code>, <code class="docutils literal notranslate"><span class="pre">compound</span> <span class="pre">name</span></code>, <code class="docutils literal notranslate"><span class="pre">drug</span></code>, <code class="docutils literal notranslate"><span class="pre">drugs</span></code>, <code class="docutils literal notranslate"><span class="pre">immunosuppressive</span> <span class="pre">drugs</span></code></p></td> 296</tr> 297<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">time</span></code></p></td> 298<td><p><code class="docutils literal notranslate"><span class="pre">time</span></code>, <code class="docutils literal notranslate"><span class="pre">initial</span> <span class="pre">time</span> <span class="pre">point</span></code>, <code class="docutils literal notranslate"><span class="pre">start</span> <span class="pre">time</span></code>, <code class="docutils literal notranslate"><span class="pre">stop</span> <span class="pre">time</span></code>, <code class="docutils literal notranslate"><span class="pre">time</span> <span class="pre">point</span></code>, <code class="docutils literal notranslate"><span class="pre">sampling</span> <span class="pre">time</span> <span class="pre">point</span></code>, <code class="docutils literal notranslate"><span class="pre">sampling</span> <span class="pre">time</span></code>, <code class="docutils literal notranslate"><span class="pre">time</span> <span class="pre">post</span> <span class="pre">infection</span></code></p></td> 299</tr> 300<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">age</span></code></p></td> 301<td><p><code class="docutils literal notranslate"><span class="pre">age</span></code>, <code class="docutils literal notranslate"><span class="pre">patient</span> <span class="pre">age</span></code>, <code class="docutils literal notranslate"><span class="pre">age</span> <span class="pre">of</span> <span class="pre">patient</span></code>, <code class="docutils literal notranslate"><span class="pre">age</span> <span class="pre">(years)</span></code>, <code class="docutils literal notranslate"><span class="pre">age</span> <span class="pre">at</span> <span class="pre">diagnosis</span></code>, <code class="docutils literal notranslate"><span class="pre">age</span> <span class="pre">at</span> <span class="pre">diagnosis</span> <span class="pre">years</span></code></p></td> 302</tr> 303<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">cell_line</span></code></p></td> 304<td><p><code class="docutils literal notranslate"><span class="pre">cell</span> <span class="pre">line</span></code></p></td> 305</tr> 306<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">sex</span></code></p></td> 307<td><p><code class="docutils literal notranslate"><span class="pre">sex</span></code>, <code class="docutils literal notranslate"><span class="pre">gender</span></code>, <code class="docutils literal notranslate"><span class="pre">subject</span> <span class="pre">gender</span></code>, <code class="docutils literal notranslate"><span class="pre">subject</span> <span class="pre">sex</span></code></p></td> 308</tr> 309</tbody> 310</table> 311<p>Values are stripped of white space and forced to lowercase.</p> 312<p>When multiple source keys that map to the same harmonized key are present in metadata from sources, we sort values in alphanumeric ascending order and concatenate them, separated by <code class="docutils literal notranslate"><span class="pre">;</span></code>. 313For example, a sample with <code class="docutils literal notranslate"><span class="pre">tissue:</span> <span class="pre">kidney</span></code> and <code class="docutils literal notranslate"><span class="pre">cell</span> <span class="pre">type:</span> <span class="pre">B</span> <span class="pre">cell</span></code> would become <code class="docutils literal notranslate"><span class="pre">specimen_part:</span> <span class="pre">B</span> <span class="pre">cell;kidney</span></code> when harmonized.</p> 314<p>We type-cast age values to doubles (e.g., <code class="docutils literal notranslate"><span class="pre">12</span></code> and <code class="docutils literal notranslate"><span class="pre">12</span> <span class="pre">weeks</span></code> both become <code class="docutils literal notranslate"><span class="pre">12.000</span></code>). 315Because of this type-casting behavior, we do not support multiple source keys; the value harmonized to <code class="docutils literal notranslate"><span class="pre">age</span></code> will be the first value that is encountered. 316If the values can not be type-cast to doubles (e.g., â9yrs 2mosâ), these are not added to the harmonized field. 317We do not attempt to normalize differences in units (e.g., months, years, days) for the harmonized age key. 318Users should consult the submitter-supplied information to determine what unit is used.</p> 319<p>Sex is a special case; we map to <code class="docutils literal notranslate"><span class="pre">female</span></code> and <code class="docutils literal notranslate"><span class="pre">male</span></code> values if the values are one of the following:</p> 320<table class="docutils align-default"> 321<thead> 322<tr class="row-odd"><th class="head text-center"><p>Harmonized <code class="docutils literal notranslate"><span class="pre">sex</span></code> value</p></th> 323<th class="head"><p>Values</p></th> 324</tr> 325</thead> 326<tbody> 327<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">female</span></code></p></td> 328<td><p><code class="docutils literal notranslate"><span class="pre">f</span></code>, <code class="docutils literal notranslate"><span class="pre">female</span></code>, <code class="docutils literal notranslate"><span class="pre">woman</span></code></p></td> 329</tr> 330<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">male</span></code></p></td> 331<td><p><code class="docutils literal notranslate"><span class="pre">m</span></code>, <code class="docutils literal notranslate"><span class="pre">male</span></code>, <code class="docutils literal notranslate"><span class="pre">man</span></code></p></td> 332</tr> 333</tbody> 334</table> 335<p>Only harmonized values are displayed in the sample table on the web interface. 336When downloading refine.bio data, these harmonized metadata are denoted with the <code class="docutils literal notranslate"><span class="pre">
336refinebio_</span></code> prefix.</p> 337<p>We recommend that users confirm metadata fields that are particularly important via the submitter-supplied metadata. If you find that the harmonized metadata does not accurately reflect the metadata supplied by the submitter, please <a href ="https://github.com/AlexsLemonade/refinebio/issues" target = "blank">file an issue on GitHub</a> so that we can resolve it. 338If you would prefer to report issues via e-mail, you can also email <a class="reference external" href="mailto:requests%40ccdatalab.org">requests<span>@</span>ccdatalab<span>.</span>org</a>.</p> 339</section> 340<section id="submitter-supplied-metadata"> 341<h3>Submitter Supplied Metadata<a class="headerlink" href="#submitter-supplied-metadata" title="Permalink to this heading">ï</a></h3> 342<p>We also capture the metadata as submitted to the source repositories. 343This includes experiment titles and descriptions or abstracts. 344Protocol information, which generally contains the type of sample preparation, is handled differently by different databases. 345We push this information down to the sample level and provide it that way. 346In the case of data imported from ArrayExpress, the protocol <em>may</em> be the information from the full experiment and not just the sample in question. 347Sample metadata in their original, unharmonized form are available as part of <a class="reference internal" href="#downloadable-files"><span class="xref myst">refine.bio downloads</span></a>.</p> 348</section> 349</section> 350</section> 351<section id="processing-information"> 352<h1>Processing Information<a class="headerlink" href="#processing-information" title="Permalink to this heading">ï</a></h1> 353<section id="refine-bio-processed"> 354<h2>refine.bio processed <img alt="refinebio-processedibadge" src="https://user-images.githubusercontent.com/15315514/44549308-b2621600-a6ee-11e8-897b-5cdcb5d1ed9d.png" /><a class="headerlink" href="#refine-bio-processed" title="Permalink to this heading">ï</a></h2> 355<p>Because refine.bio is designed to be consistently updated, we use processing and normalization methods that operate on a single sample wherever possible. 356Processing and normalization methods that require multiple samples (e.g., Robust Multi-array Average or RMA) generally have outputs that are influenced by whatever samples are included and rerunning these methods whenever a new sample is added to the system would be impractical.</p> 357<section id="microarray-pipelines"> 358<h3>Microarray pipelines<a class="headerlink" href="#microarray-pipelines" title="Permalink to this heading">ï</a></h3> 359<p><img alt="microarray-pipeline" src="https://user-images.githubusercontent.com/15315514/44549355-d3c30200-a6ee-11e8-9e1a-d5140df1f7d6.png" /></p> 360<section id="affymetrix"> 361<h4>Affymetrix<a class="headerlink" href="#affymetrix" title="Permalink to this heading">ï</a></h4> 362<p>SCAN (Single Channel Array Normalization) is a normalization method for develop for single channel Affymetrix microarrays that allows us to process individual samples. 363SCAN models and corrects for the effect of technical bias, such as GC content, using a mixture-modeling approach. 364For more information about this approach, see the primary publication (<a href = "http://dx.doi.org/10.1016/j.ygeno.2012.08.003" target = "blank">Piccolo, et al. <em>Genomics.</em> 2012.</a>) and the <a href = "https://www.bioconductor.org/packages/release/bioc/html/SCAN.UPC.html" target = "blank">SCAN.UPC Bioconductor package</a> documentation. 365We specifically use the <code class="docutils literal notranslate"><span class="pre">SCANfast</span></code> implementation of SCAN and the Brainarray packages as probe-summary packages when available. 366When available, we use <a href = "http://brainarray.mbni.med.umich.edu/Brainarray/Database/CustomCDF/CDF_download.asp" target = "blank">BrainArray Custom CDFs</a> during processing with SCAN.</p> 367<section id="affymetrix-platform-detection"> 368<h5>Affymetrix platform detection<a class="headerlink" href="#affymetrix-platform-detection" title="Permalink to this heading">ï</a></h5> 369<p>We have encountered instances where the platform label from the source repository and the metadata included in the sampleâs raw data file (<code class="docutils literal notranslate"><span class="pre">.CEL</span></code> file) itself do not match. 370In these cases, we take the platform information included in the raw data (<code class="docutils literal notranslate"><span class="pre">.CEL</span></code>) file header to be the true platform label.</p> 371</section> 372</section> 373<section id="illumina-beadarrays"> 374<h4>Illumina BeadArrays<a class="headerlink" href="#illumina-beadarrays" title="Permalink to this heading">ï</a></h4> 375<p>
375Dr. Stephen Piccolo, the developer of SCAN, has adapted the algorithm for use with Illumina BeadArrays for refine.bio. 376Because this Illumina SCAN methodology is not yet incorporated into the SCAN.UPC package, we briefly summarize the methods below.</p> 377<p>We require that non-normalized or raw expression values and detection p-values to be present in Illumina non-normalized data. 378If we infer that background correction has not occurred in the non-normalized data (e.g., there are no negative expression values), the data are background corrected using the <a href = "http://web.mit.edu/%7Er/current/arch/i386_linux26/lib/R/library/limma/html/nec.html" target = "blank"><code class="docutils literal notranslate"><span class="pre">limma::nec</span></code></a> function (<a href = "https://doi.org/10.1093/nar/gkq871" target = "blank">Shi, Oshlack, and Smyth. <em>Nucleic Acids Research.</em> 2010.</a>). 379Following background correction â either upstream presumably in the Illumina BeadStudio software or in our processor, arrays are normalized with SCAN. 380SCAN requires probe sequence information obtained from the <a href = "https://www.bioconductor.org/packages/release/BiocViews.html#___IlluminaChip" target = "blank">Illumina BeadArray Bioconductor annotation packages</a> (e.g., <a href = "https://www.bioconductor.org/packages/release/data/annotation/html/illuminaHumanv1.db.html" target = "blank"><code class="docutils literal notranslate"><span class="pre">illuminaHumanv1.db</span></code></a>). 381We only retain probes that have a âGoodâ or âPerfectâ rating in these packages; this quality rating is in reference to how well a probe is likely to measure its target transcript.</p> 382<section id="illumina-platform-detection"> 383<h5>Illumina platform detection<a class="headerlink" href="#illumina-platform-detection" title="Permalink to this heading">ï</a></h5> 384<p>We infer the Illumina BeadArray platform that a sample is likely to be run on by comparing the probe identifiers in the unprocessed file to probes for each of the Illumina expression arrays for a given organism. 385We again use the Illumina Bioconductor annotation packages for this step. 386For instance, the overlap between the probe identifiers in a human sample and the probe identifiers in each human platform (<a href = "https://www.bioconductor.org/packages/release/data/annotation/html/illuminaHumanv1.db.html" target = "blank"><code class="docutils literal notranslate"><span class="pre">v1</span></code></a>, <a href = "https://www.bioconductor.org/packages/release/data/annotation/html/illuminaHumanv2.db.html" target = "blank"><code class="docutils literal notranslate"><span class="pre">v2</span></code></a>, <a href = "https://www.bioconductor.org/packages/release/data/annotation/html/illuminaHumanv3.db.html" target = "blank"><code class="docutils literal notranslate"><span class="pre">v3</span></code></a>, and <a href = "https://www.bioconductor.org/packages/release/data/annotation/html/illuminaHumanv4.db.html" target = "blank"><code class="docutils literal notranslate"><span class="pre">v4</span></code></a>) is calculated. 387The platform with the highest overlap (provided it is >75%) is inferred to be the true platform. 388Some analyses around this platform detection procedure can be found in <a href = "https://github.com/jaclyn-taroni/beadarray-platform-detection" target = "blank">this repository</a>.</p> 389</section> 390<section id="handling-illumina-probes-that-map-to-multiple-ensembl-gene-identifiers"> 391<h5>Handling Illumina probes that map to multiple Ensembl gene identifiers<a class="headerlink" href="#handling-illumina-probes-that-map-to-multiple-ensembl-gene-identifiers" title="Permalink to this heading">ï</a></h5> 392<p>Illumina probes sometimes map to multiple Ensembl gene identifiers when using the annotation in <a href = "https://www.bioconductor.org/packages/release/BiocViews.html#___IlluminaChip" target = "blank">Illumina BeadArray Bioconductor annotation packages</a> (e.g., <a href = "https://www.bioconductor.org/packages/release/data/annotation/html/illuminaHumanv1.db.html" target = "blank"><code class="docutils literal notranslate"><span class="pre">illuminaHumanv1.db</span></code></a>). 393For human platforms in particular, these genes tend to be from highly polymorphic loci, e.g., Killer-cell immunoglobulin-like receptors. 394Because refine.bio allows users to combine samples from multiple platforms, we prioritize Ensembl gene identifiers that maximize compatibility with other platforms. 395Specifically, we select genes in order of priority as follows:</p> 396<ul class="simple"> 397<li><p>Pick the gene ID with the most appearances in BrainArray packages for Affymetrix platforms of the same species as the input Illumina platform</p></li> 398<li><p>If two or more of the associated gene IDs appear an equal number of times in BrainArray packages, or if none of the associated gene IDs appear in any BrainArray package, we break ties as follows:</p> 399<ul> 400<li><p>First, we check Ensembl and filter out any gene IDs that are no longer valid</p></li> 401<li><p>Next, if there are any Ensembl genes on the primary assembly, we take only the genes on the primary assembly and discard genes in <a hrefâhttps://www.ensembl.org/info/genome/genebuild/haplotypes_patches.htmlâ target = âblankâ>haplotypes (alternative versions of the genome) or patches</a>.</p></li> 402<li><p>If there are still two or more genes left, we pick the gene with the lowest Ensembl gene identifier to break the tie. This is an arbitrary but consistent way to break ties.</p></li> 403</ul> 404</li> 405</ul> 406<p>This is implemented in <a href="https://github.com/AlexsLemonade/illumina-refinery" target = "blank"><code class="docutils literal notranslate"><span class="pre">AlexsLemonade/illumina-refinery</span></code></a>.</p> 407</section> 408</section> 409</section> 410<section id="rna-seq-pipelines"> 411<h3>RNA-seq pipelines<a class="headerlink" href="#rna-seq-pipelines" title="Permalink to this heading">ï</a></h3> 412<p><img alt="rna-seq-pipeline" src="https://user-images.githubusercontent.com/15315514/44549339-c86fd680-a6ee-11e8-8d62-419ae7f10a94.png" /></p> 413<p>We use <a href = "https://combine-lab.github.io/salmon/" target = "blank">Salmon</a> and <a href = "https://bioconductor.org/packages/release/bioc/html/tximport.html" target = "blank">tximport</a> to process all RNA-seq data in refine.bio. 414We obtain sra files run on our <a href = "https://github.com/AlexsLemonade/refinebio/blob/dev/config/supported_rnaseq_platforms.txt" target = "blank">supported short-read platforms</a> from NCBI Sequence Read Archive. 415We use <a href = "https://ncbi.github.io/sra-tools/fastq-dump.html" target = "blank"><code class="docutils literal notranslate"><span class="pre">fastq-dump</span></code></a> to named pipes, which allows us to support paired-end experiments, and pass these to Salmon. 416Note that any unmated reads from paired experiments are discarded.</p> 417<p>We use the <strong>library strategy</strong> and <strong>library source</strong> metadata fields to identify RNA-seq experiments. 418Itâs possible that experiments that are inappropriate for use with Salmon will still appear in refine.bio (e.g., long-read platforms that are labeled incorrectly in the source repository). 419We also encounter trouble distinguishing single-cell and bulk RNA-seq data from these fields. 420We strongly recommend exercising caution when using single-cell data from refine.bio as the pipeline we use may be inappropriate (e.g., correcting for gene length in 3â tagged RNA-seq data induces bias [<a href = "https://bioconductor.org/packages/release/bioc/vignettes/tximport/inst/doc/tximport.html#tagged-rna-seq" target = "blank">ref</a>], Salmon TPM may overcorrect expression of long genes [<a href = "http://hemberg-lab.github.io/scRNA.seq.course/construction-of-expression-matrix.html#reads-alignment" target = "blank">ref</a>]). 421If you find an experiment that you believe is inappropriate for use with our pipeline, please <a href ="https://github.com/AlexsLemonade/refinebio/issues" target = "blank">file an issue on GitHub</a> so that we can resolve it. 422If you would prefer to report issues via e-mail, you can also email <a class="reference external" href="mailto:requests%40ccdatalab.org">requests<span>@</span>
422ccdatalab<span>.</span>org</a>.</p> 423<section id="salmon"> 424<h4>Salmon<a class="headerlink" href="#salmon" title="Permalink to this heading">ï</a></h4> 425<p>Salmon is an alignment-free method for estimating transcript abundances from RNA-seq data (<a href = "http://dx.doi.org/10.1038/nmeth.4197" target = "blank">Patro, et al. <em>Nature Methods</em>. 2017.</a>). 426We use it in <a href = "http://salmon.readthedocs.io/en/latest/salmon.html#preparing-transcriptome-indices-quasi-index-and-fmd-index-based-modes" target = "blank">quasi-mapping mode</a>, which is significantly faster than alignment-based approaches and requires us to build a Salmon transcriptome index.</p> 427<section id="transcriptome-index"> 428<h5>Transcriptome index<a class="headerlink" href="#transcriptome-index" title="Permalink to this heading">ï</a></h5> 429<p>We build a custom reference transcriptome (using <a href = "https://github.com/deweylab/RSEM" target = "blank">RSEM</a> <code class="docutils literal notranslate"><span class="pre">rsem-prepare-reference</span></code>) by filtering the Ensembl genomic DNA assembly to remove <em>pseudogenes</em>, which we expect could negatively impact the quantification of protein-coding genes. 430This means weâre obtaining abundance estimates for coding as well as non-coding transcripts.</p> 431<p>Building a transcriptome index with <code class="docutils literal notranslate"><span class="pre">salmon</span> <span class="pre">index</span></code> requires us to specify a value for the parameter <code class="docutils literal notranslate"><span class="pre">-k</span></code> that determines the size of the k-mers used for the index. 432The length of a read determines what k-mer size is appropriate. 433Consistent with the recommendations of the authors of Salmon, we use an index build with <em>k</em> = 31 when quantifying samples with reads with length > 75bp. 434We use <em>k</em> = 23 for shorter read lengths.</p> 435<p>The refine.bio processed Salmon indices are available for download. 436You can make use of our API like so:</p> 437<div class="highlight-default notranslate"><div class="highlight"><pre><span></span>https://api.refine.bio/v1/transcriptome_indices/?organism__name=<ORGANISM>&length=<LENGTH> 438</pre></div> 439</div> 440<p>Where <code class="docutils literal notranslate"><span class="pre"><ORGANISM></span></code> is the scientific name of the species in all caps separated by underscores and <code class="docutils literal notranslate"><span class="pre"><LENGTH></span></code> is either <code class="docutils literal notranslate"><span class="pre">SHORT</span></code> or <code class="docutils literal notranslate"><span class="pre">LONG</span></code>.</p> 441<p>To obtain the zebrafish (<em>Danio rerio</em>) index used for >75bp reads, use:</p> 442<div class="highlight-default notranslate"><div class="highlight"><pre><span></span>https://api.refine.bio/v1/transcriptome_indices/?organism__name=DANIO_RERIO&length=LONG 443</pre></div> 444</div> 445<p>The <code class="docutils literal notranslate"><span class="pre">download_url</span></code> field will allow you to download the index.</p> 446<section id="microbial-transcriptome-indices"> 447<h6>Microbial transcriptome indices<a class="headerlink" href="#microbial-transcriptome-indices" title="Permalink to this heading">ï</a></h6> 448<p>For an individual microbial species, there are often multiple genome assemblies available. 449Multiple assemblies from a species often reflect that multiple strains from that species have been characterized or established as laboratory strains. 450We use a single genome assembly, selected by domain experts, when we generate the transcriptome index used for each microbial species. 451The single genome assembly is typically from a common laboratory strain. 452This strainâs transcriptome index is used to quantify <em>all</em> RNA-seq data from that species, regardless of a sampleâs reported strain of origin. 453Quantifying using a single strain background allows us to generate compendia using a single transcriptome index, but most likely results in some loss of information for samples that do not originate from that strain.</p> 454<p>For a list of supported microbial species and the assemblies used, please see the <a class="reference external" href="https://github.com/AlexsLemonade/refinebio/blob/master/config/organism_strain_mapping.csv"><code class="docutils literal notranslate"><span class="pre">config/organism_strain_mapping.csv</span></code></a> file. 455For more context, please also visit the original GitHub issue at <a class="reference external" href="https://github.com/AlexsLemonade/refinebio/issues/1722"><code class="docutils literal notranslate"><span class="pre">AlexsLemonade/refinebio#1722</span></code></a>.</p> 456</section> 457</section> 458<section id="quantification-with-salmon"> 459<h5>Quantification with Salmon<a class="headerlink" href="#quantification-with-salmon" title="Permalink to this heading">ï</a></h5> 460<p>When quantifying transcripts with <code class="docutils literal notranslate"><span class="pre">salmon</span> <span class="pre">quant</span></code>, we take advantage of options that allow Salmon to learn and attempt to correct for certain biases in sequencing data. 461We include the flags <code class="docutils literal notranslate"><span class="pre">--seqBias</span></code> to correct for random hexamer priming and, if this is a <strong>
461paired-end</strong> experiment, <code class="docutils literal notranslate"><span class="pre">--gcBias</span></code> to correct for GC content when running salmon quant. 462We set the library type parameter such that Salmon will <a href ="http://salmon.readthedocs.io/en/latest/salmon.html#what-s-this-libtype" target = "blank">infer the sequencing library type automatically </a> for the reads it is quantifying (<code class="docutils literal notranslate"><span class="pre">-l</span> <span class="pre">A</span></code>).</p> 463</section> 464</section> 465<section id="tximport"> 466<h4>tximport<a class="headerlink" href="#tximport" title="Permalink to this heading">ï</a></h4> 467<p>Salmon quantification is at the <em>transcript-level</em>. 468To better integrate with the microarray data contained in refine.bio, we summarize the transcript-level information to the <em>gene-level</em> with <code class="docutils literal notranslate"><span class="pre">tximport</span></code> (<a href = "http://dx.doi.org/10.12688/f1000research.7563.1" target = "blank">Soneson, Love, and Robinson. <em>F1000 Research.</em> 2015.</a>).</p> 469<p>Our tximport implementation generates <a href ="https://www.rdocumentation.org/packages/tximport/versions/1.0.3/topics/tximport" target = "blank"> âlengthScaledTPMâ</a>, which are gene-level count-scale values that are generated by scaling TPM using the average transcript length across samples and the library size. 470Note that tximport is applied at the <em>experiment-level</em> rather than to single samples. 471For additional information, see the <a href= "http://bioconductor.org/packages/release/bioc/html/tximport.html" target = "blank"> tximport Bioconductor page </a>, the <a href="http://bioconductor.org/packages/release/bioc/vignettes/tximport/inst/doc/tximport.html" target ="blank">tximport tutorial <em>Importing transcript abundance datasets with tximport</em></a>, and <a href ="http://dx.doi.org/10.12688/f1000research.7563.1" target = "blank"> Soneson, Love, and Robinson. <em>F1000Research.</em> 2015.</a>.</p> 472<p>In some cases, all samples in an experiment can not be processing using Salmon (e.g., the file available for one sample is malformed). 473When experiments are >80% complete and contain more than 20 samples, we may run tximport on all available samples at the time. 474Weâve found the effect of running tximport âearlyâ on the resulting values to be small under these conditions. 475As a result, the tximport may be run on the same example multiple times; the most recent values will be available from refine.bio.</p> 476</section> 477</section> 478</section> 479<section id="submitter-processed"> 480<h2>Submitter processed <img alt="submitter-processed-badge" src="https://user-images.githubusercontent.com/15315514/44549307-b2621600-a6ee-11e8-9ef4-17b81d7728fd.png" /><a class="headerlink" href="#submitter-processed" title="Permalink to this heading">ï</a></h2> 481<p>Sometimes raw data for a sample is either unavailable at the source repository or exists in a form that we can not process. 482For microarray platforms that we support, we obtain the submitter processed expression data and use these values in refine.bio with some modification (e.g., log2-transformation where we detect it has not been performed).</p> 483<p>As noted above, we use Ensembl gene identifiers throughout refine.bio. 484Submitter processed data may use other gene (or probe) identifiers that we must convert to Ensembl gene identifiers. 485We describe the processes for Affymetrix and Illumina data below. 486Note in the case of one-to-many mappings when going from the ID used by the submitter to the Ensembl gene ID, expression values are duplicated: 487if a probe maps to two Ensembl gene ids, those two Ensembl gene ids will both have the probeâs expression value following conversion.</p> 488<section id="affymetrix-identifier-conversion"> 489<h3>Affymetrix identifier conversion<a class="headerlink" href="#affymetrix-identifier-conversion" title="Permalink to this heading">ï</a></h3> 490<p>We have created custom gene mapping files for most of the Affymetrix platforms we support.
491Briefly, for Brainarray supported platforms, we use the Brainarray (e.g., <code class="docutils literal notranslate"><span class="pre">hgu133plus2hsensgprobe</span></code>) and the platform-specific annotation package from Bioconductor (e.g., <code class="docutils literal notranslate"><span class="pre">hgu133plus2.db</span></code>) to generate a platform-specific mapping file that includes probe IDs, Ensembl gene IDs, gene symbols, Entrez IDs, RefSeq and Unigene identifiers. 492The rationale for only using probes or IDs that are accounted for in the Brainarray package is two-fold: 1) Brainarray packages are updated as we learn more about the genome and 2) it allows for these submitter processed data to be more consistent with refine.bio processed data. 493We support identifier conversion for a limited number of platforms that either do not have a Brainarray or Bioconductory annotation packages.</p> 494<p>The code for deriving these mappings and more details are available at https://github.com/AlexsLemonade/identifier-refinery. 495If you find an issue with these mappings, please <a href ="https://github.com/AlexsLemonade/refinebio/issues" target = "blank">file an issue on GitHub </a> so that we can resolve it. 496If you would prefer to report issues via e-mail, you can also email <a class="reference external" href="mailto:requests%40ccdatalab.org">requests<span>@</span>
496ccdatalab<span>.</span>org</a>.</p> 497</section> 498<section id="illumina-identifier-conversion"> 499<h3>Illumina identifier conversion<a class="headerlink" href="#illumina-identifier-conversion" title="Permalink to this heading">ï</a></h3> 500<p>We support conversion from Illumina BeadArray probe IDs to Ensembl gene IDs using 501<a href = "https://www.bioconductor.org/packages/release/BiocViews.html#___IlluminaChip" target ="blank">Bioconductor Illumina BeadArray expression packages</a>, 502allowing for one-to-many mappings.</p> 503</section> 504</section> 505<section id="aggregations"> 506<h2>Aggregations<a class="headerlink" href="#aggregations" title="Permalink to this heading">ï</a></h2> 507<p>refine.bio allows users to aggregate their selected samples in two ways: by experiment or by species. 508We use the term aggregate or aggregation to refer to the process of combining <em>individual samples</em> to form a <em>multi-sample</em> gene expression matrix (see also: <a class="reference internal" href="#downloadable-files"><span class="xref myst">Downloadable Files</span></a>).</p> 509<ul class="simple"> 510<li><p><strong>By experiment:</strong> Samples that belong to the same experiment will become a single gene expression matrix. 511If you have selected all samples from two experiments with 10 and 15 samples, respectively, and have chosen the <code class="docutils literal notranslate"><span class="pre">by</span> <span class="pre">experiment</span></code> option, you will receive two gene expression matrices with 10 and 15 samples, respectively.</p></li> 512<li><p><strong>By species:</strong> All samples assaying the same species will be aggregated into a single gene expression matrix. 513If you have selected three experiments each from human and mouse and the <code class="docutils literal notranslate"><span class="pre">by</span> <span class="pre">species</span></code> option, you receive two gene expression matrices that contain all human and all mouse samples, respectively.</p></li> 514</ul> 515<p>For either aggregation method, we summarize duplicate Ensembl gene IDs to the mean expression value and only include genes (rows) that are represented in <strong>all</strong> samples being aggregated. 516This is also known as an inner join and is illustrated below. 517<img alt="inner join" src="https://user-images.githubusercontent.com/15315514/44534751-7a46dd00-a6c6-11e8-9760-e8daa91a500f.png" /> 518Note that some early generation microarrays measure fewer genes than their more recent counterparts, so their inclusion when aggregating <code class="docutils literal notranslate"><span class="pre">by</span> <span class="pre">species</span></code> may result in a small number of genes being returned.</p> 519<section id="limitations-of-gene-identifiers-when-combining-across-platforms"> 520<h3>Limitations of gene identifiers when combining across platforms<a class="headerlink" href="#limitations-of-gene-identifiers-when-combining-across-platforms" title="Permalink to this heading">ï</a></h3> 521<p>We use Ensembl gene identifiers across refine.bio and, within platform, we use the same annotation to maintain consistency (e.g., all samples from the same Affymetrix platform use the same Brainarray package or identifier mapping derived from said package). 522However, Brainarray packages or Bioconductor annotation packages may be assembled from different genome builds compared each other or compared to the genome build used to generate transcriptome indices. 523If there tend to be considerable differences between (relatively) recent genome builds for your organism of interest or you are performing downstream analysis that would be sensitive to these differences, we do not recommend aggregating by species.</p> 524</section> 525</section> 526<section id="transformations"> 527<h2>Transformations<a class="headerlink" href="#transformations" title="Permalink to this heading">ï</a></h2> 528<section id="quantile-normalization"> 529<h3>Quantile normalization<a class="headerlink" href="#quantile-normalization" title="Permalink to this heading">ï</a></h3> 530<p>refine.bio is designed to allow for the aggregation of multiple platforms and even multiple technologies. 531With that in mind, we would like the distributions of samples from different platforms/technologies to be as similar as possible. 532We use <a href ="https://en.wikipedia.org/wiki/Quantile_normalization" target = "blank"> quantile normalization</a> to accomplish this. 533Specifically, we generate a reference distribution for each organism from a large body of data with the <code class="docutils literal notranslate"><span class="pre">normalize.quantiles.determine.target</span></code>
533 function from the <a href = "http://www.bioconductor.org/packages/release/bioc/html/preprocessCore.html" target ="blank">preprocessCore</a> R package and quantile normalize samples that a user selects for download with this target (using the <code class="docutils literal notranslate"><span class="pre">normalize.quantiles.use.target</span></code> function of <code class="docutils literal notranslate"><span class="pre">preprocessCore</span></code>). 534There is a single reference distribution per species, used to normalize all samples from that species regardless of platform or technology. 535We go into more detail below.</p> 536<section id="reference-distribution"> 537<h4>Reference distribution<a class="headerlink" href="#reference-distribution" title="Permalink to this heading">ï</a></h4> 538<p>By performing quantile normalization, we assume that the differences in expression values between samples arise solely from technical differences. 539This is not always the case; for instance, samples included in refine.bio are from multiple tissues. 540Weâll use as many samples as possible to generate the reference or target distribution. 541By including as diverse biological conditions as we have available to us to inform the reference distribution, we attempt to generate a tissue-agnostic consensus. 542To that end, we use the Affymetrix microarray platform with the largest number of samples for a given organism (e.g., <code class="docutils literal notranslate"><span class="pre">hgu133plus2</span></code> in humans) and only samples we have processed from raw as shown below.</p> 543<p><img alt="docs-ref-dist" src="https://user-images.githubusercontent.com/15315514/45969124-cb692a00-c000-11e8-9cfc-6317c92202c8.png" /></p> 544<section id="quantile-normalizing-your-own-data-with-refine-bio-reference-distribution"> 545<h5>Quantile normalizing your own data with refine.bio reference distribution<a class="headerlink" href="#quantile-normalizing-your-own-data-with-refine-bio-reference-distribution" title="Permalink to this heading">ï</a></h5> 546<p>refine.bio quantile normalization reference distribution or âtargetsâ are available for download. 547You may wish to use these to normalize your own data to make it more comparable to data you obtain from refine.bio.</p> 548<p>Quantile normalization targets can be obtained by first querying the API like so:</p> 549<div class="highlight-default notranslate"><div class="highlight"><pre><span></span><span class="n">https</span><span class="p">:</span><span class="o">//</span><span class="n">api</span><span class="o">.</span><span class="n">refine</span><span class="o">.</span><span class="n">bio</span><span class="o">/</span><span class="n">v1</span><span class="o">/</span><span class="n">qn_targets</span><span class="o">/<</span><span class="n">ORGANISM</span><span class="o">></span> 550</pre></div> 551</div> 552<p>Where <code class="docutils literal notranslate"><span class="pre"><ORGANISM></span></code> is the scientific name of the species separated by underscores.</p> 553<p>To obtain the zebrafish (<em>Danio rerio</em>) reference distribution, use:</p> 554<div class="highlight-default notranslate"><div class="highlight"><pre><span></span><span class="n">https</span><span class="p">:</span><span class="o">//</span><span class="n">api</span><span class="o">.</span><span class="n">refine</span><span class="o">.</span><span class="n">bio</span><span class="o">/</span><span class="n">v1</span><span class="o">/</span><span class="n">qn_targets</span><span class="o">/</span><span class="n">danio_rerio</span> 555</pre></div> 556</div> 557<p>The <code class="docutils literal notranslate"><span class="pre">s3_url</span></code> field will allow you to download the index.</p> 558</section> 559</section> 560<section id="quantile-normalizing-samples-for-delivery"> 561<h4>Quantile normalizing samples for delivery<a class="headerlink" href="#quantile-normalizing-samples-for-delivery" title="Permalink to this heading">ï</a></h4> 562<p>Once we have a reference/target distribution for a given organism, we use it to quantile normalize any samples that a user has selected for download. 563This quantile normalization step takes place <em>after</em> the summarization and inner join steps described above and illustrated below.</p> 564<p><img alt="docs-normalization" src="https://user-images.githubusercontent.com/15315514/46034575-f9b53b00-c0ce-11e8-893d-868a2aa98520.png" /></p> 565<p>As a result of the quantile normalization shown above, Sample 1 now has the same underlying distribution as the reference for that organism.</p> 566<p>Note that only gene expression matrices that we are able to successfully quantile normalize will be available for download.</p> 567</section> 568<section id="limitations-of-quantile-normalization-across-platforms-with-many-zeroes"> 569<h4>
569Limitations of quantile normalization across platforms with many zeroes<a class="headerlink" href="#limitations-of-quantile-normalization-across-platforms-with-many-zeroes" title="Permalink to this heading">ï</a></h4> 570<p>Quantile normalization is a strategy that can address many technical effects, generally at the cost of retaining certain sources of biological variability. 571We use a single reference distribution per organism, generated from the Affymetrix microarray platform with the largest number of samples we were able to process from raw data (see <a class="reference internal" href="#reference=distribution"><span class="xref myst"><em>Reference distribution</em></span></a>). 572In cases where the unnormalized data contains many ties within samples (and the ties are different between samples) the transformation can produce outputs with somewhat different distributions. 573This situation arises most often when RNA-seq data and microarray data are combined into a single dataset or matrix. 574To confirm that we have quantile normalized data correctly before returning results to the user, we evaluate the top half of expression values and confirm that a KS test produces a non-significant p-value. 575Users who seek to analyze RNA-seq and microarray data together should be aware that the low-expressing genes may not be comparable across the sets.</p> 576</section> 577<section id="skipping-quantile-normalization-for-rna-seq-experiments"> 578<h4>Skipping quantile normalization for RNA-seq experiments<a class="headerlink" href="#skipping-quantile-normalization-for-rna-seq-experiments" title="Permalink to this heading">ï</a></h4> 579<p>When selecting RNA-seq samples for download and to aggregate by experiment, users have the option to skip quantile normalization by first selecting Advanced Options and checking the âSkip quantile normalization for RNA-seq samplesâ box. 580In this case, the output of tximport will be delivered in TSV files (see <a class="reference internal" href="#tximport"><span class="xref myst">our section on RNA-seq data processing with tximport</span></a>). 581These data can be used for differential expression analysis as âbias corrected counts without an offsetâ as described in the <a href = "https://bioconductor.org/packages/release/bioc/vignettes/tximport/inst/doc/tximport.html#use-with-downstream-bioconductor-dge-packages" target = "blank"><em>Use with downstream Bioconductor DGE packages</em> section of tximport vignette</a>. 582Note that these data will be less comparable to other datasets from refine.bio because this step has been skipped.</p> 583</section> 584</section> 585<section id="gene-transformations"> 586<h3>Gene transformations<a class="headerlink" href="#gene-transformations" title="Permalink to this heading">ï</a></h3> 587<p>In some cases, it may be useful to row-normalize or transform the gene expression values in a matrix (e.g., following aggregation and quantile normalization). 588We offer the following options for transformations:</p> 589<ul class="simple"> 590<li><p><strong>None:</strong> No row-wise transformation is performed.</p></li> 591<li><p><strong>Z-score:</strong> Row values are <a href = "https://en.wikipedia.org/wiki/Standard_score" target = "blank">z-scored</a> using the <a href = "http://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.StandardScaler.html" target = "blank"><code class="docutils literal notranslate"><span class="pre">StandardScaler</span></code></a> from <a href = "http://scikit-learn.org/stable/index.html" target = "blank"><code class="docutils literal notranslate"><span class="pre">scikit-learn</span></code></a>. 592This transformation is useful for examining samplesâ gene expression values relative to the rest of the samples in the expression matrix (either all selected samples from that <em>species</em> when aggregating by species or all selected samples in an <em>experiment</em> when aggregating by experiment). 593If a sample has a positive value for a gene, that gene is more highly expressed in that sample compared to the mean of all samples; if that value is negative, that gene is less expressed compared to the population. 594It assumes that the data are normally distributed.</p></li> 595<li><p><strong>Zero to one:</strong> Rows are scaled to values <code class="docutils literal notranslate"><span class="pre">[0,1]</span></code> using the <a href = "http://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.MinMaxScaler.html" target = "blank"><code class="docutils literal notranslate"><span class="pre">MinMaxScaler</span></code></a> from <a href = "http://scikit-learn.org/stable/index.html" target = "blank"><code class="docutils literal notranslate"><span class="pre">scikit-learn</span></code></a>. 596We expect this transformation to be most useful for certain machine learning applications (e.g., those using cross-entropy as a loss function).</p></li> 597</ul> 598<p>In the plot below, we demonstrate the effect of different scaling options on gene expression values (using a randomly selected human dataset, microarray platform, and gene):</p> 599<img src="https://user-images.githubusercontent.com/19534205/46036353-50247880-c0d3-11e8-9b7f-07818e545d68.png" width="480"> 600<p>Note that the distributions retain the same general <em>shape</em>, but the range of values and the density are altered by the transformations.</p> 601</section> 602</section> 603</section> 604<section id="downloadable-files"> 605<h1>Downloadable Files<a class="headerlink" href="#downloadable-files" title="Permalink to this heading">ï</a></h1> 606<p>Users can download gene expression data and associated sample and experiment metadata from refine.bio.
607These files are delivered as a zip file. 608The folder structure within the zip file is determined by whether a user selected to aggregate by <strong>experiment</strong> or by <strong>species</strong>.</p> 609<section id="the-download-folder-structure-for-data-aggregated-by-experiment"> 610<h2>The download folder structure for data aggregated by experiment:<a class="headerlink" href="#the-download-folder-structure-for-data-aggregated-by-experiment" title="Permalink to this heading">ï</a></h2> 611<p><img alt="docs-downloads-experiment-agg" src="https://user-images.githubusercontent.com/15315514/45906716-2f9eaa80-bdc3-11e8-9855-2aaeb74e588d.png" /></p> 612<p>In this example, two experiments were selected. 613There will be as many folders as there are selected experiments.</p> 614</section> 615<section id="the-download-folder-structure-for-data-aggregated-by-species"> 616<h2>The download folder structure for data aggregated by species:<a class="headerlink" href="#the-download-folder-structure-for-data-aggregated-by-species" title="Permalink to this heading">ï</a></h2> 617<p><img alt="docs-downloads-species-agg" src="https://user-images.githubusercontent.com/15315514/45906715-2f9eaa80-bdc3-11e8-8ab3-90ccc40cfa11.png" /></p> 618<p>In this example, samples from two species were selected. 619There will be as many folders as there are selected experiments and this will be the case regardless of how many individual experiments were included.</p> 620<p>In both cases, <code class="docutils literal notranslate"><span class="pre">aggregated_metadata.json</span></code> contains metadata, including both <em>experiment</em> metadata (e.g., experiment description and title) and <em>sample</em> metadata for everything included in the download. 621Below we describe the files included in the delivered zip file.</p> 622</section> 623<section id="gene-expression-matrix"> 624<h2>Gene Expression Matrix<a class="headerlink" href="#gene-expression-matrix" title="Permalink to this heading">ï</a></h2> 625<p>Gene expression matrices are delived in <a href = "https://en.wikipedia.org/wiki/Tab-separated_values" target = "blank">tab-separated value</a> (TSV) format. 626In these matrices, rows are <em>genes</em> or <em>features</em> and columns are <em>samples</em>. 627Note that this format is consistent with the input expected by many programs specifically designed for working with gene expression matrices, but some machine learning libraries will expect this to be transposed. 628The column names or header will contain values corresponding to sample accessions (denoted <code class="docutils literal notranslate"><span class="pre">refinebio_accession_code</span></code> in metadata files). 629You can use these values in the header to map between a sampleâs gene expression data and its metadata (e.g., disease label or age). See also <a class="reference internal" href="#downstream-analysis-with-refinebio-examples"><span class="xref myst">Downstream Analysis with refine.bio Examples</span></a>.</p> 630</section> 631<section id="sample-metadata"> 632<h2>Sample Metadata<a class="headerlink" href="#sample-metadata" title="Permalink to this heading">ï</a></h2> 633<p>Sample metadata is delivered in the <code class="docutils literal notranslate"><span class="pre">metadata_<experiment-accession-id>.tsv</span></code>, <code class="docutils literal notranslate"><span class="pre">metadata_<species>.json</span></code>, <code class="docutils literal notranslate"><span class="pre">metadata_<species>.tsv</span></code>, and <code class="docutils literal notranslate"><span class="pre">aggregated_metadata.json</span></code> files.</p> 634<p>The primary way we identify samples is by using the sample accession, denoted by <code class="docutils literal notranslate"><span class="pre">refinebio_accession_code</span></code>. 635Harmonized metadata fields (see the <a class="reference internal" href="#refine.bio-harmonized-metadata"><span class="xref myst">section on harmonized metadata</span></a>) are noted with a <code class="docutils literal notranslate"><span class="pre">refinebio_</span></code> prefix. 636The <code class="docutils literal notranslate"><span class="pre">refinebio_source_archive_url</span></code> and <code class="docutils literal notranslate"><span class="pre">refinebio_source_database</span></code> fields indicate where the sample was obtained from. 637If there are no keys from the source data associated with a harmonized key, the harmonized metadata field will be empty. 638We also deliver submitter-supplied data; see below for more details. 639<strong>We recommend that users confirm metadata fields that are particularly important via the submitter-supplied metadata.</strong> 640If you find that refine.bio metadata does not accurately reflect the metadata supplied by the submitter, please <a href ="https://github.com/AlexsLemonade/refinebio/issues" target = "blank">file an issue on GitHub</a> so that we can resolve it. 641If you would prefer to report issues via e-mail, you can also email <a class="reference external" href="mailto:requests%40ccdatalab.org">requests<span>@</span>
641ccdatalab<span>.</span>org</a>.</p> 642<section id="tsv-files"> 643<h3>TSV files<a class="headerlink" href="#tsv-files" title="Permalink to this heading">ï</a></h3> 644<p>In metadata TSV files, samples are represented as rows. 645The first column contains the <code class="docutils literal notranslate"><span class="pre">refinebio_accession_code</span></code> field, which match the header/column names in the gene expression matrix, followed by refine.bio-harmonized fields (e.g., <code class="docutils literal notranslate"><span class="pre">refinebio_</span></code>), and finally submitter-supplied values. 646Some information from source repositories comes in the form of nested values, which we attempt to âflattenâ where possible. 647Note that some information from source repositories is redundantâArrayExpress samples often have the same information in <code class="docutils literal notranslate"><span class="pre">characteristic</span></code> and <code class="docutils literal notranslate"><span class="pre">variable</span></code> fieldsâand we <em>assume</em> that if a field appears in both, the values are identical. 648For samples run on Illumina BeadArray platforms, information about what platform we detected and the metrics used to make that determination will also be included.</p> 649<p>Columns in these files will often have missing values, as not all fields will be available for every sample included in a file. 650This will be particularly evident when aggregating by experiments that have different submitter-supplied information associated with them (e.g., one experiment contains a <code class="docutils literal notranslate"><span class="pre">imatinib</span></code> key and all others do not).</p> 651</section> 652<section id="json-files"> 653<h3>JSON files<a class="headerlink" href="#json-files" title="Permalink to this heading">ï</a></h3> 654<p>Submitter supplied metadata and source urls are delivered in <code class="docutils literal notranslate"><span class="pre">refinebio_annotations</span></code>. 655As described above, harmonized values are noted with a <code class="docutils literal notranslate"><span class="pre">refinebio_</span></code> prefix.</p> 656</section> 657</section> 658<section id="experiment-metadata"> 659<h2>Experiment Metadata<a class="headerlink" href="#experiment-metadata" title="Permalink to this heading">ï</a></h2> 660<p>Experiment metadata (e.g., experiment description and title) is delivered in the <code class="docutils literal notranslate"><span class="pre">metadata_<species>.json</span></code> and <code class="docutils literal notranslate"><span class="pre">aggregated_metadata.json</span></code> files.</p> 661<p>The <code class="docutils literal notranslate"><span class="pre">aggregated_metadata.json</span></code> file contains additional information regarding the processing of your dataset. 662Specifically, the <code class="docutils literal notranslate"><span class="pre">aggregate_by</span></code> and <code class="docutils literal notranslate"><span class="pre">scale_by</span></code> fields note how the samples are grouped into gene expression matrices and how the gene expression data values were transformed, respectively. 663The <code class="docutils literal notranslate"><span class="pre">quantile_normalized</span></code> fields notes whether or not quantile normalization was performed. 664Currently, we only support skipping quantile normalization for RNA-seq experiments when aggregating by experiment on the web interface.</p> 665</section> 666</section> 667<section id="refine-bio-compendia"> 668<h1>refine.bio Compendia<a class="headerlink" href="#refine-bio-compendia" title="Permalink to this heading">ï</a></h1> 669<p>We periodically release compendia comprised of all the samples from a species that we were able to process. 670We refer to these as <strong>refine.bio compendia</strong>. 671We offer two kinds of refine.bio compendia: <a class="reference internal" href="#normalized-compendia"><span class="xref myst">normalized compendia</span></a> and <a class="reference internal" href="#rna-seq-sample-compendia"><span class="xref myst">RNA-seq sample compendia</span></a>.</p> 672<section id="normalized-compendia"> 673<h2>Normalized compendia<a class="headerlink" href="#normalized-compendia" title="Permalink to this heading">ï</a></h2> 674<p>refine.bio normalized compendia are comprised of all the samples from a species that we were able to process, aggregate, and normalize. 675Normalized compendia provide a snapshot of the most complete collection of gene expression that refine.bio can produce for each supported organism. 676We process these compendia in a manner that is different from the options that are available via the web user interface. 677Note that <a class="reference internal" href="#submitter-processed--"><span class="xref myst">submitter processed</span></a> samples that are available through the web user interface are omitted from normalized compendia because <a class="reference external" href="https://github.com/AlexsLemonade/refinebio/issues/2114">
677these samples can introduce unwanted technical variation</a>.</p> 678<p>The refine.bio web interface does an inner join when datasets are combined, so only genes present in all datasets are included in the final matrix. 679For compendia, we take the union of all genes, filling in any missing values with <code class="docutils literal notranslate"><span class="pre">NA</span></code>. 680This is a âfull outer joinâ as illustrated below. 681We use a full outer join because it allows us to retain more genes in a compendium and we impute missing values during compendia creation.</p> 682<p><img alt="outer join" src="https://user-images.githubusercontent.com/15315514/44534241-4dde9100-a6c5-11e8-8a9c-aa147e294e81.png" /></p> 683<p>We perform an outer join each time samples are combined in the process of building normalized compendia.</p> 684<p><img alt="docs-normalized-compendia" src="https://user-images.githubusercontent.com/15315514/65698014-ddcdd780-e049-11e9-8ed6-f1d2f8ac2ee7.png" /></p> 685<p>Samples from each technologyâmicroarray and RNA-seqâare combined separately. 686In RNA-seq samples, we filter out genes with low total counts and then <code class="docutils literal notranslate"><span class="pre">log2(x</span> <span class="pre">+</span> <span class="pre">1)</span></code> the data. 687We join samples from both technologies. 688We then drop genes that have missing values in greater than 30% of samples. 689Finally, we drop samples that have missing values in greater than 50% of genes. 690We impute the remaining missing values with IterativeSVD from <a href = "https://pypi.org/project/fancyimpute/" target = "blank">fancyimpute</a>. 691We then quantile normalize all samples as described above.</p> 692<p>Weâve made our analyses underlying processing choices and exploring test compendia available at our <a href = "https://github.com/AlexsLemonade/compendium-processing" target = "blank"><code class="docutils literal notranslate"><span class="pre">compendium-processing</span></code></a> repository.</p> 693<section id="collapsing-by-genus"> 694<h3>Collapsing by genus<a class="headerlink" href="#collapsing-by-genus" title="Permalink to this heading">ï</a></h3> 695<p>Microarray platforms are generally designed to assay samples from a specific species. 696In some cases, publicly available data surveyed by refine.bio may include samples where the microarray platform used was not specifically designed for the species as described (e.g., samples labeled <em>Bos indicus</em> were run on <em>Bos taurus</em> microarrays or mouse crosses that are not labeled <em>Mus musculus</em> were run on <em>Mus musculus</em> microarrays). 697When we encounter this in refine.bio, we will include samples in a compendium from species that differ from the primary platform species when the two species share a genus (e.g., <em>Bos indicus</em> samples run on <em>Bos taurus</em> microarrays are included in the <em>Bos taurus</em> normalized compendium, and <em>Mus</em> crosses are included in the <em>Mus musculus</em> normalized compendium). 698Such non-primary species samples generally account for a small fraction of the total samples included in a normalized compendium. 699If you would like to filter a normalized compendium based on a sampleâs species label, you can use the <code class="docutils literal notranslate"><span class="pre">refinebio_organism</span></code> column in the metadata TSV file or the <code class="docutils literal notranslate"><span class="pre">.samples[].refinebio_organism</span></code> field in the metadata JSON file included as part of the download.</p> 700<p>Note that non-primary species samples from species that are outside the genus of the primary platform species are not currently available in any normalized compendium (e.g., <em>Pan troglodytes</em> samples assayed on <em>Homo sapiens</em> microarrays are not included in the <em>Pan troglodytes</em> or <em>Homo sapiens</em> compendia), but can be included in datasets from refine.bio.</p> 701<p>Below is the list of organisms and their primary organisms:</p> 702<table class="docutils align-default"> 703<thead> 704<tr class="row-odd"><th class="head text-center"><p>Primary Organism</p></th> 705<th class="head"><p>Organisms included in compendium</p></th> 706</tr> 707</thead> 708<tbody> 709<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Anopheles</span> <span class="pre">
709gambiae</span></code></p></td> 710<td><p><code class="docutils literal notranslate"><span class="pre">Anopheles</span> <span class="pre">gambiae</span></code></p></td> 711</tr> 712<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Arabidopsis</span> <span class="pre">thaliana</span></code></p></td> 713<td><p><code class="docutils literal notranslate"><span class="pre">Arabidopsis</span> <span class="pre">thaliana</span></code>, <code class="docutils literal notranslate"><span class="pre">Arabidopsis</span> <span class="pre">halleri</span></code>, <code class="docutils literal notranslate"><span class="pre">Arabidopsis</span> <span class="pre">thaliana</span> <span class="pre">x</span> <span class="pre">arabidopsis</span> <span class="pre">halleri</span> <span class="pre">subsp.</span> <span class="pre">gemmifera</span></code>, <code class="docutils literal notranslate"><span class="pre">Arabidopsis</span> <span class="pre">lyrata</span> <span class="pre">subsp.</span> <span class="pre">petraea</span></code>, <code class="docutils literal notranslate"><span class="pre">Arabidopsis</span> <span class="pre">lyrata</span> <span class="pre">subsp.</span> <span class="pre">lyrata</span></code>, <code class="docutils literal notranslate"><span class="pre">Arabidopsis</span> <span class="pre">thaliana</span> <span class="pre">x</span> <span class="pre">arabidopsis</span> <span class="pre">lyrata</span></code>, <code class="docutils literal notranslate"><span class="pre">Arabidopsis</span> <span class="pre">halleri</span> <span class="pre">subsp.</span> <span class="pre">gemmifera</span></code>, <code class="docutils literal notranslate"><span class="pre">Arabidopsis</span> <span class="pre">lyrata</span></code></p></td> 714</tr> 715<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Bos</span> <span class="pre">indicus</span></code></p></td> 716<td><p><code class="docutils literal notranslate"><span class="pre">Bos</span> <span class="pre">indicus</span></code></p></td> 717</tr> 718<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Bos</span> <span class="pre">taurus</span></code></p></td> 719<td><p><code class="docutils literal notranslate"><span class="pre">Bos</span> <span class="pre">taurus</span></code>, <code class="docutils literal notranslate"><span class="pre">Bos</span> <span class="pre">indicus</span></code>, <code class="docutils literal notranslate"><span class="pre">Bos</span> <span class="pre">grunniens</span></code></p></td> 720</tr> 721<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Caenorhabditis</span> <span class="pre">elegans</span></code></p></td> 722<td><p><code class="docutils literal notranslate"><span class="pre">Caenorhabditis</span> <span class="pre">elegans</span></code></p></td> 723</tr> 724<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Citrus</span> <span class="pre">sinensis</span></code></p></td> 725<td><p><code class="docutils literal notranslate"><span class="pre">Citrus</span> <span class="pre">x</span> <span class="pre">paradisi</span></code>, <code class="docutils literal notranslate"><span class="pre">Citrus</span> <span class="pre">reticulata</span></code>, <code class="docutils literal notranslate"><span class="pre">Citrus</span> <span class="pre">sinensis</span></code>, <code class="docutils literal notranslate"><span class="pre">Citrus</span> <span class="pre">limon</span></code>, <code class="docutils literal notranslate"><span class="pre">Citrus</span> <span class="pre">reticulata</span> <span class="pre">x</span> <span class="pre">citrus</span> <span class="pre">trifoliata</span></code>, <code class="docutils literal notranslate"><span class="pre">Citrus</span> <span class="pre">clementina</span></code>, <code class="docutils literal notranslate"><span class="pre">Citrus</span> <span class="pre">unshiu</span></code>, <code class="docutils literal notranslate"><span class="pre">Citrus</span> <span class="pre">x</span> <span class="pre">tangelo</span></code>, <code class="docutils literal notranslate"><span class="pre">Citrus</span> <span class="pre">maxima</span></code></p></td> 726</tr> 727<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Danio</span> <span class="pre">rerio</span></code></p></td> 728<td><p><code class="docutils literal notranslate"><span class="pre">Danio</span> <span class="pre">rerio</span></code></p></td> 729</tr> 730<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Drosophila</span> <span class="pre">melanogaster</span></code></p></td> 731<td><p><code class="docutils literal notranslate"><span class="pre">Drosophila</span> <span class="pre">melanogaster</span></code>, <code class="docutils literal notranslate"><span class="pre">Drosophila</span> <span class="pre">simulans</span></code>, <code class="docutils literal notranslate"><span class="pre">Drosophila</span> <span class="pre">mauritiana</span></code>, <code class="docutils literal notranslate"><span class="pre">Drosophila</span> <span class="pre">sechellia</span></code>, <code class="docutils literal notranslate"><span class="pre">Drosophila</span> <span class="pre">teissieri</span></code>, <code class="docutils literal notranslate"><span class="pre">Drosophila</span> <span class="pre">santomea</span></code>, <code class="docutils literal notranslate"><span class="pre">Drosophila</span> <span class="pre">yakuba</span></code></p></td> 732</tr> 733<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span></code></p></td> 734<td><p><code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span></code>, <code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">str.</span> <span class="pre">k-12</span> <span class="pre">substr.</span> <span class="pre">mg1655</span></code>, <code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">k-12</span></code>, <code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">cft073</span></code>, <code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">str.</span> <span class="pre">k-12</span> <span class="pre">substr.</span> <span class="pre">w3110</span></code>, <code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">uti89</span></code>, <code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">b</span> <span class="pre">str.</span> <span class="pre">rel606</span></code>, <code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">o157</span></code>, <code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">8624</span></code>, <code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">sci-07</span></code>, <code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">bw25113</span></code>, <code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">apec</span> <span class="pre">o2</span></code>, <code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">str.</span> <span class="pre">k-12</span> <span class="pre">substr.</span> <span class="pre">mc4100</span></code>, <code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">o08</span></code>, <code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">str.</span> <span class="pre">k-12</span> <span class="pre">substr.</span> <span class="pre">dh10b</span></code></p></td> 735</tr> 736<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">k-12</span></code></p></td> 737<td><p><code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">k-12</span></code></p></td> 738</tr> 739<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">str.</span> <span class="pre">k-12</span> <span class="pre">substr.</span> <span class="pre">mg1655</span></code></p></td> 740<td><p><code class="docutils literal notranslate"><span class="pre">Escherichia</span> <span class="pre">coli</span> <span class="pre">str.</span> <span class="pre">k-12</span> <span class="pre">substr.</span> <span class="pre">mg1655</span></code></p></td> 741</tr> 742<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Gallus</span> <span class="pre">gallus</span></code></p></td> 743<td><p><code class="docutils literal notranslate"><span class="pre">Gallus</span> <span class="pre">gallus</span></code></p></td> 744</tr> 745<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Glycine</span> <span class="pre">max</span></code></p></td> 746<td><p><code class="docutils literal notranslate"><span class="pre">Glycine</span> <span class="pre">max</span></code>, <code class="docutils literal notranslate"><span class="pre">Glycine</span> <span class="pre">soja</span></code></p></td> 747</tr> 748<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Gossypium</span> <span class="pre">hirsutum</span></code></p></td> 749<td><p><code class="docutils literal notranslate"><span class="pre">Gossypium</span> <span class="pre">herbaceum</span></code>, <code class="docutils literal notranslate"><span class="pre">Gossypium</span> <span class="pre">hirsutum</span></code>, <code class="docutils literal notranslate"><span class="pre">Gossypium</span> <span class="pre">barbadense</span></code>, <code class="docutils literal notranslate"><span class="pre">Gossypium</span> <span class="pre">arboreum</span></code></p></td> 750</tr> 751<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Homo</span> <span class="pre">sapiens</span></code></p></td> 752<td><p><code class="docutils literal notranslate"><span class="pre">Homo</span> <span class="pre">sapiens</span></code></p></td> 753</tr> 754<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Hordeum</span> <span class="pre">vulgare</span></code></p></td> 755<td><p><code class="docutils literal notranslate"><span class="pre">Hordeum</span> <span class="pre">vulgare</span></code>, <code class="docutils literal notranslate"><span class="pre">Hordeum</span> <span class="pre">vulgare</span> <span class="pre">subsp.</span> <span class="pre">spontaneum</span></code></p></td> 756</tr> 757<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Lepidium</span> <span class="pre">sativum</span></code></p></td> 758<td><p><code class="docutils literal notranslate"><span class="pre">Lepidium</span> <span class="pre">sativum</span></code></p></td> 759</tr> 760<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Macaca</span> <span class="pre">fascicularis</span></code></p></td> 761<td><p><code class="docutils literal notranslate"><span class="pre">Macaca</span> <span class="pre">fascicularis</span></code></p></td> 762</tr> 763<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Macaca</span> <span class="pre">mulatta</span></code></p></td> 764<td><p><code class="docutils literal notranslate"><span class="pre">Macaca</span> <span class="pre">mulatta</span></code></p></td> 765</tr> 766<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Musculus</span></code></p></td> 767<td><p><code class="docutils literal notranslate"><span class="pre">Musculus</span></code></p></td> 768</tr> 769<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Mus</span> <span class="pre">musculus</span></code></p></td> 770<td><p><code class="docutils literal notranslate"><span class="pre">Mus</span> <span class="pre">musculus</span></code>, <code class="docutils literal notranslate"><span class="pre">Mus</span> <span class="pre">spretus</span></code>, <code class="docutils literal notranslate"><span class="pre">Mus</span> <span class="pre">
770caroli</span></code>, <code class="docutils literal notranslate"><span class="pre">Mus</span> <span class="pre">musculus</span> <span class="pre">musculus</span> <span class="pre">x</span> <span class="pre">m.</span> <span class="pre">m.</span> <span class="pre">domesticus</span></code>, <code class="docutils literal notranslate"><span class="pre">Mus</span> <span class="pre">musculus</span> <span class="pre">domesticus</span></code>, <code class="docutils literal notranslate"><span class="pre">Mus</span> <span class="pre">musculus</span> <span class="pre">x</span> <span class="pre">mus</span> <span class="pre">spretus</span></code>, <code class="docutils literal notranslate"><span class="pre">Mus</span> <span class="pre">musculus</span> <span class="pre">musculus</span> <span class="pre">x</span> <span class="pre">m.</span> <span class="pre">m.</span> <span class="pre">castaneus</span></code>, <code class="docutils literal notranslate"><span class="pre">Mus</span> <span class="pre">musculus</span> <span class="pre">musculus</span></code>, <code class="docutils literal notranslate"><span class="pre">Mus</span> <span class="pre">musculus</span> <span class="pre">castaneus</span></code>, <code class="docutils literal notranslate"><span class="pre">Mus</span> <span class="pre">sp.</span></code></p></td> 771</tr> 772<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Mustela</span> <span class="pre">putorius</span> <span class="pre">furo</span></code></p></td> 773<td><p><code class="docutils literal notranslate"><span class="pre">Mustela</span> <span class="pre">putorius</span> <span class="pre">furo</span></code></p></td> 774</tr> 775<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Oryza</span> <span class="pre">sativa</span></code></p></td> 776<td><p><code class="docutils literal notranslate"><span class="pre">Oryza</span> <span class="pre">sativa</span> <span class="pre">japonica</span></code>, <code class="docutils literal notranslate"><span class="pre">Oryza</span> <span class="pre">sativa</span></code>, <code class="docutils literal notranslate"><span class="pre">Oryza</span> <span class="pre">sativa</span> <span class="pre">indica</span> <span class="pre">group</span></code>, <code class="docutils literal notranslate"><span class="pre">Oryza</span> <span class="pre">longistaminata</span></code></p></td> 777</tr> 778<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Oryza</span> <span class="pre">sativa</span> <span class="pre">indica</span> <span class="pre">group</span></code></p></td> 779<td><p><code class="docutils literal notranslate"><span class="pre">Oryza</span> <span class="pre">sativa</span> <span class="pre">indica</span> <span class="pre">group</span></code></p></td> 780</tr> 781<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Plasmodium</span> <span class="pre">falciparum</span></code></p></td> 782<td><p><code class="docutils literal notranslate"><span class="pre">Plasmodium</span> <span class="pre">falciparum</span> <span class="pre">3d7</span></code>, <code class="docutils literal notranslate"><span class="pre">Plasmodium</span> <span class="pre">falciparum</span></code></p></td> 783</tr> 784<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Populus</span> <span class="pre">tremula</span> <span class="pre">x</span> <span class="pre">populus</span> <span class="pre">alba</span></code></p></td> 785<td><p><code class="docutils literal notranslate"><span class="pre">Populus</span> <span class="pre">tremula</span> <span class="pre">x</span> <span class="pre">populus</span> <span class="pre">alba</span></code></p></td> 786</tr> 787<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Populus</span> <span class="pre">trichocarpa</span></code></p></td> 788<td><p><code class="docutils literal notranslate"><span class="pre">Populus</span> <span class="pre">trichocarpa</span></code></p></td> 789</tr> 790<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Populus</span> <span class="pre">x</span> <span class="pre">canadensis</span></code></p></td> 791<td><p><code class="docutils literal notranslate"><span class="pre">Populus</span> <span class="pre">x</span> <span class="pre">canadensis</span></code></p></td> 792</tr> 793<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Pseudomonas</span> <span class="pre">aeruginosa</span></code></p></td> 794<td><p><code class="docutils literal notranslate"><span class="pre">Pseudomonas</span> <span class="pre">aeruginosa</span></code>, <code class="docutils literal notranslate"><span class="pre">Pseudomonas</span> <span class="pre">aeruginosa</span> <span class="pre">pao1</span></code>, <code class="docutils literal notranslate"><span class="pre">Pseudomonas</span> <span class="pre">putida</span></code>, <code class="docutils literal notranslate"><span class="pre">Pseudomonas</span> <span class="pre">aeruginosa</span> <span class="pre">ucbpp-pa14</span></code>, <code class="docutils literal notranslate"><span class="pre">Pseudomonas</span> <span class="pre">aeruginosa</span> <span class="pre">pa14</span></code>, <code class="docutils literal notranslate"><span class="pre">Pseudomonas</span> <span class="pre">aeruginosa</span> <span class="pre">tbcf10839</span></code>, <code class="docutils literal notranslate"><span class="pre">Pseudomonas</span> <span class="pre">aeruginosa</span> <span class="pre">pahm4</span></code></p></td> 795</tr> 796<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Pseudomonas</span> <span class="pre">aeruginosa</span> <span class="pre">pao1</span></code></p></td> 797<td><p><code class="docutils literal notranslate"><span class="pre">Pseudomonas</span> <span class="pre">aeruginosa</span> <span class="pre">pao1</span></code></p></td> 798</tr> 799<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Rattus</span> <span class="pre">norvegicus</span></code></p></td> 800<td><p><code class="docutils literal notranslate"><span class="pre">Rattus</span> <span class="pre">norvegicus</span></code>, <code class="docutils literal notranslate"><span class="pre">Rattus</span> <span class="pre">rattus</span></code>, <code class="docutils literal notranslate"><span class="pre">Rattus</span> <span class="pre">norvegicus</span> <span class="pre">albus</span></code></p></td> 801</tr> 802<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">
802Saccharomyces</span> <span class="pre">cerevisiae</span></code></p></td> 803<td><p><code class="docutils literal notranslate"><span class="pre">Saccharomyces</span> <span class="pre">cerevisiae</span></code>, <code class="docutils literal notranslate"><span class="pre">Saccharomyces</span> <span class="pre">cerevisiae</span> <span class="pre">s288c</span></code>, <code class="docutils literal notranslate"><span class="pre">Saccharomyces</span> <span class="pre">pastorianus</span></code>, <code class="docutils literal notranslate"><span class="pre">Saccharomyces</span> <span class="pre">pastorianus</span> <span class="pre">weihenstephan</span> <span class="pre">34/70</span></code>, <code class="docutils literal notranslate"><span class="pre">Saccharomyces</span> <span class="pre">cerevisiae</span> <span class="pre">vin13</span></code>, <code class="docutils literal notranslate"><span class="pre">Saccharomyces</span> <span class="pre">uvarum</span></code>, <code class="docutils literal notranslate"><span class="pre">Saccharomyces</span> <span class="pre">cerevisiae</span> <span class="pre">ec1118</span></code>, <code class="docutils literal notranslate"><span class="pre">Saccharomyces</span> <span class="pre">cerevisiae</span> <span class="pre">cen.pk113-7d</span></code>, <code class="docutils literal notranslate"><span class="pre">Saccharomyces</span> <span class="pre">cerevisiae</span> <span class="pre">by4741</span></code>, <code class="docutils literal notranslate"><span class="pre">Saccharomyces</span> <span class="pre">cerevisiae</span> <span class="pre">sk1</span></code>, <code class="docutils literal notranslate"><span class="pre">Saccharomyces</span> <span class="pre">bayanus</span></code>, <code class="docutils literal notranslate"><span class="pre">Saccharomyces</span> <span class="pre">cerevisiae</span> <span class="pre">x</span> <span class="pre">saccharomyces</span> <span class="pre">kudriavzevii</span></code>, <code class="docutils literal notranslate"><span class="pre">Saccharomyces</span> <span class="pre">boulardii</span></code></p></td> 804</tr> 805<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Schizosaccharomyces</span> <span class="pre">pombe</span></code></p></td> 806<td><p><code class="docutils literal notranslate"><span class="pre">Schizosaccharomyces</span> <span class="pre">pombe</span></code>, <code class="docutils literal notranslate"><span class="pre">Schizosaccharomyces</span> <span class="pre">pombe</span> <span class="pre">972h-</span></code></p></td> 807</tr> 808<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Staphylococcus</span> <span class="pre">aureus</span></code></p></td> 809<td><p><code class="docutils literal notranslate"><span class="pre">Staphylococcus</span> <span class="pre">aureus</span></code>, <code class="docutils literal notranslate"><span class="pre">Staphylococcus</span> <span class="pre">aureus</span> <span class="pre">subsp.</span> <span class="pre">aureus</span> <span class="pre">rn4220</span></code>, <code class="docutils literal notranslate"><span class="pre">Staphylococcus</span> <span class="pre">aureus</span> <span class="pre">subsp.</span> <span class="pre">aureus</span> <span class="pre">n315</span></code>, <code class="docutils literal notranslate"><span class="pre">Staphylococcus</span> <span class="pre">aureus</span> <span class="pre">subsp.</span> <span class="pre">aureus</span> <span class="pre">usa300</span></code>, <code class="docutils literal notranslate"><span class="pre">Staphylococcus</span> <span class="pre">aureus</span> <span class="pre">subsp.</span> <span class="pre">aureus</span> <span class="pre">mu50</span></code>, <code class="docutils literal notranslate"><span class="pre">Staphylococcus</span> <span class="pre">aureus</span> <span class="pre">subsp.</span> <span class="pre">aureus</span> <span class="pre">str.</span> <span class="pre">newman</span></code></p></td> 810</tr> 811<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Sus</span> <span class="pre">scrofa</span></code></p></td> 812<td><p><code class="docutils literal notranslate"><span class="pre">Sus</span> <span class="pre">scrofa</span> <span class="pre">domesticus</span></code>, <code class="docutils literal notranslate"><span class="pre">Sus</span> <span class="pre">scrofa</span></code></p></td> 813</tr> 814<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">
814Triticum</span> <span class="pre">aestivum</span></code></p></td> 815<td><p><code class="docutils literal notranslate"><span class="pre">Triticum</span> <span class="pre">aestivum</span></code>, <code class="docutils literal notranslate"><span class="pre">Triticum</span> <span class="pre">turgidum</span> <span class="pre">subsp.</span> <span class="pre">dicoccoides</span></code>, <code class="docutils literal notranslate"><span class="pre">Triticum</span> <span class="pre">turgidum</span> <span class="pre">subsp.</span> <span class="pre">durum</span></code>, <code class="docutils literal notranslate"><span class="pre">Triticum</span> <span class="pre">turgidum</span></code>, <code class="docutils literal notranslate"><span class="pre">Triticum</span> <span class="pre">carthlicum</span></code>, <code class="docutils literal notranslate"><span class="pre">Triticum</span> <span class="pre">monococcum</span></code></p></td> 816</tr> 817<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Vitis</span> <span class="pre">hybrid</span> <span class="pre">cultivar</span></code></p></td> 818<td><p><code class="docutils literal notranslate"><span class="pre">Vitis</span> <span class="pre">hybrid</span> <span class="pre">cultivar</span></code></p></td> 819</tr> 820<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Vitis</span> <span class="pre">riparia</span></code></p></td> 821<td><p><code class="docutils literal notranslate"><span class="pre">Vitis</span> <span class="pre">riparia</span></code></p></td> 822</tr> 823<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Vitis</span> <span class="pre">vinifera</span></code></p></td> 824<td><p><code class="docutils literal notranslate"><span class="pre">Vitis</span> <span class="pre">vinifera</span></code>, <code class="docutils literal notranslate"><span class="pre">Vitis</span> <span class="pre">rotundifolia</span></code>, <code class="docutils literal notranslate"><span class="pre">Vitis</span> <span class="pre">hybrid</span> <span class="pre">cultivar</span></code>, <code class="docutils literal notranslate"><span class="pre">Vitis</span> <span class="pre">aestivalis</span></code>, <code class="docutils literal notranslate"><span class="pre">Vitis</span> <span class="pre">riparia</span></code>, <code class="docutils literal notranslate"><span class="pre">Vitis</span> <span class="pre">cinerea</span> <span class="pre">var.</span> <span class="pre">helleri</span> <span class="pre">x</span> <span class="pre">vitis</span> <span class="pre">vinifera</span></code>, <code class="docutils literal notranslate"><span class="pre">Vitis</span> <span class="pre">cinerea</span> <span class="pre">var.</span> <span class="pre">helleri</span> <span class="pre">x</span> <span class="pre">vitis</span> <span class="pre">rupestris</span></code>, <code class="docutils literal notranslate"><span class="pre">Vitis</span> <span class="pre">cinerea</span> <span class="pre">var.</span> <span class="pre">helleri</span> <span class="pre">x</span> <span class="pre">vitis</span> <span class="pre">riparia</span></code></p></td> 825</tr> 826<tr class="row-odd"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Xenopus</span> <span class="pre">laevis</span></code></p></td> 827<td><p><code class="docutils literal notranslate"><span class="pre">Xenopus</span> <span class="pre">muelleri</span></code>, <code class="docutils literal notranslate"><span class="pre">Xenopus</span> <span class="pre">laevis</span> <span class="pre">x</span> <span class="pre">xenopus</span> <span class="pre">muelleri</span></code>, <code class="docutils literal notranslate"><span class="pre">Xenopus</span> <span class="pre">laevis</span></code>, <code class="docutils literal notranslate"><span class="pre">Xenopus</span> <span class="pre">laevis</span> <span class="pre">x</span> <span class="pre">xenopus</span> <span class="pre">borealis</span></code>, <code class="docutils literal notranslate"><span class="pre">Xenopus</span> <span class="pre">borealis</span></code></p></td> 828</tr> 829<tr class="row-even"><td class="text-center"><p><code class="docutils literal notranslate"><span class="pre">Zea</span> <span class="pre">mays</span></code></p></td> 830<td><p><code class="docutils literal notranslate"><span class="pre">Zea</span> <span class="pre">mays</span></code></p></td> 831</tr> 832</tbody> 833</table> 834</section> 835<section id="normalized-compendium-download-folder"> 836<h3>Normalized Compendium Download Folder<a class="headerlink" href="#normalized-compendium-download-folder" title="Permalink to this heading">ï</a></h3> 837<p>Users will receive a zipped folder with a gene expression matrix aggregated by species, along with associated metadata. 838Below is the detailed folder structure:</p> 839<p><img alt="docs-downloads-species-compendia" src="https://user-images.githubusercontent.com/15315514/65180873-ddbb4f80-da2b-11e9-97e9-127c68106182.png" /></p> 840</section> 841</section> 842<section id="rna-seq-sample-compendia"> 843<h2>RNA-Seq Sample Compendia<a class="headerlink" href="#rna-seq-sample-compendia" title="Permalink to this heading">ï</a></h2> 844<p>refine.bio RNA-seq sample compendia are comprised of the Salmon output for the collection of RNA-seq samples from an organism that we have processed with refine.bio. 845Each individual sample has its own <code class="docutils literal notranslate"><span class="pre">quant.sf</span></code> file; the samples have not been aggregated and normalized. 846RNA-seq sample compendia are designed to allow users that are comfortable handling these files to generate output that is most useful for their downstream applications. 847Please see the <a class="reference external" href="https://salmon.readthedocs.io/en/latest/file_formats.html#quantification-file">Salmon documentation on the <code class="docutils literal notranslate"><span class="pre">quant.sf</span></code> output format</a> for more information.</p> 848<section id="rna-seq-sample-compendium-download-folder"> 849<h3>RNA-Seq Sample Compendium Download Folder<a class="headerlink" href="#rna-seq-sample-compendium-download-folder" title="Permalink to this heading">ï</a></h3> 850<p>Users will receive a zipped folder with individual <code class="docutils literal notranslate"><span class="pre">quant.sf</span></code> files for each sample that we were able to process with Salmon, grouped into folders based on the experiment those samples come from, along with any associated metadata in refine.bio. 851Please note that our RNA-seq sample metadata is limited at this time and in some cases, we could not successfully run Salmon on every sample within an experiment (e.g., our processing infrastru
851cture encountered an error with the sample, the sequencing files were malformed). 852In addition, we use the terms âsampleâ and âexperimentâ to be consistent with the rest of refine.bio, but files will use run identifiers (e.g., SRR, ERR, DRR) and project identifiers (e.g., SRP, ERP, DRP), respectively. 853Below is the detailed folder structure:</p> 854<p><img alt="docs-downloads-quantpendia" src="https://user-images.githubusercontent.com/15315514/65271488-4289af00-daeb-11e9-9006-2d7536c4a103.png" /></p> 855</section> 856</section> 857</section> 858<section id="api"> 859<h1>API<a class="headerlink" href="#api" title="Permalink to this heading">ï</a></h1> 860<p>You can use the refine.bio API to build your own applications utilizing the refine.bio processed data. 861You can also select samples for aggregation and download via the API with additional options for download (e.g., without quantile normalization, selecting sample-specific <code class="docutils literal notranslate"><span class="pre">quant.sf</span></code> files). 862Our quantile normalization targets and transcriptome indices are available only via the API (see <a class="reference internal" href="#quantile-normalizing-your-own-data-with-refine-bio-reference-distribution"><span class="xref myst"><em>Quantile normalizing your own data with refine.bio reference distribution</em></span></a> and <a class="reference internal" href="#transcriptome-index"><span class="xref myst"><em>Transcriptome index</em></span></a>, respectively).</p> 863<p><strong>For more information see our API documentation at <a href = "https://api.refine.bio/" target = "blank">api.refine.bio</a>.</strong></p> 864</section> 865<section id="downstream-analysis-with-refine-bio-examples"> 866<h1>Downstream Analysis with refine.bio Examples<a class="headerlink" href="#downstream-analysis-with-refine-bio-examples" title="Permalink to this heading">ï</a></h1> 867<p>Our <a href = "https://alexslemonade.github.io/refinebio-examples/01-getting-started/getting-started.html" target = "blank">refine.bio examples site</a> includes a number of different analyses you can perform with data from refine.bio in the R programming language. 868You can view our examples in your browser or download the <a href = "https://rmarkdown.rstudio.com/" target = "blank">R Markdown (<code class="docutils literal notranslate"><span class="pre">.Rmd</span></code>)</a> files to run the code locally (find more information on required software <a href = "https://alexslemonade.github.io/refinebio-examples/01-getting-started/getting-started.html#03_What_you_need_to_install_to_run_the_examples" target = "blank">here</a> and how to use <code class="docutils literal notranslate"><span class="pre">.Rmd</span></code> <a href = "https://alexslemonade.github.io/refinebio-examples/01-getting-started/getting-started.html#05_How_to_use_R_Markdown_Documents" target = "blank">here</a>). 869Example analyses are designed to guide you through obtaining data from the refine.bio web interface and provide instruction for adapting the analysis for a dataset of your choice. 870View our <a href = "https://alexslemonade.github.io/refinebio-examples/01-getting-started/getting-started.html" target = "blank">Getting Started page</a>, <a href = "https://alexslemonade.github.io/refinebio-examples/02-microarray/00-intro-to-microarray.html" target = "blank">Introduction to Microarray</a>, or <a href = "https://alexslemonade.github.io/refinebio-examples/03-rnaseq/00-intro-to-rnaseq.html" target = "blank">Introduction to RNA-seq</a> to start using refine.bio examples.</p> 871<p>Hereâs a sneak peek of examples that are available:</p> 872<ul class="simple"> 873<li><p><a href = "https://alexslemonade.github.io/refinebio-examples/02-microarray/differential-expression_microarray_02_several-groups.html" target = "blank">Differential gene expression analysis for several groups (Microarray)</a></p></li> 874<li><p><a href = "https://alexslemonade.github.io/refinebio-examples/03-rnaseq/differential-expression_rnaseq_01.html#analysis" target = "blank">Differential gene expression analysis for 2 groups (RNA-seq)</a></p></li> 875<li><p><a href = "https://alexslemonade.github.io/refinebio-examples/03-rnaseq/clustering_rnaseq_01_heatmap.html" target = "blank">Hierarchical clustering and heatmap creation (RNA-seq)</a></p></li> 876<li><p><a href = "https://alexslemonade.github.io/refinebio-examples/02-microarray/pathway-analysis_microarray_02_gsea.html" target = "blank">Gene Set Enrichment Analysis (Microarray)</a></p></li> 877<li><p><a href = "https://alexslemonade.github.io/refinebio-examples/03-rnaseq/dimension-reduction_rnaseq_02_umap.html" target = "blank">Visualization with Uniform Manifold Approximation and Projection (RNA-seq)</a></p></li> 878</ul> 879<p><a href = "https://share.hsforms.com/1pQ8pPB70TuG37sVjr-KtXA336z0" target = "blank">Give us feedback on our refine.bio examples!</a> 880If you have a question about or find a problem with one of our examples and need to get in touch with one of our staff, please <a href = "https://github.com/AlexsLemonade/refinebio-examples/issues/new?assignees=&labels=user+report+or+question&template=report-a-problem-or-ask-a-question.md&title=User+issue%3A++" target = "blank">file a GitHub issue</a>. 881If you would prefer to report issues via e-mail, you can also email <a class="reference external" href="mailto:requests%40ccdatalab.org">requests<span>@</span>
881ccdatalab<span>.</span>org</a>.</p> 882</section> 883 884 885 </div> 886 </div> 887 <footer><div class="rst-footer-buttons" role="navigation" aria-label="Footer"> 888 <a href="index.html" class="btn btn-neutral float-left" title="refine.bio Documentation" accesskey="p" rel="prev"><span class="fa fa-arrow-circle-left" aria-hidden="true"></span> Previous</a> 889 <a href="getting_started.html" class="btn btn-neutral float-right" title="Getting Started with a refine.bio dataset" accesskey="n" rel="next">Next <span class="fa fa-arrow-circle-right" aria-hidden="true"></span></a> 890 </div> 891 892 <hr/> 893 894 <div role="contentinfo"> 895 <p>© Copyright 2023, The Childhood Cancer Data Lab. 896 <span class="commit">Revision <code>babbd0cf</code>. 897 </span></p> 898 </div> 899 900 Built with <a href="https://www.sphinx-doc.org/">Sphinx</a> using a 901 <a href="https://github.com/readthedocs/sphinx_rtd_theme">theme</a> 902 provided by <a href="https://readthedocs.org">Read the Docs</a>. 903 904 905</footer> 906 </div> 907 </div> 908 </section> 909 </div> 910 911 912
913<script> 914 jQuery(function () { 915 SphinxRtdTheme.Navigation.enable(true); 916 }); 917 </script>
917 918 919
919<script type="text/javascript" src="_static/js/sticky-sidebar.min.js"></script>
919 920
920<script type="text/javascript"> 921 (function() { 922 var stickySidebar = new StickySidebar('.wy-side-scroll'); 923 document.addEventListener('click', function(e) { 924 if (!event.target.matches('.wy-side-scroll a span')) return; 925 stickySidebar.updateSticky(); 926 }, true) 927 })(); 928 </script>
928 929 930
930<script type="text/javascript"> 931 (function(){ 932 const versionElements = document.getElementsByClassName('version'); 933 if (versionElements.length === 0) return false; 934 fetch('https://api.refine.bio/v1/transcriptome_indices/?limit=1') 935 .then(function(response) { 936 const version = response.headers.get('x-source-revision'); 937 console.log(versionElements) 938 if (versionElements && version) { 939 versionElements[0].textContent = version; 940 } 941 }) 942 })() 943 </script>
943 944 945 946</body> 947</html>
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.