Benjamin-Lee
diff --git a/‎citations.tsv
Lines changed: 1 addition & 0 deletions b/‎citations.tsv
Lines changed: 1 addition & 0 deletions
diff --git a/‎manuscript.html
Lines changed: 81 additions & 74 deletions b/‎manuscript.html
Lines changed: 81 additions & 74 deletions
diff --git a/‎manuscript.md
Lines changed: 13 additions & 12 deletions b/‎manuscript.md
Lines changed: 13 additions & 12 deletions
diff --git a/‎manuscript.pdf
14.4 KB b/‎manuscript.pdf
14.4 KB
diff --git a/‎references.json
Lines changed: 101 additions & 33 deletions b/‎references.json
Lines changed: 101 additions & 33 deletions
@@ -78,6 +78,7 @@ doi:10.1109/TBDATA.2016.2573280	doi:10.1109/TBDATA.2016.2573280	doi:10.1109/tbda
 doi:10/bjjdg2	doi:10/bjjdg2	doi:10.1016/s0893-6080(05)80131-5	AE3ehMCc
 tag:srivastava-dropout	http://dl.acm.org/citation.cfm?id=2670313	url:http://dl.acm.org/citation.cfm?id=2670313	wgOFUxdw
 tag:ioffe-batchnorm	https://dl.acm.org/citation.cfm?id=3045118.3045167	url:https://dl.acm.org/citation.cfm?id=3045118.3045167	4oKcgKmU
+doi:10.1073/pnas.1903070116	doi:10.1073/pnas.1903070116	doi:10.1073/pnas.1903070116	qCKLXDUQ
 arxiv:1811.12808	arxiv:1811.12808	arxiv:1811.12808	1CDx6NYSj
 doi:10.1162/089976698300017197	doi:10.1162/089976698300017197	doi:10.1162/089976698300017197	hJQdIoO3
 url:http://jmlr.csail.mit.edu/papers/v15/srivastava14a.html	url:http://jmlr.csail.mit.edu/papers/v15/srivastava14a.html	url:http://jmlr.csail.mit.edu/papers/v15/srivastava14a.html	R1RpVu06
 
@@ -22,7 +22,7 @@ author-meta:
 - Juan Jose Carmona
 bibliography:
 - content/manual-references.json
-date-meta: '2021-01-21'
+date-meta: '2021-01-23'
 header-includes: '<!--
 
   Manubot generated metadata rendered from header-includes-template.html.
@@ -41,9 +41,9 @@ header-includes: '<!--
 
   <meta property="twitter:title" content="Ten Quick Tips for Deep Learning in Biology" />
 
-  <meta name="dc.date" content="2021-01-21" />
+  <meta name="dc.date" content="2021-01-23" />
 
-  <meta name="citation_publication_date" content="2021-01-21" />
+  <meta name="citation_publication_date" content="2021-01-23" />
 
   <meta name="dc.language" content="en-US" />
 
@@ -217,19 +217,19 @@ header-includes: '<!--
 
   <link rel="alternate" type="application/pdf" href="https://Benjamin-Lee.github.io/deep-rules/manuscript.pdf" />
 
-  <link rel="alternate" type="text/html" href="https://Benjamin-Lee.github.io/deep-rules/v/cdf8ae16f5b10a5ef134a5f56af5fde5c173d11b/" />
+  <link rel="alternate" type="text/html" href="https://Benjamin-Lee.github.io/deep-rules/v/cc09f2b48953fe4b7a52ad5ff955bfb0d25e62bd/" />
 
-  <meta name="manubot_html_url_versioned" content="https://Benjamin-Lee.github.io/deep-rules/v/cdf8ae16f5b10a5ef134a5f56af5fde5c173d11b/" />
+  <meta name="manubot_html_url_versioned" content="https://Benjamin-Lee.github.io/deep-rules/v/cc09f2b48953fe4b7a52ad5ff955bfb0d25e62bd/" />
 
-  <meta name="manubot_pdf_url_versioned" content="https://Benjamin-Lee.github.io/deep-rules/v/cdf8ae16f5b10a5ef134a5f56af5fde5c173d11b/manuscript.pdf" />
+  <meta name="manubot_pdf_url_versioned" content="https://Benjamin-Lee.github.io/deep-rules/v/cc09f2b48953fe4b7a52ad5ff955bfb0d25e62bd/manuscript.pdf" />
 
   <meta property="og:type" content="article" />
 
   <meta property="twitter:card" content="summary_large_image" />
 
-  <meta property="og:image" content="https://github.com/Benjamin-Lee/deep-rules/raw/cdf8ae16f5b10a5ef134a5f56af5fde5c173d11b/content/images/thumbnail_tips_overview.png" />
+  <meta property="og:image" content="https://github.com/Benjamin-Lee/deep-rules/raw/cc09f2b48953fe4b7a52ad5ff955bfb0d25e62bd/content/images/thumbnail_tips_overview.png" />
 
-  <meta property="twitter:image" content="https://github.com/Benjamin-Lee/deep-rules/raw/cdf8ae16f5b10a5ef134a5f56af5fde5c173d11b/content/images/thumbnail_tips_overview.png" />
+  <meta property="twitter:image" content="https://github.com/Benjamin-Lee/deep-rules/raw/cc09f2b48953fe4b7a52ad5ff955bfb0d25e62bd/content/images/thumbnail_tips_overview.png" />
 
   <link rel="icon" type="image/png" sizes="192x192" href="https://manubot.org/favicon-192x192.png" />
 
@@ -258,10 +258,10 @@ title: Ten Quick Tips for Deep Learning in Biology
 
 <small><em>
 This manuscript
-([permalink](https://Benjamin-Lee.github.io/deep-rules/v/cdf8ae16f5b10a5ef134a5f56af5fde5c173d11b/))
+([permalink](https://Benjamin-Lee.github.io/deep-rules/v/cc09f2b48953fe4b7a52ad5ff955bfb0d25e62bd/))
 was automatically generated
-from [Benjamin-Lee/deep-rules@cdf8ae1](https://github.com/Benjamin-Lee/deep-rules/tree/cdf8ae16f5b10a5ef134a5f56af5fde5c173d11b)
-on January 21, 2021.
+from [Benjamin-Lee/deep-rules@cc09f2b](https://github.com/Benjamin-Lee/deep-rules/tree/cc09f2b48953fe4b7a52ad5ff955bfb0d25e62bd)
+on January 23, 2021.
 </em></small>
 
 ## Authors
@@ -695,8 +695,9 @@ In other words, the model fits patterns that are overly specific to the data it
 This subtle distinction is made clearer by seeing what happens when a model is tested on data to which it was not exposed during training: just as a student who memorizes exam materials struggles to correctly answer questions for which they have not studied, a machine learning model that has overfit to its training data will perform poorly on unseen test data.
 Deep learning models are particularly susceptible to overfitting due to their relatively large number of parameters and associated representational capacity.
 Just as some students may have greater potential for memorization, deep learning models seem more prone to overfitting than machine learning models with fewer parameters.
+However, having a large number of parameters does not always imply that a neural network will overfit [@doi:10.1073/pnas.1903070116].
 
-![A visual example of overfitting and failure to generalize. While a high-degree polynomial achieves high accuracy on its training data, it performs poorly on data with specificities that have not been seen before. That is, the model has learned the training dataset specifically rather than learning a generalizable pattern that represents data of this type. In contrast, a simple linear regression works well on both datasets. The greater representational capacity of the polynomial is analogous to using a larger or deeper neural network.](images/overfitting.png){#fig:overfitting-fig}
+![A visual example of overfitting and failure to generalize. While a high-degree polynomial achieves high accuracy on its training data, it performs poorly on the test data that have not been seen before. That is, the model has memorized the training dataset specifically rather than learning a generalizable pattern that represents data of this type. In contrast, a simple linear regression works equally well on both datasets.](images/overfitting.png){#fig:overfitting-fig}
 
 In general, one of the most effective ways to combat overfitting is to detect it in the first place.
 One way to do this is to split the main dataset being worked on into three independent parts: a training set, a tuning set (also commonly called a validation set in the machine learning literature), and a test set.
 
@@ -2920,6 +2920,49 @@
     "URL": "https://doi.org/ghfwxq",
     "note": "This CSL JSON Item was automatically generated by Manubot v0.4.1 using citation-by-identifier.\nstandard_id: doi:10.1038/s42256-020-0218-x"
   },
+  {
+    "type": "article-journal",
+    "id": "qCKLXDUQ",
+    "author": [
+      {
+        "family": "Belkin",
+        "given": "Mikhail"
+      },
+      {
+        "family": "Hsu",
+        "given": "Daniel"
+      },
+      {
+        "family": "Ma",
+        "given": "Siyuan"
+      },
+      {
+        "family": "Mandal",
+        "given": "Soumik"
+      }
+    ],
+    "issued": {
+      "date-parts": [
+        [
+          2019,
+          8,
+          6
+        ]
+      ]
+    },
+    "abstract": "Breakthroughs in machine learning are rapidly changing science and society, yet our fundamental understanding of this technology has lagged far behind. Indeed, one of the central tenets of the field, the bias–variance trade-off, appears to be at odds with the observed behavior of methods used in modern machine-learning practice. The bias–variance trade-off implies that a model should balance underfitting and overfitting: Rich enough to express underlying structure in data and simple enough to avoid fitting spurious patterns. However, in modern practice, very rich models such as neural networks are trained to exactly fit (i.e., interpolate) the data. Classically, such models would be considered overfitted, and yet they often obtain high accuracy on test data. This apparent contradiction has raised questions about the mathematical foundations of machine learning and their relevance to practitioners. In this paper, we reconcile the classical understanding and the modern practice within a unified performance curve. This “double-descent” curve subsumes the textbook U-shaped bias–variance trade-off curve by showing how increasing model capacity beyond the point of interpolation results in improved performance. We provide evidence for the existence and ubiquity of double descent for a wide spectrum of models and datasets, and we posit a mechanism for its emergence. This connection between the performance and the structure of machine-learning models delineates the limits of classical analyses and has implications for both the theory and the practice of machine learning.",
+    "container-title": "Proceedings of the National Academy of Sciences",
+    "DOI": "10.1073/pnas.1903070116",
+    "volume": "116",
+    "issue": "32",
+    "page": "15849-15854",
+    "publisher": "Proceedings of the National Academy of Sciences",
+    "title": "Reconciling modern machine-learning practice and the classical bias–variance trade-off",
+    "URL": "https://doi.org/gf5dmw",
+    "PMCID": "PMC6689936",
+    "PMID": "31341078",
+    "note": "This CSL JSON Item was automatically generated by Manubot v0.4.1 using citation-by-identifier.\nstandard_id: doi:10.1073/pnas.1903070116"
+  },
   {
     "type": "article-journal",
     "id": "1AyQuG5x7",
@@ -2962,8 +3005,20 @@
     "note": "This CSL JSON Item was automatically generated by Manubot v0.4.1 using citation-by-identifier.\nstandard_id: doi:10.1089/omi.2018.0097"
   },
   {
-    "type": "article-journal",
     "id": "QobI7Hyv",
+    "type": "article-journal",
+    "title": "Correct machine learning on protein sequences: a peer-reviewing perspective",
+    "container-title": "Briefings in Bioinformatics",
+    "page": "831-840",
+    "volume": "17",
+    "issue": "5",
+    "source": "DOI.org (Crossref)",
+    "URL": "https://doi.org/f89ms7",
+    "DOI": "10.1093/bib/bbv082",
+    "ISSN": "1467-5463, 1477-4054",
+    "shortTitle": "Correct machine learning on protein sequences",
+    "journalAbbreviation": "Brief Bioinform",
+    "language": "en",
     "author": [
       {
         "family": "Walsh",
@@ -2981,25 +3036,37 @@
     "issued": {
       "date-parts": [
         [
-          2016,
+          "2016",
           9
         ]
       ]
     },
-    "container-title": "Briefings in Bioinformatics",
-    "DOI": "10.1093/bib/bbv082",
-    "volume": "17",
-    "issue": "5",
-    "page": "831-840",
-    "publisher": "Oxford University Press (OUP)",
-    "title": "Correct machine learning on protein sequences: a peer-reviewing perspective",
-    "URL": "https://doi.org/f89ms7",
+    "accessed": {
+      "date-parts": [
+        [
+          "2021",
+          1,
+          23
+        ]
+      ]
+    },
     "PMID": "26411473",
     "note": "This CSL JSON Item was automatically generated by Manubot v0.4.1 using citation-by-identifier.\nstandard_id: doi:10.1093/bib/bbv082"
   },
   {
-    "type": "article-journal",
     "id": "1GGrbeMvT",
+    "type": "article-journal",
+    "title": "Correcting for experiment-specific variability in expression compendia can remove underlying signals",
+    "container-title": "GigaScience",
+    "page": "giaa117",
+    "volume": "9",
+    "issue": "11",
+    "source": "DOI.org (Crossref)",
+    "abstract": "Abstract\r\n            \r\n              Motivation\r\n              In the past two decades, scientists in different laboratories have assayed gene expression from millions of samples. These experiments can be combined into compendia and analyzed collectively to extract novel biological patterns. Technical variability, or \"batch effects,\" may result from combining samples collected and processed at different times and in different settings. Such variability may distort our ability to extract true underlying biological patterns. As more integrative analysis methods arise and data collections get bigger, we must determine how technical variability affects our ability to detect desired patterns when many experiments are combined.\r\n            \r\n            \r\n              Objective\r\n              We sought to determine the extent to which an underlying signal was masked by technical variability by simulating compendia comprising data aggregated across multiple experiments.\r\n            \r\n            \r\n              Method\r\n              We developed a generative multi-layer neural network to simulate compendia of gene expression experiments from large-scale microbial and human datasets. We compared simulated compendia before and after introducing varying numbers of sources of undesired variability.\r\n            \r\n            \r\n              Results\r\n              The signal from a baseline compendium was obscured when the number of added sources of variability was small. Applying statistical correction methods rescued the underlying signal in these cases. However, as the number of sources of variability increased, it became easier to detect the original signal even without correction. In fact, statistical correction reduced our power to detect the underlying signal.\r\n            \r\n            \r\n              Conclusion\r\n              When combining a modest number of experiments, it is best to correct for experiment-specific noise. However, when many experiments are combined, statistical correction reduces our ability to extract underlying patterns.",
+    "URL": "https://doi.org/ghhtpf",
+    "DOI": "10.1093/gigascience/giaa117",
+    "ISSN": "2047-217X",
+    "language": "en",
     "author": [
       {
         "family": "Lee",
@@ -3025,20 +3092,21 @@
     "issued": {
       "date-parts": [
         [
-          2020,
+          "2020",
           11,
           3
         ]
       ]
     },
-    "container-title": "GigaScience",
-    "DOI": "10.1093/gigascience/giaa117",
-    "volume": "9",
-    "issue": "11",
-    "page": "giaa117",
-    "publisher": "Oxford University Press (OUP)",
-    "title": "Correcting for experiment-specific variability in expression compendia can remove underlying signals",
-    "URL": "https://doi.org/ghhtpf",
+    "accessed": {
+      "date-parts": [
+        [
+          "2021",
+          1,
+          23
+        ]
+      ]
+    },
     "PMCID": "PMC7607552",
     "PMID": "33140829",
     "note": "This CSL JSON Item was automatically generated by Manubot v0.4.1 using citation-by-identifier.\nstandard_id: doi:10.1093/gigascience/giaa117"
@@ -5066,7 +5134,7 @@
         [
           "2021",
           1,
-          19
+          22
         ]
       ]
     },
@@ -5119,7 +5187,7 @@
         [
           "2021",
           1,
-          20
+          23
         ]
       ]
     },
@@ -5136,7 +5204,7 @@
         [
           "2021",
           1,
-          20
+          23
         ]
       ]
     },
@@ -5160,7 +5228,7 @@
         [
           "2021",
           1,
-          20
+          23
         ]
       ]
     },
@@ -5203,7 +5271,7 @@
         [
           "2021",
           1,
-          19
+          22
         ]
       ]
     },
@@ -5254,7 +5322,7 @@
         [
           "2021",
           1,
-          19
+          22
         ]
       ]
     },
@@ -5318,7 +5386,7 @@
         [
           "2021",
           1,
-          20
+          23
         ]
       ]
     },
@@ -5327,7 +5395,7 @@
         [
           "2021",
           1,
-          20
+          23
         ]
       ]
     }
@@ -5343,7 +5411,7 @@
         [
           "2021",
           1,
-          20
+          23
         ]
       ]
     },
@@ -5403,7 +5471,7 @@
         [
           "2021",
           1,
-          20
+          23
         ]
       ]
     },
@@ -5428,7 +5496,7 @@
         [
           "2021",
           1,
-          20
+          23
         ]
       ]
     },
@@ -5445,7 +5513,7 @@
         [
           "2021",
           1,
-          20
+          23
         ]
       ]
     },
@@ -5541,7 +5609,7 @@
         [
           "2021",
           1,
-          20
+          23
         ]
       ]
     },
Original file line number	Diff line number	Diff line change
`@@ -2920,6 +2920,49 @@`
`2920`	`2920`	`"URL": "https://doi.org/ghfwxq",`
`2921`	`2921`	`"note": "This CSL JSON Item was automatically generated by Manubot v0.4.1 using citation-by-identifier.\nstandard_id: doi:10.1038/s42256-020-0218-x"`
`2922`	`2922`	`},`
	`2923`	`+ {`
	`2924`	`+ "type": "article-journal",`
	`2925`	`+ "id": "qCKLXDUQ",`
	`2926`	`+ "author": [`
	`2927`	`+ {`
	`2928`	`+ "family": "Belkin",`
	`2929`	`+ "given": "Mikhail"`
	`2930`	`+ },`
	`2931`	`+ {`
	`2932`	`+ "family": "Hsu",`
	`2933`	`+ "given": "Daniel"`
	`2934`	`+ },`
	`2935`	`+ {`
	`2936`	`+ "family": "Ma",`
	`2937`	`+ "given": "Siyuan"`
	`2938`	`+ },`
	`2939`	`+ {`
	`2940`	`+ "family": "Mandal",`
	`2941`	`+ "given": "Soumik"`
	`2942`	`+ }`
	`2943`	`+ ],`
	`2944`	`+ "issued": {`
	`2945`	`+ "date-parts": [`
	`2946`	`+ [`
	`2947`	`+ 2019,`
	`2948`	`+ 8,`
	`2949`	`+ 6`
	`2950`	`+ ]`
	`2951`	`+ ]`
	`2952`	`+ },`
	`2953`	+ "abstract": "Breakthroughs in machine learning are rapidly changing science and society, yet our fundamental understanding of this technology has lagged far behind. Indeed, one of the central tenets of the field, the bias–variance trade-off, appears to be at odds with the observed behavior of methods used in modern machine-learning practice. The bias–variance trade-off implies that a model should balance underfitting and overfitting: Rich enough to express underlying structure in data and simple enough to avoid fitting spurious patterns. However, in modern practice, very rich models such as neural networks are trained to exactly fit (i.e., interpolate) the data. Classically, such models would be considered overfitted, and yet they often obtain high accuracy on test data. This apparent contradiction has raised questions about the mathematical foundations of machine learning and their relevance to practitioners. In this paper, we reconcile the classical understanding and the modern practice within a unified performance curve. This “double-descent” curve subsumes the textbook U-shaped bias–variance trade-off curve by showing how increasing model capacity beyond the point of interpolation results in improved performance. We provide evidence for the existence and ubiquity of double descent for a wide spectrum of models and datasets, and we posit a mechanism for its emergence. This connection between the performance and the structure of machine-learning models delineates the limits of classical analyses and has implications for both the theory and the practice of machine learning.",
	`2954`	`+ "container-title": "Proceedings of the National Academy of Sciences",`
	`2955`	`+ "DOI": "10.1073/pnas.1903070116",`
	`2956`	`+ "volume": "116",`
	`2957`	`+ "issue": "32",`
	`2958`	`+ "page": "15849-15854",`
	`2959`	`+ "publisher": "Proceedings of the National Academy of Sciences",`
	`2960`	`+ "title": "Reconciling modern machine-learning practice and the classical bias–variance trade-off",`
	`2961`	`+ "URL": "https://doi.org/gf5dmw",`
	`2962`	`+ "PMCID": "PMC6689936",`
	`2963`	`+ "PMID": "31341078",`
	`2964`	`+ "note": "This CSL JSON Item was automatically generated by Manubot v0.4.1 using citation-by-identifier.\nstandard_id: doi:10.1073/pnas.1903070116"`
	`2965`	`+ },`
`2923`	`2966`	`{`
`2924`	`2967`	`"type": "article-journal",`
`2925`	`2968`	`"id": "1AyQuG5x7",`
`@@ -2962,8 +3005,20 @@`
`2962`	`3005`	`"note": "This CSL JSON Item was automatically generated by Manubot v0.4.1 using citation-by-identifier.\nstandard_id: doi:10.1089/omi.2018.0097"`
`2963`	`3006`	`},`
`2964`	`3007`	`{`
`2965`		`- "type": "article-journal",`
`2966`	`3008`	`"id": "QobI7Hyv",`
	`3009`	`+ "type": "article-journal",`
	`3010`	`+ "title": "Correct machine learning on protein sequences: a peer-reviewing perspective",`
	`3011`	`+ "container-title": "Briefings in Bioinformatics",`
	`3012`	`+ "page": "831-840",`
	`3013`	`+ "volume": "17",`
	`3014`	`+ "issue": "5",`
	`3015`	`+ "source": "DOI.org (Crossref)",`
	`3016`	`+ "URL": "https://doi.org/f89ms7",`
	`3017`	`+ "DOI": "10.1093/bib/bbv082",`
	`3018`	`+ "ISSN": "1467-5463, 1477-4054",`
	`3019`	`+ "shortTitle": "Correct machine learning on protein sequences",`
	`3020`	`+ "journalAbbreviation": "Brief Bioinform",`
	`3021`	`+ "language": "en",`
`2967`	`3022`	`"author": [`
`2968`	`3023`	`{`
`2969`	`3024`	`"family": "Walsh",`
`@@ -2981,25 +3036,37 @@`
`2981`	`3036`	`"issued": {`
`2982`	`3037`	`"date-parts": [`
`2983`	`3038`	`[`
`2984`		`- 2016,`
	`3039`	`+ "2016",`
`2985`	`3040`	`9`
`2986`	`3041`	`]`
`2987`	`3042`	`]`
`2988`	`3043`	`},`
`2989`		`- "container-title": "Briefings in Bioinformatics",`
`2990`		`- "DOI": "10.1093/bib/bbv082",`
`2991`		`- "volume": "17",`
`2992`		`- "issue": "5",`
`2993`		`- "page": "831-840",`
`2994`		`- "publisher": "Oxford University Press (OUP)",`
`2995`		`- "title": "Correct machine learning on protein sequences: a peer-reviewing perspective",`
`2996`		`- "URL": "https://doi.org/f89ms7",`
	`3044`	`+ "accessed": {`
	`3045`	`+ "date-parts": [`
	`3046`	`+ [`
	`3047`	`+ "2021",`
	`3048`	`+ 1,`
	`3049`	`+ 23`
	`3050`	`+ ]`
	`3051`	`+ ]`
	`3052`	`+ },`
`2997`	`3053`	`"PMID": "26411473",`
`2998`	`3054`	`"note": "This CSL JSON Item was automatically generated by Manubot v0.4.1 using citation-by-identifier.\nstandard_id: doi:10.1093/bib/bbv082"`
`2999`	`3055`	`},`
`3000`	`3056`	`{`
`3001`		`- "type": "article-journal",`
`3002`	`3057`	`"id": "1GGrbeMvT",`
	`3058`	`+ "type": "article-journal",`
	`3059`	`+ "title": "Correcting for experiment-specific variability in expression compendia can remove underlying signals",`
	`3060`	`+ "container-title": "GigaScience",`
	`3061`	`+ "page": "giaa117",`
	`3062`	`+ "volume": "9",`
	`3063`	`+ "issue": "11",`
	`3064`	`+ "source": "DOI.org (Crossref)",`
	`3065`	+ "abstract": "Abstract\r\n \r\n Motivation\r\n In the past two decades, scientists in different laboratories have assayed gene expression from millions of samples. These experiments can be combined into compendia and analyzed collectively to extract novel biological patterns. Technical variability, or \"batch effects,\" may result from combining samples collected and processed at different times and in different settings. Such variability may distort our ability to extract true underlying biological patterns. As more integrative analysis methods arise and data collections get bigger, we must determine how technical variability affects our ability to detect desired patterns when many experiments are combined.\r\n \r\n \r\n Objective\r\n We sought to determine the extent to which an underlying signal was masked by technical variability by simulating compendia comprising data aggregated across multiple experiments.\r\n \r\n \r\n Method\r\n We developed a generative multi-layer neural network to simulate compendia of gene expression experiments from large-scale microbial and human datasets. We compared simulated compendia before and after introducing varying numbers of sources of undesired variability.\r\n \r\n \r\n Results\r\n The signal from a baseline compendium was obscured when the number of added sources of variability was small. Applying statistical correction methods rescued the underlying signal in these cases. However, as the number of sources of variability increased, it became easier to detect the original signal even without correction. In fact, statistical correction reduced our power to detect the underlying signal.\r\n \r\n \r\n Conclusion\r\n When combining a modest number of experiments, it is best to correct for experiment-specific noise. However, when many experiments are combined, statistical correction reduces our ability to extract underlying patterns.",
	`3066`	`+ "URL": "https://doi.org/ghhtpf",`
	`3067`	`+ "DOI": "10.1093/gigascience/giaa117",`
	`3068`	`+ "ISSN": "2047-217X",`
	`3069`	`+ "language": "en",`
`3003`	`3070`	`"author": [`
`3004`	`3071`	`{`
`3005`	`3072`	`"family": "Lee",`
`@@ -3025,20 +3092,21 @@`
`3025`	`3092`	`"issued": {`
`3026`	`3093`	`"date-parts": [`
`3027`	`3094`	`[`
`3028`		`- 2020,`
	`3095`	`+ "2020",`
`3029`	`3096`	`11,`
`3030`	`3097`	`3`
`3031`	`3098`	`]`
`3032`	`3099`	`]`
`3033`	`3100`	`},`
`3034`		`- "container-title": "GigaScience",`
`3035`		`- "DOI": "10.1093/gigascience/giaa117",`
`3036`		`- "volume": "9",`
`3037`		`- "issue": "11",`
`3038`		`- "page": "giaa117",`
`3039`		`- "publisher": "Oxford University Press (OUP)",`
`3040`		`- "title": "Correcting for experiment-specific variability in expression compendia can remove underlying signals",`
`3041`		`- "URL": "https://doi.org/ghhtpf",`
	`3101`	`+ "accessed": {`
	`3102`	`+ "date-parts": [`
	`3103`	`+ [`
	`3104`	`+ "2021",`
	`3105`	`+ 1,`
	`3106`	`+ 23`
	`3107`	`+ ]`
	`3108`	`+ ]`
	`3109`	`+ },`
`3042`	`3110`	`"PMCID": "PMC7607552",`
`3043`	`3111`	`"PMID": "33140829",`
`3044`	`3112`	`"note": "This CSL JSON Item was automatically generated by Manubot v0.4.1 using citation-by-identifier.\nstandard_id: doi:10.1093/gigascience/giaa117"`
`@@ -5066,7 +5134,7 @@`
`5066`	`5134`	`[`
`5067`	`5135`	`"2021",`
`5068`	`5136`	`1,`
`5069`		`- 19`
	`5137`	`+ 22`
`5070`	`5138`	`]`
`5071`	`5139`	`]`
`5072`	`5140`	`},`
`@@ -5119,7 +5187,7 @@`
`5119`	`5187`	`[`
`5120`	`5188`	`"2021",`
`5121`	`5189`	`1,`
`5122`		`- 20`
	`5190`	`+ 23`
`5123`	`5191`	`]`
`5124`	`5192`	`]`
`5125`	`5193`	`},`
`@@ -5136,7 +5204,7 @@`
`5136`	`5204`	`[`
`5137`	`5205`	`"2021",`
`5138`	`5206`	`1,`
`5139`		`- 20`
	`5207`	`+ 23`
`5140`	`5208`	`]`
`5141`	`5209`	`]`
`5142`	`5210`	`},`
`@@ -5160,7 +5228,7 @@`
`5160`	`5228`	`[`
`5161`	`5229`	`"2021",`
`5162`	`5230`	`1,`
`5163`		`- 20`
	`5231`	`+ 23`
`5164`	`5232`	`]`
`5165`	`5233`	`]`
`5166`	`5234`	`},`
`@@ -5203,7 +5271,7 @@`
`5203`	`5271`	`[`
`5204`	`5272`	`"2021",`
`5205`	`5273`	`1,`
`5206`		`- 19`
	`5274`	`+ 22`
`5207`	`5275`	`]`
`5208`	`5276`	`]`
`5209`	`5277`	`},`
`@@ -5254,7 +5322,7 @@`
`5254`	`5322`	`[`
`5255`	`5323`	`"2021",`
`5256`	`5324`	`1,`
`5257`		`- 19`
	`5325`	`+ 22`
`5258`	`5326`	`]`
`5259`	`5327`	`]`
`5260`	`5328`	`},`
`@@ -5318,7 +5386,7 @@`
`5318`	`5386`	`[`
`5319`	`5387`	`"2021",`
`5320`	`5388`	`1,`
`5321`		`- 20`
	`5389`	`+ 23`
`5322`	`5390`	`]`
`5323`	`5391`	`]`
`5324`	`5392`	`},`
`@@ -5327,7 +5395,7 @@`
`5327`	`5395`	`[`
`5328`	`5396`	`"2021",`
`5329`	`5397`	`1,`
`5330`		`- 20`
	`5398`	`+ 23`
`5331`	`5399`	`]`
`5332`	`5400`	`]`
`5333`	`5401`	`}`
`@@ -5343,7 +5411,7 @@`
`5343`	`5411`	`[`
`5344`	`5412`	`"2021",`
`5345`	`5413`	`1,`
`5346`		`- 20`
	`5414`	`+ 23`
`5347`	`5415`	`]`
`5348`	`5416`	`]`
`5349`	`5417`	`},`
`@@ -5403,7 +5471,7 @@`
`5403`	`5471`	`[`
`5404`	`5472`	`"2021",`
`5405`	`5473`	`1,`
`5406`		`- 20`
	`5474`	`+ 23`
`5407`	`5475`	`]`
`5408`	`5476`	`]`
`5409`	`5477`	`},`
`@@ -5428,7 +5496,7 @@`
`5428`	`5496`	`[`
`5429`	`5497`	`"2021",`
`5430`	`5498`	`1,`
`5431`		`- 20`
	`5499`	`+ 23`
`5432`	`5500`	`]`
`5433`	`5501`	`]`
`5434`	`5502`	`},`
`@@ -5445,7 +5513,7 @@`
`5445`	`5513`	`[`
`5446`	`5514`	`"2021",`
`5447`	`5515`	`1,`
`5448`		`- 20`
	`5516`	`+ 23`
`5449`	`5517`	`]`
`5450`	`5518`	`]`
`5451`	`5519`	`},`
`@@ -5541,7 +5609,7 @@`
`5541`	`5609`	`[`
`5542`	`5610`	`"2021",`
`5543`	`5611`	`1,`
`5544`		`- 20`
	`5612`	`+ 23`
`5545`	`5613`	`]`
`5546`	`5614`	`]`
`5547`	`5615`	`},`