<?xml version="1.0" encoding="UTF-8"?><article xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="pmc-domain-id">4382</journal-id><journal-id journal-id-type="pmc-domain">cgen</journal-id><journal-title-group><journal-title>Cell Genomics</journal-title><abbrev-journal-title>Cell Genom</abbrev-journal-title></journal-title-group><publisher><publisher-name>Elsevier</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="pmcid">PMC9105345</article-id><article-id pub-id-type="pmcaid">9105345</article-id><article-id pub-id-type="pmcaiid">9903758</article-id><article-id pub-id-type="pmid">35573091</article-id><article-id pub-id-type="doi">10.1016/j.xgen.2022.100111</article-id><article-id pub-id-type="nihmsid">NIHMS1798679</article-id><title-group><article-title>High-throughput characterization of the role of non-B DNA motifs on promoter function</article-title></title-group><contrib-group content-type="author"><contrib><name name-style="western"><surname>Georgakopoulos-Soares</surname><given-names initials="I">Ilias</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib><name name-style="western"><surname>Victorino</surname><given-names initials="J">Jesus</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref><xref rid="fn3" ref-type="author-notes">14</xref></contrib><contrib><name name-style="western"><surname>Parada</surname><given-names initials="GE">Guillermo E</given-names></name><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref><xref rid="fn3" ref-type="author-notes">14</xref></contrib><contrib><name name-style="western"><surname>Agarwal</surname><given-names initials="V">Vikram</given-names></name><xref ref-type="aff" rid="aff7">7</xref></contrib><contrib><name name-style="western"><surname>Zhao</surname><given-names initials="J">Jingjing</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib><name name-style="western"><surname>Wong</surname><given-names initials="HY">Hei Yuen</given-names></name><xref ref-type="aff" rid="aff8">8</xref></contrib><contrib><name name-style="western"><surname>Umar</surname><given-names initials="MI">Mubarak Ishaq</given-names></name><xref ref-type="aff" rid="aff8">8</xref></contrib><contrib><name name-style="western"><surname>Elor</surname><given-names initials="O">Orry</given-names></name><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib><name name-style="western"><surname>Muhwezi</surname><given-names initials="A">Allan</given-names></name><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib><name name-style="western"><surname>An</surname><given-names initials="JY">Joon-Yong</given-names></name><xref ref-type="aff" rid="aff9">9</xref><xref ref-type="aff" rid="aff10">10</xref></contrib><contrib><name name-style="western"><surname>Sanders</surname><given-names initials="SJ">Stephan J</given-names></name><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff9">9</xref></contrib><contrib><name name-style="western"><surname>Kwok</surname><given-names initials="CK">Chun Kit</given-names></name><xref ref-type="aff" rid="aff8">8</xref><xref ref-type="aff" rid="aff11">11</xref></contrib><contrib><name name-style="western"><surname>Inoue</surname><given-names initials="F">Fumitaka</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref rid="fn1" ref-type="author-notes">12</xref></contrib><contrib><name name-style="western"><surname>Hemberg</surname><given-names initials="M">Martin</given-names></name><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref><xref rid="fn2" ref-type="author-notes">13</xref><xref rid="cor1" ref-type="author-notes">∗</xref></contrib><contrib><name name-style="western"><surname>Ahituv</surname><given-names initials="N">Nadav</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref rid="fn4" ref-type="author-notes">15</xref><xref rid="cor2" ref-type="author-notes">∗∗</xref></contrib></contrib-group><aff id="aff1"><label>1</label>Department of Bioengineering and Therapeutic Sciences, University of California San Francisco, San Francisco, CA, USA</aff><aff id="aff2"><label>2</label>Institute for Human Genetics, University of California San Francisco, San Francisco, CA, USA</aff><aff id="aff3"><label>3</label>Centro Nacional de Investigaciones Cardiovasculares Carlos III (CNIC), 28029 Madrid, Spain</aff><aff id="aff4"><label>4</label>Departamento de Bioquímica, Facultad de Medicina, Universidad Autónoma de Madrid (UAM), 28029 Madrid, Spain</aff><aff id="aff5"><label>5</label>Wellcome Sanger Institute, Wellcome Genome Campus, Hinxton CB10 1SA, UK</aff><aff id="aff6"><label>6</label>Wellcome Trust Cancer Research UK Gurdon Institute, University of Cambridge, Tennis Court Road, Cambridge CB2 1QN, UK</aff><aff id="aff7"><label>7</label>Calico Life Sciences LLC, South San Francisco, CA, USA</aff><aff id="aff8"><label>8</label>Department of Chemistry and State Key Laboratory of Marine Pollution, City University of Hong Kong, Kowloon Tong, Hong Kong SAR, China</aff><aff id="aff9"><label>9</label>Department of Psychiatry, UCSF Weill Institute for Neurosciences, University of California San Francisco, San Francisco, CA, USA</aff><aff id="aff10"><label>10</label>School of Biosystem and Biomedical Science, College of Health Science, Korea University, Seoul, Republic of Korea</aff><aff id="aff11"><label>11</label>Shenzhen Research Institute of City University of Hong Kong, Shenzhen, China</aff><author-notes><fn id="cor1"><label>∗</label><p>Corresponding author <email>mhemberg@bwh.harvard.edu</email></p></fn><fn id="cor2"><label>∗∗</label><p>Corresponding author <email>nadav.ahituv@ucsf.edu</email></p></fn><fn id="fn1"><label>12</label><p id="ntpara0010">Present address: Institute for the Advanced Study of Human Biology (WPI-ASHBi), Kyoto University, Kyoto 606-8501, Japan</p></fn><fn id="fn2"><label>13</label><p id="ntpara0015">Present address: Evergrande Center for Immunologic Diseases, Harvard Medical School and Brigham and Women’s Hospital, Boston, MA, USA</p></fn><fn id="fn3"><label>14</label><p id="ntpara0020">These authors contributed equally</p></fn><fn id="fn4"><label>15</label><p id="ntpara0025">Lead contact</p></fn></author-notes><pub-date><day>15</day><month>3</month><year>2022</year></pub-date><volume>2</volume><issue>4</issue><fpage>100111</fpage><page-range>100111</page-range><pub-history><event event-type="pmc-release"><date><day>13</day><month>5</month><year>2022</year></date></event></pub-history><permissions><copyright-statement>© 2022 The Author(s)</copyright-statement><license><license-p>This is an open access article under the CC BY license (http://creativecommons.org/licenses/by/4.0/).</license-p></license></permissions><self-uri xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="main.pdf" content-type="pmc-pdf"><?cloudpmc-path 393b/9903758/8a9afb0ae3fa/main.pdf?><?cloudpmc-bucket app?><?size 7249848?></self-uri><abstract id="abs0010"><title>Summary</title><p>Alternative DNA conformations, termed non-B DNA structures, can affect transcription, but the underlying mechanisms and their functional impact have not been systematically characterized. Here, we used computational genomic analyses coupled with massively parallel reporter assays (MPRAs) to show that certain non-B DNA structures have a substantial effect on gene expression. Genomic analyses found that non-B DNA structures at promoters harbor an excess of germline variants. Analysis of multiple MPRAs, including a promoter library specifically designed to perturb non-B DNA structures, functionally validated that Z-DNA can significantly affect promoter activity. We also observed that biophysical properties of non-B DNA motifs, such as the length of Z-DNA motifs and the orientation of G-quadruplex structures relative to transcriptional direction, have a significant effect on promoter activity. Combined, their higher mutation rate and functional effect on transcription implicate a subset of non-B DNA motifs as major drivers of human gene-expression-associated phenotypes.</p><sec id="kwrds0010" sec-type="kwd-group" disp-level="2"><p><bold>Keywords:</bold> non-B DNA, Z-DNA, G-quadruplex, MPRA, promoter, mutations</p></sec></abstract><abstract id="abs0015" abstract-type="graphical"><title>Graphical abstract</title><fig id="undfig1" position="anchor"><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="fx1.jpg"><?cloudpmc-path blobs/393b/9903758/16eb8cb5d437/fx1.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 996?><?original-width 996?><?scaled-height 664?><?scaled-width 664?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="fx1.gif"><?cloudpmc-path blobs/393b/9903758/62c17a18b697/fx1.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig></abstract><abstract id="abs0020" abstract-type="author-highlights"><title>Highlights</title><list list-type="label" id="ulist0010"><list-item id="u0010"><label>•</label><p id="p0010">Excess of germline variants at non-B DNA motif loci</p></list-item><list-item id="u0015"><label>•</label><p id="p0015">Massively parallel reporter assays measure the impact of non-B DNA motifs on expression</p></list-item><list-item id="u0020"><label>•</label><p id="p0020">Z-DNA significantly affects promoter activity across cell types and experiments</p></list-item><list-item id="u0025"><label>•</label><p id="p0025">The orientation of G-quadruplexes influences their formation and promoter activity</p></list-item></list></abstract><abstract id="abs0025" abstract-type="teaser"><p>Georgakopoulos-Soares et al. performed computational analyses of germline mutations and identified increased mutability at non-B DNA motifs. The contribution of non-B DNA motifs on gene expression was investigated using massively parallel reporter assays, identifying Z-DNA as a positive regulator of gene expression and finding that the orientation of G-quadruplexes influences promoter activity.</p></abstract><custom-meta-group><custom-meta><meta-name>status</meta-name><meta-value>released</meta-value></custom-meta><custom-meta><meta-name>display-pdf</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>is-olf</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>is-manuscript</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>is-preprint</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>is-journal-matter</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>is-scanned</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>is-retracted</meta-name><meta-value>no</meta-value></custom-meta></custom-meta-group></article-meta><notes notes-type="article-notes"><sec id="historyarticle-meta1" sec-type="history" disp-level="2"><p>Received 2021 Mar 23; Revised 2021 Oct 21; Accepted 2022 Feb 18; Collection date 2022 Apr 13.</p></sec></notes></front><body><sec id="sec1" disp-level="1"><title>Introduction</title><p id="p0045">Under physiological conditions, the favored conformation of DNA is a right-handed double helix, also known as B-DNA (<xref rid="fig1" ref-type="fig">Figure 1</xref>A). However, alternative DNA conformations, collectively termed non-B DNA structures, have been recognized and shown to affect transcription, replication, recombination, and DNA repair, either transiently or for longer periods.<xref rid="bib1" ref-type="bibr"><sup>1</sup></xref> The propensity to form non-canonical structures and their biophysical properties are determined by non-B DNA motifs that can be identified from the primary sequence.<xref rid="bib2" ref-type="bibr">2</xref>, <xref rid="bib3" ref-type="bibr">3</xref>, <xref rid="bib4" ref-type="bibr">4</xref>, <xref rid="bib5" ref-type="bibr">5</xref> For example, Z-DNA is a left-handed double-helical structure that is formed by alternating purine-pyrimidine tracts (<xref rid="fig1" ref-type="fig">Figure 1</xref>B). G-quadruplexes (G4s) consist of four or more G-runs that are interspersed with loop elements (<xref rid="fig1" ref-type="fig">Figure 1</xref>C). Direct and tandem repeats, including mononucleotide repeat tracts, can form slipped structures (<xref rid="fig1" ref-type="fig">Figure 1</xref>D); mirror repeats with high A/G content can form triple-stranded DNA structures (<xref rid="fig1" ref-type="fig">Figure 1</xref>E); and inverted repeats can form hairpins and cruciforms (<xref rid="fig1" ref-type="fig">Figures 1</xref>F and 1G).</p><fig id="fig1" position="float"><?disp-level 2?><label>Figure 1</label><caption><p>Genomic variants are enriched at non-B DNA motifs</p><p>(A–G) Schematic representation of non-B DNA motifs.</p><p>(A) Canonical B-DNA structure.</p><p>(B) Left-handed double-stranded DNA, known as Z-DNA conformation.</p><p>(C) G-quadruplex formation at sites of four G-runs interspersed by looping regions.</p><p>(D) Direct and tandem repeats misalign and form slipped DNA structures. The arms are the repeating unit and the spacer the intervening non-repeating part.</p><p>(E) A subset of mirror repeats with high AG/TC-content fold into intramolecular DNA structures known as H-DNA. The arms are the repeating unit with mirror symmetry and the spacer the intervening non-repeating part.</p><p>(F and G) (F) Inverted repeats fold into hairpin structures, and (G) Inverted repeats can fold into cruciform structures. The arms are the repeating unit with inverted symmetry and the spacer the intervening non-repeating part.</p><p>In schematics (D)–(G), spacer denotes the region of the non-B DNA motif that remains single stranded and exposed, whereas arms hybridize into double-stranded DNA.</p><p>(H) Distribution of non-B DNA motifs relative to 204,063,503 SNPs on the left. Distribution of non-B DNA motifs relative to 25,925,202 small indel variants in the center. Distribution of non-B DNA motifs relative to 505,529 structural variants on the right. Enrichment is corrected for trinucleotide context. DR, G4, IR, MR, and STR refer to direct repeats, G-quadruplexes, inverted repeats, mirror repeats, and short tandem repeats, respectively.</p><p>(I) Association between structural-variant-breakpoint category and enrichment at non-B DNA motifs. INV, CPX, CTX, DEL, DUP, and INS refer to inversions, complex rearrangements, translocations, deletions, duplications, and insertions, respectively. Adjusted p values displayed as ∗p &lt; 0.05, ∗∗p &lt; 0.01, and ∗∗∗p &lt; 0.001.</p><p>(J) Enrichment patterns of eQTLs at non-B DNA motifs relative to proximal regions.</p><p>(K) eQTL density at G4 peaks from G4 antibody treatment.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="gr1.jpg"><?cloudpmc-path blobs/393b/9903758/00ca6cc9d09e/gr1.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 3594?><?original-width 2908?><?scaled-height 899?><?scaled-width 727?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="gr1.gif"><?cloudpmc-path blobs/393b/9903758/24f771af2587/gr1.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig><p id="p0050">Previous studies have shown that non-B DNA structures are mutational hotspots because they are more likely to be exposed as single-stranded DNA, making them vulnerable to damage.<xref rid="bib6" ref-type="bibr"><sup>6</sup></xref><sup>,</sup><xref rid="bib7" ref-type="bibr"><sup>7</sup></xref> Their increased mutability results in an excess of population variants overlapping non-B DNA motifs<xref rid="bib8" ref-type="bibr"><sup>8</sup></xref><sup>,</sup><xref rid="bib9" ref-type="bibr"><sup>9</sup></xref> and an excess of somatic mutagenesis at those sites in cancer.<xref rid="bib10" ref-type="bibr">10</xref>, <xref rid="bib11" ref-type="bibr">11</xref>, <xref rid="bib12" ref-type="bibr">12</xref>, <xref rid="bib13" ref-type="bibr">13</xref>, <xref rid="bib14" ref-type="bibr">14</xref>, <xref rid="bib15" ref-type="bibr">15</xref> Although variants overlapping non-B DNA motifs are frequently neutral in their effect, it is clear that non-B DNA motifs are a major source of genetic variation in the human genome. They are enriched in regulatory regions<xref rid="bib16" ref-type="bibr">16</xref>, <xref rid="bib17" ref-type="bibr">17</xref>, <xref rid="bib18" ref-type="bibr">18</xref>, <xref rid="bib19" ref-type="bibr">19</xref> and likely cause numerous disorders such as cancer, fragile X syndrome, and Friedreich ataxia.<xref rid="bib20" ref-type="bibr">20</xref>, <xref rid="bib21" ref-type="bibr">21</xref>, <xref rid="bib22" ref-type="bibr">22</xref> As a result, they are likely hotspots for disease and genetic variation.<xref rid="bib23" ref-type="bibr"><sup>23</sup></xref> Thus, it is important to take non-B DNA motifs into consideration when modeling mutation rates and pathogenicity.<xref rid="bib7" ref-type="bibr"><sup>7</sup></xref><sup>,</sup><xref rid="bib15" ref-type="bibr"><sup>15</sup></xref><sup>,</sup><xref rid="bib24" ref-type="bibr"><sup>24</sup></xref></p><p id="p0055">In the human genome, non-B DNA motifs are unevenly distributed. They are enriched in certain regulatory regions, including open chromatin, promoters, and 5′ and 3′ UTRs.<xref rid="bib16" ref-type="bibr">16</xref>, <xref rid="bib17" ref-type="bibr">17</xref>, <xref rid="bib18" ref-type="bibr">18</xref>, <xref rid="bib19" ref-type="bibr">19</xref> At the base-pair level, specific non-B DNA motifs are over-represented and positioned relative to critical gene features, such as the transcription start and end sites, splice junctions, and translation initiation regions, while their formation is often associated with transcriptionally active loci.<xref rid="bib25" ref-type="bibr">25</xref>, <xref rid="bib26" ref-type="bibr">26</xref>, <xref rid="bib27" ref-type="bibr">27</xref>, <xref rid="bib28" ref-type="bibr">28</xref>, <xref rid="bib29" ref-type="bibr">29</xref>, <xref rid="bib30" ref-type="bibr">30</xref>, <xref rid="bib31" ref-type="bibr">31</xref> A number of studies have shown, primarily in cancer when targeting selected loci, that non-B DNA motifs can have an impact on the expression levels of various genes. For example, G4s were shown to modulate the expression of key cancer genes, such as <italic>MYC</italic>, <italic>c-Kit</italic>, <italic>BCL2</italic>, and <italic>KRAS</italic>, with their disruption resulting in pronounced expression changes.<xref rid="bib25" ref-type="bibr"><sup>25</sup></xref><sup>,</sup><xref rid="bib32" ref-type="bibr"><sup>32</sup></xref> Furthermore, recurrent mutations across cancer types and patients, including highly recurrent promoter mutations in the <italic>TERT</italic> and <italic>PLEKHS1</italic> genes, overlap non-B DNA motifs<xref rid="bib33" ref-type="bibr">33</xref>, <xref rid="bib34" ref-type="bibr">34</xref>, <xref rid="bib35" ref-type="bibr">35</xref> and likely predispose these regions to increased mutagenesis. However, the functional consequences of non-B DNA motif disruptions, either due to germline or somatic mutations at promoter regions, have not been studied in a systematic manner and remain poorly understood. Additionally, although the impact of promoter non-B DNA structures at individual genes on the regulation of gene expression has been demonstrated at individual loci,<xref rid="bib34" ref-type="bibr"><sup>34</sup></xref><sup>,</sup><xref rid="bib36" ref-type="bibr"><sup>36</sup></xref><sup>,</sup><xref rid="bib37" ref-type="bibr"><sup>37</sup></xref> the results are conflicting regarding the role of non-B DNA motifs acting as either activators or repressors.<xref rid="bib38" ref-type="bibr"><sup>38</sup></xref></p><p id="p0060">Here, we set out to systematically identify the role of non-B DNA motifs on promoter transcriptional regulation. We find that non-B DNA motifs harbor an excess of polymorphisms, many of which affect gene expression levels. To gain further insights regarding the impact of non-B DNA motifs on gene expression, we analyzed various lentivirus-based massively parallel reporter assays (lentiMPRAs<xref rid="bib39" ref-type="bibr"><sup>39</sup></xref>) to systematically test the effect of non-B DNA motifs on promoter activity. We observed a causal link between specific non-B DNA sequences and gene expression levels. We also show that biophysical properties, which influence the likelihood of secondary-structure formation and stability, are linked to these regulatory effects. Our results demonstrate that non-B DNA motifs are important determinants of promoter activity, and their increased mutability implicates them as major drivers of gene-expression-associated phenotypes.</p></sec><sec id="sec2" disp-level="1"><title>Results</title><sec id="sec2.1" disp-level="2"><title>Non-B DNA motifs harbor an excess of standing genetic variation</title><p id="p0065">As previous studies demonstrated that non-B DNA motifs are enriched for somatic mutations,<xref rid="bib11" ref-type="bibr"><sup>11</sup></xref><sup>,</sup><xref rid="bib14" ref-type="bibr"><sup>14</sup></xref><sup>,</sup><xref rid="bib15" ref-type="bibr"><sup>15</sup></xref> we set out to analyze whether this enrichment also exists for germline variation. We took advantage of available whole-genome sequencing (WGS) datasets for thousands of individuals and analyzed them to determine whether non-B DNA sequences are enriched for variants. We measured the genome-wide distribution of 204,063,503 single-nucleotide polymorphisms (SNPs), including both rare and common variants as well as 25,925,202 small insertions and deletions (indels; &lt;50 bp) derived from 15,496 genomes from the gnomAD project<xref rid="bib40" ref-type="bibr"><sup>40</sup></xref> relative to seven non-B DNA motifs: inverted repeats (IRs), direct repeats (DRs), mirror repeats (MRs), short tandem repeats (STRs), G4s, Z-DNA, and H-DNA motifs (<xref rid="fig1" ref-type="fig">Figures 1</xref>A–1G). To form a null distribution, we generated simulated SNPs, controlling for trinucleotide context and proximity to the original SNP (<xref rid="sec4" ref-type="sec">STAR Methods</xref>). We observed an excess of SNPs directly overlapping non-B DNA motifs (<xref rid="mmc1" ref-type="supplementary-material">Figure S1</xref>A; Mann-Whitney U, p &lt; 0.0001), but the magnitude of the effect was small, and the highly significant p value was due to the large sample size. Of note, H-DNA motifs and IRs showed the highest (1.56) and lowest (1.05) fold enrichments, respectively (<xref rid="fig1" ref-type="fig">Figures 1</xref>H, <xref rid="mmc1" ref-type="supplementary-material">S1</xref>B, and S1C). Similarly, the proportion of indels overlapping non-B DNA motifs was substantially elevated relative to the simulated controls (2.26-fold, Mann-Whitney U, p &lt; 0.0001; <xref rid="mmc1" ref-type="supplementary-material">Figure S1</xref>D). The enrichment of genetic variants at individual non-B DNA motifs was higher for small indels than for SNPs, ranging from 2.44-fold for IRs to 13.68-fold for STRs (<xref rid="fig1" ref-type="fig">Figures 1</xref>H, <xref rid="mmc1" ref-type="supplementary-material">S1</xref>E, and S1F). We further separated indels into insertions and deletions, finding differences depending on the non-B DNA motif category (<xref rid="mmc1" ref-type="supplementary-material">Figure S1</xref>G). For example, STRs had a higher frequency of deletions, whereas G4s had a higher frequency of insertions.</p><p id="p0070">Extending our analysis to 505,529 structural-variant breakpoints derived from the gnomAD project,<xref rid="bib40" ref-type="bibr"><sup>40</sup></xref> we found a strong association with non-B DNA motifs, with 14.61% of structural-variant breakpoints directly overlapping a non-B DNA motif versus 8.83% for simulated controls (Mann-Whitney U, p &lt; 0.0001; <xref rid="mmc1" ref-type="supplementary-material">Figure S1</xref>H), representing a 1.66-fold enrichment. For individual non-B DNA motifs, the enrichments ranged from 1.23-fold for G4s to 3.50-fold for H-DNA motifs (<xref rid="fig1" ref-type="fig">Figures 1</xref>H and <xref rid="mmc1" ref-type="supplementary-material">S1</xref>I–S1K; Mann-Whitney p &lt; 0.0001 for all non-B DNA motifs), consistent with previous reports finding an excess of non-B DNA motifs at structural-variant breakpoints.<xref rid="bib41" ref-type="bibr"><sup>41</sup></xref> We separated structural variants into six categories: insertions, deletions, duplications, inversions, translocations, and complex.<xref rid="bib40" ref-type="bibr"><sup>40</sup></xref> We found that deletions, insertions, and duplications were the most enriched across non-B DNA motifs (<xref rid="fig1" ref-type="fig">Figure 1</xref>I). Taken together, these results suggest that non-B DNA motifs are hotspots of genetic variation in the human population across different categories of population variants.</p></sec><sec id="sec2.2" disp-level="2"><title>Non-B DNA motifs are enriched for gene-regulatory-associated variants</title><p id="p0075">To gain further insights regarding the regulatory potential of these variants, we investigated the relative frequency of variants overlapping non-B DNA motifs across six regulatory-element-associated sequences/functions defined by the Ensembl Regulatory Build:<xref rid="bib42" ref-type="bibr"><sup>42</sup></xref> promoters, CTCF-binding sites, open chromatin regions, transcription factor binding sites, promoter flanking regions, and enhancers. The analysis was performed across twelve different cell lines (<xref rid="sec4" ref-type="sec">STAR Methods</xref>), finding that most non-B DNA motifs were enriched for SNPs, indels, and structural variants across the regulatory elements, but more so for indels than for SNPs and structural variants (<xref rid="mmc1" ref-type="supplementary-material">Figures S2</xref>A–S2C). We also investigated the increase in mutagenicity for non-B DNA motifs across the seven annotated genic sub-compartments: genic, intronic, coding, and 5′ and 3′ UTRs as well as 1 kilobase (kb) upstream of the transcription start site (TSS) and 1 kb downstream of the transcription end site (TES). Most regions had elevated mutation rates, although the magnitude varied by mutation type and genic sub-compartment (<xref rid="mmc1" ref-type="supplementary-material">Figures S2</xref>D–S2F). As expected, coding regions showed the lowest mutagenicity relative to other regions, most likely due to selection constraints and increased DNA repair.<xref rid="bib43" ref-type="bibr"><sup>43</sup></xref></p><p id="p0080">To analyze whether variants in non-B DNA motifs could have a substantial impact on gene expression, we analyzed expression quantitative trait loci (eQTL). We examined the frequency of eQTLs, characterized by the GTEx consortium,<xref rid="bib44" ref-type="bibr"><sup>44</sup></xref> at each of the seven non-B DNA motifs genome wide. We found an enrichment of eQTLs across all non-B DNA categories relative to their flanking regions, with the most pronounced effect for G4s (<xref rid="fig1" ref-type="fig">Figure 1</xref>J). Although the excess of eQTLs in the vicinity of non-B DNA motifs can be explained by the higher background frequency of substitution and indel SNPs (<xref rid="fig1" ref-type="fig">Figure 1</xref>H), our results indicate that a subset of mutations overlapping non-B DNA motifs impact gene expression.</p><p id="p0085">As G4s had the most pronounced effect on gene expression, we next analyzed G4 sequencing (G4-seq) and G4 chromatin immunoprecipitation (ChIP)-seq datasets for their overlap with population variants and eQTLs. We investigated the association between population variants and G4s using previously published G4-seq datasets from the HEK-293T cell line with Pyridostatin (PDS) and K<sup>+</sup> treatments that provide <italic>in vitro</italic> evidence of G4 formation potential<xref rid="bib45" ref-type="bibr"><sup>45</sup></xref> and G4 ChIP-seq-derived peaks from the HaCat cell line that provide <italic>in vivo</italic> evidence of sites that form G4 structures.<xref rid="bib16" ref-type="bibr"><sup>16</sup></xref> In accordance with the G4 motif analysis, we found that SNPs, indels, and structural variants were enriched at G4-seq and G4 ChIP-seq peaks (<xref rid="mmc1" ref-type="supplementary-material">Figures S3</xref>A–S3F; Mann-Whitney U, p &lt; 0.001). We considered the G4 ChIP-seq sites that overlapped both G4-seq K<sup>+</sup> and G4-seq PDS peaks as the highest confidence, experimentally derived G4s (<xref rid="mmc1" ref-type="supplementary-material">Figure S3</xref>G) and found consistent enrichments of 1.14-fold, 1.41-fold, and 1.36-fold for substitutions, small indels, and structural variants (<xref rid="mmc1" ref-type="supplementary-material">Figures S3</xref>H and S3I). Next, we found that eQTLs are found more frequently than expected by chance in the experimentally derived G4 sites. In total, 20,310 eQTLs overlapped with the 8,955 ChIP-seq peaks, with 34% of the peaks having one or more eQTL (<xref rid="fig1" ref-type="fig">Figures 1</xref>K and <xref rid="mmc1" ref-type="supplementary-material">S3</xref>J). Interestingly, the enrichment for the experimentally derived G4s was more pronounced than our results derived from the G4 motif analysis. This is likely the result of G4 formation occurring more frequently in open chromatin and transcribed regions.<xref rid="bib16" ref-type="bibr"><sup>16</sup></xref></p><p id="p0090">We also investigated if G4 ChIP-seq peaks overlapping genes display a preference for the template (non-coding) or non-template (coding) strands, using the G4 motif orientation within the peaks as proxy. After correcting for the background bias in the orientation of G4 motifs (<xref rid="mmc1" ref-type="supplementary-material">Figure S4</xref>A), we found that G4 motifs on the non-template strand overlap G4 ChIP-seq peaks 1.71-fold more frequently than motifs on the template strand (binomial test, p &lt; 1 × 10<sup>−12</sup>) (<xref rid="mmc1" ref-type="supplementary-material">Figures S4</xref>B and S4C), suggesting significant bias in the formation of G4s, dependent on their orientation.</p></sec><sec id="sec2.3" disp-level="2"><title>Non-B DNA motifs are enriched in promoter regions</title><p id="p0095">We next investigated the distribution of non-B DNA motifs across the six regulatory elements defined by the Ensembl Regulatory Build (promoters, CTCF-binding sites, open chromatin regions, transcription factor binding sites, promoter flanking regions, and enhancers). For most non-B DNA motifs, we found an enrichment at promoters and CTCF-binding sites relative to other regulatory elements (<xref rid="fig2" ref-type="fig">Figures 2</xref>A and <xref rid="mmc1" ref-type="supplementary-material">S5</xref>A), in accordance with previous findings.<xref rid="bib46" ref-type="bibr"><sup>46</sup></xref> Next, we separated the gene body into six compartments: a 1 kb window upstream from the TSS, the 5′ and 3′ UTRs, coding exons and introns, and a 1 kb window downstream of the TES. Consistently, promoter regions displayed a higher density of non-B DNA motifs than the gene body for most non-B DNA motifs, with the enrichment ranging from 0.97-fold for IRs to 3.02-fold for G4s (<xref rid="fig2" ref-type="fig">Figures 2</xref>B and <xref rid="mmc1" ref-type="supplementary-material">S5</xref>B). We also found a significant enrichment of G4-seq-derived peaks for both PDS and K<sup>+</sup> treatments and for G4 ChIP-seq-derived peaks at promoters relative to other regulatory elements (<xref rid="fig2" ref-type="fig">Figure 2</xref>C). Across the gene body, we found the highest enrichments at promoters, coding regions, and 5′ UTRs (<xref rid="fig2" ref-type="fig">Figure 2</xref>D).</p><fig id="fig2" position="float"><?disp-level 3?><label>Figure 2</label><caption><p>Non-B DNA motifs at functional elements</p><p>(A) Median relative enrichment across 12 cell lines for non-B DNA motif enrichment at Ensembl Regulatory Features.</p><p>(B) Non-B DNA motif enrichment at functional genomic compartments for each non-B DNA motif. Statistical significance was estimated using Binomial tests with Bonferroni correction.</p><p>(C) <italic>Z</italic> score of G4-seq and G4 ChIP-seq peak density across Ensembl Regulatory Features.</p><p>(D) <italic>Z</italic> score of G4-seq and G4 ChIP-seq peak density across the gene body.</p><p>For (C) and (D), two treatments that stabilize G4s, PDS and K<sup>+</sup>, were used in G4-seq.</p><p>(E) Enrichment of non-B DNA motifs in the [–250, 0] region relative to the wider promoter region (–1 kB, 0). Error bars represent standard deviation from bootstrapping.</p><p>(F) Base-pair resolution of distribution of nucleotide motifs relative to the TSS. IRs, MRs, DRs, STRs, and G4s are abbreviations for inverted repeats, mirror repeats, direct repeats, short tandem repeats, and G-quadruplexes, respectively.</p><p>(G) G4 enrichment patterns relative to the TSS for G4 motif, G4-seq peaks in K<sup>+</sup> and PDS treatments, and from G4 ChIP-seq peaks.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="gr2.jpg"><?cloudpmc-path blobs/393b/9903758/a107fc17645f/gr2.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 2776?><?original-width 3333?><?scaled-height 616?><?scaled-width 740?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="gr2.gif"><?cloudpmc-path blobs/393b/9903758/ff2d04a10e34/gr2.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig><p id="p0100">At promoters, for most non-B DNA motifs, the enrichment was higher upstream of the TSS than in the broader promoter region (<xref rid="fig2" ref-type="fig">Figure 2</xref>E). A close investigation of the distribution of non-B DNA motifs relative to the TSS showed an enrichment of peaks ∼50 bp upstream of the TSS ranging between 1.28- and 1.89-fold for DRs and G4 motifs, respectively (<xref rid="fig2" ref-type="fig">Figure 2</xref>F). Importantly, we observed a 5-fold enrichment approximately 100 bp upstream of the TSS for G4 ChIP-seq peaks, consistent with the literature.<xref rid="bib16" ref-type="bibr"><sup>16</sup></xref> Interestingly, the ChIP-seq-derived enrichment was substantially larger than that of the G4 motif and the G4-seq datasets (<xref rid="fig2" ref-type="fig">Figure 2</xref>G), reflecting a preference in structure formation at promoters <italic>in vivo</italic>. We also performed a Gene Ontology (GO) term analysis in promoter upstream regions. For G4s, Z-DNA motifs, and MRs, we found multiple terms associated with developmental processes, such as pattern specification process (GO: 0007389), embryonic organ development (GO: 0048568), and positive regulation of neuron differentiation (GO: 0045666) (<xref rid="mmc1" ref-type="supplementary-material">Figure S6</xref>A). As these analyses suggest that some non-B DNA motifs could control tissue-specific gene expression, we used TissueEnrich to calculate the enrichment of tissue-specific genes and found sets of tissue-specific genes where a set of neuronal-specific genes were enriched for genes containing G4, MR, DR, and STR at their upstream promoter regions (<xref rid="mmc1" ref-type="supplementary-material">Figure S6</xref>B). Altogether, these results demonstrate that promoters are enriched for non-B DNA motifs relative to other regulatory elements and to other genic compartments and that some non-B DNA motifs are more likely to occur at developmental and neuronal genes. Therefore, the excess of genetic variants at non-B DNA motifs identified earlier could have broad implications on gene regulation expression levels across tissues and developmental stages.</p></sec><sec id="sec2.4" disp-level="2"><title>MPRAs identify G4 and Z-DNA to have a substantial effect on gene expression</title><p id="p0105">The enrichment of non-B DNA motifs at promoters and the excess of eQTLs localized within certain non-B DNA motifs prompted us to investigate their functional impact on gene transcription utilizing MPRAs. We first analyzed two lentiMPRA datasets generated by our group as part of the ENCODE consortium,<xref rid="bib47" ref-type="bibr"><sup>47</sup></xref> where a total of 14,625 and 7,346 candidate promoter sequences were examined in both orientations in K562 and HepG2 cell lines. We identified non-B DNA motifs across the lentiMPRA tested sequences (<xref rid="sec4" ref-type="sec">STAR Methods</xref>) and examined their association with gene expression. We found that sequences with G4 and Z-DNA motifs showed significantly increased expression levels in both cell lines (<xref rid="fig3" ref-type="fig">Figures 3</xref>A and 3B; t tests, Bonferroni correction, p &lt; 0.001), while for IRs, DRs, STRs, and MRs we did not observe consistent results (<xref rid="mmc1" ref-type="supplementary-material">Figure S7</xref>A). As there is a known positive correlation between expression and guanine-cytosine (GC) content,<xref rid="bib48" ref-type="bibr"><sup>48</sup></xref> which was also observed in our lentiMPRA datasets (Pearson r = 0.398 and 0.261 in K562 and HepG2, respectively), we constructed a linear model to account for the contribution of GC content toward expression (<xref rid="mmc1" ref-type="supplementary-material">Figure S7</xref>B). Sequences with Z-DNA motifs had substantially elevated expression levels relative to sequences without them, even after controlling for GC content in both cell lines (t tests, Bonferroni correction p &lt; 0.001; <xref rid="fig3" ref-type="fig">Figures 3</xref>C and <xref rid="mmc1" ref-type="supplementary-material">S7</xref>C). However, after GC-content correction, G4s were not associated with increased expression, and in HepG2, they were instead significantly associated with reduced expression levels (<xref rid="fig3" ref-type="fig">Figures 3</xref>C and <xref rid="mmc1" ref-type="supplementary-material">S7</xref>C). Similar results were obtained after removing outliers from the linear model (absolute <italic>Ζ</italic> score &gt;2.5). Also, G4s on the template strand were associated with reduced expression relative to non-template strands in both cell lines, but the difference reached statistical significance only in the HepG2 cell line (<xref rid="mmc1" ref-type="supplementary-material">Figure S7</xref>D). For the other non-B DNA motifs, we could not find consistent effects in both cell lines, suggesting that nucleotide composition contributed to the observed effects before GC-content correction.</p><fig id="fig3" position="float"><?disp-level 3?><label>Figure 3</label><caption><p>Contribution of sequences with non-B DNA motifs toward gene expression</p><p>(A) Association between presence of different non-B DNA motifs and expression. Median differences in expression of sequences with and without each non-B DNA motif are shown. Error bars show standard deviation from bootstrapping.</p><p>(B) Comparative analysis of sequences with and without G4s and Z-DNA motifs.</p><p>(C) Comparative analysis of sequences with and without G4s and Z-DNA motifs controlling for GC content.</p><p>(B and C) Statistical significance was calculated with t tests and Bonferroni correction.</p><p>(D and E) Relative expression differences between the median expression for sequences with and without non-B DNA motifs and transcription factor binding sites in (D) HepG2 and (E) K562 lentiMPRA.</p><p>Statistical significance is estimated with t tests and Bonferroni correction. In (B) and (C), adjusted p values displayed as ∗p &lt; 0.05, ∗∗p &lt; 0.01, and ∗∗∗p &lt; 0.001.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="gr3.jpg"><?cloudpmc-path blobs/393b/9903758/d7eae3e35116/gr3.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 2144?><?original-width 3333?><?scaled-height 476?><?scaled-width 740?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="gr3.gif"><?cloudpmc-path blobs/393b/9903758/af04d1697ac9/gr3.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig><p id="p0110">Finally, we identified transcription factor binding sites (TFBSs) across the MPRA sequences using the JASPAR vertebrate non-redundant list of transcription factor motifs.<xref rid="bib49" ref-type="bibr"><sup>49</sup></xref> We compared the contribution of non-B DNA motifs relative to TFBSs toward expression levels, both before and after GC-content correction. We found that G4 and Z-DNA motifs had similar contributions to known TFBSs, such as EGR1, YY1, and SP9, resulting in increased expression levels relative to sequences without them (<xref rid="mmc1" ref-type="supplementary-material">Figure S8</xref>). However, only Z-DNA motifs had comparable effects when we accounted for GC content (<xref rid="fig3" ref-type="fig">Figures 3</xref>D and 3E), and the results were consistent between HepG2 and K562 lentiMPRAs.</p><p id="p0115">To further validate our findings, we analyzed lentiMPRA results from a library that characterized the effect of 3,623 <italic>de novo</italic> promoter mutations that were identified in the Simons Simplex Collection.<xref rid="bib50" ref-type="bibr"><sup>50</sup></xref> This library tested both alleles, centered around the variant, totaling 7,246 sequences along with 150 positive and 150 negative controls for their effect on promoter activity in neural progenitor cells (NPCs) (<xref rid="mmc1" ref-type="supplementary-material">Figures S9</xref>A–S9C). This library had 1,234 sequences harboring one or more non-B DNA motifs (<xref rid="mmc1" ref-type="supplementary-material">Figure S9</xref>D). We observed that sequences harboring G4, DR, and Z-DNA motifs displayed a significantly higher expression than sequences without them (t tests, Bonferroni corrected p values, G4s, DRs, and Z-DNA p &lt; 0.001), whereas sequences with IRs, MRs, and STRs did not show a significant association (p &gt; 0.05) (<xref rid="fig4" ref-type="fig">Figure 4</xref>A).</p><fig id="fig4" position="float"><?disp-level 3?><label>Figure 4</label><caption><p>Expression-associated variants relative to non-B DNA motifs</p><p>(A and B) Expression of sequences with and without each of the non-B DNA motifs: (A) without adjusting for GC content and (B) adjusting for GC content. t tests with Bonferroni correction were performed.</p><p>(C) Expression is associated with the orientation of G4s at promoters.</p><p>(D) The length of Z-DNA motifs was associated with increased gene expression (Kruskal-Wallis H test, p &lt; 0.001).</p><p>(E) Circular dichroism (CD) spectra of the four candidate targets for G4 formation potential in presence of two cations.</p><p>(F) UV-melting profiles of the four G4 candidates in presence of K<sup>+</sup>. The reverse melting profile (K<sup>+</sup><sub>rev</sub>) is also shown and matched well with the forward melting profile (K<sup>+</sup>). Hypochromic shift at 295 nm is a hallmark for G4 formation, which can be transformed into a negative peak in derivative plot (dAbs/dT) for G4 stability analysis. The melting temperature (Tm) of a G4 can be identified at the maximum negative value.</p><p>(G) Fluorescence emission associated with NMM ligand binding to G4 candidates in the presence of Li<sup>+</sup> or K<sup>+</sup> ions.</p><p>(H) Intrinsic fluorescence of four candidate DNA oligonucleotides under Li<sup>+</sup> or K<sup>+</sup> conditions.</p><p>In (A)–(C), adjusted p values displayed as ∗p &lt; 0.05, ∗∗p &lt; 0.01, and ∗∗∗p &lt; 0.001.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="gr4.jpg"><?cloudpmc-path blobs/393b/9903758/0dbc945ed6b9/gr4.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 3520?><?original-width 3333?><?scaled-height 782?><?scaled-width 740?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="gr4.gif"><?cloudpmc-path blobs/393b/9903758/4cd7baa8ddf9/gr4.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig><p id="p0120">Similar to the analysis of the ENCODE MPRA libraries, we observed a significant contribution of the GC content toward the effects on expression of certain non-B DNA motifs. After constructing a linear model to adjust for GC content, we observed that G4 motifs are associated with decreased expression, while only Z-DNA sequences remained associated with higher expression (<xref rid="fig4" ref-type="fig">Figure 4</xref>B), consistent with previous results. In this case, removing outliers maintained a positive association with G4s and gene expression. We also observed a substantial difference in the expression dependent on the orientation of G4 motifs, with G4s on the template strand having lower expression than those on the non-template strand before and after GC-content adjustment (<xref rid="fig4" ref-type="fig">Figures 4</xref>C and <xref rid="mmc1" ref-type="supplementary-material">S9</xref>E; Mann-Whitney U, p &lt; 0.001). The primary sequence comprising consecutive G-runs that are interspersed by loop elements can form G4 structures (<xref rid="fig1" ref-type="fig">Figure 1</xref>). The association between G-runs and gene expression was further investigated, finding that consecutive G-runs result in decreased expression when accounting for their GC-content contribution (<xref rid="mmc1" ref-type="supplementary-material">Figure S9</xref>F). Furthermore, we found that the length of the Z-DNA motif was positively associated with the expression levels (Kruskal-Wallis H test, p &lt; 0.001; <xref rid="fig4" ref-type="fig">Figure 4</xref>D).</p><p id="p0125">Similar to the previous MPRAs, we identified TFBSs across the MPRA sequences and compared the contribution of non-B DNA motifs relative to TFBSs toward expression levels before and after GC-content correction. We found that G4 and Z-DNA motifs had comparable contributions to TFBSs toward increasing expression levels with increases of 1.27- and 1.51-fold over sequences without them (<xref rid="mmc1" ref-type="supplementary-material">Figure S10</xref>A). However, when we accounted for GC content, the effect of non-B DNA motifs was not comparable to the best TFBS motifs (<xref rid="mmc1" ref-type="supplementary-material">Figure S10</xref>B). Therefore, we find substantial differences in the results in NPCs relative to HepG2 and K562 cell lines, with a lower contribution of Z-DNA motifs in NPCs, which might be due to the selection of loci that were not necessarily proximal to the TSS or due to the lower number of Z-DNA-containing sequences, with only 311 sequences having them.</p><p id="p0130">To validate if the G4s we observed in this NPC lentiMPRA form these structures, we selected ten candidate promoter-proximal sequences with the lowest and highest expression among sequences with G4s (<xref rid="mmc1" ref-type="supplementary-material">Table S1</xref>) and performed multiple spectroscopic assays to characterize their structures (<xref rid="fig4" ref-type="fig">Figures 4</xref>E and 4F), as G4 structures possess distinct spectroscopic features.<xref rid="bib51" ref-type="bibr"><sup>51</sup></xref><sup>,</sup><xref rid="bib52" ref-type="bibr"><sup>52</sup></xref> We first used circular dichroism spectroscopy measurements of the G4-containing DNA oligonucleotides, in the presence of lithium ions (non-G4 stabilizing) or potassium ions (G4 stabilizing), to examine the formation potential of DNA G4s, which indicated that our candidate sequences can fold into G4 structures (<xref rid="fig4" ref-type="fig">Figures 4</xref>E, 4F, <xref rid="mmc1" ref-type="supplementary-material">S11</xref>A, and S11B). In addition, we conducted UV melting and found a hypochromic shift at 295 nm for the potassium-ion condition, which supported the formation of the G4 structure, with a melting temperature above physiological temperature (<xref rid="fig4" ref-type="fig">Figures 4</xref>E, 4F, <xref rid="mmc1" ref-type="supplementary-material">S11</xref>A, and S11B).</p><p id="p0135">To confirm the results from the circular dichroism and UV-melting experiments, we used fluorescent-based arrays, including N-methyl mesoporphyrin IX (NMM)-ligand-enhanced fluorescence and intrinsic fluorescence experiments (<xref rid="fig4" ref-type="fig">Figures 4</xref>G, 4H,<xref rid="mmc1" ref-type="supplementary-material">S12</xref>, and S12B). In the absence of NMM ligand, no fluorescence was observed at ∼610 nm. Upon NMM addition, weak fluorescence was observed under Li<sup>+</sup>, which was substantially enhanced when substituted with K<sup>+</sup>, supporting the formation of G4 that allows recognition of NMM and enhances its fluorescence (<xref rid="fig4" ref-type="fig">Figure 4</xref>G). Similarly, the intrinsic fluorescence of G4s was increased when replacing Li<sup>+</sup> with K<sup>+</sup>, highlighting the formation of DNA G4s (<xref rid="fig4" ref-type="fig">Figure 4</xref>H). Corroborating our results, we observed increased fluorescence intensity under conditions that promote G4 formation for all candidates. We also carried out two positive G4 controls and a negative B-DNA control to verify our findings above (<xref rid="mmc1" ref-type="supplementary-material">Figure S13</xref>). Combined, these results validate that these sequences form G4 structures <italic>in vitro</italic>.</p></sec><sec id="sec2.5" disp-level="2"><title>Non-B DNA motifs have a significant effect on promoter activity</title><p id="p0140">To directly test the effect of non-B DNA structures on promoter activity, we generated an MPRA library that introduces various non-B DNA perturbations to ten disease-associated genes. This set of genes included cancer oncogenes (<italic>CMYC</italic>, <italic>CKIT</italic>, <italic>BCL2</italic>, <italic>KRAS</italic>) and genes associated with different cancer types (<italic>ADAM12</italic>, <italic>ALOX5</italic>, <italic>SRSF6</italic>, <italic>VEGF12</italic>) as well as <italic>FMR1</italic>, associated with fragile X syndrome (OMIM: <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://omim.org/entry/300624" ext-link-type="uri">300624</ext-link>), and <italic>SNX12</italic>, which is associated with neurodegenerative diseases (<xref rid="mmc1" ref-type="supplementary-material">Table S2</xref>). As our MPRA-tested sequences are 200 bp in length, we first validated whether our selected 200 bp sequences could drive promoter activity using luciferase assays in K562, MCF7, IMR90, and HEK293T cells, finding the majority to be active in most cell lines (<xref rid="mmc1" ref-type="supplementary-material">Figures S14</xref>A and S14B).</p><p id="p0145">Following validation of these 200 bp sequences, we next generated an MPRA library that included the following manipulations: (1) disruption of existing non-B DNA motifs and (2) introduction of different non-B DNA motifs with varied biophysical properties, including spacer- and arm-length changes in IRs, DRs, and MRs, orientation and loop length in G4s, and length in Z-DNA motifs. lentiMPRAs and subsequent computational analyses were carried out as previously described.<xref rid="bib53" ref-type="bibr"><sup>53</sup></xref> Briefly, oligonucleotides were synthesized and cloned into a lentiviral MPRA promoter vector (<xref rid="fig5" ref-type="fig">Figure 5</xref>A; <xref rid="mmc1" ref-type="supplementary-material">Table S2</xref>), and lentivirus libraries were generated. Libraries were used to infect both K562 and HEK293T cells for 3 days, to allow non-integrating lentivirus to degenerate, and DNA and RNA barcodes were sequenced. Since previous work in our lab showed that lower basal activity can have a significant effect on MPRA results,<xref rid="bib54" ref-type="bibr"><sup>54</sup></xref> these two cell lines were chosen as almost all the selected promoters showed ≥2-fold activity compared with empty vectors (except for <italic>CKIT</italic> in HEK293T). All experiments were done in triplicate, and computational analyses were carried out using MPRAflow<xref rid="bib53" ref-type="bibr"><sup>53</sup></xref> and MPRAnalyze.<xref rid="bib55" ref-type="bibr"><sup>55</sup></xref> We observed a strong correlation between all three replicates (Pearson r ≥ 0.9 in all cases; <xref rid="mmc1" ref-type="supplementary-material">Figure S15</xref>A) and between the two cell lines (Pearson r = 0.87; <xref rid="mmc1" ref-type="supplementary-material">Figure S15</xref>B).</p><fig id="fig5" position="float"><?disp-level 3?><label>Figure 5</label><caption><p>Characterization of non-B DNA motifs across nine promoter templates</p><p>(A) Schematic summary of the experimental design for the promoter lentiMPRA. An example of one of the promoters is depicted at the top left with several non-B DNA motifs, and several mutations are shown at the bottom (site mutations) and on the right (duplication/substitution) for G4s. The collection of all promoters is ordered as an oligonucleotide library of 230-mer. The oligonucleotide library is PCR amplified and barcoded at the 5′ UTR using a degenerate reverse primer. Cloning of PCR products into a lentiviral promoter assay vector was performed next. Cloning of PCR products into a promoter-less lentiviral vector was then performed. This plasmid library was sequenced to assign every barcode to one of the promoters in the library (left) and used to produce the lentiviral library (bottom), which was then used to infect the cell lines of interest (K562 and HEK293T). RNA and DNA were collected after 3 days post-infection, and the barcodes were sequenced. Promoter activity was calculated as the log(RNA/DNA). LTR, long terminal repeats; ARE, antirepressor element.</p><p>(B) Expression levels of nine genes and their sequence variants for K562 and HEK-293T cell lines.</p><p>(C) Boxplot displaying the <italic>Z</italic> score, for sequences with and without each non-B DNA motif, calculated separately for each gene, in K562 and HEK293T cell lines.</p><p>(D) Sequences with Z-DNA motifs display higher expression than sequences with Z-DNA disruptions for <italic>SNX12</italic> and <italic>SRSF6</italic> genes.</p><p>(E) Sequences with G4 motifs on the non-template strand have a higher expression than sequences with G4 motifs at the template strand.</p><p>(F) Sequences with longer Z-DNA motifs display higher expression.</p><p>(A–F) Mann-Whitney U tests with Bonferroni correction were performed, showing significant difference in sequences with and without the displayed non-B DNA motifs, p &lt; 0.05 in all cases.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="gr5.jpg"><?cloudpmc-path blobs/393b/9903758/8f86af34d6db/gr5.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 1946?><?original-width 3321?><?scaled-height 432?><?scaled-width 738?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="gr5.gif"><?cloudpmc-path blobs/393b/9903758/b3c588c0e0a6/gr5.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig><p id="p0150">The promoters in our MPRAs showed variable expression, with the highest levels observed for <italic>SRSF6</italic> and the lowest for <italic>ADAM12</italic> (<xref rid="fig5" ref-type="fig">Figure 5</xref>B). We investigated the contribution of each non-B DNA motif toward expression in both cell lines across the promoters, adjusting across genes using <italic>Z</italic> score normalization. Specifically, for each gene we calculated the <italic>Z</italic> score of each sequence, which was calculated by subtracting the expression levels of that sequence from the mean across all sequences of that gene and dividing by the standard deviation. In concordance with our previous MPRA analyses, we observed that sequences with Z-DNA and G4 motifs had significantly higher expression (<xref rid="fig5" ref-type="fig">Figures 5</xref>C and 5D). Interestingly, while we did not observe consistent results in our previous MPRA analyses for MRs, DRs, and IRs, here, we observed significantly higher expression levels when MRs and DRs were present, whereas for IRs we found significantly lower expression (<xref rid="fig5" ref-type="fig">Figure 5</xref>C). For STRs, we did not find consistent patterns in the two cell lines. The above results across non-B DNA motifs did not change when we accounted for GC content; however, this was most likely due to our experimental design having only a small number of loci targeted, which, as a result, had a narrow and uninformative GC-content range.</p><p id="p0155">For G4s, we introduced a single, two, or three mutations in one, two, three, or every G-run at the original G4 genomic sites. We compared the mutated sequences with the original sequence and found that sequences with disruptions in the G-runs did not display significant expression differences from the original sequences (<xref rid="mmc1" ref-type="supplementary-material">Figure S15</xref>C). We designed MPRA sequences with scrambled Z-DNA motifs or with disruptions of purines to pyrimidines in the alternating purine-pyrimidine tract, which served as Z-DNA controls. We found that there was a statistically significant reduction in expression following the disruption of Z-DNA motifs (<xref rid="fig5" ref-type="fig">Figure 5</xref>D), supporting the notion that they are activating sequences. We also observed that non-template G4s had higher expression than those at the template strand in both cell lines and both before and after GC-content correction (Mann-Whitney U, Bonferroni corrected; <xref rid="fig5" ref-type="fig">Figure 5</xref>E), consistent with our earlier results. For Z-DNA, longer motifs resulted in higher expression (<xref rid="fig5" ref-type="fig">Figure 5</xref>F). These results suggest that the non-B DNA motifs and their biophysical properties contribute to expression across promoter templates.</p></sec></sec><sec id="sec3" disp-level="1"><title>Discussion</title><p id="p0160">By analyzing thousands of WGS datasets, we found that non-B DNA motifs are hotspots for genetic variation, fitting with their known increased mutability properties. Their increased mutability is consistently observed across mutation types, including substitutions but also larger and more disruptive indels and structural variants. The increased likelihood of mutagenesis at non-B DNA motifs is also consistent with previous analyses of somatic mutations in cancer genomes.<xref rid="bib15" ref-type="bibr"><sup>15</sup></xref> Different mechanisms underlying the higher mutation rate at individual non-B DNA motifs have been previously identified, such as DNA polymerase slippage errors at microsatellites causing deletions,<xref rid="bib20" ref-type="bibr"><sup>20</sup></xref> which was also observed in this study. We also observed an excess of eQTLs in the vicinity of non-B DNA motifs. In particular, at experimentally identified G4s, the eQTL enrichment was even larger than that observed across G4 motifs (<xref rid="fig1" ref-type="fig">Figures 1</xref>J and 1K), which is likely due to the formation of G4 motifs being more frequent in open chromatin regions and nucleosome-depleted regions.<xref rid="bib16" ref-type="bibr"><sup>16</sup></xref> We further show that non-B DNA motifs are enriched in promoters where they can directly influence downstream gene expression levels. Specifically, we observed that Z-DNA motifs increase expression, whereas the effect of G4s is dependent on the gene studied. Combined, these results suggest that gene-regulatory variants are more likely to occur at non-B DNA structures and that they have a substantial impact on gene expression.</p><p id="p0165">The promoter effects of G4s have previously been shown to be inhibitory or activating depending on the target gene.<xref rid="bib56" ref-type="bibr">56</xref>, <xref rid="bib57" ref-type="bibr">57</xref>, <xref rid="bib58" ref-type="bibr">58</xref> Similarly, previous work has suggested that Z-DNA sequences can act as both activating and repressing elements in promoters.<xref rid="bib29" ref-type="bibr"><sup>29</sup></xref><sup>,</sup><xref rid="bib59" ref-type="bibr"><sup>59</sup></xref><sup>,</sup><xref rid="bib60" ref-type="bibr"><sup>60</sup></xref> Here, we found that in the absence of chemical perturbations, Z-DNA sequences are more likely to be activating, while G4s are more likely to be inhibitory and promoter dependent. One of the mechanisms by which Z-DNA motifs might increase gene expression might be the reduction of nucleosome occupancy that they elicit.<xref rid="bib60" ref-type="bibr"><sup>60</sup></xref> The reduction of expression at promoters with G4 motifs could be due to interference with transcription factor or RNA polymerase II binding. In addition, template G4s have a more inhibitory effect than non-template ones. The stronger inhibitory effect at the template strand is also aligned with potentially interfering with RNA polymerase II binding. These results are suggestive of inhibitory effects of G4s in promoters, which can be mischaracterized if the effect of GC content is not taken into consideration, as well as orientation-dependent regulatory effects.</p><p id="p0170">Non-B DNA structure formation depends on a plethora of factors, including DNA superhelicity as well as the activity of multiple enzymes such as topoisomerases and helicases.<xref rid="bib61" ref-type="bibr"><sup>61</sup></xref><sup>,</sup><xref rid="bib62" ref-type="bibr"><sup>62</sup></xref> Small molecules that stabilize G4s can substantially alter the thermodynamic equilibrium of structure formation, resulting in dramatic changes in gene expression.<xref rid="bib63" ref-type="bibr"><sup>63</sup></xref><sup>,</sup><xref rid="bib64" ref-type="bibr"><sup>64</sup></xref> Thus, targeting these sequences in key regulatory sites could be a potential novel therapeutic path.<xref rid="bib65" ref-type="bibr"><sup>65</sup></xref> Although the selectivity of such compounds is usually limited, molecules that discriminate among G4s have also been characterized.<xref rid="bib66" ref-type="bibr"><sup>66</sup></xref> These can modulate the activity of clinically important genes, as recently shown for the telomerase gene (<italic>TERT</italic>), where promoter mutations have been associated with a variety of cancers.<xref rid="bib67" ref-type="bibr"><sup>67</sup></xref> By targeting a G4 in the <italic>TERT</italic> promoter with a small molecule, the expression of telomerase was down-regulated in cancer cells.<xref rid="bib34" ref-type="bibr"><sup>34</sup></xref> However, small molecules targeting G4s could cause concomitant DNA damage and telomere dysfunction, influence telomere length, and interfere with other biological processes.<xref rid="bib63" ref-type="bibr"><sup>63</sup></xref> Targeting these non-B DNA structures via <italic>cis</italic>-regulation therapy could be an alternate approach to alter target gene expression.<xref rid="bib68" ref-type="bibr"><sup>68</sup></xref></p><p id="p0175">It is increasingly recognized that non-B DNA motifs are involved in a plethora of cellular processes, such as transcription and translation initiation, splicing, and transcription termination.<xref rid="bib26" ref-type="bibr">26</xref>, <xref rid="bib27" ref-type="bibr">27</xref>, <xref rid="bib28" ref-type="bibr">28</xref>, <xref rid="bib29" ref-type="bibr">29</xref><sup>,</sup><xref rid="bib69" ref-type="bibr">69</xref>, <xref rid="bib70" ref-type="bibr">70</xref>, <xref rid="bib71" ref-type="bibr">71</xref>, <xref rid="bib72" ref-type="bibr">72</xref>, <xref rid="bib73" ref-type="bibr">73</xref>, <xref rid="bib74" ref-type="bibr">74</xref>, <xref rid="bib75" ref-type="bibr">75</xref>, <xref rid="bib76" ref-type="bibr">76</xref>, <xref rid="bib77" ref-type="bibr">77</xref>, <xref rid="bib78" ref-type="bibr">78</xref>, <xref rid="bib79" ref-type="bibr">79</xref>, <xref rid="bib80" ref-type="bibr">80</xref>, <xref rid="bib81" ref-type="bibr">81</xref> Therefore, future work is required to explore the regulatory effects of mutations at non-B DNA motifs genome wide and to estimate their overall pathogenicity by integrating the topology of non-B DNA motifs and the downstream biological effects of their disruption. In addition, measuring the likelihood of mutagenesis for individual non-B DNA motifs per cell division in somatic and cancer cells could have important implications relevant to modeling cancer evolution and aging. Further systematic and high-throughput functional assays could extend our understanding of the functional diversity and clinical evaluation of particular non-B DNA motifs and the variants within them.</p><sec id="sec3.1" disp-level="2"><title>Limitations of the study</title><p id="p0180">Our study has multiple limitations. First, the examination of the regulatory roles of non-B DNA motifs through MPRA experiments did not investigate how molecules that stabilize their formation affect the conclusions reached. Secondly, the MPRA results are based on specific cell lines, and it would be of interest to examine which of these findings can be generalized across cell types and which effects are cell-type specific. We also cannot exclude the influence of the experimental design in our findings. Furthermore, additional experiments and mechanistic work are required to further our understanding, including biophysical and molecular experiments. Lastly, future work would be needed to resolve the relevance of mutations at non-B DNA motifs in the development and progression of human diseases. The aforementioned limitations could be of high interest for future work.</p></sec></sec><sec id="sec4" disp-level="1"><title>STAR★Methods</title><sec id="sec4.1" disp-level="2"><title>Key resources table</title>
<table-wrap id="undtbl1" position="float"><table frame="hsides" rules="groups"><thead><tr><th colspan="1" rowspan="1">REAGENT or RESOURCE</th><th colspan="1" rowspan="1">SOURCE</th><th colspan="1" rowspan="1">IDENTIFIER</th></tr></thead><tbody><tr><td colspan="3" rowspan="1"><bold>Bacterial and virus strains</bold></td></tr><tr><td colspan="3" rowspan="1"><hr/></td></tr><tr><td colspan="1" rowspan="1">ElectroMAX Stbl4</td><td colspan="1" rowspan="1">ThermoFisher Scientific</td><td colspan="1" rowspan="1">Cat#11635018</td></tr><tr><td colspan="3" rowspan="1"><hr/></td></tr><tr><td colspan="3" rowspan="1"><bold>Chemicals, peptides, and recombinant proteins</bold></td></tr><tr><td colspan="3" rowspan="1"><hr/></td></tr><tr><td colspan="1" rowspan="1">Polybrene</td><td colspan="1" rowspan="1">Sigma</td><td colspan="1" rowspan="1">Cat#TR-1003-G</td></tr><tr><td colspan="1" rowspan="1">Lithium Hydroxide</td><td colspan="1" rowspan="1">Acros Organics</td><td colspan="1" rowspan="1">Cat# 413325000</td></tr><tr><td colspan="1" rowspan="1">Cacodylic Acid</td><td colspan="1" rowspan="1">Acros Organics</td><td colspan="1" rowspan="1">Cat# 318150100</td></tr><tr><td colspan="1" rowspan="1">Lithium Chloride</td><td colspan="1" rowspan="1">Sigma</td><td colspan="1" rowspan="1">t# L7026-1L</td></tr><tr><td colspan="1" rowspan="1">Potassium Chloride</td><td colspan="1" rowspan="1">Thermo Fisher</td><td colspan="1" rowspan="1">Cat# J/2892/15</td></tr><tr><td colspan="1" rowspan="1">N-Methyl Mesoporphyrin IX (NMM)</td><td colspan="1" rowspan="1">Frontier Specialty Chemicals</td><td colspan="1" rowspan="1">Cat# NMM580-5mg</td></tr><tr><td colspan="1" rowspan="1">Dimethyl sulfoxide</td><td colspan="1" rowspan="1">J&amp;K Scientific</td><td colspan="1" rowspan="1">Cat# 292271</td></tr><tr><td colspan="3" rowspan="1"><hr/></td></tr><tr><td colspan="3" rowspan="1"><bold>Critical commercial assays</bold></td></tr><tr><td colspan="3" rowspan="1"><hr/></td></tr><tr><td colspan="1" rowspan="1">Lenti-Pac HIV Expression Packaging Kit</td><td colspan="1" rowspan="1">Genecopoeia</td><td colspan="1" rowspan="1">Cat#LT001</td></tr><tr><td colspan="1" rowspan="1">Lenti-X concentrator</td><td colspan="1" rowspan="1">Takara</td><td colspan="1" rowspan="1">Cat#631231</td></tr><tr><td colspan="1" rowspan="1">Dual-Luciferase Reporter Assay System</td><td colspan="1" rowspan="1">Promega</td><td colspan="1" rowspan="1">Cat#E1910</td></tr><tr><td colspan="1" rowspan="1">NEBuilder HiFi DNA Assembly Master Mix</td><td colspan="1" rowspan="1">New England Biolabs</td><td colspan="1" rowspan="1">Cat#E2621S</td></tr><tr><td colspan="1" rowspan="1">NEBNext High-Fidelity 2X PCR Master Mix</td><td colspan="1" rowspan="1">New England Biolabs</td><td colspan="1" rowspan="1">Cat#M0541S</td></tr><tr><td colspan="1" rowspan="1">QIAquick gel extraction kit</td><td colspan="1" rowspan="1">Qiagen</td><td colspan="1" rowspan="1">Cat#28704</td></tr><tr><td colspan="1" rowspan="1">MinElute Reaction Cleanup Kit</td><td colspan="1" rowspan="1">Qiagen</td><td colspan="1" rowspan="1">Cat#28204</td></tr><tr><td colspan="1" rowspan="1">Allprep DNA/RNA mini kit</td><td colspan="1" rowspan="1">Qiagen</td><td colspan="1" rowspan="1">Cat#80204</td></tr><tr><td colspan="1" rowspan="1">Oligotex mRNA mini kit</td><td colspan="1" rowspan="1">Qiagen</td><td colspan="1" rowspan="1">Cat#70022</td></tr><tr><td colspan="1" rowspan="1">SuperScriptII</td><td colspan="1" rowspan="1">Life Technologies</td><td colspan="1" rowspan="1">Cat#18064-071</td></tr><tr><td colspan="3" rowspan="1"><hr/></td></tr><tr><td colspan="3" rowspan="1"><bold>Deposited data</bold></td></tr><tr><td colspan="3" rowspan="1"><hr/></td></tr><tr><td colspan="1" rowspan="1">ENCODE MPRA for K562 and HEPG2 cell lines</td><td colspan="1" rowspan="1">Consortium, Encode Project<xref rid="bib47" ref-type="bibr"><sup>47</sup></xref></td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://www.encodeproject.org/" ext-link-type="uri">https://www.encodeproject.org/</ext-link></td></tr><tr><td colspan="1" rowspan="1">HEK-293T and K562 MPRPA</td><td colspan="1" rowspan="1">This paper</td><td colspan="1" rowspan="1">PRJNA763774</td></tr><tr><td colspan="1" rowspan="1">NPC MPRA</td><td colspan="1" rowspan="1">This paper</td><td colspan="1" rowspan="1">PRJNA763774</td></tr><tr><td colspan="1" rowspan="1">Non-B DNA motif maps</td><td colspan="1" rowspan="1">Cer et al.<xref rid="bib82" ref-type="bibr"><sup>82</sup></xref></td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://nonb-abcc.ncifcrf.gov/" ext-link-type="uri">https://nonb-abcc.ncifcrf.gov/</ext-link></td></tr><tr><td colspan="1" rowspan="1">Ensembl Regulatory Build</td><td colspan="1" rowspan="1">Zerbino et al.<xref rid="bib42" ref-type="bibr"><sup>42</sup></xref></td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://m.ensembl.org/info/genome/funcgen/regulatory_build.html" ext-link-type="uri">https://m.ensembl.org/info/genome/funcgen/regulatory_build.html</ext-link></td></tr><tr><td colspan="1" rowspan="1">G4-seq and G4-ChIP-seq data</td><td colspan="1" rowspan="1">Marsico et al.,<xref rid="bib45" ref-type="bibr"><sup>45</sup></xref> Hänsel-Hertsch et al.<xref rid="bib83" ref-type="bibr"><sup>83</sup></xref></td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="http://ncbi-geo:GSE63874" ext-link-type="uri">GSE63874</ext-link>; <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="http://ncbi-geo:GSE107690" ext-link-type="uri">GSE107690</ext-link></td></tr><tr><td colspan="1" rowspan="1">eQTLs from GTEx consortium</td><td colspan="1" rowspan="1">GTEx Consortium<xref rid="bib44" ref-type="bibr"><sup>44</sup></xref></td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://gtexportal.org/home/" ext-link-type="uri">https://gtexportal.org/home/</ext-link></td></tr><tr><td colspan="1" rowspan="1">Population variants</td><td colspan="1" rowspan="1">Karczewski et al.<xref rid="bib40" ref-type="bibr"><sup>40</sup></xref></td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://gnomad.broadinstitute.org/" ext-link-type="uri">https://gnomad.broadinstitute.org/</ext-link></td></tr><tr><td colspan="1" rowspan="1">Transcription factor binding profiles</td><td colspan="1" rowspan="1">Fornes et al.<xref rid="bib49" ref-type="bibr"><sup>49</sup></xref></td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://jaspar.genereg.net/" ext-link-type="uri">https://jaspar.genereg.net/</ext-link></td></tr><tr><td colspan="3" rowspan="1"><hr/></td></tr><tr><td colspan="3" rowspan="1"><bold>Experimental models: Cell lines</bold></td></tr><tr><td colspan="3" rowspan="1"><hr/></td></tr><tr><td colspan="1" rowspan="1">293T</td><td colspan="1" rowspan="1">ATCC</td><td colspan="1" rowspan="1">Cat#CRL-3216, RRID:CVCL_0063</td></tr><tr><td colspan="1" rowspan="1">K562</td><td colspan="1" rowspan="1">ATCC</td><td colspan="1" rowspan="1">Cat#CCL-243, RRID:CVCL_0004</td></tr><tr><td colspan="1" rowspan="1">MCF-7</td><td colspan="1" rowspan="1">ATCC</td><td colspan="1" rowspan="1">Cat#HTB-22, RRID:CVCL_0031</td></tr><tr><td colspan="1" rowspan="1">IMR-90</td><td colspan="1" rowspan="1">ATCC</td><td colspan="1" rowspan="1">Cat#CCL-186, RRID:CVCL_0347</td></tr><tr><td colspan="3" rowspan="1"><hr/></td></tr><tr><td colspan="3" rowspan="1"><bold>Software and algorithms</bold></td></tr><tr><td colspan="3" rowspan="1"><hr/></td></tr><tr><td colspan="1" rowspan="1">Code associated with this manuscript</td><td colspan="1" rowspan="1">This paper</td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://doi.org/10.5281/zenodo.6098968" ext-link-type="uri">https://doi.org/10.5281/zenodo.6098968</ext-link></td></tr><tr><td colspan="1" rowspan="1">MPRAnator</td><td colspan="1" rowspan="1">MPRAnator et al.<xref rid="bib84" ref-type="bibr"><sup>84</sup></xref></td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://genomegeek.com/" ext-link-type="uri">https://genomegeek.com/</ext-link></td></tr><tr><td colspan="1" rowspan="1">BEDTools utilities v2.21.0</td><td colspan="1" rowspan="1">Quinlan et al.<xref rid="bib85" ref-type="bibr"><sup>85</sup></xref></td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://github.com/arq5x/bedtools2" ext-link-type="uri">https://github.com/arq5x/bedtools2</ext-link></td></tr><tr><td colspan="1" rowspan="1">BWA MEM</td><td colspan="1" rowspan="1">Li<xref rid="bib86" ref-type="bibr"><sup>86</sup></xref></td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="http://bio-bwa.sourceforge.net/" ext-link-type="uri">http://bio-bwa.sourceforge.net/</ext-link></td></tr><tr><td colspan="1" rowspan="1">FIMO</td><td colspan="1" rowspan="1">Grant et al.<xref rid="bib87" ref-type="bibr"><sup>87</sup></xref></td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://meme-suite.org/meme/doc/fimo.html" ext-link-type="uri">https://meme-suite.org/meme/doc/fimo.html</ext-link></td></tr><tr><td colspan="1" rowspan="1">MPRAflow</td><td colspan="1" rowspan="1">Gordon et al.<xref rid="bib53" ref-type="bibr"><sup>53</sup></xref></td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://github.com/shendurelab/MPRAflow" ext-link-type="uri">https://github.com/shendurelab/MPRAflow</ext-link></td></tr><tr><td colspan="1" rowspan="1">clusterProfiler</td><td colspan="1" rowspan="1">Yu et al.<xref rid="bib88" ref-type="bibr"><sup>88</sup></xref></td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://bioconductor.org/packages/release/bioc/html/clusterProfiler.html" ext-link-type="uri">https://bioconductor.org/packages/release/bioc/html/clusterProfiler.html</ext-link></td></tr><tr><td colspan="1" rowspan="1">TissueEnrich</td><td colspan="1" rowspan="1">Jain et al.<xref rid="bib89" ref-type="bibr"><sup>89</sup></xref></td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://bioconductor.org/packages/release/bioc/html/TissueEnrich.html" ext-link-type="uri">https://bioconductor.org/packages/release/bioc/html/TissueEnrich.html</ext-link></td></tr><tr><td colspan="1" rowspan="1">SciPy</td><td colspan="1" rowspan="1">Virtanen et al.<xref rid="bib90" ref-type="bibr"><sup>90</sup></xref></td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://scipy.org/" ext-link-type="uri">https://scipy.org/</ext-link></td></tr><tr><td colspan="1" rowspan="1">FluorEssence™</td><td colspan="1" rowspan="1">HORIBA</td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://www.horiba.com/int/products/detail/action/show/Product/fluoressence-1378/" ext-link-type="uri">https://www.horiba.com/int/products/detail/action/show/Product/fluoressence-1378/</ext-link></td></tr><tr><td colspan="1" rowspan="1">Spectra Manager™</td><td colspan="1" rowspan="1">JASCO</td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://jascoinc.com/products/spectroscopy/circular-dichroism/software/spectra-manager/" ext-link-type="uri">https://jascoinc.com/products/spectroscopy/circular-dichroism/software/spectra-manager/</ext-link></td></tr><tr><td colspan="1" rowspan="1">Cary WinUV Software</td><td colspan="1" rowspan="1">Agilent</td><td colspan="1" rowspan="1"><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://www.agilent.com/en/product/molecular-spectroscopy/uv-vis-uv-vis-nir-spectroscopy/uv-vis-uv-vis-nir-software/cary-winuv-software" ext-link-type="uri">https://www.agilent.com/en/product/molecular-spectroscopy/uv-vis-uv-vis-nir-spectroscopy/uv-vis-uv-vis-nir-software/cary-winuv-software</ext-link></td></tr></tbody></table></table-wrap>
</sec><sec id="sec4.2" disp-level="2"><title>Resource availability</title><sec id="sec4.2.1" disp-level="3"><title>Lead contact</title><p id="p0190">Further information and requests for resources should be directed to and will be fulfilled by the lead contact, Nadav Ahituv (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="http://mailto:nadav.ahituv@ucsf.edu" ext-link-type="uri">nadav.ahituv@ucsf.edu</ext-link>).</p></sec><sec id="sec4.2.2" disp-level="3"><title>Materials availability</title><p id="p0195">This study did not generate new unique reagents.</p></sec></sec><sec id="sec4.3" disp-level="2"><title>Experimental model and subject details</title><p id="p0200">Cell culturing was performed for HEK293T (RRID CVCL_0063), K562 (RRID CVCL_0004), MCF-7 (RRID: CVCL_0031) and IMR-90 (RRID: CVCL_0347) cell lines. Human HEK293T embryonic kidney cells were cultured in Dulbecco’s modified Eagle’s medium (DMEM, Sigma) supplemented with 10% FBS and 2mmol/ L L-glutamine. Human K562 erythroleukemia cells were cultured in Iscove’s modified Dulbecco’s medium (IMDM, Sigma-Aldrich) supplemented with 10% FBS. Human MCF-7 breast cancer cells were cultured in Eagle’s minimal essential medium (MEM, Sigma-Aldrich) supplemented with 10% FBS, 10μg/ml insulin, 1mM sodium pyruvate and 0.1 mM non-essential amino acids. Human IMR-90 fibroblasts were cultured in MEM supplemented with 10% FBS and 0.1 mM non-essential amino acids. Neural progenitor cells were differentiated from H1 hESCs following the dual-Smad inhibition protocol as described in <xref rid="bib91" ref-type="bibr"><sup>91</sup></xref>. All cell lines were grown at 37°C and 5% CO<sub>2</sub>.</p></sec><sec id="sec4.4" disp-level="2"><title>Method details</title><sec id="sec4.4.1" disp-level="3"><title>Genomic elements</title><p id="p0205">Gene annotation from Ensembl was followed throughout. Genic regions were separated into introns, coding exons, 5′ UTRs and 3′ UTRs, 1kb upstream of the TSS, 1kb downstream of the TES based on UCSC Table Browser using browser extensive data selection files. BEDTools utilities v2.21.0 were used to manipulate genomic files and intervals.<xref rid="bib85" ref-type="bibr"><sup>85</sup></xref></p></sec><sec id="sec4.4.2" disp-level="3"><title>Ensembl regulatory build</title><p id="p0210">Regulatory features were derived from the Ensembl regulatory build for twelve commonly used cell lines across human tissues, namely A549, HMEC, HUVEC, IMR-90, K562, HepG2, HSMM, MCF-7, NHEK, H1-ESC, GM12878 and HCT116.<xref rid="bib42" ref-type="bibr"><sup>42</sup></xref> The enrichment in <xref rid="fig2" ref-type="fig">Figures 2</xref>A and 2C were calculated from the median enrichment across the cell lines.</p></sec><sec id="sec4.4.3" disp-level="3"><title>Non-B DNA motif identification</title><p id="p0215">The genome-wide analysis of non-B DNA motifs was performed using the positions derived from <xref rid="bib82" ref-type="bibr"><sup>82</sup></xref>. Custom scripts were developed in Python to identify STRs, DRs, IRs, MRs, Z-DNA and G4s across the MPRA sequences. Consensus G4 motifs were derived using the regular expression ([gG]{3,}\w{1,7}){3,}[gG]{3,}. IR, DR and MRs with arm lengths of 10bp and spacer sequences of up to 4bps were identified, unless otherwise defined in the particular figure. Z-DNA sequences were defined as alternating purine-pyrimidine tracts of at least 10bp length. The subset of MRs that have high AG content (&gt;90%) and which are more likely to form H-DNA structures. Here, H-DNA motifs were defined as the subset of MRs that have a high (&gt;90%) AG content, arm lengths of &gt;=10bp and spacer size of less than 8bp. Custom scripts were developed in Python to identify the size and positions of the non-B DNA motif sub-components. For DR motif identification, the STR repeat threshold within the arm was set to 80%, in order to separate them from STR motifs. Enrichment of mutations at non-B DNA motifs was estimated as described in <xref rid="bib15" ref-type="bibr"><sup>15</sup></xref>.</p></sec><sec id="sec4.4.4" disp-level="3"><title>G4-seq and G4 ChIP-seq maps</title><p id="p0220">G4-seq BedGraph data were derived from GEO accession code <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="http://ncbi-geo:GSE63874" ext-link-type="uri">GSE63874</ext-link> for the human genome, in two conditions, PDS and K<sup>+</sup> treatments.<xref rid="bib45" ref-type="bibr"><sup>45</sup></xref> G4 ChIP-seq data were derived from GEO accession code <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="http://ncbi-geo:GSE107690" ext-link-type="uri">GSE107690</ext-link> for K562 cell line.<xref rid="bib83" ref-type="bibr"><sup>83</sup></xref></p><p id="p0225">G4 motifs were oriented as template and non-template based on their orientation relative to gene direction, across genic regions. Strand orientation of G4 motifs at G4-seq and G4 ChIP-seq peaks was performed by subsetting the strand of G4 motifs overlapping the peaks.</p></sec><sec id="sec4.4.5" disp-level="3"><title>Transcription factor binding site maps</title><p id="p0230">Position frequency matrices (PFMs) of transcription factors were derived from JASPAR (release 2020)<xref rid="bib49" ref-type="bibr"><sup>49</sup></xref> for the non-redundant CORE collection (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="http://jaspar.genereg.net/download/CORE/JASPAR2020_CORE_vertebrates_non-redundant_pfms_meme.zip" ext-link-type="uri">http://jaspar.genereg.net/download/CORE/JASPAR2020_CORE_vertebrates_non-redundant_pfms_meme.zip</ext-link>) and motif scanning was performed with FIMO.<xref rid="bib87" ref-type="bibr"><sup>87</sup></xref></p></sec><sec id="sec4.4.6" disp-level="3"><title>Population variant analysis</title><p id="p0235">Nucleotide variants, indels as well as structural variants were derived from the GnomAD project for whole genome sequenced datasets.<xref rid="bib40" ref-type="bibr"><sup>40</sup></xref> Only variants with the filter flag PASS were analyzed.</p></sec><sec id="sec4.4.7" disp-level="3"><title>eQTL analysis</title><p id="p0240">eQTLs were derived from the GTEx consortium<xref rid="bib44" ref-type="bibr"><sup>44</sup></xref> and analyzed with the commands “intersect” and “closest” from BEDTools to investigate their intersection and distribution patterns with motifs from each non-B DNA category as well as with G4-seq and G4 ChIP-seq peaks.</p></sec><sec id="sec4.4.8" disp-level="3"><title>Gene set enrichment analysis</title><p id="p0245">For each type of non-B DNA motif, we extracted a group of genes that contain a non-B DNA motif within a 200 bp upstream window from their TSS and these were used to perform gene set enrichment analyses. GO analyses were performed using clusterProfiler,<xref rid="bib88" ref-type="bibr"><sup>88</sup></xref> where GO terms with at least 20 genes and gene ratio greater than 0.01 for at least one of the non-B DNA sets were considered. For visualization purposes, we only displayed a maximum of 10 GO terms with the highest gene ratio per non-B DNA set. Finally, we calculated the enrichment of each non-B DNA group across sets of tissue-specific genes using TissueEnrich<xref rid="bib89" ref-type="bibr"><sup>89</sup></xref> using default arguments.</p></sec><sec id="sec4.4.9" disp-level="3"><title>Luciferase assay</title><p id="p0250">Candidate promoters of 200 bp were PCR amplified using AccuPrime™ GC-Rich DNA Polymerase (ThermoFisher Scientific 12337016) and cloned into the pLSmP-Luciferase vector after digestion with SbfI and AgeI restriction enzymes (remove minimal promoter). Primers with 20bp homology to the vector cloning site were designed and PCR products were assembled to the lentiviral vector using NEBuilder® HiFi DNA Assembly Master Mix (E2621S). Lentiviruses were produced using Lenti-Pac HIV Expression Packaging Kit (Genecopoeia, LT001) according to manufacturer's instructions. Small scale viral productions on HEK293T cells (2x800,000 cells seeded on a p6 well 24h prior to transfection; virus-containing culture media was collected 48h post-transfection and was used to infect desired cells) were performed of all different constructs to test luciferase activity in four different cell lines (MCF-7, IMR-90, K562, HEK293T). 50,000 cells were seeded on 96-well plates in a volume of 50 μL and another 50 μL of virus-containing medium was added in order to transduce them. Luminescence was measured 24h or 48h post-infection using Dual-Luciferase® Reporter Assay System (Promega, E1910).</p></sec><sec id="sec4.4.10" disp-level="3"><title>lentiMPRA of promoters</title><p id="p0255">Each of the sequences was synthesized on a 7,500-feature microarray (Agilent OLS; 15 bp primer + 200 bp promoter + 15 bp primer = 230 mers). For the G4s that we studied across these genes, all selected loci overlapped G4-seq or G4 ChIP-seq peaks. lentiMPRA was performed as described previously with modifications.<xref rid="bib53" ref-type="bibr"><sup>53</sup></xref></p><p id="p0260">In brief, PCR amplification of OLS library was performed using NEBNext® High-Fidelity 2X PCR Master Mix (New England Biolabs, M0541S)(4x50 μl reactions using 20 ng of template OLS library and primers L1.Amp.F and L1.Amp.R; PCR program: 95°C, 2 min; (95°C, 15 sec; 65°C, 20 sec; 72°C, 1 min) x12 cycles; 72°C, 5min). Barcodes were added by PCR in the library amplification step in the 5′ UTR of the GFP gene. This PCR 5′-tagging strategy allowed us to eliminate the confounding effect that lentiviral genome recombination might have on 3′-tagged libraries. Additionally, tagging barcodes in the PCR amplification step via primers harboring degenerate nucleotides enabled us to assay larger promoter sequences (200 bp instead of 171 bp of previous MPRA designs) and the cost-effective use of an oligonucleotide library 100 times smaller in size to obtain 100 barcodes per promoter (we ordered 7,500 different sequences instead of 750,000). 20 μg of lentiviral vector (pLSmP-GFP) were digested with <italic>Sbf</italic>I and <italic>Age</italic>I restriction enzymes. Linearized vector and PCR products were run on a 1% agarose gel and purified using QIAquick gel extraction kit (QIAGEN, 28704). 5x20 μl ligations containing 1:10 molar ratio between vector and inserts were performed using NEBuilder® HiFi DNA Assembly Master Mix (E2621S). Ligations were pooled and purified using MinElute Reaction Cleanup Kit (QIAGEN 28204) and electroporated into ElectroMAX™ Stbl4™ Competent Cells (ThermoFisher Scientific 11635018). 50 μl of electrocompetent bacteria and 60 ng of DNA were used per reaction in a 0.1 cm cuvette (Program: 1.2kV; 200 ohms; 25 μF; 1 pulse). 1:1,000 and 1:10,000 dilutions were seeded on LB plates with ampicillin in order to estimate the number of clones. Approximately 800,000 different clones were obtained and, thus, the complexity of the plasmid library with an estimated of 100 barcodes per insert. Insert-barcode fragment was amplified from the plasmid library and sequenced using NextSeq PE150 for the insert-barcode association.</p><p id="p0265">Lentiviral particles were produced from the plasmid library as in the luciferase assay but scaling the process to 6x150 mm plates. In summary, 6x10<sup>6</sup> HEK293T cells were seeded per plate 48 hour before transfection, 5 μg of plasmid library and 5 μg of HIV packaging mix were co-transfected using 30 μl of EndoFectin. Media were collected 48h post-transfection and lentiviral particles were concentrated using Lenti-X™ Concentrator (Takara 631231). Lentiviral library was tested in a small scale experiment with HEK293T and K562 cell lines in order to titrate the number of desired integrations. Three million HEK293T and 4 million K562 cells were infected with the library with the multiplicity of infection (MOI) of 400 and 40, respectively, as calculated in small scale titration experiments. In order to improve infection, polybrene was added together with the lentiviral library at a final concentration of 8 μg/ml. After three days of culture, barcoded DNA and RNA were extracted from the cells using Allprep DNA/RNA mini kit (QIAGEN 80204). mRNA was purified using Oligotex mRNA mini kit (QIAGEN 70022), and reverse-transcribed using SuperScriptII (Life Technologies, 18064-071), according to manufacturer's instructions. Barcodes were amplified and sequenced using NextSeq PE15, as described previously.<xref rid="bib53" ref-type="bibr"><sup>53</sup></xref> We performed three independent replicates of infection for each cell line.</p></sec><sec id="sec4.4.11" disp-level="3"><title>MPRA analysis pipeline</title><p id="p0270">The design of MPRA sequences was performed with algorithms adjusted from <xref rid="bib84" ref-type="bibr"><sup>84</sup></xref>. For barcode insert mapping and filtering, we called a consensus sequence from the paired-end reads associating with barcode sequence from the index read. We aligned all consensus sequences back to all designed sequences (inserts) using BWA MEM (version 0.7.17-r1188).<xref rid="bib86" ref-type="bibr"><sup>86</sup></xref> As many of the designed sequences are either only 1bp mutation from each other, or the inverted orientation, we use CIGAR string with perfect sequence match and 0 mismatches as a strict filter. For RNA/DNA barcode counting and ratio normalization, RNA and DNA barcodes for each of three replicates were sequenced on an Illumina NextSeq instrument, UMI is used to remove PCR duplicates and the inserts with associated barcode counts lower than 3 are removed. Evaluating the effect of GC-content in the contribution to expression across the MPRA was performed by fitting a linear model and subtracting from each sequence the expected score due to GC-content.</p></sec><sec id="sec4.4.12" disp-level="3"><title>NMM ligand enhanced fluorescence</title><p id="p0275">Experiments were carried out as previously reported with slight modification.<xref rid="bib92" ref-type="bibr"><sup>92</sup></xref> Sample solutions of 100 μL total volume were prepared containing 1 μM DNA, 10 mM lithium cacodylate (LiCac) buffer (pH 7.0), 150 mM LiCl or KCl solution and 1 μM NMM ligand. HORIBA FluoroMax-4 Fluorometer was used to measure the fluorescence spectra. Before sample measurement, samples were first prepared without ligand and heated for denaturation at 95°C for 3 minutes followed by cooling down for 15 minutes by placing the sample solution at room temperature so as to undergo renaturation. The samples were then transferred into a quartz cuvette which had a path length of 1-cm and excited at 394 nm. The range from 550 to 750 nm of emission spectra were needed. All data were measured at 25°C in every 2 nm and the exit and entrance slit widths were 5 nm. The enhanced fluorescence spectra of samples in the absence of ligand were used for normalization. All of the above calculations were analyzed in Microsoft Excel.</p></sec><sec id="sec4.4.13" disp-level="3"><title>Circular dichroism (CD) spectroscopy</title><p id="p0280">Experiments were carried out as previously reported with slight modification.<xref rid="bib93" ref-type="bibr"><sup>93</sup></xref> Jasco J-1500 CD spectrophotometer was used to carry out the CD spectroscopy. A total of 2 mL sample solution was contained with a quartz cuvette which had a path length of 1-cm. Sample reactions consisting of 5 μM DNA, 150 mM KCl or LiCl and 10 mM LiCac (pH 7.0) were prepared. Then mixed thoroughly and denatured the sample solution for 5 minutes at 95°C and then incubated for 15 minutes by placing the sample solution at room temperature to undergo renaturation. All samples were measured at 25°C in a range from 220 to 310 nm. The spectra were needed every 1 nm. The time for responding was 0.5 s/nm and all of the spectra stated were 2 scans in average. By normalizing the data collected, the molar residue ellipticity was obtained and then smoothed over 5 nm. Spectra Manager Suite (Jasco Software) was used to analyze the collected data.</p></sec><sec id="sec4.4.14" disp-level="3"><title>Thermal melting monitored by UV spectroscopy</title><p id="p0285">Experiments were carried out as previously reported with slight modification.<xref rid="bib93" ref-type="bibr"><sup>93</sup></xref> Sample reactions of 2 mL consisting of 5 μM DNA (except for concentration dependent melting that ranged from 1 – 10 μM), 150 mM KCl and 10 mM LiCac (pH 7.0) were prepared. Samples were then mixed completely and heated for 3 minutes at 95°C for DNA denaturation and followed by renaturation for 15 minutes by placing the sample solution at room temperature. Samples were then transferred into a quartz cuvette which had a path length of 1-cm then sealed with 2 layers of Teflon tape in order to lower the chance of evaporation of the sample when the measurement reached high temperature. Measurements were conducted using Agilent Cary 100 UV-Vis Spectrophotometer with sample block initially set at 20°C for 5 minutes.</p><p id="p0290">The samples were measured from 20 to 95°C (forward scan) with a 0.5°C/min temperature increment rate. There was a reverse scan measurement (95 to 20°C) that also had a 0.5°C/min increment rate after holding for 5 minutes at 95°C. At 295 nm (or 260nm for the B-DNA oligonucleotide), both of the forward and reverse scans were recorded for the folding and unfolding transitions.</p><p id="p0295">The collected data were deducted by the blanked solutions which had the identical concentrations of the KCl and LiCac buffer (pH 7.0) only. The data’s first derivatives were obtained by smoothing the data over 11 nm where all the processes and results were marked in Microsoft Excel. By taking average of the melting temperatures in both of the reversed and forward measurements, the final melting temperature was determined.</p></sec><sec id="sec4.4.15" disp-level="3"><title>Intrinsic fluorescence spectroscopy</title><p id="p0300">Experiments were carried out as previously reported with slight modification.<xref rid="bib93" ref-type="bibr"><sup>93</sup></xref> Samples were prepared as done for UV-melting and CD-spectroscopy. HORIBA FluoroMax-4 Fluorometer was used to measure the intrinsic fluorescence spectra. After denaturation and renaturation of samples, the samples were transferred into a quartz cuvette which had a path length of 1-cm and excited at 260 nm. The range from 300 to 500 nm of the emission spectra were needed. All data were measured at 25°C of every 2 nm and the exit and entrance slit widths were 5 nm. The collected data were smoothed over 5 nm using Microsoft Excel.</p></sec></sec><sec id="sec4.5" disp-level="2"><title>Quantification and statistical analysis</title><sec id="sec4.5.1" disp-level="3"><title>Population variant analysis</title><p id="p0305">For SNP variants, simulated controls were generated within 10kb from the original variant, controlling for trinucleotide content. To achieve this, the base-pair at the randomly selected simulated position, within 10kb from the original mutation, and both the 5’ and the 3’ adjacent base-pairs had to match those at the mutated sites, and the mutation and simulation sites had to be different from one another. In addition, in the simulations regions of the human genome for which mutation calling by GnomAD was not performed were excluded. For indels, we generated simulated indels within 10kb of the original indel site, correcting for indel length and with local GC content at a 100bp window each side of the indel site within 2.5% difference from the original. For structural variants, we simulated an equal number of breakpoints at random locations within 10kb of the original breakpoints, correcting for local GC content, with 2.5% maximum difference from the original GC content. Statistical significance was estimated with non-parametric Mann-Whitney U tests in Python using the SciPy library.<xref rid="bib90" ref-type="bibr"><sup>90</sup></xref> Across regulatory elements, z-scores were calculated from the density of mutations at non-B DNA motifs at that element, relative to the mean mutational density at that element, divided by the standard deviation.</p></sec><sec id="sec4.5.2" disp-level="3"><title>Transcription factor binding</title><p id="p0310">PFMs were used to identify transcription factor binding sites with FIMO,<xref rid="bib87" ref-type="bibr"><sup>87</sup></xref> which was used with background model the nucleotide frequencies across the human genome and requiring a minimum p-value &lt;10−6.</p></sec><sec id="sec4.5.3" disp-level="3"><title>MPRA analysis</title><p id="p0315">Statistical significance of expression difference between sequences with and without a non-B DNA motif was estimated with Mann-Whitney U tests with Bonferroni correction in Python using the SciPy library.<xref rid="bib90" ref-type="bibr"><sup>90</sup></xref></p></sec></sec></sec><sec id="ack0010" sec-type="ack" disp-level="1"><title>Acknowledgments</title><p id="p0320">This work is supported by the National Human Genome Research Institute (1UM1HG009408, R01HG010333, 1R21HG010065, UM1HG011966, and 1R21HG010683 to N.A.), the National Institute of Mental Health (1R01MH109907 and 1U01MH116438 to N.A.), and the National Heart, Lung, and Blood Institute (R35HL145235 to N.A.). G.E.P. and M.H. were supported by a core grant from the Wellcome Trust. Funding for open access charge was provided by NHGRI. The C.K.K. lab is supported by the Shenzhen Basic Research Project (JCYJ20180507181642811); the Research Grants Council of the Hong Kong SAR, China Projects (CityU 11100421, CityU 11101519, CityU 11100218, and N_CityU110/17); the Croucher Foundation Project (9509003); the State Key Laboratory of Marine Pollution Director Discretionary Fund; City University of Hong Kong projects (7005503, 9667222, and 9680261) to C.K.K.; and National Research Foundation of Korea grant 2020R1C1C1003426 (to J.-Y.A.). O.E. is supported by a California Institute of Regenerative Medicine fellowship (SFSU EDUC2-08391). The sequencing was carried by the DNA Technologies and Expression Analysis Core at the UC Davis Genome Center, supported by NIH Shared Instrumentation grant 1S10OD010786-01.</p><sec id="sec5" disp-level="2"><title>Author contributions</title><p id="p0325">I.G.-S., M.H., and N.A. conceived the study. I.G.-S., G.E.P., V.A., A.M., and J.Z. wrote the code and performed the analyses. I.G.-S., G.E.P., J.Z., and J.V. generated the visualizations. J.V., O.E., and F.I. performed the MPRA experiments. H.Y.W. and M.I.U. performed the circular dichroism (CD) titration, UV melting, and fluorescence assays, and H.Y.W., M.I.U., and C.K.K. analyzed and interpreted the spectroscopic data. J.-Y.A. and S.J.S worked on the NPC MPRA design. M.H. and N.A. supervised the research. I.G.-S., M.H., and N.A. wrote the manuscript with input from all authors.</p></sec><sec id="sec6" disp-level="2"><title>Declaration of interests</title><p id="p0330">The authors declare no competing interests.</p></sec></sec><sec id="notes1" disp-level="1"><p id="misc0010">Published: March 15, 2022</p></sec><sec id="fn-group1" sec-type="fn-group" disp-level="1"><title>Footnotes</title><fn-group><fn id="appsec1"><p id="p0335">Supplemental information can be found online at <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://doi.org/10.1016/j.xgen.2022.100111" ext-link-type="uri">https://doi.org/10.1016/j.xgen.2022.100111</ext-link>.</p></fn></fn-group></sec><sec id="_ci93_" xml:lang="en" sec-type="contrib-info" disp-level="1"><title>Contributor Information</title><p>Martin Hemberg, Email: mhemberg@bwh.harvard.edu.</p><p>Nadav Ahituv, Email: nadav.ahituv@ucsf.edu.</p></sec><sec id="appsec2" disp-level="1"><title>Supplemental information</title>
<supplementary-material id="mmc1" position="float"><?disp-level 2?><caption><title>Document S1. Figures S1–S15 and Tables S1 and S2</title></caption><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="mmc1.pdf" mimetype="application" mime-subtype="pdf"><?cloudpmc-path 393b/9903758/09dd25e70914/mmc1.pdf?><?cloudpmc-bucket app?><?size 3262927?></media></supplementary-material>
<supplementary-material id="mmc2" position="float"><?disp-level 2?><caption><title>Document S2. Transparent peer review</title><p>for Georgakopoulos-Soares et al.</p></caption><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="mmc2.pdf" mimetype="application" mime-subtype="pdf"><?cloudpmc-path 393b/9903758/602219a9ba2e/mmc2.pdf?><?cloudpmc-bucket app?><?size 664694?></media></supplementary-material>
<supplementary-material id="mmc3" position="float"><?disp-level 2?><caption><title>Document S3. Article plus supplemental information</title></caption><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="mmc3.pdf" mimetype="application" mime-subtype="pdf"><?cloudpmc-path 393b/9903758/8fcf4e39ab25/mmc3.pdf?><?cloudpmc-bucket app?><?size 10511800?></media></supplementary-material>
</sec><sec id="da0010" disp-level="1"><title>Data and code availability</title><p id="p0030">The MPRA data for the NPC cell line targeting autism-related loci and the MPRA data for the non-B DNA associated loci in HEK-293T and K562 cell lines are deposited in NCBI BioProject with accession number PRJNA763774. The MPRA data for HEPG2 and K562 cell lines (<xref rid="fig3" ref-type="fig">Figure 3</xref>) have been deposited in the ENCODE portal with IDs ENCSR463IRX and ENCSR460LZI.</p><p id="p0035">All original code and data tables to perform the analyses can be found on the GitHub page (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://github.com/IliasGeoSo/High_Throughput_MPRA_Non_B_DNA" ext-link-type="uri">https://github.com/IliasGeoSo/High_Throughput_MPRA_Non_B_DNA</ext-link>) and are publicly available.</p><p id="p0040">Any additional information required to reanalyze the data reported in this paper is available from the lead contact upon request.</p></sec><sec id="cebib0010" sec-type="ref-list" disp-level="1"><title>References</title><sec id="cebib0010_sec2" disp-level="2"><ref-list><ref id="bib1"><label>1.</label><mixed-citation id="sref1"><named-content content-type="citation-string">Ghosh A., Bansal M. A glossary of DNA structures from A to Z. Acta Crystallogr. D Biol. Crystallogr. 2003;59:620–626. doi: 10.1107/s0907444903003251.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1107/s0907444903003251"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="12657780"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Acta Crystallogr. D Biol. Crystallogr.&amp;title=A glossary of DNA structures from A to Z&amp;author=A. Ghosh&amp;author=M. Bansal&amp;volume=59&amp;publication_year=2003&amp;pages=620-626&amp;pmid=12657780&amp;doi=10.1107/s0907444903003251&amp;"/></mixed-citation></ref><ref id="bib2"><label>2.</label><mixed-citation id="sref2"><named-content content-type="citation-string">Nag D.K., Petes T.D. Seven-base-pair inverted repeats in DNA form stable hairpins in vivo in Saccharomyces cerevisiae. Genetics. 1991;129:669–673. doi: 10.1093/genetics/129.3.669.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/genetics/129.3.669"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC1204734"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="1752412"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Genetics&amp;title=Seven-base-pair inverted repeats in DNA form stable hairpins in vivo in Saccharomyces cerevisiae&amp;author=D.K. Nag&amp;author=T.D. Petes&amp;volume=129&amp;publication_year=1991&amp;pages=669-673&amp;pmid=1752412&amp;doi=10.1093/genetics/129.3.669&amp;"/></mixed-citation></ref><ref id="bib3"><label>3.</label><mixed-citation id="sref3"><named-content content-type="citation-string">Leach D.R. Long DNA palindromes, cruciform structures, genetic instability and secondary structure repair. Bioessays. 1994;16:893–900. doi: 10.1002/bies.950161207.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1002/bies.950161207"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="7840768"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Bioessays&amp;title=Long DNA palindromes, cruciform structures, genetic instability and secondary structure repair&amp;author=D.R. Leach&amp;volume=16&amp;publication_year=1994&amp;pages=893-900&amp;pmid=7840768&amp;doi=10.1002/bies.950161207&amp;"/></mixed-citation></ref><ref id="bib4"><label>4.</label><mixed-citation id="sref4"><named-content content-type="citation-string">Lobachev K.S., Shor B.M., Tran H.T., Taylor W., Keen J.D., Resnick M.A., Gordenin D.A. Factors affecting inverted repeat stimulation of recombination and deletion in Saccharomyces cerevisiae. Genetics. 1998;148:1507–1524. doi: 10.1093/genetics/148.4.1507.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/genetics/148.4.1507"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC1460095"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="9560370"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Genetics&amp;title=Factors affecting inverted repeat stimulation of recombination and deletion in Saccharomyces cerevisiae&amp;author=K.S. Lobachev&amp;author=B.M. Shor&amp;author=H.T. Tran&amp;author=W. Taylor&amp;author=J.D. Keen&amp;volume=148&amp;publication_year=1998&amp;pages=1507-1524&amp;pmid=9560370&amp;doi=10.1093/genetics/148.4.1507&amp;"/></mixed-citation></ref><ref id="bib5"><label>5.</label><mixed-citation id="sref5"><named-content content-type="citation-string">Tippana R., Xiao W., Myong S. G-quadruplex conformation and dynamics are determined by loop length and sequence. Nucleic Acids Res. 2014;42:8106–8114. doi: 10.1093/nar/gku464.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gku464"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC4081081"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="24920827"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res.&amp;title=G-quadruplex conformation and dynamics are determined by loop length and sequence&amp;author=R. Tippana&amp;author=W. Xiao&amp;author=S. Myong&amp;volume=42&amp;publication_year=2014&amp;pages=8106-8114&amp;pmid=24920827&amp;doi=10.1093/nar/gku464&amp;"/></mixed-citation></ref><ref id="bib6"><label>6.</label><mixed-citation id="sref6"><named-content content-type="citation-string">Pannunzio N.R., Lieber M.R. Concept of DNA lesion longevity and chromosomal translocations. Trends Biochem. Sci. 2018;43:490–498. doi: 10.1016/j.tibs.2018.04.004.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.tibs.2018.04.004"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6014902"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="29735400"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Trends Biochem. Sci.&amp;title=Concept of DNA lesion longevity and chromosomal translocations&amp;author=N.R. Pannunzio&amp;author=M.R. Lieber&amp;volume=43&amp;publication_year=2018&amp;pages=490-498&amp;pmid=29735400&amp;doi=10.1016/j.tibs.2018.04.004&amp;"/></mixed-citation></ref><ref id="bib7"><label>7.</label><mixed-citation id="sref7"><named-content content-type="citation-string">Gonzalez-Perez A., Sabarinathan R., Lopez-Bigas N. Local determinants of the mutational landscape of the human genome. Cell. 2019;177:101–114. doi: 10.1016/j.cell.2019.02.051.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.cell.2019.02.051"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="30901533"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Cell&amp;title=Local determinants of the mutational landscape of the human genome&amp;author=A. Gonzalez-Perez&amp;author=R. Sabarinathan&amp;author=N. Lopez-Bigas&amp;volume=177&amp;publication_year=2019&amp;pages=101-114&amp;pmid=30901533&amp;doi=10.1016/j.cell.2019.02.051&amp;"/></mixed-citation></ref><ref id="bib8"><label>8.</label><mixed-citation id="sref8"><named-content content-type="citation-string">Du X., Gertz E.M., Wojtowicz D., Zhabinskaya D., Levens D., Benham C.J., Schäffer A.A., Przytycka T.M. Potential non-B DNA regions in the human genome are associated with higher rates of nucleotide mutation and expression variation. Nucleic Acids Res. 2014;42:12367–12379. doi: 10.1093/nar/gku921.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gku921"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC4227770"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="25336616"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res.&amp;title=Potential non-B DNA regions in the human genome are associated with higher rates of nucleotide mutation and expression variation&amp;author=X. Du&amp;author=E.M. Gertz&amp;author=D. Wojtowicz&amp;author=D. Zhabinskaya&amp;author=D. Levens&amp;volume=42&amp;publication_year=2014&amp;pages=12367-12379&amp;pmid=25336616&amp;doi=10.1093/nar/gku921&amp;"/></mixed-citation></ref><ref id="bib9"><label>9.</label><mixed-citation id="sref9"><named-content content-type="citation-string">Guiblet W.M., Cremona M.A., Harris R.S., Chen D., Eckert K.A., Chiaromonte F., Huang Y.-F., Makova K.D. Non-B DNA: a major contributor to small- and large-scale variation in nucleotide substitution frequencies across the genome. Nucleic Acids Res. 2021 doi: 10.1093/nar/gkaa1269.9.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gkaa1269.9"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC7897504"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="33450015"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res.&amp;title=Non-B DNA: a major contributor to small- and large-scale variation in nucleotide substitution frequencies across the genome&amp;author=W.M. Guiblet&amp;author=M.A. Cremona&amp;author=R.S. Harris&amp;author=D. Chen&amp;author=K.A. Eckert&amp;publication_year=2021&amp;pmid=33450015&amp;doi=10.1093/nar/gkaa1269.9&amp;"/></mixed-citation></ref><ref id="bib10"><label>10.</label><mixed-citation id="sref10"><named-content content-type="citation-string">Wang G., Vasquez K.M. Non-B DNA structure-induced genetic instability. Mutat. Res. 2006;598:103–119. doi: 10.1016/j.mrfmmm.2006.01.019.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.mrfmmm.2006.01.019"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="16516932"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Mutat. Res.&amp;title=Non-B DNA structure-induced genetic instability&amp;author=G. Wang&amp;author=K.M. Vasquez&amp;volume=598&amp;publication_year=2006&amp;pages=103-119&amp;pmid=16516932&amp;doi=10.1016/j.mrfmmm.2006.01.019&amp;"/></mixed-citation></ref><ref id="bib11"><label>11.</label><mixed-citation id="sref11"><named-content content-type="citation-string">Wang G., Christensen L.A., Vasquez K.M. Z-DNA-forming sequences generate large-scale deletions in mammalian cells. Proc. Natl. Acad. Sci. U S A. 2006;103:2677–2682. doi: 10.1073/pnas.0511084103.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1073/pnas.0511084103"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC1413824"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="16473937"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Proc. Natl. Acad. Sci. U S A&amp;title=Z-DNA-forming sequences generate large-scale deletions in mammalian cells&amp;author=G. Wang&amp;author=L.A. Christensen&amp;author=K.M. Vasquez&amp;volume=103&amp;publication_year=2006&amp;pages=2677-2682&amp;pmid=16473937&amp;doi=10.1073/pnas.0511084103&amp;"/></mixed-citation></ref><ref id="bib12"><label>12.</label><mixed-citation id="sref12"><named-content content-type="citation-string">Lu S., Wang G., Bacolla A., Zhao J., Spitser S., Vasquez K.M. Short inverted repeats are hotspots for genetic instability: relevance to cancer genomes. Cell Rep. 2015;10:1674–1680. doi: 10.1016/j.celrep.2015.02.039.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.celrep.2015.02.039"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6013304"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="25772355"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Cell Rep.&amp;title=Short inverted repeats are hotspots for genetic instability: relevance to cancer genomes&amp;author=S. Lu&amp;author=G. Wang&amp;author=A. Bacolla&amp;author=J. Zhao&amp;author=S. Spitser&amp;volume=10&amp;publication_year=2015&amp;pages=1674-1680&amp;pmid=25772355&amp;doi=10.1016/j.celrep.2015.02.039&amp;"/></mixed-citation></ref><ref id="bib13"><label>13.</label><mixed-citation id="sref13"><named-content content-type="citation-string">Bacolla A., Tainer J.A., Vasquez K.M., Cooper D.N. Translocation and deletion breakpoints in cancer genomes are associated with potential non-B DNA-forming sequences. Nucleic Acids Res. 2016;44:5673–5688. doi: 10.1093/nar/gkw261.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gkw261"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC4937311"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="27084947"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res.&amp;title=Translocation and deletion breakpoints in cancer genomes are associated with potential non-B DNA-forming sequences&amp;author=A. Bacolla&amp;author=J.A. Tainer&amp;author=K.M. Vasquez&amp;author=D.N. Cooper&amp;volume=44&amp;publication_year=2016&amp;pages=5673-5688&amp;pmid=27084947&amp;doi=10.1093/nar/gkw261&amp;"/></mixed-citation></ref><ref id="bib14"><label>14.</label><mixed-citation id="sref14"><named-content content-type="citation-string">Kamat M.A., Bacolla A., Cooper D.N., Chuzhanova N. A role for non-B DNA forming sequences in mediating microlesions causing human inherited disease. Hum. Mutat. 2016;37:65–73. doi: 10.1002/humu.22917.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1002/humu.22917"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="26466920"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Hum. Mutat.&amp;title=A role for non-B DNA forming sequences in mediating microlesions causing human inherited disease&amp;author=M.A. Kamat&amp;author=A. Bacolla&amp;author=D.N. Cooper&amp;author=N. Chuzhanova&amp;volume=37&amp;publication_year=2016&amp;pages=65-73&amp;pmid=26466920&amp;doi=10.1002/humu.22917&amp;"/></mixed-citation></ref><ref id="bib15"><label>15.</label><mixed-citation id="sref15"><named-content content-type="citation-string">Georgakopoulos-Soares I., Morganella S., Jain N., Hemberg M., Nik-Zainal S. Noncanonical secondary structures arising from non-B DNA motifs are determinants of mutagenesis. Genome Res. 2018;28:1264–1271. doi: 10.1101/gr.231688.117.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1101/gr.231688.117"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6120622"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="30104284"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Genome Res.&amp;title=Noncanonical secondary structures arising from non-B DNA motifs are determinants of mutagenesis&amp;author=I. Georgakopoulos-Soares&amp;author=S. Morganella&amp;author=N. Jain&amp;author=M. Hemberg&amp;author=S. Nik-Zainal&amp;volume=28&amp;publication_year=2018&amp;pages=1264-1271&amp;pmid=30104284&amp;doi=10.1101/gr.231688.117&amp;"/></mixed-citation></ref><ref id="bib16"><label>16.</label><mixed-citation id="sref16"><named-content content-type="citation-string">Hänsel-Hertsch R., Beraldi D., Lensing S.V., Marsico G., Zyner K., Parry A., Di Antonio M., Pike J., Kimura H., Narita M., et al.  G-quadruplex structures mark human regulatory chromatin. Nat. Genet. 2016;48:1267–1272. doi: 10.1038/ng.3662.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/ng.3662"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="27618450"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat. Genet.&amp;title=G-quadruplex structures mark human regulatory chromatin&amp;author=R. Hänsel-Hertsch&amp;author=D. Beraldi&amp;author=S.V. Lensing&amp;author=G. Marsico&amp;author=K. Zyner&amp;volume=48&amp;publication_year=2016&amp;pages=1267-1272&amp;pmid=27618450&amp;doi=10.1038/ng.3662&amp;"/></mixed-citation></ref><ref id="bib17"><label>17.</label><mixed-citation id="sref17"><named-content content-type="citation-string">Bogard N., Linder J., Rosenberg A.B., Seelig G. A deep neural network for predicting and engineering alternative polyadenylation. Cell. 2019;178:91–106.e23. doi: 10.1016/j.cell.2019.04.046.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.cell.2019.04.046"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6599575"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="31178116"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Cell&amp;title=A deep neural network for predicting and engineering alternative polyadenylation&amp;author=N. Bogard&amp;author=J. Linder&amp;author=A.B. Rosenberg&amp;author=G. Seelig&amp;volume=178&amp;publication_year=2019&amp;pages=91-106.e23&amp;pmid=31178116&amp;doi=10.1016/j.cell.2019.04.046&amp;"/></mixed-citation></ref><ref id="bib18"><label>18.</label><mixed-citation id="sref18"><named-content content-type="citation-string">Shin S.-I., Ham S., Park J., Seo S.H., Lim C.H., Jeon H., Huh J., Roh T.-Y. Z-DNA-forming sites identified by ChIP-Seq are associated with actively transcribed regions in the human genome. DNA Res. 2016;23:477–486. doi: 10.1093/dnares/dsw031.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/dnares/dsw031"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC5066173"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="27374614"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=DNA Res.&amp;title=Z-DNA-forming sites identified by ChIP-Seq are associated with actively transcribed regions in the human genome&amp;author=S.-I. Shin&amp;author=S. Ham&amp;author=J. Park&amp;author=S.H. Seo&amp;author=C.H. Lim&amp;volume=23&amp;publication_year=2016&amp;pages=477-486&amp;pmid=27374614&amp;doi=10.1093/dnares/dsw031&amp;"/></mixed-citation></ref><ref id="bib19"><label>19.</label><mixed-citation id="sref19"><named-content content-type="citation-string">Gymrek M., Willems T., Guilmatre A., Zeng H., Markus B., Georgiev S., Daly M.J., Price A.L., Pritchard J.K., Sharp A.J., et al.  Abundant contribution of short tandem repeats to gene expression variation in humans. Nat. Genet. 2016;48:22–29. doi: 10.1038/ng.3461.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/ng.3461"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC4909355"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="26642241"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat. Genet.&amp;title=Abundant contribution of short tandem repeats to gene expression variation in humans&amp;author=M. Gymrek&amp;author=T. Willems&amp;author=A. Guilmatre&amp;author=H. Zeng&amp;author=B. Markus&amp;volume=48&amp;publication_year=2016&amp;pages=22-29&amp;pmid=26642241&amp;doi=10.1038/ng.3461&amp;"/></mixed-citation></ref><ref id="bib20"><label>20.</label><mixed-citation id="sref20"><named-content content-type="citation-string">Bacolla A., Wells R.D. Non-B DNA conformations, genomic rearrangements, and human disease. J. Biol. Chem. 2004;279:47411–47414. doi: 10.1074/jbc.R400028200.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1074/jbc.R400028200"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="15326170"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=J. Biol. Chem.&amp;title=Non-B DNA conformations, genomic rearrangements, and human disease&amp;author=A. Bacolla&amp;author=R.D. Wells&amp;volume=279&amp;publication_year=2004&amp;pages=47411-47414&amp;pmid=15326170&amp;doi=10.1074/jbc.R400028200&amp;"/></mixed-citation></ref><ref id="bib21"><label>21.</label><mixed-citation id="sref21"><named-content content-type="citation-string">Wells R.D. Non-B DNA conformations, mutagenesis and disease. Trends Biochem. Sci. 2007;32:271–278. doi: 10.1016/j.tibs.2007.04.003.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.tibs.2007.04.003"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="17493823"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Trends Biochem. Sci.&amp;title=Non-B DNA conformations, mutagenesis and disease&amp;author=R.D. Wells&amp;volume=32&amp;publication_year=2007&amp;pages=271-278&amp;pmid=17493823&amp;doi=10.1016/j.tibs.2007.04.003&amp;"/></mixed-citation></ref><ref id="bib22"><label>22.</label><mixed-citation id="sref22"><named-content content-type="citation-string">Bacolla A., Wells R.D. Non-B DNA conformations as determinants of mutagenesis and human disease. Mol. Carcinog. 2009;48:273–285. doi: 10.1002/mc.20507.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1002/mc.20507"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="19306308"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Mol. Carcinog.&amp;title=Non-B DNA conformations as determinants of mutagenesis and human disease&amp;author=A. Bacolla&amp;author=R.D. Wells&amp;volume=48&amp;publication_year=2009&amp;pages=273-285&amp;pmid=19306308&amp;doi=10.1002/mc.20507&amp;"/></mixed-citation></ref><ref id="bib23"><label>23.</label><mixed-citation id="sref23"><named-content content-type="citation-string">Xie K.T., Wang G., Thompson A.C., Wucherpfennig J.I., Reimchen T.E., MacColl A.D.C., Schluter D., Bell M.A., Vasquez K.M., Kingsley D.M. DNA fragility in the parallel evolution of pelvic reduction in stickleback fish. Science. 2019;363:81–84. doi: 10.1126/science.aan1425.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1126/science.aan1425"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6677656"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="30606845"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Science&amp;title=DNA fragility in the parallel evolution of pelvic reduction in stickleback fish&amp;author=K.T. Xie&amp;author=G. Wang&amp;author=A.C. Thompson&amp;author=J.I. Wucherpfennig&amp;author=T.E. Reimchen&amp;volume=363&amp;publication_year=2019&amp;pages=81-84&amp;pmid=30606845&amp;doi=10.1126/science.aan1425&amp;"/></mixed-citation></ref><ref id="bib24"><label>24.</label><mixed-citation id="sref24"><named-content content-type="citation-string">Buisson R., Langenbucher A., Bowen D., Kwan E.E., Benes C.H., Zou L., Lawrence M.S. Passenger hotspot mutations in cancer driven by APOBEC3A and mesoscale genomic features. Science. 2019;364 doi: 10.1126/science.aaw2872.24.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1126/science.aaw2872.24"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6731024"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="31249028"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Science&amp;title=Passenger hotspot mutations in cancer driven by APOBEC3A and mesoscale genomic features&amp;author=R. Buisson&amp;author=A. Langenbucher&amp;author=D. Bowen&amp;author=E.E. Kwan&amp;author=C.H. Benes&amp;volume=364&amp;publication_year=2019&amp;pmid=31249028&amp;doi=10.1126/science.aaw2872.24&amp;"/></mixed-citation></ref><ref id="bib25"><label>25.</label><mixed-citation id="sref25"><named-content content-type="citation-string">Siddiqui-Jain A., Grand C.L., Bearss D.J., Hurley L.H. Direct evidence for a G-quadruplex in a promoter region and its targeting with a small molecule to repress c-MYC transcription. Proc. Natl. Acad. Sci. U S A. 2002;99:11593–11598. doi: 10.1073/pnas.182256799.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1073/pnas.182256799"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC129314"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="12195017"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Proc. Natl. Acad. Sci. U S A&amp;title=Direct evidence for a G-quadruplex in a promoter region and its targeting with a small molecule to repress c-MYC transcription&amp;author=A. Siddiqui-Jain&amp;author=C.L. Grand&amp;author=D.J. Bearss&amp;author=L.H. Hurley&amp;volume=99&amp;publication_year=2002&amp;pages=11593-11598&amp;pmid=12195017&amp;doi=10.1073/pnas.182256799&amp;"/></mixed-citation></ref><ref id="bib26"><label>26.</label><mixed-citation id="sref26"><named-content content-type="citation-string">Belotserkovskii B.P., De Silva E., Tornaletti S., Wang G., Vasquez K.M., Hanawalt P.C. A triplex-forming sequence from the human c-MYC promoter interferes with DNA transcription. J. Biol. Chem. 2007;282:32433–32441. doi: 10.1074/jbc.M704618200.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1074/jbc.M704618200"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="17785457"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=J. Biol. Chem.&amp;title=A triplex-forming sequence from the human c-MYC promoter interferes with DNA transcription&amp;author=B.P. Belotserkovskii&amp;author=E. De Silva&amp;author=S. Tornaletti&amp;author=G. Wang&amp;author=K.M. Vasquez&amp;volume=282&amp;publication_year=2007&amp;pages=32433-32441&amp;pmid=17785457&amp;doi=10.1074/jbc.M704618200&amp;"/></mixed-citation></ref><ref id="bib27"><label>27.</label><mixed-citation id="sref27"><named-content content-type="citation-string">Ditlevson J.V., Tornaletti S., Belotserkovskii B.P., Teijeiro V., Wang G., Vasquez K.M., Hanawalt P.C. Inhibitory effect of a short Z-DNA forming sequence on transcription elongation by T7 RNA polymerase. Nucleic Acids Res. 2008;36:3163–3170. doi: 10.1093/nar/gkn136.27.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gkn136.27"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC2425487"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="18400779"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res.&amp;title=Inhibitory effect of a short Z-DNA forming sequence on transcription elongation by T7 RNA polymerase&amp;author=J.V. Ditlevson&amp;author=S. Tornaletti&amp;author=B.P. Belotserkovskii&amp;author=V. Teijeiro&amp;author=G. Wang&amp;volume=36&amp;publication_year=2008&amp;pages=3163-3170&amp;pmid=18400779&amp;doi=10.1093/nar/gkn136.27&amp;"/></mixed-citation></ref><ref id="bib28"><label>28.</label><mixed-citation id="sref28"><named-content content-type="citation-string">Kumari S., Bugaut A., Huppert J.L., Balasubramanian S. An RNA G-quadruplex in the 5′ UTR of the NRAS proto-oncogene modulates translation. Nat. Chem. Biol. 2007;3:218–221. doi: 10.1038/nchembio864.28.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/nchembio864.28"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC2206252"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="17322877"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat. Chem. Biol.&amp;title=An RNA G-quadruplex in the 5′ UTR of the NRAS proto-oncogene modulates translation&amp;author=S. Kumari&amp;author=A. Bugaut&amp;author=J.L. Huppert&amp;author=S. Balasubramanian&amp;volume=3&amp;publication_year=2007&amp;pages=218-221&amp;pmid=17322877&amp;doi=10.1038/nchembio864.28&amp;"/></mixed-citation></ref><ref id="bib29"><label>29.</label><mixed-citation id="sref29"><named-content content-type="citation-string">Ray B.K., Dhar S., Shakya A., Ray A. Z-DNA-forming silencer in the first exon regulates human ADAM-12 gene expression. Proc. Natl. Acad. Sci. U S A. 2011;108:103–108. doi: 10.1073/pnas.1008831108.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1073/pnas.1008831108"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC3017174"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="21173277"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Proc. Natl. Acad. Sci. U S A&amp;title=Z-DNA-forming silencer in the first exon regulates human ADAM-12 gene expression&amp;author=B.K. Ray&amp;author=S. Dhar&amp;author=A. Shakya&amp;author=A. Ray&amp;volume=108&amp;publication_year=2011&amp;pages=103-108&amp;pmid=21173277&amp;doi=10.1073/pnas.1008831108&amp;"/></mixed-citation></ref><ref id="bib30"><label>30.</label><mixed-citation id="sref30"><named-content content-type="citation-string">Agarwala P., Pandey S., Mapa K., Maiti S. The G-quadruplex augments translation in the 5′ untranslated region of transforming growth factor β2. Biochemistry. 2013;52:1528–1538. doi: 10.1021/bi301365g.30.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1021/bi301365g.30"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="23387555"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Biochemistry&amp;title=The G-quadruplex augments translation in the 5′ untranslated region of transforming growth factor β2&amp;author=P. Agarwala&amp;author=S. Pandey&amp;author=K. Mapa&amp;author=S. Maiti&amp;volume=52&amp;publication_year=2013&amp;pages=1528-1538&amp;pmid=23387555&amp;doi=10.1021/bi301365g.30&amp;"/></mixed-citation></ref><ref id="bib31"><label>31.</label><mixed-citation id="sref31"><named-content content-type="citation-string">Georgakopoulos-Soares I., Parada G.E., Wong H.Y., Miska E.A., Kwok C.K., Hemberg M. Alternative splicing modulation by G-quadruplexes. </named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1101/700575.31"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC9065059"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="35504902"/></mixed-citation></ref><ref id="bib32"><label>32.</label><mixed-citation id="sref32"><named-content content-type="citation-string">Shirude P.S., Okumus B., Ying L., Ha T., Balasubramanian S. Single-molecule conformational analysis of G-quadruplex formation in the promoter DNA duplex of the proto-oncogene c-kit. J. Am. Chem. Soc. 2007;129:7484–7485. doi: 10.1021/ja070497d.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1021/ja070497d"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC2195893"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="17523641"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=J. Am. Chem. Soc.&amp;title=Single-molecule conformational analysis of G-quadruplex formation in the promoter DNA duplex of the proto-oncogene c-kit&amp;author=P.S. Shirude&amp;author=B. Okumus&amp;author=L. Ying&amp;author=T. Ha&amp;author=S. Balasubramanian&amp;volume=129&amp;publication_year=2007&amp;pages=7484-7485&amp;pmid=17523641&amp;doi=10.1021/ja070497d&amp;"/></mixed-citation></ref><ref id="bib33"><label>33.</label><mixed-citation id="sref33"><named-content content-type="citation-string">Weinhold N., Jacobsen A., Schultz N., Sander C., Lee W. Genome-wide analysis of noncoding regulatory mutations in cancer. Nat. Genet. 2014;46:1160–1165. doi: 10.1038/ng.3101.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/ng.3101"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC4217527"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="25261935"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat. Genet.&amp;title=Genome-wide analysis of noncoding regulatory mutations in cancer&amp;author=N. Weinhold&amp;author=A. Jacobsen&amp;author=N. Schultz&amp;author=C. Sander&amp;author=W. Lee&amp;volume=46&amp;publication_year=2014&amp;pages=1160-1165&amp;pmid=25261935&amp;doi=10.1038/ng.3101&amp;"/></mixed-citation></ref><ref id="bib34"><label>34.</label><mixed-citation id="sref34"><named-content content-type="citation-string">Song J.H., Kang H.-J., Luevano L.A., Gokhale V., Wu K., Pandey R., Sherry Chow H.-H., Hurley L.H., Kraft A.S. Small-molecule-targeting hairpin loop of hTERT promoter G-quadruplex induces cancer cell death. Cell Chem. Biol. 2019;26:1110–1121.e4. doi: 10.1016/j.chembiol.2019.04.009.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.chembiol.2019.04.009"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6713458"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="31155510"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Cell Chem. Biol.&amp;title=Small-molecule-targeting hairpin loop of hTERT promoter G-quadruplex induces cancer cell death&amp;author=J.H. Song&amp;author=H.-J. Kang&amp;author=L.A. Luevano&amp;author=V. Gokhale&amp;author=K. Wu&amp;volume=26&amp;publication_year=2019&amp;pages=1110-1121.e4&amp;pmid=31155510&amp;doi=10.1016/j.chembiol.2019.04.009&amp;"/></mixed-citation></ref><ref id="bib35"><label>35.</label><mixed-citation id="sref35"><named-content content-type="citation-string">Monsen R.C., DeLeeuw L., Dean W.L., Gray R.D., Sabo T.M., Chakravarthy S., Chaires J.B., Trent J.O. The hTERT core promoter forms three parallel G-quadruplexes. Nucleic Acids Res. 2020;48:5720–5734. doi: 10.1093/nar/gkaa107.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gkaa107"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC7261196"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="32083666"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res.&amp;title=The hTERT core promoter forms three parallel G-quadruplexes&amp;author=R.C. Monsen&amp;author=L. DeLeeuw&amp;author=W.L. Dean&amp;author=R.D. Gray&amp;author=T.M. Sabo&amp;volume=48&amp;publication_year=2020&amp;pages=5720-5734&amp;pmid=32083666&amp;doi=10.1093/nar/gkaa107&amp;"/></mixed-citation></ref><ref id="bib36"><label>36.</label><mixed-citation id="sref36"><named-content content-type="citation-string">Seenisamy J., Rezler E.M., Powell T.J., Tye D., Gokhale V., Joshi C.S., Siddiqui-Jain A., Hurley L.H. The dynamic character of the G-quadruplex element in the c-MYC promoter and modification by TMPyP4. J. Am. Chem. Soc. 2004;126:8702–8709. doi: 10.1021/ja040022b.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1021/ja040022b"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="15250722"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=J. Am. Chem. Soc.&amp;title=The dynamic character of the G-quadruplex element in the c-MYC promoter and modification by TMPyP4&amp;author=J. Seenisamy&amp;author=E.M. Rezler&amp;author=T.J. Powell&amp;author=D. Tye&amp;author=V. Gokhale&amp;volume=126&amp;publication_year=2004&amp;pages=8702-8709&amp;pmid=15250722&amp;doi=10.1021/ja040022b&amp;"/></mixed-citation></ref><ref id="bib37"><label>37.</label><mixed-citation id="sref37"><named-content content-type="citation-string">Kaiser C.E., Van Ert N.A., Agrawal P., Chawla R., Yang D., Hurley L.H. Insight into the complexity of the i-motif and G-quadruplex DNA structures formed in the KRAS promoter and subsequent drug-induced gene repression. J. Am. Chem. Soc. 2017;139:8522–8536. doi: 10.1021/jacs.7b02046.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1021/jacs.7b02046"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC5978000"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="28570076"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=J. Am. Chem. Soc.&amp;title=Insight into the complexity of the i-motif and G-quadruplex DNA structures formed in the KRAS promoter and subsequent drug-induced gene repression&amp;author=C.E. Kaiser&amp;author=N.A. Van Ert&amp;author=P. Agrawal&amp;author=R. Chawla&amp;author=D. Yang&amp;volume=139&amp;publication_year=2017&amp;pages=8522-8536&amp;pmid=28570076&amp;doi=10.1021/jacs.7b02046&amp;"/></mixed-citation></ref><ref id="bib38"><label>38.</label><mixed-citation id="sref38"><named-content content-type="citation-string">Kim N. The interplay between G-quadruplex and transcription. Curr. Med. Chem. 2019;26:2898–2917. doi: 10.2174/0929867325666171229132619.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.2174/0929867325666171229132619"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6026074"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="29284393"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Curr. Med. Chem.&amp;title=The interplay between G-quadruplex and transcription&amp;author=N. Kim&amp;volume=26&amp;publication_year=2019&amp;pages=2898-2917&amp;pmid=29284393&amp;doi=10.2174/0929867325666171229132619&amp;"/></mixed-citation></ref><ref id="bib39"><label>39.</label><mixed-citation id="sref39"><named-content content-type="citation-string">Inoue F., Kircher M., Martin B., Cooper G.M., Witten D.M., McManus M.T., Ahituv N., Shendure J. A systematic comparison reveals substantial differences in chromosomal versus episomal encoding of enhancer activity. Genome Res. 2017;27:38–52. doi: 10.1101/gr.212092.116.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1101/gr.212092.116"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC5204343"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="27831498"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Genome Res.&amp;title=A systematic comparison reveals substantial differences in chromosomal versus episomal encoding of enhancer activity&amp;author=F. Inoue&amp;author=M. Kircher&amp;author=B. Martin&amp;author=G.M. Cooper&amp;author=D.M. Witten&amp;volume=27&amp;publication_year=2017&amp;pages=38-52&amp;pmid=27831498&amp;doi=10.1101/gr.212092.116&amp;"/></mixed-citation></ref><ref id="bib40"><label>40.</label><mixed-citation id="sref40"><named-content content-type="citation-string">Karczewski K.J., Francioli L.C., Tiao G., Cummings B.B., Alföldi J., Wang Q., Collins R.L., Laricchia K.M., Ganna A., Birnbaum D.P., et al.  The mutational constraint spectrum quantified from variation in 141,456 humans. Nature. 2020;581:434–443. doi: 10.1038/s41586-020-2308-7.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/s41586-020-2308-7"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC7334197"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="32461654"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nature&amp;title=The mutational constraint spectrum quantified from variation in 141,456 humans&amp;author=K.J. Karczewski&amp;author=L.C. Francioli&amp;author=G. Tiao&amp;author=B.B. Cummings&amp;author=J. Alföldi&amp;volume=581&amp;publication_year=2020&amp;pages=434-443&amp;pmid=32461654&amp;doi=10.1038/s41586-020-2308-7&amp;"/></mixed-citation></ref><ref id="bib41"><label>41.</label><mixed-citation id="sref41"><named-content content-type="citation-string">Bacolla A., Jaworski A., Larson J.E., Jakupciak J.P., Chuzhanova N., Abeysinghe S.S., O’Connell C.D., Cooper D.N., Wells R.D. Breakpoints of gross deletions coincide with non-B DNA conformations. Proc. Natl. Acad. Sci. U S A. 2004;101:14162–14167. doi: 10.1073/pnas.0405974101.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1073/pnas.0405974101"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC521098"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="15377784"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Proc. Natl. Acad. Sci. U S A&amp;title=Breakpoints of gross deletions coincide with non-B DNA conformations&amp;author=A. Bacolla&amp;author=A. Jaworski&amp;author=J.E. Larson&amp;author=J.P. Jakupciak&amp;author=N. Chuzhanova&amp;volume=101&amp;publication_year=2004&amp;pages=14162-14167&amp;pmid=15377784&amp;doi=10.1073/pnas.0405974101&amp;"/></mixed-citation></ref><ref id="bib42"><label>42.</label><mixed-citation id="sref42"><named-content content-type="citation-string">Zerbino D.R., Wilder S.P., Johnson N., Juettemann T., Flicek P.R. The ensembl regulatory build. Genome Biol. 2015;16:56. doi: 10.1186/s13059-015-0621-5.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1186/s13059-015-0621-5"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC4407537"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="25887522"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Genome Biol.&amp;title=The ensembl regulatory build&amp;author=D.R. Zerbino&amp;author=S.P. Wilder&amp;author=N. Johnson&amp;author=T. Juettemann&amp;author=P.R. Flicek&amp;volume=16&amp;publication_year=2015&amp;pages=56&amp;pmid=25887522&amp;doi=10.1186/s13059-015-0621-5&amp;"/></mixed-citation></ref><ref id="bib43"><label>43.</label><mixed-citation id="sref43"><named-content content-type="citation-string">Frigola J., Sabarinathan R., Mularoni L., Muiños F., Gonzalez-Perez A., López-Bigas N. Reduced mutation rate in exons due to differential mismatch repair. Nat. Genet. 2017;49:1684–1692. doi: 10.1038/ng.3991.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/ng.3991"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC5712219"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="29106418"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat. Genet.&amp;title=Reduced mutation rate in exons due to differential mismatch repair&amp;author=J. Frigola&amp;author=R. Sabarinathan&amp;author=L. Mularoni&amp;author=F. Muiños&amp;author=A. Gonzalez-Perez&amp;volume=49&amp;publication_year=2017&amp;pages=1684-1692&amp;pmid=29106418&amp;doi=10.1038/ng.3991&amp;"/></mixed-citation></ref><ref id="bib44"><label>44.</label><mixed-citation id="sref44"><named-content content-type="citation-string">GTEx Consortium. Laboratory, Data Analysis &amp;Coordinating Center (LDACC)—Analysis Working Group. Statistical Methods groups—analysis Working Group. Enhancing GTEx (eGTEx) groups. NIH Common Fund. NIH/NCI. NIH/NHGRI. NIH/NIMH. NIH/NIDA. Biospecimen Collection Source Site—NDRI. et al.  Genetic effects on gene expression across human tissues. Nature. 2017;550:204–213. doi: 10.1038/nature24277.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/nature24277"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC5776756"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="29022597"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nature&amp;title=Genetic effects on gene expression across human tissues&amp;volume=550&amp;publication_year=2017&amp;pages=204-213&amp;pmid=29022597&amp;doi=10.1038/nature24277&amp;"/></mixed-citation></ref><ref id="bib45"><label>45.</label><mixed-citation id="sref45"><named-content content-type="citation-string">Marsico G., Chambers V.S., Sahakyan A.B., McCauley P., Boutell J.M., Antonio M.D., Balasubramanian S. Whole genome experimental maps of DNA G-quadruplexes in multiple species. Nucleic Acids Res. 2019;47:3862–3874. doi: 10.1093/nar/gkz179.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gkz179"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6486626"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="30892612"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res.&amp;title=Whole genome experimental maps of DNA G-quadruplexes in multiple species&amp;author=G. Marsico&amp;author=V.S. Chambers&amp;author=A.B. Sahakyan&amp;author=P. McCauley&amp;author=J.M. Boutell&amp;volume=47&amp;publication_year=2019&amp;pages=3862-3874&amp;pmid=30892612&amp;doi=10.1093/nar/gkz179&amp;"/></mixed-citation></ref><ref id="bib46"><label>46.</label><mixed-citation id="sref46"><named-content content-type="citation-string">Hou Y., Li F., Zhang R., Li S., Liu H., Qin Z.S., Sun X. Integrative characterization of G-Quadruplexes in the three-dimensional chromatin structure. Epigenetics. 2019;14:894–911. doi: 10.1080/15592294.2019.1621140.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1080/15592294.2019.1621140"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6691997"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="31177910"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Epigenetics&amp;title=Integrative characterization of G-Quadruplexes in the three-dimensional chromatin structure&amp;author=Y. Hou&amp;author=F. Li&amp;author=R. Zhang&amp;author=S. Li&amp;author=H. Liu&amp;volume=14&amp;publication_year=2019&amp;pages=894-911&amp;pmid=31177910&amp;doi=10.1080/15592294.2019.1621140&amp;"/></mixed-citation></ref><ref id="bib47"><label>47.</label><mixed-citation id="sref47"><named-content content-type="citation-string">Consortium, Encode Project. Dunham I., Kundaje A., Aldred S.F., Collins P.J., Davis C.A., Doyle F., Epstein C.B., Frietze S., Harrow J., et al.  An integrated encyclopedia of DNA elements in the human genome. Nature. 2012;489:57–74. doi: 10.1038/nature11247.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/nature11247"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC3439153"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="22955616"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nature&amp;title=An integrated encyclopedia of DNA elements in the human genome&amp;author=I. Dunham&amp;author=A. Kundaje&amp;author=S.F. Aldred&amp;author=P.J. Collins&amp;author=C.A. Davis&amp;volume=489&amp;publication_year=2012&amp;pages=57-74&amp;pmid=22955616&amp;doi=10.1038/nature11247&amp;"/></mixed-citation></ref><ref id="bib48"><label>48.</label><mixed-citation id="sref48"><named-content content-type="citation-string">Sémon M., Mouchiroud D., Duret L. Relationship between gene expression and GC-content in mammals: statistical significance and biological relevance. Hum. Mol. Genet. 2005;14:421–427. doi: 10.1093/hmg/ddi038.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/hmg/ddi038"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="15590696"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Hum. Mol. Genet.&amp;title=Relationship between gene expression and GC-content in mammals: statistical significance and biological relevance&amp;author=M. Sémon&amp;author=D. Mouchiroud&amp;author=L. Duret&amp;volume=14&amp;publication_year=2005&amp;pages=421-427&amp;pmid=15590696&amp;doi=10.1093/hmg/ddi038&amp;"/></mixed-citation></ref><ref id="bib49"><label>49.</label><mixed-citation id="sref49"><named-content content-type="citation-string">Fornes O., Castro-Mondragon J.A., Khan A., van der Lee R., Zhang X., Richmond P.A., Modi B.P., Correard S., Gheorghe M., Baranašić D., et al.  JASPAR 2020: update of the open-access database of transcription factor binding profiles. Nucleic Acids Res. 2020;48:D87–D92. doi: 10.1093/nar/gkz1001.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gkz1001"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC7145627"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="31701148"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res.&amp;title=JASPAR 2020: update of the open-access database of transcription factor binding profiles&amp;author=O. Fornes&amp;author=J.A. Castro-Mondragon&amp;author=A. Khan&amp;author=R. van der Lee&amp;author=X. Zhang&amp;volume=48&amp;publication_year=2020&amp;pages=D87-D92&amp;pmid=31701148&amp;doi=10.1093/nar/gkz1001&amp;"/></mixed-citation></ref><ref id="bib50"><label>50.</label><mixed-citation id="sref50"><named-content content-type="citation-string">An J.-Y., Lin K., Zhu L., Werling D.M., Dong S., Brand H., Wang H.Z., Zhao X., Schwartz G.B., Collins R.L., et al.  Genome-wide de novo risk score implicates promoter variation in autism spectrum disorder. Science. 2018;362 doi: 10.1126/science.aat6576.50.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1126/science.aat6576.50"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6432922"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="30545852"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Science&amp;title=Genome-wide de novo risk score implicates promoter variation in autism spectrum disorder&amp;author=J.-Y. An&amp;author=K. Lin&amp;author=L. Zhu&amp;author=D.M. Werling&amp;author=S. Dong&amp;volume=362&amp;publication_year=2018&amp;pmid=30545852&amp;doi=10.1126/science.aat6576.50&amp;"/></mixed-citation></ref><ref id="bib51"><label>51.</label><mixed-citation id="sref51"><named-content content-type="citation-string">Kwok C.K., Merrick C.J. G-quadruplexes: prediction, characterization, and biological application. Trends Biotechnol. 2017;35:997–1013. doi: 10.1016/j.tibtech.2017.06.012.51.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.tibtech.2017.06.012.51"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="28755976"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Trends Biotechnol.&amp;title=G-quadruplexes: prediction, characterization, and biological application&amp;author=C.K. Kwok&amp;author=C.J. Merrick&amp;volume=35&amp;publication_year=2017&amp;pages=997-1013&amp;pmid=28755976&amp;doi=10.1016/j.tibtech.2017.06.012.51&amp;"/></mixed-citation></ref><ref id="bib52"><label>52.</label><mixed-citation id="sref52"><named-content content-type="citation-string">Umar M.I., Ji D., Chan C.-Y., Kwok C.K. G-quadruplex-based fluorescent turn-on ligands and aptamers: from development to applications. Molecules. 2019;24 doi: 10.3390/molecules24132416.52.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.3390/molecules24132416.52"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6650947"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="31262059"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Molecules&amp;title=G-quadruplex-based fluorescent turn-on ligands and aptamers: from development to applications&amp;author=M.I. Umar&amp;author=D. Ji&amp;author=C.-Y. Chan&amp;author=C.K. Kwok&amp;volume=24&amp;publication_year=2019&amp;pmid=31262059&amp;doi=10.3390/molecules24132416.52&amp;"/></mixed-citation></ref><ref id="bib53"><label>53.</label><mixed-citation id="sref53"><named-content content-type="citation-string">Gordon M.G., Inoue F., Martin B., Schubach M., Agarwal V., Whalen S., Feng S., Zhao J., Ashuach T., Ziffra R., et al.  lentiMPRA and MPRAflow for high-throughput functional characterization of gene regulatory elements. Nat. Protoc. 2020;15:2387–2412. doi: 10.1038/s41596-020-0333-5.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/s41596-020-0333-5"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC7550205"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="32641802"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat. Protoc.&amp;title=lentiMPRA and MPRAflow for high-throughput functional characterization of gene regulatory elements&amp;author=M.G. Gordon&amp;author=F. Inoue&amp;author=B. Martin&amp;author=M. Schubach&amp;author=V. Agarwal&amp;volume=15&amp;publication_year=2020&amp;pages=2387-2412&amp;pmid=32641802&amp;doi=10.1038/s41596-020-0333-5&amp;"/></mixed-citation></ref><ref id="bib54"><label>54.</label><mixed-citation id="sref54"><named-content content-type="citation-string">Kircher M., Xiong C., Martin B., Schubach M., Inoue F., Bell R.J.A., Costello J.F., Shendure J., Ahituv N. Saturation mutagenesis of twenty disease-associated regulatory elements at single base-pair resolution. Nat. Commun. 2019;10:3583. doi: 10.1038/s41467–019–11526–w.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/s41467–019–11526–w"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6687891"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="31395865"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat. Commun.&amp;title=Saturation mutagenesis of twenty disease-associated regulatory elements at single base-pair resolution&amp;author=M. Kircher&amp;author=C. Xiong&amp;author=B. Martin&amp;author=M. Schubach&amp;author=F. Inoue&amp;volume=10&amp;publication_year=2019&amp;pages=3583&amp;pmid=31395865&amp;doi=10.1038/s41467–019–11526–w&amp;"/></mixed-citation></ref><ref id="bib55"><label>55.</label><mixed-citation id="sref55"><named-content content-type="citation-string">Ashuach T., Fischer D.S., Kreimer A., Ahituv N., Theis F.J., Yosef N. MPRAnalyze: statistical framework for massively parallel reporter assays. Genome Biol. 2019;20:183. doi: 10.1186/s13059–019–1787–z.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1186/s13059–019–1787–z"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6717970"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="31477158"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Genome Biol.&amp;title=MPRAnalyze: statistical framework for massively parallel reporter assays&amp;author=T. Ashuach&amp;author=D.S. Fischer&amp;author=A. Kreimer&amp;author=N. Ahituv&amp;author=F.J. Theis&amp;volume=20&amp;publication_year=2019&amp;pages=183&amp;pmid=31477158&amp;doi=10.1186/s13059–019–1787–z&amp;"/></mixed-citation></ref><ref id="bib56"><label>56.</label><mixed-citation id="sref56"><named-content content-type="citation-string">Brooks T.A., Hurley L.H. Targeting MYC expression through G-quadruplexes. Genes Cancer. 2010;1:641–649. doi: 10.1177/1947601910377493.56.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1177/1947601910377493.56"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC2992328"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="21113409"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Genes Cancer&amp;title=Targeting MYC expression through G-quadruplexes&amp;author=T.A. Brooks&amp;author=L.H. Hurley&amp;volume=1&amp;publication_year=2010&amp;pages=641-649&amp;pmid=21113409&amp;doi=10.1177/1947601910377493.56&amp;"/></mixed-citation></ref><ref id="bib57"><label>57.</label><mixed-citation id="sref57"><named-content content-type="citation-string">Lam E.Y.N., Beraldi D., Tannahill D., Balasubramanian S. G-quadruplex structures are stable and detectable in human genomic DNA. Nat. Commun. 2013;4:1796. doi: 10.1038/ncomms2792.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/ncomms2792"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC3736099"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="23653208"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat. Commun.&amp;title=G-quadruplex structures are stable and detectable in human genomic DNA&amp;author=E.Y.N. Lam&amp;author=D. Beraldi&amp;author=D. Tannahill&amp;author=S. Balasubramanian&amp;volume=4&amp;publication_year=2013&amp;pages=1796&amp;pmid=23653208&amp;doi=10.1038/ncomms2792&amp;"/></mixed-citation></ref><ref id="bib58"><label>58.</label><mixed-citation id="sref58"><named-content content-type="citation-string">Armas P., David A., Calcaterra N.B. Transcriptional control by G-quadruplexes: in vivo roles and perspectives for specific intervention. Transcription. 2017;8:21–25. doi: 10.1080/21541264.2016.1243505.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1080/21541264.2016.1243505"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC5279714"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="27696937"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Transcription&amp;title=Transcriptional control by G-quadruplexes: in vivo roles and perspectives for specific intervention&amp;author=P. Armas&amp;author=A. David&amp;author=N.B. Calcaterra&amp;volume=8&amp;publication_year=2017&amp;pages=21-25&amp;pmid=27696937&amp;doi=10.1080/21541264.2016.1243505&amp;"/></mixed-citation></ref><ref id="bib59"><label>59.</label><mixed-citation id="sref59"><named-content content-type="citation-string">Wittig B., Wölfl S., Dorbic T., Vahrson W., Rich A. Transcription of human c-myc in permeabilized nuclei is associated with formation of Z-DNA in three discrete regions of the gene. EMBO J. 1992;11:4653–4663. doi: 10.1002/j.1460-2075.1992.tb05567.x.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1002/j.1460-2075.1992.tb05567.x"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC557041"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="1330542"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=EMBO J.&amp;title=Transcription of human c-myc in permeabilized nuclei is associated with formation of Z-DNA in three discrete regions of the gene&amp;author=B. Wittig&amp;author=S. Wölfl&amp;author=T. Dorbic&amp;author=W. Vahrson&amp;author=A. Rich&amp;volume=11&amp;publication_year=1992&amp;pages=4653-4663&amp;pmid=1330542&amp;doi=10.1002/j.1460-2075.1992.tb05567.x&amp;"/></mixed-citation></ref><ref id="bib60"><label>60.</label><mixed-citation id="sref60"><named-content content-type="citation-string">Maruyama A., Mimura J., Harada N., Itoh K. Nrf2 activation is associated with Z-DNA formation in the human HO-1 promoter. Nucleic Acids Res. 2013;41:5223–5234. doi: 10.1093/nar/gkt243.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gkt243"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC3664823"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="23571756"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res.&amp;title=Nrf2 activation is associated with Z-DNA formation in the human HO-1 promoter&amp;author=A. Maruyama&amp;author=J. Mimura&amp;author=N. Harada&amp;author=K. Itoh&amp;volume=41&amp;publication_year=2013&amp;pages=5223-5234&amp;pmid=23571756&amp;doi=10.1093/nar/gkt243&amp;"/></mixed-citation></ref><ref id="bib61"><label>61.</label><mixed-citation id="sref61"><named-content content-type="citation-string">Mendoza O., Bourdoncle A., Boulé J.-B., Brosh R.M., Jr., Mergny J.-L. G-quadruplexes and helicases. Nucleic Acids Res. 2016;44:1989–2006. doi: 10.1093/nar/gkw079.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gkw079"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC4797304"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="26883636"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res.&amp;title=G-quadruplexes and helicases&amp;author=O. Mendoza&amp;author=A. Bourdoncle&amp;author=J.-B. Boulé&amp;author=R.M. Brosh&amp;author=J.-L. Mergny&amp;volume=44&amp;publication_year=2016&amp;pages=1989-2006&amp;pmid=26883636&amp;doi=10.1093/nar/gkw079&amp;"/></mixed-citation></ref><ref id="bib62"><label>62.</label><mixed-citation id="sref62"><named-content content-type="citation-string">Sharma S. Non-B DNA secondary structures and their resolution by RecQ helicases. J. Nucleic Acids. 2011;2011:724215. doi: 10.4061/2011/724215.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.4061/2011/724215"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC3185257"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="21977309"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=J. Nucleic Acids&amp;title=Non-B DNA secondary structures and their resolution by RecQ helicases&amp;author=S. Sharma&amp;volume=2011&amp;publication_year=2011&amp;pages=724215&amp;pmid=21977309&amp;doi=10.4061/2011/724215&amp;"/></mixed-citation></ref><ref id="bib63"><label>63.</label><mixed-citation id="sref63"><named-content content-type="citation-string">Neidle S. Quadruplex nucleic acids as targets for anticancer therapeutics. Nat. Rev. Chem. 2017;1 doi: 10.1038/s41570-017-0041.63.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/s41570-017-0041.63"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat. Rev. Chem.&amp;title=Quadruplex nucleic acids as targets for anticancer therapeutics&amp;author=S. Neidle&amp;volume=1&amp;publication_year=2017&amp;doi=10.1038/s41570-017-0041.63&amp;"/></mixed-citation></ref><ref id="bib64"><label>64.</label><mixed-citation id="sref64"><named-content content-type="citation-string">Hänsel-Hertsch R., Di Antonio M., Balasubramanian S. DNA G-quadruplexes in the human genome: detection, functions and therapeutic potential. Nat. Rev. Mol. Cell Biol. 2017;18:279–284. doi: 10.1038/nrm.2017.3.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/nrm.2017.3"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="28225080"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat. Rev. Mol. Cell Biol.&amp;title=DNA G-quadruplexes in the human genome: detection, functions and therapeutic potential&amp;author=R. Hänsel-Hertsch&amp;author=M. Di Antonio&amp;author=S. Balasubramanian&amp;volume=18&amp;publication_year=2017&amp;pages=279-284&amp;pmid=28225080&amp;doi=10.1038/nrm.2017.3&amp;"/></mixed-citation></ref><ref id="bib65"><label>65.</label><mixed-citation id="sref65"><named-content content-type="citation-string">Balasubramanian S., Hurley L.H., Neidle S. Targeting G-quadruplexes in gene promoters: a novel anticancer strategy? Nat. Rev. Drug Discov. 2011;10:261–275. doi: 10.1038/nrd3428.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/nrd3428"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC3119469"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="21455236"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat. Rev. Drug Discov.&amp;title=Targeting G-quadruplexes in gene promoters: a novel anticancer strategy?&amp;author=S. Balasubramanian&amp;author=L.H. Hurley&amp;author=S. Neidle&amp;volume=10&amp;publication_year=2011&amp;pages=261-275&amp;pmid=21455236&amp;doi=10.1038/nrd3428&amp;"/></mixed-citation></ref><ref id="bib66"><label>66.</label><mixed-citation id="sref66"><named-content content-type="citation-string">Sun Z.-Y., Wang X.-N., Cheng S.-Q., Su X.-X., Ou T.-M. Developing novel G-quadruplex ligands: from interaction with nucleic acids to interfering with nucleic acid–protein interaction. Molecules. 2019;24:396. doi: 10.3390/molecules24030396.66.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.3390/molecules24030396.66"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6384609"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="30678288"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Molecules&amp;title=Developing novel G-quadruplex ligands: from interaction with nucleic acids to interfering with nucleic acid–protein interaction&amp;author=Z.-Y. Sun&amp;author=X.-N. Wang&amp;author=S.-Q. Cheng&amp;author=X.-X. Su&amp;author=T.-M. Ou&amp;volume=24&amp;publication_year=2019&amp;pages=396&amp;pmid=30678288&amp;doi=10.3390/molecules24030396.66&amp;"/></mixed-citation></ref><ref id="bib67"><label>67.</label><mixed-citation id="sref67"><named-content content-type="citation-string">Yuan X., Larsson C., Xu D. Mechanisms underlying the activation of TERT transcription and telomerase activity in human cancer: old actors and new players. Oncogene. 2019;38:6172–6183. doi: 10.1038/s41388-019-0872-9.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/s41388-019-0872-9"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6756069"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="31285550"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Oncogene&amp;title=Mechanisms underlying the activation of TERT transcription and telomerase activity in human cancer: old actors and new players&amp;author=X. Yuan&amp;author=C. Larsson&amp;author=D. Xu&amp;volume=38&amp;publication_year=2019&amp;pages=6172-6183&amp;pmid=31285550&amp;doi=10.1038/s41388-019-0872-9&amp;"/></mixed-citation></ref><ref id="bib68"><label>68.</label><mixed-citation id="sref68"><named-content content-type="citation-string">Matharu N., Ahituv N. Modulating gene regulation to treat genetic disorders. Nat. Rev. Drug Discov. 2020;19:757–775. doi: 10.1038/s41573-020-0083-7.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/s41573-020-0083-7"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC12922764"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="33020616"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat. Rev. Drug Discov.&amp;title=Modulating gene regulation to treat genetic disorders&amp;author=N. Matharu&amp;author=N. Ahituv&amp;volume=19&amp;publication_year=2020&amp;pages=757-775&amp;pmid=33020616&amp;doi=10.1038/s41573-020-0083-7&amp;"/></mixed-citation></ref><ref id="bib69"><label>69.</label><mixed-citation id="sref69"><named-content content-type="citation-string">Varshney D., Spiegel J., Zyner K., Tannahill D., Balasubramanian S. The regulation and functions of DNA and RNA G-quadruplexes. Nat. Rev. Mol. Cell Biol. 2020;21:459–474. doi: 10.1038/s41580-020-0236-x.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/s41580-020-0236-x"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC7115845"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="32313204"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat. Rev. Mol. Cell Biol.&amp;title=The regulation and functions of DNA and RNA G-quadruplexes&amp;author=D. Varshney&amp;author=J. Spiegel&amp;author=K. Zyner&amp;author=D. Tannahill&amp;author=S. Balasubramanian&amp;volume=21&amp;publication_year=2020&amp;pages=459-474&amp;pmid=32313204&amp;doi=10.1038/s41580-020-0236-x&amp;"/></mixed-citation></ref><ref id="bib70"><label>70.</label><mixed-citation id="sref70"><named-content content-type="citation-string">Beaudoin J.-D., Perreault J.-P. Exploring mRNA 3′-UTR G-quadruplexes: evidence of roles in both alternative polyadenylation and mRNA shortening. Nucleic Acids Res. 2013;41:5898–5911. doi: 10.1093/nar/gkt265.70.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gkt265.70"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC3675481"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="23609544"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res.&amp;title=Exploring mRNA 3′-UTR G-quadruplexes: evidence of roles in both alternative polyadenylation and mRNA shortening&amp;author=J.-D. Beaudoin&amp;author=J.-P. Perreault&amp;volume=41&amp;publication_year=2013&amp;pages=5898-5911&amp;pmid=23609544&amp;doi=10.1093/nar/gkt265.70&amp;"/></mixed-citation></ref><ref id="bib71"><label>71.</label><mixed-citation id="sref71"><named-content content-type="citation-string">Yang F., Sun X., Wang L., Li Q., Guan A., Shen G., Tang Y. Selective recognition of c-myc promoter G-quadruplex and down-regulation of oncogene c-myc transcription in human cancer cells by 3,8a-disubstituted indolizinone. RSC Adv. 2017;7:51965–51969. doi: 10.1039/c7ra09870g.71.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1039/c7ra09870g.71"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=RSC Adv.&amp;title=Selective recognition of c-myc promoter G-quadruplex and down-regulation of oncogene c-myc transcription in human cancer cells by 3,8a-disubstituted indolizinone&amp;author=F. Yang&amp;author=X. Sun&amp;author=L. Wang&amp;author=Q. Li&amp;author=A. Guan&amp;volume=7&amp;publication_year=2017&amp;pages=51965-51969&amp;doi=10.1039/c7ra09870g.71&amp;"/></mixed-citation></ref><ref id="bib72"><label>72.</label><mixed-citation id="sref72"><named-content content-type="citation-string">Nagesh N., Buscaglia R., Dettler J.M., Lewis E.A. Studies on the site and mode of TMPyP4 interactions with Bcl-2 promoter sequence G-quadruplexes. Biophys. J. 2010;98:2628–2633. doi: 10.1016/j.bpj.2010.02.050.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.bpj.2010.02.050"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC2877325"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="20513407"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Biophys. J.&amp;title=Studies on the site and mode of TMPyP4 interactions with Bcl-2 promoter sequence G-quadruplexes&amp;author=N. Nagesh&amp;author=R. Buscaglia&amp;author=J.M. Dettler&amp;author=E.A. Lewis&amp;volume=98&amp;publication_year=2010&amp;pages=2628-2633&amp;pmid=20513407&amp;doi=10.1016/j.bpj.2010.02.050&amp;"/></mixed-citation></ref><ref id="bib73"><label>73.</label><mixed-citation id="sref73"><named-content content-type="citation-string">Agrawal P., Lin C., Mathad R.I., Carver M., Yang D. The major G-quadruplex formed in the human BCL-2 proximal promoter adopts a parallel structure with a 13-nt loop in K solution. J. Am. Chem. Soc. 2014;136:1750–1753. doi: 10.1021/ja4118945.73.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1021/ja4118945.73"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC4732354"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="24450880"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=J. Am. Chem. Soc.&amp;title=The major G-quadruplex formed in the human BCL-2 proximal promoter adopts a parallel structure with a 13-nt loop in K solution&amp;author=P. Agrawal&amp;author=C. Lin&amp;author=R.I. Mathad&amp;author=M. Carver&amp;author=D. Yang&amp;volume=136&amp;publication_year=2014&amp;pages=1750-1753&amp;pmid=24450880&amp;doi=10.1021/ja4118945.73&amp;"/></mixed-citation></ref><ref id="bib74"><label>74.</label><mixed-citation id="sref74"><named-content content-type="citation-string">Phan A.T., Kuryavyi V., Burge S., Neidle S., Patel D.J. Structure of an unprecedented G-quadruplex scaffold in the human c-kit promoter. J. Am. Chem. Soc. 2007;129:4386–4392. doi: 10.1021/ja068739h.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1021/ja068739h"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC4693632"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="17362008"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=J. Am. Chem. Soc.&amp;title=Structure of an unprecedented G-quadruplex scaffold in the human c-kit promoter&amp;author=A.T. Phan&amp;author=V. Kuryavyi&amp;author=S. Burge&amp;author=S. Neidle&amp;author=D.J. Patel&amp;volume=129&amp;publication_year=2007&amp;pages=4386-4392&amp;pmid=17362008&amp;doi=10.1021/ja068739h&amp;"/></mixed-citation></ref><ref id="bib75"><label>75.</label><mixed-citation id="sref75"><named-content content-type="citation-string">Crawford D.C., Acuña J.M., Sherman S.L. FMR1 and the fragile X syndrome: human genome epidemiology review. Genet. Med. 2001;3:359–371. doi: 10.1097/00125817-200109000-00006.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1097/00125817-200109000-00006"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC4493892"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="11545690"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Genet. Med.&amp;title=FMR1 and the fragile X syndrome: human genome epidemiology review&amp;author=D.C. Crawford&amp;author=J.M. Acuña&amp;author=S.L. Sherman&amp;volume=3&amp;publication_year=2001&amp;pages=359-371&amp;pmid=11545690&amp;doi=10.1097/00125817-200109000-00006&amp;"/></mixed-citation></ref><ref id="bib76"><label>76.</label><mixed-citation id="sref76"><named-content content-type="citation-string">Khateb S., Weisman-Shomer P., Hershco-Shani I., Ludwig A.L., Fry M. The tetraplex (CGG)n destabilizing proteins hnRNP A2 and CBF-A enhance the in vivo translation of fragile X premutation mRNA. Nucleic Acids Res. 2007;35:5775–5788. doi: 10.1093/nar/gkm636.76.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gkm636.76"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC2034458"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="17716999"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res.&amp;title=The tetraplex (CGG)n destabilizing proteins hnRNP A2 and CBF-A enhance the in vivo translation of fragile X premutation mRNA&amp;author=S. Khateb&amp;author=P. Weisman-Shomer&amp;author=I. Hershco-Shani&amp;author=A.L. Ludwig&amp;author=M. Fry&amp;volume=35&amp;publication_year=2007&amp;pages=5775-5788&amp;pmid=17716999&amp;doi=10.1093/nar/gkm636.76&amp;"/></mixed-citation></ref><ref id="bib77"><label>77.</label><mixed-citation id="sref77"><named-content content-type="citation-string">Cogoi S., Xodo L.E. G-quadruplex formation within the promoter of the KRAS proto-oncogene and its effect on transcription. Nucleic Acids Res. 2006;34:2536–2549. doi: 10.1093/nar/gkl286.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gkl286"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC1459413"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="16687659"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res.&amp;title=G-quadruplex formation within the promoter of the KRAS proto-oncogene and its effect on transcription&amp;author=S. Cogoi&amp;author=L.E. Xodo&amp;volume=34&amp;publication_year=2006&amp;pages=2536-2549&amp;pmid=16687659&amp;doi=10.1093/nar/gkl286&amp;"/></mixed-citation></ref><ref id="bib78"><label>78.</label><mixed-citation id="sref78"><named-content content-type="citation-string">Cogoi S., Paramasivam M., Spolaore B., Xodo L.E. Structural polymorphism within a regulatory element of the human KRAS promoter: formation of G4-DNA recognized by nuclear proteins. Nucleic Acids Res. 2008;36:3765–3780. doi: 10.1093/nar/gkn120.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gkn120"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC2441797"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="18490377"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res.&amp;title=Structural polymorphism within a regulatory element of the human KRAS promoter: formation of G4-DNA recognized by nuclear proteins&amp;author=S. Cogoi&amp;author=M. Paramasivam&amp;author=B. Spolaore&amp;author=L.E. Xodo&amp;volume=36&amp;publication_year=2008&amp;pages=3765-3780&amp;pmid=18490377&amp;doi=10.1093/nar/gkn120&amp;"/></mixed-citation></ref><ref id="bib79"><label>79.</label><mixed-citation id="sref79"><named-content content-type="citation-string">Zhang H., Huang T., Hong Y., Yang W., Zhang X., Luo H., Xu H., Wang X. The retromer complex and sorting nexins in neurodegenerative diseases. Front. Aging Neurosci. 2018;10:79. doi: 10.3389/fnagi.2018.00079.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.3389/fnagi.2018.00079"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC5879135"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="29632483"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Front. Aging Neurosci.&amp;title=The retromer complex and sorting nexins in neurodegenerative diseases&amp;author=H. Zhang&amp;author=T. Huang&amp;author=Y. Hong&amp;author=W. Yang&amp;author=X. Zhang&amp;volume=10&amp;publication_year=2018&amp;pages=79&amp;pmid=29632483&amp;doi=10.3389/fnagi.2018.00079&amp;"/></mixed-citation></ref><ref id="bib80"><label>80.</label><mixed-citation id="sref80"><named-content content-type="citation-string">Agrawal P., Hatzakis E., Guo K., Carver M., Yang D. Solution structure of the major G-quadruplex formed in the human VEGF promoter in K : insights into loop interactions of the parallel G-quadruplexes. Nucleic Acids Res. 2013;41:10584–10592. doi: 10.1093/nar/gkt784.80.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gkt784.80"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC3905851"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="24005038"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res.&amp;title=Solution structure of the major G-quadruplex formed in the human VEGF promoter in K : insights into loop interactions of the parallel G-quadruplexes&amp;author=P. Agrawal&amp;author=E. Hatzakis&amp;author=K. Guo&amp;author=M. Carver&amp;author=D. Yang&amp;volume=41&amp;publication_year=2013&amp;pages=10584-10592&amp;pmid=24005038&amp;doi=10.1093/nar/gkt784.80&amp;"/></mixed-citation></ref><ref id="bib81"><label>81.</label><mixed-citation id="sref81"><named-content content-type="citation-string">Schlag K., Steinhilber D., Karas M., Sorg B.L. Analysis of proximal ALOX5 promoter binding proteins by quantitative proteomics. FEBS J. 2020 doi: 10.1111/febs.15259.81.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1111/febs.15259.81"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="32096311"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=FEBS J.&amp;title=Analysis of proximal ALOX5 promoter binding proteins by quantitative proteomics&amp;author=K. Schlag&amp;author=D. Steinhilber&amp;author=M. Karas&amp;author=B.L. Sorg&amp;publication_year=2020&amp;pmid=32096311&amp;doi=10.1111/febs.15259.81&amp;"/></mixed-citation></ref><ref id="bib82"><label>82.</label><mixed-citation id="sref82"><named-content content-type="citation-string">Cer R.Z., Donohue D.E., Mudunuri U.S., Temiz N.A., Loss M.A., Starner N.J., Halusa G.N., Volfovsky N., Yi M., Luke B.T., et al.  Non-B DB v2.0: a database of predicted non-B DNA-forming motifs and its associated tools. Nucleic Acids Res. 2013;41:D94–D100. doi: 10.1093/nar/gks955.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gks955"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC3531222"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="23125372"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res.&amp;title=Non-B DB v2.0: a database of predicted non-B DNA-forming motifs and its associated tools&amp;author=R.Z. Cer&amp;author=D.E. Donohue&amp;author=U.S. Mudunuri&amp;author=N.A. Temiz&amp;author=M.A. Loss&amp;volume=41&amp;publication_year=2013&amp;pages=D94-D100&amp;pmid=23125372&amp;doi=10.1093/nar/gks955&amp;"/></mixed-citation></ref><ref id="bib83"><label>83.</label><mixed-citation id="sref83"><named-content content-type="citation-string">Hänsel-Hertsch R., Spiegel J., Marsico G., Tannahill D., Balasubramanian S. Genome-wide mapping of endogenous G-quadruplex DNA structures by chromatin immunoprecipitation and high-throughput sequencing. Nat. Protoc. 2018;13:551–564. doi: 10.1038/nprot.2017.150.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/nprot.2017.150"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="29470465"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat. Protoc.&amp;title=Genome-wide mapping of endogenous G-quadruplex DNA structures by chromatin immunoprecipitation and high-throughput sequencing&amp;author=R. Hänsel-Hertsch&amp;author=J. Spiegel&amp;author=G. Marsico&amp;author=D. Tannahill&amp;author=S. Balasubramanian&amp;volume=13&amp;publication_year=2018&amp;pages=551-564&amp;pmid=29470465&amp;doi=10.1038/nprot.2017.150&amp;"/></mixed-citation></ref><ref id="bib84"><label>84.</label><mixed-citation id="sref84"><named-content content-type="citation-string">Georgakopoulos-Soares I., Jain N., Gray J.M., Hemberg M. MPRAnator: a web-based tool for the design of massively parallel reporter assay experiments. Bioinformatics. 2017;33:137–138. doi: 10.1093/bioinformatics/btw584.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/bioinformatics/btw584"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC5198521"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="27605100"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Bioinformatics&amp;title=MPRAnator: a web-based tool for the design of massively parallel reporter assay experiments&amp;author=I. Georgakopoulos-Soares&amp;author=N. Jain&amp;author=J.M. Gray&amp;author=M. Hemberg&amp;volume=33&amp;publication_year=2017&amp;pages=137-138&amp;pmid=27605100&amp;doi=10.1093/bioinformatics/btw584&amp;"/></mixed-citation></ref><ref id="bib85"><label>85.</label><mixed-citation id="sref85"><named-content content-type="citation-string">Quinlan A.R., Hall I.M. BEDTools: a flexible suite of utilities for comparing genomic features. Bioinformatics. 2010;26:841–842. doi: 10.1093/bioinformatics/btq033.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/bioinformatics/btq033"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC2832824"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="20110278"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Bioinformatics&amp;title=BEDTools: a flexible suite of utilities for comparing genomic features&amp;author=A.R. Quinlan&amp;author=I.M. Hall&amp;volume=26&amp;publication_year=2010&amp;pages=841-842&amp;pmid=20110278&amp;doi=10.1093/bioinformatics/btq033&amp;"/></mixed-citation></ref><ref id="bib86"><label>86.</label><mixed-citation id="sref86"><named-content content-type="citation-string">Li H. Aligning sequence reads, clone sequences and assembly contigs with BWA-MEM. arXiv. 2013 Preprint at. 1303.3997v2.90.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=arXiv&amp;title=Aligning sequence reads, clone sequences and assembly contigs with BWA-MEM&amp;author=H. Li&amp;publication_year=2013&amp;"/></mixed-citation></ref><ref id="bib87"><label>87.</label><mixed-citation id="sref87"><named-content content-type="citation-string">Grant C.E., Bailey T.L., Noble W.S. FIMO: scanning for occurrences of a given motif. Bioinformatics. 2011;27:1017–1018. doi: 10.1093/bioinformatics/btr064.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/bioinformatics/btr064"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC3065696"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="21330290"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Bioinformatics&amp;title=FIMO: scanning for occurrences of a given motif&amp;author=C.E. Grant&amp;author=T.L. Bailey&amp;author=W.S. Noble&amp;volume=27&amp;publication_year=2011&amp;pages=1017-1018&amp;pmid=21330290&amp;doi=10.1093/bioinformatics/btr064&amp;"/></mixed-citation></ref><ref id="bib88"><label>88.</label><mixed-citation id="sref88"><named-content content-type="citation-string">Yu G., Wang L.-G., Han Y., He Q.-Y. clusterProfiler: an R package for comparing biological themes among gene clusters. OMICS. 2012;16:284–287. doi: 10.1089/omi.2011.0118.87.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1089/omi.2011.0118.87"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC3339379"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="22455463"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=OMICS&amp;title=clusterProfiler: an R package for comparing biological themes among gene clusters&amp;author=G. Yu&amp;author=L.-G. Wang&amp;author=Y. Han&amp;author=Q.-Y. He&amp;volume=16&amp;publication_year=2012&amp;pages=284-287&amp;pmid=22455463&amp;doi=10.1089/omi.2011.0118.87&amp;"/></mixed-citation></ref><ref id="bib89"><label>89.</label><mixed-citation id="sref89"><named-content content-type="citation-string">Jain A., Tuteja G. TissueEnrich: tissue-specific gene enrichment analysis. Bioinformatics. 2019;35:1966–1967. doi: 10.1093/bioinformatics/bty890.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/bioinformatics/bty890"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6546155"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="30346488"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Bioinformatics&amp;title=TissueEnrich: tissue-specific gene enrichment analysis&amp;author=A. Jain&amp;author=G. Tuteja&amp;volume=35&amp;publication_year=2019&amp;pages=1966-1967&amp;pmid=30346488&amp;doi=10.1093/bioinformatics/bty890&amp;"/></mixed-citation></ref><ref id="bib90"><label>90.</label><mixed-citation id="sref90"><named-content content-type="citation-string">Virtanen P., Gommers R., Oliphant T.E., Haberland M., Reddy T., Cournapeau D., Burovski E., Peterson P., Weckesser W., Bright J., et al.  SciPy 1.0: fundamental algorithms for scientific computing in Python. Nat. Methods. 2020;17:261–272. doi: 10.1038/s41592-019-0686-2.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/s41592-019-0686-2"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC7056644"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="32015543"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat. Methods&amp;title=SciPy 1.0: fundamental algorithms for scientific computing in Python&amp;author=P. Virtanen&amp;author=R. Gommers&amp;author=T.E. Oliphant&amp;author=M. Haberland&amp;author=T. Reddy&amp;volume=17&amp;publication_year=2020&amp;pages=261-272&amp;pmid=32015543&amp;doi=10.1038/s41592-019-0686-2&amp;"/></mixed-citation></ref><ref id="bib91"><label>91.</label><mixed-citation id="sref91"><named-content content-type="citation-string">Inoue F., Kreimer A., Ashuach T., Ahituv N., Yosef N. Identification and massively parallel characterization of regulatory elements driving neural induction. Cell Stem Cell. 2019;25:713–727.e10. doi: 10.1016/j.stem.2019.09.010.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.stem.2019.09.010"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6850896"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="31631012"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Cell Stem Cell&amp;title=Identification and massively parallel characterization of regulatory elements driving neural induction&amp;author=F. Inoue&amp;author=A. Kreimer&amp;author=T. Ashuach&amp;author=N. Ahituv&amp;author=N. Yosef&amp;volume=25&amp;publication_year=2019&amp;pages=713-727.e10&amp;pmid=31631012&amp;doi=10.1016/j.stem.2019.09.010&amp;"/></mixed-citation></ref><ref id="bib92"><label>92.</label><mixed-citation id="sref92"><named-content content-type="citation-string">Chan K.L., Peng B., Umar M.I., Chan C.-Y., Sahakyan A.B., Le M.T.N., Kwok C.K. Structural analysis reveals the formation and role of RNA G-quadruplex structures in human mature microRNAs. Chem. Commun. 2018;54:10878–10881. doi: 10.1039/c8cc04635b.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1039/c8cc04635b"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="30204160"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Chem. Commun.&amp;title=Structural analysis reveals the formation and role of RNA G-quadruplex structures in human mature microRNAs&amp;author=K.L. Chan&amp;author=B. Peng&amp;author=M.I. Umar&amp;author=C.-Y. Chan&amp;author=A.B. Sahakyan&amp;volume=54&amp;publication_year=2018&amp;pages=10878-10881&amp;pmid=30204160&amp;doi=10.1039/c8cc04635b&amp;"/></mixed-citation></ref><ref id="bib93"><label>93.</label><mixed-citation id="sref93"><named-content content-type="citation-string">Chan C.-Y., Umar M.I., Kwok C.K. Spectroscopic analysis reveals the effect of a single nucleotide bulge on G-quadruplex structures. Chem. Commun. 2019;55:2616–2619. doi: 10.1039/c8cc09929d.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1039/c8cc09929d"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="30724299"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Chem. Commun.&amp;title=Spectroscopic analysis reveals the effect of a single nucleotide bulge on G-quadruplex structures&amp;author=C.-Y. Chan&amp;author=M.I. Umar&amp;author=C.K. Kwok&amp;volume=55&amp;publication_year=2019&amp;pages=2616-2619&amp;pmid=30724299&amp;doi=10.1039/c8cc09929d&amp;"/></mixed-citation></ref></ref-list></sec></sec><sec id="_ad93_" xml:lang="en" sec-type="associated-data" disp-level="1"><title>Associated Data</title><sec id="_adsm93_" xml:lang="en" sec-type="supplementary-materials" disp-level="2"><title>Supplementary Materials</title><supplementary-material id="db_ds_supplementary-material1_reqid_" position="float"><?disp-level 2?><caption><title>Document S1. Figures S1–S15 and Tables S1 and S2</title></caption><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="mmc1.pdf" mimetype="application" mime-subtype="pdf"><?cloudpmc-path 393b/9903758/09dd25e70914/mmc1.pdf?><?cloudpmc-bucket app?><?size 3262927?></media></supplementary-material><supplementary-material id="db_ds_supplementary-material2_reqid_" position="float"><?disp-level 2?><caption><title>Document S2. Transparent peer review</title><p>for Georgakopoulos-Soares et al.</p></caption><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="mmc2.pdf" mimetype="application" mime-subtype="pdf"><?cloudpmc-path 393b/9903758/602219a9ba2e/mmc2.pdf?><?cloudpmc-bucket app?><?size 664694?></media></supplementary-material><supplementary-material id="db_ds_supplementary-material3_reqid_" position="float"><?disp-level 2?><caption><title>Document S3. Article plus supplemental information</title></caption><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="mmc3.pdf" mimetype="application" mime-subtype="pdf"><?cloudpmc-path 393b/9903758/8fcf4e39ab25/mmc3.pdf?><?cloudpmc-bucket app?><?size 10511800?></media></supplementary-material></sec><sec id="_adda93_" xml:lang="en" sec-type="data-availability-statement" disp-level="2"><title>Data Availability Statement</title><p>The MPRA data for the NPC cell line targeting autism-related loci and the MPRA data for the non-B DNA associated loci in HEK-293T and K562 cell lines are deposited in NCBI BioProject with accession number PRJNA763774. The MPRA data for HEPG2 and K562 cell lines (<xref rid="fig3" ref-type="fig">Figure 3</xref>) have been deposited in the ENCODE portal with IDs ENCSR463IRX and ENCSR460LZI.</p><p>All original code and data tables to perform the analyses can be found on the GitHub page (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://github.com/IliasGeoSo/High_Throughput_MPRA_Non_B_DNA" ext-link-type="uri">https://github.com/IliasGeoSo/High_Throughput_MPRA_Non_B_DNA</ext-link>) and are publicly available.</p><p>Any additional information required to reanalyze the data reported in this paper is available from the lead contact upon request.</p></sec></sec></body></article>