<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.3 20210610//EN"  "JATS-archivearticle1-3-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.3"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">97682</article-id><article-id pub-id-type="doi">10.7554/eLife.97682</article-id><article-id pub-id-type="doi" specific-use="version">10.7554/eLife.97682.3</article-id><article-version article-version-type="publication-state">version of record</article-version><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Computational and Systems Biology</subject></subj-group><subj-group subj-group-type="heading"><subject>Genetics and Genomics</subject></subj-group></article-categories><title-group><article-title>Multiplexed assays of human disease-relevant mutations reveal UTR dinucleotide composition as a major determinant of RNA stability</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name><surname>Su</surname><given-names>Jia-Ying</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-5934-5458</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund2"/><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" equal-contrib="yes"><name><surname>Wang</surname><given-names>Yun-Lin</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="other" rid="fund3"/><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Hsieh</surname><given-names>Yu-Tung</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund4"/><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Chang</surname><given-names>Yu-Chi</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund4"/><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Yang</surname><given-names>Cheng-Han</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund4"/><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Kang</surname><given-names>YoonSoon</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund3"/><xref ref-type="fn" rid="con6"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Huang</surname><given-names>Yen-Tsung</given-names></name><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund2"/><xref ref-type="other" rid="fund5"/><xref ref-type="fn" rid="con7"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes"><name><surname>Lin</surname><given-names>Chien-Ling</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-5730-799X</contrib-id><email>mbcllin@gate.sinica.edu.tw</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="pa1">‡</xref><xref ref-type="other" rid="fund3"/><xref ref-type="other" rid="fund4"/><xref ref-type="fn" rid="con8"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/047sbcx71</institution-id><institution>Institute of Molecular Biology, Academia Sinica</institution></institution-wrap><addr-line><named-content content-type="city">Taipei</named-content></addr-line><country>Taiwan</country></aff><aff id="aff2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/044gv5910</institution-id><institution>Institute of Statistical Science, Academia Sinica</institution></institution-wrap><addr-line><named-content content-type="city">Taipei</named-content></addr-line><country>Taiwan</country></aff><aff id="aff3"><label>3</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/05bxb3784</institution-id><institution>Bioinformatics Program, Taiwan International Graduate Program, Academia Sinica</institution></institution-wrap><addr-line><named-content content-type="city">Taipei</named-content></addr-line><country>Taiwan</country></aff><aff id="aff4"><label>4</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00se2k293</institution-id><institution>Institute of Biomedical Informatics, National Yang Ming Chiao Tung University</institution></institution-wrap><addr-line><named-content content-type="city">Taipei</named-content></addr-line><country>Taiwan</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Calarco</surname><given-names>John</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/03dbr7087</institution-id><institution>University of Toronto</institution></institution-wrap><country>Canada</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Moses</surname><given-names>Alan M</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/03dbr7087</institution-id><institution>University of Toronto</institution></institution-wrap><country>Canada</country></aff></contrib></contrib-group><author-notes><fn fn-type="con" id="equal-contrib1"><p><sup>†</sup>These authors contributed equally</p></fn><fn fn-type="present-address" id="pa1"><label>‡</label><p>Institute of Molecular Biology, Academia Sinica, Taipei City, Taiwan</p></fn></author-notes><pub-date publication-format="electronic" date-type="publication"><day>18</day><month>02</month><year>2025</year></pub-date><volume>13</volume><elocation-id>RP97682</elocation-id><history><date date-type="sent-for-review" iso-8601-date="2024-04-10"><day>10</day><month>04</month><year>2024</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint.</event-desc><date date-type="preprint" iso-8601-date="2024-04-10"><day>10</day><month>04</month><year>2024</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2024.04.10.588845"/></event><event><event-desc>This manuscript was published as a reviewed preprint.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2024-06-18"><day>18</day><month>06</month><year>2024</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.97682.1"/></event><event><event-desc>The reviewed preprint was revised.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2025-01-16"><day>16</day><month>01</month><year>2025</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.97682.2"/></event></pub-history><permissions><copyright-statement>© 2024, Su, Wang et al</copyright-statement><copyright-year>2024</copyright-year><copyright-holder>Su, Wang et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-97682-v1.pdf"/><self-uri content-type="figures-pdf" xlink:href="elife-97682-figures-v1.pdf"/><abstract><p>Untranslated regions (UTRs) contain crucial regulatory elements for RNA stability, translation and localization, so their integrity is indispensable for gene expression. Approximately 3.7% of genetic variants associated with diseases occur in UTRs, yet a comprehensive understanding of UTR variant functions remains limited due to inefficient experimental and computational assessment methods. To systematically evaluate the effects of UTR variants on RNA stability, we established a massively parallel reporter assay on 6555 UTR variants reported in human disease databases. We examined the RNA degradation patterns mediated by the UTR library in two cell lines, and then applied LASSO regression to model the influential regulators of RNA stability. We found that UA dinucleotides and UA-rich motifs are the most prominent destabilizing element. Gain of UA dinucleotide outlined mutant UTRs with reduced stability. Studies on endogenous transcripts indicate that high UA-dinucleotide ratios in UTRs promote RNA degradation. Conversely, elevated GC content and protein binding on UA dinucleotides protect high-UA RNA from degradation. Further analysis reveals polarized roles of UA-dinucleotide-binding proteins in RNA protection and degradation. Furthermore, the UA-dinucleotide ratio of both UTRs is a common characteristic of genes in innate immune response pathways, implying a coordinated stability regulation through UTRs at the transcriptomic level. We also demonstrate that stability-altering UTRs are associated with changes in biobank-based health indices, underscoring the importance of precise UTR regulation for wellness. Our study highlights the importance of RNA stability regulation through UTR primary sequences, paving the way for further exploration of their implications in gene networks and precision medicine.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>untranslated region</kwd><kwd>RNA stability</kwd><kwd>UTR variants</kwd><kwd>massively parallel reporter assay</kwd><kwd>statistical learning</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Human</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100001869</institution-id><institution>Academia Sinica</institution></institution-wrap></funding-source><award-id>AS-CDA-108-M03</award-id><principal-award-recipient><name><surname>Su</surname><given-names>Jia-Ying</given-names></name><name><surname>Huang</surname><given-names>Yen-Tsung</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100001869</institution-id><institution>Academia Sinica</institution></institution-wrap></funding-source><award-id>AS-PH-109-01-3</award-id><principal-award-recipient><name><surname>Su</surname><given-names>Jia-Ying</given-names></name><name><surname>Huang</surname><given-names>Yen-Tsung</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100004737</institution-id><institution>National Health Research Institutes</institution></institution-wrap></funding-source><award-id>NHRI-EX112-10908BC</award-id><principal-award-recipient><name><surname>Wang</surname><given-names>Yun-Lin</given-names></name><name><surname>Kang</surname><given-names>YoonSoon</given-names></name><name><surname>Lin</surname><given-names>Chien-Ling</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100020950</institution-id><institution>National Science and Technology Council</institution></institution-wrap></funding-source><award-id>MOST 111-2628-B-001-003</award-id><principal-award-recipient><name><surname>Hsieh</surname><given-names>Yu-Tung</given-names></name><name><surname>Chang</surname><given-names>Yu-Chi</given-names></name><name><surname>Yang</surname><given-names>Cheng-Han</given-names></name><name><surname>Lin</surname><given-names>Chien-Ling</given-names></name></principal-award-recipient></award-group><award-group id="fund5"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100020950</institution-id><institution>National Science and Technology Council</institution></institution-wrap></funding-source><award-id>108-2118-M-001-013-MY5</award-id><principal-award-recipient><name><surname>Huang</surname><given-names>Yen-Tsung</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>The UA-dinucleotide ratio in UTRs is negatively correlated with RNA stability both in the massively parallel reporter assay and in vivo, and is prevalent in fast-turnover genes.</meta-value></custom-meta><custom-meta specific-use="meta-only"><meta-name>publishing-route</meta-name><meta-value>prc</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Untranslated regions of RNAs are indispensable for post-transcriptional regulation of gene expression</title><p>A mature RNA consists of three regions - a 5' untranslated region (5' UTR), the protein coding region, and a 3' untranslated region (3' UTR; <xref ref-type="bibr" rid="bib42">Mignone et al., 2002</xref>). UTRs are indispensable for gene expression. For most mRNAs of higher eukaryotes, the 5’ UTR is essential for ribosome entry and the 3’ UTR is responsible for polyadenylation to stabilize the RNA and enhance its translation efficiency. The average length of the 5’ and 3’ UTRs in human is ~210 nucleotides (nt) and ~1030 nt, respectively, with mean 3' UTR length being more diverse among species and higher eukaryotes hosting longer 3' UTRs (<xref ref-type="bibr" rid="bib49">Pesole et al., 2001</xref>). These structural differences among species are consistent with genomic complexity and more complex post-transcriptional regulatory mechanisms. The UTRs contain <italic>cis</italic>-regulatory elements that contribute to post-transcriptionally regulating gene expression, such as via protein translation control, subcellular mRNA localization, and mRNA stability upon interactions with <italic>trans</italic>-acting factors such as RNA-binding proteins (RBPs) and microRNAs (miRNAs) (<xref ref-type="bibr" rid="bib5">Barrett et al., 2012</xref>).</p></sec><sec id="s1-2"><title>UTRs control RNA stability</title><p>RNA stability regulation serves as an mRNA quality control mechanism (e.g. nonsense-mediated mRNA decay) to gate protein production (<xref ref-type="bibr" rid="bib55">Schoenberg and Maquat, 2012</xref>). Decades of research have shown that the <italic>cis</italic>-regulatory elements in UTRs affect mRNA stability. These sequence elements control mRNA integrity and can trigger mRNA degradation pathways by interacting with RBPs or small regulatory RNAs such as miRNAs and small interfering RNAs (siRNAs; <xref ref-type="bibr" rid="bib21">Garneau et al., 2007</xref>; <xref ref-type="bibr" rid="bib43">Mitchell and Tollervey, 2001</xref>). For instance, the miRNA-Argonaute complex recruits the CCR4-NOT (Carbon Catabolite Repression—Negative On TATA-less) complex to initiate deadenylation and decay (<xref ref-type="bibr" rid="bib30">Huntzinger and Izaurralde, 2011</xref>). Another example is AU-rich elements (AREs, sequence elements rich in adenosine and uridine) present in the 3' UTRs of many mRNAs that provide binding sites for ARE-binding proteins that trigger the RNA degradation pathway (<xref ref-type="bibr" rid="bib21">Garneau et al., 2007</xref>). Many ARE-binding proteins have been characterized to date that are involved in stability regulation of ARE-hosting mRNAs (<xref ref-type="bibr" rid="bib4">Barreau et al., 2005</xref>), including Tristetrapolin (TTP), Butyrate Response Factor 1 (BRF1), Heterogeneous Nuclear Ribonucleoprotein (hnRNP) D (also known as AU-Rich Element-Binding Protein 1, AUF1), and KH-type splicing regulatory protein (KSRP), all of which destabilize mRNA, unlike ELAV Like RNA Binding Protein 1 (ELAVL1, also known as HuR) that stabilizes it (<xref ref-type="bibr" rid="bib55">Schoenberg and Maquat, 2012</xref>). This regulatory mechanism necessitates physical access to the sequence elements, so structural contexts are critical (<xref ref-type="bibr" rid="bib47">Paschoud et al., 2006</xref>). Complex secondary structures such as RNA G-quadruplexes (RG4s) and pseudoknots may also play a role in stability regulation. RG4s are enriched in UTRs where they regulate many post-transcriptional regulatory processes, including RNA stability (<xref ref-type="bibr" rid="bib17">Dumas et al., 2021</xref>), although the detailed mechanisms remain to be elucidated. Although 3’ UTR-mediated regulation gains more attention, 5’ UTRs may also contribute to mRNA stability regulation. For instance, upstream open reading frames (uORF) in 5’ UTRs can facilitate RNA decay in a translation-dependent manner, and RG4s in 5’ UTRs reduce RNA stability in a ribosome-independent manner (<xref ref-type="bibr" rid="bib31">Jia et al., 2020</xref>).</p></sec><sec id="s1-3"><title>Multiplexed reporter assays to elucidate UTR stability control</title><p>Efforts have been made to elucidate RNA stability regulation on a genome-wide scale to understand its general regulatory mechanisms. However, due to the complexity of post-transcriptional regulation, features inferred from endogenous steady-state RNA levels can be obscured by other dominant factors. For instance, studies examining cellular RNA degradation alongside transcriptional inhibition have shown that coding region length and ribosome occupancy are key determinants of RNA stability (<xref ref-type="bibr" rid="bib44">Neymotin et al., 2015</xref>). Additionally, RNA stability inferred from steady-state RNA concentrations normalized against transcription rates has indicated that splice junction density is a major factor promoting RNA stability (<xref ref-type="bibr" rid="bib2">Agarwal and Kelley, 2022</xref>; <xref ref-type="bibr" rid="bib10">Blumberg et al., 2021</xref>). Nonetheless, these studies have not systematically decoded the influence of primary sequences on RNA stability regulation. Therefore, efforts have been made to establish massively parallel reporter assays for bulk-synthesized UTRs to unveil their governance of RNA regulation in various species. Bidirectional promoters driving a control transcript and a green fluorescence protein (GFP) hosting various test UTRs were first used to evaluate the effect of the UTRs on fluorescence signals (<xref ref-type="bibr" rid="bib45">Oikonomou et al., 2014</xref>; <xref ref-type="bibr" rid="bib54">Sample et al., 2019</xref>; <xref ref-type="bibr" rid="bib62">Vainberg Slutskin et al., 2018</xref>; <xref ref-type="bibr" rid="bib66">Wissink et al., 2016</xref>; <xref ref-type="bibr" rid="bib68">Zhao et al., 2014</xref>). Alternatively, various UTRs have been inserted into plasmids prior to cellular expression, and then the DNA and RNA levels of each construct have been compared to infer the effect of the UTRs on RNA expression (<xref ref-type="bibr" rid="bib24">Griesemer et al., 2021</xref>; <xref ref-type="bibr" rid="bib36">Litterman et al., 2019</xref>; <xref ref-type="bibr" rid="bib57">Siegel et al., 2022</xref>). Nevertheless, because the steady-state level is a result of production and decay, these approaches cannot differentiate the effect of the UTRs on transcription, stability and, in some cases, even protein production, greatly limiting scientific interpretation. Injections or transfections of in vitro-transcribed RNAs have been used to study RNA stability. These studies have identified AREs and miRNA-binding sites as destabilizing elements and U-rich sequences as stabilizing elements in zebrafish embryos (<xref ref-type="bibr" rid="bib52">Rabani et al., 2017</xref>; <xref ref-type="bibr" rid="bib64">Vejnar et al., 2019</xref>; <xref ref-type="bibr" rid="bib68">Zhao et al., 2014</xref>). Similar studies in human cell lines have shown that RG4 structures and A-rich sequences in 5’ UTRs promote RNA decay (<xref ref-type="bibr" rid="bib31">Jia et al., 2020</xref>).</p></sec><sec id="s1-4"><title>UTR mutations and disease</title><p>UTR sequence variation affects mRNA stability, translation and localization. RNA dysregulation arising from mutations in UTRs significantly and negatively affects gene regulation, which can promote phenotypical and even pathological change. According to the NHGRI-EBI GWAS Catalog, genome-wide association studies up to 2018 had uncovered that ~3.7% of disease risk/quantitative trait-associated genetic variants are located in UTRs (<xref ref-type="bibr" rid="bib40">MacArthur et al., 2017</xref>; <xref ref-type="bibr" rid="bib60">Steri et al., 2018</xref>). Indeed, certain studies have provided evidence that alterations to even a single nucleotide in a UTR can impact mRNA translation or transcript half-life in disease contexts. For example, a single nucleotide substitution of the 36th position in the 5' UTR of <italic>transforming growth factor-β3</italic> (<italic>TGFβ3</italic>; OMIM# 190230) or its 1723th 3’ UTR position is associated with arrhythmogenic right ventricular cardiomyopathy (<xref ref-type="bibr" rid="bib8">Beffagna et al., 2005</xref>). Similarly, point mutation of the <italic>GFPT1</italic> 3’ UTR results in congenital myasthenic syndrome. GFPT1 (Glutamine-Fructose-6-Phosphate Transaminase 1) is the rate-limiting enzyme for hexosamine biosynthesis, and mutation of its 3’ UTR results in a 90% reduction in protein production, potentially due to gain of a miRNA binding site (<xref ref-type="bibr" rid="bib18">Dusl et al., 2015</xref>). Collectively, these studies support that the precise RNA regulation exerted by UTRs plays a critical role in controlling gene expression, with UTR mutations potentially eliciting divergent phenotypes and even severe disease.</p><p>Despite their disease relevance, a comprehensive overview of the pathogenic effects of UTR mutations is still lacking. The medical genetics community has earnestly advocated for consideration of ‘UTR variants in genetic diagnostic procedures’ (<xref ref-type="bibr" rid="bib18">Dusl et al., 2015</xref>). Specifically, the impact of 5’ and 3’ UTR sequence variations on RNA stability regulation remains unclear. While there have been efforts to examine how 3’ UTR variants affect steady-state RNA levels, systematic assessments of the explicit effects of disease-relevant variants in both UTRs on stability regulation are still absent. To examine potentially pathogenic UTR mutations and their links to RNA stability, we developed a massively parallel reporter assay in which human 5’/3’ UTRs with disease-relevant mutations were generated in vitro, ligated with the enhanced green fluorescence protein (EGFP) coding region, and then directly transfected into human cell lines to assess their decay patterns by next-generation sequencing. Taking redundancy and interdependency of regulatory features into consideration, our approach identified that UA dinucleotides are the most influential destabilizing sequence element for RNA stability. Moreover, we found that joint regulation by 5’ and 3’ UTRs shapes the expression kinetics of functional gene groups. Our study unveils RNA stability determinants and delineates the importance of precise UTR control in maintaining harmonious genetic networks for human health.</p></sec></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>Massively parallel reporter assay (MPRA) for RNA stability</title><p>The above-described genome-wide analysis prompted the hypothesis that UTR variants that disrupt critical RNA regulatory elements may be linked to pathogenicity. Since one of the major roles of UTRs is to control RNA stability, we hypothesized that disease-relevant UTR variants may alter RNA stability. Therefore, we designed 6555 pairs of 155-nt UTR fragments centering on the variant collected from the HGMD and ClinVar disease databases, and performed time-course assays to examine the relative stability. First, we fused UTRs to the EGFP coding regions, transcribed them in vitro, and then transfected them into human embryonic kidney cells (HEK293T) or neuroblastoma cells (SH-SY5Y), considering pervasive neurological diseases in the mutation collection. Then, we monitored the relative abundance of the reference (ref) and mutant (mt) alleles by amplicon sequencing over a time course (30, 75, 120 min for HEK293T; 20, 40, 60 min for SH-SY5Y) (<xref ref-type="fig" rid="fig1">Figure 1A</xref>). Primers targeting common reporter regions were utilized for the retrieval of UTR sequences at each time point. We estimated the decay constant and half-life (t<sub>1/2</sub>) of each UTR according to its relative abundance over time (see Methods). We defined stability-altering variants as those for which the decay constants significantly changed relative to their ref counterparts, as determined by weighted linear regression (see Methods). We observed that variants in both 5’ and 3’ UTRs significantly altered RNA half-life, with slightly more variants having a negative impact on RNA stability (<xref ref-type="fig" rid="fig1">Figure 1B&amp;C</xref>; <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>).</p><fig-group><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Massively parallel reporter assay (MPRA) to determine the effects of UTR variants on RNA stability.</title><p>(<bold>A</bold>) MPRA workflow. In brief, 6555 reference (ref) and mutant (mt) UTR pairs were synthesized in bulk, ligated with promoters and reporter sequences, in vitro-transcribed into capped and tailed RNAs, transfected into human cell lines, and then the remaining RNAs were collected over a time-course. The collected RNAs were reverse-transcribed, amplified and sequenced to resolve the genotype of each UTR. The unique sequences were used to calculate RNA half-life. Mutational effects were inferred from those pairs significantly differing in half-life (see Methods). (<bold>B</bold>) Volcano plot of MPRA data from three repeated experiments. The colored dots indicate significant stability-altering variants. (<bold>C</bold>) Examples of significant stability-altering UTR mutations in both UTR types. Data are presented as mean ± SD (n = 3 experimental replicates).</p><p><supplementary-material id="fig1sdata1"><label>Figure 1—source data 1.</label><caption><title>UTR stability through a time course (labeled image).</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-97682-fig1-data1-v1.zip"/></supplementary-material></p><p><supplementary-material id="fig1sdata2"><label>Figure 1—source data 2.</label><caption><title>UTR stability through a time course (original image).</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-97682-fig1-data2-v1.zip"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-fig1-v1.tif"/></fig><fig id="fig1s1" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 1.</label><caption><title>Correlation of experimental results.</title><p>(<bold>A</bold>) Spearman’s correlation of the sequencing read counts among three biological repeats in HEK293T cells. (<bold>B</bold>) Spearman’s correlation of the sequencing outcomes among three repeated experiments in SH-SY5Y cells. (<bold>C</bold>) In vitro polyadenylation (poly(A)) prior to transfection. (<bold>D</bold>) Comparisons of half-lives in log scale between UTRs and cell lines.</p><p><supplementary-material id="fig1s1sdata1"><label>Figure 1—figure supplement 1—source data 1.</label><caption><title>Polyadenylation of in vitro transcribed RNA (original image).</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-97682-fig1-figsupp1-data1-v1.zip"/></supplementary-material></p><p><supplementary-material id="fig1s1sdata2"><label>Figure 1—figure supplement 1—source data 2.</label><caption><title>Polyadenylation of in vitro transcribed RNA (labeled image).</title></caption><media mimetype="application" mime-subtype="zip" xlink:href="elife-97682-fig1-figsupp1-data2-v1.zip"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-fig1-figsupp1-v1.tif"/></fig></fig-group><p>Our results from three independent experiments are highly consistent, with a Spearman’s correlation coefficient &gt;0.93 (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1A and B</xref>). From among 3,700 pairs of valid comparisons, 40 (1.1%) and 839 (22.8%) variants displayed significantly altered stability compared to their ref counterparts in HEK293T and SH-SY5Y cells, respectively (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1B and C</xref>). Thus, we observed a significant effect of UTR variation on RNA stability, but their regulatory impact was strikingly divergent between the two tested cell lines (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1D</xref>). We attribute this divergence to potential differences in translation capacity, as well as variations in the composition and concentration of RNA-binding proteins and RNases between the two cell lines.</p></sec><sec id="s2-2"><title>Impact of bi-functional AREs on RNA stability</title><p>AREs are well-recognized regulatory motifs controlling RNA stability. Accordingly, we examined if the regulatory effect of AREs could be captured by our MPRA approach. Multiple approaches have revealed AREs as exerting a destabilizing effect on RNA stability (<xref ref-type="bibr" rid="bib4">Barreau et al., 2005</xref>). However, ARE motifs and ARE-binding proteins are diverse, so the impact of binding may vary considerably. Therefore, we examined the effect of AREs on RNA stability of the ref alleles according to specific sequence content. Based on the definition of AREsite2 (<ext-link ext-link-type="uri" xlink:href="http://nibiru.tbi.univie.ac.at/AREsite2">http://nibiru.tbi.univie.ac.at/AREsite2</ext-link>), we categorized AREs as either WUUUW or its longer derivatives, UUUGUUU or AWUAAA (W:A/G; <xref ref-type="bibr" rid="bib19">Fallmann et al., 2016</xref>). We observed that AREs in either the 5’ or 3’ UTRs generally destabilized RNA (<xref ref-type="fig" rid="fig2">Figure 2A</xref>; <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>; <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>). More specifically, AUUUA/AUUA-containing AREs are associated with RNA destabilization when present in either UTR type, whereas in SH-SY5Y cells, extremely U-dominant AREs (U<sub>8-10</sub>A<sub>1-2</sub>) stabilized it (<xref ref-type="fig" rid="fig2">Figure 2B</xref>), similar to the stabilizing effect of U-stretches described for zebrafish (<xref ref-type="bibr" rid="bib52">Rabani et al., 2017</xref>; <xref ref-type="bibr" rid="bib64">Vejnar et al., 2019</xref>). These results suggest that although mostly destabilizing, AREs can play dual roles in regulating RNA stability by recruiting binding proteins of diverse functions.</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>AREs generally destabilize RNAs (except extremely U-rich AREs).</title><p>(<bold>A</bold>) AREs of both UTR types destabilize RNA. (<bold>B</bold>) The ten most influential AREs in terms of RNA stability in SH-SY5Y cells. Coefficients are determined by regression analysis, representing the effect size of each motif.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-fig2-v1.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Various destabilizing effects of AREs in HEK293T, related to <xref ref-type="fig" rid="fig2">Figure 2</xref>.</title><p>(<bold>A</bold>) AREs of both UTRs destabilize RNA. (<bold>B</bold>) The ten most influential AREs of RNA stability.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-fig2-figsupp1-v1.tif"/></fig></fig-group></sec><sec id="s2-3"><title>Modeling the impact of UTR-mediated regulation on RNA stability</title><p>Given our discovery that the effect of AREs is heavily dependent on sequence content, we decided to further explore the effects of other sequence elements, that is beyond known regulatory motifs, in more detail. Since most reported RBP motifs are 6-mers, we initiated a search for novel motifs by analyzing the presence of all 7-mers in our massively parallel reporter assay (MPRA) library, correlating their occurrence with mRNA half-life. For those with significant stabilizing or destabilizing effects, we clustered similar ones into motifs (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref>). The motifs suggest a G-rich stabilizing profile and an A-rich destabilizing profile, with the latter being more pronounced for the 3’ UTR. Next, to gain a comprehensive understanding of the contextual effect of each sequence element, we took advantage of LASSO regression, which minimizes coefficients of explanatory factors to select the most influential factors. We considered as many factors as possible to explain the half-life of our ref UTR libraries, including primary sequences, RBP binding sites (ATtRACT database <xref ref-type="bibr" rid="bib22">Giudice et al., 2016</xref>), miRNA seed sites, secondary structures, and folding energy. Furthermore, to avoid collinearity confounding our model, for example the effects of very similar factors (such as ‘AA’ and ‘AAA’ sequences), we clustered the factors according to their properties, and then only one representative factor from within a cluster (i.e. the one with the highest correlation to half-life within a cluster) was subjected to LASSO regression (<xref ref-type="fig" rid="fig3">Figure 3A</xref>, <xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref>, <xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref> and Methods). LASSO regression renders as zero the coefficients of factors with minimal explanatory power (see Methods for details). Overall, we started with 1231 (5’ UTR) or 1475 (3’ UTR) factors, but only 5–19 factors were selected ultimately for each trained model (<xref ref-type="fig" rid="fig3">Figure 3B–E</xref>; <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>). The selected explanatory factors all represent small kmer motifs (k=2–3) and RBP binding motifs. Unexpectedly, we identified some unique regulatory factors in each cell line, indicating that RNA decay pathways are typically shared but can be strongly influenced by the cellular environment. Overall, motifs that are at least two nucleotides long proved critical for RNA stability, supporting the sequence specificity of the decay process.</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Inferential statistical analysis of RNA stability determinants.</title><p>(<bold>A</bold>) Workflow of variable selection to build models of influencers of RNA stability. (<bold>B–E</bold>) Influential regulators for the 5’ UTR library from HEK293T cells (<bold>B</bold>), the 3’ UTR library from HEK293T cells (<bold>C</bold>), the 5’ UTR library from SH-SY5Y cells (<bold>D</bold>), and the 3’ UTR library from SH-SY5Y cells (<bold>E</bold>). The error bars represent 95% confidence intervals of the coefficients. Note that the factors presented on the figure are representative of their respective clusters (see Methods and <xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref>).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-fig3-v1.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>Stabilizing or destabilizing motifs in UTRs.</title><p>7-mer motifs &gt;20 occurrences in the MPRA library were examined by LASSO regression to test their association with half-life. Motifs that were selected &gt;1600 times out of 2000 bootstraps (see Methods) are presented in this figure: (<bold>A</bold>) HEK293T 5’ UTR motifs. (<bold>B</bold>) HEK293T 3’ UTR motifs. (<bold>C</bold>) SH-SY5Y 5’ UTR motifs. (<bold>D</bold>) SH-SY5Y 3’ UTR motifs.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-fig3-figsupp1-v1.tif"/></fig><fig id="fig3s2" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 2.</label><caption><title>Workflow of variable selection to build models of stability influence.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-fig3-figsupp2-v1.tif"/></fig></fig-group><p>In both of the cell lines we tested, GU-rich sequences in 5’ UTRs stabilized RNAs (<xref ref-type="fig" rid="fig3">Figure 3B and D</xref>). In contrast, CA- and UG-repeat sequences—potential binding sites for Insulin Like Growth Factor 2 mRNA Binding Protein 3 (IGF2BP3) and CUGBP Elav-Like Family Member 1 (CELF1)—in 3’ UTRs proved the most destabilizing factors in HEK293T cells (<xref ref-type="fig" rid="fig3">Figure 3C</xref>). Moreover, we noticed that for both UTRs, UA dinucleotides and other UA-rich sequences—such as WWWWWW (W=A/U; a potential binding motif of Peptidylprolyl Isomerase E (PPIE)), AUUUA (a potential binding motif of ELAV Like RNA Binding Protein 1 (ELAVL1)), and UUUAUA (a potential binding motif of hnRNPA1)—are strongly destabilizing (<xref ref-type="fig" rid="fig3">Figure 3B-E</xref> and <xref ref-type="fig" rid="fig4">Figure 4A,C</xref>). UA dinucleotides and WWWWWW belong to the same cluster, but they were respectively selected for LASSO regression in the two cell lines because they displayed the highest explanatory power (largest coefficient by univariate regression) for RNA half-life in each cell line (<xref ref-type="fig" rid="fig3">Figures 3A</xref>, <xref ref-type="fig" rid="fig4">4B and D</xref>). Most prominently, UA dinucleotides in both UTRs overwhelmed other factors in robustly destabilizing RNAs in SH-SY5Y cells (<xref ref-type="fig" rid="fig3">Figure 3D and E</xref>). Therefore, UA dinucleotides seem to be a universal destabilizing motif, so we investigated how UA dinucleotides regulate RNA stability in further detail.</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>The UTR UA dinucleotides and UA-rich motifs are the most common and influential RNA destabilizing factor.</title><p>(<bold>A</bold>) Correlation of the 5’ UTR UA dinucleotide ratio and half-life. (<bold>B</bold>) Top 15 influential factors in the UA cluster of 5’ UTR. UTRs are arranged by half-life, and factors by their coefficient to half-life. Note that there are destabilizing factors (such as UA and AU dinucleotides) as well as stabilizing factors (such as GC content and G monomers) in this cluster. UA dinucleotide and WWWWWW (PPIE) (where W represents A/U) are representative of the cluster for modeling UTR stability in SH-SY5Y and HEK293T cells, respectively. (<bold>C</bold>) Correlation of the 3’ UTR UA dinucleotide ratio and half-life. (<bold>D</bold>) Top 15 influential factors in the UA cluster of 3’ UTR. (<bold>E</bold>) Mutational gain of a UA dinucleotide by 3’ UTRs significantly reduces RNA stability (lower panel). (<bold>F</bold>) Gain of UA dinucleotides in a random 5' UTR library led to RNA destabilization. We categorized pairs with a≥1.5 fold change as ’significant' (Sig) and those with less than this threshold as 'non-significant' (Non-sig). (<bold>G</bold>) High UA-nucleotide ratios of both UTRs reduce endogenous RNA stability in HEK293 cells. Q1-Q4 denote quantile groups categorized based on the UA-dinucleotide ratio.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-fig4-v1.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>UA cluster is the most potent destabilizing factor in each cell line.</title><p>(<bold>A</bold>) Correlation of top 15 5’ UTR influential factors of stability in the UA cluster. (<bold>B</bold>) Correlation of top 15 3’ UTR influential factors of stability in the UA cluster, related to <xref ref-type="fig" rid="fig4">Figure 4A–D</xref>. Factors are arranged by their coefficient to half-lives. Please note that there are destabilizing factors, such as UA and AU dinucleotides, and stabilizing factors, such as GC-content and G-monomer, in this cluster. UA dinucleotide and WWWWWW (PPIE, W represents A/T) represent the cluster to model UTR stability in SH-SY5Y and HEK293T cells, respectively. (<bold>C</bold>) Commonalities and composition of significant stability-regulating factors among the four experimental groups. (<bold>D</bold>) The workflow of sliding window analysis. A 10-mer window progressing 1 nt at a time calculates the TA/AT dinucleotide ratio within the window. Spearman’s correlation between TA/AT dinucleotide ratio and half-lives by groups estimated the importance of regional TA/AT dinucleotide ratio. (<bold>E</bold>) Results of (<bold>D</bold>) showed a higher correlation between TA/AT dinucleotide and half-lives in SH-SY5Y cells, and a moderate correlation at the end of 3’ UTR in HEK293T cells. (<bold>F</bold>) High TA-nucleotide ratios of both UTRs reduce endogenous RNA stability in K562 cells. Q1-Q4 denote quantile groups categorized based on the TA-dinucleotide ratio.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-fig4-figsupp1-v1.tif"/></fig></fig-group></sec><sec id="s2-4"><title>UA dinucleotides and UA-rich motifs are the most common and effective RNA destabilizing factor</title><p>UA dinucleotides proved to be the strongest stability determinant for both UTR types in SH-SY5Y cells. UA dinucleotides alone present a negative correlation with RNA stability, with a Pearson’s correlation coefficient of –0.287 for 5’ UTRs and –0.377 for 3’ UTRs (<xref ref-type="fig" rid="fig4">Figure 4A and C</xref>). UA-rich motifs (in the same cluster as UA dinucleotides) behave similarly to UA dinucleotides in regulating RNA stability, whereas GC-rich motifs have the opposite effect (<xref ref-type="fig" rid="fig4">Figure 4B and D</xref>; <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1A and B</xref>). Within the same cluster, UA dinucleotides and the WWWWWW motif were the strongest RNA stability regulators in each cell line (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>). Given the strong destabilizing effect of factors in the UA-associated cluster for both UTR types and both cell lines, we further analyzed their commonalities. An UpSet analysis revealed that all features contributing to RNA stability across four experimental groups (HEK293T 5’ UTRs, HEK293T 3’ UTRs, SH-SY5Y 5’ UTRs, SH-SY5Y 3’ UTRs) occur in the UA dinucleotide/WWWWWW cluster (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1C</xref>), indicating a universal destabilizing effect of UA-rich sequences. Next, to examine if there is a region-specific effect of UA and closely-related AU dinucleotides, we used a sliding window to establish the localization-associated relationship between the UA/AU dinucleotide ratio and RNA half-life. Correlation coefficients between UA/AU dinucleotide ratios and UTR stability were calculated for each window, and we assumed that regions displaying a strong correlation between UA/AU dinucleotide ratios and stability rank hosted UA/AU dinucleotides that control RNA stability (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1D</xref>). We found that UA/AU dinucleotides in the UTRs of SH-SY5Y cells were generally strongly correlated with RNA stability, but only weakly associated with RNA stability in HEK293T cells (apart from a relatively strong correlation at the ends of 3’ UTRs, implying a protective role against exonuclease digestion) (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1E</xref>). Together, these results support that UA dinucleotides are a common and prominent RNA destabilizing motif.</p><p>Next, we examined the effect of mutating the most effective destabilizing UA dinucleotide (resulting in dinucleotide gain or loss) in terms of altering RNA stability. We found a clear propensity for 3’ UTRs with stability loss to accumulate gain of UA dinucleotide mutations, compared to stabilizing or non-significant mutations (<xref ref-type="fig" rid="fig4">Figure 4E</xref>). To further validate the impact of UA dinucleotides, we curated a subset of oligo pairs from a 5' UTR random library (<xref ref-type="bibr" rid="bib31">Jia et al., 2020</xref>) with the sole difference being the presence of one additional UA dinucleotide, resulting in a discrepancy of one UA dinucleotide between them. This selection allowed us to investigate whether these variations influenced RNA half-lives. We defined significant pairs as those with half-life differences greater than or equal to a 1.5-fold change. Notably, the acquisition of an additional UA dinucleotide resulted in the destabilization of RNAs (<xref ref-type="fig" rid="fig4">Figure 4F</xref>, Left: 0–1 UA dinucleotide; Right: 1–2 UA dinucleotides). Moreover, to explore the influence of the UA-destabilizing effect on endogenous mRNA stability, we assessed the UA-dinucleotide ratio in relation to RNA half-life of HEK293 and human erythroleukemia K562 cells. While exon junction density and transcript length have been suggested as major determinants of RNA half-life in vivo (<xref ref-type="bibr" rid="bib2">Agarwal and Kelley, 2022</xref>; <xref ref-type="bibr" rid="bib10">Blumberg et al., 2021</xref>), our findings indicate that elevated UA-dinucleotide ratios in both UTRs significantly promote RNA decay (<xref ref-type="fig" rid="fig4">Figure 4G</xref> and <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1G</xref>). This effect is specific, as such ratios in the coding region are inconsequential. Thus, we have identified UA dinucleotides as being the strongest RNA destabilizing feature in UTRs, with their mutational addition proving the most prevalent cause of reduced RNA stability.</p></sec><sec id="s2-5"><title>Intrinsic features of UTRs determine RNA stability</title><p>Apart from their shared regulatory mechanisms, our MPRA revealed distinct UTR behaviors. Despite undergoing exactly the same procedure, library RNAs in SH-SY5Y cells degraded much faster than those in HEK293T cells (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1D</xref>). Distinct RNA stability control in neuronal cells by GC content of transcripts has been reported (<xref ref-type="bibr" rid="bib25">Guvenek et al., 2022</xref>). Additionally, cell-specific contexts, such as expression of RBPs and miRNAs, may have intensified the discrepancy in stability control between the two cell lines. Moreover, for mutant UTRs that significantly destabilized or stabilized RNAs, we found that their ref counterparts are with significantly longer or shorter RNA half-life, respectively (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1A</xref>). This intrinsic difference in effects on RNA stability for ref UTRs was observed for both cell lines and it was amplified in their mutant counterparts (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1B and C</xref>). A motif analysis revealed that ref UTRs whose mutant counterparts significantly altered RNA stability tended to harbor more NOVA1 (NOVA Alternative Splicing Regulator 1) and PPIE binding sites, but fewer FMR1 (fragile X mental retardation 1) binding sites (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1D</xref>). These observations indicate that intrinsic properties of the UTRs where mutations occur greatly influence the mutational effect.</p></sec><sec id="s2-6"><title>GC content RBP- and ribosome-binding hinders the destabilizing effect of UA dinucleotides</title><p>Many previous studies have reported GC content to be a major RNA stability determinant (<xref ref-type="bibr" rid="bib15">Courel et al., 2019</xref>; <xref ref-type="bibr" rid="bib36">Litterman et al., 2019</xref>; <xref ref-type="bibr" rid="bib68">Zhao et al., 2014</xref>). A univariate regression analysis on our UTR libraries revealed that both GC content and UA dinucleotides are strongly associated with RNA half-life, with the 3’ UTR library from SH-SY5Y cells displaying the strongest association. However, overall, UA dinucleotides are more strongly correlated with RNA half-life than GC content (<xref ref-type="fig" rid="fig4">Figure 4B and D</xref>; <xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2A</xref>). We reasoned that GC content and UA dinucleotides may represent confounding factors in our models. Therefore, to further dissect their respective contributions to RNA stability, we examined their relationship by regressing both factors as well as their interaction term against RNA half-life (t<sub>1/2</sub> ~ UA-diNT% + GC% + UA-diNT% × GC%). For both 5’ UTRs and 3’ UTRs, we observed that UA dinucleotides exhibited a stronger association (i.e. smaller p values) with RNA half-life than GC content. Indeed, the link between GC content and RNA half-life became non-significant when we also accounted for UA dinucleotides (<xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2A</xref>). Moreover, the interaction term (UA-diNT% × GC%) proved significant for three out of four experimental groups (SH-SY5Y 5’ UTR: p = 0.18, 3’ UTR: p = 0.00047, HEK293T 5’ UTR: p = 0.0062, 3’ UTR: p = 1.2e-7), supporting that GC content influences the effect of UA dinucleotides. To further explore this interaction, we stratified degrees of GC content and examined their effects on UA dinucleotides. Notably, despite a somewhat anti-correlation between UA dinucleotides and GC content, they do not simply oppose each other. Substantial counts of UA dinucleotides are observed in regions characterized by high GC content. We found that UA dinucleotides strongly destabilized RNA in the context of low GC content (bottom 50%), but their effects were neutralized somewhat under scenarios of high GC content (top 50%; <xref ref-type="fig" rid="fig5">Figure 5A</xref>). The GC protective effect also supports the observation that change of UA dinucleotides in high GC-content 5’ UTR did not always translate into a change of stability (<xref ref-type="fig" rid="fig4">Figure 4F</xref>). Conversely, the protective effect of GC content was remarkably strong for high UA-dinucleotide ratios but was barely detectable for low UA-dinucleotide ratios (<xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2B</xref>). Thus, the UA destabilizing effect is most pronounced under conditions of a low UA-dinucleotide ratio and low GC content (<xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2C</xref>). Moreover, the stabilizing or destabilizing effects of UA deletion or addition, respectively, were only observed for low GC content, further supporting that high GC content hinders the destabilizing effect of UA dinucleotides (<xref ref-type="fig" rid="fig5">Figure 5B</xref>, <xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2D</xref>).</p><fig-group><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>GC content, RBP and ribosome binding shields RNA from the destabilizing effect of UA dinucleotides.</title><p>(<bold>A</bold>) MPRA data of SH-SY5Y cells stratified according to the GC content (GC%) of UTRs. The data was divided into high and low groups according to the median of GC%. In both UTRs, the destabilizing effect of UA dinucleotides is more evident in the context of low GC content (right panels). p values were determined by linear regression. (<bold>B</bold>) High GC content hinders the effect of altered UA dinucleotides in mutant UTRs. The destabilizing effect of UA-addition (blue) and the stabilizing effect of UA-deletion (crimson) are only observed under the condition of low GC content. p values were determined by a two-sided Wilcoxon rank sum test. (<bold>C</bold>) Destabilizing effect of UA dinucleotide is observed with 5’ UTR random library. High GC content hinders the UA-destabilizing effect. (<bold>D</bold>) UA dinucleotides are enriched in P-body-resident mRNAs. <inline-formula><mml:math id="inf1"><mml:mi>ρ</mml:mi></mml:math></inline-formula> represents Spearman’s correlation coefficient. (<bold>E</bold>) High GC content inhibits enrichment of UA dinucleotide-hosting mRNAs in P-bodies. For medium or low GC%, a high UA-dinucleotide ratio promoted the P-body localization of mRNAs, but this was not the case for high GC%. This effect was more prominent for 3’ UTRs. p values were determined by a two-sided Wilcoxon rank sum test. (<bold>F</bold>) UTRs with more eCLIP RBP binding signals per UA dinucleotide are more stable. The high and low groups was stratified based on the median of number of RBPs per UA. (<bold>G</bold>) UTRs harboring more predicted RBP-binding sites per UA dinucleotide are more stable, as determined by MPRA. p values were determined by two-sided Wilcoxon rank sum test (<bold>F and G</bold>). (<bold>H</bold>) Comparison of RNA half-life of high-UA UTRs determined by MPRA and transcription inhibition with actinomycin D (ActD) (<bold>H</bold>). <inline-formula><mml:math id="inf2"><mml:mi>ρ</mml:mi></mml:math></inline-formula> represents Spearman’s correlation coefficient. (<bold>I</bold>) RNA stability assay with Actinomycin D treatment. Error bars are standard errors computed from three experimental replicates. **: FDR-adjusted p-value = 0.002. (<bold>J</bold>) Association of UA dinucleotide-binding protein motifs with RNA half-life in SH-SY5Y cells. Note that UA-binding RBPs can have both positive and negative effects on RNA stability.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-fig5-v1.tif"/></fig><fig id="fig5s1" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 1.</label><caption><title>Distinct intrinsic properties associated with stability alteration.</title><p>(<bold>A</bold>) Half-life comparisons among ref UTRs whose mutant counterpart did not alter stability (left), significantly decreased stability (middle) or significantly increased stability (right) in SH-SY5Y cells. Please note that only a few UTR mutations altered stability in HEK293T cells, so this comparison was not done for HEK293T cells due to the limits in sample sizes. (<bold>B–C</bold>) Half-life comparisons among UTRs with no effect from mutations (left), ref UTRs whose mutant counterpart significantly altered stability (middle), mutant UTRs that significantly differed from their ref counterpart in stability (right) in HEK293T (<bold>B</bold>) or SH-SY5Y cells (<bold>C</bold>). (<bold>D</bold>) UTRs that are sensitive to mutations (whose mutant altered stability, red lines) are more likely to contain NOVA1, PPIE binding sites, but fewer FMR1 binding sites.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-fig5-figsupp1-v1.tif"/></fig><fig id="fig5s2" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 2.</label><caption><title>Interplay of GC content and UA dinucleotide on stability regulation, related to <xref ref-type="fig" rid="fig5">Figure 5</xref>.</title><p>(<bold>A</bold>) Statistical interaction of GC content and UA dinucleotide on RNA stability. GC content becomes insignificant when considering the effect of UA dinucleotide or UA dinucleotide and their interaction in the multivariate regression. p values were determined by linear regression. (<bold>B</bold>) The protective effect of GC content on RNA half-lives depends on the UA dinucleotide ratio. (<bold>C</bold>) Stratifications of both UA dinucleotide ratio and GC content showed that the destabilizing effect of UA dinucleotide is the most prominent under conditions of low UA dinucleotide ratio and low GC content. The same trend was observed for 5’ UTR (left) and 3’ UTR (right). (<bold>D</bold>) High GC content mitigates the destabilizing effect caused by the gain of UA dinucleotides in a 5’ UTR random library. (<bold>E</bold>) Association of the binding motifs of TA-dinucleotide-binding protein motifs with RNA half-life in HEK293T cells.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-fig5-figsupp2-v1.tif"/></fig></fig-group><p>To corroborate the interplay between the UA-dinucleotide ratio and GC content, we investigated their combined impact on RNA stability using a fixed-length 5' UTR random library (<xref ref-type="bibr" rid="bib31">Jia et al., 2020</xref>). Our findings reveal a negative association between the UA-dinucleotide count within the library and RNA half-life, with this effect being attenuated under conditions of high GC content (top 50%; <xref ref-type="fig" rid="fig5">Figure 5C</xref>). As another layer of validation, we further explored their relative contributions to mRNA enrichment in P-bodies (<xref ref-type="bibr" rid="bib29">Hubstenberger et al., 2017</xref>). P-bodies are membraneless granules for RNA storage and turnover (<xref ref-type="bibr" rid="bib7">Beadle et al., 2023</xref>). We uncovered a positive correlation between UA dinucleotides and RNA P-body localization, especially for those occurring in 3’ UTRs (<xref ref-type="fig" rid="fig5">Figure 5D</xref>). For both UTR types, we observed a greater UA dinucleotide-stabilizing effect in P-bodies when GC content is low (<xref ref-type="fig" rid="fig5">Figure 5D</xref>), consistent with our MPRA dataset.</p><p>We hypothesized that the protective effect of GC content arises from extensive intramolecular interactions that shield UA dinucleotides from being recognized by nucleases, though other physical hindrances may also diminish the destabilizing effect of UA dinucleotides. Therefore, we examined the influence of RBP binding on the destabilizing effect of UA dinucleotides. In <xref ref-type="fig" rid="fig5">Figure 5F and G</xref>, we calculated the number of RBP species binding to UA dinucleotides using both experimental data (enhanced crosslinking immunoprecipitation or eCLIP) and predicted RBP-binding motifs (ATtRACT). We then organized our MPRA-derived results based on whether they exhibited a high (top 50%) or low (bottom 50%) number of RBP binding partners per UA dinucleotide. In doing so, we revealed that the RNA group with multiple binding partners indeed displayed longer half-life compared to the group with few binding partners. This result proved consistent for both UTR types and based on experimental (<xref ref-type="fig" rid="fig5">Figure 5F</xref>) or predicted (<xref ref-type="fig" rid="fig5">Figure 5G</xref>) RBP binding data. To further validate the RBP protective effect against UA-destabilization, we selected four 3’ UTRs (APC, WDR35, SH3TC2, and MTR) with a UA-dinucleotide ratio greater than 90% quantile and assessed their RNA stability by transcription inhibition with actinomycin D (<xref ref-type="fig" rid="fig5">Figure 5H, I</xref>). Half-life obtained by the actinomycin D treatment were highly correlated with the MPRA result (ρ=0.8) in accordance with numbers of RBP binding sites per UA dinucleotide. Together, the results argue a protective role of RBP binding on UA dinucleotides.</p><p>In acknowledgment of the dual potential of RBPs to either promote or inhibit RNA degradation, we conducted a detailed analysis of the impact of each RBP’s binding on RNA stability. By correlating the binding of UA dinucleotide-binding proteins with RNA half-life, we identified UA-binding RBPs that play roles in either safeguarding or promoting RNA degradation (<xref ref-type="fig" rid="fig5">Figure 5J</xref>; <xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2E</xref>; <xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>). Notably, the protective UA-containing motifs were found to be U-rich, reinforcing the observation that U-stretch sequences may enhance RNA stability. Thus, we have identified interplay between GC content, RBP-binding and the destabilizing effect on RNAs of UA dinucleotides. The fact that these regulatory factors may control RNA stability via synergistic or antagonistic mechanisms emphasizes the need to consider all contributory factors simultaneously, as achieved by our modeling approach, to gain a complete overview of the regulatory network (<xref ref-type="fig" rid="fig3">Figure 3</xref>).</p></sec><sec id="s2-7"><title>UA-dinucleotide ratio of UTRs reflects functional enrichment</title><p>Since we identified the UA dinucleotides of UTRs as a major stability determinant, we argued that if UA dinucleotides do indeed represent a functional motif regulating RNA stability, then this property could be a proxy of expression dynamics for classifying genes into functional groups. Therefore, we calculated the average UA-dinucleotide ratio of each gene and compared this distribution within a given GO term against the entire genome background. We found that the 5’ UTRs of genes responsible for regulating appetite, apoptosis and synaptic signaling display the highest UA-dinucleotide ratios and those involved in glutathione metabolism exhibit the lowest (<xref ref-type="fig" rid="fig6">Figure 6A</xref>; <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6</xref>). In terms of 3’ UTRs, genes linked to integrin activation, the interleukin-mediated pathway and B-cell differentiation have the highest UA-dinucleotide ratios (<xref ref-type="fig" rid="fig6">Figure 6B</xref>). To investigate how UA dinucleotide localization contributes to biological functions, we utilized a sliding window approach to identify UTR regions with UA dinucleotides above or below genomic background levels (<xref ref-type="fig" rid="fig6">Figure 6C and D</xref>). UA dinucleotide in each 10-nt window with 1-nt step was calculated and normalized to the UTR length (Methods). For 5’ UTRs, we found that synaptic signaling and striated muscle contraction pathways represent the functional groups with genes having most UA dinucleotide-enriched windows, whereas genes responsible for regulating cell proliferation had most UA dinucleotides-depleted windows (<xref ref-type="fig" rid="fig6">Figure 6C</xref>). For 3’ UTRs, RNA translation and immune-related functions proved to be the GO terms with most UA dinucleotide-enriched windows (<xref ref-type="fig" rid="fig6">Figure 6D</xref>). Since 5’ and 3’ UTRs may synergistically regulate gene expression, we examined genes for which the UA-dinucleotide ratio in both UTRs significantly differed from the genomic background, which revealed that the UA-dinucleotide ratio in both UTRs was consistently either above or below background values (<xref ref-type="fig" rid="fig6">Figure 6E and F</xref>). Plotting these positive or negative effects two-dimensionally, we observed that the gene groups mostly lie in the first and third quadrants, which we interpret as indicative of a synergistic stabilizing or destabilizing effect of both UTRs. Notably, the gene groups in the first quadrant, reflecting UA-dinucleotide ratios in both UTRs being above background levels, are all related to the immune response. It is also noteworthy that the high UA-dinucleotide ratio of genes regulating appetite, synaptic signaling and the immune response reflects the transient expression nature of these gene groups, further supporting that UA dinucleotides exert a destabilizing effect on RNA. Together, this genome-wide analysis indicates that the UA-dinucleotide ratio of UTRs reflects global regulation of gene expression dynamics.</p><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>UA dinucleotides delineate functional gene groups.</title><p>(<bold>A</bold>) The top ten biological processes for which the 5’ UTR UA-dinucleotide ratio most significantly deviated from the genomic background (dashed line). (<bold>B</bold>) The top ten biological processes for which the 3’ UTR UA-dinucleotide ratio most significantly deviated from the genomic background. (<bold>C</bold>) Functional gene groups for which the 5’ UTR UA-dinucleotide ratio was significantly above or below the genomic background in more than ten sliding windows. (<bold>D</bold>) Functional gene groups for which the 3’ UTR UA-dinucleotide ratio was significantly above or below the genomic background in more than ten sliding windows. (<bold>E</bold>) Biological processes for RNAs in which the UA-dinucleotide ratios of both 5’ and 3’ UTRs are significantly different from the genomic background (dashed lines). (<bold>F</bold>) Molecular functions for RNAs in which the UA-dinucleotide ratios of both 5’ and 3’ UTRs are significantly different from the genomic background (dashed lines). The thin solid lines represent the standard deviation of the UA-dinucleotide ratio within the gene group.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-fig6-v1.tif"/></fig></sec><sec id="s2-8"><title>UTR variants associated with disease</title><p>The results from our MPRA stability assay and the genome-wide functional classification support the hypothesis that UTR-mediated RNA stability and gene expression may be disrupted by SNPs within the UTRs. To expand our findings from controlled MPRA experiments to human physiological conditions, we explored the effect of genetic variations in UTRs by surveying human disease databases and biobanks.</p><p>Since dysregulated RNA stability is known to contribute to cancer progression (<xref ref-type="bibr" rid="bib48">Perron et al., 2022</xref>), we first investigated UTR mutations in samples taken from cancer patients (Harmonized Cancer Datasets: <ext-link ext-link-type="uri" xlink:href="https://portal.gdc.cancer.gov/">https://portal.gdc.cancer.gov/</ext-link>). We identified several SNPs in UTRs that correlated with aberrant RNA expression and/or protein expression (<xref ref-type="supplementary-material" rid="supp7">Supplementary file 7</xref>). Interestingly, two 3’ UTR mutations that resulted in aberrant expression of both RNA and proteins (<italic>DDP4</italic> [Dipeptidyl Peptidase 4] and CASP7 [Caspase 7], respectively) were A/T deletions in TA-rich sequences with increased RNA and protein expression levels (<xref ref-type="fig" rid="fig7">Figure 7A and B</xref>), in agreement with our findings that UA-rich sequences are the most influential stability determinants (<xref ref-type="fig" rid="fig3">Figure 3</xref>).</p><fig id="fig7" position="float"><label>Figure 7.</label><caption><title>UTR variants associated with disease.</title><p>(<bold>A–B</bold>) 3’ UTR mutations that increase RNA and protein expression in carcinoma samples. Protein level was determined by reverse phase protein arrays (RPPA). (<bold>C</bold>) QQ plot of the p value distribution of stability-altering UTR variants in association with health biomarkers or self-reported diseases against a theoretical distribution. (<bold>D</bold>) The G allele of the most significant UTR variant (rs5128) identified in (<bold>C</bold>) is associated with plasma triglyceride levels in the Taiwanese population (TWB dataset).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-fig7-v1.tif"/></fig><p>Finally, to establish a correlation between UTR variants and health outcomes, we examined if stability-altering UTR variations identified in our MPRA experiments are associated with abnormal physiological/pathological presentations among the Taiwanese community (Taiwan Biobank (TWB) data). We employed a quantile-quantile (Q-Q) plot to compare the p-value distribution of significant stability-altering UTR SNPs associated with skewed biochemical indices or self-reported diseases against theoretical p values (<xref ref-type="fig" rid="fig7">Figure 7C</xref>). We observed that p values were skewed towards the stability-altering variants, indicating that TWB subjects harboring stability-altering UTR variants are more likely to display abnormal biochemical phenotypes. The most significant association was detected for the 3’ UTR of APOC3 (apolipoprotein C3 c.*40G&gt;C) and blood triglyceride levels (<italic>p =</italic> <inline-formula><mml:math id="inf3"><mml:mn>3.5</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>−</mml:mo><mml:mn>74</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> as determined by linear regression) (<xref ref-type="fig" rid="fig7">Figure 7C and D</xref>), total cholesterol (<italic>p =</italic> <inline-formula><mml:math id="inf4"><mml:mn>7.6</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mrow><mml:mn>10</mml:mn></mml:mrow><mml:mrow><mml:mo>−</mml:mo><mml:mn>12</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>) and self-reported hyperlipidemia (p = 0.00038; <xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref>). APOC3 is involved in metabolizing triglyceride-rich lipoproteins, and its mutation has been associated with low plasma triglyceride levels (<xref ref-type="bibr" rid="bib12">Borén et al., 2020</xref>; <xref ref-type="bibr" rid="bib23">Goyal et al., 2021</xref>). Our findings provide compelling evidence that regulation of RNA turnover by UTRs controls metabolic equilibrium, which can be perturbed by SNPs within the UTRs. Overall, we have demonstrated that pathogenic UTR variants are enriched in critical regulatory regions and may elicit disease.</p></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><sec id="s3-1"><title>Small kmers in UTRs determine RNA stability</title><p>The function of UTRs in regulating RNA stability has long been recognized. However, very few reports have systematically addressed the functional impact of genetic variations in UTRs (<xref ref-type="bibr" rid="bib24">Griesemer et al., 2021</xref>; <xref ref-type="bibr" rid="bib54">Sample et al., 2019</xref>). Therefore, UTR variants are typically classified as being benign or of unknown functions, without experimental support. To methodically dissect the impact of such UTR variants, we have established an MPRA to test the effect of disease-related UTR variants on RNA stability. Unlike previous assays that measure the RNA:DNA ratio or protein output to infer RNA stability, our assay directly measures RNA survival and thereby circumvents confounding effects on transcription and/or protein production. However, while our approach effectively assesses the stability of synthesized RNA in human cells, it may not fully capture the decay dynamics of nuclear-synthesized RNA, which can be influenced by endogenous modifications and <italic>trans</italic>-acting RNA binding factors.</p><p>From among almost 1500 potential stability regulators, we applied LASSO regression to select 5–19 independent factors that best explained RNA half-life. The major destabilizing factors proved to be UA dinucleotides, as well as WWWWWW (W:A/U) and AUUUA (ELAV1 binding site) motifs, all highlighting the importance of UA-rich sequences in UTRs for RNA stability (<xref ref-type="fig" rid="fig3">Figure 3</xref>). Among these destabilizing factors, UA dinucleotides were best correlated with RNA half-life (<xref ref-type="fig" rid="fig4">Figure 4A–D</xref>). The UA-dinucleotide ratio rather than GC content or folding energy explained our RNA stability data better, implying that specific sequence recognition by <italic>trans</italic> factors and not simply regulation according to structuredness is the underlying control mechanism. Similarly, MPRA on 3’ UTR variants identified mono- or di-nucleotide composition as primary factors controlling RNA expression (<xref ref-type="bibr" rid="bib24">Griesemer et al., 2021</xref>). This di-nucleotide specificity may partially be attributed to the sequence preference of a human ribonuclease superfamily in which eight catalytically-active RNases (numbered 1–8) all share homology with bovine pancreatic ribonuclease A. RNase A cleaves 3’–5’ phosphodiester bonds with a specificity for pyrimidines (U/C) at the main anchoring site and purines (A/G) at the secondary site (<xref ref-type="bibr" rid="bib58">Sorrentino, 2010</xref>). Kinetic analysis has revealed a &gt;100-fold preference for UpA than UpG substrates for the human RNase A family (RNase 1–7, RNase 8 was not studied in this report) (<xref ref-type="bibr" rid="bib50">Prats-Ejarque et al., 2019</xref>), consistent with our finding that UA dinucleotides represent the most destabilizing sequence motifs. While most of the studies on RNA degradation mechanism concerned deadenylation and decapping followed by exonuclease digestion, the contribution of endonucleases on overall RNA degradation might be underestimated. It was reported that transcript and UTR length is negatively correlated with RNA stability (<xref ref-type="bibr" rid="bib44">Neymotin et al., 2015</xref>; <xref ref-type="bibr" rid="bib10">Blumberg et al., 2021</xref>), implying that increased RNA mass could be susceptible to endonuclease attack. Thus, our findings and those of others demonstrate that specific sequence recognition, especially di-nucleotide composition, determines RNA stability.</p></sec><sec id="s3-2"><title>Interplay of GC content, RBP binding and UA dinucleotides for RNA stability</title><p>Although it was intuitive to infer a negative correlation between UA dinucleotides and GC content, the best-known stability regulator, we found that UA dinucleotides cannot simply be viewed as an inverse proxy for GC content. Instead, the UA-dinucleotide ratio more adequately explained our RNA stability data, and the protective effect of GC content on stability was almost abolished in the context of a low UA-dinucleotide ratio (<xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2</xref>). Consequently, UA dinucleotides destabilize RNA more robustly under conditions of low GC content (<xref ref-type="fig" rid="fig5">Figure 5A</xref>; <xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2</xref>). This reciprocal interaction was also apparent in independent RNA stability assays and P-body transcriptomic analysis that inferred RNA degradation in vivo (<xref ref-type="bibr" rid="bib29">Hubstenberger et al., 2017</xref>; <xref ref-type="bibr" rid="bib31">Jia et al., 2020</xref>; <xref ref-type="fig" rid="fig5">Figure 5C–E</xref>). Thus, the effect of GC content was also impacted by the sequence context, potentially explaining discrepancies among previous reports on the effects of GC content on RNA stability (<xref ref-type="bibr" rid="bib15">Courel et al., 2019</xref>; <xref ref-type="bibr" rid="bib24">Griesemer et al., 2021</xref>; <xref ref-type="bibr" rid="bib36">Litterman et al., 2019</xref>; <xref ref-type="bibr" rid="bib68">Zhao et al., 2014</xref>). In contrast, although RBPs do not always protect RNA from degradation, with their effect on RNA stability depending on the factors recruited (<xref ref-type="fig" rid="fig5">Figure 5J</xref>), the destabilizing effect of UA dinucleotides is generally hampered by RBP binding (<xref ref-type="fig" rid="fig5">Figure 5F and G</xref>). We reason that, similar to structural hindrance, recognition of UA dinucleotides can be blocked by the physical occupancy of RBPs, resulting in an overall protective effect of RBPs against UA dinucleotide-mediated degradation that may overwhelm the destabilizing effect of a subset of ARE-binding proteins.</p></sec><sec id="s3-3"><title>UTR variants linked to RNA stability and population health</title><p>We identified many crucial regulatory features in 5’ UTRs. Thus, our results evidence that 5’ UTRs are an indispensable region for controlling RNA stability. Our MPRA data revealed that more 5’ UTR variants than those in 3’ UTRs caused stability changes (<xref ref-type="fig" rid="fig1">Figure 1B</xref>). Moreover, we found that stability-altering variants of both UTR types are associated with skewed biomarker or disease accessions in the TWB. Although we did not intend to dissect translation-dependent or -independent decay pathways in our system, it demonstrated the critical contribution of both UTR types to regulating RNA stability. Significantly, our presentation of UA dinucleotide enrichment or depletion in the UTRs of functional gene groups indicates that both UTR types may jointly regulate RNA stability to achieve kinetic control of a functional pathway (<xref ref-type="fig" rid="fig6">Figure 6E and F</xref>). At the sequence level, our results indicate that mutational gain of UA dinucleotides is linked to diminished RNA half-life (<xref ref-type="fig" rid="fig4">Figure 4E</xref>). This simple rule can be adopted as a primary screening for UTR mutation-mediated pathologies and a principle for sequence design of synthetic RNA to be expressed in human cells. Further validation of the sequence effects can be undertaken in pertinent cell or tissue types. Collectively, our findings have uncovered the RNA stability regulation exerted by the sequence composition of UTRs, revealed health-related UTR variants, and provided a foundation for precise diagnosis of non-coding genetic variants.</p></sec></sec><sec id="s4" sec-type="methods"><title>Methods</title><sec id="s4-1"><title>Metagene analysis</title><p>Both variants on all UTR and uORF use dbSNP version 151 (<xref ref-type="bibr" rid="bib56">Sherry et al., 1999</xref>) as baseline, disease variants, UTR sources are as described above. uORF definitions were downloaded from TIS-db (<xref ref-type="bibr" rid="bib65">Wan and Qian, 2014</xref>), and their genome coordinates were lifted from hg19 to hg38 before analysis.</p></sec><sec id="s4-2"><title>Disease variant collection, UTR library construction</title><sec id="s4-2-1"><title>Variant collection</title><p>Disease-related variants were collected from ClinVar (<xref ref-type="bibr" rid="bib34">Landrum et al., 2016</xref>) and HGMD database (<xref ref-type="bibr" rid="bib59">Stenson et al., 2017</xref>). Variants located in the coding region or labeled as benign variants were removed. The UTRs were defined by NCBI RefSeq and ENCODE V27. We then intersected selected variant positions with the UTR regions by using BEDTools (<xref ref-type="bibr" rid="bib51">Quinlan and Hall, 2010</xref>) to collect UTR variants.</p></sec><sec id="s4-2-2"><title>Library construction</title><p>For each variant, we extracted 115 bp of sequence around variant position, and built one oligo pair, reference and mutant, which only differs only by the variant position. The variant was placed in the middle of the sequence, unless it is located near the boundary of UTR. UTR-specific primers were then added on each sequence. (5’ UTR-F: 5’-<named-content content-type="sequence">CGCTAGGGATCCTCTAGTCA</named-content>-3’, 5’ UTR-R: 5’-<named-content content-type="sequence">ACCGGTCGCCACCATGGTGA</named-content>-3’; 3’ UTR-F: 5’-<named-content content-type="sequence">GGACGAGCTGTACAAGTAAA</named-content>-3’, 3’ UTR-R: 5’-<named-content content-type="sequence">GCGGCCGCGCAATAACTAGC</named-content>-3’). Overall, we built an oligo library containing 12,472 sequences (6555 pairs) of both 5’ and 3’ UTR, and synthesized them by CustomArray Inc (U.S.).</p><p>UTR-library DNA templates were assembled by overlap extension PCR with Herculase II Fusion Enzyme (Agilent Technologies). Initially, the oligonucleotide library sequences were double-strandized and amplified by PCR. After being subjected to PCR clean-up (QIAGEN), the UTR-library amplicons were then appended with EGFP CDS (derived from EGFP-N1 vector) and T7 promoter sequence through overlapping PCR. Eventually, the assembled full-length DNA templates (T7-5’ UTR-EGFP or T7-EGFP-3’ UTR) were subjected to PCR clean-up again for the following in vitro transcription.</p></sec><sec id="s4-2-3"><title>In vitro transcription and polyadenylation</title><p>200 ng of the PCR product was subject to in vitro transcription using MEGAscript T7 Transcription Kit (Thermo Fisher Scientific). To cap the RNA product, m7G(5')ppp(5')G RNA Cap Structure Analog (New England Biolabs, 4:1 to GTP) was supplemented to the in vitro transcription reaction. After 3 hr incubation in 37 °C, DNA template was removed by 1 μl Turbo DNase at 37 °C for 15 min. The RNA product was purified by illustra microspin G-50 column (GE Healthcare Life Sciences). 10 μg of the purified RNA was then polyadenylated by 4 U poly(A) polymerase (New England Biolabs) at 37 °C for 1 hr and purified again by illustra microspin G-50 column or Direct-zol RNA Miniprep kits (Zymo Research) before transfection (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1C</xref>).</p></sec><sec id="s4-2-4"><title>Cell lines</title><p>HEK293T or SH-SY5Y cells were authenticated and purchased from ATCC (American Type Culture Collection), catalog numbers CRL-11268 and CRL-2266. They were tested negative for mycoplasma contamination.</p></sec><sec id="s4-2-5"><title>Transfection</title><p>Seed HEK293T or SH-SY5Y cells 20 hr before transfection. Transfect 500 ng capped and polyadenylated RNA per well into a 12-well plate by Lipofectamine 3000 reagent (Thermo Fisher Scientific). Wash the cells twice before collecting the first time point (30 min for HEK293T, 20 min for SH-SY5Y), and then harvest RNA along the time course.</p></sec><sec id="s4-2-6"><title>Amplicon preparation for sequencing</title><p>RNA was extracted using Trizol reagent (Invitrogen) and total RNA was extracted with Direct-zol RNA Miniprep kits (Zymo Research). Afterwards, cDNA was reverse transcribed from 1 μg of RNA using library-specific primers (5’ UTR: EGFP3-f_UMI_5Lib-RT; 3’ UTR: CMV3-f_UMI_M13-rev, see <xref ref-type="supplementary-material" rid="supp9">Supplementary file 9</xref>) with SuperScript IV Reverse Transcriptase (Thermo Fisher Scientific). The cDNA was amplified first by primers EGFP3-f and CMV3-f 15 cycles by Phusion High-Fidelity DNA Polymerase (Thermo Fisher Scientific) with GC buffer. The resultant product was cleaned up using QIAGEN PCR Purification kit and subject to second round of PCR with primers Forward/P5_fractionsID_CMV3f and Reverse/P7_#N_EGFP3f for 5’ UTR library, and Forward/P5_fractionsID_EGFP3f and Reverse/P7_#N_CMV3f (<xref ref-type="supplementary-material" rid="supp9">Supplementary file 9</xref>) for 3’ UTR library by 10 cycles. The PCR product was purified again by QIAGEN PCR Purification kit for Illumina NextSeq paired-end 150 sequencing.</p></sec></sec><sec id="s4-3"><title>RNA-seq data processing</title><sec id="s4-3-1"><title>QC</title><p>PCR duplication of single-end raw fastq files (150 bp) was first removed by nubeam-dedup (<xref ref-type="bibr" rid="bib16">Dai and Guan, 2020</xref>), then adapter and low-quality base (sliding mean quality &lt;20) was trimmed by trimmomatic (<xref ref-type="bibr" rid="bib11">Bolger et al., 2014</xref>). Trimmed reads of length less than 80 bp were discarded.</p></sec><sec id="s4-3-2"><title>Alignment</title><p>FASTQ data alignment was done by HISAT2 (<xref ref-type="bibr" rid="bib32">Kim et al., 2015</xref>), genome information was built from the UTR library.</p><p>Detail parameters are as below, HISAT2: hisat2 <monospace>--no-spliced-alignment</monospace> <monospace>--score-min </monospace>L,0,–0.7 -x index -U trimmed_fastq.gz.</p></sec><sec id="s4-3-3"><title>Count</title><p>After alignment, BEDTools multicov (<xref ref-type="bibr" rid="bib51">Quinlan and Hall, 2010</xref>) was used to build the count matrix. Since accuracy is critical in this project, alignment results were filtered by mapping quality (HISAT2: MAPQ = 60), and count only ‘exactly one alignment’ result.</p></sec></sec><sec id="s4-4"><title>Half-life estimation</title><p>For each oligonucleotide, we estimated decay constant λ and half-life (<inline-formula><mml:math id="inf5"><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mrow></mml:mrow></mml:msub></mml:math></inline-formula>) by the following equations:<disp-formula id="equ1"><mml:math id="m1"><mml:mrow><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="0.7em 0.3em" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>ln</mml:mi><mml:mo>⁡</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub></mml:mfrac></mml:mstyle><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo></mml:mstyle></mml:mtd><mml:mtd><mml:mi/><mml:mo>−</mml:mo><mml:mi>λ</mml:mi><mml:msub><mml:mi>t</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>ϵ</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo></mml:mtd><mml:mtd><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mrow><mml:mi>ln</mml:mi><mml:mo>⁡</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>2</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>λ</mml:mi></mml:mfrac></mml:mstyle></mml:mstyle></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf6"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>t</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf7"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> are read count values of the i<sup>th</sup> replicate at time points <inline-formula><mml:math id="inf8"><mml:mi>t</mml:mi></mml:math></inline-formula> and 0, and <inline-formula><mml:math id="inf9"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>ϵ</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> is error term. For a more accurate estimate of λ, we used the first time point (HEK293T: 30 min, SH-SY5Y: 20 min) as the time point of t=0, and the models were weighted by their inverse standard deviations. All the read count values were normalized by the average of <inline-formula><mml:math id="inf10"><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow></mml:mfenced></mml:mrow></mml:msub></mml:math></inline-formula> before constructing the linear models. After that, half-life λ was estimated by linear regression function ‘lm’ of R. We only selected oligonucleotide that <italic>R<sup>2</sup></italic> &gt;0.5 and mean squared error (MSE) &lt; 1 for further analysis.</p></sec><sec id="s4-5"><title>Statistical method to detect the mutation effect on stability</title><p>To determine variants that changed oligonucleotide stability, we used the normalized counts to build a linear model weighted with the inverse of their standard deviations as follows:<disp-formula id="equ2"><mml:math id="m2"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>ln</mml:mi><mml:mo>⁡</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>t</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:mfrac><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>β</mml:mi><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>θ</mml:mi><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>G</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>ϵ</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf11"><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> represented reference (ref) or mutant (mt), and <inline-formula><mml:math id="inf12"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>ϵ</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> is error term.</p><p>We tested the significance of <inline-formula><mml:math id="inf13"><mml:mi>θ</mml:mi></mml:math></inline-formula> to determine the effect of the variant. We then corrected the p value by FDR.</p><p>All statistical analyses were performed by R language (version 4.0.5/4.2.0), linear models were constructed by ‘lm’ function and p value adjustment was conducted by ‘p.adjust’ function and ‘qvalue’ function from the R package ‘qvalue’.</p></sec><sec id="s4-6"><title>in silico feature prediction</title><sec id="s4-6-1"><title>Free Energy</title><p>Free Energy was estimated by the ViennaRNA package RNAfold command (<xref ref-type="bibr" rid="bib37">Lorenz et al., 2011</xref>), which predicted minimum free energy (MFE) by default.</p></sec><sec id="s4-6-2"><title>GC content and kmer ratio</title><p>GC content, the ratio of G/C nucleotide of a sequence, and short kmer (k=1–3) were calculated by letterFrequency and oligonucleotideFrequency function of the R package Biostrings (<ext-link ext-link-type="uri" xlink:href="https://bioconductor.org/packages/Biostrings">https://bioconductor.org/packages/Biostrings</ext-link>).</p></sec><sec id="s4-6-3"><title>Motifs</title><p>For motif scanning such as PUM motif, RBP motif, ARE and novel motif, were done by Biostrings functions, countPWM or countPattern (<ext-link ext-link-type="uri" xlink:href="https://bioconductor.org/packages/Biostrings">https://bioconductor.org/packages/Biostrings</ext-link>). PUM motif was retrieved from <xref ref-type="bibr" rid="bib52">Rabani et al., 2017</xref>. RBP motif PWMs were download from ATtRACT database (<xref ref-type="bibr" rid="bib22">Giudice et al., 2016</xref>) and ARE definition followed by AREsite database (<xref ref-type="bibr" rid="bib19">Fallmann et al., 2016</xref>), including ATTTA, WTTTW, WWTTTWW, WWWTTTWWW, WWWWTTTWWWW, WWWWWTTTWWWWW, TTTGTTT, GTTTG, AWTAAA.</p></sec><sec id="s4-6-4"><title>KOZAK/uAUG</title><p>KOZAK sequence was defined as in the previous study (<xref ref-type="bibr" rid="bib26">Hernández et al., 2019</xref>), optimal: GCCRCCAUGG; strong: NNNRNNAUGG; moderate_H: NNNRNNAUG(A/C/U); moderate_Y: NNN(C/U)NNAUGG and weak: NNN(C/U)NNAUG(A/C/U). We scanned these motifs on the library sequence by R package Biostrings countPattern function.</p></sec><sec id="s4-6-5"><title>miRNA</title><p>miRNA binding site was predicted by TargetScan 7.0 (<xref ref-type="bibr" rid="bib1">Agarwal et al., 2015</xref>; <xref ref-type="bibr" rid="bib41">McGeary et al., 2019</xref>). miRNA expression information was downloaded from miRmine database (<xref ref-type="bibr" rid="bib46">Panwar et al., 2017</xref>), only those miRNAs that have at least 1 RPM in HEK293T or SH-SY5Y cells were selected as features. Since the 6-mer seed sequence is known to be less effective (<xref ref-type="bibr" rid="bib6">Bartel, 2009</xref>), only predicted 7mer (7mer-m8, 7mer-1a) and 8mer seed sequences were kept as features.</p></sec><sec id="s4-6-6"><title>RNA binding protein binding</title><p>eCLIP (enhanced crosslinking immunoprecipitation) data were downloaded from ENCODE (<xref ref-type="bibr" rid="bib38">Luo et al., 2020a</xref>) and GSE117290 datasets (<xref ref-type="bibr" rid="bib39">Luo et al., 2020b</xref>), and selected only eCLIP regions identified in two biological replicates. eCLIP signals with value &lt;1 or p value &gt;0.05 were removed. All genome coordinates were lifted to hg38 by liftover command (<xref ref-type="bibr" rid="bib27">Hinrichs et al., 2006</xref>). RBP bindings were determined by intersecting eCLIP signals and variant coordinates by BEDTools intersect command (<xref ref-type="bibr" rid="bib51">Quinlan and Hall, 2010</xref>).</p></sec><sec id="s4-6-7"><title>G-quadruplex (RG4)</title><p>G-quadruplex structures were predicted by the RNAfold command from the ViennaRNA package (<xref ref-type="bibr" rid="bib37">Lorenz et al., 2011</xref>), with -g parameter.</p></sec><sec id="s4-6-8"><title>Conservation level</title><p>phastCons scores (pC4way-pC100way) were downloaded from UCSC genome browser (<xref ref-type="bibr" rid="bib35">Lee et al., 2022</xref>) and intersected with UTR genome coordinates by the intersect function of BEDTools (<xref ref-type="bibr" rid="bib51">Quinlan and Hall, 2010</xref>). An average phastCons score was calculated for each UTR.</p></sec><sec id="s4-6-9"><title>Novel 7-mer motifs</title><p>Primer sequences were first removed from the UTR library sequence. Then a 7-mer table was built by manual R script (<xref ref-type="bibr" rid="bib53">R Development Core Team, 2021</xref>). UTR library half-life was transformed by log2() and the extreme values were filtered out. 7-mers occurring less than 20 times in the UTR library were filtered out. Regress 7-mers to log2(half-life) by glmnet::glmnet() LASSO selection (<xref ref-type="bibr" rid="bib20">Friedman et al., 2010</xref>) to obtain the coefficients. Repeat the regression 2000 times with bootstrap resampling and select 7-mers that were selected &gt;1600 times for further analysis. The non-zero distribution of coefficients of a 7-mer was then tested by permutation coin::oneway_test (<xref ref-type="bibr" rid="bib28">Hothorn et al., 2008</xref>), and the resulting p values were adjusted by p.adjust() Bonferroni adjustment. 7-mers with non-zero coefficients (adjusted p-values &lt;0.05) were further divided into positive and negative effects according to their mean coefficients. To generate motifs, the text distance in each group was calculated using stringdist::stringdistmatrix() function with ‘lv’ distance (<xref ref-type="bibr" rid="bib63">van der Loo, 2014</xref>) and clustered by hclust(). Finally, use cutree(), h=2 to subgroup the 7-mers. 7-mers of a subgroup were combined to build PWM according to the weight of their mean coefficients.</p></sec><sec id="s4-6-10"><title>Feature selection</title><p>A LASSO model was utilized to perform feature selection. Features with a high proportion of 0 (≥ 90%) were excluded to ensure the robustness of the selection model. In addition, to avoid multicollinearity caused by similar features that perturb feature selection, all features were clustered using single-linkage hierarchical clustering with the distance metric defined as one minus the absolute value of the Spearman correlation coefficient. We cut the tree at a specific height, and the feature that had the greatest influence on RNA stability, which was examined using a simple linear regression model, was selected to be the representative of each cluster. Then we calculated the variance inflation factor (VIF) value of the representative features. The VIFs were obtained by the following linear model and equations:<disp-formula id="equ3"><mml:math id="m3"><mml:mrow><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="0.7em 0.7em 0.3em" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mover><mml:mi>X</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mrow><mml:mover><mml:mi>a</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>0</mml:mn><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:munder><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>≠</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:munder><mml:msub><mml:mrow><mml:mover><mml:mi>a</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mstyle></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msubsup><mml:mi>R</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mn>2</mml:mn></mml:msubsup></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mrow><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>X</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>X</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>⋅</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mrow><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>X</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>⋅</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mfrac></mml:mstyle></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mtext>VIF</mml:mtext><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msubsup><mml:mi>R</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mn>2</mml:mn></mml:msubsup></mml:mrow></mml:mfrac></mml:mstyle></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf14"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>X</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf15"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> are the estimated value of the j<sup>th</sup> feature and the value of the k<sup>th</sup> feature of the i<sup>th</sup> UTR (note that the k<sup>th</sup> feature is a feature other than the j<sup>th</sup> feature), <inline-formula><mml:math id="inf16"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>α</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>0</mml:mn><mml:mrow><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf17"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>α</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>j</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> are the intercept and the regression coefficients of the linear model that regressed the j<sup>th</sup> feature on the other remaining features, and <inline-formula><mml:math id="inf18"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>X</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>⋅</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> is the mean level of the j<sup>th</sup> feature of all UTRs.</p><p>The height for cutting the tree was gradually increased until the VIF values of all representative features were less than 5. By identifying and excluding highly collinear variables, we aimed to minimize multicollinearity and improve the accuracy of our regression models (<xref ref-type="bibr" rid="bib3">Akinwande et al., 2015</xref>). Finally, the LASSO-based feature selection was conducted on the representative features (<xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref>). The LASSO model was carried out using R package glmnet (<xref ref-type="bibr" rid="bib20">Friedman et al., 2010</xref>) and can be written as follows:<disp-formula id="equ4"><mml:math id="m4"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:munder><mml:mrow><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">g</mml:mi><mml:mi mathvariant="normal">m</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">n</mml:mi></mml:mrow><mml:mi>β</mml:mi></mml:munder><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>ln</mml:mi><mml:mo>⁡</mml:mo><mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:munder><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:mi>λ</mml:mi><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:munder><mml:mrow><mml:mo>|</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf19"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf20"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> are the intercept and regression coefficients of the j<sup>th</sup> feature, and <inline-formula><mml:math id="inf21"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> is the level of the j<sup>th</sup> feature of the i<sup>th</sup> UTR. λ was determined through a ten-fold cross-validation process, which yielded the minimum mean cross-validated error. We performed a 2:1 split of the data into training and testing sets to validate the robustness of the model. The performance metrics, including the correlation coefficient between observed and predicted values, mean average error (MAE), root mean squared error (RMSE), mean absolute percentage error (MAPE), and R-squared, are provided in <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>.</p></sec></sec><sec id="s4-7"><title>In vivo validation</title><p>To validate UA-diNT effect in in vivo data, we collected public SLAM-seq data set GSE126523 (K562) and GSE214396 (HEK293) to estimate the half-life of endogenous transcripts. The first step was to remove adapters and low-quality sequences (sliding window average sequencing quality &lt;20 and read length &lt;50) by trimmomatic. SLAM-seq fastq was then aligned to transcript sequences by HISAT-3N (<xref ref-type="bibr" rid="bib67">Zhang et al., 2021</xref>) with annotation file gencode.v43.primary_assembly. After alignment, ‘exactly one alignment’ result was further used to calculate T to C conversion by hisat-3n-table. Nucleotide positions with less than 10 coverage or potential SNPs (T to C conversion rate larger than 0.8) were removed. Each transcript T to C conversion rate were calculated by Total TtoC ⁄ (Total TtoC +Total T). Transcripts which has at least 2 batches in each timepoint were kept for further analysis. After all time points T to C conversion rate normalized to time 0 average T to C conversion rate, normalized data then fit<disp-formula id="equ5"> <mml:math id="m5"><mml:mrow><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="0.7em 0.3em" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>ln</mml:mi><mml:mo>⁡</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub></mml:mfrac><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mo>−</mml:mo><mml:mi>λ</mml:mi><mml:msub><mml:mi>t</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>ϵ</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mrow><mml:mi>ln</mml:mi><mml:mo>⁡</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>2</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>λ</mml:mi></mml:mfrac></mml:mstyle></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf22"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>t</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="inf23"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> are T to C conversion rate of the i<sup>th</sup> transcript at time points <inline-formula><mml:math id="inf24"><mml:mi>t</mml:mi></mml:math></inline-formula> and 0, and <inline-formula><mml:math id="inf25"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>ϵ</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> is error term. R lm() function was used to estimate λ and calculate transcript half-life t<sub>1/2</sub>.</p><p>UA-diNT counts were calculated by R Biostrings::countPattern and sequence length by nchar(). UA-diNT ratios in each UTR sequences were UA-diNT counts normalized to the sequence length.</p></sec><sec id="s4-8"><title>RNA stability assay with actinomycin D treatment</title><p>SH-SY5Y cells were seeded approximately 4x10<sup>5</sup> cells per well in 12-well plates before the day of the transfection. The cells were transfected with 800 ng pEGFP-3’ UTR plasmids by using Viromer ONE RED (Lipocalyx GmbH). 24 hr post-transfection, the cells per 12-well were equally divided into 3 wells in 24-well plate. After 16 hr incubation, cells were treated with actinomycin D (5 μg/ml) and were harvested by TRIzol at the respective time points (0, 2, 4 hr after stopping the transcription). The RNA was purified by the Direct-zol RNA Microprep kits (Zymo Research). Equal amount of the total RNA for each sample were reverse transcribed using SuperScript IV (Thermo Fisher) with random hexamer. Real-time PCR was performed with FastSYBR Green Plus Master Mix (Applied Biosystems) and QuantStudio 12 K Flex Real-Time PCR System (Applied Biosystems). Relative mRNA abundance in different time points were normalized to time points 0 hr (2<sup>-(△Ct –△Ct0)</sup>) for each pEGFP-3’ UTR.</p></sec><sec id="s4-9"><title>Construction of UTR-library DNA templates</title><p>UTR-library DNA templates were assembled by overlap extension PCR with Herculase II Fusion Enzyme (Agilent Technologies). Initially, the oligonucleotide library sequences (CustomArray, Inc USA) were double-strandized and amplified by PCR. After subjected to PCR clean-up (Qiagen), the UTR-library amplicons were then appended with EGFP CDS and CMV promoter sequence (derived from EGFP-N1 vector) sequentially through overlapping PCR. Eventually, the assembled full-length DNA templates (CMVP-5’ UTR-EGFP or CMVP-EGFP-3’ UTR) were subjected to PCR clean-up again for the following transfection.</p></sec><sec id="s4-10"><title>UA-ratio analysis for GO annotation</title><p>GO annotations were downloaded from QuickGO (<xref ref-type="bibr" rid="bib9">Binns et al., 2009</xref>) and Ensembl BioMart (<xref ref-type="bibr" rid="bib33">Kinsella et al., 2011</xref>). Full Human genome sequence and transcript annotation were from R Bioconductor package BSgenome.Hsapiens.UCSC.hg38 v.1.4.4 (UCSC version hg38, based on GRCh38.p13) and TxDb.Hsapiens.UCSC.hg38.knownGene v.3.15.0. UTR length &lt;10 nucleotides were removed from analysis.</p><p>UA-dinucleotide ratio in each UTR sequence was calculated by (UA count ÷ (sequence length - 1)). For every GO annotation containing &gt;20 qualified transcripts, UA-ratio of those transcripts were compared with all transcripts by Fisher-Pitman permutation test (R coin package <xref ref-type="bibr" rid="bib28">Hothorn et al., 2008</xref>). Bonferroni corrected p-value &lt;0.01 was defined as significant enrichment/depletion.</p><p>For the sliding UA-dinucleotide ratio, UA-ratio of the most 5’ 10-nt window was first calculated and the window was slid by 1 nt in each movement to the 3’end for each transcript. Then each UTR was normalized to length by splitting sliding UA-ratio to 100 fragments and the mean sliding UA-dinucleotide ratio in each fragment was calculated. Finally calculate the mean ratio in each fragment across all transcripts within a GO term was calculated to represent the sliding UA-dinucleotide ratio.</p></sec><sec id="s4-11"><title>TCGA data</title><p>TCGA data were downloaded by R/Bioconductor package TCGABiolinks (<xref ref-type="bibr" rid="bib14">Colaprico et al., 2016</xref>) with the following query code:</p><p>RNA-seq: GDCquery(TCGA-id, data.category = &quot;Transcriptome Profiling&quot;, experimental.strategy = &quot;RNA-Seq&quot;, data.type = &quot;Gene Expression Quantification&quot;,workflow.type = &quot;STAR - Counts&quot;).</p><p>Protein: GDCquery(project = TCGA id, data.type = &quot;Protein expression quantification&quot;, legacy = TRUE, data.category = &quot;Protein expression&quot;, platform = &quot;MDA_RPPA_Core&quot;).</p><p>Only donor IDs and variants valid in both RNA/Protein datasets underwent further analyses.</p></sec><sec id="s4-12"><title>The Taiwan Biobank Data, TWB</title><p>In order to identify the association between RNA stability altering variants and common chronic diseases and biochemical indices, the genotyping and phenotypic data of 68,978 Taiwanese people were obtained from the TWB (<ext-link ext-link-type="uri" xlink:href="https://www.biobank.org.tw/">https://www.biobank.org.tw/</ext-link>). The information on the diseases was self-reported and collected through questionnaires. Each participant was genotyped on the Affymetrix Axiom genome-wide TWB 2.0 array containing 752,921 SNP (single nucleotide polymorphism) probes. The study was approved by the Institutional Review Board of Academia Sinica (AS-IRB-BM-19020).</p><p>We conducted statistical analyses to assess the association of 21 variants significantly affecting RNA stability with 23 reported traits and the levels of 24 biochemical indices. Before carrying out association tests, the quality control for genotyping data was performed as described previously to ensure its reliability (<xref ref-type="bibr" rid="bib13">Chiang et al., 2022</xref>). Linear regression and logistic regression were performed to examine the association between each SNP and continuous biochemical index and dichotomous trait, respectively. Multinomial logistic regression followed by a likelihood ratio test was used to determine p-values for biochemical indices containing more than two levels and was fitted by R package nnet (v 7.3–17). All SNPs were tested as co-dominant genetic models. The square root of age, gender, dwelling place, and the batch of array were included in the regression models to adjust for potential confounding effects. The first 10 principal components were also included as covariates in all regression models to control the population stratification.</p></sec></sec></body><back><sec sec-type="additional-information" id="s5"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Investigation, Visualization, Methodology, Writing – original draft, Project administration, Writing – review and editing</p></fn><fn fn-type="con" id="con2"><p>Data curation, Formal analysis, Investigation, Visualization, Methodology, Writing – original draft, Project administration</p></fn><fn fn-type="con" id="con3"><p>Validation</p></fn><fn fn-type="con" id="con4"><p>Validation, Project administration</p></fn><fn fn-type="con" id="con5"><p>Validation, Project administration</p></fn><fn fn-type="con" id="con6"><p>Project administration</p></fn><fn fn-type="con" id="con7"><p>Supervision, Funding acquisition</p></fn><fn fn-type="con" id="con8"><p>Conceptualization, Supervision, Funding acquisition, Writing – original draft, Writing – review and editing</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s6"><title>Additional files</title><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>Results of the massively parallel reporter assays.</title></caption><media xlink:href="elife-97682-supp1-v1.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="supp2"><label>Supplementary file 2.</label><caption><title>Correlation of AREs and RNA half-lives.</title></caption><media xlink:href="elife-97682-supp2-v1.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="supp3"><label>Supplementary file 3.</label><caption><title>Univariable correlation of sequence features and RNA half-lives.</title></caption><media xlink:href="elife-97682-supp3-v1.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="supp4"><label>Supplementary file 4.</label><caption><title>LASSO feature selection for RNA half-lives.</title></caption><media xlink:href="elife-97682-supp4-v1.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="supp5"><label>Supplementary file 5.</label><caption><title>Correlation of UA-binding RBPs and RNA half-lives.</title></caption><media xlink:href="elife-97682-supp5-v1.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="supp6"><label>Supplementary file 6.</label><caption><title>Frequency of UTR UA-dinucleotides in functional gene groups.</title></caption><media xlink:href="elife-97682-supp6-v1.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="supp7"><label>Supplementary file 7.</label><caption><title>RNA and protein expression level in association with genotypes in cancers.</title></caption><media xlink:href="elife-97682-supp7-v1.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="supp8"><label>Supplementary file 8.</label><caption><title>Correlation tests of health markers and genotypes in TWB.</title></caption><media xlink:href="elife-97682-supp8-v1.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="supp9"><label>Supplementary file 9.</label><caption><title>Primer list.</title></caption><media xlink:href="elife-97682-supp9-v1.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-97682-mdarchecklist1-v1.pdf" mimetype="application" mime-subtype="pdf"/></supplementary-material></sec><sec sec-type="data-availability" id="s7"><title>Data availability</title><p>All raw and processed sequencing data generated in this study have been submitted to the NCBI Gene Expression Omnibus (GEO; <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/geo/">https://www.ncbi.nlm.nih.gov/geo/</ext-link>) under accession number GSE217518. Codes used for the analysis in this study have been deposited at <ext-link ext-link-type="uri" xlink:href="https://github.com/chienlinglin/modeling-UTR-variants-stability">https://github.com/chienlinglin/modeling-UTR-variants-stability</ext-link> (copy archived at <xref ref-type="bibr" rid="bib61">Su, 2025</xref>). Other databases used in the study: <ext-link ext-link-type="uri" xlink:href="https://genome.ucsc.edu/cgi-bin/hgTracks?hgsid=1351580935_14MOQtNDW7V78RaXEDp3Yy4m4PTb&amp;c=chr2&amp;hgTracksConfigPage=configure&amp;hgtgroup_compGeno_close=0#compGenoGroup">UCSC PhyloP</ext-link>; <ext-link ext-link-type="uri" xlink:href="http://nibiru.tbi.univie.ac.at/AREsite2">AREsite2</ext-link>; <ext-link ext-link-type="uri" xlink:href="https://attract.cnic.es/download">ATtRACT</ext-link>; <ext-link ext-link-type="uri" xlink:href="https://asia.ensembl.org/info/data/ftp/index.html">Ensembl</ext-link>; <ext-link ext-link-type="uri" xlink:href="https://portal.gdc.cancer.gov/">Harmonized Cancer Datasets</ext-link>; and <ext-link ext-link-type="uri" xlink:href="https://www.biobank.org.tw/">Taiwan Biobanks</ext-link>.</p><p>The following dataset was generated:</p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset1"><person-group person-group-type="author"><name><surname>Lin</surname><given-names>C</given-names></name><name><surname>Wang</surname><given-names>Y</given-names></name><name><surname>Su</surname><given-names>J</given-names></name><name><surname>Chang</surname><given-names>Y</given-names></name><name><surname>Yang</surname><given-names>C</given-names></name><name><surname>Lin</surname><given-names>P</given-names></name><name><surname>Kang</surname><given-names>Y</given-names></name><name><surname>Hsieh</surname><given-names>Y</given-names></name><name><surname>Huang</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2025">2025</year><data-title>Multiplexed Assays of Human Disease-relevant Mutations Reveal UTR Dimer Composition as a Major Determinant of RNA Stability</data-title><source>NCBI Gene Expression Omnibus</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE217518">GSE217518</pub-id></element-citation></p><p>The following previously published datasets were used:</p><p><element-citation publication-type="data" specific-use="references" id="dataset2"><person-group person-group-type="author"><name><surname>Wu</surname><given-names>Q</given-names></name><name><surname>Medina</surname><given-names>SG</given-names></name><name><surname>Kushawah</surname><given-names>G</given-names></name><name><surname>DeVore</surname><given-names>ML</given-names></name></person-group><year iso-8601-date="2019">2019</year><data-title>Translation affects mRNA stability in a codon dependent manner in human cells</data-title><source>NCBI Gene Expression Omnibus</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE126523">GSE126523</pub-id></element-citation></p><p><element-citation publication-type="data" specific-use="references" id="dataset3"><person-group person-group-type="author"><name><surname>Mueller</surname><given-names>MB</given-names></name><name><surname>Jayaraj</surname><given-names>GG</given-names></name><name><surname>Hartl</surname><given-names>FU</given-names></name></person-group><year iso-8601-date="2023">2023</year><data-title>Mechanisms of stop codon readthrough mitigation reveal principles of GCN1 mediated translational quality control</data-title><source>NCBI Gene Expression Omnibus</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE214396">GSE214396</pub-id></element-citation></p></sec><ack id="ack"><title>Acknowledgements</title><p>We thank the Genomics Core and the Bioinformatics Core of the Institute of Molecular Biology (IMB), Academia Sinica, for performing the amplicon sequencing and for providing computing resources. We thank all members of IMB, particularly Drs. Jun-Yi Leu and Hung-Lun Chiang, for tremendous help and support. This work was supported by Career Development Award and Multidisciplinary Health Cloud Research Program of Academia Sinica (AS-CDA-108-M03 and AS-PH-109-01-3), Career Development Award of National Health Research Institute, Taiwan (NHRI-EX112-10908BC) and Excellent Young Scholar Research Grants and Ta-You Wu Memorial Award of National Science and Technology Council, Taiwan (MOST 111–2628-B-001–003 and 108–2118 M-001-013-MY5).</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Agarwal</surname><given-names>V</given-names></name><name><surname>Bell</surname><given-names>GW</given-names></name><name><surname>Nam</surname><given-names>JW</given-names></name><name><surname>Bartel</surname><given-names>DP</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Predicting effective microRNA target sites in mammalian mRNAs</article-title><source>eLife</source><volume>4</volume><elocation-id>e05005</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.05005</pub-id><pub-id pub-id-type="pmid">26267216</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Agarwal</surname><given-names>V</given-names></name><name><surname>Kelley</surname><given-names>DR</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>The genetic and biochemical determinants of mRNA degradation rates in mammals</article-title><source>Genome Biology</source><volume>23</volume><elocation-id>245</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-022-02811-x</pub-id><pub-id pub-id-type="pmid">36419176</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Akinwande</surname><given-names>MO</given-names></name><name><surname>Dikko</surname><given-names>HG</given-names></name><name><surname>Samson</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Variance inflation factor: as a condition for the inclusion of suppressor variable(s) in regression analysis</article-title><source>Open Journal of Statistics</source><volume>05</volume><fpage>754</fpage><lpage>767</lpage><pub-id pub-id-type="doi">10.4236/ojs.2015.57075</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Barreau</surname><given-names>C</given-names></name><name><surname>Paillard</surname><given-names>L</given-names></name><name><surname>Osborne</surname><given-names>HB</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>AU-rich elements and associated factors: are there unifying principles?</article-title><source>Nucleic Acids Research</source><volume>33</volume><fpage>7138</fpage><lpage>7150</lpage><pub-id pub-id-type="doi">10.1093/nar/gki1012</pub-id><pub-id pub-id-type="pmid">16391004</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Barrett</surname><given-names>LW</given-names></name><name><surname>Fletcher</surname><given-names>S</given-names></name><name><surname>Wilton</surname><given-names>SD</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Regulation of eukaryotic gene expression by the untranslated gene regions and other non-coding elements</article-title><source>Cellular and Molecular Life Sciences</source><volume>69</volume><fpage>3613</fpage><lpage>3634</lpage><pub-id pub-id-type="doi">10.1007/s00018-012-0990-9</pub-id><pub-id pub-id-type="pmid">22538991</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bartel</surname><given-names>DP</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>MicroRNAs: target recognition and regulatory functions</article-title><source>Cell</source><volume>136</volume><fpage>215</fpage><lpage>233</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2009.01.002</pub-id><pub-id pub-id-type="pmid">19167326</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Beadle</surname><given-names>LF</given-names></name><name><surname>Love</surname><given-names>JC</given-names></name><name><surname>Shapovalova</surname><given-names>Y</given-names></name><name><surname>Artemev</surname><given-names>A</given-names></name><name><surname>Rattray</surname><given-names>M</given-names></name><name><surname>Ashe</surname><given-names>HL</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Combined modelling of mRNA decay dynamics and single-molecule imaging in the embryo uncovers a role for P-bodies in 5′ to 3′ degradation</article-title><source>PLOS Biology</source><volume>21</volume><elocation-id>e3001956</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.3001956</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Beffagna</surname><given-names>G</given-names></name><name><surname>Occhi</surname><given-names>G</given-names></name><name><surname>Nava</surname><given-names>A</given-names></name><name><surname>Vitiello</surname><given-names>L</given-names></name><name><surname>Ditadi</surname><given-names>A</given-names></name><name><surname>Basso</surname><given-names>C</given-names></name><name><surname>Bauce</surname><given-names>B</given-names></name><name><surname>Carraro</surname><given-names>G</given-names></name><name><surname>Thiene</surname><given-names>G</given-names></name><name><surname>Towbin</surname><given-names>JA</given-names></name><name><surname>Danieli</surname><given-names>GA</given-names></name><name><surname>Rampazzo</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Regulatory mutations in transforming growth factor-beta3 gene cause arrhythmogenic right ventricular cardiomyopathy type 1</article-title><source>Cardiovascular Research</source><volume>65</volume><fpage>366</fpage><lpage>373</lpage><pub-id pub-id-type="doi">10.1016/j.cardiores.2004.10.005</pub-id><pub-id pub-id-type="pmid">15639475</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Binns</surname><given-names>D</given-names></name><name><surname>Dimmer</surname><given-names>E</given-names></name><name><surname>Huntley</surname><given-names>R</given-names></name><name><surname>Barrell</surname><given-names>D</given-names></name><name><surname>O’Donovan</surname><given-names>C</given-names></name><name><surname>Apweiler</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>QuickGO: a web-based tool for gene ontology searching</article-title><source>Bioinformatics</source><volume>25</volume><fpage>3045</fpage><lpage>3046</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btp536</pub-id><pub-id pub-id-type="pmid">19744993</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Blumberg</surname><given-names>A</given-names></name><name><surname>Zhao</surname><given-names>YX</given-names></name><name><surname>Huang</surname><given-names>YF</given-names></name><name><surname>Dukler</surname><given-names>N</given-names></name><name><surname>Rice</surname><given-names>EJ</given-names></name><name><surname>Chivu</surname><given-names>AG</given-names></name><name><surname>Krumholz</surname><given-names>K</given-names></name><name><surname>Danko</surname><given-names>CG</given-names></name><name><surname>Siepel</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Characterizing RNA stability genome-wide through combined analysis of PRO-seq and RNA-seq data</article-title><source>BMC Biology</source><volume>19</volume><elocation-id>30</elocation-id><pub-id pub-id-type="doi">10.1186/s12915-021-00949-x</pub-id><pub-id pub-id-type="pmid">33588838</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bolger</surname><given-names>AM</given-names></name><name><surname>Lohse</surname><given-names>M</given-names></name><name><surname>Usadel</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Trimmomatic: a flexible trimmer for Illumina sequence data</article-title><source>Bioinformatics</source><volume>30</volume><fpage>2114</fpage><lpage>2120</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btu170</pub-id><pub-id pub-id-type="pmid">24695404</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Borén</surname><given-names>J</given-names></name><name><surname>Packard</surname><given-names>CJ</given-names></name><name><surname>Taskinen</surname><given-names>M-R</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>The roles of apoC-III on the metabolism of triglyceride-rich lipoproteins in humans</article-title><source>Frontiers in Endocrinology</source><volume>11</volume><elocation-id>474</elocation-id><pub-id pub-id-type="doi">10.3389/fendo.2020.00474</pub-id><pub-id pub-id-type="pmid">32849270</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chiang</surname><given-names>HL</given-names></name><name><surname>Chen</surname><given-names>YT</given-names></name><name><surname>Su</surname><given-names>JY</given-names></name><name><surname>Lin</surname><given-names>HN</given-names></name><name><surname>Yu</surname><given-names>CHA</given-names></name><name><surname>Hung</surname><given-names>YJ</given-names></name><name><surname>Wang</surname><given-names>YL</given-names></name><name><surname>Huang</surname><given-names>YT</given-names></name><name><surname>Lin</surname><given-names>CL</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Mechanism and modeling of human disease-associated near-exon intronic variants that perturb RNA splicing</article-title><source>Nature Structural &amp; Molecular Biology</source><volume>29</volume><fpage>1043</fpage><lpage>1055</lpage><pub-id pub-id-type="doi">10.1038/s41594-022-00844-1</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Colaprico</surname><given-names>A</given-names></name><name><surname>Silva</surname><given-names>TC</given-names></name><name><surname>Olsen</surname><given-names>C</given-names></name><name><surname>Garofano</surname><given-names>L</given-names></name><name><surname>Cava</surname><given-names>C</given-names></name><name><surname>Garolini</surname><given-names>D</given-names></name><name><surname>Sabedot</surname><given-names>TS</given-names></name><name><surname>Malta</surname><given-names>TM</given-names></name><name><surname>Pagnotta</surname><given-names>SM</given-names></name><name><surname>Castiglioni</surname><given-names>I</given-names></name><name><surname>Ceccarelli</surname><given-names>M</given-names></name><name><surname>Bontempi</surname><given-names>G</given-names></name><name><surname>Noushmehr</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>TCGAbiolinks: an R/Bioconductor package for integrative analysis of TCGA data</article-title><source>Nucleic Acids Research</source><volume>44</volume><elocation-id>e71</elocation-id><pub-id pub-id-type="doi">10.1093/nar/gkv1507</pub-id><pub-id pub-id-type="pmid">26704973</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Courel</surname><given-names>M</given-names></name><name><surname>Clément</surname><given-names>Y</given-names></name><name><surname>Bossevain</surname><given-names>C</given-names></name><name><surname>Foretek</surname><given-names>D</given-names></name><name><surname>Vidal Cruchez</surname><given-names>O</given-names></name><name><surname>Yi</surname><given-names>Z</given-names></name><name><surname>Bénard</surname><given-names>M</given-names></name><name><surname>Benassy</surname><given-names>M-N</given-names></name><name><surname>Kress</surname><given-names>M</given-names></name><name><surname>Vindry</surname><given-names>C</given-names></name><name><surname>Ernoult-Lange</surname><given-names>M</given-names></name><name><surname>Antoniewski</surname><given-names>C</given-names></name><name><surname>Morillon</surname><given-names>A</given-names></name><name><surname>Brest</surname><given-names>P</given-names></name><name><surname>Hubstenberger</surname><given-names>A</given-names></name><name><surname>Roest Crollius</surname><given-names>H</given-names></name><name><surname>Standart</surname><given-names>N</given-names></name><name><surname>Weil</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>GC content shapes mRNA storage and decay in human cells</article-title><source>eLife</source><volume>8</volume><elocation-id>e49708</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.49708</pub-id><pub-id pub-id-type="pmid">31855182</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dai</surname><given-names>H</given-names></name><name><surname>Guan</surname><given-names>YT</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>The Nubeam reference-free approach to analyze metagenomic sequencing reads</article-title><source>Genome Research</source><volume>30</volume><fpage>1364</fpage><lpage>1375</lpage><pub-id pub-id-type="doi">10.1101/gr.261750.120</pub-id><pub-id pub-id-type="pmid">32883749</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dumas</surname><given-names>L</given-names></name><name><surname>Herviou</surname><given-names>P</given-names></name><name><surname>Dassi</surname><given-names>E</given-names></name><name><surname>Cammas</surname><given-names>A</given-names></name><name><surname>Millevoi</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>G-Quadruplexes in RNA biology: recent advances and future directions</article-title><source>Trends in Biochemical Sciences</source><volume>46</volume><fpage>270</fpage><lpage>283</lpage><pub-id pub-id-type="doi">10.1016/j.tibs.2020.11.001</pub-id><pub-id pub-id-type="pmid">33303320</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dusl</surname><given-names>M</given-names></name><name><surname>Senderek</surname><given-names>J</given-names></name><name><surname>Müller</surname><given-names>JS</given-names></name><name><surname>Vogel</surname><given-names>JG</given-names></name><name><surname>Pertl</surname><given-names>A</given-names></name><name><surname>Stucka</surname><given-names>R</given-names></name><name><surname>Lochmüller</surname><given-names>H</given-names></name><name><surname>David</surname><given-names>R</given-names></name><name><surname>Abicht</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>A 3’-UTR mutation creates A microRNA target site in the GFPT1 gene of patients with congenital myasthenic syndrome</article-title><source>Human Molecular Genetics</source><volume>24</volume><fpage>3418</fpage><lpage>3426</lpage><pub-id pub-id-type="doi">10.1093/hmg/ddv090</pub-id><pub-id pub-id-type="pmid">25765662</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fallmann</surname><given-names>J</given-names></name><name><surname>Sedlyarov</surname><given-names>V</given-names></name><name><surname>Tanzer</surname><given-names>A</given-names></name><name><surname>Kovarik</surname><given-names>P</given-names></name><name><surname>Hofacker</surname><given-names>IL</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>AREsite2: an enhanced database for the comprehensive investigation of AU/GU/U-rich elements</article-title><source>Nucleic Acids Research</source><volume>44</volume><fpage>D90</fpage><lpage>D95</lpage><pub-id pub-id-type="doi">10.1093/nar/gkv1238</pub-id><pub-id pub-id-type="pmid">26602692</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Friedman</surname><given-names>J</given-names></name><name><surname>Hastie</surname><given-names>T</given-names></name><name><surname>Tibshirani</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Regularization paths for generalized linear models via coordinate descent</article-title><source>Journal of Statistical Software</source><volume>33</volume><fpage>1</fpage><lpage>22</lpage><pub-id pub-id-type="doi">10.18637/jss.v033.i01</pub-id><pub-id pub-id-type="pmid">20808728</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Garneau</surname><given-names>NL</given-names></name><name><surname>Wilusz</surname><given-names>J</given-names></name><name><surname>Wilusz</surname><given-names>CJ</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>The highways and byways of mRNA decay</article-title><source>Nature Reviews. Molecular Cell Biology</source><volume>8</volume><fpage>113</fpage><lpage>126</lpage><pub-id pub-id-type="doi">10.1038/nrm2104</pub-id><pub-id pub-id-type="pmid">17245413</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Giudice</surname><given-names>G</given-names></name><name><surname>Sánchez-Cabo</surname><given-names>F</given-names></name><name><surname>Torroja</surname><given-names>C</given-names></name><name><surname>Lara-Pezzi</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>ATtRACT-a database of RNA-binding proteins and associated motifs</article-title><source>Database</source><volume>2016</volume><elocation-id>baw035</elocation-id><pub-id pub-id-type="doi">10.1093/database/baw035</pub-id><pub-id pub-id-type="pmid">27055826</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Goyal</surname><given-names>S</given-names></name><name><surname>Tanigawa</surname><given-names>Y</given-names></name><name><surname>Zhang</surname><given-names>W</given-names></name><name><surname>Chai</surname><given-names>J-F</given-names></name><name><surname>Almeida</surname><given-names>M</given-names></name><name><surname>Sim</surname><given-names>X</given-names></name><name><surname>Lerner</surname><given-names>M</given-names></name><name><surname>Chainakul</surname><given-names>J</given-names></name><name><surname>Ramiu</surname><given-names>JG</given-names></name><name><surname>Seraphin</surname><given-names>C</given-names></name><name><surname>Apple</surname><given-names>B</given-names></name><name><surname>Vaughan</surname><given-names>A</given-names></name><name><surname>Muniu</surname><given-names>J</given-names></name><name><surname>Peralta</surname><given-names>J</given-names></name><name><surname>Lehman</surname><given-names>DM</given-names></name><name><surname>Ralhan</surname><given-names>S</given-names></name><name><surname>Wander</surname><given-names>GS</given-names></name><name><surname>Singh</surname><given-names>JR</given-names></name><name><surname>Mehra</surname><given-names>NK</given-names></name><name><surname>Sidorov</surname><given-names>E</given-names></name><name><surname>Peyton</surname><given-names>MD</given-names></name><name><surname>Blackett</surname><given-names>PR</given-names></name><name><surname>Curran</surname><given-names>JE</given-names></name><name><surname>Tai</surname><given-names>ES</given-names></name><name><surname>van Dam</surname><given-names>R</given-names></name><name><surname>Cheng</surname><given-names>C-Y</given-names></name><name><surname>Duggirala</surname><given-names>R</given-names></name><name><surname>Blangero</surname><given-names>J</given-names></name><name><surname>Chambers</surname><given-names>JC</given-names></name><name><surname>Sabanayagam</surname><given-names>C</given-names></name><name><surname>Kooner</surname><given-names>JS</given-names></name><name><surname>Rivas</surname><given-names>MA</given-names></name><name><surname>Aston</surname><given-names>CE</given-names></name><name><surname>Sanghera</surname><given-names>DK</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>APOC3 genetic variation, serum triglycerides, and risk of coronary artery disease in Asian Indians, Europeans, and other ethnic groups</article-title><source>Lipids in Health and Disease</source><volume>20</volume><elocation-id>113</elocation-id><pub-id pub-id-type="doi">10.1186/s12944-021-01531-8</pub-id><pub-id pub-id-type="pmid">34548093</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Griesemer</surname><given-names>D</given-names></name><name><surname>Xue</surname><given-names>JR</given-names></name><name><surname>Reilly</surname><given-names>SK</given-names></name><name><surname>Ulirsch</surname><given-names>JC</given-names></name><name><surname>Kukreja</surname><given-names>K</given-names></name><name><surname>Davis</surname><given-names>JR</given-names></name><name><surname>Kanai</surname><given-names>M</given-names></name><name><surname>Yang</surname><given-names>DK</given-names></name><name><surname>Butts</surname><given-names>JC</given-names></name><name><surname>Guney</surname><given-names>MH</given-names></name><name><surname>Luban</surname><given-names>J</given-names></name><name><surname>Montgomery</surname><given-names>SB</given-names></name><name><surname>Finucane</surname><given-names>HK</given-names></name><name><surname>Novina</surname><given-names>CD</given-names></name><name><surname>Tewhey</surname><given-names>R</given-names></name><name><surname>Sabeti</surname><given-names>PC</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Genome-wide functional screen of 3’UTR variants uncovers causal variants for human disease and evolution</article-title><source>Cell</source><volume>184</volume><fpage>5247</fpage><lpage>5260</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2021.08.025</pub-id><pub-id pub-id-type="pmid">34534445</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guvenek</surname><given-names>A</given-names></name><name><surname>Shin</surname><given-names>J</given-names></name><name><surname>De Filippis</surname><given-names>L</given-names></name><name><surname>Zheng</surname><given-names>D</given-names></name><name><surname>Wang</surname><given-names>W</given-names></name><name><surname>Pang</surname><given-names>ZP</given-names></name><name><surname>Tian</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Neuronal cells display distinct stability controls of alternative polyadenylation mRNA isoforms, long non-coding RNAs, and mitochondrial RNAs</article-title><source>Frontiers in Genetics</source><volume>13</volume><elocation-id>840369</elocation-id><pub-id pub-id-type="doi">10.3389/fgene.2022.840369</pub-id><pub-id pub-id-type="pmid">35664307</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hernández</surname><given-names>G</given-names></name><name><surname>Osnaya</surname><given-names>VG</given-names></name><name><surname>Pérez-Martínez</surname><given-names>X</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Conservation and variability of the aug initiation codon context in eukaryotes</article-title><source>Trends in Biochemical Sciences</source><volume>44</volume><fpage>1009</fpage><lpage>1021</lpage><pub-id pub-id-type="doi">10.1016/j.tibs.2019.07.001</pub-id><pub-id pub-id-type="pmid">31353284</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hinrichs</surname><given-names>AS</given-names></name><name><surname>Karolchik</surname><given-names>D</given-names></name><name><surname>Baertsch</surname><given-names>R</given-names></name><name><surname>Barber</surname><given-names>GP</given-names></name><name><surname>Bejerano</surname><given-names>G</given-names></name><name><surname>Clawson</surname><given-names>H</given-names></name><name><surname>Diekhans</surname><given-names>M</given-names></name><name><surname>Furey</surname><given-names>TS</given-names></name><name><surname>Harte</surname><given-names>RA</given-names></name><name><surname>Hsu</surname><given-names>F</given-names></name><name><surname>Hillman-Jackson</surname><given-names>J</given-names></name><name><surname>Kuhn</surname><given-names>RM</given-names></name><name><surname>Pedersen</surname><given-names>JS</given-names></name><name><surname>Pohl</surname><given-names>A</given-names></name><name><surname>Raney</surname><given-names>BJ</given-names></name><name><surname>Rosenbloom</surname><given-names>KR</given-names></name><name><surname>Siepel</surname><given-names>A</given-names></name><name><surname>Smith</surname><given-names>KE</given-names></name><name><surname>Sugnet</surname><given-names>CW</given-names></name><name><surname>Sultan-Qurraie</surname><given-names>A</given-names></name><name><surname>Thomas</surname><given-names>DJ</given-names></name><name><surname>Trumbower</surname><given-names>H</given-names></name><name><surname>Weber</surname><given-names>RJ</given-names></name><name><surname>Weirauch</surname><given-names>M</given-names></name><name><surname>Zweig</surname><given-names>AS</given-names></name><name><surname>Haussler</surname><given-names>D</given-names></name><name><surname>Kent</surname><given-names>WJ</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>The UCSC genome browser database: update 2006</article-title><source>Nucleic Acids Research</source><volume>34</volume><fpage>D590</fpage><lpage>D598</lpage><pub-id pub-id-type="doi">10.1093/nar/gkj144</pub-id><pub-id pub-id-type="pmid">16381938</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hothorn</surname><given-names>T</given-names></name><name><surname>Hornik</surname><given-names>K</given-names></name><name><surname>Wiel</surname><given-names>MA</given-names></name><name><surname>van de Zeileis</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Implementing a class of permutation tests: the coin package</article-title><source>Journal of Statistical Software</source><volume>28</volume><fpage>1</fpage><lpage>23</lpage><pub-id pub-id-type="doi">10.18637/jss.v028.i08</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hubstenberger</surname><given-names>A</given-names></name><name><surname>Courel</surname><given-names>M</given-names></name><name><surname>Bénard</surname><given-names>M</given-names></name><name><surname>Souquere</surname><given-names>S</given-names></name><name><surname>Ernoult-Lange</surname><given-names>M</given-names></name><name><surname>Chouaib</surname><given-names>R</given-names></name><name><surname>Yi</surname><given-names>Z</given-names></name><name><surname>Morlot</surname><given-names>JB</given-names></name><name><surname>Munier</surname><given-names>A</given-names></name><name><surname>Fradet</surname><given-names>M</given-names></name><name><surname>Daunesse</surname><given-names>M</given-names></name><name><surname>Bertrand</surname><given-names>E</given-names></name><name><surname>Pierron</surname><given-names>G</given-names></name><name><surname>Mozziconacci</surname><given-names>J</given-names></name><name><surname>Kress</surname><given-names>M</given-names></name><name><surname>Weil</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>P-Body purification reveals the condensation of repressed mRNA regulons</article-title><source>Molecular Cell</source><volume>68</volume><fpage>144</fpage><lpage>157</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2017.09.003</pub-id><pub-id pub-id-type="pmid">28965817</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huntzinger</surname><given-names>E</given-names></name><name><surname>Izaurralde</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Gene silencing by microRNAs: contributions of translational repression and mRNA decay</article-title><source>Nature Reviews. Genetics</source><volume>12</volume><fpage>99</fpage><lpage>110</lpage><pub-id pub-id-type="doi">10.1038/nrg2936</pub-id><pub-id pub-id-type="pmid">21245828</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jia</surname><given-names>LF</given-names></name><name><surname>Mao</surname><given-names>YH</given-names></name><name><surname>Ji</surname><given-names>QQ</given-names></name><name><surname>Dersh</surname><given-names>D</given-names></name><name><surname>Yewdell</surname><given-names>JW</given-names></name><name><surname>Qian</surname><given-names>SB</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Decoding mRNA translatability and stability from the 5′ UTR</article-title><source>Nature Structural &amp; Molecular Biology</source><volume>27</volume><fpage>814</fpage><lpage>821</lpage><pub-id pub-id-type="doi">10.1038/s41594-020-0465-x</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname><given-names>D</given-names></name><name><surname>Langmead</surname><given-names>B</given-names></name><name><surname>Salzberg</surname><given-names>SL</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>HISAT: a fast spliced aligner with low memory requirements</article-title><source>Nature Methods</source><volume>12</volume><fpage>357</fpage><lpage>360</lpage><pub-id pub-id-type="doi">10.1038/nmeth.3317</pub-id><pub-id pub-id-type="pmid">25751142</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kinsella</surname><given-names>RJ</given-names></name><name><surname>Kähäri</surname><given-names>A</given-names></name><name><surname>Haider</surname><given-names>S</given-names></name><name><surname>Zamora</surname><given-names>J</given-names></name><name><surname>Proctor</surname><given-names>G</given-names></name><name><surname>Spudich</surname><given-names>G</given-names></name><name><surname>Almeida-King</surname><given-names>J</given-names></name><name><surname>Staines</surname><given-names>D</given-names></name><name><surname>Derwent</surname><given-names>P</given-names></name><name><surname>Kerhornou</surname><given-names>A</given-names></name><name><surname>Kersey</surname><given-names>P</given-names></name><name><surname>Flicek</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Ensembl BioMarts: a hub for data retrieval across taxonomic space</article-title><source>Database</source><volume>2011</volume><elocation-id>bar030</elocation-id><pub-id pub-id-type="doi">10.1093/database/bar030</pub-id><pub-id pub-id-type="pmid">21785142</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Landrum</surname><given-names>MJ</given-names></name><name><surname>Lee</surname><given-names>JM</given-names></name><name><surname>Benson</surname><given-names>M</given-names></name><name><surname>Brown</surname><given-names>G</given-names></name><name><surname>Chao</surname><given-names>C</given-names></name><name><surname>Chitipiralla</surname><given-names>S</given-names></name><name><surname>Gu</surname><given-names>B</given-names></name><name><surname>Hart</surname><given-names>J</given-names></name><name><surname>Hoffman</surname><given-names>D</given-names></name><name><surname>Hoover</surname><given-names>J</given-names></name><name><surname>Jang</surname><given-names>W</given-names></name><name><surname>Katz</surname><given-names>K</given-names></name><name><surname>Ovetsky</surname><given-names>M</given-names></name><name><surname>Riley</surname><given-names>G</given-names></name><name><surname>Sethi</surname><given-names>A</given-names></name><name><surname>Tully</surname><given-names>R</given-names></name><name><surname>Villamarin-Salomon</surname><given-names>R</given-names></name><name><surname>Rubinstein</surname><given-names>W</given-names></name><name><surname>Maglott</surname><given-names>DR</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>ClinVar: public archive of interpretations of clinically relevant variants</article-title><source>Nucleic Acids Research</source><volume>44</volume><fpage>D862</fpage><lpage>D868</lpage><pub-id pub-id-type="doi">10.1093/nar/gkv1222</pub-id><pub-id pub-id-type="pmid">26582918</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname><given-names>BT</given-names></name><name><surname>Barber</surname><given-names>GP</given-names></name><name><surname>Benet-Pagès</surname><given-names>A</given-names></name><name><surname>Casper</surname><given-names>J</given-names></name><name><surname>Clawson</surname><given-names>H</given-names></name><name><surname>Diekhans</surname><given-names>M</given-names></name><name><surname>Fischer</surname><given-names>C</given-names></name><name><surname>Gonzalez</surname><given-names>JN</given-names></name><name><surname>Hinrichs</surname><given-names>AS</given-names></name><name><surname>Lee</surname><given-names>CM</given-names></name><name><surname>Muthuraman</surname><given-names>P</given-names></name><name><surname>Nassar</surname><given-names>LR</given-names></name><name><surname>Nguy</surname><given-names>B</given-names></name><name><surname>Pereira</surname><given-names>T</given-names></name><name><surname>Perez</surname><given-names>G</given-names></name><name><surname>Raney</surname><given-names>BJ</given-names></name><name><surname>Rosenbloom</surname><given-names>KR</given-names></name><name><surname>Schmelter</surname><given-names>D</given-names></name><name><surname>Speir</surname><given-names>ML</given-names></name><name><surname>Wick</surname><given-names>BD</given-names></name><name><surname>Zweig</surname><given-names>AS</given-names></name><name><surname>Haussler</surname><given-names>D</given-names></name><name><surname>Kuhn</surname><given-names>RM</given-names></name><name><surname>Haeussler</surname><given-names>M</given-names></name><name><surname>Kent</surname><given-names>WJ</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>The UCSC genome browser database: 2022 update</article-title><source>Nucleic Acids Research</source><volume>50</volume><fpage>D1115</fpage><lpage>D1122</lpage><pub-id pub-id-type="doi">10.1093/nar/gkab959</pub-id><pub-id pub-id-type="pmid">34718705</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Litterman</surname><given-names>AJ</given-names></name><name><surname>Kageyama</surname><given-names>R</given-names></name><name><surname>Le Tonqueze</surname><given-names>O</given-names></name><name><surname>Zhao</surname><given-names>WX</given-names></name><name><surname>Gagnon</surname><given-names>JD</given-names></name><name><surname>Goodarzi</surname><given-names>H</given-names></name><name><surname>Erle</surname><given-names>DJ</given-names></name><name><surname>Ansel</surname><given-names>KM</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>A massively parallel 3’ UTR reporter assay reveals relationships between nucleotide content, sequence conservation, and mRNA destabilization</article-title><source>Genome Research</source><volume>29</volume><fpage>896</fpage><lpage>906</lpage><pub-id pub-id-type="doi">10.1101/gr.242552.118</pub-id><pub-id pub-id-type="pmid">31152051</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lorenz</surname><given-names>R</given-names></name><name><surname>Bernhart</surname><given-names>SH</given-names></name><name><surname>Höner Zu Siederdissen</surname><given-names>C</given-names></name><name><surname>Tafer</surname><given-names>H</given-names></name><name><surname>Flamm</surname><given-names>C</given-names></name><name><surname>Stadler</surname><given-names>PF</given-names></name><name><surname>Hofacker</surname><given-names>IL</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>ViennaRNA Package 2.0</article-title><source>Algorithms for Molecular Biology</source><volume>6</volume><elocation-id>26</elocation-id><pub-id pub-id-type="doi">10.1186/1748-7188-6-26</pub-id><pub-id pub-id-type="pmid">22115189</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Luo</surname><given-names>Y</given-names></name><name><surname>Hitz</surname><given-names>BC</given-names></name><name><surname>Gabdank</surname><given-names>I</given-names></name><name><surname>Hilton</surname><given-names>JA</given-names></name><name><surname>Kagda</surname><given-names>MS</given-names></name><name><surname>Lam</surname><given-names>B</given-names></name><name><surname>Myers</surname><given-names>Z</given-names></name><name><surname>Sud</surname><given-names>P</given-names></name><name><surname>Jou</surname><given-names>J</given-names></name><name><surname>Lin</surname><given-names>K</given-names></name><name><surname>Baymuradov</surname><given-names>UK</given-names></name><name><surname>Graham</surname><given-names>K</given-names></name><name><surname>Litton</surname><given-names>C</given-names></name><name><surname>Miyasato</surname><given-names>SR</given-names></name><name><surname>Strattan</surname><given-names>JS</given-names></name><name><surname>Jolanki</surname><given-names>O</given-names></name><name><surname>Lee</surname><given-names>J-W</given-names></name><name><surname>Tanaka</surname><given-names>FY</given-names></name><name><surname>Adenekan</surname><given-names>P</given-names></name><name><surname>O’Neill</surname><given-names>E</given-names></name><name><surname>Cherry</surname><given-names>JM</given-names></name></person-group><year iso-8601-date="2020">2020a</year><article-title>New developments on the Encyclopedia of DNA Elements (ENCODE) data portal</article-title><source>Nucleic Acids Research</source><volume>48</volume><fpage>D882</fpage><lpage>D889</lpage><pub-id pub-id-type="doi">10.1093/nar/gkz1062</pub-id><pub-id pub-id-type="pmid">31713622</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Luo</surname><given-names>EC</given-names></name><name><surname>Nathanson</surname><given-names>JL</given-names></name><name><surname>Tan</surname><given-names>FE</given-names></name><name><surname>Schwartz</surname><given-names>JL</given-names></name><name><surname>Schmok</surname><given-names>JC</given-names></name><name><surname>Shankar</surname><given-names>A</given-names></name><name><surname>Markmiller</surname><given-names>S</given-names></name><name><surname>Yee</surname><given-names>BA</given-names></name><name><surname>Sathe</surname><given-names>S</given-names></name><name><surname>Pratt</surname><given-names>GA</given-names></name><name><surname>Scaletta</surname><given-names>DB</given-names></name><name><surname>Ha</surname><given-names>Y</given-names></name><name><surname>Hill</surname><given-names>DE</given-names></name><name><surname>Aigner</surname><given-names>S</given-names></name><name><surname>Yeo</surname><given-names>GW</given-names></name></person-group><year iso-8601-date="2020">2020b</year><article-title>Large-scale tethered function assays identify factors that regulate mRNA stability and translation</article-title><source>Nature Structural &amp; Molecular Biology</source><volume>27</volume><fpage>989</fpage><lpage>1000</lpage><pub-id pub-id-type="doi">10.1038/s41594-020-0477-6</pub-id><pub-id pub-id-type="pmid">32807991</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>MacArthur</surname><given-names>J</given-names></name><name><surname>Bowler</surname><given-names>E</given-names></name><name><surname>Cerezo</surname><given-names>M</given-names></name><name><surname>Gil</surname><given-names>L</given-names></name><name><surname>Hall</surname><given-names>P</given-names></name><name><surname>Hastings</surname><given-names>E</given-names></name><name><surname>Junkins</surname><given-names>H</given-names></name><name><surname>McMahon</surname><given-names>A</given-names></name><name><surname>Milano</surname><given-names>A</given-names></name><name><surname>Morales</surname><given-names>J</given-names></name><name><surname>Pendlington</surname><given-names>ZM</given-names></name><name><surname>Welter</surname><given-names>D</given-names></name><name><surname>Burdett</surname><given-names>T</given-names></name><name><surname>Hindorff</surname><given-names>L</given-names></name><name><surname>Flicek</surname><given-names>P</given-names></name><name><surname>Cunningham</surname><given-names>F</given-names></name><name><surname>Parkinson</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>The new NHGRI-EBI Catalog of published genome-wide association studies (GWAS Catalog)</article-title><source>Nucleic Acids Research</source><volume>45</volume><fpage>D896</fpage><lpage>D901</lpage><pub-id pub-id-type="doi">10.1093/nar/gkw1133</pub-id><pub-id pub-id-type="pmid">27899670</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McGeary</surname><given-names>SE</given-names></name><name><surname>Lin</surname><given-names>KS</given-names></name><name><surname>Shi</surname><given-names>CY</given-names></name><name><surname>Pham</surname><given-names>TM</given-names></name><name><surname>Bisaria</surname><given-names>N</given-names></name><name><surname>Kelley</surname><given-names>GM</given-names></name><name><surname>Bartel</surname><given-names>DP</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>The biochemical basis of microRNA targeting efficacy</article-title><source>Science</source><volume>366</volume><elocation-id>1470</elocation-id><pub-id pub-id-type="doi">10.1126/science.aav1741</pub-id><pub-id pub-id-type="pmid">31806698</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mignone</surname><given-names>F</given-names></name><name><surname>Gissi</surname><given-names>C</given-names></name><name><surname>Liuni</surname><given-names>S</given-names></name><name><surname>Pesole</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Untranslated regions of mRNAs</article-title><source>Genome Biology</source><volume>3</volume><elocation-id>ARTN</elocation-id><pub-id pub-id-type="doi">10.1186/gb-2002-3-3-reviews0004</pub-id><pub-id pub-id-type="pmid">11897027</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mitchell</surname><given-names>P</given-names></name><name><surname>Tollervey</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>MRNA turnover</article-title><source>Current Opinion in Cell Biology</source><volume>13</volume><fpage>320</fpage><lpage>325</lpage><pub-id pub-id-type="doi">10.1016/s0955-0674(00)00214-3</pub-id><pub-id pub-id-type="pmid">11343902</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Neymotin</surname><given-names>B</given-names></name><name><surname>Ettorre</surname><given-names>V</given-names></name><name><surname>Gresham</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Global Determinants of mRNA Degradation rates in <italic>Saccharomyces cerevisiae</italic></article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/014845</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Oikonomou</surname><given-names>P</given-names></name><name><surname>Goodarzi</surname><given-names>H</given-names></name><name><surname>Tavazoie</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Systematic identification of regulatory elements in conserved 3’ UTRs of human transcripts</article-title><source>Cell Reports</source><volume>7</volume><fpage>281</fpage><lpage>292</lpage><pub-id pub-id-type="doi">10.1016/j.celrep.2014.03.001</pub-id><pub-id pub-id-type="pmid">24656821</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Panwar</surname><given-names>B</given-names></name><name><surname>Omenn</surname><given-names>GS</given-names></name><name><surname>Guan</surname><given-names>YF</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>miRmine: a database of human miRNA expression profiles</article-title><source>Bioinformatics</source><volume>33</volume><fpage>1554</fpage><lpage>1560</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btx019</pub-id><pub-id pub-id-type="pmid">28108447</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Paschoud</surname><given-names>S</given-names></name><name><surname>Dogar</surname><given-names>AM</given-names></name><name><surname>Kuntz</surname><given-names>C</given-names></name><name><surname>Grisoni-Neupert</surname><given-names>B</given-names></name><name><surname>Richman</surname><given-names>L</given-names></name><name><surname>Kühn</surname><given-names>LC</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Destabilization of interleukin-6 mRNA requires a putative RNA stem-loop structure, an AU-rich element, and the RNA-binding protein AUF1</article-title><source>Molecular and Cellular Biology</source><volume>26</volume><fpage>8228</fpage><lpage>8241</lpage><pub-id pub-id-type="doi">10.1128/MCB.01155-06</pub-id><pub-id pub-id-type="pmid">16954375</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Perron</surname><given-names>G</given-names></name><name><surname>Jandaghi</surname><given-names>P</given-names></name><name><surname>Moslemi</surname><given-names>E</given-names></name><name><surname>Nishimura</surname><given-names>T</given-names></name><name><surname>Rajaee</surname><given-names>M</given-names></name><name><surname>Alkallas</surname><given-names>R</given-names></name><name><surname>Lu</surname><given-names>TY</given-names></name><name><surname>Riazalhosseini</surname><given-names>Y</given-names></name><name><surname>Najafabadi</surname><given-names>HS</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Pan-cancer analysis of mRNA stability for decoding tumour post-transcriptional programs</article-title><source>Communications Biology</source><volume>5</volume><elocation-id>851</elocation-id><pub-id pub-id-type="doi">10.1038/s42003-022-03796-w</pub-id><pub-id pub-id-type="pmid">35987939</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pesole</surname><given-names>G</given-names></name><name><surname>Mignone</surname><given-names>F</given-names></name><name><surname>Gissi</surname><given-names>C</given-names></name><name><surname>Grillo</surname><given-names>G</given-names></name><name><surname>Licciulli</surname><given-names>F</given-names></name><name><surname>Liuni</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>Structural and functional features of eukaryotic mRNA untranslated regions</article-title><source>Gene</source><volume>276</volume><fpage>73</fpage><lpage>81</lpage><pub-id pub-id-type="doi">10.1016/s0378-1119(01)00674-6</pub-id><pub-id pub-id-type="pmid">11591473</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Prats-Ejarque</surname><given-names>G</given-names></name><name><surname>Lu</surname><given-names>L</given-names></name><name><surname>Salazar</surname><given-names>VA</given-names></name><name><surname>Moussaoui</surname><given-names>M</given-names></name><name><surname>Boix</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Evolutionary trends in RNA base selectivity within the rnase a superfamily</article-title><source>Frontiers in Pharmacology</source><volume>10</volume><elocation-id>1170</elocation-id><pub-id pub-id-type="doi">10.3389/fphar.2019.01170</pub-id><pub-id pub-id-type="pmid">31649540</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Quinlan</surname><given-names>AR</given-names></name><name><surname>Hall</surname><given-names>IM</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>BEDTools: a flexible suite of utilities for comparing genomic features</article-title><source>Bioinformatics</source><volume>26</volume><fpage>841</fpage><lpage>842</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btq033</pub-id><pub-id pub-id-type="pmid">20110278</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rabani</surname><given-names>M</given-names></name><name><surname>Pieper</surname><given-names>L</given-names></name><name><surname>Chew</surname><given-names>GL</given-names></name><name><surname>Schier</surname><given-names>AF</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>A massively parallel reporter assay of 3’ utr sequences identifies in vivo rules for mrna degradation</article-title><source>Molecular Cell</source><volume>68</volume><fpage>1083</fpage><lpage>1094</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2017.11.014</pub-id><pub-id pub-id-type="pmid">29225039</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="software"><person-group person-group-type="author"><collab>R Development Core Team</collab></person-group><year iso-8601-date="2021">2021</year><data-title>R: a language and environment for statistical computing</data-title><publisher-loc>Vienna, Austria</publisher-loc><publisher-name>R Foundation for Statistical Computing</publisher-name><ext-link ext-link-type="uri" xlink:href="https://www.R-project.org">https://www.R-project.org</ext-link></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sample</surname><given-names>PJ</given-names></name><name><surname>Wang</surname><given-names>B</given-names></name><name><surname>Reid</surname><given-names>DW</given-names></name><name><surname>Presnyak</surname><given-names>V</given-names></name><name><surname>McFadyen</surname><given-names>IJ</given-names></name><name><surname>Morris</surname><given-names>DR</given-names></name><name><surname>Seelig</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Human 5’ UTR design and variant effect prediction from a massively parallel translation assay</article-title><source>Nature Biotechnology</source><volume>37</volume><fpage>803</fpage><lpage>809</lpage><pub-id pub-id-type="doi">10.1038/s41587-019-0164-5</pub-id><pub-id pub-id-type="pmid">31267113</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schoenberg</surname><given-names>DR</given-names></name><name><surname>Maquat</surname><given-names>LE</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Regulation of cytoplasmic mRNA decay</article-title><source>Nature Reviews. Genetics</source><volume>13</volume><fpage>246</fpage><lpage>259</lpage><pub-id pub-id-type="doi">10.1038/nrg3160</pub-id><pub-id pub-id-type="pmid">22392217</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sherry</surname><given-names>ST</given-names></name><name><surname>Ward</surname><given-names>MH</given-names></name><name><surname>Sirotkin</surname><given-names>K</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>dbSNP-database for single nucleotide polymorphisms and other classes of minor genetic variation</article-title><source>Genome Research</source><volume>9</volume><fpage>677</fpage><lpage>679</lpage><pub-id pub-id-type="pmid">10447503</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Siegel</surname><given-names>DA</given-names></name><name><surname>Le Tonqueze</surname><given-names>O</given-names></name><name><surname>Biton</surname><given-names>A</given-names></name><name><surname>Zaitlen</surname><given-names>N</given-names></name><name><surname>Erle</surname><given-names>DJ</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Massively 1063 parallel analysis of human 3 ’ UTRs reveals that AU-rich element length and registration predict mRNA destabilization</article-title><source>G3-Genes Genomes Genetics</source><volume>12</volume><elocation-id>jkab404</elocation-id><pub-id pub-id-type="doi">10.1093/g3journal/jkab404</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sorrentino</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>The eight human “canonical” ribonucleases: molecular diversity, catalytic properties, and special biological actions of the enzyme proteins</article-title><source>FEBS Letters</source><volume>584</volume><fpage>2194</fpage><lpage>2200</lpage><pub-id pub-id-type="doi">10.1016/j.febslet.2010.04.018</pub-id><pub-id pub-id-type="pmid">20388512</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stenson</surname><given-names>PD</given-names></name><name><surname>Mort</surname><given-names>M</given-names></name><name><surname>Ball</surname><given-names>EV</given-names></name><name><surname>Evans</surname><given-names>K</given-names></name><name><surname>Hayden</surname><given-names>M</given-names></name><name><surname>Heywood</surname><given-names>S</given-names></name><name><surname>Hussain</surname><given-names>M</given-names></name><name><surname>Phillips</surname><given-names>AD</given-names></name><name><surname>Cooper</surname><given-names>DN</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>The human gene mutation database: towards a comprehensive repository of inherited mutation data for medical research, genetic diagnosis and next-generation sequencing studies</article-title><source>Human Genetics</source><volume>136</volume><fpage>665</fpage><lpage>677</lpage><pub-id pub-id-type="doi">10.1007/s00439-017-1779-6</pub-id><pub-id pub-id-type="pmid">28349240</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Steri</surname><given-names>M</given-names></name><name><surname>Idda</surname><given-names>ML</given-names></name><name><surname>Whalen</surname><given-names>MB</given-names></name><name><surname>Orrù</surname><given-names>V</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Genetic variants in mRNA untranslated regions</article-title><source>Wiley Interdisciplinary Reviews. RNA</source><volume>9</volume><elocation-id>e1474</elocation-id><pub-id pub-id-type="doi">10.1002/wrna.1474</pub-id><pub-id pub-id-type="pmid">29582564</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Su</surname><given-names>LY</given-names></name></person-group><year iso-8601-date="2025">2025</year><data-title>Modeling-UTR-variants-stability</data-title><version designator="swh:1:rev:30b9983b77a5673a090316d0068a28ec1c0ef510">swh:1:rev:30b9983b77a5673a090316d0068a28ec1c0ef510</version><source>Software Heritage</source><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:f498ed477a99d9dd4b1acaeca2a2c0f6a4d01da1;origin=https://github.com/chienlinglin/modeling-UTR-variants-stability;visit=swh:1:snp:0d14ef25591a724a4389c40e686b3b7281afd68c;anchor=swh:1:rev:30b9983b77a5673a090316d0068a28ec1c0ef510">https://archive.softwareheritage.org/swh:1:dir:f498ed477a99d9dd4b1acaeca2a2c0f6a4d01da1;origin=https://github.com/chienlinglin/modeling-UTR-variants-stability;visit=swh:1:snp:0d14ef25591a724a4389c40e686b3b7281afd68c;anchor=swh:1:rev:30b9983b77a5673a090316d0068a28ec1c0ef510</ext-link></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vainberg Slutskin</surname><given-names>I</given-names></name><name><surname>Weingarten-Gabbay</surname><given-names>S</given-names></name><name><surname>Nir</surname><given-names>R</given-names></name><name><surname>Weinberger</surname><given-names>A</given-names></name><name><surname>Segal</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Unraveling the determinants of microRNA mediated regulation using a massively parallel reporter assay</article-title><source>Nature Communications</source><volume>9</volume><elocation-id>529</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-018-02980-z</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>van der Loo</surname><given-names>MPJ</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>The stringdist package for approximate string matching</article-title><source>The R Journal</source><volume>6</volume><elocation-id>111</elocation-id><pub-id pub-id-type="doi">10.32614/RJ-2014-011</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vejnar</surname><given-names>CE</given-names></name><name><surname>Abdel Messih</surname><given-names>M</given-names></name><name><surname>Takacs</surname><given-names>CM</given-names></name><name><surname>Yartseva</surname><given-names>V</given-names></name><name><surname>Oikonomou</surname><given-names>P</given-names></name><name><surname>Christiano</surname><given-names>R</given-names></name><name><surname>Stoeckius</surname><given-names>M</given-names></name><name><surname>Lau</surname><given-names>S</given-names></name><name><surname>Lee</surname><given-names>MT</given-names></name><name><surname>Beaudoin</surname><given-names>J-D</given-names></name><name><surname>Musaev</surname><given-names>D</given-names></name><name><surname>Darwich-Codore</surname><given-names>H</given-names></name><name><surname>Walther</surname><given-names>TC</given-names></name><name><surname>Tavazoie</surname><given-names>S</given-names></name><name><surname>Cifuentes</surname><given-names>D</given-names></name><name><surname>Giraldez</surname><given-names>AJ</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Genome wide analysis of 3’ UTR sequence elements and proteins regulating mRNA stability during maternal-to-zygotic transition in zebrafish</article-title><source>Genome Research</source><volume>29</volume><fpage>1100</fpage><lpage>1114</lpage><pub-id pub-id-type="doi">10.1101/gr.245159.118</pub-id><pub-id pub-id-type="pmid">31227602</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wan</surname><given-names>J</given-names></name><name><surname>Qian</surname><given-names>SB</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>TISdb: a database for alternative translation initiation in mammalian cells</article-title><source>Nucleic Acids Research</source><volume>42</volume><fpage>D845</fpage><lpage>D850</lpage><pub-id pub-id-type="doi">10.1093/nar/gkt1085</pub-id><pub-id pub-id-type="pmid">24203712</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wissink</surname><given-names>EM</given-names></name><name><surname>Fogarty</surname><given-names>EA</given-names></name><name><surname>Grimson</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>High-throughput discovery of post-transcriptional cis-regulatory elements</article-title><source>BMC Genomics</source><volume>17</volume><elocation-id>177</elocation-id><pub-id pub-id-type="doi">10.1186/s12864-016-2479-7</pub-id><pub-id pub-id-type="pmid">26941072</pub-id></element-citation></ref><ref id="bib67"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Park</surname><given-names>C</given-names></name><name><surname>Bennett</surname><given-names>C</given-names></name><name><surname>Thornton</surname><given-names>M</given-names></name><name><surname>Kim</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Rapid and accurate alignment of nucleotide conversion sequencing reads with HISAT-3N</article-title><source>Genome Research</source><volume>31</volume><fpage>1290</fpage><lpage>1295</lpage><pub-id pub-id-type="doi">10.1101/gr.275193.120</pub-id><pub-id pub-id-type="pmid">34103331</pub-id></element-citation></ref><ref id="bib68"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname><given-names>WX</given-names></name><name><surname>Pollack</surname><given-names>JL</given-names></name><name><surname>Blagev</surname><given-names>DP</given-names></name><name><surname>Zaitlen</surname><given-names>N</given-names></name><name><surname>McManus</surname><given-names>MT</given-names></name><name><surname>Erle</surname><given-names>DJ</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Massively parallel functional annotation of 3’ untranslated regions</article-title><source>Nature Biotechnology</source><volume>32</volume><fpage>387</fpage><lpage>391</lpage><pub-id pub-id-type="doi">10.1038/nbt.2851</pub-id><pub-id pub-id-type="pmid">24633241</pub-id></element-citation></ref></ref-list></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.97682.3.sa0</article-id><title-group><article-title>eLife Assessment</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Calarco</surname><given-names>John</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution>University of Toronto</institution><country>Canada</country></aff></contrib></contrib-group><kwd-group kwd-group-type="evidence-strength"><kwd>Solid</kwd></kwd-group><kwd-group kwd-group-type="claim-importance"><kwd>Valuable</kwd></kwd-group></front-stub><body><p>This <bold>valuable</bold> study combines massively parallel reporter assays and regression analysis to identify sequence features in untranslated regions contributing to the stability of in vitro transcribed mRNA delivered to cells. The strength of evidence presented is <bold>solid</bold>, although some points about half-life measurements and the relevance of identified sequence features to native transcript stability will inform future discussion surrounding the present study. Taken together, the work will be of interest to a broad swath of colleagues studying post-transcriptional gene regulation and especially to those using massively parallel reporter assays.</p></body></sub-article><sub-article article-type="referee-report" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.97682.3.sa1</article-id><title-group><article-title>Reviewer #1 (Public review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>In the manuscript by Su et al., the authors present a massively parallel reporter assay (MPRA) measuring the stability of in vitro transcribed mRNAs carrying wild-type or mutant 5' or 3' UTRs transfected into two different human cell lines. The goal presented at the beginning of the manuscript was to screen for effects of disease-associated point mutations on the stability of the reporter RNAs carrying partial human 5' or 3' UTRs. However, the majority of the manuscript is dedicated to identifying sequence components underlying the differential stability of reporter constructs. This analysis showed that UA dinucleotides are the most predictive feature of RNA stability in both cell lines and both UTRs.</p><p>The effect of AU rich elements (AREs) on RNA stability is well established in multiple systems, and the present study confirms this general trend, but points out variability in the consequence of seemingly similar motifs on RNA stability. For example, the authors report that a long stretch of Us has extreme opposite effects on RNA stability depending on whether it is preceded by an A (strongly destabilizing) or followed by an A (strongly stabilizing). While the authors interpretation of a context-dependence of the effect is certainly well-founded, it seems counterintuitive that the preceding or following A would be the (only) determining factor. This points to a generally reductionist approach taken by the authors in the analysis of the data and in their attempt to dissect the contribution of &quot;AU rich sequences&quot; to RNA stability, with a general tendency to reduce the size and complexity of the features (e.g. to dinucleotides). While this certainly increases the statistical power of the analysis due to the number of occurrences of these motifs, it limits the interpretability of the results. How do UA dinucleotides per se contribute to destabilizing the RNA, both in 5' and 3' UTRs, but (according to limited data presented) not in coding sequences? What is the mechanism? RBPs binding to UA dinucleotide containing sequences are suggested to &quot;mask&quot; the destabilizing effect, thereby leading to a more stable RNA. Gain of UA dinucleotides is reported to have a destabilizing effect, but again no hypothesis is provided as to the underlying molecular mechanism. In addition to reducing the motif length to dinucleotides, the notion of &quot;context dependence&quot; is used in a very narrow sense.</p><p>The present MPRA measures the effect of UTR sequences in one specific reporter context and using one experimental approach (following the decay of in vitro transcribed and transfected RNAs). While this method certainly has its merits compared to other approaches, it also comes with some caveats: RNA is delivered naked, without bound RBPs and no nuclear history, e.g. of splicing (no EJCs), editing and modifications. Therefore, it remains to be seen whether UA dinucleotide frequency is a substantial factor in determining the half-lives of endogenous mRNAs.</p><p>The authors conclude their study with a meta-analysis of genes with increased UA dinucleotides in 5' and 3'UTRs, showing that specific functional groups are overrepresented among these genes. In addition, they provide evidence for an effect of disease-associated UTR mutations on endogenous RNA stability. While these elements link back to the original motivation of the study (screening for effects of point mutations in 5' and 3' UTRs), they provide only a limited amount of additional insights.</p><p>In summary, this manuscript presents an interesting addition to the long-standing attempts at dissecting the sequence basis of RNA stability in human cells. The analysis is in general comprehensive and sound; however, it remains unclear to what extent the findings can be generalized beyond the method and the experimental system used here.</p><p>Comments on revisions:</p><p>Parts of my original comments have been adequately addressed by the reviewers.</p><p>After reading the revised manuscript and the rebuttal, my main concern is related to the figure comparing the half-lives as measured in the two different cell lines that was included in the response to reviewer 2, but not in the revised manuscript. The complete lack of correlation between the half-lives of the 3'UTR library measured in the two cell lines is concerning. While variability and cell type-specific effects can be expected, some principles should be the same (such as the effect of UA dinucleotides that the authors report), leading to at least some correlation.</p><p>In addition, it is unclear to me why the half-lives measured for the two libraries in HEK cells are shifted (median ln(t 1/2)=6-7 for the 5'UTR library and ln(t 1/2)=4-4.5 for the 3'UTR library), but not in SH.</p><p>I feel that this figure contains important information that should be included in the final manuscript.</p></body></sub-article><sub-article article-type="referee-report" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.97682.3.sa2</article-id><title-group><article-title>Reviewer #2 (Public review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>Summary of goals:</p><p>Untranslated regions are key cis-regulatory elements that control mRNA stability, translation, and translocation. Through interactions with small RNAs and RNA binding proteins, UTRs form complex transcriptional circuitry that allows cells to fine-tune gene expression. Functional annotation of UTR variants has been very limited, and improvements could offer insights into disease relevant regulatory mechanisms. The goals were to advance our understanding of the determinants of UTR regulatory elements and characterize the effects of a set of &quot;disease-relevant&quot; UTR variants.</p><p>Strengths:</p><p>The use of a massively parallel reporter assay allowed for analysis of a substantial set (6,555 pairs) of 5' and 3' UTR fragments compiled from known disease associated variants. Two cell types were used.</p><p>The findings confirm previous work about the importance of AREs, which helps show validity and adds some detailed comparisons of specific AU-rich motif effects in these two cell types.</p><p>Using a Lasso regression, TA-dinucleotide content is identified as a strong regulator of RNA stability in a context dependent manner based on GC content and presence of RNA binding protein binding motifs. The findings have potential importance, drawing attention to a UTR feature that is not well characterized.</p><p>The use of complementary datasets, including from half-life analyses of RNAs and from random sequence library MRPA's, is a useful addition and supports several important findings. The finding the TA dinucleotides have explanatory power separate from (and in some cases interacting with) GC content is valuable.</p><p>The functional enrichment analysis suggests some new ideas about how UTRs may contribute to regulation of certain classes of genes.</p><p>Weaknesses:</p><p>In this section, original reviewer comments about the initial submission and the responses of the authors are listed together with new reviewer responses to the authors:</p><p>Reviewer original comment 1: It is difficult to understand how the calculations for half-life were performed. The sequencing approach measures the relative frequency of each sequence at each time point (less stable sequences become relatively less frequent after time 0, whereas more stable sequences become relatively more frequent after time 0). Since there is no discussion of whether the abundance of the transfected RNA population is referenced to some external standard (e.g., housekeeping RNAs), it is not clear how absolute (rather than relative) half-lives were determined.</p><p>Author response: [The authors showed the equations used to calculate half lives based on read counts.] They stated that &quot;The absolute abundance was not required for the half-life calculation.&quot;</p><p>Reviewer response to authors: The methods section states that DESeq2 was used to normalize read counts. DESeq2 normalization assumes that levels of most RNAs are not different between samples. That assumption is not valid here, since RNAs in the library are introduced into cells at time 0 and all RNAs decrease over time. If DESeq2 is applied without modification to normalize across timepoints, normalized reads from less stable RNAs will decrease over time (as expected) but normalized reads from more stable RNAs will increase. Can the authors please clarify in the methods how the read counts were normalized to account for this issue?</p><p>Reviewer original comment 2: Fig. S1A and B are used to assess reproducibility. They show that read counts at a given time point correlate well across replicate experiments. However, this is not a good way to assess reproducibility or accuracy of the measurements of t1/2 are. (The major source of variability in read counts in these plots - especially at early time points - is likely starting abundance of each RNA sequence, not stability.) This creates concerns about how well the method is measuring t1/2. Also creating concern is the observation that many RNAs are associated with half-lives that are much longer than the time points analyzed in the study. For example, based upon Figure S1 and Table S1 correctly, the median t1/2 for the 5' UTR library in HEK cells appears to be &gt;700 minutes. Given that RNA was collected at 30, 75, and 120 minutes, accurate measurements of RNAs with such long half lives would seem to be very difficult.</p><p>Author response: ... The calculation of the half-life involves first determining the decay constant λ, which represents a constant rate of decay. Since λ is a constant, it is possible to accurately calculate it without needing data over the entire decay range. Our experimental design considers this by selecting appropriate time points to ensure a reliable estimation of λ, and thus, the half-life. To determine the most suitable time points, we conducted preliminary experiments using RT-PCR. These experiments indicated that 30, 75, and 120 minutes provided an effective range for capturing the decay dynamics of the transcripts.</p><p>Reviewer response to author comments: Based on Fig. S1D, for 3' UTRs in both cell types and for 5' UTRs in SH-SY5Y cells, median t1/2 is in the range of ~30 to 90 minutes (corresponding to ln t1/2 = 3.5 to 4.5). Measuring RNAs at 30, 75, and 120 minutes would therefore be a good choice for these cases, However, median t1/2 in HEK cells appears to be ~600 minutes (corresponding to ln t1/2 ~6.4) for HEK cells. For t1/2 of 600 minutes, RNA levels at the final time point (120 minutes) would be 90% of the those at the first time point (30 minutes), which illustrates why the method would need to be able to reliably capture very small changes in RNA abundance to accurately measure t1/2 for transcripts with half-lives much longer than 120 minutes. As suggested in our original review, this concern could be addressed by showing the correlation of half-lives across replicates for the 5' and 3' UTR libraries in both cell types. Alternatively, the authors could show other measures of reproducibility for the half-life measurements across replicates. This requires no additional experimentation and can be done using the data from replicate runs shown in Fig. S1A and B. We remain concerned that for sequences with very long half-lives, extrapolating the half-life from small changes between 30 and 120 minutes will lead to imprecise measurements.</p><p>Reviewer original comment 3: There is no direct comparison of t1/2 between the two cell types studied for the full set of sequences studied. This would be helpful in understanding whether the regulatory effects of UTRs are generally similar across cell lines (as has been shown in some previous studies) or whether there are fundamental differences. The distribution of t1/2's is clearly quite different in the two cell lines, but it is important to know if this reflects generally slow RNA turnover in HEK cells or whether there are a large number of sequence-specific effects on stability between cell lines. A related issue is that it is not clear whether the relatively small number of significant variant effects detected in HEK cells versus SH-SY5Y cells is attributable to real biological differences between cell types or to technical issues (many fewer read counts and much longer half lives in HEK cells).</p><p>Author response: For both cell lines, we selected oligonucleotides with R2 &gt; 0.5 and mean squared error (MSE) &lt; 1 for analysis when estimating half-life (λ) by linear regression. This selection criterion was implemented to minimize the effect of experimental noise. After quality control, we selected common UTRs and compared the RNA half-lives of the two cell lines using a scatter plot. The figure below shows that RNA half-lives are quite different between the cell lines, with a moderate similarity observed in the 5' UTRs (R = 0.21), while the correlation in the 3' UTRs is non-significant. Despite the low correlation of mRNA half-life between the two cell lines, UA-dinucleotide and UA-rich sequences consistently emerge as the most significant destabilizing features, suggesting a shared regulatory mechanism across diverse cellular environments.</p><p>Reviewer response to author comments: We appreciate that the authors shared this additional analysis of the data. We believe that this is an important finding and that the additional figure showing correlations of half-lives across cell types should be included in the manuscript or supplement. Discussion of this result in the manuscript would also be useful for readers. This result is surprising to us since we would have expected that widely expressed RNA-binding proteins would have led to more similar effects between the two cell types, as previously found using other approaches (e.g., studies of 3' UTR effects in MPRAs). It would also be appropriate to discuss that differences seen between the two cell types indicate that caution is warranted when trying to generalize the results of this study to other cell types.</p><p>Reviewer original comment 4 has been addressed adequately in the revised manuscript.</p><p>Appraisal and impact:</p><p>Reviewer original comment 1: The work adds to existing studies that previously identified sequence features, including AREs and other RNA binding protein motifs, that regulate stability and puts a new emphasis on the role of &quot;TA&quot; (better &quot;UA&quot;) dinucleotides. It is not clear how potential problems with the RNA stability measurements discussed above might influence the overall conclusions, which may limit the impact unless these can be addressed.</p><p>It is difficult to understand whether the importance of TA dinucleotides is best explained by their occurrence in a related set of longer RBP binding motifs (see Fig 5J, these motifs may be encompassed by the &quot;WWWWWW cluster&quot;) or whether some other explanation applies. Further discussion of this would be helpful. Does the LASSO method tend to collapse a more diverse set of longer motifs that are each relatively rare compared to the dinucleotide? It remains unclear whether TA dinucleotides are associated with less stability independent of the presence of the known larger WWWWWWW motif. As noted above, the importance of TA dinucleotides in the HEK experiments appears to be less than is implied in the text.</p><p>Author response: To ensure the representativeness of the features entered into the LASSO model, we pre-selected those with an occurrence greater than 10% among all UTRs. There is no evidence to support a preference for dinucleotides by LASSO. To address whether the destabilizing effect of UA dinucleotides is part of the broader WWWWWW motif, we divided UA dinucleotides into two groups: those within the WWWWWW motif and those outside of it. Specifically, we divided UTRs into two categories: 'at least one UA within a WWWWWW motif' and 'no UA within a WWWWWW motif,' and visualized the results using a boxplot. As shown in [figures provided to the reviewers], the destabilizing trend still remains for UA dinucleotides outside of the WWWWWW motif, although the effect appears to be more pronounced when UA is within the WWWWWW motif. This suggests that while UA dinucleotides have a destabilizing effect independently, their impact is amplified when they are part of the broader WWWWWW motif.</p><p>Reviewer response to authors: These are useful additional analyses, and we suggest that the additional figure and discussion should be included in the manuscript/supplement so that readers can benefit from them.</p><p>Reviewer original comment 2: The inclusion of more than a single cell type is an acknowledgement of the importance of evaluating cell type-specific effects. The work suggests a number of cell type-specific differences, but due to technical issues (especially with the HEK data, as outlined above) and the use of only two cell lines, it is difficult to understand cell type effects from the work.</p><p>The inclusion of both 3' and 5' UTR sequences distinguishes this work from most prior studies in the field. Contrasting the effects of these regions on stability is of interest, although the role of these UTRs (especially the 5' UTR) in translational regulation is not assessed here.</p><p>Author response: We examined the role of UTR and UTR variants in translation regulation using polysome profiling. By both univariate analysis and an elastic regression model, we identified motifs of short repeated sequences, including SRSF2 binding sites, as mutation hotspots that lead to aberrant translation. Furthermore, these polysome-shifting mutations had a considerable impact on RNA secondary structures, particularly in upstream AUG-containing 5' UTRs. Integrating these features, our model achieved high accuracy (AUROC &gt; 0.8) in predicting polysome-shifting mutations in the test dataset. Additionally, metagene analysis indicated that pathogenic variants were enriched at the upstream open reading frame (uORF) translation start site, suggesting changes in uORF usage underlie the translation deficiencies caused by these mutations. Illustrating this, we demonstrated that a pathogenic mutation in the IRF6 5' UTR suppresses translation of the primary open reading frame by creating a uORF. Remarkably, site-directed ADAR editing of the mutant mRNA rescued this translation deficiency. Because the regulation of translation and stability does not converge, we illustrate these two mechanisms in two separate manuscripts (this one and doi.org/10.1101/2024.04.11.589132).</p><p>Reviewer response to authors: This is useful context. No further comment.</p></body></sub-article><sub-article article-type="referee-report" id="sa3"><front-stub><article-id pub-id-type="doi">10.7554/eLife.97682.3.sa3</article-id><title-group><article-title>Reviewer #3 (Public review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>Summary:</p><p>In their manuscript titled &quot;Multiplexed Assays of Human Disease‐relevant Mutations Reveal UTR Dinucleotide Composition as a Major Determinant of RNA Stability&quot; the authors aim to investigate the effect of sequence variations in 3'UTR and 5'UTRs on the stability of mRNAs in two different human cell lines.</p><p>To do so, the authors use a massively parallel reporter assay (MPRA). They transfect cells with a set of mRNA reporters that contain sequence variants in their 3' or 5' UTRs, which were previously reported in human diseases. They follow their clearance from cells over time relative to the matching non-variant sequence. To analyze their results, they define a set of factors (RBP and miRNA binding sites, sequence features, secondary structure etc.) and test their association with differences in mRNA stability. For features with a significant association, they use clustering to select a subset of factors for LASSO regression and identify factors that affect mRNA stability.</p><p>They conclude that the TA dinucleotide content of UTRs is the strongest destabilizing sequence feature. Within that context, elevated GC content and protein binding can protect susceptible mRNAs from degradation. They also show that TA dinucleotide content of UTRs affects native mRNA stability and that it is associated with specific functional groups. Finally, they link disease associated sequence variants with differences in mRNA stability of reporters.</p><p>Strengths:</p><p>(1) This work introduces a different MPRA approach to analyze the effect of genetic variants. While previous works in tissue culture use DNA transfections that require normalization for transcription efficiency, here the mRNA is directly introduced into cells at fixed amounts, allowing a more direct view of the mRNA regulation.</p><p>(2) The authors also introduce a unique analysis approach, which takes into account multiple factors that might affect mRNA stability. This approach allows them to identify general sequence features that affect mRNA stability beyond specific genetic variants, and reach important insights on mRNA stability regulation. Indeed, while the conclusions to genetic variants identified in this work are interesting, the main strength of the work involves general effect of sequence features rather than specific variants.</p><p>(3) The authors provide adequate support for their claims and validate their analysis using both their reporter data and native genes. For the main feature identified, TA di-nucleotides, they perform follow-up experiments with modified reporters that further strengthen their claims, and also validate the effect on native cellular transcripts (beyond reporters), demonstrating its validity also within native scenarios.</p><p>(4) The work provides a broad analysis of mRNA stability, across two mRNA regulatory segments (3'UTR and 5'UTR) and is performed in two separate cell-types. Comparison between two different cell-types is adequate, and the results demonstrate, as expected, the dependence of mRNA stability on the cellular context. Analysis of 3'UTR and 5'UTR regulatory effects also shows interesting differences and similarities between these two regulatory regions.</p><p>Weaknesses:</p><p>In their revised manuscripts, the authors successfully address many of the weaknesses raised in the original review, including the effect of possible confounding effects, and additional methodology details. Notably, two of the issues raised in the original report, have only been partially addressed in the revision.</p><p>(1) The analysis and regression models built in this work are not thoroughly investigated relative to native genes within cells.</p><p>While using MPRAs indeed allows to isolate regulatory effects that are less influential in-vivo, the resulting effects still provide some regulatory function in-vivo. The goal of such an analysis would not be to demonstrate the predictive power of the models, or to make any claims regarding using these models to fully explain or predict the stability of native transcripts. Clearly, additional more prominent factors could function in controlling endogenous RNA stability.</p><p>Instead, the goal of such an investigation is to simply assess the fraction of in-vivo regulation that the factors identified in this work contribute in native contexts, and what is the relative contribution of the phenomena captured by the well-controlled MPRA study.</p><p>This reviewer believes that even if the effects identified by the current MPRA study only contribute a small fraction of in-vivo variation, an analysis that aim to estimate what this fraction is, will be very relevant to this study for several reasons. First, in order to appreciate the results of this study within their in-vivo context. Second, in light of the questions raised as motivation for this study, and particularly the need to identify the effect of disease-associated 3'UTR variants, which clearly have an in-vivo effect.</p><p>(2) Methodology validation can be performed with simulated data (generated in-silico by the authors) to provide an independent support for the ability of the current methodology to correctly extract regulatory effects from the data.</p></body></sub-article><sub-article article-type="author-comment" id="sa4"><front-stub><article-id pub-id-type="doi">10.7554/eLife.97682.3.sa4</article-id><title-group><article-title>Author response</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Su</surname><given-names>Jia-Ying</given-names></name><role specific-use="author">Author</role><aff><institution>Institute of Statistical Science, Academia Sinica</institution><addr-line><named-content content-type="city">Taipei</named-content></addr-line><country>Taiwan</country></aff></contrib><contrib contrib-type="author"><name><surname>Wang</surname><given-names>Yun-Lin</given-names></name><role specific-use="author">Author</role><aff><institution>Institute of Molecular Biology, Academia Sinica</institution><addr-line><named-content content-type="city">Taipei</named-content></addr-line><country>Taiwan</country></aff></contrib><contrib contrib-type="author"><name><surname>Hsieh</surname><given-names>Yu-Tung</given-names></name><role specific-use="author">Author</role><aff><institution>Institute of Molecular Biology, Academia Sinica</institution><addr-line><named-content content-type="city">Taipei</named-content></addr-line><country>Taiwan</country></aff></contrib><contrib contrib-type="author"><name><surname>Chang</surname><given-names>Yu-Chi</given-names></name><role specific-use="author">Author</role><aff><institution>Institute of Molecular Biology, Academia Sinica</institution><addr-line><named-content content-type="city">Taipei</named-content></addr-line><country>Taiwan</country></aff></contrib><contrib contrib-type="author"><name><surname>Yang</surname><given-names>Cheng-Han</given-names></name><role specific-use="author">Author</role><aff><institution>Institute of Molecular Biology, Academia Sinica</institution><addr-line><named-content content-type="city">Taipei</named-content></addr-line><country>Taiwan</country></aff></contrib><contrib contrib-type="author"><name><surname>Kang</surname><given-names>YoonSoon</given-names></name><role specific-use="author">Author</role><aff><institution>Institute of Molecular Biology, Academia Sinica</institution><addr-line><named-content content-type="city">Taipei</named-content></addr-line><country>Taiwan</country></aff></contrib><contrib contrib-type="author"><name><surname>Huang</surname><given-names>Yen-Tsung</given-names></name><role specific-use="author">Author</role><aff><institution>Institute of Statistical Science, Academia Sinica</institution><addr-line><named-content content-type="city">Taipei</named-content></addr-line><country>Taiwan</country></aff></contrib><contrib contrib-type="author"><name><surname>Lin</surname><given-names>Chien-Ling</given-names></name><role specific-use="author">Author</role><aff><institution>Institute of Molecular Biology, Academia Sinica</institution><addr-line><named-content content-type="city">Taipei</named-content></addr-line><country>Taiwan</country></aff></contrib></contrib-group></front-stub><body><p>The following is the authors’ response to the original reviews.</p><disp-quote content-type="editor-comment"><p><bold>Public Reviews:</bold></p><p><bold>Reviewer #1 (Public Review):</bold></p><p>In the manuscript by Su et al., the authors present a massively parallel reporter assay (MPRA) measuring the stability of in vitro transcribed mRNAs carrying wild-type or mutant 5' or 3' UTRs transfected into two different human cell lines. The goal presented at the beginning of the manuscript was to screen for effects of disease-associated point mutations on the stability of the reporter RNAs carrying partial human 5' or 3' UTRs. However, the majority of the manuscript is dedicated to identifying sequence components underlying the differential stability of reporter constructs. This shows that TA dinucleotides are the most predictive feature of RNA stability in both cell lines and both UTRs.</p><p>The effect of AU rich elements (AREs) on RNA stability is well established in multiple systems, and the present study confirms this general trend but points out variability in the consequence of seemingly similar motifs on RNA stability. For example, the authors report that a long stretch of Us has extreme opposite effects on RNA stability depending on whether it is preceded by an A (strongly destabilizing) or followed by an A (strongly stabilizing). While the authors interpretation of a context- dependence of the effect is certainly well-founded, it seems counterintuitive that the preceding or following A would be the (only) determining factor. This points to a generally reductionist approach taken by the authors in the analysis of the data and in their attempt to dissect the contribution of &quot;AU rich sequences&quot; to RNA stability, with a general tendency to reduce the size and complexity of the features (e.g. to dinucleotides). While this certainly increases the statistical power of the analysis due to the number of occurrences of these motifs, it limits the interpretability of the results. How do TA dinucleotides per se contribute to destabilizing the RNA, both in 5' and 3' UTRs, but (according to limited data presented) not in coding sequences? What is the mechanism? RBPs binding to TA dinucleotide containing sequences are suggested to &quot;mask&quot; the destabilizing effect, thereby leading to a more stable RNA. Gain of TA dinucleotides is reported to have a destabilizing effect, but again no hypothesis is provided as to the underlying molecular mechanism. In addition to reducing the motif length to dinucleotides, the notion of &quot;context dependence&quot; is used in a very narrow sense; especially when focusing on simple and short motifs, a more extensive analysis of the interdependence of these features (beyond the existing analysis of the relationship between TA- diNTs and GC content) could potentially reveal more of the context dependence underlying the seemingly opposite behavior of very similar motifs.</p></disp-quote><p>(We have used UA instead of TA, as per the reviewer's suggestion)</p><p>The contribution of coding region sequence to RNA stability has been extensively discussed (For example: doi.org/10.1016/j.molcel.2022.03.032; doi.org/10.1186/s13059-020-02251-5; doi.org/10.15252/embr.201948220; doi.org/10.1371/journal.pone.0228730; doi.org/10.7554/eLife.45396). While UA content at the third codon position (wobble position) has been implicated as a pro-degradation signal, codon optimality has emerged as the most prominent determinant for RNA stability. This indicates that the role of coding regions in RNA stability differs from that of UTRs due to the involvement of translation elongation. We did not intend to suggest that UA-dinucleotides in UTRs and coding regions have the same effect.</p><p>To ensure the representativeness of the features entered into the LASSO model, we pre-selected those with an occurrence greater than 10% among all UTRs. As a result, while motifs with very low occurrences were excluded from the analysis, there is no evidence to indicate a preference for dinucleotides by the LASSO model.</p><p>We hypothesize that UA-dinucleotide may recruit endonucleases RNase A family, whose catalytic pockets exhibit a strong bias for UA dinucleotide (doi.org/10.1016/j.febslet.2010.04.018). Structures or protein bindings that block this recognition might stabilize RNAs. To gain further insight into the motif interactions, we investigated the interactions between UA and other 15 dinucleotides through more detailed analyses. We conducted a linear regression analysis investigating interactions between UA and the other 15 dinucleotides. The formula used below includes UA:</p><p><inline-formula><mml:math id="sa4m1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>ln</mml:mi><mml:mo>⁡</mml:mo><mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mi>D</mml:mi><mml:mi>i</mml:mi><mml:mi>N</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>D</mml:mi><mml:mi>i</mml:mi><mml:mi>N</mml:mi><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mi>C</mml:mi><mml:mi mathvariant="normal">%</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mi>C</mml:mi><mml:msub><mml:mi mathvariant="normal">%</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> <inline-formula><mml:math id="sa4m2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mo>+</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>X</mml:mi><mml:mi>G</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mi>C</mml:mi><mml:msub><mml:mi mathvariant="normal">%</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> <inline-formula><mml:math id="sa4m3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mo>+</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:mi>A</mml:mi><mml:mi>X</mml:mi><mml:mi>D</mml:mi><mml:mi>i</mml:mi><mml:mi>N</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:msub><mml:mi>A</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:msub><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>D</mml:mi><mml:mi>i</mml:mi><mml:mi>N</mml:mi><mml:mi>T</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> <inline-formula><mml:math id="sa4m4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mo>+</mml:mo><mml:msub><mml:mi>ϵ</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>, where all <italic>β</italic> terms represent the regression coefficients, and <inline-formula><mml:math id="sa4m5"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>U</mml:mi><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>, <inline-formula><mml:math id="sa4m6"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>D</mml:mi><mml:mi>i</mml:mi><mml:mi>N</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>, and <sup>th</sup> UTR, respectively, and 𝜖<sub>i</sub> denotes the error term. For each dinucleotide, we tested the significance of <italic>β</italic> <inline-formula><mml:math id="sa4m7"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:mi mathvariant="normal">%</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> represent the number of UA dinucleotides, the number of other dinucleotides (other than UA), and the GC content of the i<italic>UAxGC%</italic> and <italic>βUAxDiNT</italic>, and compared their p-values using a quantile-quantile (QQ) plot. Author response image 1 shows that the interaction effect of UA dinucleotides with GC% is much more significant than interactions with the other 15 dinucleotides, as indicated by the inflated QQ plot of p-values. This suggests that GC content is a more critical contextual factor influencing UA dinucleotides' impact on RNA stability.</p><fig id="sa4fig1" position="float"><label>Author response image 1.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-sa4-fig1-v1.tif"/></fig><disp-quote content-type="editor-comment"><p>The present MPRAs measures the effect of UTR sequences in one specific reporter context and using one experimental approach (following the decay of in vitro transcribed and transfected RNAs). While this approach certainly has its merits compared to other approaches, it also comes with some caveats: RNA is delivered naked, without bound RBPs and no nuclear history, e.g. of splicing (no EJCs), editing and modifications. One way to assess the generalizability of the results as well as the context dependence of the effects is to perform the same analysis on existing datasets of RNA stability measurements obtained through other methods (e.g. transcription inhibition). Are TA dinucleotides universally the most predictive feature of RNA half-lives?</p></disp-quote><p>Our system studies the stability control of RNA synthesized in vitro and delivered into human cells. While we did not intend to generalize our conclusions to endogenous RNAs, our approach contributes to the understanding of in vitro synthesized RNA used for cellular expression, such as in vaccines. It is known that endogenous RNAs undergo very different regulation. The most prominent factors controlling endogenous RNA stability are the density of splice junctions and the length of UTRs (doi.org/10.1186/s13059-022-02811-x; doi.org/10.1186/s12915-021-00949-x). To decipher the sequence regulation, these factors are controlled in our experiments. Therefore, we do not expect the dinucleotide features found by our approach to be generalized as the most predictive feature of RNA half-life in vivo.</p><disp-quote content-type="editor-comment"><p>The authors conclude their study with a meta-analysis of genes with increased TA dinucleotides in 5' and 3'UTRs, showing that specific functional groups are overrepresented among these genes. In addition, they provide evidence for an effect of disease-associated UTR mutations on endogenous RNA stability. While these elements link back to the original motivation of the study (screening for effects of point mutations in 5' and 3' UTRs), they provide only a limited amount of additional insights.</p></disp-quote><p>We utilized the Taiwan Biobank to investigate whether mutations significantly affecting RNA stability also impact human biochemical measurements. Our findings indicate that these mutations indeed have a significant effect on various biochemical indices. This highlights the importance of our study, as it bridges basic science with potential applications in precision medicine. By linking specific UTR mutations with measurable changes in biochemical indices, our research underscores the potential for these findings to inform targeted medical interventions in the future.</p><disp-quote content-type="editor-comment"><p>In summary, this manuscript presents an interesting addition to the long-standing attempts at dissecting the sequence basis of RNA stability in human cells. The analysis is in general very comprehensive and sound; however, at times the goal of the authors to find novelty and specificity in the data overshadows some analyses. One example is the case where the authors try to show that TA-dinucleotides and GC content are decoupled and not merely two sides of the same coin.</p><p>They claim that the effect of TA dinucleotides is different between high- and low-GC content contexts but do not control for the fact that low GC-content regions naturally will contain more TA dinucleotides and therefore the effect sizes and the resulting correlation between TA-diNT rate and stability will be stronger (Fig. 5A). A more thorough analysis and greater caution in some of the claims could further improve the credibility of the conclusions.</p></disp-quote><p>Low GC content implies a higher UA content but does not directly equate to a high UA-dinucleotide ratio. For instance, the sequence AUUGAACCUU has a lower GC content (0.3) compared to UAUAGGCCGC (0.6), yet it also has a lower UA-dinucleotide ratio (0 vs. 0.22). To address this concern more rigorously, we performed a stratified analysis based on UA-diNT rate. As shown in our Fig. S7C, even after stratifying by UA- dinucleotide ratio (upper panel high UA- dinucleotide ratio / lower panel low UA- dinucleotide ratio), we still observe that the destabilizing effect of UA is stronger in the low GC content group.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #2 (Public Review):</bold></p><p>Summary of goals:</p><p>Untranslated regions are key cis-regulatory elements that control mRNA stability, translation, and translocation. Through interactions with small RNAs and RNA binding proteins, UTRs form complex transcriptional circuitry that allows cells to fine-tune gene expression. Functional annotation of UTR variants has been very limited, and improvements could offer insights into disease relevant regulatory mechanisms. The goals were to advance our understanding of the determinants of UTR regulatory elements and characterize the effects of a set of &quot;disease-relevant&quot; UTR variants.</p><p>Strengths:</p><p>The use of a massively parallel reporter assay allowed for analysis of a substantial set (6,555 pairs) of 5' and 3' UTR fragments compiled from known disease associated variants. Two cell types were used.</p><p>The findings confirm previous work about the importance of AREs, which helps show validity and adds some detailed comparisons of specific AU-rich motif effects in these two cell types.</p><p>Using a Lasso regression, TA-dinucleotide content is identified as a strong regulator of RNA stability in a context dependent manner based on GC content and presence of RNA binding protein binding motifs. The findings have potential importance, drawing attention to a UTR feature that is not well characterized.</p><p>The use of complementary datasets, including from half-life analyses of RNAs and from random sequence library MRPA's, is a useful addition and supports several important findings. The finding the TA dinucleotides have explanatory power separate from (and in some cases interacting with) GC content is valuable.</p><p>The functional enrichment analysis suggests some new ideas about how UTRs may contribute to regulation of certain classes of genes.</p><p>Weaknesses:</p><p>It is difficult to understand how the calculations for half-life were performed. The sequencing approach measures the relative frequency of each sequence at each time point (less stable sequences become relatively less frequent after time 0, whereas more stable sequences become relatively more frequent after time 0). Since there is no discussion of whether the abundance of the transfected RNA population is referenced to some external standard (e.g., housekeeping RNAs), it is not clear how absolute (rather than relative) half-lives were determined.</p></disp-quote><p>We estimated decay constant λ and half-life (<italic>t</italic><sub>1/2</sub>) by the following equations:<disp-formula id="sa4equ1"><mml:math id="sa4m8"><mml:mrow><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>ln</mml:mi><mml:mo>⁡</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub></mml:mfrac><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mo>−</mml:mo><mml:mi>λ</mml:mi><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mrow><mml:mi>ln</mml:mi><mml:mo>⁡</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>2</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>λ</mml:mi></mml:mfrac></mml:mstyle></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula></p><p>where <italic>C</italic><sub>i(t)</sub> and <italic>C</italic><sub>i(t=0)</sub> are read count values of the ith replicate at time points <italic>t</italic> and 0 (see also Methods). The absolute abundance was not required for the half-life calculation.</p><disp-quote content-type="editor-comment"><p>Fig. S1A and B are used to assess reproducibility. They show that read counts at a given time point correlate well across replicate experiments. However, this is not a good way to assess reproducibility or accuracy of the measurements of t1/2 are. (The major source of variability in read counts in these plots - especially at early time points - is likely the starting abundance of each RNA sequence, not stability.) This creates concerns about how well the method is measuring t1/2. Also creating concern is the observation that many RNAs are associated with half-lives that are much longer than the time points analyzed in the study. For example, based upon Figure S1 and Table S1 correctly, the median t1/2 for the 5' UTR library in HEK cells appears to be &gt;700 minutes. Given that RNA was collected at 30, 75, and 120 minutes, accurate measurements of RNAs with such long half lives would seem to be very difficult.</p></disp-quote><p>We estimated the half-life based on the following equations:<disp-formula id="sa4equ2"><mml:math id="sa4m9"><mml:mrow><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>ln</mml:mi><mml:mo>⁡</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub></mml:mfrac><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mo>−</mml:mo><mml:mi>λ</mml:mi><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mstyle></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>ln</mml:mi><mml:mo>⁡</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>2</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>λ</mml:mi></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula></p><p>where <italic>C</italic><sub>i(t)</sub> and <italic>C</italic><sub>i(t=0)</sub> are read count values of the ith replicate at time points <italic>t</italic> and 0 (see also Methods). The calculation of the half-life involves first determining the decay constant λ, which represents a constant rate of decay. Since λ is a constant, it is possible to accurately calculate it without needing data over the entire decay range. Our experimental design considers this by selecting appropriate time points to ensure a reliable estimation of λ, and thus, the half-life. To determine the most suitable time points, we conducted preliminary experiments using RT-PCR.</p><p>These experiments indicated that 30, 75, and 120 minutes provided an effective range for capturing the decay dynamics of the transcripts.</p><disp-quote content-type="editor-comment"><p>There is no direct comparison of t1/2 between the two cell types studied for the full set of sequences studied. This would be helpful in understanding whether the regulatory effects of UTRs are generally similar across cell lines (as has been shown in some previous studies) or whether there are fundamental differences. The distribution of t1/2's is clearly quite different in the two cell lines, but it is important to know if this reflects generally slow RNA turnover in HEK cells or whether there are a large number of sequence-specific effects on stability between cell lines. A related issue is that it is not clear whether the relatively small number of significant variant effects detected in HEK cells versus SH-SY5Y cells is attributable to real biological differences between cell types or to technical issues (many fewer read counts and much longer half lives in HEK cells).</p></disp-quote><p>For both cell lines, we selected oligonucleotides with R<sup>2</sup> &gt; 0.5 and mean squared error (MSE) &lt; 1 for analysis when estimating half-life (λ) by linear regression. This selection criterion was implemented to minimize the effect of experimental noise. After quality control, we selected common UTRs and compared the RNA half-lives of the two cell lines using a scatter plot. Author response image 2 shows that RNA half-lives are quite different between the cell lines, with a moderate similarity observed in the 5' UTRs (R = 0.21), while the correlation in the 3' UTRs is non-significant.</p><fig id="sa4fig2" position="float"><label>Author response image 2.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-sa4-fig2-v1.tif"/></fig><p>Despite the low correlation of mRNA half-life between the two cell lines, UA-dinucleotide and UA-rich sequences consistently emerge as the most significant destabilizing features, suggesting a shared regulatory mechanism across diverse cellular environments.</p><disp-quote content-type="editor-comment"><p>The general assertion is made in many places that TA dinucleotides are the most prominent destabilizing element in UTRs (e.g., in the title, the abstract, Fig. 4 legend, and on p. 12). This appears to be true for only one of the two cell lines tested based on Fig. 3.</p></disp-quote><p>UA-dinucleotides and other UA-rich sequences exhibit similar effects on RNA stability, as illustrated in Fig. S5A-C. In two cell lines, UA-dinucleotide and WWWWWW sequences were representatives of the same stability-affecting cluster. While the impact of UA-dinucleotides can be generalized, we have rephrased some statements for clarification to avoid any potential misunderstanding. For examples:</p><p>Abstract: “...We found that UA dinucleotides and UA-rich motifs are the most prominent destabilizing element.“</p><p>p.10: “UA dinucleotides and UA-rich motifs are the most common and effective RNA destabilizing factor”</p><p>Figure 4: “The UTR UA dinucleotides and UA-rich motifs are the most common and influential RNA destabilizing factor.”</p><disp-quote content-type="editor-comment"><p>Appraisal and impact:</p><p>The work adds to existing studies that previously identified sequence features, including AREs and other RNA binding protein motifs, that regulate stability and puts a new emphasis on the role of &quot;TA&quot; (better &quot;UA&quot;) dinucleotides. It is not clear how potential problems with the RNA stability measurements discussed above might influence the overall conclusions, which may limit the impact unless these can be addressed.</p><p>It is difficult to understand whether the importance of TA dinucleotides is best explained by their occurrence in a related set of longer RBP binding motifs (see Fig 5J, these motifs may be encompassed by the &quot;WWWWWW cluster&quot;) or whether some other explanation applies. Further discussion of this would be helpful. Does the LASSO method tend to collapse a more diverse set of longer motifs that are each relatively rare compared to the dinucleotide? It remains unclear whether TA dinucleotides are associated with less stability independent of the presence of the known larger WWWWWWW motif. As noted above, the importance of TA dinucleotides in the HEK experiments appears to be less than is implied in the text.</p></disp-quote><p>To ensure the representativeness of the features entered into the LASSO model, we pre-selected those with an occurrence greater than 10% among all UTRs. There is no evidence to support a preference for dinucleotides by LASSO. To address whether the destabilizing effect of UA dinucleotides is part of the broader WWWWWW motif, we divided UA dinucleotides into two groups: those within the WWWWWW motif and those outside of it. Specifically, we divided UTRs into two categories: 'at least one UA within a WWWWWW motif' and 'no UA within a WWWWWW motif,' and visualized the results using a boxplot. As shown in Author response image 3, the destabilizing trend still remains for UA dinucleotides outside of the WWWWWW motif, although the effect appears to be more pronounced when UA is within the WWWWWW motif. This suggests that while UA dinucleotides have a destabilizing effect independently, their impact is amplified when they are part of the broader WWWWWW motif.</p><fig id="sa4fig3" position="float"><label>Author response image 3.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-sa4-fig3-v1.tif"/></fig><disp-quote content-type="editor-comment"><p>The inclusion of more than a single cell type is an acknowledgement of the importance of evaluating cell type-specific effects. The work suggests a number of cell type-specific differences, but due to technical issues (especially with the HEK data, as outlined above) and the use of only two cell lines, it is difficult to understand cell type effects from the work.</p><p>The inclusion of both 3' and 5' UTR sequences distinguishes this work from most prior studies in the field. Contrasting the effects of these regions on stability is of interest, although the role of these UTRs (especially the 5' UTR) in translational regulation is not assessed here.</p></disp-quote><p>We examined the role of UTR and UTR variants in translation regulation using polysome profiling. By both univariate analysis and an elastic regression model, we identified motifs of short repeated sequences, including SRSF2 binding sites, as mutation hotspots that lead to aberrant translation. Furthermore, these polysome-shifting mutations had a considerable impact on RNA secondary structures, particularly in upstream AUG-containing 5’ UTRs. Integrating these features, our model achieved high accuracy (AUROC &gt; 0.8) in predicting polysome-shifting mutations in the test dataset. Additionally, metagene analysis indicated that pathogenic variants were enriched at the upstream open reading frame (uORF) translation start site, suggesting changes in uORF usage underlie the translation deficiencies caused by these mutations. Illustrating this, we demonstrated that a pathogenic mutation in the IRF6 5’ UTR suppresses translation of the primary open reading frame by creating a uORF. Remarkably, site-directed ADAR editing of the mutant mRNA rescued this translation deficiency. Because the regulation of translation and stability does not converge, we illustrate these two mechanisms in two separate manuscripts (this one and doi.org/10.1101/2024.04.11.589132).</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #3 (Public Review):</bold></p><p>Summary:</p><p>In their manuscript titled &quot;Multiplexed Assays of Human Disease‐relevant Mutations Reveal UTR</p><p>Dinucleotide Composition as a Major Determinant of RNA Stability&quot; the authors aim to investigate the effect of sequence variations in 3'UTR and 5'UTRs on the stability of mRNAs in two different human cell lines.</p><p>To do so, the authors use a massively parallel reporter assay (MPRA). They transfect cells with a set of mRNA reporters that contain sequence variants in their 3' or 5' UTRs, which were previously reported in human diseases. They follow their clearance from cells over time relative to the matching non-variant sequence. To analyze their results, they define a set of factors (RBP and miRNA binding sites, sequence features, secondary structure etc.) and test their association with differences in mRNA stability. For features with a significant association, they use clustering to select a subset of factors for LASSO regression and identify factors that affect mRNA stability.</p><p>They conclude that the TA dinucleotide content of UTRs is the strongest destabilizing sequence feature. Within that context, elevated GC content and protein binding can protect susceptible mRNAs from degradation. They also show that TA dinucleotide content of UTRs affects native mRNA stability, and that it is associated with specific functional groups. Finally, they link disease associated sequence variants with differences in mRNA stability of reporters.</p><p>Strengths:</p><p>This work introduces a different MPRA approach to analyze the effect of genetic variants. While previous works in tissue culture use DNA transfections that require normalization for transcription efficiency, here the mRNA is directly introduced into cells at fixed amounts, allowing a more direct view of the mRNA regulation.</p><p>The authors also introduce a unique analysis approach, which takes into account multiple factors that might affect mRNA stability. This approach allows them to identify general sequence features that affect mRNA stability beyond specific genetic variants, and reach important insights on mRNA stability regulation. Indeed, while the conclusions to genetic variants identified in this work are interesting, the main strength of the work involve general effect of sequence features rather than specific variants.</p><p>The authors provide adequate supports for their claims, and validate their analysis using both their reporter data and native genes. For the main feature identified, TA di-nucleotides, they perform follow-up experiments with modified reporters that further strengthen their claims, and also validate the effect on native cellular transcripts (beyond reporters), demonstrating its validity also within native scenarios.</p><p>The work provides a broad analysis of mRNA stability, across two mRNA regulatory segments (3'UTR and 5'UTR) and is performed in two separate cell-types. Comparison between two different cell-types is adequate, and the results demonstrate, as expected, the dependence of mRNA stability on the cellular context. Analysis of 3'UTR and 5'UTR regulatory effects also shows interesting differences and similarities between these two regulatory regions.</p><p>Weaknesses:</p><p>(1) The authors fail to acknowledge several possible confounding factors of their MPRA approach in the discussion.</p><p>First, while transfection of mRNA directly into cells allows to avoid the need to normalize for differences in transcription, the introduction of naked mRNA molecules is different than native cellular mRNAs and could introduce biases due to differences in mRNA modifications, protein associations etc. that may occur co-transcriptionally.</p><p>Second, along those lines, the authors also use in-vitro polyadenylation. The length of the polyA tail of the transfected transcripts could potentially be very different than that of native mRNAs and also affect stability.</p></disp-quote><p>The transcripts used in our study were polyadenylated in vitro with approximately 100 nucleotides</p><p>(Fig. S1C), similar to the polyA tail lengths typically observed in vivo (dx.doi.org/10.1016/j.molcel.2014.02.007). Additionally, these transcripts were capped to emulate essential mRNA characteristics and to minimize immune responses in recipient cells. This design allows us to study RNA decay for in vitro-synthesized RNA delivered into human cells, akin to RNA vaccines, but it does not necessarily extend to endogenous RNAs. As mentioned, endogenous RNAs undergo nuclear processing and are decorated by numerous trans factors, resulting in distinct regulatory mechanisms. We therefore provided a more discussion on these differences and their implications in the revised manuscript: “However, while our approach effectively assesses the stability of synthesized RNA in human cells, it may not fully capture the decay dynamics of nuclear-synthesized RNA, which can be influenced by endogenous modifications and trans-acting RNA binding factors. (p. 18)”</p><disp-quote content-type="editor-comment"><p>(2) The analysis approach used in this work for identifying regulatory features in UTRs was not previously used. As such, lack of in-depth details of the methodology, and possibly also more general validation of the approach, is a drawback in convincing the reader in the validity of this approach and its results.</p><p>In particular, a main point that is not addressed is how the authors decide on the set of &quot;factors&quot; used in their analysis? As choosing different sets of factors might affect the results of the analysis.</p></disp-quote><p>In our study, we employed the calculation of the Variance Inflation Factor (VIF) as a basis for selecting variables. This well-established method is widely used to detect variables with high collinearity, thus ensuring the robustness and reliability of our analysis. By identifying and excluding highly collinear variables, we aimed to minimize multicollinearity and improve the accuracy of our regression models. For more detailed information on the use of VIF in regression analysis, please refer to Akinwande, M., Dikko, H., and Samson, A. (2015). Variance Inflation Factor: As a Condition for the Inclusion of Suppressor Variable(s) in Regression Analysis. Open Journal of Statistics, 5, 754-767. doi: 10.4236/ojs.2015.57075. We have included the method details in the revised manuscript (p. 28) :”… to avoid multicollinearity caused by similar features that perturb feature selection, all features were clustered using single-linkage hierarchical clustering with the distance metric defined as one minus the absolute value of the Spearman correlation coefficient. We cut the tree at a specific height, and the feature that had the greatest influence on RNA stability, which was examined using a simple linear regression model, was selected to be the representative of each cluster. Then we calculated the variance inflation factor (VIF) value of the representative features. The VIFs were obtained by the following linear model and equations:<disp-formula id="sa4equ3"><mml:math id="sa4m10"><mml:mrow><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mover><mml:mi>X</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mrow><mml:mover><mml:mi>α</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>0</mml:mn><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>≠</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:munder><mml:msub><mml:mrow><mml:mover><mml:mi>α</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mstyle></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msubsup><mml:mi>R</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mrow><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>X</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>X</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>⋅</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:munder><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>X</mml:mi><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>⋅</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mstyle></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>V</mml:mi><mml:mi>I</mml:mi><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msubsup><mml:mi>R</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:mfrac></mml:mstyle></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="sa4m11"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>X</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="sa4m12"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> are the estimated value of the jth feature and the value of the kth feature of the ith UTR (note that the kth feature is a feature other than the jth feature), <inline-formula><mml:math id="sa4m13"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>α</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>0</mml:mn><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> and <inline-formula><mml:math id="sa4m14"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>α</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> are the intercept and the regression coefficients of the linear model that regressed the jth feature on the other remaining features, and <inline-formula><mml:math id="sa4m15"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:munder><mml:mi>X</mml:mi><mml:mo>_</mml:mo></mml:munder><mml:mo>⋅</mml:mo><mml:munder><mml:mi>j</mml:mi><mml:mo>_</mml:mo></mml:munder></mml:mrow></mml:mstyle></mml:math></inline-formula> is the mean level of the jth feature of all UTRs.”</p><disp-quote content-type="editor-comment"><p>For example, the choice to use 7-mer sequences within the factors set is not explained, particularly when almost all motifs that are eventually identified (Figure 3B-E) are shorter.</p></disp-quote><p>The known RBP motifs are primarily 6-mer. To explore the possibility of discovering novel motifs that could significantly impact our model, we started with 7-mer sequences. However, our analysis revealed that including these additional variables did not improve the explanatory power of the model; instead, it reduced it. Consequently, our final model focuses on motifs shorter than 7-mer. We explained the motif selections in the revised manuscript (p. 9): “Given our discovery that the effect of AREs is heavily dependent on sequence content, we decided to further explore the effects of other sequence elements, i.e., beyond known regulatory motifs, in more detail. Since most reported RBP motifs are 6-mers, we initiated a search for novel motifs by analyzing the presence of all 7-mers in our massively parallel reporter assay (MPRA) library, correlating their occurrence with mRNA half-life.”</p><disp-quote content-type="editor-comment"><p>In addition, the authors do not perform validations to demonstrate the validity of their approach on simulated data or well-established control datasets. Such analysis would be helpful to further convince the reader in the usefulness and robustness of the analysis.</p></disp-quote><p>We acknowledge the importance of validating our approach on simulated data or well-established control datasets to demonstrate its robustness and reliability. However, to the best of our knowledge, there are currently no well-established control datasets available that perfectly correspond to our specific study context. Despite this, we will continue to search for any relevant datasets that could be utilized for this purpose in future work. This effort will help to further reinforce the confidence in our methodology and its findings.</p><disp-quote content-type="editor-comment"><p>(3) The analysis and regression models built in this work are not thoroughly investigated relative to native genes within cells. The effect of sequence &quot;factors&quot; on native cellular transcripts' stability is not investigated beyond TA di-nucleotides, and it is unclear to what degree do other predicted factors also affect native transcripts.</p></disp-quote><p>Our system studies the stability control of RNA synthesized in vitro and delivered into human cells. While we validated the UTR UA-dinucleotide effect in vivo, we did not intend to conclude that this is the most influential regulation for endogenous RNAs. It is known that endogenous RNAs undergo very different regulation. The most prominent factors controlling endogenous RNA stability are the density of splice junctions and the length of UTRs (doi.org/10.1186/s13059-022-02811-x; doi.org/10.1186/s12915-021-00949-x). To decipher the sequence regulation, we controlled for these factors in our experiments. Therefore, we acknowledge that several endogenous features, which were excluded by our approach, may serve as predictive features of RNA half-life in vivo.</p><disp-quote content-type="editor-comment"><p><bold>Recommendations for the authors:</bold></p><p><bold>Reviewer #1 (Recommendations For The Authors):</bold></p><p>Specific comments:</p><p>Some references are missing, e.g for the sentence:</p></disp-quote><p>Please see the response below.</p><p>&quot;Similarly, point mutation of the GFPT1 3' UTR results in congenital myasthenic syndrome.&quot; (p5)</p><p>The reference has been added to the text:</p><p>Dusl, M., Senderek, J., Muller, J. S., Vogel, J. G., Pertl, A., Stucka, R., Lochmuller, H., David, R., &amp; Abicht, A. (2015). A 3'-UTR mutation creates a microRNA target site in the GFPT1 gene of patients with congenital myasthenic syndrome. Human Molecular Genetics, 24(12), 34183426. <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/hmg/ddv090">https://doi.org/10.1093/hmg/ddv090</ext-link></p><disp-quote content-type="editor-comment"><p>&quot;...but there have been no systematic assessments of the explicit effects of variants of both UTRs on stability regulation.&quot; (not true in the current phrasing; e.g. PMIDs 32719458, 36156153, 34849835)</p></disp-quote><p>These references have been added to the text. However, we have to point out that these studies do not focus on the effects of the disease-relevant variants. To clarify, we modified the sentence to &quot;... systematic assessments of the explicit effects of disease-relevant variants in both UTRs on stability regulation are still absent.&quot;</p><disp-quote content-type="editor-comment"><p>&quot;Multiple approaches have revealed AREs as exerting a destabilizing effect on RNA stability (Barreau et al., 2005). (p8)</p></disp-quote><p>The reference has been added to the text:</p><p>Barreau, C., Paillard, L., &amp; Osborne, H. B. (2005). AU-rich elements and associated factors: are there unifying principles? Nucleic Acids Research, 33(22), 7138-7150. <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/nar/gki1012">https://doi.org/10.1093/nar/gki1012</ext-link></p><disp-quote content-type="editor-comment"><p>&quot;This effect is specific, as such ratios in the coding region are inconsequential.&quot; (p12)</p></disp-quote><p>This refers to our findings of Fig. 4G and Supplemental Fig. S5F.</p><disp-quote content-type="editor-comment"><p>What are the sequences at the 5' and 3'UTR without insertion of a library? 5'UTR library (especially in SH) has much longer half-life compared to 3'utr library (Fig S1D).</p></disp-quote><p>There is no designed 5’UTR of the 3’UTR library, only the Kozak sequence derived from the pEGFPC1 vector. This may partially underlie the shorter half-life of the 3’ UTR library.</p><disp-quote content-type="editor-comment"><p>Fig2A: What are the units? &quot;half-life (log)&quot; Do the numbers correspond to log10(min)?</p></disp-quote><p>It represents ln (min). To clarify, we now use ‘ln t<sub>1/2</sub> (min)’ in all figures.</p><disp-quote content-type="editor-comment"><p>Fig 2 and 3: This was done only on the wild-type sequences? Or all tested sequences together, wt and mut?</p></disp-quote><p>It was done only on the wild-type sequences. To clarify, we modified the text to “we examined the effect of AREs on RNA stability of the ref alleles according to specific sequence content….(p.8)” and “We considered as many factors as possible to explain the half-life of our ref UTR libraries,…. (p.9)”. ‘ref’ stands for reference.</p><disp-quote content-type="editor-comment"><p>&quot;Furthermore, to avoid collinearity confounding our model, e.g., the effects of very similar factors (such as 'AA' and 'AAA' sequences), we clustered the factors according to their properties, and then only one representative factor from within a cluster (i.e., the one with the highest correlation to halflife within a cluster) was subjected to LASSO regression&quot;: Given the observed context dependence, e.g. in the case of poly-U stretches: Isn't this clustering leading to similar/identical motifs with different context being grouped together such as polyU preceded by an A (strongly destabilizing, according to Fig 2B) or followed by one (strongly stabilizing, according to Fig 2B), resulting in ignoring the context or using one potential outcome while a motif from the same cluster can have the opposite effect?</p></disp-quote><p>Thank you very much for pointing this out. To determine if considering different contextual effects within each feature cluster would enhance model performance, we modified our feature selection by choosing both the feature with the largest positive and the largest negative effect on RNA half-life in Step III of Figure 3A. We then split the data into a 2:1 training and testing set and repeated this process 100 times. Model performance was evaluated using mean average error (MAE), root mean squared error (RMSE), and adjusted R-squared. From Author response image 4, we observed no significant improvement in model performance using this new approach. Notably, in the SH-SY5Y 5' UTR model, our original method even outperformed the modified one, with statistically lower MAE and RMSE and a higher adjusted R-squared. Therefore, we believe our current approach remains appropriate.</p><fig id="sa4fig4" position="float"><label>Author response image 4.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-sa4-fig4-v1.tif"/></fig><disp-quote content-type="editor-comment"><p>&quot;Overall, motifs that are at least two nucleotides long proved critical for RNA stability, supporting the sequence specificity of the decay process.&quot; Unclear why this supports the &quot;sequence specificity&quot;</p></disp-quote><p>No monomers were selected as an explanatory factor. On the contrary, specific sequence combinations and order are important for the regulation. These findings suggest sequence-specific recognition for the decay process.</p><disp-quote content-type="editor-comment"><p>Fig3: The same features were used in both cell lines? If yes: Since they were selected for their highest correlation with half-life, how was a common set chosen? If no: problematic to compare.</p></disp-quote><p>Thank you for your question regarding feature selection across cell lines. Initially, the features were collected uniformly for both cell lines. However, subsequent feature selection steps were cell-type specific, focusing on identifying features with the greatest impact on RNA half-life in each context. This approach allows us to still compare model performance and discuss the similarities and differences in selected features across cell types. By maintaining a consistent starting point, we ensure that any observed differences reflect cell-specific regulatory dynamics.</p><disp-quote content-type="editor-comment"><p>uORFs were not used as features?</p></disp-quote><p>Thank you for pointing this out. At the beginning of our study, we investigated the impact of Kozak sequence strength (categorized as weak, moderate, strong, or optimal) on RNA half-life. However, we found that this feature performed poorly in predicting RNA stability, and as a result, we decided not to include upstream open reading frames (uORFs) or Kozak sequences in our subsequent analyses.</p><disp-quote content-type="editor-comment"><p>Experimental reproducibility: Only correlations between replicates for the same time point is shown, but no comparison between time points or between decay rates. How reproducible were the paired differences between mut/wt?</p></disp-quote><p>The decay rate was calculated by modeling the slope of a linear regression of all time points. Therefore, there is only one decay rate associated with a genotype. To rule out inconsistent data, we excluded any regression with a mean square error greater than 1, as this indicates a poor fit of the data points.</p><disp-quote content-type="editor-comment"><p>Fig 7C/p17: This does not establish a &quot;causal relationship&quot; as the authors claim.</p></disp-quote><p>We agree with the reviewer’s suggestion. We have modified the text on p.17 to “to establish a correlation between UTR variants and health outcomes,…..”</p><disp-quote content-type="editor-comment"><p>In the discussion, the authors claim that TA-diNTs are not only an opposite of the GC percentage and base this on Fig 5A.</p></disp-quote><p>Fig 5A: The range of TA-diNTs is naturally much higher in the low GC group. To make the high and low GC content comparable (as the authors aim to do), the correlation should be assessed for the same range of TA dint in both cases.</p><p>To address this concern more rigorously, we performed a stratified analysis based on UA-diNT rate. As shown in our Fig. S7C, even after stratifying by UA- dinucleotide ratio (upper panel high UA- dinucleotide ratio / lower panel low UA- dinucleotide ratio), we still observe that the destabilizing effect of UA is stronger in the low GC content group.</p><p>Supplemental Figure S7. Interplay of GC content and TA dinucleotide on stability regulation, related to Figure 5. (C) Stratifications of both TA dinucleotide ratio and GC content showed that the destabilizing effect of TA dinucleotide is the most prominent under conditions of low TA dinucleotide ratio and low GC content. The same trend was observed for 5’ UTR (left) and 3’ UTR (right).</p><disp-quote content-type="editor-comment"><p>The injection of in vitro transcribed and polyA/capped RNA certainly has advantages over other methods, but delivering naked mRNA without nuclear history might also lead to artifacts. The caveats of the approach should be discussed more extensively.</p></disp-quote><p>We appreciate the suggestion and have hence added the following in the Discussion (p.18): “However, while our approach effectively assesses the stability of synthesized RNA in human cells, it may not fully capture the decay dynamics of nuclear-synthesized RNA, which can be influenced by endogenous modifications and trans-acting RNA binding factors.”</p><disp-quote content-type="editor-comment"><p>&quot;We unexpectedly identified many crucial regulatory features in 5' UTRs.&quot; Why was this unexpected?</p></disp-quote><p>We initially thought the 3’ UTR would play a major role in stability regulation. To avoid confusion, we have removed the word ‘unexpected’ from the text (p. 20): &quot;We identified many crucial regulatory features in 5' UTRs.&quot;</p><disp-quote content-type="editor-comment"><p>&quot;...a massively parallel reporter assay in which coding regions and human 5'/3' UTRs with diseaserelevant mutations were generated in vitro and then directly transfected into human cell lines to assess their decay patterns by next‐generation sequencing&quot;: also coding regions?</p></disp-quote><p>Thanks for the question. Indeed, the coding region was not synthesized together with the UTR library. Therefore, we modified the text of p. 6 to “…we developed a massively parallel reporter assay in which human 5’/3’ UTRs with disease-relevant mutations were generated in vitro, ligated with the enhanced green fluorescence protein (EGFP) coding region, and then directly transfected into human cell lines to assess their decay patterns by next-generation sequencing.”</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #2 (Recommendations For The Authors):</bold></p><p>Nomenclature: When discussing RNA sequences, &quot;U&quot; should be used in place of &quot;T&quot; (e.g., &quot;UA dinucleotide&quot;).</p></disp-quote><p>We have replaced the RNA sequence “T” with “U” of the text and figures.</p><disp-quote content-type="editor-comment"><p>Abstract: &quot;We examined the RNA degradation patterns mediated by the UTR library in multiple cell lines&quot; - It would be clearer to state that two cell lines (rather than multiple) were used.</p></disp-quote><p>We appreciate the suggestion. We have modified the abstract as suggested: “We examined the RNA degradation patterns mediated by the UTR library in two cell lines…&quot;</p><disp-quote content-type="editor-comment"><p>The manuscript refers to &quot;wild-type (WT) and mutant (mt) alleles.&quot; (p. 7 and elsewhere). It would be better to use &quot;reference&quot; instead of &quot;wild type&quot; given that these are human populations.</p></disp-quote><p>We appreciate the suggestion. All instances of ‘wild-type’ or ‘WT’ in the text and figures have been replaced with ‘reference’ or ‘ref’.</p><disp-quote content-type="editor-comment"><p>In the introduction, it is stated that traditional MPRAs &quot;cannot differentiate the effect of the UTRs on transcription, stability and, in some cases, even protein production, greatly limiting scientific interpretation.&quot; This is confusing, since these assays can and have been used in association with both RNA decay measurements and measurements of reporter protein levels that allow assessment of effects on stability and protein production (including in the cited references).</p></disp-quote><p>We reason that the RNA steady-state level (e.g., sequencing the overall RNA normalized to DNA) or protein steady-state level (e.g., detecting the fluorescence signal) does not precisely reveal the decay kinetics of the RNA. Steady-state level is a result of production and decay, both of which UTRs contribute to. Similarly, the protein level is not a perfect estimate of the RNA decay.</p><p>To clarify, we have modified the introduction (p. 5) to “Nevertheless, because the steady-state level is a result of production and decay, these approaches cannot differentiate the effect of the UTRs on transcription, stability and, in some cases, even protein production, greatly limiting scientific interpretation.”</p><p>Adding raw and normalized read count data from individual experiments (e.g., to Table S1) would make it more likely for others to use this dataset to address additional questions.</p><p>All raw and processed sequencing data generated in this study have been submitted to the NCBI Gene Expression Omnibus (GEO; <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/geo/">https://www.ncbi.nlm.nih.gov/geo/</ext-link>) under accession number GSE217518 (reviewer token snspaakujtsdpcv).</p><disp-quote content-type="editor-comment"><p>The manuscript would benefit from further clarification about model selection. Additional details regarding how the features were clustered, and the actual clusters themselves should be included.</p><p>It should be discussed why Lasso was chosen vs Ridge or Elastic Net, in the context of handling multicollinearity. Often, data is subsetted for training and validation, and model performance metrics are presented.</p></disp-quote><p>Thank you for pointing out the need for further clarification on model selection. The features were clustered using single-linkage hierarchical clustering with the distance metric defined as one minus the absolute value of the Spearman correlation coefficient (this information has been added to the manuscript on p. 28: “…to avoid multicollinearity caused by similar features that perturb feature selection, all features were clustered using single-linkage hierarchical clustering with the distance metric defined as one minus the absolute value of the Spearman correlation coefficient.”). The resulting feature clusters are available in Supplemental Table S3.</p><p>Regarding model selection, we chose LASSO over ridge and elastic net primarily for feature selection, as ridge does not perform feature selection. Elastic net is essentially a hybrid of ridge regression and LASSO regularization, but we opted for LASSO for its simplicity and effectiveness in selecting a sparse set of important features.</p><p>We also performed a 2:1 training and testing set analysis and have included these details in the manuscript. Model performance metrics, including correlation coefficient between observed and predicted values in the testing set, mean absolute error (MAE), root mean squared error (RMSE), mean absolute percentage error (MAPE), and R-squared, are provided in new Supplemental Table S4.</p><disp-quote content-type="editor-comment"><p>Recommend reviewing and correcting verb tenses in the methods section.</p></disp-quote><p>We appreciate the reviewer’s suggestion. We have corrected verb tenses in the methods section, which includes “The UTRs were defined by NCBI RefSeq and ENCODE V27. (p.21)”, “The variant was placed in the middle of the sequence….(p.22)”, and “eCLIP signals with value &lt; 1 or p value &gt; 0.05 were removed. (p.26)”</p><disp-quote content-type="editor-comment"><p>Please add information about which cell type(s) are being used in each of the figure legends (e.g., in Figs. 2B and 5).</p></disp-quote><p>We appreciate the reviewer’s suggestion. We have added the cell type information in the figure legends: “Figure 2…. (B) The ten most influential AREs in terms of RNA stability in SH-SY5Y cells.” And “Figure 5…..(A) MPRA data of SH-SY5Y cells stratified according to the GC content (GC%) of UTRs.”</p><disp-quote content-type="editor-comment"><p>Recommend review of axis labels and consistency in formatting the log(half-lives) and including the base of the log and the time unit (minutes). Even better, converting axis labels from log minutes to minutes would make this easier to understand.</p></disp-quote><p>Thank you for the suggestion regarding axis labels and consistency. We have unified the half-life label to ‘ln t<sub>1/2</sub> (min)’ in all figures. We chose not to convert the axis from logarithmic minutes to minutes because the original scale is highly skewed, which would hinder clear data visualization.</p><disp-quote content-type="editor-comment"><p>The discussion refers to Figure 1D but Figure 1 only has A-C</p></disp-quote><p>Thank you for pointing out this mistake. ‘Fig. 1D’ has been changed to ‘Fig. 1B’ in the text (p. 7 and p. 20).</p><disp-quote content-type="editor-comment"><p>The analyses in Fig. 2 are interpreted as demonstrating that AREs destabilize RNAs. These analyses are examining associations, so it would be more appropriate to say that AREs are associated with destabilization (since it is formally possible that other sequences that are present in these UTR fragment cause destabilization). A similar issue arises on p. 10: &quot;TA dinucleotides alone can negatively regulate RNA stability, with a Pearson's correlation coefficient of ‐0.287 for 5' UTRs and ‐0.377 for 3' UTRs (Fig. 4A,C).&quot; This is an association and does not establish causation. Again on p. 17: &quot;We identified several SNPs in UTRs that induce aberrant RNA expression and/or protein expression (Supplemental Table S7).&quot; These may be causal but may simply be in LD with other variants that are causal.</p></disp-quote><p>We agree that the association observed is not proven to be causal. Therefore, we modified the text as suggested:</p><p>“AUUUA/AUUA-containing AREs are associated with RNA destabilization.” (p. 8)</p><p>“UA dinucleotides alone present a negative correlation with RNA stability, with a Pearson’s correlation coefficient of -0.287 for 5’ UTRs and -0.377 for 3’ UTRs.” (p.10)</p><p>“We identified several SNPs in UTRs that correlated with aberrant RNA expression and/or protein expression.” (p. 17)</p><disp-quote content-type="editor-comment"><p>Figure 4C is important in that it examines whether variant sequences that differ in a manner that changes the number of dinucleotide repeats affect stability. Please show the number (not just the percentage) of sequences in each category.</p></disp-quote><p>Thank you for your insightful comment. We believe the figure you referred to is Figure 4E. We have updated the figure to include the number of sequences in each category.</p><disp-quote content-type="editor-comment"><p>Figure 6A and B: The horizontal axes appear to be misaligned since the dotted vertical lines do not cross at 0. ?</p></disp-quote><p>The dotted vertical lines represent the genomic background of the UA-diNT ratio. To clarify it, we have modified the legend to: “Figure 6……(A) The top ten biological processes for which the 5’ UTR UA-dinucleotide ratio most significantly deviated from the genomic background (dashed line).”</p><disp-quote content-type="editor-comment"><p>It may be helpful to state what the dashed and solid lines represent on Figure 6 E/F. Please correct spelling of &quot;Biological&quot; in 6E.</p></disp-quote><p>As per the reviewer’s suggestions, we have modified the legend of Figure 6 to: “………..(E) Biological processes for RNAs in which the UA-dinucleotide ratios of both 5’ and 3’ UTRs are significantly different from the genomic background (dashed lines). (F) Molecular functions for RNAs in which the UA-dinucleotide ratios of both 5’ and 3’ UTRs are significantly different from the genomic background (dashed lines). The thin solid lines represent the standard deviation of the UAdinucleotide ratio within the gene group.”</p><p>In addition, the spelling of “Biological” in Fig. 6E has been corrected.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #3 (Recommendations For The Authors):</bold></p><p>I have 3 points that I think could improve science and its presentation within the manuscript.</p><p>(1) Most importantly, how well do LASSO regression models predict the stability of native transcripts? Such analysis can also be useful for comparison between two different cell-types. How well does the regression model learned (on reporters) within one cell-type predict mRNA stability (of reporters and native genes) in this cell-type and in the other cell-type? Similarly, models can also help to analyze the effects of 5'UTR and 3'UTR sequences on mRNA stability. In particular, how well does the regression model of each separate regulatory sequence (3'UTR or 5'UTR) is able to predict the stability of native genes in the cell? Can the predictions be improved by combining both 3'UTR and 5'UTR sequence features within the regression models?</p></disp-quote><p>The decay model for native transcripts has been established in prior research (doi.org/10.1186/s13059-022-02811-x; doi.org/10.1186/s12915-021-00949-x), which indicates that exon junction density and transcript length are the primary determinants of RNA stability. Based on these findings, we designed the MPRA with fixed length and without splicing to focus on the contribution of primary sequences. We validated the destabilizing effect of UA dinucleotide on endogenous RNAs (Fig. 4G and Supplemental Fig. S5F) but do not recommend using our model to fully explain or predict the stability of native transcripts.</p><p>To assess the model's cross-cell type predictive performance for RNA half-life, we employed the Regression Error Characteristic (REC) curve (Bi &amp; Bennett, 2003). Similar to the receiver operating characteristic (ROC) curve, the REC curve illustrates the trade-off between error tolerance and accuracy, with better performance indicated by curves trending toward the upper left. We also computed the Area Over the Curve (AOC) as a performance metric, where lower values indicate better predictive ability. From Author response image 5, the REC curves reveal that cross-cell type prediction performance is suboptimal. The y-axis represents prediction accuracy, while the x-axis denotes error tolerance for the natural logarithm of RNA half-life (ln(<italic>t1/2</italic>), in minutes).</p><fig id="sa4fig5" position="float"><label>Author response image 5.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-97682-sa4-fig5-v1.tif"/></fig><p>In response to the suggestion of combining 5' and 3' UTR sequence features in the regression model, we believe this approach may not be ideal. As shown in Figure S1D, the distribution of RNA half-lives between 5' and 3' UTRs is significantly different, reflecting their distinct regulatory roles. Additionally, the base composition differs, with 5' UTRs having a higher GC content compared to 3' UTRs. Combining these datasets would likely make the origin of the sequence (5' or 3' UTR) the most predictive feature, thereby reducing the model's interpretability. Furthermore, our MPRA results, derived from separate 5’ or 3’ UTR library, do not support a combined model, further suggesting this approach may not be suitable with our data.</p><disp-quote content-type="editor-comment"><p>The conclusions regarding genetic variants are interesting, yet the main strength of the work involves identifying general sequence features that affect mRNA stability rather than specific variants. I wonder if the authors have considered to shift the focus of the motivation part to reflect that?</p></disp-quote><p>We appreciated the reviewer’s suggestion. We have revised the abstract and introductions to emphasize the general UTR regulation. Here is the revised abstract:</p><p>UTRs contain crucial regulatory elements for RNA stability, translation and localization, so their integrity is indispensable for gene expression. Approximately 3.7% of genetic variants associated with diseases occur in UTRs, yet a comprehensive understanding of UTR variant functions remains limited due to inefficient experimental and computational assessment methods. To systematically evaluate the effects of UTR variants on RNA stability, we established a massively parallel reporter assay on 6,555 UTR variants reported in human disease databases. We examined the RNA degradation patterns mediated by the UTR library in two cell lines, and then applied LASSO regression to model the influential regulators of RNA stability. We found that UA dinucleotides and UA-rich motifs are the most prominent destabilizing element. Gain of UA dinucleotide outlined mutant UTRs with reduced stability. Studies on endogenous transcripts indicate that high UA-dinucleotide ratios in UTRs promote RNA degradation. Conversely, elevated GC content and protein binding on UA dinucleotides protect high-UA RNA from degradation. Further analysis reveals polarized roles of UA-dinucleotide-binding proteins in RNA protection and degradation. Furthermore, the UA-dinucleotide ratio of both UTRs is a common characteristic of genes in innate immune response pathways, implying a coordinated stability regulation through UTRs at the transcriptomic level. We also demonstrate that stability-altering UTRs are associated with changes in biobank-based health indices, underscoring the importance of precise UTR regulation for wellness. Our study highlights the importance of RNA stability regulation through UTR primary sequences, paving the way for further exploration of their implications in gene networks and precision medicine.</p><disp-quote content-type="editor-comment"><p>Plots presenting correlations (e.g., Figure 4A, 4C) are more informative when plotted as density plots (i.e., using colorscale to show density of the dots at each part of the plot).</p></disp-quote><p>We greatly appreciate the reviewer's insightful suggestion regarding the use of density plots for presenting correlations. We have modified Figures 4A and 4C in the revised manuscript to implement density plotting. The updated figures now utilize a colorscale that highlights areas of high and low data density.</p></body></sub-article></article>