<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Bioinform Biotech</journal-id><journal-id journal-id-type="publisher-id">bioinform</journal-id><journal-id journal-id-type="index">19</journal-id><journal-title>JMIR Bioinformatics and Biotechnology</journal-title><abbrev-journal-title>JMIR Bioinform Biotech</abbrev-journal-title><issn pub-type="epub">2563-3570</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v7i1e92529</article-id><article-id pub-id-type="doi">10.2196/92529</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Enriching Public Pathogen Genomic Records With Patient Metadata: Quantitative Study and Exploratory Case Analysis</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Pavia</surname><given-names>Michael J</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>O'Connor</surname><given-names>Karen</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Gonzalez-Hernandez</surname><given-names>Graciela</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Scotch</surname><given-names>Matthew</given-names></name><degrees>MPH, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff4">4</xref></contrib></contrib-group><aff id="aff1"><institution>Biodesign Center for Environmental Health Engineering, Arizona State University</institution><addr-line>1151 S. Forest Ave</addr-line><addr-line>Tempe</addr-line><addr-line>AZ</addr-line><country>United States</country></aff><aff id="aff2"><institution>Department of Biostatistics, Epidemiology and Informatics, Perelman School of Medicine, University of Pennsylvania</institution><addr-line>Philadelphia</addr-line><addr-line>PA</addr-line><country>United States</country></aff><aff id="aff3"><institution>Department of Computational Biomedicine, Cedars-Sinai Medical Center</institution><addr-line>Los Angeles</addr-line><addr-line>CA</addr-line><country>United States</country></aff><aff id="aff4"><institution>College of Health Solutions, Arizona State University</institution><addr-line>Phoenix</addr-line><addr-line>AZ</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Yue</surname><given-names>Zongliang</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Giap</surname><given-names>Binh Duong</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Wu</surname><given-names>Hao</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Kurmashev</surname><given-names>Ruslan</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Matthew Scotch, MPH, PhD, Biodesign Center for Environmental Health Engineering, Arizona State University, 1151 S. Forest Ave, Tempe, AZ, United States, 1 855-278-5080; <email>Matthew.Scotch@asu.edu</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>6</day><month>10</month><year>2026</year></pub-date><volume>7</volume><elocation-id>e92529</elocation-id><history><date date-type="received"><day>02</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>07</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>18</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Michael J Pavia, Karen O'Connor, Graciela Gonzalez-Hernandez, Matthew Scotch. Originally published in JMIR Bioinformatics and Biotechnology (<ext-link ext-link-type="uri" xlink:href="https://bioinform.jmir.org">https://bioinform.jmir.org</ext-link>), 6.10.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">http://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Bioinformatics and Biotechnology, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://bioinform.jmir.org/">https://bioinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://bioinform.jmir.org/2026/1/e92529"/><abstract><sec><title>Background</title><p>During the COVID-19 pandemic, large-scale sequencing generated millions of SARS-CoV-2 genomes in public repositories including GenBank and GISAID. However, most records lack detailed patient metadata, including demographic information and clinical outcomes. This lack of host-associated information limits their utility for large-scale pathogen genomics analyses. Although sequence records linked to journal publications may contain relevant metadata, systematically extracting and linking this information requires substantial manual effort.</p></sec><sec><title>Objective</title><p>This study aimed to assess host metadata completeness in GenBank SARS-CoV-2 records and to demonstrate, through an exploratory case study, analytical opportunities enabled by enriched clinical and demographic annotations for genomic epidemiology.</p></sec><sec sec-type="methods"><title>Methods</title><p>The authors searched LitCovid for PubMed Central (PMC) articles published between January 2023 and December 2024 that reported original complete SARS-CoV-2 genome sequences deposited in GenBank with sequence-specific patient metadata from human hosts. Two independent reviewers screened eligible articles and manually extracted metadata on sample collection, demographics, treatments, serology, vaccination status, infection presentation, and clinical outcomes. Enriched metadata were defined as patient information in the publication but absent from GenBank records. Synonymous clinical and demographic terms were standardized using SNOMED CT, and Charlson Comorbidity Index scores were calculated when metadata were available. SARS-CoV-2 genomes were retrieved, assembled, and subjected to quality control using standard bioinformatics tools. To demonstrate enriched metadata&#x2019;s analytical value, we selected a subset of genomes with longitudinal sequencing, comprising 100 genomes from 34 patients across 4 studies for exploratory analysis of within-host viral evolution and patient outcomes.</p></sec><sec sec-type="results"><title>Results</title><p>Among 116,600 articles identified through our PMC/LitCovid screening framework, approximately 0.02% (n=21) reported original complete SARS-CoV-2 genomes deposited in GenBank with accessible sequence-specific metadata. Within this eligible set, GenBank records contained on average 22% completeness for host metadata in our extraction schema. Completeness was confined to sample fields (averaging 81%); host demographic and clinical fields averaged 2%. Manual enrichment increased overall completeness by 30% on average, recovering 56% (14/25) of metadata types absent from GenBank records. In the longitudinal subset, enriched metadata enabled host-stratified analyses, showing nominal associations between immunocompromised status and higher within-host evolutionary rates (<italic>P</italic>=.02) and unique amino acid mutations (<italic>P</italic>=.04). Models with enriched patient and treatment metadata outperformed mutation-only models for mortality, hospitalization, and infection duration.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Enriched host metadata improve the utility of pathogen genomic data by enabling analyses linking viral variation with demographics and clinical outcomes. Despite limited clinical and demographic information in examined GenBank records, manual enrichment facilitated a more comprehensive view of viral evolution and disease dynamics than sequence data alone. The case study&#x2019;s genotype-phenotype associations are exploratory and require validation in larger, independently collected cohorts. These findings highlight the need for more standardized, structured, and accessible patient metadata deposition with genomic sequences to strengthen pathogen genomics and precision public health research.</p></sec></abstract><kwd-group><kwd>pathogen genomics</kwd><kwd>genomic surveillance</kwd><kwd>metadata standards</kwd><kwd>bioinformatics</kwd><kwd>data interoperability</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>The COVID-19 pandemic brought about significant changes in pathogen genomics, with large numbers of viral sequences being shared openly in nucleotide repositories like GenBank [<xref ref-type="bibr" rid="ref1">1</xref>] and Global Initiative on Sharing All Influenza Data (GISAID) [<xref ref-type="bibr" rid="ref2">2</xref>]. These databases are essential for answering scientific questions on SARS-CoV-2 or other pathogens, including differences in rates of evolution [<xref ref-type="bibr" rid="ref3">3</xref>], transmission patterns [<xref ref-type="bibr" rid="ref4">4</xref>], and geographic variation in the diversity of variants of concern [<xref ref-type="bibr" rid="ref5">5</xref>]. While these platforms facilitate rapid sharing of genomic data, their utility is limited by inconsistent or incomplete reporting of patient metadata, such as demographics, clinical outcomes, and/or comorbidities [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. This metadata may be described in the accompanying publications, yet sequence records often lack a manuscript reference, either because one does not exist or because the record was never updated to include the publication link [<xref ref-type="bibr" rid="ref8">8</xref>]. Consequently, the lack of metadata in pathogen genomics limits our capacity to connect viral sequences with patient phenotypes, impeding the identification of key epidemiological patterns.</p><p>Integrating metadata with viral genomic data is essential for understanding how viral variation contributes to clinical outcomes. For example, we know that genetic variations in the SARS-CoV-2 spike glycoprotein and host factors, such as angiotensin-converting enzyme 2 (<italic>ACE2</italic>) and transmembrane serine protease 2 (<italic>TMPRSS2</italic>) polymorphisms, directly influence viral infectivity and disease severity [<xref ref-type="bibr" rid="ref9">9</xref>]. In addition, while comparative analyses across variants have demonstrated that hospitalization rates vary by lineage [<xref ref-type="bibr" rid="ref10">10</xref>], the associations often lose statistical significance once models account for population-level factors, such as changes in clinical care standards and testing practices [<xref ref-type="bibr" rid="ref11">11</xref>]. Furthermore, without comorbidity data such as obesity (as measured by BMI), the impact of the A20268G mutation on hospitalization would have been attributed only to viral lineage [<xref ref-type="bibr" rid="ref12">12</xref>].</p><p>Understanding the relationship between viral genotypes and patient clinical outcomes depends on the granularity of available metadata. For instance, Patel et al [<xref ref-type="bibr" rid="ref13">13</xref>] linked SARS-CoV-2 mutation profiles to age and geography to find that working-age individuals (18&#x2010;64 years) carry the highest burden of unique mutations, with high regional variability. Incorporating metadata on vaccination phases further showed distinct spikes in unique mutations that coincided with the vaccine rollout, indicating that demographics, geography, and public health initiatives together influence evolutionary rates. In another example, longitudinal sequencing of SARS-CoV-2 infections in immunosuppressed patients receiving antiviral treatments showed that these individuals harbor a greater number of private (nonlineage) mutations, which would have been missed in coarser datasets [<xref ref-type="bibr" rid="ref14">14</xref>]. Furthermore, Larsen et al [<xref ref-type="bibr" rid="ref15">15</xref>] found that the spike protein mutation D614G is linked to shifts in symptom progression (a tendency for cough to precede fever), a relationship that emerged only after the inclusion of detailed clinical symptom data in the analysis. These studies demonstrate that integrating detailed, high-quality patient metadata is critical for clarifying the clinical and public health implications of viral evolutionary dynamics.</p><p>Incomplete or ambiguous metadata pose significant challenges for accurate genomic epidemiology. As an example, in a global analysis of SARS-CoV-2 sequences from GISAID, it was observed that 63% of records lacked demographic data and more than 95% were missing patient-level clinical information [<xref ref-type="bibr" rid="ref16">16</xref>]. Additionally, we have described the lack of host location metadata in pathogen sequence records in GenBank [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>]. To address this issue, we have developed a framework integrating uncertainty in sampling locations into phylodynamic analysis, rather than relying on fixed geographic assignments [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>]. This approach outperforms conventional methods when reconstructing viral persistence times, migration rates, and ancestral origins. Additionally, efforts have been made to develop a natural language processing (NLP)&#x2013;based system designed to automatically extract and refine geospatial data directly from scientific literature [<xref ref-type="bibr" rid="ref21">21</xref>].</p><p>The goal of this study is to demonstrate how sequence-specific host metadata affect the utility of public pathogen sequence data, using SARS-CoV-2 as the primary case study. First, we quantify the availability of patient-level metadata linked to SARS-CoV-2 genomes in GenBank, among recent open-access publications in LitCovid [<xref ref-type="bibr" rid="ref22">22</xref>]. We then describe the enrichment workflow for extraction, normalization, and linking of patient metadata to sequence records. From this enriched dataset, we use a subset of samples with longitudinal sequencing to demonstrate analyses that are either not possible or are limited when using the metadata stored in GenBank. This case study is intended as exploratory rather than as a definitive test of genotype-phenotype associations. The objective is to illustrate the analytical opportunities enabled by metadata enrichment and to identify candidate associations that warrant evaluation in larger, independently collected datasets. By leveraging metadata-enriched pathogen genomes, we demonstrate an untapped potential for informing more effective responses to emerging viral threats and precision public health research.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Systematic Search, Screening, and Data Enrichment</title><p>We searched LitCovid [<xref ref-type="bibr" rid="ref22">22</xref>] for full-text PubMed Central (PMC) articles published between January 2023 and December 2024. We used regular expressions to screen for (1) mentions of sequence databases (GenBank, BioProject, BioSample, Sequence Read Archive [SRA], and GISAID), (2) strings that matched the alphanumeric GenBank accession number format, and (3) references to variants of interest (VOI) or variants under monitoring (VUM), including BA.2, BA.2.86, JN.1, KP.2, KP.3, KP.3.1.1, LB.1, XEC, JN.1.7, and JN.1.18, as defined by the World Health Organization (WHO) in November 2024. Two reviewers independently assessed articles for all 3 mentions. We specified an inclusion criterion where an article needed to report (1) original SARS-CoV-2 complete genome sequence data, (2) the deposit of the raw reads or consensus sequences in GenBank, and (3) sequence-specific patient metadata derived from human hosts. For this effort, we only considered sequences deposited in GenBank to leverage the National Center for Biotechnology Information (NCBI) <italic>Entrez</italic> environment, which links PMC and GenBank (nucleotide) databases [<xref ref-type="bibr" rid="ref23">23</xref>]. We therefore excluded articles that submitted data to GISAID or that were derived from nonhuman hosts or from environmental sources such as wastewater.</p><p>For articles that met our inclusion criteria, we manually extracted sequence-specific patient metadata encompassing sample collection details, patient demographics, treatment regimens, laboratory results, vaccination status, infection presentation, and clinical outcomes when available. We defined enriched metadata as patient information that was not reported in GenBank. Pre-enriched refers to the metadata that were recovered from GenBank; this metric was different for each article. Metadata that were not reported, missing, or not applicable were recorded as absent without distinguishing between them, and the number of assessed fields remained fixed across studies. Additionally, for each article where enriched metadata contained enough demographic and clinical information, such as age and comorbidities, we calculated the Charlson Comorbidity Index (CCI) [<xref ref-type="bibr" rid="ref24">24</xref>]; CCI was not calculated when comorbidity data were incomplete. To ensure consistency across studies, we grouped synonymous terms (including sampling location/method, comorbidities, and treatment coding) extracted from articles according to SNOMED CT [<xref ref-type="bibr" rid="ref25">25</xref>]. Terms that did not map unambiguously were left as initially extracted. A complete breakdown of the LitCovid screening and metadata enrichment protocols can be found in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendices 1 to 3</xref><xref ref-type="supplementary-material" rid="app2"/>-<xref ref-type="supplementary-material" rid="app3">3</xref>.</p><p>We assigned each extracted field to 1 of 2 tiers. Universally expected fields&#x2014;collection date, country of origin, geographic location, and biospecimen type&#x2014;describe the sample and apply to any deposited sequence regardless of study design. Context-dependent fields describe demographics, clinical presentation, treatment, and outcome, and their relevance varies based on study design. This distinction was necessary because a missing value carries a different meaning in each tier. Absence of a universally expected field reflects incomplete deposition, whereas absence of a context-dependent field may instead indicate a variable that was not measured or was not applicable.</p><p>To evaluate metadata enrichment, we used 4 complementary metrics, each calculated per study and reported as the mean across the 21 studies unless otherwise stated. First, metadata completeness was calculated as the proportion of predefined metadata fields available in GenBank relative to the total number assessed, with missing metadata defined using the same denominator. Second, enrichment completeness was calculated after incorporating metadata extracted from the associated publications. Third, recovered metadata types refer to distinct metadata categories (age, sex, symptoms, hospitalization, mortality, etc) that were absent from GenBank but obtained through manual extraction; the reported percentage represents the proportion of previously unavailable metadata types recovered. Finally, additional metadata variables per sample were calculated as the number of new metadata fields linked to an individual sequence following enrichment. Completeness before and after enrichment was additionally calculated within each tier. Together, these metrics capture changes in overall completeness, recovery of previously unavailable metadata categories, and the number of additional patient-level variables linked to genomic records.</p></sec><sec id="s2-2"><title>SARS-CoV-2 Genome Retrieval and Assembly</title><p>Using NCBI Entrez Direct [<xref ref-type="bibr" rid="ref26">26</xref>], we obtained SARS-CoV-2 genomes from GenBank and SRA libraries for 8 of the studies. We assembled the raw Illumina reads by first removing low-quality reads and adapters with fastp [<xref ref-type="bibr" rid="ref27">27</xref>], followed by IRMA (iterative refinement meta-assembler) for reference-based assembly [<xref ref-type="bibr" rid="ref28">28</xref>]. For 3 studies where both GISAID-deposited genomes and SRA libraries were available, we used the GISAID genomes solely as an external reference to validate our assemblies, comparing them with BLAST (Basic Local Alignment Search Tool) [<xref ref-type="bibr" rid="ref29">29</xref>]; GISAID sequences were not used in any downstream analyses. We applied a threshold of &#x003E;99.5% identity or fewer than 6 combined mismatches and gaps; assemblies that did not meet these criteria were excluded from downstream analyses. We used Nextclade [<xref ref-type="bibr" rid="ref30">30</xref>] to assign clades, evaluate genome quality, and identify both nucleotide and amino acid mutations relative to the Wuhan-1 genome [NC_045512.2]. Finally, we excluded any genome classified as &#x201C;bad,&#x201D; by Nextclade [<xref ref-type="bibr" rid="ref30">30</xref>] for its overall QC status from downstream analyses.</p></sec><sec id="s2-3"><title>Sequence Selection and Metadata Integration for Within-Host Analyses</title><p>While genomic data alone can be used to identify amino acid mutations, linking or stratifying these observations with clinical outcomes requires integration with patient metadata. Using the enriched dataset, we studied how the addition of immune status, comorbidities, and treatment regimen metadata improved our ability to understand viral evolution and patient outcomes for SARS-CoV-2 infections. To strengthen generalizability, we combined data from multiple articles in which longitudinal SARS-CoV-2 sequence data were available, as well as corresponding patient metadata encompassing immune status, treatments, and outcomes. This curated dataset consisted of 100 genomes from 34 patients (<xref ref-type="fig" rid="figure1">Figure 1</xref> and Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), drawn from Gonzalez_Reiche_2023 [<xref ref-type="bibr" rid="ref31">31</xref>], Igari_2024 [<xref ref-type="bibr" rid="ref32">32</xref>], Manuto_2024 [<xref ref-type="bibr" rid="ref33">33</xref>], and Pavia_2024 [<xref ref-type="bibr" rid="ref34">34</xref>].</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Heatmap of sequence-specific patient metadata collected [<xref ref-type="bibr" rid="ref31">31</xref>-<xref ref-type="bibr" rid="ref51">51</xref>]. Dark blocks represent metadata available in GenBank, light blocks represent metadata extracted from publication text, and white blocks represent missing metadata. Fields are grouped into 6 blocks and ordered from study metadata to clinical treatment: study metadata, sample collection, patient characteristics, infection characteristics, disease severity and outcomes, and treatment. Studies highlighted in red were used in the case study. Local refers to state or city vs county level location. Ct refers to the cycle threshold in reverse transcription polymerase chain reaction, with Ct precision indicating either exact value or a reported range. Symptom status refers to whether cases are asymptomatic or symptomatic cases, and symptoms lists the specified reported symptoms. Vaccination status refers to whether a patient was vaccinated, and vaccine dose specifies the number of doses received prior to sequencing. Comorbidity status refers to the presence or absence of comorbidities, and comorbidity lists the specific conditions reported. Ab: antibody; CCI: Charlson Comorbidity Index.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="bioinform_v7i1e92529_fig01.png"/></fig></sec><sec id="s2-4"><title>Case Study: Linking Immune Status to Within-Host Evolution and Outcomes via Enriched Metadata</title><p>We analyzed longitudinal SARS-CoV-2 genomes from 4 articles encompassing individuals with varying immune status, comorbidities, and treatment regimens. We performed phylogenetic reconstruction using IQ-TREE (GTR+G model) [<xref ref-type="bibr" rid="ref52">52</xref>] with 1000 bootstrap replicates. For time calibration, we used the treedater package [<xref ref-type="bibr" rid="ref53">53</xref>] in R (R Foundation for Statistical Computing) under an uncorrelated clock model. We estimated evolutionary rates for patients with two or more time points by fitting a linear regression of root-to-tip genetic distances against sampling dates and further assessed the dependence of these rates (mutations per site per year per patient) on immune status and viral clade using multiple linear regression. We identified nonrandom recurrent mutations by applying an empirical binomial model to estimate the expected frequencies of amino acid and nucleotide mutations across the cohort. The resulting binomial <italic>P</italic> values were then adjusted using the Benjamini-Hochberg method [<xref ref-type="bibr" rid="ref54">54</xref>]. We defined empirical recurrence thresholds as the minimum number of patients for which the mutation&#x2019;s observed frequency yielded a cumulative binomial <italic>P</italic>&#x003C;.05. Finally, we classified as <italic>recurrent</italic> any mutations that exceeded the threshold and appeared in two or more viral clades. We set recurrence thresholds at a minimum of 11 patients for amino acid mutations and 10 patients for nucleotide mutations (Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). To assess associations between amino acid mutations, patient risk factors (immune status and CCI), treatment regimens, and outcomes (mortality, hospitalization, and infection duration), we used generalized linear models with logistic regression for binary outcomes (mortality and hospitalization) and linear regression for infection duration. We grouped recurrent amino acid mutations with identical patient-level patterns to prevent redundancy and reduce model complexity. Our models consisted of (1) mutation group only, (2) mutation group plus antiviral treatment, (3) mutation group plus patient factors, and (4) mutation group with antiviral treatment and patient factors. Patient-level covariates and outcomes (immune status, CCI, treatment regimen, hospitalization, mortality, and infection duration) were linked to each genome sequence, such that multiple longitudinal sequences from the same patient shared identical clinical metadata. Models were fit using complete-case analysis; observations with missing values for any predictor or outcome included in a given model were excluded, and no imputation was performed.</p><p>To guard against overfitting, given the small cohort, we performed 5-fold cross-validation [<xref ref-type="bibr" rid="ref55">55</xref>] using caret [<xref ref-type="bibr" rid="ref56">56</xref>] to report changes in Akaike information criterion (&#x0394;AIC) and Bayesian information criterion (&#x0394;BIC), which we calculated relative to &#x0394;=0, and the Brier score [<xref ref-type="bibr" rid="ref56">56</xref>] for binary outcomes and root mean square error (RMSE) for infection length. Because the dataset consisted of 100 genomes derived from only 34 patients, with multiple longitudinal samples contributed by some individuals, the folds were not fully independent at the patient level. Consequently, the cross-validation results should be interpreted as measures of internal consistency rather than evidence of external validity or generalizability. The reported performance metrics therefore provide an indication of model robustness within this dataset but do not constitute independent validation of predictive performance. All statistical analyses were carried out in R (version 4.4.2). Reported <italic>P</italic> values are uncorrected for multiple comparisons, and results with <italic>P</italic>&#x003C;.05 were considered exploratory rather than confirmatory.</p></sec><sec id="s2-5"><title>Ethical Considerations</title><p>All genomic and clinical metadata used in this study were obtained from publicly available, deidentified sources. No identifiable patient information was collected, accessed, or analyzed in this study. As this study involved only secondary analysis of publicly available, deidentified data, no additional institutional review board approval or informed consent was required.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Availability of Patient Metadata in GenBank</title><p>A systematic screening of 116,600 recent publications from LitCovid identified a final set of 21 articles that met all inclusion criteria for metadata enrichment (Figure S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). In 2023, a total of 71,692 publications were listed in LitCovid, of which 68% (n=48,959) of publications were open access and eligible for further analysis. In contrast, in 2024 there were only 44,908 publications with 49% (n=21,790) available as open access. After applying regular expressions to identify GenBank accessions, sequence databases, and SARS-CoV-2 variants, we considered 640 and 442 candidate articles in 2023 and 2024, respectively. Through manual review, we narrowed the list to a total of 21 articles that met all inclusion criteria (10 in 2023 and 11 in 2024) and subjected these to comprehensive metadata collection (Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><p>The studies included in our analysis predominantly sampled SARS-CoV-2 in 2022 (16/21, 76%), coinciding with the emergence and global dominance of the Omicron variant and its sublineages [<xref ref-type="bibr" rid="ref57">57</xref>]. Geographically, most studies were conducted in Asia (13/21, 62%), followed by Europe (6/21, 29%) and the Americas (2/21, 9%). Across the 21 included articles, GenBank completeness was concentrated entirely in the universally expected tier (<xref ref-type="fig" rid="figure1">Figure 1</xref>). Sample collection date and country of origin were reported in all 21 studies, and completeness across the 4 fields averaged 81%. Completeness across the 19 context-dependent fields averaged 2%, and 14 of these types, including immune status, comorbidities, treatment regimen, and clinical outcomes, were absent from every GenBank record and were available only in the article text, requiring manual enrichment.</p><p>We found a considerable discrepancy between the metadata reported in GenBank records and the metadata available in the text and <xref ref-type="supplementary-material" rid="app1">Multimedia Appendices 1 to 3</xref><xref ref-type="supplementary-material" rid="app2"/>-<xref ref-type="supplementary-material" rid="app3">3</xref>. Across all studies, there was a wide range of metadata completeness, measured by the number of metadata items extractable from our studies. Pooled across both tiers, GenBank records captured an average of only 22% of host metadata across studies, leaving more than 75% of host metadata missing. Manual enrichment increased metadata completeness by an average of 30% across studies for all metadata that could be extracted. Enrichment acted almost exclusively on the context-dependent tier, raising the average from 2% to 40%, while the universally expected tier changed from 81% to 94%. Additionally, manual enrichment recovered an average of 56% of metadata types absent from the corresponding GenBank records, although the number of recoverable metadata categories varied across studies. Demographic variables such as age (15/21, 71% of articles) and sex (13/21, 62% of articles) were primarily recovered through enrichment. However, disease outcomes (hospitalization, mortality, and symptoms) were reported less frequently: mortality was found in 38% (8/21) of articles, hospitalization in 43% (9/21), and symptoms in 48% (10/21). Reporting was the lowest for infection length (4/21, 19% of articles) and specific treatment regimens, with frequencies for oxygen therapy, antivirals, biologics, and monoclonal antibodies ranging from 9% (2/21) to 19% (4/21).</p><p>Some studies, such as Manuto_2024 (37 genomes [<xref ref-type="bibr" rid="ref33">33</xref>]) and Pavia_2024 (9 genomes [<xref ref-type="bibr" rid="ref34">34</xref>]), reported nearly all metadata types extractable in our dataset (24/25, 96% complete), representing best practices for data sharing and facilitating downstream analysis and cross-cohort comparisons. In contrast, Pe&#x00F1;as_Utrilla_2023 (6 genomes [<xref ref-type="bibr" rid="ref42">42</xref>]) baseline GenBank metadata were not improved by manual enrichment.</p></sec><sec id="s3-2"><title>Evolutionary Dynamics and Risk Predictors in Persistent SARS-CoV-2 Infections</title><p>The analyses below are exploratory. They are presented to illustrate what enriched metadata make analytically accessible, not to establish genotype-phenotype associations. Nextclade placed our 100 sequences (from 34 patients across 4 articles) into 5 distinct PANGO (Phylogenetic Assignment of Named Global Outbreak) lineages, led by BA.1 and BA.5, with contributions from Italy, Japan, and the United States (<xref ref-type="fig" rid="figure2">Figure 2A</xref>). Apart from collection date, this reflects the analytical limits of the metadata deposited in GenBank for these sequences. Using the enriched metadata available, we could then explore whether within-host evolutionary rates varied as a function of immune status and viral clade using multiple linear regression (<xref ref-type="fig" rid="figure2">Figure 2B</xref>). In this exploratory analysis, immunocompromised individuals were significantly associated with higher within-host evolutionary rates (<italic>P</italic>=.02). Among clades, BA.1 (21K) had a near-significant association with higher rates of amino acid mutations (<italic>P</italic>=.052), while the other clades showed no significant differences. We further explored whether the diversity of unique mutations differed between immunocompromised and immunocompetent patients at the nucleotide and amino acid levels, motivated by the effect of immune status on selection, infection duration, and thus the effective evolutionary substitution rate of the pathogen (<xref ref-type="fig" rid="figure2">Figure 2C</xref>). While the total number of nucleotide changes did not differ significantly between the 2 groups, immunocompromised individuals did harbor a significantly higher number of unique amino acid mutations (<italic>P</italic>=.04). Overall, we were able to explore the mutational diversity observed across genomes identifying 497 unique nucleotide mutations (299 missense, 143 synonymous, 38 intergenic, and 17 deletions), which resulted in 348 unique amino acid mutations.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Mutation burden and recurrence across immune states and viral clades. (A) Distribution of SARS-CoV-2 lineages by country. (B) Dot plot of mutation rate per patient. (C) Boxplots comparing the number of unique amino acid and nucleotide mutations between immunocompetent and immunocompromised patients; <italic>P</italic> values from Wilcoxon tests are shown. (D) Heatmap of recurrent amino acid mutations by clade. Black cells represent mutations in each clade (column), white cells represent their absence, and the adjacent color code indicates the corresponding gene for each mutation.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="bioinform_v7i1e92529_fig02.png"/></fig><p>To categorize recurrent mutations from those arising by chance, we performed a rank-frequency analysis and used an empirical binomial model to set the patient-count threshold at which a mutation is unlikely to be a chance event (Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). Applying this threshold, we identified 63 recurrent amino acid changes and 69 synonymous nucleotide mutations (<xref ref-type="fig" rid="figure2">Figure 2C</xref>). Among the recurrent mutations, we identified 2 nucleocapsid mutations R203K+G204R and the &#x0394;31&#x2010;33 deletion, both of which are hallmark mutations of the Omicron lineage [<xref ref-type="bibr" rid="ref58">58</xref>]. Additionally, we identified 2 rare mutations, I1566V (<italic>ORF1b</italic>) and P10S (<italic>ORF9b</italic>), present in less than 0.01% of global sequences [<xref ref-type="bibr" rid="ref59">59</xref>]. Every recurrent nucleotide mutation corresponded to an amino acid change, except for 2 intergenic mutations, C241T and A28271T. Notably, 35 (55%) of these recurrent amino acid mutations were localized to the spike protein: D405N, E484A, G446S, K417N, L452R, N440K, N501Y, N969K, Q493R, and S371F. While these observations are exploratory and not intended to establish clinical associations, the enriched clinical and genomic metadata provide a framework for the investigation of recurrent mutations and their potential clinical relevance.</p><p>We contextualized the amino acid mutations by first grouping them based on congruent patient-level presence and by integrating these groups with patient metadata, including immune status, CCI, and treatment regimens. To evaluate how the addition of clinical and treatment metadata improves the prediction of patient outcomes, we fit a series of nested regression models from mutation-only (equivalent to pre-enrichment analytical capabilities) to those fully adjusted for treatment regimen and patient characteristics (<xref ref-type="fig" rid="figure3">Figure 3</xref>). Within each mutation group, we subtracted the lowest AIC (or BIC) value from all models in that group, so the most supported model has &#x0394;=0 and higher values indicate weaker support. Because BIC penalizes each added covariate more heavily than AIC, a model favored by both suggests that the improvement in fit is not explained by model complexity alone. Across 23 (17 were singletons, 74%) groups, the fully adjusted models (including all covariates) demonstrated the strongest performance for &#x0394;AIC and &#x0394;BIC scores in 83% (n=19) of groups for mortality, 100% (n=23) of groups for hospitalization, and 96% (n=22) of groups for infection duration. Cross-validated performance for the fully adjusted models averaged area under the curve (AUC) of 0.93 (Brier=0.09) for mortality, AUC of 0.94 (Brier=0.09) for hospitalization, and RMSE of 19.2 for infection duration (mean across 5-folds). These performance estimates reflect internal evaluation within the cohort and should not be interpreted as evidence that the models will achieve similar performance in independent datasets. External validation in larger and more diverse cohorts will be required to assess generalizability.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Model comparison for clinical outcomes using changes in Akaike information criterion (&#x0394;AIC) and Bayesian information criterion (&#x0394;BIC). Vertical faceted plots represent the outcome tested by each model. Patient factors include age, Charlson Comorbidity Index, and immune status. Lines link mutation groups across models. The best-fitting model (lowest &#x0394;AIC or &#x0394;BIC) per mutation is highlighted in green.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="bioinform_v7i1e92529_fig03.png"/></fig><p>Using these exploratory models, we identified 2 mutation groups and 10 singletons nominally associated with the tested clinical outcomes (<xref ref-type="table" rid="table1">Table 1</xref>). Because outcomes were patient-level and the number of events was limited (6 deaths and 13 hospitalizations among 34 patients; infection duration ranged from 2 to 86 days; Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), all genotype-phenotype analyses should be interpreted as hypothesis-generating rather than as estimates of effect. For mortality, 2 groups were nominally associated: a large cluster spanning structural genes (mutations in <italic>S</italic> and <italic>N</italic>) and nonstructural genes (mutations in <italic>ORF1ab</italic> and <italic>ORF3a</italic>). The second group showing nominal association with mortality was a collection of 4 mutations, all occurring within <italic>ORF1a</italic>. Amino acid mutations nominally associated with hospitalization were primarily localized to the <italic>S</italic> gene, within the receptor-binding domain and upstream of the polybasic furin cleavage site. For infection length, only 2 mutations (G446S and T547K) were nominally associated, both within the spike protein.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Mutation groups nominally associated with clinical outcomes in the fully adjusted model (n=100)<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Outcome, group, and gene</td><td align="left" valign="bottom">Mutation</td><td align="left" valign="bottom">Effect size (95% CI)</td><td align="left" valign="bottom">Direction</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Mortality<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top">&#x2003;</td><td align="left" valign="top"/><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Group 1</td><td align="left" valign="top"/><td align="left" valign="top">6.28 (1.16&#x2010;50.2)</td><td align="left" valign="top">Higher</td><td align="left" valign="top">.03</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>N</italic></td><td align="left" valign="top">S431-</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>ORF1a</italic><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content></td><td align="left" valign="top">T3090I</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>ORF1b</italic></td><td align="left" valign="top">T2163I</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>ORF3a</italic></td><td align="left" valign="top">T223I</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>S</italic></td><td align="left" valign="top">T19I, V213G, T376A, D405N, and R408S</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Group 2</td><td align="left" valign="top"/><td align="left" valign="top">7.69 (1.49&#x2010;61.0)</td><td align="left" valign="top">Higher</td><td align="left" valign="top">.01</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>ORF1a</italic></td><td align="left" valign="top">S135R, T842I, G1307, and L3027F</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Singleton</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>ORF1a</italic></td><td align="left" valign="top">T3255I</td><td align="char" char="." valign="top">0.05 (0-0.43)</td><td align="left" valign="top">Lower</td><td align="char" char="." valign="top">.01</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>ORF1b</italic></td><td align="left" valign="top">R1315C</td><td align="left" valign="top">6.28 (1.16&#x2010;50.2)</td><td align="left" valign="top">Higher</td><td align="left" valign="top">.03</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>S</italic></td><td align="left" valign="top">Y144-</td><td align="left" valign="top">0.14 (0.02-0.71)</td><td align="left" valign="top">Lower</td><td align="left" valign="top">.01</td></tr><tr><td align="left" valign="top">Hospitalization<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top">&#x2003;</td><td align="left" valign="top"/><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Singleton</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>S</italic></td><td align="left" valign="top">S477N</td><td align="left" valign="top">40.4 (5.74&#x2010;510)</td><td align="left" valign="top">Higher</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>S</italic></td><td align="left" valign="top">E484A</td><td align="left" valign="top">5.56 (1.09&#x2010;40.2)</td><td align="left" valign="top">Higher</td><td align="left" valign="top">.04</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>S</italic></td><td align="left" valign="top">P681H</td><td align="left" valign="top">8.19 (1.62&#x2010;57.7)</td><td align="left" valign="top">Higher</td><td align="left" valign="top">.01</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>ORF1a</italic></td><td align="left" valign="top">P3395H</td><td align="left" valign="top">40.4 (5.74&#x2010;510)</td><td align="left" valign="top">Higher</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>ORF1b</italic></td><td align="left" valign="top">I1566V</td><td align="left" valign="top">16.9 (3.08&#x2010;137)</td><td align="left" valign="top">Higher</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Infection length<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top">&#x2003;</td><td align="left" valign="top"/><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Singleton</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>S</italic></td><td align="left" valign="top">T547K</td><td align="left" valign="top">15.4 (6.28&#x2010;24.5)</td><td align="left" valign="top">Longer</td><td align="left" valign="top">.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>S</italic></td><td align="left" valign="top">G446S</td><td align="left" valign="top">11.7 (2.39&#x2010;21)</td><td align="left" valign="top">Longer</td><td align="left" valign="top">.02</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Groups were retained if the mutation group had an uncorrected <italic>P</italic>&#x003C;.05. Logistic regression was used for mortality and hospitalization, and linear regression for infection duration.</p></fn><fn id="table1fn2"><p><sup>b</sup>The outcome was assessed using odds ratios as the effect measure.</p></fn><fn id="table1fn3"><p><sup>c</sup>The outcome was assessed using &#x03B2;, expressed in days, as the effect measure.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>In this study, we demonstrate that systematically enriching pathogen genomic records with sequence-specific patient metadata expands the analytical scope of downstream epidemiological investigations. Of the 116,000 articles captured by our screen in 2023 to 2024, approximately 0.02% provided readily accessible sequence-specific patient metadata. This highlights the limitations faced by scientists performing secondary data analysis, as well as deficiencies in data deposition standards, corroborating earlier reports of gaps in genomic data stewardship [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref60">60</xref>]. Through manual curation, we recovered a median of 14 additional metadata variables per sample, including age, sex, comorbidities, treatment, and clinical outcomes, increasing overall metadata completeness by an average of 30%. This enriched dataset supported 2 exploratory analyses that the GenBank records alone could not support: within-host evolutionary rate as a function of immune status, and genotype-phenotype association models incorporating clinical covariates. Collectively, we illustrate how richer host information can significantly increase the amount of usable information in sequence databases for precision public health inquiry.</p><p>The volume of metadata stored exclusively in unstructured formats, such as text, figures, and supplementary materials, presents significant challenges for scalability and data reuse. Although our manual extraction efforts partially address this gap, the approach is labor-intensive and impractical as the number of articles increases. Demographic data, such as age and sex, were the most commonly recovered enriched metadata types (relative to pre-enriched metadata), reflecting standardized sampling strategies and their importance in epidemiologic study design [<xref ref-type="bibr" rid="ref61">61</xref>]. In contrast, clinical information was rarely recoverable and, when present, was found as unstructured text or embedded in figures. This scarcity is further compounded by inconsistencies in clinical documentation and by privacy regulations enforced under the Health Insurance Portability and Accountability Act or the General Data Protection Regulation.</p><p>Addressing this gap will require both policy and technical interventions, such as enforcement of metadata deposition standards, data sharing frameworks that provide as much metadata as possible without sacrificing patient confidentiality, and NLP-based tools for metadata extraction. Traditional clinical NLP has been used to map narrative text to standardized codes with expert-level accuracy, accounting for modifiers, including negation, temporal information, and family history [<xref ref-type="bibr" rid="ref62">62</xref>]. More recently, machine learning pipelines have been developed for the large-scale detection of patient metadata from COVID-19 literature [<xref ref-type="bibr" rid="ref63">63</xref>]. Incomplete or ambiguous geospatial metadata pose significant challenges for accurate genomic epidemiology [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>]. To address this issue, frameworks have been developed that integrate uncertainty in sampling locations into phylodynamic analysis [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref><ext-link ext-link-type="uri" xlink:href="https://paperpile.com/c/Zp5v0x/ximwu+7TbmR">]. </ext-link>Additionally, NLP-based systems have been developed to automatically extract and refine geospatial data directly from scientific literature [<xref ref-type="bibr" rid="ref21">21</xref>]. Klein et al [<xref ref-type="bibr" rid="ref63">63</xref>] show that fine-tuned models pretrained on biomedical corpora, such as BiomedBERT, outperform large language models for classifying text-containing metadata. Until NLP-based tools are developed and adopted, demographic data will remain more accessible than clinical metadata, limiting population-level efforts to identify epidemiological trends.</p><p>In our case study, genomes recovered from immunocompromised patients contained more amino acid mutations than those from immunocompetent patients, though the small cohort of 34 patients limits the extent to which this comparison can be generalized. Although lineage BA.1 showed a borderline association with elevated rates, we found that, in our small cohort, overall viral clades had minimal impact on evolutionary dynamics compared to host immune status. Infections in severely immunosuppressed individuals, such as those with hematologic malignancies or organ transplants, can persist for months, effectively allowing the virus to undergo continuous adaptation under reduced immune pressure [<xref ref-type="bibr" rid="ref64">64</xref>]. This extended evolutionary window not only increases the total number of mutations but also shifts the selective pressures acting on the viral population. We observed that immunocompromised patients harbored significantly more unique amino acid mutations, while unique nucleotide changes showed no significant difference between groups. The higher rates of nonsynonymous mutations may suggest that this pattern is consistent with positive selection rather than neutral drift alone influencing the viral population [<xref ref-type="bibr" rid="ref65">65</xref>], but we did not test this selection directly.</p><p>We have shown the utility of integrating detailed clinical metadata with pathogen genomic data for exploratory evaluation of mutational dynamics and clinical outcomes. Specifically, we identified 63 mutations that had a low probability of occurring by chance and recurred across multiple patients and viral lineages. Of these, the majority were localized to the spike protein, including many known to be associated with antibody escape (D405N, E484A, G446S, K417N, L452R, N440K, N501Y, N969K, Q493R, and S371F) [<xref ref-type="bibr" rid="ref66">66</xref>-<xref ref-type="bibr" rid="ref71">71</xref>]. These patterns of convergent mutations are comparable to those noted in other studies of persistent SARS-CoV-2 infections in immunocompromised hosts [<xref ref-type="bibr" rid="ref72">72</xref>]. Separately, we also recovered 2 intergenic nucleotide mutations (C241T and A28271T), whose recurrence might suggest some potential regulatory functions that warrant further investigation.</p><p>By leveraging the increased analytical resolution provided by enriched clinical covariates, we identified exploratory associations between mortality and a pattern of mutations that spans both structural and nonstructural genes. Specifically, mutations in the <italic>S</italic> gene (T19I, V213G, T376A, D405N, and R408S) have been shown to enhance the virus&#x2019;s ability to bind and enter host cells [<xref ref-type="bibr" rid="ref66">66</xref>,<xref ref-type="bibr" rid="ref67">67</xref>]. The nonstructural mutations (T3090I, T2163I, and R1315C) are found in genes known to be involved in both disabling host innate immune signaling and controlling the kinetics of viral replication [<xref ref-type="bibr" rid="ref73">73</xref>]. These functional roles were established in experimental systems other than ours and describe why the mutations are plausible candidates; they are not evidence that the mutations influenced mortality in this cohort. Together, these mutations co-occurred with mortality in our models and may reflect combined viral and host factors (age and comorbidities) that delay or prevent the immune response, although this association will need to be explored further.</p><p>Most of the mutations we identified as being associated with hospitalization were localized to the binding domain of the spike gene. Mutation S477N is common among Omicron variants and is known to increase ACE2 affinity, suggesting that one possible explanation is enhanced viral entry, as reported in prior studies [<xref ref-type="bibr" rid="ref74">74</xref>]. The mutations associated with infection length, G446S and T547K, have been linked to altered T-cell recognition [<xref ref-type="bibr" rid="ref75">75</xref>] and viral fusion phenotypes [<xref ref-type="bibr" rid="ref76">76</xref>], respectively. While more efficient viral entry has been associated with more severe symptoms, SARS-CoV-2 persistence was associated with mutations previously linked to immune status [<xref ref-type="bibr" rid="ref72">72</xref>]. Together, these observations illustrate the potential of integrated, metadata-enriched pathogen genomics in facilitating analysis that can lead to genotype-phenotype relationships with larger cohorts.</p><p>This study illustrates the significant, yet often unrecognized, advantage of enriching sequence databases with demographic and clinical metadata. Recent audits have shown that approximately 60% of GISAID submissions are missing data [<xref ref-type="bibr" rid="ref59">59</xref>] or contain errors or ambiguities [<xref ref-type="bibr" rid="ref7">7</xref>]. Combined, these lead to misguided genotype-phenotype analyses, compromising the accuracy and reliability of epidemiological studies. Adopting structured and harmonized contextual data standards, such as those developed by the Public Health Alliance for Genomic Epidemiology, would mitigate these issues and maximize the utility of sequencing data across databases and institutions [<xref ref-type="bibr" rid="ref77">77</xref>]. These standards separate fields by requirement level, and our results indicate where each level matters. The universally expected fields are already nearly complete in GenBank and could be enforced upon required at deposition. The larger deficit lies in the context-dependent clinical fields, whose deposition is constrained both by their relevance to a given study and by patient confidentiality.</p></sec><sec id="s4-2"><title>Limitations</title><p>Our study has several limitations that should be considered when interpreting our findings. First, our literature search was restricted to open-access articles published in 2023 and 2024 with genomic data available in GenBank; this approach omits relevant articles behind paywalls and excludes genomic data that are only available in other databases such as GISAID. Additionally, our screening strategy required mentions of designated VOIs or VUMs, as defined by the WHO, to identify candidate articles for manual review. While this approach improved specificity and reduced the number of articles that required full-text assessment, it may have excluded otherwise eligible studies. Consequently, our estimate of metadata availability should be interpreted as representative of the literature captured by this screening strategy rather than an exhaustive assessment of all SARS-CoV-2 sequencing studies.</p><p>Our approach led to a dataset that was both geographically and temporally skewed toward studies conducted in Asia (13/21, 62%) and in 2022 (16/21, 76%), which limits the broader applicability of our findings to other regions, populations, and phases of the pandemic. Furthermore, despite combining data across 4 articles, our analysis included only 34 patients. We did not adjust for study source, calendar period, or geographic origin because they were correlated with immune status and treatment availability across the 4 studies. In addition, many mutation groups were rare, with 74% (17/23) observed in only a single patient. As a result, coefficient estimates may be unstable, and intervals should be interpreted cautiously. Although 5-fold cross-validation was used to evaluate internal model consistency, it does not provide independent validation and may overestimate predictive performance in small datasets. Finally, while manual enrichment was effective, the labor-intensive process may introduce inconsistencies in metadata collection, reinforcing the critical need for standardized, structured, and accessible deposition of patient metadata alongside genomic sequences.</p></sec><sec id="s4-3"><title>Conclusions</title><p>Here, we show the importance of enriched host metadata in understanding both viral evolution and clinical dynamics. Linking pathogen genomic data with patient characteristics provides a more comprehensive picture of disease behavior. There have been few large-scale SARS-CoV-2 sequencing studies that have incorporated rich patient metadata, largely because most publicly available genomic sequences contain limited clinical information. Our research directly quantifies this problem and shows how improvements in metadata availability strengthen pathogen genomics. Incorporating enriched metadata has facilitated the identification of genotypic and phenotypic patterns that sequence data alone cannot demonstrate. Comprehensive patient metadata can provide insights into the interplay among viral mutations and clinical outcomes, offering a more complete understanding of viral evolutionary behavior within populations.</p></sec></sec></body><back><ack><p>The authors used Arizona State University&#x2019;s institutional license for ChatGPT (OpenAI [<xref ref-type="bibr" rid="ref78">78</xref>]), including ChatGPT (GPT-4) and ChatGPT (GPT-5) models, to support Python and R code generation for data analysis and visualization, including the creation of graphs and plots.</p></ack><notes><sec><title>Funding</title><p>Research reported in this publication was supported by the National Institute of Allergy and Infectious Diseases of the National Institutes of Health under Award Number R01AI164481 to GG-H and MS. The content is solely the responsibility of the authors and does not necessarily represent the official views of the National Institutes of Health.</p></sec><sec><title>Data Availability</title><p>The metadata-enriched genomic dataset analyzed in this study is available in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref> and is also dynamically available through the HLP Gonzalez Lab [<xref ref-type="bibr" rid="ref79">79</xref>]. Most genomes are publicly available in the National Center for Biotechnology Information database, and their corresponding accession IDs are provided in the &#x201C;genbank&#x201D; column. This dataset also includes 167 genomes for which raw sequencing data were obtained from the Sequence Read Archive (SRA) and assembled (as described in the &#x201C;Methods&#x201D; section). The raw SRA accession numbers are also stored in the &#x201C;genbank&#x201D; column. Of those SRA-assembled genomes, 81 are deposited in the Global Initiative on Sharing All Influenza Data (GISAID) database and their GISAID accession IDs are provided in the &#x201C;id&#x201D; column. All code used for data processing, analysis, and figure generation in this study is publicly available through Zenodo [<xref ref-type="bibr" rid="ref80">80</xref>].</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: MP, GG-H, MS</p><p>Data curation: MP, KO</p><p>Formal analysis: MP, KO, MS</p><p>Funding acquisition: GG-H, MS</p><p>Methodology: MP, KO, MS</p><p>Project administration: GG-H, MS</p><p>Resources: KO, GG-H</p><p>Supervision: GG-H, MS</p><p>Validation: MP, MS</p><p>Visualization: MP</p><p>Writing &#x2013; original draft: MP, MS</p><p>Writing &#x2013; review &#x0026; editing: MP, KO, GG-H, MS</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1"><italic>ACE2</italic></term><def><p>angiotensin-converting enzyme 2</p></def></def-item><def-item><term id="abb2">AIC</term><def><p>Akaike information criterion</p></def></def-item><def-item><term id="abb3">AUC</term><def><p>area under the curve</p></def></def-item><def-item><term id="abb4">BIC</term><def><p>Bayesian information criterion</p></def></def-item><def-item><term id="abb5">BLAST</term><def><p>Basic Local Alignment Search Tool</p></def></def-item><def-item><term id="abb6">CCI</term><def><p>Charlson Comorbidity Index</p></def></def-item><def-item><term id="abb7">GISAID</term><def><p>Global Initiative on Sharing All Influenza Data</p></def></def-item><def-item><term id="abb8">IRMA</term><def><p>iterative refinement meta-assembler</p></def></def-item><def-item><term id="abb9">NCBI</term><def><p>National Center for Biotechnology Information</p></def></def-item><def-item><term id="abb10">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb11">PANGO</term><def><p>Phylogenetic Assignment of Named Global Outbreak</p></def></def-item><def-item><term id="abb12">PMC</term><def><p>PubMed Central</p></def></def-item><def-item><term id="abb13">RMSE</term><def><p>root mean square error</p></def></def-item><def-item><term id="abb14">SRA</term><def><p>Sequence Read Archive</p></def></def-item><def-item><term id="abb15"><italic>TMPRSS2</italic></term><def><p>transmembrane serine protease 2</p></def></def-item><def-item><term id="abb16">VOI</term><def><p>variants of interest</p></def></def-item><def-item><term id="abb17">VUM</term><def><p>variants under monitoring</p></def></def-item><def-item><term id="abb18">WHO</term><def><p>World Health Organization</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sayers</surname><given-names>EW</given-names> </name><name name-style="western"><surname>Bolton</surname><given-names>EE</given-names> </name><name name-style="western"><surname>Brister</surname><given-names>JR</given-names> </name><etal/></person-group><article-title>Database resources of the national center for biotechnology information</article-title><source>Nucleic Acids Res</source><year>2022</year><month>01</month><day>7</day><volume>50</volume><issue>D1</issue><fpage>D20</fpage><lpage>D26</lpage><pub-id pub-id-type="doi">10.1093/nar/gkab1112</pub-id><pub-id pub-id-type="medline">34850941</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>McCauley</surname><given-names>J</given-names> </name></person-group><article-title>GISAID: global initiative on sharing all influenza data - from vision to reality</article-title><source>Euro Surveill</source><year>2017</year><month>03</month><day>30</day><volume>22</volume><issue>13</issue><fpage>30494</fpage><pub-id pub-id-type="doi">10.2807/1560-7917.ES.2017.22.13.30494</pub-id><pub-id pub-id-type="medline">28382917</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cai</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Lavine</surname><given-names>CL</given-names> </name><etal/></person-group><article-title>Structural and functional impact by SARS-CoV-2 Omicron spike mutations</article-title><source>Cell Rep</source><year>2022</year><month>04</month><day>26</day><volume>39</volume><issue>4</issue><fpage>110729</fpage><pub-id pub-id-type="doi">10.1016/j.celrep.2022.110729</pub-id><pub-id pub-id-type="medline">35452593</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Scotch</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lauer</surname><given-names>K</given-names> </name><name name-style="western"><surname>Wieben</surname><given-names>ED</given-names> </name><etal/></person-group><article-title>Genomic epidemiology reveals the dominance of Hennepin County in the transmission of SARS-CoV-2 in Minnesota from 2020 to 2022</article-title><source>mSphere</source><year>2023</year><month>12</month><day>20</day><volume>8</volume><issue>6</issue><fpage>e0023223</fpage><pub-id pub-id-type="doi">10.1128/msphere.00232-23</pub-id><pub-id pub-id-type="medline">37882516</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Grimaldi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Panariello</surname><given-names>F</given-names> </name><name name-style="western"><surname>Annunziata</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Improved SARS-CoV-2 sequencing surveillance allows the identification of new variants and signatures in infected patients</article-title><source>Genome Med</source><year>2022</year><month>08</month><day>12</day><volume>14</volume><issue>1</issue><fpage>90</fpage><pub-id pub-id-type="doi">10.1186/s13073-022-01098-8</pub-id><pub-id pub-id-type="medline">35962405</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schriml</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Chuvochina</surname><given-names>M</given-names> </name><name name-style="western"><surname>Davies</surname><given-names>N</given-names> </name><etal/></person-group><article-title>COVID-19 pandemic reveals the peril of ignoring metadata standards</article-title><source>Sci Data</source><year>2020</year><month>06</month><day>19</day><volume>7</volume><issue>1</issue><fpage>188</fpage><pub-id pub-id-type="doi">10.1038/s41597-020-0524-5</pub-id><pub-id pub-id-type="medline">32561801</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gozashti</surname><given-names>L</given-names> </name><name name-style="western"><surname>Corbett-Detig</surname><given-names>R</given-names> </name></person-group><article-title>Shortcomings of SARS-CoV-2 genomic metadata</article-title><source>BMC Res Notes</source><year>2021</year><month>05</month><day>17</day><volume>14</volume><issue>1</issue><fpage>189</fpage><pub-id pub-id-type="doi">10.1186/s13104-021-05605-9</pub-id><pub-id pub-id-type="medline">34001211</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sintchenko</surname><given-names>V</given-names> </name><name name-style="western"><surname>Sim</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Suster</surname><given-names>CJE</given-names> </name></person-group><article-title>Estimating the deferred value of pathogen genomic data for secondary use</article-title><source>Sci Data</source><year>2025</year><month>05</month><day>13</day><volume>12</volume><issue>1</issue><fpage>784</fpage><pub-id pub-id-type="doi">10.1038/s41597-025-05049-x</pub-id><pub-id pub-id-type="medline">40360501</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>SW</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>SF</given-names> </name></person-group><article-title>SARS-CoV-2 entry related viral and host genetic variations: implications on COVID-19 severity, immune escape, and infectivity</article-title><source>Int J Mol Sci</source><year>2021</year><month>03</month><day>17</day><volume>22</volume><issue>6</issue><fpage>3060</fpage><pub-id pub-id-type="doi">10.3390/ijms22063060</pub-id><pub-id pub-id-type="medline">33802729</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Khongsiri</surname><given-names>W</given-names> </name><name name-style="western"><surname>Poolchanuan</surname><given-names>P</given-names> </name><name name-style="western"><surname>Dulsuk</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Associations between clinical data, vaccination status, antibody responses, and post-COVID-19 symptoms in Thais infected with SARS-CoV-2 Delta and Omicron variants: a 1-year follow-up study</article-title><source>BMC Infect Dis</source><year>2024</year><month>10</month><day>7</day><volume>24</volume><issue>1</issue><fpage>1116</fpage><pub-id pub-id-type="doi">10.1186/s12879-024-09999-2</pub-id><pub-id pub-id-type="medline">39375604</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ling-Hu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Simons</surname><given-names>LM</given-names> </name><name name-style="western"><surname>Dean</surname><given-names>TJ</given-names> </name><etal/></person-group><article-title>Integration of individualized and population-level molecular epidemiology data to model COVID-19 outcomes</article-title><source>Cell Rep Med</source><year>2024</year><month>01</month><day>16</day><volume>5</volume><issue>1</issue><fpage>101361</fpage><pub-id pub-id-type="doi">10.1016/j.xcrm.2023.101361</pub-id><pub-id pub-id-type="medline">38232695</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mart&#x00ED;nez-Martinez</surname><given-names>AB</given-names> </name><name name-style="western"><surname>Tristancho-Bar&#x00F3;</surname><given-names>A</given-names> </name><name name-style="western"><surname>Garcia-Rodriguez</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Impact of obesity-associated SARS-CoV-2 mutations on COVID-19 severity and clinical outcomes</article-title><source>Viruses</source><year>2024</year><month>12</month><day>30</day><volume>17</volume><issue>1</issue><fpage>38</fpage><pub-id pub-id-type="doi">10.3390/v17010038</pub-id><pub-id pub-id-type="medline">39861827</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Patel</surname><given-names>M</given-names> </name><name name-style="western"><surname>Shamim</surname><given-names>U</given-names> </name><name name-style="western"><surname>Umang</surname><given-names>U</given-names> </name><name name-style="western"><surname>Pandey</surname><given-names>R</given-names> </name><name name-style="western"><surname>Narayan</surname><given-names>J</given-names> </name></person-group><article-title>SARS-CoV-2 alchemy: understanding the dynamics of age, vaccination, and geography in the evolution of SARS-CoV-2 in India</article-title><source>PLoS Negl Trop Dis</source><year>2025</year><month>03</month><volume>19</volume><issue>3</issue><fpage>e0012918</fpage><pub-id pub-id-type="doi">10.1371/journal.pntd.0012918</pub-id><pub-id pub-id-type="medline">40063870</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Feng</surname><given-names>S</given-names> </name><name name-style="western"><surname>Reid</surname><given-names>GE</given-names> </name><name name-style="western"><surname>Clark</surname><given-names>NM</given-names> </name><name name-style="western"><surname>Harrington</surname><given-names>A</given-names> </name><name name-style="western"><surname>Uprichard</surname><given-names>SL</given-names> </name><name name-style="western"><surname>Baker</surname><given-names>SC</given-names> </name></person-group><article-title>Evidence of SARS-CoV-2 convergent evolution in immunosuppressed patients treated with antiviral therapies</article-title><source>Virol J</source><year>2024</year><month>05</month><day>7</day><volume>21</volume><issue>1</issue><fpage>105</fpage><pub-id pub-id-type="doi">10.1186/s12985-024-02378-y</pub-id><pub-id pub-id-type="medline">38715113</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Larsen</surname><given-names>JR</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>MR</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>JD</given-names> </name><name name-style="western"><surname>Hicks</surname><given-names>JB</given-names> </name><name name-style="western"><surname>Kuhn</surname><given-names>P</given-names> </name></person-group><article-title>Modeling the onset of symptoms of COVID-19: effects of SARS-CoV-2 variant</article-title><source>PLOS Comput Biol</source><year>2021</year><month>12</month><volume>17</volume><issue>12</issue><fpage>e1009629</fpage><pub-id pub-id-type="doi">10.1371/journal.pcbi.1009629</pub-id><pub-id pub-id-type="medline">34914688</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Azman</surname><given-names>AS</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Global landscape of SARS-CoV-2 genomic surveillance and data sharing</article-title><source>Nat Genet</source><year>2022</year><month>04</month><volume>54</volume><issue>4</issue><fpage>499</fpage><lpage>507</lpage><pub-id pub-id-type="doi">10.1038/s41588-022-01033-y</pub-id><pub-id pub-id-type="medline">35347305</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tahsin</surname><given-names>T</given-names> </name><name name-style="western"><surname>Beard</surname><given-names>R</given-names> </name><name name-style="western"><surname>Rivera</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Natural language processing methods for enhancing geographic metadata for phylogeography of zoonotic viruses</article-title><source>AMIA Jt Summits Transl Sci Proc</source><year>2014</year><volume>2014</volume><fpage>102</fpage><lpage>111</lpage><pub-id pub-id-type="medline">25717409</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Scotch</surname><given-names>M</given-names> </name><name name-style="western"><surname>Sarkar</surname><given-names>IN</given-names> </name><name name-style="western"><surname>Mei</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Enhancing phylogeography by improving geographical information from GenBank</article-title><source>J Biomed Inform</source><year>2011</year><month>12</month><volume>44 Suppl 1</volume><issue>Suppl 1</issue><fpage>S44</fpage><lpage>S47</lpage><pub-id pub-id-type="doi">10.1016/j.jbi.2011.06.005</pub-id><pub-id pub-id-type="medline">21723960</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vaiente</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Scotch</surname><given-names>M</given-names> </name></person-group><article-title>Going back to the roots: evaluating Bayesian phylogeographic models with discrete trait uncertainty</article-title><source>Infect Genet Evol</source><year>2020</year><month>11</month><volume>85</volume><fpage>104501</fpage><pub-id pub-id-type="doi">10.1016/j.meegid.2020.104501</pub-id><pub-id pub-id-type="medline">32798768</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Scotch</surname><given-names>M</given-names> </name><name name-style="western"><surname>Tahsin</surname><given-names>T</given-names> </name><name name-style="western"><surname>Weissenbacher</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Incorporating sampling uncertainty in the geospatial assignment of taxa for virus phylogeography</article-title><source>Virus Evol</source><year>2019</year><month>01</month><volume>5</volume><issue>1</issue><fpage>vey043</fpage><pub-id pub-id-type="doi">10.1093/ve/vey043</pub-id><pub-id pub-id-type="medline">30838129</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Weissenbacher</surname><given-names>D</given-names> </name><name name-style="western"><surname>Tahsin</surname><given-names>T</given-names> </name><name name-style="western"><surname>Beard</surname><given-names>R</given-names> </name><etal/></person-group><article-title>Knowledge-driven geospatial location resolution for phylogeographic models of virus migration</article-title><source>Bioinformatics</source><year>2015</year><month>06</month><day>15</day><volume>31</volume><issue>12</issue><fpage>i348</fpage><lpage>56</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btv259</pub-id><pub-id pub-id-type="medline">26072502</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Allot</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>Z</given-names> </name></person-group><article-title>LitCovid: an open database of COVID-19 literature</article-title><source>Nucleic Acids Res</source><year>2021</year><month>01</month><day>8</day><volume>49</volume><issue>D1</issue><fpage>D1534</fpage><lpage>D1540</lpage><pub-id pub-id-type="doi">10.1093/nar/gkaa952</pub-id><pub-id pub-id-type="medline">33166392</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schuler</surname><given-names>GD</given-names> </name><name name-style="western"><surname>Epstein</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Ohkawa</surname><given-names>H</given-names> </name><name name-style="western"><surname>Kans</surname><given-names>JA</given-names> </name></person-group><article-title>Entrez: molecular biology database and retrieval system</article-title><source>Methods Enzymol</source><year>1996</year><volume>266</volume><fpage>141</fpage><lpage>162</lpage><pub-id pub-id-type="doi">10.1016/s0076-6879(96)66012-1</pub-id><pub-id pub-id-type="medline">8743683</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Charlson</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Pompei</surname><given-names>P</given-names> </name><name name-style="western"><surname>Ales</surname><given-names>KL</given-names> </name><name name-style="western"><surname>MacKenzie</surname><given-names>CR</given-names> </name></person-group><article-title>A new method of classifying prognostic comorbidity in longitudinal studies: development and validation</article-title><source>J Chronic Dis</source><year>1987</year><volume>40</volume><issue>5</issue><fpage>373</fpage><lpage>383</lpage><pub-id pub-id-type="doi">10.1016/0021-9681(87)90171-8</pub-id><pub-id pub-id-type="medline">3558716</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>El-Sappagh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Franda</surname><given-names>F</given-names> </name><name name-style="western"><surname>Ali</surname><given-names>F</given-names> </name><name name-style="western"><surname>Kwak</surname><given-names>KS</given-names> </name></person-group><article-title>SNOMED CT standard ontology based on the ontology for general medical science</article-title><source>BMC Med Inform Decis Mak</source><year>2018</year><month>08</month><day>31</day><volume>18</volume><issue>1</issue><fpage>76</fpage><pub-id pub-id-type="doi">10.1186/s12911-018-0651-5</pub-id><pub-id pub-id-type="medline">30170591</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Kans</surname><given-names>J</given-names> </name></person-group><article-title>Entrez Direct: E-utilities on the Unix command line</article-title><source>Entrez&#x00AE; Programming Utilities Help</source><year>2024</year><access-date>2026-09-12</access-date><publisher-name>National Center for Biotechnology Information (US)</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/books/NBK179288/">https://www.ncbi.nlm.nih.gov/books/NBK179288/</ext-link></comment></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>S</given-names> </name></person-group><article-title>Ultrafast one-pass FASTQ data preprocessing, quality control, and deduplication using fastp</article-title><source>Imeta</source><year>2023</year><month>05</month><volume>2</volume><issue>2</issue><fpage>e107</fpage><pub-id pub-id-type="doi">10.1002/imt2.107</pub-id><pub-id pub-id-type="medline">38868435</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shepard</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Meno</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bahl</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wilson</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Barnes</surname><given-names>J</given-names> </name><name name-style="western"><surname>Neuhaus</surname><given-names>E</given-names> </name></person-group><article-title>Viral deep sequencing needs an adaptive approach: IRMA, the iterative refinement meta-assembler</article-title><source>BMC Genomics</source><year>2016</year><month>09</month><day>5</day><volume>17</volume><issue>1</issue><fpage>708</fpage><pub-id pub-id-type="doi">10.1186/s12864-016-3030-6</pub-id><pub-id pub-id-type="medline">27595578</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Camacho</surname><given-names>C</given-names> </name><name name-style="western"><surname>Coulouris</surname><given-names>G</given-names> </name><name name-style="western"><surname>Avagyan</surname><given-names>V</given-names> </name><etal/></person-group><article-title>BLAST+: architecture and applications</article-title><source>BMC Bioinformatics</source><year>2009</year><month>12</month><day>15</day><volume>10</volume><fpage>421</fpage><pub-id pub-id-type="doi">10.1186/1471-2105-10-421</pub-id><pub-id pub-id-type="medline">20003500</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aksamentov</surname><given-names>I</given-names> </name><name name-style="western"><surname>Roemer</surname><given-names>C</given-names> </name><name name-style="western"><surname>Hodcroft</surname><given-names>EB</given-names> </name><name name-style="western"><surname>Neher</surname><given-names>RA</given-names> </name></person-group><article-title>Nextclade: clade assignment, mutation calling and quality control for viral genomes</article-title><source>JOSS</source><year>2021</year><month>11</month><day>30</day><volume>6</volume><issue>67</issue><fpage>3773</fpage><pub-id pub-id-type="doi">10.21105/joss.03773</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gonzalez-Reiche</surname><given-names>AS</given-names> </name><name name-style="western"><surname>Alshammary</surname><given-names>H</given-names> </name><name name-style="western"><surname>Schaefer</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Sequential intrahost evolution and onward transmission of SARS-CoV-2 variants</article-title><source>Nat Commun</source><year>2023</year><month>06</month><day>3</day><volume>14</volume><issue>1</issue><fpage>3235</fpage><pub-id pub-id-type="doi">10.1038/s41467-023-38867-x</pub-id><pub-id pub-id-type="medline">37270625</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Igari</surname><given-names>H</given-names> </name><name name-style="western"><surname>Sakao</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ishige</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Dynamic diversity of SARS-CoV-2 genetic mutations in a lung transplantation patient with persistent COVID-19</article-title><source>Nat Commun</source><year>2024</year><month>04</month><day>29</day><volume>15</volume><issue>1</issue><fpage>3604</fpage><pub-id pub-id-type="doi">10.1038/s41467-024-47941-x</pub-id><pub-id pub-id-type="medline">38684722</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Manuto</surname><given-names>L</given-names> </name><name name-style="western"><surname>Bado</surname><given-names>M</given-names> </name><name name-style="western"><surname>Cola</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Immune system deficiencies do not alter SARS-CoV-2 evolutionary rate but favour the emergence of mutations by extending viral persistence</article-title><source>Viruses</source><year>2024</year><month>03</month><day>13</day><volume>16</volume><issue>3</issue><fpage>447</fpage><pub-id pub-id-type="doi">10.3390/v16030447</pub-id><pub-id pub-id-type="medline">38543811</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pavia</surname><given-names>G</given-names> </name><name name-style="western"><surname>Quirino</surname><given-names>A</given-names> </name><name name-style="western"><surname>Marascio</surname><given-names>N</given-names> </name><etal/></person-group><article-title>Persistence of SARS-CoV-2 infection and viral intra- and inter-host evolution in COVID-19 hospitalized patients</article-title><source>J Med Virol</source><year>2024</year><month>06</month><volume>96</volume><issue>6</issue><fpage>e29708</fpage><pub-id pub-id-type="doi">10.1002/jmv.29708</pub-id><pub-id pub-id-type="medline">38804179</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sayama</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Sakagami</surname><given-names>A</given-names> </name><name name-style="western"><surname>Okamoto</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Identification of various recombinants in a patient coinfected with the different SARS-CoV-2 variants</article-title><source>Influenza Other Respir Viruses</source><year>2024</year><month>06</month><volume>18</volume><issue>6</issue><fpage>e13340</fpage><pub-id pub-id-type="doi">10.1111/irv.13340</pub-id><pub-id pub-id-type="medline">38890805</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>B</given-names> </name><name name-style="western"><surname>Oyama</surname><given-names>R</given-names> </name><name name-style="western"><surname>Tabe</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Investigation of the individual genetic evolution of SARS-CoV-2 in a small cluster during the rapid spread of the BF.5 lineage in Tokyo, Japan</article-title><source>Front Microbiol</source><year>2023</year><volume>14</volume><fpage>1229234</fpage><pub-id pub-id-type="doi">10.3389/fmicb.2023.1229234</pub-id><pub-id pub-id-type="medline">37744926</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Ng</surname><given-names>RWY</given-names> </name><name name-style="western"><surname>Lui</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Quantitative and qualitative subgenomic RNA profiles of SARS-CoV-2 in respiratory samples: a comparison between Omicron BA.2 and non-VOC-D614G</article-title><source>Virol Sin</source><year>2024</year><month>04</month><volume>39</volume><issue>2</issue><fpage>218</fpage><lpage>227</lpage><pub-id pub-id-type="doi">10.1016/j.virs.2024.01.010</pub-id><pub-id pub-id-type="medline">38316363</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jony</surname><given-names>MHK</given-names> </name><name name-style="western"><surname>Alam</surname><given-names>AN</given-names> </name><name name-style="western"><surname>Nasif</surname><given-names>MAO</given-names> </name><etal/></person-group><article-title>Emergence of SARS-CoV-2 Omicron sub-lineage JN.1 in Bangladesh</article-title><source>Microbiol Resour Announc</source><year>2024</year><month>06</month><day>11</day><volume>13</volume><issue>6</issue><fpage>e0013024</fpage><pub-id pub-id-type="doi">10.1128/mra.00130-24</pub-id><pub-id pub-id-type="medline">38651907</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>LT</given-names> </name><name name-style="western"><surname>Chiou</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>PC</given-names> </name><etal/></person-group><article-title>Epidemiology and analysis of SARS-CoV-2 Omicron subvariants BA.1 and 2 in Taiwan</article-title><source>Sci Rep</source><year>2023</year><month>10</month><day>3</day><volume>13</volume><issue>1</issue><fpage>16583</fpage><pub-id pub-id-type="doi">10.1038/s41598-023-43357-7</pub-id><pub-id pub-id-type="medline">37789031</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Deng</surname><given-names>P</given-names> </name><name name-style="western"><surname>He</surname><given-names>J</given-names> </name><etal/></person-group><article-title>A regional genomic surveillance program is implemented to monitor the occurrence and emergence of SARS-CoV-2 variants in Yubei District, China</article-title><source>Virol J</source><year>2024</year><month>01</month><day>8</day><volume>21</volume><issue>1</issue><fpage>13</fpage><pub-id pub-id-type="doi">10.1186/s12985-023-02279-6</pub-id><pub-id pub-id-type="medline">38191416</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Misra</surname><given-names>G</given-names> </name><name name-style="western"><surname>Manzoor</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chopra</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Genomic epidemiology of SARS-CoV-2 from Uttar Pradesh, India</article-title><source>Sci Rep</source><year>2023</year><month>09</month><day>8</day><volume>13</volume><issue>1</issue><fpage>14847</fpage><pub-id pub-id-type="doi">10.1038/s41598-023-42065-6</pub-id><pub-id pub-id-type="medline">37684328</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pe&#x00F1;as-Utrilla</surname><given-names>D</given-names> </name><name name-style="western"><surname>Sanz</surname><given-names>A</given-names> </name><name name-style="western"><surname>Catal&#x00E1;n</surname><given-names>P</given-names> </name><etal/></person-group><article-title>A mutation responsible for impaired detection by the Xpert SARS-CoV-2 assay independently emerged in different lineages during the SARS-CoV-2 pandemic</article-title><source>BMC Microbiol</source><year>2023</year><month>07</month><day>17</day><volume>23</volume><issue>1</issue><fpage>190</fpage><pub-id pub-id-type="doi">10.1186/s12866-023-02924-8</pub-id><pub-id pub-id-type="medline">37460980</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Perez-Florido</surname><given-names>J</given-names> </name><name name-style="western"><surname>Casimiro-Soriguer</surname><given-names>CS</given-names> </name><name name-style="western"><surname>Ortu&#x00F1;o</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Detection of high level of co-infection and the emergence of novel SARS CoV-2 Delta-Omicron and Omicron-Omicron recombinants in the epidemiological surveillance of Andalusia</article-title><source>Int J Mol Sci</source><year>2023</year><month>01</month><day>26</day><volume>24</volume><issue>3</issue><fpage>2419</fpage><pub-id pub-id-type="doi">10.3390/ijms24032419</pub-id><pub-id pub-id-type="medline">36768752</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>de Prost</surname><given-names>N</given-names> </name><name name-style="western"><surname>Audureau</surname><given-names>E</given-names> </name><name name-style="western"><surname>Pr&#x00E9;au</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Clinical phenotypes and outcomes associated with SARS-CoV-2 Omicron variants BA.2, BA.5 and BQ.1.1 in critically ill patients with COVID-19: a prospective, multicenter cohort study</article-title><source>Intensive Care Med Exp</source><year>2023</year><month>08</month><day>7</day><volume>11</volume><issue>1</issue><fpage>48</fpage><pub-id pub-id-type="doi">10.1186/s40635-023-00536-0</pub-id><pub-id pub-id-type="medline">37544942</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Selvavinayagam</surname><given-names>ST</given-names> </name><name name-style="western"><surname>Karishma</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Hemashree</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Clinical characteristics and novel mutations of omicron subvariant XBB in Tamil Nadu, India - a cohort study</article-title><source>Lancet Reg Health Southeast Asia</source><year>2023</year><month>12</month><volume>19</volume><fpage>100272</fpage><pub-id pub-id-type="doi">10.1016/j.lansea.2023.100272</pub-id><pub-id pub-id-type="medline">38076717</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Singh</surname><given-names>P</given-names> </name><name name-style="western"><surname>Sharma</surname><given-names>K</given-names> </name><name name-style="western"><surname>Bhargava</surname><given-names>A</given-names> </name><name name-style="western"><surname>Negi</surname><given-names>SS</given-names> </name></person-group><article-title>Genomic characterization of Influenza A (H1N1)pdm09 and SARS-CoV-2 from influenza like illness (ILI) and severe acute respiratory illness (SARI) cases reported between July-December, 2022</article-title><source>Sci Rep</source><year>2024</year><month>05</month><day>9</day><volume>14</volume><issue>1</issue><fpage>10660</fpage><pub-id pub-id-type="doi">10.1038/s41598-024-58993-w</pub-id><pub-id pub-id-type="medline">38724525</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Taboada</surname><given-names>BI</given-names> </name><name name-style="western"><surname>Z&#x00E1;rate</surname><given-names>S</given-names> </name><name name-style="western"><surname>Garc&#x00ED;a-L&#x00F3;pez</surname><given-names>R</given-names> </name><etal/></person-group><article-title>SARS-CoV-2 Omicron variants BA.4 and BA.5 dominated the fifth COVID-19 epidemiological wave in Mexico</article-title><source>Microb Genom</source><year>2023</year><month>12</month><volume>9</volume><issue>12</issue><fpage>001120</fpage><pub-id pub-id-type="doi">10.1099/mgen.0.001120</pub-id><pub-id pub-id-type="medline">38112714</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tahsin</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hasan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Rahman</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Coding-complete genomes of 18 SARS-CoV-2 Omicron JN.1, JN.1.4, and JN.1.11 sub-lineages in Bangladesh</article-title><source>Microbiol Resour Announc</source><year>2024</year><month>06</month><day>11</day><volume>13</volume><issue>6</issue><fpage>e0013524</fpage><pub-id pub-id-type="doi">10.1128/mra.00135-24</pub-id><pub-id pub-id-type="medline">38656213</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tsai</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Chiou</surname><given-names>SS</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>PC</given-names> </name><etal/></person-group><article-title>The epidemiology and phylogenetic trends of Omicron subvariants from BA.5 to XBB.1 in Taiwan</article-title><source>J Infect Public Health</source><year>2024</year><month>11</month><volume>17</volume><issue>11</issue><fpage>102556</fpage><pub-id pub-id-type="doi">10.1016/j.jiph.2024.102556</pub-id><pub-id pub-id-type="medline">39388868</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ulhuq</surname><given-names>FR</given-names> </name><name name-style="western"><surname>Barge</surname><given-names>M</given-names> </name><name name-style="western"><surname>Falconer</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Analysis of the ARTIC V4 and V4.1 SARS-CoV-2 primers and their impact on the detection of Omicron BA.1 and BA.2 lineage-defining mutations</article-title><source>Microb Genom</source><year>2023</year><month>04</month><volume>9</volume><issue>4</issue><fpage>mgen000991</fpage><pub-id pub-id-type="doi">10.1099/mgen.0.000991</pub-id><pub-id pub-id-type="medline">37083576</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhao</surname><given-names>N</given-names> </name><name name-style="western"><surname>He</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Genomic epidemiology reveals the variation and transmission properties of SARS-CoV-2 in a single-source community outbreak</article-title><source>Virus Evol</source><year>2024</year><volume>10</volume><issue>1</issue><fpage>veae085</fpage><pub-id pub-id-type="doi">10.1093/ve/veae085</pub-id><pub-id pub-id-type="medline">39493536</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Minh</surname><given-names>BQ</given-names> </name><name name-style="western"><surname>Schmidt</surname><given-names>HA</given-names> </name><name name-style="western"><surname>Chernomor</surname><given-names>O</given-names> </name><etal/></person-group><article-title>IQ-TREE 2: new models and efficient methods for phylogenetic inference in the genomic era</article-title><source>Mol Biol Evol</source><year>2020</year><month>05</month><day>1</day><volume>37</volume><issue>5</issue><fpage>1530</fpage><lpage>1534</lpage><pub-id pub-id-type="doi">10.1093/molbev/msaa015</pub-id><pub-id pub-id-type="medline">32011700</pub-id></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Volz</surname><given-names>EM</given-names> </name><name name-style="western"><surname>Frost</surname><given-names>SDW</given-names> </name></person-group><article-title>Scalable relaxed clock phylogenetic dating</article-title><source>Virus Evol</source><year>2017</year><month>07</month><volume>3</volume><issue>2</issue><pub-id pub-id-type="doi">10.1093/ve/vex025</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Benjamini</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Hochberg</surname><given-names>Y</given-names> </name></person-group><article-title>Controlling the false discovery rate: a practical and powerful approach to multiple testing</article-title><source>J R Stat Soc Series B Stat Methodol</source><year>1995</year><month>01</month><day>1</day><volume>57</volume><issue>1</issue><fpage>289</fpage><lpage>300</lpage><pub-id pub-id-type="doi">10.1111/j.2517-6161.1995.tb02031.x</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Kohavi</surname><given-names>R</given-names> </name></person-group><article-title>A study of cross-validation and bootstrap for accuracy estimation and model selection</article-title><source>IJCAI&#x2019;95: Proceedings of the 14th International Joint Conference on Artificial Intelligence - Volume 2</source><year>1995</year><access-date>2026-09-12</access-date><publisher-name>Morgan Kaufmann Publishers</publisher-name><fpage>1137</fpage><lpage>1143</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/10.5555/1643031.1643047">https://dl.acm.org/doi/10.5555/1643031.1643047</ext-link></comment></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kuhn</surname><given-names>M</given-names> </name></person-group><article-title>Building predictive models in R using the caret package</article-title><source>J Stat Soft</source><year>2008</year><volume>28</volume><issue>5</issue><fpage>1</fpage><lpage>26</lpage><pub-id pub-id-type="doi">10.18637/jss.v028.i05</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hyug Choi</surname><given-names>J</given-names> </name><name name-style="western"><surname>Sook Jun</surname><given-names>M</given-names> </name><name name-style="western"><surname>Yong Jeon</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Global lineage evolution pattern of SARS-CoV-2 in Africa, America, Europe, and Asia: a comparative analysis of variant clusters and their relevance across continents</article-title><source>J Transl Int Med</source><year>2023</year><month>12</month><volume>11</volume><issue>4</issue><fpage>410</fpage><lpage>422</lpage><pub-id pub-id-type="doi">10.2478/jtim-2023-0118</pub-id><pub-id pub-id-type="medline">38130632</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nguyen</surname><given-names>A</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>H</given-names> </name><name name-style="western"><surname>Myagmarsuren</surname><given-names>D</given-names> </name></person-group><article-title>Modulation of biophysical properties of nucleocapsid protein in the mutant spectrum of SARS-CoV-2</article-title><source>Elife</source><year>2024</year><volume>13</volume><pub-id pub-id-type="doi">10.7554/eLife.94836</pub-id><pub-id pub-id-type="medline">38941236</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tzou</surname><given-names>PL</given-names> </name><name name-style="western"><surname>Tao</surname><given-names>K</given-names> </name><name name-style="western"><surname>Pond</surname><given-names>SLK</given-names> </name><name name-style="western"><surname>Shafer</surname><given-names>RW</given-names> </name></person-group><article-title>Coronavirus Resistance Database (CoV-RDB): SARS-CoV-2 susceptibility to monoclonal antibodies, convalescent plasma, and plasma from vaccinated persons</article-title><source>PLoS One</source><year>2022</year><volume>17</volume><issue>3</issue><fpage>e0261045</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0261045</pub-id><pub-id pub-id-type="medline">35263335</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>O&#x2019;Connor</surname><given-names>K</given-names> </name><name name-style="western"><surname>Weissenbacher</surname><given-names>D</given-names> </name><name name-style="western"><surname>Elyaderani</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lautenbach</surname><given-names>E</given-names> </name><name name-style="western"><surname>Scotch</surname><given-names>M</given-names> </name><name name-style="western"><surname>Gonzalez-Hernandez</surname><given-names>G</given-names> </name></person-group><article-title>Patient-related metadata reported in sequencing studies of SARS-CoV-2: protocol for a scoping review and bibliometric analysis</article-title><source>JMIR Res Protoc</source><year>2025</year><month>04</month><day>22</day><volume>14</volume><fpage>e58567</fpage><pub-id pub-id-type="doi">10.2196/58567</pub-id><pub-id pub-id-type="medline">40262134</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Inward</surname><given-names>RPD</given-names> </name><name name-style="western"><surname>Parag</surname><given-names>KV</given-names> </name><name name-style="western"><surname>Faria</surname><given-names>NR</given-names> </name></person-group><article-title>Using multiple sampling strategies to estimate SARS-CoV-2 epidemiological parameters from genomic sequencing data</article-title><source>Nat Commun</source><year>2022</year><month>09</month><day>23</day><volume>13</volume><issue>1</issue><fpage>5587</fpage><pub-id pub-id-type="doi">10.1038/s41467-022-32812-0</pub-id><pub-id pub-id-type="medline">36151084</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Friedman</surname><given-names>C</given-names> </name><name name-style="western"><surname>Shagina</surname><given-names>L</given-names> </name><name name-style="western"><surname>Lussier</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Hripcsak</surname><given-names>G</given-names> </name></person-group><article-title>Automated encoding of clinical documents based on natural language processing</article-title><source>J Am Med Inform Assoc</source><year>2004</year><volume>11</volume><issue>5</issue><fpage>392</fpage><lpage>402</lpage><pub-id pub-id-type="doi">10.1197/jamia.M1552</pub-id><pub-id pub-id-type="medline">15187068</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Klein</surname><given-names>AZ</given-names> </name><name name-style="western"><surname>Weissenbacher</surname><given-names>D</given-names> </name><name name-style="western"><surname>O&#x2019;Connor</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Detection of patient metadata in published articles for genomic epidemiology using machine learning and large language models</article-title><source>medRxiv</source><year>2025</year><month>04</month><day>28</day><fpage>2025.04.25.25326298</fpage><pub-id pub-id-type="doi">10.1101/2025.04.25.25326298</pub-id><pub-id pub-id-type="medline">40343027</pub-id></nlm-citation></ref><ref id="ref64"><label>64</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Marques</surname><given-names>AD</given-names> </name><name name-style="western"><surname>Graham-Wooten</surname><given-names>J</given-names> </name><name name-style="western"><surname>Fitzgerald</surname><given-names>AS</given-names> </name><etal/></person-group><article-title>SARS-CoV-2 evolution during prolonged infection in immunocompromised patients</article-title><source>MBio</source><year>2024</year><month>03</month><day>13</day><volume>15</volume><issue>3</issue><fpage>e0011024</fpage><pub-id pub-id-type="doi">10.1128/mbio.00110-24</pub-id><pub-id pub-id-type="medline">38364100</pub-id></nlm-citation></ref><ref id="ref65"><label>65</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Du</surname><given-names>P</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Two-step fitness selection for intra-host variations in SARS-CoV-2</article-title><source>Cell Rep</source><year>2022</year><month>01</month><day>11</day><volume>38</volume><issue>2</issue><fpage>110205</fpage><pub-id pub-id-type="doi">10.1016/j.celrep.2021.110205</pub-id><pub-id pub-id-type="medline">34982968</pub-id></nlm-citation></ref><ref id="ref66"><label>66</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pastorio</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zech</surname><given-names>F</given-names> </name><name name-style="western"><surname>Noettger</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Determinants of spike infectivity, processing, and neutralization in SARS-CoV-2 Omicron subvariants BA.1 and BA.2</article-title><source>Cell Host Microbe</source><year>2022</year><month>09</month><day>14</day><volume>30</volume><issue>9</issue><fpage>1255</fpage><lpage>1268</lpage><pub-id pub-id-type="doi">10.1016/j.chom.2022.07.006</pub-id><pub-id pub-id-type="medline">35931073</pub-id></nlm-citation></ref><ref id="ref67"><label>67</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bugatti</surname><given-names>A</given-names> </name><name name-style="western"><surname>Filippini</surname><given-names>F</given-names> </name><name name-style="western"><surname>Messali</surname><given-names>S</given-names> </name><etal/></person-group><article-title>The D405N mutation in the spike protein of SARS-CoV-2 Omicron BA.5 inhibits spike/integrins interaction and viral infection of human lung microvascular endothelial cells</article-title><source>Viruses</source><year>2023</year><month>01</month><day>24</day><volume>15</volume><issue>2</issue><fpage>332</fpage><pub-id pub-id-type="doi">10.3390/v15020332</pub-id><pub-id pub-id-type="medline">36851546</pub-id></nlm-citation></ref><ref id="ref68"><label>68</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schr&#x00F6;der</surname><given-names>S</given-names> </name><name name-style="western"><surname>Richter</surname><given-names>A</given-names> </name><name name-style="western"><surname>Veith</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Characterization of intrinsic and effective fitness changes caused by temporarily fixed mutations in the SARS-CoV-2 spike E484 epitope and identification of an epistatic precondition for the evolution of E484A in variant Omicron</article-title><source>Virol J</source><year>2023</year><month>11</month><day>8</day><volume>20</volume><issue>1</issue><fpage>257</fpage><pub-id pub-id-type="doi">10.1186/s12985-023-02154-4</pub-id><pub-id pub-id-type="medline">37940989</pub-id></nlm-citation></ref><ref id="ref69"><label>69</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Iketani</surname><given-names>S</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Striking antibody evasion manifested by the Omicron variant of SARS-CoV-2</article-title><source>Nature</source><year>2022</year><month>02</month><volume>602</volume><issue>7898</issue><fpage>676</fpage><lpage>681</lpage><pub-id pub-id-type="doi">10.1038/s41586-021-04388-0</pub-id><pub-id pub-id-type="medline">35016198</pub-id></nlm-citation></ref><ref id="ref70"><label>70</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Luan</surname><given-names>B</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Huynh</surname><given-names>T</given-names> </name></person-group><article-title>Enhanced binding of the N501Y-mutated SARS-CoV-2 spike protein to the human ACE2 receptor: insights from molecular dynamics simulations</article-title><source>FEBS Lett</source><year>2021</year><month>05</month><volume>595</volume><issue>10</issue><fpage>1454</fpage><lpage>1461</lpage><pub-id pub-id-type="doi">10.1002/1873-3468.14076</pub-id><pub-id pub-id-type="medline">33728680</pub-id></nlm-citation></ref><ref id="ref71"><label>71</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Greaney</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Starr</surname><given-names>TN</given-names> </name><name name-style="western"><surname>Gilchuk</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Complete mapping of mutations to the SARS-CoV-2 spike receptor-binding domain that escape antibody recognition</article-title><source>Cell Host Microbe</source><year>2021</year><month>01</month><day>13</day><volume>29</volume><issue>1</issue><fpage>44</fpage><lpage>57</lpage><pub-id pub-id-type="doi">10.1016/j.chom.2020.11.007</pub-id><pub-id pub-id-type="medline">33259788</pub-id></nlm-citation></ref><ref id="ref72"><label>72</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Futatsusako</surname><given-names>H</given-names> </name><name name-style="western"><surname>Hashimoto</surname><given-names>R</given-names> </name><name name-style="western"><surname>Yamamoto</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Longitudinal analysis of genomic mutations in SARS-CoV-2 isolates from persistent COVID-19 patient</article-title><source>iScience</source><year>2024</year><month>05</month><day>17</day><volume>27</volume><issue>5</issue><fpage>109597</fpage><pub-id pub-id-type="doi">10.1016/j.isci.2024.109597</pub-id><pub-id pub-id-type="medline">38638575</pub-id></nlm-citation></ref><ref id="ref73"><label>73</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Michel</surname><given-names>HA</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>PH</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>GL</given-names> </name></person-group><article-title>Manipulation of innate immune signaling pathways by SARS-CoV-2 non-structural proteins</article-title><source>Front Microbiol</source><year>2022</year><volume>13</volume><fpage>1027015</fpage><pub-id pub-id-type="doi">10.3389/fmicb.2022.1027015</pub-id><pub-id pub-id-type="medline">36478862</pub-id></nlm-citation></ref><ref id="ref74"><label>74</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mondeali</surname><given-names>M</given-names> </name><name name-style="western"><surname>Etemadi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Barkhordari</surname><given-names>K</given-names> </name><etal/></person-group><article-title>The role of S477N mutation in the molecular behavior of SARS-CoV-2 spike protein: an in-silico perspective</article-title><source>J Cell Biochem</source><year>2023</year><month>02</month><volume>124</volume><issue>2</issue><fpage>308</fpage><lpage>319</lpage><pub-id pub-id-type="doi">10.1002/jcb.30367</pub-id><pub-id pub-id-type="medline">36609701</pub-id></nlm-citation></ref><ref id="ref75"><label>75</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Motozono</surname><given-names>C</given-names> </name><name name-style="western"><surname>Toyoda</surname><given-names>M</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>TS</given-names> </name><etal/></person-group><article-title>The SARS-CoV-2 Omicron BA.1 spike G446S mutation potentiates antiviral T-cell recognition</article-title><source>Nat Commun</source><year>2022</year><month>09</month><day>21</day><volume>13</volume><issue>1</issue><fpage>5440</fpage><pub-id pub-id-type="doi">10.1038/s41467-022-33068-4</pub-id><pub-id pub-id-type="medline">36130929</pub-id></nlm-citation></ref><ref id="ref76"><label>76</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Park</surname><given-names>SB</given-names> </name><name name-style="western"><surname>Khan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Chiliveri</surname><given-names>SC</given-names> </name><etal/></person-group><article-title>SARS-CoV-2 Omicron variants harbor spike protein mutations responsible for their attenuated fusogenic phenotype</article-title><source>Commun Biol</source><year>2023</year><month>05</month><day>24</day><volume>6</volume><issue>1</issue><fpage>556</fpage><pub-id pub-id-type="doi">10.1038/s42003-023-04923-x</pub-id><pub-id pub-id-type="medline">37225764</pub-id></nlm-citation></ref><ref id="ref77"><label>77</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Griffiths</surname><given-names>EJ</given-names> </name><name name-style="western"><surname>Timme</surname><given-names>RE</given-names> </name><name name-style="western"><surname>Mendes</surname><given-names>CI</given-names> </name><etal/></person-group><article-title>Future-proofing and maximizing the utility of metadata: the PHA4GE SARS-CoV-2 contextual data specification package</article-title><source>Gigascience</source><year>2022</year><month>02</month><day>16</day><volume>11</volume><fpage>giac003</fpage><pub-id pub-id-type="doi">10.1093/gigascience/giac003</pub-id><pub-id pub-id-type="medline">35169842</pub-id></nlm-citation></ref><ref id="ref78"><label>78</label><nlm-citation citation-type="web"><source>ChatGPT</source><access-date>2026-09-22</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://chatgpt.com">https://chatgpt.com</ext-link></comment></nlm-citation></ref><ref id="ref79"><label>79</label><nlm-citation citation-type="web"><article-title>Dynamic data visualization dashboard</article-title><source>HLP Gonzalez Lab</source><access-date>2026-09-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://dataviz.hlpgonzalezlab.com/genomic-metadata">https://dataviz.hlpgonzalezlab.com/genomic-metadata</ext-link></comment></nlm-citation></ref><ref id="ref80"><label>80</label><nlm-citation citation-type="web"><article-title>pavia27/pathogen-genome-framework-manuscript: pathogen-genome-framework-manuscript</article-title><source>Zenodo</source><access-date>2026-09-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.21365814">https://doi.org/10.5281/zenodo.21365814</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Rank frequency and empirical threshold for nonrandom mutations (Figure S1); number of article records retained at each stage of the filtering process (Figure S2); characteristics of the 21 SARS-CoV-2 sequencing studies included in metadata enrichment (Table S1); and characteristics of the 21 SARS-CoV-2 sequencing studies included in metadata enrichment (Table S2).</p><media xlink:href="bioinform_v7i1e92529_app1.pdf" xlink:title="PDF File, 387 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Supplementary methods detailing the screening protocol used to select the 21 articles and the enrichment procedures applied to extract and normalize patient metadata.</p><media xlink:href="bioinform_v7i1e92529_app2.pdf" xlink:title="PDF File, 93 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>Enriched dataset of patient metadata linked to each GenBank accession.</p><media xlink:href="bioinform_v7i1e92529_app3.xlsx" xlink:title="XLSX File, 5544 KB"/></supplementary-material></app-group></back></article>