﻿<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.0 20120330//EN" "JATS-journalpublishing1.dtd">
<article xml:lang="en" article-type="research-article" xmlns:xlink="http://www.w3.org/1999/xlink">
<?release-delay 0|0?>
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">Ann Lab Med</journal-id>
<journal-title-group>
<journal-title>Annals of Laboratory Medicine</journal-title>
<abbrev-journal-title abbrev-type="publisher">Ann Lab Med</abbrev-journal-title>
</journal-title-group>
<issn pub-type="ppub">2234-3806</issn>
<issn pub-type="epub">2234-3814</issn>
<publisher>
<publisher-name>Korean Society for Laboratory Medicine</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3343/alm.2022.42.4.438</article-id>
<article-id pub-id-type="publisher-id">alm-42-4-438</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Article</subject>
<subj-group>
<subject>Clinical Microbiology</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Comparing Genomic Characteristics of <italic>Streptococcus pyogenes</italic> Associated with Invasiveness over a 20-year Period in Korea</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<contrib-id contrib-id-type="orcid">https://orcid.org/0000-0001-9737-2393</contrib-id>
<name><surname>Shin</surname><given-names>Hyoshim</given-names></name>
<degrees>M.D.</degrees>
<xref rid="aff1" ref-type="aff">1</xref>
</contrib>
<contrib contrib-type="author">
<contrib-id contrib-id-type="orcid">https://orcid.org/0000-0003-4131-2062</contrib-id>
<name><surname>Takahashi</surname><given-names>Takashi</given-names></name>
<degrees>M.D., Ph.D.</degrees>
<xref rid="aff2" ref-type="aff">2</xref>
</contrib>
<contrib contrib-type="author">
<contrib-id contrib-id-type="orcid">https://orcid.org/0000-0002-3377-4833</contrib-id>
<name><surname>Lee</surname><given-names>Seungjun</given-names></name>
<degrees>M.D.</degrees>
<xref rid="aff3" ref-type="aff">3</xref>
</contrib>
<contrib contrib-type="author">
<contrib-id contrib-id-type="orcid">https://orcid.org/0000-0002-5857-0749</contrib-id>
<name><surname>Choi</surname><given-names>Eun Hwa</given-names></name>
<degrees>M.D., Ph.D.</degrees>
<xref rid="aff4" ref-type="aff">4</xref>
</contrib>
<contrib contrib-type="author">
<contrib-id contrib-id-type="orcid">https://orcid.org/0000-0003-0899-2860</contrib-id>
<name><surname>Maeda</surname><given-names>Takahiro</given-names></name>
<degrees>B.P.</degrees>
<xref rid="aff2" ref-type="aff">2</xref>
</contrib>
<contrib contrib-type="author">
<contrib-id contrib-id-type="orcid">https://orcid.org/0000-0003-3284-3056</contrib-id>
<name><surname>Fukushima</surname><given-names>Yasuto</given-names></name>
<degrees>B.P.</degrees>
<xref rid="aff2" ref-type="aff">2</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<contrib-id contrib-id-type="orcid">https://orcid.org/0000-0001-8099-8891</contrib-id>
<name><surname>Kim</surname><given-names>Sunjoo</given-names></name>
<degrees>M.D., Ph.D.</degrees>
<xref rid="aff3" ref-type="aff">3</xref>
<xref rid="aff5" ref-type="aff">5</xref>
<xref rid="cor1" ref-type="corresp"/>
</contrib>
</contrib-group>
<aff id="aff1"><label>1</label>Department of Laboratory Medicine, Gyeongsang National University Hospital, Jinju, <country>Korea</country></aff>
<aff id="aff2"><label>2</label>Laboratory of Infectious Diseases, Graduate School of Infection Control Sciences &#38; &#332;mura Satoshi Memorial Institute, Kitasato University, Tokyo, <country>Japan</country></aff>
<aff id="aff3"><label>3</label>Department of Laboratory Medicine, Gyeongsang National University Changwon Hospital, Changwon, <country>Korea</country></aff>
<aff id="aff4"><label>4</label>Department of Pediatrics, Seoul National University College of Medicine, Seoul, <country>Korea</country></aff>
<aff id="aff5"><label>5</label>Department of Laboratory Medicine, Gyeongsang National University College of Medicine, Institute of Health Sciences, Jinju, <country>Korea</country></aff>
<author-notes>
<corresp id="cor1"><bold>Corresponding author:</bold> Sunjoo Kim, M.D., Ph.D. Department of Laboratory Medicine, Gyeongsang National University Changwon Hospital, 11 Samjungja-ro, Seongsan-gu, Changwon 51472, Korea Tel: +82-55-214-3072 Fax: +82-55-214-3087 E-mail: <email xlink:href="sjkim8239@hanmail.net">sjkim8239@hanmail.net</email></corresp>
</author-notes>
<pub-date pub-type="ppub">
<day>1</day>
<month>7</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="epub">
<day>1</day>
<month>7</month>
<year>2022</year>
</pub-date>
<volume>42</volume>
<issue>4</issue>
<fpage>438</fpage>
<lpage>446</lpage>
<history>
<date date-type="received">
<day>11</day>
<month>6</month>
<year>2021</year>
</date>
<date date-type="rev-recd">
<day>23</day>
<month>7</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>6</day>
<month>12</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>&#169; Korean Society for Laboratory Medicine</copyright-statement>
<copyright-year>2022</copyright-year>
<license license-type="open-access">
<license-p>This is an open-access article distributed under the terms of the Creative Commons Attribution Non-Commercial License (<ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by-nc/4.0">http://creativecommons.org/licenses/by-nc/4.0</ext-link>) which permits unrestricted non-commercial use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<abstract>
<sec sec-type="background">
<title>Background</title>
<p>Few studies have investigated the invasiveness of <italic>Streptococcus pyogenes</italic> based on whole-genome sequencing (WGS). Using WGS, we determined the genomic features associated with invasiveness of <italic>S. pyogenes</italic> strains in Korea.</p>
</sec>
<sec sec-type="methods">
<title>Methods</title>
<p>Forty-five <italic>S. pyogenes</italic> strains from 1997, 2006, and 2017, including common emm types, were selected from the repository at Gyeongsang National University Hospital in Korea. In addition, 48 <italic>S. pyogenes</italic> strains were randomly selected depending on their invasiveness between 1997 and 2017 to evaluate the genetic evolution and the associations between invasiveness and genetic profiles. Using WGS datasets, we conducted virulence-associated DNA sequence determination, <italic>emm</italic> genotyping, multi-locus sequence typing (MLST), and superantigen gene profiling.</p>
</sec>
<sec sec-type="results">
<title>Results</title>
<p>In total, 87 strains were included in this study. There were no significant differences in the genomic features throughout the study periods. Four genes, <italic>csn1</italic>, <italic>ispE</italic>, <italic>nisK</italic>, and <italic>citC</italic>, were detected only in invasive strains. There was a significant association between invasiveness and <italic>emm</italic> cluster type A-C3, including, <italic>emm</italic>1.0, <italic>emm</italic>1.18, <italic>emm</italic>1.3, and <italic>emm</italic>1.76 (<italic>P</italic>&#60;0.05). The predominant <italic>emm</italic>1 lineage belonged to ST28. There were no associations between invasiveness and superantigen gene profiles.</p>
</sec>
<sec sec-type="conclusions">
<title>Conclusions</title>
<p>This is the first study using WGS datasets of <italic>S. pyogenes</italic> strains collected between 1997 and 2017 in Korea. Streptococcal invasiveness is associated with the presence of <italic>csn1</italic>, <italic>ispE</italic>, <italic>nisK</italic>, and <italic>citC</italic>. The emm1 lineage and ST28 clone are explicitly associated with invasiveness, whereas genomic features remained stable over the 20-year period.</p>
</sec>
</abstract>
<kwd-group>
<kwd><italic>Streptococcus pyogenes</italic></kwd>
<kwd>Invasiveness</kwd>
<kwd>Whole genome sequencing</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec sec-type="intro">
<title>INTRODUCTION</title>
<p><italic>Streptococcus pyogenes</italic> causes a wide spectrum of diseases in humans, from mild tonsillopharyngitis, impetigo, and scarlet fever to severe and invasive sepsis, arthritis, necrotizing fasciitis, and streptococcal toxic shock syndrome (STSS) [<xref rid="ref1" ref-type="bibr">1</xref>]. More than 500,000 deaths due to streptococcal infections are reported worldwide each year, making <italic>S. pyogenes</italic> a major pathogen associated with high morbidity and mortality [<xref rid="ref2" ref-type="bibr">2</xref>]. <italic>S. pyogenes</italic> strains are genetically diverse and have various virulence factors, including adhesion molecules, superantigens, DNases, proteases, and M protein, that are involved in its complex pathogenicity.</p>
<p><italic>S. pyogenes</italic> strains can be classified by <italic>emm</italic> typing that is based on PCR amplification and amplicon sequencing of the <italic>emm</italic> gene, encoding M protein. Multilocus sequence typing (MLST) based on the amplification and sequencing of seven housekeeping genes and pulsed-field gel electrophoresis (PFGE) of the large genomic fragment have also been used in epidemiological studies [<xref rid="ref3" ref-type="bibr">3</xref>, <xref rid="ref4" ref-type="bibr">4</xref>]. Whole-genome sequencing (WGS) datasets help in discriminating closely related strains and allow epidemiological analysis of small infection clusters. WGS analysis is an ideal molecular typing method for bacteria, as it provides complete genetic information of a strain. Until recently, because of the high cost and technical complexity, WGS was beyond the reach of average diagnostic laboratories [<xref rid="ref5" ref-type="bibr">5</xref>]. Little has been published on the invasiveness of <italic>S. pyogenes</italic> strains collected in Korea. Characterization of <italic>S. pyogenes</italic> in a longitudinal surveillance study of WGS datasets would provide important information about the genomic characteristics, virulence-associated gene profiles, and genomic dynamics during the long period of time.</p>
<p>We genomically characterized <italic>S. pyogenes</italic> strains collected in Korea between 1997 and 2017 by determining their emm types, MLST-based sequence types (STs), and superantigen gene profiles. We then investigated whether the genomic characteristics of <italic>S. pyogenes</italic> strains differed based on invasiveness and isolation year.</p>
</sec>
<sec sec-type="materials|methods">
<title>MATERIALS AND METHODS</title>
<sec>
<title>Bacterial strain selection</title>
<p>All strains used in this study were collected between 1997 and 2017 and stored in the repository at Gyeongsang National University Hospital (GNUH) in Gyeongnam Province, Korea. Forty-five <italic>S. pyogenes</italic> strains were selected according to the common <italic>emm</italic> types in three years: 1997, 2006, and 2017. Forty-eight strains were randomly selected based on their invasiveness between 1997 and 2017. An &#8220;invasive strain&#8221; was defined as one isolated from a normally sterile body fluid, such as blood, cerebrospinal fluid, pleural fluid, pericardial fluid, joint fluid, bone aspirate, or a deep-tissue abscess [<xref rid="ref6" ref-type="bibr">6</xref>]. <xref rid="F1" ref-type="fig">Fig. 1</xref> shows a flow chart of the strain selection procedure. In total, 87 strains were included in this study. For the first analysis to evaluate the genetic evolution over a 20-year time span, non-invasive strains (N=45) were selected every 10 years: 1997, 2006, and 2017. In the second analysis to evaluate assocations between invasiveness and genetic profiles, non-invasive and invasive strains were compared. Sixty-three non-invasive strains were isolated from the throats of carriers who did not have any symptoms or signs of tonsillitis. The other 24 invasive strains were isolated from blood (N=21) or joint fluid (N=3) of patients (<xref rid="F1" ref-type="fig">Fig. 1</xref>).</p>
<p>Bacteria were identified using a Vitek-2 automated identification system (BioM&#233;rieux Inc., Marcy l&#8217;&#201;toile, France). All strains were inoculated in 30% glycerol in Todd-Hewitt broth and stored at &#8211;70&#176;C. They were recovered on blood agar plates for genetic analysis.</p>
<p>The study protocol was approved by the Institutional Review Board of GNUH (approval number: GNUCH 2018-01-008). Informed consent was waived because of the retrospective nature of the study.</p>
</sec>
<sec>
<title>Genomic DNA extraction and WGS</title>
<p>Genomic DNA was extracted using a Wizard Genomic DNA Isolation Kit (Promega, Madison, WI, USA). The DNA and potential culture contamination were checked by 16S rRNA gene sequencing using an ABI 3730 DNA sequencer (Applied Biosystems, Foster City, CA, USA). A draft genome sequence of each strain was generated by MiSeq sequencing (300-bp, paired-end) using a MiSeq Reagent Kit v3 (Illumina, San Diego, CA, USA). Sequencing libraries were prepared using the TruSeq DNA LT sample Prep Kit (Illumina). The Illumina sequencing data were assembled with SPAdes v3.13.0 (Algorithmic Biology Lab, St. Petersburg Academic University of the Russian Academy of Sciences, St. Petersburg, Russia). The EzBioCloud genome database was used for gene finding and functional annotation of the whole-genome assemblies (<ext-link ext-link-type="uri" xlink:href="https://www.ezbiocloud.net">https://www.ezbiocloud.net</ext-link>). Protein-coding DNA sequences (CDSs) were predicted using Prodigal 2.6.2 [<xref rid="ref7" ref-type="bibr">7</xref>]. The CDSs were classified based on their roles, with reference to orthologous groups (EggNOG v4.5; <ext-link ext-link-type="uri" xlink:href="http://eggnogdb.embl.de">http://eggnogdb.embl.de</ext-link>). For more detailed functional annotation, the predicted CDSs were compared with those from the Swiss-Prot (<ext-link ext-link-type="uri" xlink:href="https://www.uniprot.org">https://www.uniprot.org</ext-link>), KEGG (<ext-link ext-link-type="uri" xlink:href="http://www.genome.jp/kegg/">http://www.genome.jp/kegg/</ext-link>), and SEED (<ext-link ext-link-type="uri" xlink:href="http://pubseed.theseed.org">http://pubseed.theseed.org</ext-link>) databases using the UBLAST (<ext-link ext-link-type="uri" xlink:href="https://www.drive5.com/">https://www.drive5.com/</ext-link>) program.</p>
</sec>
<sec>
<title>Comparative genome (CG) analyses</title>
<p>CG analyses comprised two steps (<xref rid="F1" ref-type="fig">Fig. 1</xref>). The first analysis was conducted according to the isolation year (strains of 1997 vs. those of 2006 vs. those of 2017). The second analysis was conducted according to invasiveness (strains from the throat vs. those from blood/joint fluid). CG analysis was conducted by comparing functional genes based on the clustering of orthologous genes. The genome sequences of all strains were obtained from the EzBioCloud database (<ext-link ext-link-type="uri" xlink:href="http://www.ezbiocloud.net/">http://www.ezbiocloud.net/</ext-link>), and average nucleotide identity (ANI) values were calculated. For ANI calculation, the query genomes were cut into small fragments (1,020 bp), and high-scoring pairs between two genome sequences were selected using the USEARCH program (<ext-link ext-link-type="uri" xlink:href="http://www.drive5.com/usearch">http://www.drive5.com/usearch</ext-link>). Using the calculated ANI values, a dendrogram was constructed using the unweighted pair group method. Homologous regions in a target genome to query open reading frames were determined using the USEARCH program and were aligned using pair-wise global alignment. The matched regions in the subject contig were extracted and saved as homologs [<xref rid="ref8" ref-type="bibr">8</xref>].</p>
</sec>
<sec>
<title><italic>emm</italic> genotyping</title>
<p>We used DDBJ Fast Annotation and Submission Tool v1.2.4 (DFAST; <ext-link ext-link-type="uri" xlink:href="https://dfast.nig.ac.jp">https://dfast.nig.ac.jp</ext-link>) for annotation and searched the sequences around the <italic>mga</italic> annotation encoding multiple virulence gene regulators based on the annotation data [<xref rid="ref9" ref-type="bibr">9</xref>]. If the sequences around <italic>mga</italic> were not found, the sequences around the <italic>emm1</italic> primer (forward: 5&#900;-TATT(C/G)GCTTAGAAAATTAA-3&#900;) were searched throughout the contig sequences using the FASTA format. We extracted the <italic>emm</italic> sequences between emm1 and <italic>emm</italic>2 primers (reverse: 5&#900;-GCAAGTTCTTCAGCTTGTTT-3&#900;) using the corresponding sequences recovered from the contig data. By inserting the extracted sequences into the Centers for Disease Control and Prevention (CDC) database (<ext-link ext-link-type="uri" xlink:href="https://www2.cdc.gov/vaccines/biotech/strepblast.asp">https://www2.cdc.gov/vaccines/biotech/strepblast.asp</ext-link>), <italic>emm</italic> genotypes (including subtypes) were assigned to the extracted sequences.</p>
<p>When sequences were incompletely matched with an <italic>emm</italic> genotype in the CDC database, we directly PCR-amplified <italic>emm</italic> using bacterial DNA templates and the <italic>emm</italic>1/<italic>emm</italic>2 primer set and sequenced the amplicons after purification using an AccuPrep Purification Kit (Bioneer Corp., Daejeon, Korea). The <italic>emm</italic> genotypes were assigned directly to the amplified sequences based on the CDC database [<xref rid="ref10" ref-type="bibr">10</xref>, <xref rid="ref11" ref-type="bibr">11</xref>].</p>
</sec>
<sec>
<title>Phylogenetic tree and superantigen gene profiling</title>
<p>Phylogenetic analysis was accomplished using ~1.4 million bps of orthologous protein-coding regions for 45 strains according to the isolation year and 48 strains according to invasiveness, respectively (data not shown).</p>
<p>To determine five target genes (<italic>speA</italic>, <italic>speB</italic>, <italic>speC</italic>, <italic>ssa</italic>, and <italic>smeZ</italic>) encoding the superantigens (also known as exotoxins), we conducted PCR simulation analysis using the Serial Cloner (<ext-link ext-link-type="uri" xlink:href="http://serialbasics.free.fr/Serial_Cloner.html">http://serialbasics.free.fr/Serial_Cloner.html</ext-link>) application with the contig sequences, as previously reported [<xref rid="ref12" ref-type="bibr">12</xref><xref rid="ref13" ref-type="bibr"/>-<xref rid="ref14" ref-type="bibr">14</xref>]. The primer sets used to amplify the <italic>speA</italic>, <italic>speB</italic>, <italic>speC</italic>, <italic>ssa</italic>, and <italic>smeZ</italic> are listed in <xref rid="T1" ref-type="table">Table 1</xref>. The <italic>speB</italic> product was included as an internal control in the PCR simulation analysis because all <italic>S. pyogenes</italic> strains possess the <italic>speB</italic> sequence (955 bp). Superantigen gene profiles were determined for each strain.</p>
</sec>
<sec>
<title>MLST</title>
<p>We determined the STs using allelic profiles consisting of seven housekeeping genes (<italic>gki</italic>, <italic>gtr</italic>, <italic>murI</italic>, <italic>mutS</italic>, <italic>recP</italic>, <italic>xpt</italic>, and <italic>yqiL</italic>) by inserting the contig sequences obtained into the online application MLST v2.0 (<ext-link ext-link-type="uri" xlink:href="https://cge.cbs.dtu.dk/services/MLST/">https://cge.cbs.dtu.dk/services/MLST/</ext-link>), which is managed by the Center for Genomic Epidemiology at the Technical University of Denmark [<xref rid="ref15" ref-type="bibr">15</xref>]. The STs were grouped into clonal complexes (CC), whereby related STs were classified as single locus variants, differing in only one housekeeping gene. An expansion of the goeBURST program implemented in PHYLOViZ was used to produce a minimum-spanning tree representing possible relationships among the STs [<xref rid="ref16" ref-type="bibr">16</xref>].</p>
<p>For novel allelic numbers/STs, we submitted the data (i.e., bacterial genotypic/phenotypic data and patient backgrounds) to the <italic>S. pyogenes</italic> PubMLST (<ext-link ext-link-type="uri" xlink:href="http://pubmlst.org/organisms/streptococcus-pyogenes">http://pubmlst.org/organisms/streptococcus-pyogenes</ext-link>) database. The PubMLST curator assigned novel allelic numbers/STs to our strains.</p>
</sec>
<sec>
<title>Statistical analysis</title>
<p>We used Fisher&#8217;s exact test (two-sided) to determine significant differences in categorical variables, and the chi-square test to compare the proportions in each <italic>emm</italic> genotype/cluster between invasive and non-invasive strains. SPSS Statistics v22.0 (IBM Corp., Armonk, NY, USA) was used for the analysis. <italic>P</italic>&#60;0.05 was considered significant.</p>
</sec>
</sec>
<sec sec-type="results">
<title>RESULTS</title>
<sec>
<title><italic>emm</italic> genotypes/clusters</title>
<p>The <italic>emm</italic> genotypes and cluster types are presented in <xref rid="S1" ref-type="supplementary-material">Supplemental Data Table 1</xref>. In total, 21 <italic>emm</italic> genotypes were identified, with <italic>emm</italic>1 (<italic>emm</italic>1.00, <italic>emm</italic>1.18, <italic>emm</italic>1.30, and <italic>emm</italic>1.76), <italic>emm</italic>4 (<italic>emm</italic>4.00), and <italic>emm</italic>12 (<italic>emm</italic>12.00, <italic>emm</italic>12.19, and <italic>emm</italic>12.49) accounting for 19.5%, 13.8%, and 20.7%, respectively. In total, eight <italic>emm</italic> cluster types were identified, among which the A-C3, A-C4, and E1 types were the most common. <xref rid="F2" ref-type="fig">Fig. 2</xref> shows the differences among the <italic>emm</italic> clusters according to invasiveness. There were significantly more invasive strains in cluster A-C3 than in the others (<italic>P</italic>&#60;0.05). The A-C3 type included the genotypes <italic>emm</italic>1.0, <italic>emm</italic>1.18, <italic>emm</italic>1.3, and <italic>emm</italic>1.76 (<xref rid="S1" ref-type="supplementary-material">Supplemental Data Table 1</xref>, <xref rid="F2" ref-type="fig">Fig. 2</xref>).</p>
</sec>
<sec>
<title>ST with goeBURST diagram</title>
<p>The STs are presented in <xref rid="S1" ref-type="supplementary-material">Supplemental Data Table 1</xref>. The 87 strains comprised 21 STs with exact loci matched against the PubMLST database. ST36, ST28, and ST39, accounting for 20.7%, 18.4%, and 11.5%, respectively, were the most frequent. There were strong associations of genetic characteristics within the MLST complex. The goeBURST diagram is shown in <xref rid="F3" ref-type="fig">Fig. 3</xref>. There were 17 singletons in the CG analysis, and ST28 showed a clonal distribution of invasive strains in the second analysis. The predominant <italic>emm</italic>1 lineage belonged to ST28 (<xref rid="F3" ref-type="fig">Fig. 3</xref>).</p>
</sec>
<sec>
<title>Phylogenetic tree and superantigen gene profiling</title>
<p>The phylogenetic tree based on the periodic comparison showed a sporadic distribution (data not shown). The second analysis revealed the genetic relationships among <italic>emm</italic> genotypes or STs. Superantigen profiling revealed that <italic>speB</italic> was present in all strains. <italic>speZ</italic>-<italic>speB</italic> and <italic>speZ</italic>-<italic>speB</italic>-<italic>speC</italic> profiles were present in 37.9% and 28.7% of the total strains, respectively. We found no significant association between the coexistence of different superantigen genes and invasiveness.</p>
</sec>
<sec>
<title>Virulence-associated CDSs</title>
<p>When comparing gene origins by pan-genome orthologous group (POG) analysis, we found <italic>csn1</italic>, <italic>ispE</italic>, <italic>nisK</italic>, and <italic>citC</italic> were more significantly present in invasive strains than in non-invasive strains (all <italic>P</italic>&#60;0.05) (<xref rid="T2" ref-type="table">Table 2</xref>).</p>
<p>We looked for common virulence-associated CDSs among all 87 strains by searching for annotated CDSs based on functional annotation of the whole-genome assemblies. We found 25 CDSs associated with bacterial virulence. Among them, 12 (lactocepin, oleate hydratase, putative glycoslytransferases, capsule biosynthesis protein [CapA], regulatory protein MsrR, internalin-I, deoxyribonuclease, biofilm-regulatory protein, listeriolysin-regulatory protein, streptokinase, C5a peptidase, and M protein) were identified in all strains. CDSs encoding exotoxin A and procollagen-proline 3-dioxygenase were frequently detected in invasive strains (all <italic>P</italic>&#60;0.05) (<xref rid="T3" ref-type="table">Table 3</xref>).</p>
</sec>
</sec>
<sec sec-type="discussion">
<title>DISCUSSION</title>
<p>WGS analyses have proven useful in unraveling the genetic diversity of strains and discriminating between closely related strains. Our study provided information about the genomic characteristics and virulence genes of 87 strains collected in Korea over a 20-year period based on longitudinal analysis of WGS datasets.</p>
<p>Up to 200 <italic>emm</italic> types have been identified, suggesting that the M protein is a polymorphic protein (<ext-link ext-link-type="uri" xlink:href="https://www.cdc.gov/streplab/index.html">https://www.cdc.gov/streplab/index.html</ext-link>). A global review of emm types revealed a total of 205 <italic>emm</italic> types, including a category of non-typeable strains. The most common <italic>emm</italic> type was <italic>emm</italic>1, which accounted for 18.3% of all strains, followed by <italic>emm</italic>12 (11.1%), <italic>emm</italic>28 (8.5%), <italic>emm</italic>3 (6.9%), and <italic>emm</italic>4 (6.9%) [<xref rid="ref17" ref-type="bibr">17</xref>]. In Europe, severe clinical manifestations, such as STSS and necrotizing fasciitis, were caused by 45 different types, of which <italic>emm</italic>1 was the most prevalent, accounting for 37% and 31% of cases, respectively [<xref rid="ref18" ref-type="bibr">18</xref>]. In Korea, <italic>emm</italic>1 was significantly more common among invasive cases, whereas <italic>emm</italic>4, <italic>emm</italic>6, and <italic>emm</italic>12 were dominant in non-invasive cases [<xref rid="ref19" ref-type="bibr">19</xref>].</p>
<p>Globally, <italic>emm</italic> types influence routine epidemiological surveillance, and MLST is excellent for exotoxin gene profiling [<xref rid="ref20" ref-type="bibr">20</xref>]. <italic>emm</italic>1 and <italic>emm</italic>3 associated with ST28 have traditionally been associated with invasive <italic>S. pyogenes</italic> strains [<xref rid="ref19" ref-type="bibr">19</xref>]. In this study, the predominant <italic>emm</italic>1 lineage belonging to ST28 showed a clonal distribution of invasive strains according to the goeBURST results. ST785, a single locus variant of ST28, also belonged to <italic>emm</italic>1. The advantages of the conservative approach used by goeBURST, in which links are shown only between STs that differ at a single locus, have been demonstrated by the analysis of meningococcal CCs using goeBURST, which allowed describing the clonal structures of populations in a quantitative way [<xref rid="ref21" ref-type="bibr">21</xref>].</p>
<p>By searching for virulence-associated CDSs, we found that four genes, <italic>csn1</italic> (<italic>cas9</italic>), <italic>ispE</italic>, <italic>nisK</italic> (<italic>spaK</italic>), and <italic>citC</italic>, were frequently present in invasive strains. <italic>Cas9</italic> is associated with the clustered regularly interspaced short palindromic repeats (CRISPR) array [<xref rid="ref22" ref-type="bibr">22</xref>]. The type II-A system of <italic>S. pyogenes</italic> contains four cas genes (<italic>cas9</italic>, <italic>cas1</italic>, <italic>cas2</italic>, and <italic>csn1</italic>) and six CRISPR spacers targeting a phage endopeptidase, superantigen (sepM), methyltransferase, hyaluronidase, hypothetical protein, and an unknown target. <italic>cas9</italic> (previously called <italic>csn1</italic>) and trans-activating CRISPR RNA are essential for all stages of immunity in the type II-A system [<xref rid="ref23" ref-type="bibr">23</xref>]. In our study, <italic>cas9</italic> was more common in invasive strains. This result indicates the role of <italic>cas9</italic> in <italic>S. pyogenes</italic> pathogenesis and the ability of <italic>S. pyogenes</italic> to counter external stimuli, while playing a direct role in bacterial immunity.</p>
<p><italic>IspE</italic> is involved in the isoprenoid (IPP) biosynthesis pathway. IPPs comprise a large, diverse class of naturally occurring organic chemicals essential for cell survival [<xref rid="ref24" ref-type="bibr">24</xref>]. The IPP pathway is essential for various vital biological functions of bacteria. IPPs are synthesized via the classical mevalonate pathway or the alternative 2C-methyl-D-erythritol 4-phosphate (MEP) pathway. The distribution of the MEP and mevalonate pathways is highly complex, but there is a clear bias towards the former in pathogenic organisms. In our study, <italic>ispE</italic> expression was significantly upregulated in invasive strains, suggesting that the IPP biosynthesis pathway is associated with virulence.</p>
<p>Lactococcal <italic>nisA</italic> is a promoter in the nis cluster that is required for the biosynthesis, immunity, and regulatory systems of <italic>S. pyogenes</italic>; <italic>nisK</italic> and <italic>spaK</italic> also belong to this cluster. The <italic>nisA</italic> promoter is dependent on NisR and NisK, which are important in the survival mechanisms of <italic>S. pyogenes</italic>. The <italic>nisA</italic> promoter allows gene expression modulation in pathogenic streptococci [<xref rid="ref26" ref-type="bibr">26</xref>]. Bacterial citrate lyase, the key enzyme in citrate fermentation, is encoded by <italic>citC</italic>. Lactic acid bacteria of the genus <italic>Leuconostoc</italic> can produce carbon dioxide and C4 aromatic compounds through lactose heterofermentation and citrate utilization [<xref rid="ref27" ref-type="bibr">27</xref>]. We confirmed that <italic>nisR</italic> and <italic>citC</italic> are significantly associated with invasive <italic>S. pyogenes</italic>. Their protein products are widely found in <italic>Lactococcus</italic>; therefore, we presume that the genes must have been transmitted via plasmid transfer, allowing efficient control of gene expression by regulatory proteins [<xref rid="ref28" ref-type="bibr">28</xref>, <xref rid="ref29" ref-type="bibr">29</xref>]. The transmitted genes allow <italic>S. pyogenes</italic> to survive in various environments, strengthening its invasiveness [<xref rid="ref22" ref-type="bibr">22</xref>].</p>
<p>We investigated virulence-associated CDSs among all 87 strains searched from annotated CDSs based on functional annotation of the whole-genome assemblies. The genes encoding exotoxin A and procollagen-proline 3-dioxygenase were frequently present in invasive strains. Streptococcal exotoxin A is encoded by <italic>speA</italic>, which is part of bacteriophage T12 [<xref rid="ref30" ref-type="bibr">30</xref>]. The presence of <italic>speA</italic> is frequently associated with scarlet fever or rheumatic fever and streptococcal disease [<xref rid="ref1" ref-type="bibr">1</xref>, <xref rid="ref31" ref-type="bibr">31</xref>]. Procollagen-proline 3-dioxygenase catalyzes procollagen L-proline to produce procollagen trans-3-hydroxy-L-proline. This enzyme belongs to the family of oxidoreductases, and its activity has been detected in several strains [<xref rid="ref32" ref-type="bibr">32</xref>]. A relationship between this enzyme and invasiveness has been rarely observed in <italic>S. pyogenes</italic> [<xref rid="ref33" ref-type="bibr">33</xref>].</p>
<p>The phylogenetic analysis revealed no significant associations between the superantigen profiles and invasiveness. Moreover, establishing links between longitudinal groups within the phylogenetic tree was difficult. These results indicate the preservation of stable genetic elements over time. The <italic>S. pyogenes</italic> population may have maintained a state of host adaptation by maintaining stable genetic elements over long periods [<xref rid="ref34" ref-type="bibr">34</xref>].</p>
<p>This study had some limitations. Although our study spanned two decades and was population-based, only 87 strains were included, explaining why we did not observe significant genome changes during the study period. We investigated virulence-associated CDSs among all 87 strains by searching only annotated CDSs based on a functional annotation pipeline of whole-genome assemblies rather than by searching the sequences around specific genes or by PCR simulation. We searched for related articles by entering the search terms &#8220;<italic>Streptococcus pyogenes</italic>,&#8221; &#8220;whole genome,&#8221; or &#8220;Korea&#8221; into the PubMed database (<ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/">https://pubmed.ncbi.nlm.nih.gov/</ext-link>). However, there were no hits for related manuscripts as of May 26, 2021. This is probably the first report on WGS datasets of <italic>S. pyogenes</italic> strains from Korea.</p>
<p>In conclusion, this study provided CG characteristics of <italic>S. pyogenes</italic> according to invasiveness over a 20-year period. Genomic dynamics were stable during this time span. Four genes, <italic>csn1</italic>, <italic>ispE</italic>, <italic>nisK</italic>, and <italic>citC</italic>, are candidate virulence-associated CDSs in host&#8211;pathogen interactions of invasive <italic>S. pyogenes</italic> strains. Our results showed considerable agreement with previous epidemiological study results, especially regarding the predominant invasive genotypes, i.e., <italic>emm</italic>1.0, <italic>emm</italic>1.18, <italic>emm</italic>1.3, and <italic>emm</italic>1.76. ST28 showed a clonal distribution of invasive strains. Further epidemiological studies using WGS datasets are needed to better understand and monitor streptococcal virulence.</p>
</sec>
<sec sec-type="supplementary-material">
<title>Supplemental Materials</title>
<supplementary-material id ="S1" content-type="local-data">
<media xlink:href="alm-42-4-438-supple.pdf" mimetype="application" mime-subtype="pdf"/>
</supplementary-material>
</sec>
</body>
<back>
<fn-group>
<fn fn-type="con">
<p><bold>AUTHOR CONTRIBUTIONS</bold></p>
<p>Kim S, Choi E, and Takahashi T conceptualized the study; Shin H, Choi E, and Lee S collected the data; Shin H, Maeda T, Fukushima Y, Lee S, Kim S, and Takahashi T analyzed the data; Shin H wrote the manuscript; Kim S and Takahashi T reviewed and edited the manuscript; all authors reviewed and approved the manuscript.</p>
</fn>
<fn fn-type="conflict">
<p><bold>CONFLICTS OF INTEREST</bold></p>
<p>None declared.</p>
</fn>
<fn fn-type="supported-by">
<p><bold>RESEARCH FUNDING</bold></p>
<p>This work was supported by a grant from the Korea Health Technology R&#38;D Project through the Korea Health Industry Development Institute (KHIDI), funded by the Ministry of Health &#38; Welfare (H19C0047), and Bio &#38; Medical Technology Development Program of the National Research Foundation (NRF) (2021M3E5E3080382, 2021R1I1A3044483) by the Korean government. The funders had no role in study design, data collection and interpretation, decision to publish, or preparation of the manuscript.</p>
</fn>
</fn-group>
<ref-list>
<title>REFERENCES</title>
<ref id="ref1">
<label>1</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Stevens</surname><given-names>DL</given-names></name>
<name><surname>Tanner</surname><given-names>MH</given-names></name>
<name><surname>Winship</surname><given-names>J</given-names></name>
<name><surname>Swarts</surname><given-names>R</given-names></name>
<name><surname>Ries</surname><given-names>KM</given-names></name>
<name><surname>Schlievert</surname><given-names>PM</given-names></name>
<etal/>
</person-group>
<year>1989</year>
<article-title>Severe group A streptococcal infections associated with a toxic shock-like syndrome and scarlet fever toxin A</article-title>
<source>N Engl J Med</source>
<volume>321</volume>
<fpage>1</fpage>
<lpage>7</lpage>
<pub-id pub-id-type="doi">10.1056/NEJM198907063210101</pub-id>
<pub-id pub-id-type="pmid">2659990</pub-id>
</element-citation>
</ref>
<ref id="ref2">
<label>2</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Carapetis</surname><given-names>JR</given-names></name>
<name><surname>Steer</surname><given-names>AC</given-names></name>
<name><surname>Mulholland</surname><given-names>EK</given-names></name>
<name><surname>Weber</surname><given-names>M</given-names></name>
</person-group>
<year>2005</year>
<article-title>The global burden of group A streptococcal diseases</article-title>
<source>Lancet Infect Dis</source>
<volume>5</volume>
<fpage>685</fpage>
<lpage>94</lpage>
<pub-id pub-id-type="doi">10.1016/S1473-3099(05)70267-X</pub-id>
<pub-id pub-id-type="pmid">16253886</pub-id>
</element-citation>
</ref>
<ref id="ref3">
<label>3</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Enright</surname><given-names>MC</given-names></name>
<name><surname>Spratt</surname><given-names>BG</given-names></name>
<name><surname>Kalia</surname><given-names>A</given-names></name>
<name><surname>Cross</surname><given-names>JH</given-names></name>
<name><surname>Bessen</surname><given-names>DE</given-names></name>
</person-group>
<year>2001</year>
<article-title>Multilocus sequence typing of <italic>Streptococcus pyogenes</italic> and the relationships between <italic>emm</italic> type and clone</article-title>
<source>Infect Immun</source>
<volume>69</volume>
<fpage>2416</fpage>
<lpage>27</lpage>
<pub-id pub-id-type="doi">10.1128/IAI.69.4.2416-2427.2001</pub-id>
<pub-id pub-id-type="pmid">11254602</pub-id>
<pub-id pub-id-type="pmcid">PMC98174</pub-id>
</element-citation>
</ref>
<ref id="ref4">
<label>4</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Luca</surname><given-names>AV</given-names></name>
<name><surname>Giovanni</surname><given-names>G</given-names></name>
<name><surname>Dezemona</surname><given-names>P</given-names></name>
</person-group>
<article-title>Pulsed field gel electrophoresis of group A streptococci</article-title>
<source>Methods Mol Biol</source>
<year>2015</year>
<volume>1301</volume>
<fpage>129</fpage>
<lpage>38</lpage>
<pub-id pub-id-type="doi">10.1007/978-1-4939-2599-5_12</pub-id>
<pub-id pub-id-type="pmid">25862054</pub-id>
</element-citation>
</ref>
<ref id="ref5">
<label>5</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Lewis</surname><given-names>T</given-names></name>
<name><surname>Loman</surname><given-names>NJ</given-names></name>
<name><surname>Bingle</surname><given-names>L</given-names></name>
<name><surname>Jumaa</surname><given-names>P</given-names></name>
<name><surname>Weinstock</surname><given-names>GM</given-names></name>
<name><surname>Mortiboy</surname><given-names>D</given-names></name>
<etal/>
</person-group>
<year>2010</year>
<article-title>High-throughput whole-genome sequencing to dissect the epidemiology of <italic>Acinetobacter baumannii</italic> isolates from a hospital outbreak</article-title>
<source>J Hosp Infect</source>
<volume>75</volume>
<fpage>37</fpage>
<lpage>41</lpage>
<pub-id pub-id-type="doi">10.1016/j.jhin.2010.01.012</pub-id>
<pub-id pub-id-type="pmid">20299126</pub-id>
</element-citation>
</ref>
<ref id="ref6">
<label>6</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Schuchat</surname><given-names>A</given-names></name>
<name><surname>Hilger</surname><given-names>T</given-names></name>
<name><surname>Zell</surname><given-names>E</given-names></name>
<name><surname>Farley</surname><given-names>MM</given-names></name>
<name><surname>Reingold</surname><given-names>A</given-names></name>
<name><surname>Harrison</surname><given-names>L</given-names></name>
<etal/>
</person-group>
<year>2001</year>
<article-title>Active bacterial core surveillance of the emerging infections program network</article-title>
<source>Emerg Infect Dis</source>
<volume>7</volume>
<fpage>92</fpage>
<lpage>9</lpage>
<pub-id pub-id-type="doi">10.3201/eid0701.010114</pub-id>
<pub-id pub-id-type="pmid">11266299</pub-id>
<pub-id pub-id-type="pmcid">PMC2631675</pub-id>
</element-citation>
</ref>
<ref id="ref7">
<label>7</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Hyatt</surname><given-names>D</given-names></name>
<name><surname>Chen</surname><given-names>GL</given-names></name>
<name><surname>LoCascio</surname><given-names>PF</given-names></name>
<name><surname>Land</surname><given-names>ML</given-names></name>
<name><surname>Larimer</surname><given-names>FW</given-names></name>
<name><surname>Hauser</surname><given-names>LJ</given-names></name>
</person-group>
<year>2010</year>
<article-title>Prodigal: prokaryotic gene recognition and translation initiation site identification</article-title>
<source>BMC Bioinformatics</source>
<volume>11</volume>
<fpage>1</fpage>
<lpage>11</lpage>
<pub-id pub-id-type="doi">10.1186/1471-2105-11-119</pub-id>
<pub-id pub-id-type="pmid">20211023</pub-id>
<pub-id pub-id-type="pmcid">PMC2848648</pub-id>
</element-citation>
</ref>
<ref id="ref8">
<label>8</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Chun</surname><given-names>J</given-names></name>
<name><surname>Grim</surname><given-names>CJ</given-names></name>
<name><surname>Hasan</surname><given-names>NA</given-names></name>
<name><surname>Lee</surname><given-names>JH</given-names></name>
<name><surname>Choi</surname><given-names>SY</given-names></name>
<name><surname>Haley</surname><given-names>BJ</given-names></name>
<etal/>
</person-group>
<year>2009</year>
<article-title>Comparative genomics reveals mechanism for short-term and long-term clonal transitions in pandemic <italic>Vibrio cholerae</italic></article-title>
<source>Proc Natl Acad Sci U S A</source>
<volume>106</volume>
<fpage>15442</fpage>
<lpage>7</lpage>
<pub-id pub-id-type="doi">10.1073/pnas.0907787106</pub-id>
<pub-id pub-id-type="pmid">19720995</pub-id>
<pub-id pub-id-type="pmcid">PMC2741270</pub-id>
</element-citation>
</ref>
<ref id="ref9">
<label>9</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Tanizawa</surname><given-names>Y</given-names></name>
<name><surname>Fujisawa</surname><given-names>T</given-names></name>
<name><surname>Nakamura</surname><given-names>Y</given-names></name>
</person-group>
<year>2018</year>
<article-title>DFAST: a flexible prokaryotic genome annotation pipeline for faster genome publication</article-title>
<source>Bioinformatics</source>
<volume>34</volume>
<fpage>1037</fpage>
<lpage>9</lpage>
<pub-id pub-id-type="doi">10.1093/bioinformatics/btx713</pub-id>
<pub-id pub-id-type="pmid">29106469</pub-id>
<pub-id pub-id-type="pmcid">PMC5860143</pub-id>
</element-citation>
</ref>
<ref id="ref10">
<label>10</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Takahashi</surname><given-names>T</given-names></name>
<name><surname>Arai</surname><given-names>K</given-names></name>
<name><surname>Lee</surname><given-names>DH</given-names></name>
<name><surname>Koh</surname><given-names>EH</given-names></name>
<name><surname>Yoshida</surname><given-names>H</given-names></name>
<name><surname>Yano</surname><given-names>H</given-names></name>
<etal/>
</person-group>
<year>2016</year>
<article-title>Epidemiological study of erythromycin-resistant <italic>Streptococcus pyogenes</italic> from Korea and Japan by <italic>emm</italic> genotyping and multilocus sequence typing</article-title>
<source>Ann Lab Med</source>
<volume>36</volume>
<fpage>9</fpage>
<lpage>14</lpage>
<pub-id pub-id-type="doi">10.3343/alm.2016.36.1.9</pub-id>
<pub-id pub-id-type="pmid">26522753</pub-id>
<pub-id pub-id-type="pmcid">PMC4697353</pub-id>
</element-citation>
</ref>
<ref id="ref11">
<label>11</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Kim</surname><given-names>S</given-names></name>
<name><surname>Byun</surname><given-names>JH</given-names></name>
<name><surname>Park</surname><given-names>H</given-names></name>
<name><surname>Lee</surname><given-names>J</given-names></name>
<name><surname>Lee</surname><given-names>HS</given-names></name>
<name><surname>Yoshida</surname><given-names>H</given-names></name>
<etal/>
</person-group>
<year>2018</year>
<article-title>Molecular epidemiological features and antibiotic susceptibility patterns of <italic>Streptococcus dysgalactiae</italic> subsp. <italic>equisimilis</italic> isolates from Korea and Japan</article-title>
<source>Ann Lab Med</source>
<volume>38</volume>
<fpage>212</fpage>
<lpage>9</lpage>
<pub-id pub-id-type="doi">10.3343/alm.2018.38.3.212</pub-id>
<pub-id pub-id-type="pmid">29401555</pub-id>
<pub-id pub-id-type="pmcid">PMC5820065</pub-id>
</element-citation>
</ref>
<ref id="ref12">
<label>12</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Tamayo</surname><given-names>E</given-names></name>
<name><surname>Montes</surname><given-names>M</given-names></name>
<name><surname>Vicente</surname><given-names>D</given-names></name>
<name><surname>P&#233;rez-Trallero</surname><given-names>E</given-names></name>
</person-group>
<year>2016</year>
<article-title><italic>Streptococcus pyogenes</italic> pneumonia in adults: clinical presentation and molecular characterization of isolates 2006-2015</article-title>
<source>PLoS One</source>
<volume>11</volume>
<elocation-id>e0152640</elocation-id>
<pub-id pub-id-type="doi">10.1371/journal.pone.0152640</pub-id>
<pub-id pub-id-type="pmid">27027618</pub-id>
<pub-id pub-id-type="pmcid">PMC4814053</pub-id>
</element-citation>
</ref>
<ref id="ref13">
<label>13</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Sakai</surname><given-names>T</given-names></name>
<name><surname>Taniyama</surname><given-names>D</given-names></name>
<name><surname>Takahashi</surname><given-names>S</given-names></name>
<name><surname>Nakamura</surname><given-names>M</given-names></name>
<name><surname>Takahashi</surname><given-names>T</given-names></name>
</person-group>
<year>2017</year>
<article-title>Pleural empyema and streptococcal toxic shock syndrome due to <italic>Streptococcus pyogenes</italic> in a healthy Spanish traveler in Japan</article-title>
<source>IDCases</source>
<volume>9</volume>
<fpage>85</fpage>
<lpage>8</lpage>
<pub-id pub-id-type="doi">10.1016/j.idcr.2017.06.006</pub-id>
<pub-id pub-id-type="pmid">28725562</pub-id>
<pub-id pub-id-type="pmcid">PMC5506862</pub-id>
</element-citation>
</ref>
<ref id="ref14">
<label>14</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Takahashi</surname><given-names>T</given-names></name>
<name><surname>Maeda</surname><given-names>T</given-names></name>
<name><surname>Lee</surname><given-names>S</given-names></name>
<name><surname>Lee</surname><given-names>DH</given-names></name>
<name><surname>Kim</surname><given-names>S</given-names></name>
</person-group>
<year>2020</year>
<article-title>Clonal distribution of clindamycin-resistant erythromycin-susceptible (CRES) <italic>Streptococcus agalactiae</italic> in Korea based on whole genome sequences</article-title>
<source>Ann Lab Med</source>
<volume>40</volume>
<fpage>370</fpage>
<lpage>81</lpage>
<pub-id pub-id-type="doi">10.3343/alm.2020.40.5.370</pub-id>
<pub-id pub-id-type="pmid">32311850</pub-id>
<pub-id pub-id-type="pmcid">PMC7169627</pub-id>
</element-citation>
</ref>
<ref id="ref15">
<label>15</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Larsen</surname><given-names>MV</given-names></name>
<name><surname>Cosentino</surname><given-names>S</given-names></name>
<name><surname>Rasmussen</surname><given-names>S</given-names></name>
<name><surname>Friis</surname><given-names>C</given-names></name>
<name><surname>Hasman</surname><given-names>H</given-names></name>
<name><surname>Marvig</surname><given-names>RL</given-names></name>
<etal/>
</person-group>
<year>2012</year>
<article-title>Multilocus sequence typing of total-genome-sequenced bacteria</article-title>
<source>J Clin Microbiol</source>
<volume>50</volume>
<fpage>1355</fpage>
<lpage>61</lpage>
<pub-id pub-id-type="doi">10.1128/JCM.06094-11</pub-id>
<pub-id pub-id-type="pmid">22238442</pub-id>
<pub-id pub-id-type="pmcid">PMC3318499</pub-id>
</element-citation>
</ref>
<ref id="ref16">
<label>16</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Nascimento</surname><given-names>M</given-names></name>
<name><surname>Sousa</surname><given-names>A</given-names></name>
<name><surname>Ramirez</surname><given-names>M</given-names></name>
<name><surname>Francisco</surname><given-names>AP</given-names></name>
<name><surname>Carri&#231;o</surname><given-names>JA</given-names></name>
<name><surname>Vaz</surname><given-names>C</given-names></name>
</person-group>
<year>2017</year>
<article-title>PHYLOViZ 2.0: providing scalable data integration and visualization for multiple phylogenetic inference methods</article-title>
<source>Bioinformatics</source>
<volume>33</volume>
<fpage>128</fpage>
<lpage>9</lpage>
<pub-id pub-id-type="doi">10.1093/bioinformatics/btw582</pub-id>
<pub-id pub-id-type="pmid">27605102</pub-id>
</element-citation>
</ref>
<ref id="ref17">
<label>17</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Steer</surname><given-names>AC</given-names></name>
<name><surname>Law</surname><given-names>I</given-names></name>
<name><surname>Matatolu</surname><given-names>L</given-names></name>
<name><surname>Beall</surname><given-names>BW</given-names></name>
<name><surname>Carapetis</surname><given-names>JR</given-names></name>
</person-group>
<year>2009</year>
<article-title>Global <italic>emm</italic> type distribution of group A streptococci: systematic review and implications for vaccine development</article-title>
<source>Lancet Infect Dis</source>
<volume>9</volume>
<fpage>611</fpage>
<lpage>6</lpage>
<pub-id pub-id-type="doi">10.1016/S1473-3099(09)70178-1</pub-id>
<pub-id pub-id-type="pmid">19778763</pub-id>
</element-citation>
</ref>
<ref id="ref18">
<label>18</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Luca-Harari</surname><given-names>B</given-names></name>
<name><surname>Darenberg</surname><given-names>J</given-names></name>
<name><surname>Neal</surname><given-names>S</given-names></name>
<name><surname>Siljander</surname><given-names>T</given-names></name>
<name><surname>Strakova</surname><given-names>L</given-names></name>
<name><surname>Tanna</surname><given-names>A</given-names></name>
<etal/>
</person-group>
<year>2009</year>
<article-title>Clinical and microbiological characteristics of severe <italic>Streptococcus pyogenes</italic> disease in Europe</article-title>
<source>J Clin Microbiol</source>
<volume>47</volume>
<fpage>1155</fpage>
<lpage>65</lpage>
<pub-id pub-id-type="doi">10.1128/JCM.02155-08</pub-id>
<pub-id pub-id-type="pmid">19158266</pub-id>
<pub-id pub-id-type="pmcid">PMC2668334</pub-id>
</element-citation>
</ref>
<ref id="ref19">
<label>19</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Ekelund</surname><given-names>K</given-names></name>
<name><surname>Darenberg</surname><given-names>J</given-names></name>
<name><surname>Norrby-Teglund</surname><given-names>A</given-names></name>
<name><surname>Hoffmann</surname><given-names>S</given-names></name>
<name><surname>Bang</surname><given-names>D</given-names></name>
<name><surname>Skinh&#248;j</surname><given-names>P</given-names></name>
<etal/>
</person-group>
<year>2005</year>
<article-title>Variations in <italic>emm</italic> type among group A streptococcal isolates causing invasive or noninvasive infections in a nationwide study</article-title>
<source>J Clin Microbiol</source>
<volume>43</volume>
<fpage>3101</fpage>
<lpage>9</lpage>
<pub-id pub-id-type="doi">10.1128/JCM.43.7.3101-3109.2005</pub-id>
<pub-id pub-id-type="pmid">16000420</pub-id>
<pub-id pub-id-type="pmcid">PMC1169105</pub-id>
</element-citation>
</ref>
<ref id="ref20">
<label>20</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Enright</surname><given-names>MC</given-names></name>
<name><surname>Day</surname><given-names>NP</given-names></name>
<name><surname>Davies</surname><given-names>CE</given-names></name>
<name><surname>Peacock</surname><given-names>SJ</given-names></name>
<name><surname>Spratt</surname><given-names>BG</given-names></name>
</person-group>
<year>2000</year>
<article-title>Multilocus sequence typing for characterization of methicillin-resistant and methicillin-susceptible clones of <italic>Staphylococcus aureus</italic></article-title>
<source>J Clin Microbiol</source>
<volume>38</volume>
<fpage>1008</fpage>
<lpage>15</lpage>
<pub-id pub-id-type="doi">10.1128/JCM.38.3.1008-1015.2000</pub-id>
<pub-id pub-id-type="pmid">10698988</pub-id>
<pub-id pub-id-type="pmcid">PMC86325</pub-id>
</element-citation>
</ref>
<ref id="ref21">
<label>21</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Feil</surname><given-names>EJ</given-names></name>
<name><surname>Li</surname><given-names>BC</given-names></name>
<name><surname>Aanensen</surname><given-names>DM</given-names></name>
<name><surname>Hanage</surname><given-names>WP</given-names></name>
<name><surname>Spratt</surname><given-names>BG</given-names></name>
</person-group>
<year>2004</year>
<article-title>eBURST: inferring patterns of evolutionary descent among clusters of related bacterial genotypes from multilocus sequence typing data</article-title>
<source>J Bacteriol</source>
<volume>186</volume>
<fpage>1518</fpage>
<lpage>30</lpage>
<pub-id pub-id-type="doi">10.1128/JB.186.5.1518-1530.2004</pub-id>
<pub-id pub-id-type="pmid">14973027</pub-id>
<pub-id pub-id-type="pmcid">PMC344416</pub-id>
</element-citation>
</ref>
<ref id="ref22">
<label>22</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Le Rhun</surname><given-names>A</given-names></name>
<name><surname>Escalera-Maurer</surname><given-names>A</given-names></name>
<name><surname>Bratovi&#269;</surname><given-names>M</given-names></name>
<name><surname>Charpentier</surname><given-names>E</given-names></name>
</person-group>
<year>2019</year>
<article-title>CRISPR-Cas in <italic>Streptococcus pyogenes</italic></article-title>
<source>RNA Biol</source>
<volume>16</volume>
<fpage>380</fpage>
<lpage>9</lpage>
<pub-id pub-id-type="doi">10.1080/15476286.2019.1582974</pub-id>
<pub-id pub-id-type="pmid">30856357</pub-id>
<pub-id pub-id-type="pmcid">PMC6546361</pub-id>
</element-citation>
</ref>
<ref id="ref23">
<label>23</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Nozawa</surname><given-names>T</given-names></name>
<name><surname>Furukawa</surname><given-names>N</given-names></name>
<name><surname>Aikawa</surname><given-names>C</given-names></name>
<name><surname>Watanabe</surname><given-names>T</given-names></name>
<name><surname>Haobam</surname><given-names>B</given-names></name>
<name><surname>Kurokawa</surname><given-names>K</given-names></name>
<etal/>
</person-group>
<year>2011</year>
<article-title>CRISPR inhibition of prophage acquisition in <italic>Streptococcus pyogenes</italic></article-title>
<source>PLoS One</source>
<volume>6</volume>
<elocation-id>e19543</elocation-id>
<pub-id pub-id-type="doi">10.1371/journal.pone.0019543</pub-id>
<pub-id pub-id-type="pmid">21573110</pub-id>
<pub-id pub-id-type="pmcid">PMC3089615</pub-id>
</element-citation>
</ref>
<ref id="ref24">
<label>24</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Heuston</surname><given-names>S</given-names></name>
<name><surname>Begley</surname><given-names>M</given-names></name>
<name><surname>Gahan</surname><given-names>CGM</given-names></name>
<name><surname>Hill</surname><given-names>C</given-names></name>
</person-group>
<year>2012</year>
<article-title>Isoprenoid biosynthesis in bacterial pathogens</article-title>
<source>Microbiology (Reading)</source>
<volume>158</volume>
<fpage>1389</fpage>
<lpage>401</lpage>
<pub-id pub-id-type="doi">10.1099/mic.0.051599-0</pub-id>
<pub-id pub-id-type="pmid">22466083</pub-id>
</element-citation>
</ref>
<ref id="ref25">
<label>25</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Voynova</surname><given-names>NE</given-names></name>
<name><surname>Rios</surname><given-names>SE</given-names></name>
<name><surname>Miziorko</surname><given-names>HM</given-names></name>
</person-group>
<year>2004</year>
<article-title><italic>Staphylococcus aureus</italic> mevalonate kinase: isolation and characterization of an enzyme of the isoprenoid biosynthetic pathway</article-title>
<source>J Bacteriol</source>
<volume>186</volume>
<fpage>61</fpage>
<lpage>7</lpage>
<pub-id pub-id-type="doi">10.1128/JB.186.1.61-67.2004</pub-id>
<pub-id pub-id-type="pmid">14679225</pub-id>
<pub-id pub-id-type="pmcid">PMC303434</pub-id>
</element-citation>
</ref>
<ref id="ref26">
<label>26</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Bekal</surname><given-names>S</given-names></name>
<name><surname>Van Beeumen</surname><given-names>J</given-names></name>
<name><surname>Samyn</surname><given-names>B</given-names></name>
<name><surname>Garmyn</surname><given-names>D</given-names></name>
<name><surname>Henini</surname><given-names>S</given-names></name>
<name><surname>Divi&#232;s</surname><given-names>C</given-names></name>
<etal/>
</person-group>
<year>1998</year>
<article-title>Purification of <italic>Leuconostoc mesenteroides</italic> citrate lyase and cloning and characterization of the <italic>citC</italic>DEFG gene cluster</article-title>
<source>J Bacteriol</source>
<volume>180</volume>
<fpage>647</fpage>
<lpage>54</lpage>
<pub-id pub-id-type="doi">10.1128/JB.180.3.647-654.1998</pub-id>
<pub-id pub-id-type="pmid">9457870</pub-id>
<pub-id pub-id-type="pmcid">PMC106934</pub-id>
</element-citation>
</ref>
<ref id="ref27">
<label>27</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Kawada-Matsuo</surname><given-names>M</given-names></name>
<name><surname>Tatsuno</surname><given-names>I</given-names></name>
<name><surname>Arii</surname><given-names>K</given-names></name>
<name><surname>Zendo</surname><given-names>T</given-names></name>
<name><surname>Oogai</surname><given-names>Y</given-names></name>
<name><surname>Noguchi</surname><given-names>K</given-names></name>
<etal/>
</person-group>
<year>2016</year>
<article-title>Two-component systems involved in susceptibility to nisin A in <italic>Streptococcus pyogenes</italic></article-title>
<source>Appl Environ Microbiol</source>
<volume>82</volume>
<fpage>5930</fpage>
<lpage>9</lpage>
<pub-id pub-id-type="doi">10.1128/AEM.01897-16</pub-id>
<pub-id pub-id-type="pmid">27474716</pub-id>
<pub-id pub-id-type="pmcid">PMC5038018</pub-id>
</element-citation>
</ref>
<ref id="ref28">
<label>28</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Quadri</surname><given-names>LE</given-names></name>
</person-group>
<year>2002</year>
<article-title>Regulation of antimicrobial peptide production by autoinducer-mediated quorum sensing in lactic acid bacteria</article-title>
<source>Antonie Van Leeuwenhoek</source>
<volume>82</volume>
<fpage>133</fpage>
<lpage>45</lpage>
<pub-id pub-id-type="doi">10.1023/A:1020624808520</pub-id>
<pub-id pub-id-type="pmid">12369185</pub-id>
</element-citation>
</ref>
<ref id="ref29">
<label>29</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Hu</surname><given-names>J</given-names></name>
<name><surname>Jin</surname><given-names>K</given-names></name>
<name><surname>He</surname><given-names>ZG</given-names></name>
<name><surname>Zhang</surname><given-names>H</given-names></name>
</person-group>
<year>2020</year>
<article-title>Citrate lyase CitE in <italic>Mycobacterium tuberculosis</italic> contributes to mycobacterial survival under hypoxic conditions</article-title>
<source>PLoS One</source>
<volume>15</volume>
<elocation-id>e0230786</elocation-id>
<pub-id pub-id-type="doi">10.1371/journal.pone.0230786</pub-id>
<pub-id pub-id-type="pmid">32302313</pub-id>
<pub-id pub-id-type="pmcid">PMC7164622</pub-id>
</element-citation>
</ref>
<ref id="ref30">
<label>30</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Weeks</surname><given-names>CR</given-names></name>
<name><surname>Ferretti</surname><given-names>JJ</given-names></name>
</person-group>
<year>1986</year>
<article-title>Nucleotide sequence of the type A streptococcal exotoxin (erythrogenic toxin) gene from <italic>Streptococcus pyogenes</italic> bacteriophage T12</article-title>
<source>Infect Immun</source>
<volume>52</volume>
<fpage>144</fpage>
<lpage>50</lpage>
<pub-id pub-id-type="doi">10.1128/iai.52.1.144-150.1986</pub-id>
<pub-id pub-id-type="pmid">3514452</pub-id>
<pub-id pub-id-type="pmcid">PMC262210</pub-id>
</element-citation>
</ref>
<ref id="ref31">
<label>31</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Yu</surname><given-names>CE</given-names></name>
<name><surname>Ferretti</surname><given-names>JJ</given-names></name>
</person-group>
<year>1989</year>
<article-title>Molecular epidemiologic analysis of the type A streptococcal exotoxin (erythrogenic toxin) gene (<italic>speA</italic>) in clinical <italic>Streptococcus pyogenes</italic> strains</article-title>
<source>Infect Immun</source>
<volume>57</volume>
<fpage>3715</fpage>
<lpage>9</lpage>
<pub-id pub-id-type="doi">10.1128/iai.57.12.3715-3719.1989</pub-id>
<pub-id pub-id-type="pmid">2553612</pub-id>
<pub-id pub-id-type="pmcid">PMC259895</pub-id>
</element-citation>
</ref>
<ref id="ref32">
<label>32</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Shibasaki</surname><given-names>T</given-names></name>
<name><surname>Mori</surname><given-names>H</given-names></name>
<name><surname>Chiba</surname><given-names>S</given-names></name>
<name><surname>Ozaki</surname><given-names>A</given-names></name>
</person-group>
<year>1999</year>
<article-title>Microbial proline 4-hydroxylase screening and gene cloning</article-title>
<source>Appl Environ Microbiol</source>
<volume>65</volume>
<fpage>4028</fpage>
<lpage>31</lpage>
<pub-id pub-id-type="doi">10.1128/AEM.65.9.4028-4031.1999</pub-id>
<pub-id pub-id-type="pmid">10473412</pub-id>
<pub-id pub-id-type="pmcid">PMC99737</pub-id>
</element-citation>
</ref>
<ref id="ref33">
<label>33</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Mori</surname><given-names>H</given-names></name>
<name><surname>Shibasaki</surname><given-names>T</given-names></name>
<name><surname>Uozaki</surname><given-names>Y</given-names></name>
<name><surname>Ochiai</surname><given-names>K</given-names></name>
<name><surname>Ozaki</surname><given-names>A</given-names></name>
</person-group>
<year>1996</year>
<article-title>Detection of novel proline 3-hydroxylase activities in <italic>Streptomyces</italic> and <italic>Bacillus</italic> spp. by regio- and streospecific hydroxylation of L-proline</article-title>
<source>Appl Environ Microbiol</source>
<volume>62</volume>
<fpage>1903</fpage>
<lpage>7</lpage>
<pub-id pub-id-type="doi">10.1128/aem.62.6.1903-1907.1996</pub-id>
<pub-id pub-id-type="pmid">16535329</pub-id>
<pub-id pub-id-type="pmcid">PMC1388867</pub-id>
</element-citation>
</ref>
<ref id="ref34">
<label>34</label>
<element-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Park</surname><given-names>HJ</given-names></name>
<name><surname>Gokhale</surname><given-names>CS</given-names></name>
<name><surname>Bertels</surname><given-names>F</given-names></name>
</person-group>
<year>2021</year>
<article-title>How sequence populations persist inside bacterial genomes</article-title>
<source>Genetics</source>
<volume>217</volume>
<elocation-id>iyab027</elocation-id>
<pub-id pub-id-type="doi">10.1093/genetics/iyab027</pub-id>
<pub-id pub-id-type="pmid">33724360</pub-id>
<pub-id pub-id-type="pmcid">PMC8049555</pub-id>
</element-citation>
</ref>
</ref-list>
<sec sec-type="display-objects">
<title>Figures and Tables</title>
<fig id="F1" position="float">
<label>Fig. 1</label>
<caption>
<p>Flow chart of strain selection according to isolation year and source of <italic>Streptococcus pyogenes</italic> isolates.</p>
</caption>
<graphic xlink:href="alm-42-4-438-f1.tif"/>
</fig>
<fig id="F2" position="float">
<label>Fig. 2</label>
<caption>
<p>Distribution of <italic>emm</italic> clusters according to invasiveness (N=48).</p>
</caption>
<graphic xlink:href="alm-42-4-438-f2.tif"/>
</fig>
<fig id="F3" position="float">
<label>Fig. 3</label>
<caption>
<p>goeBURST diagram of the relationships among STs according to invasiveness. The numbers in the circles indicate the STs, and the numbers near the lines indicate the number of different alleles between two connected STs. A putative CC is indicated by an outer dotted frame and corresponds to the STs with the highest number of single locus variants. ST785, a single locus variant of ST28, formed CC28.</p>
<p>Abbreviations: ST, sequence type; CC, clonal complex.</p>
</caption>
<graphic xlink:href="alm-42-4-438-f3.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>Table 1</label>
<caption>
<p>Primer sets used to amplify <italic>speA</italic>, <italic>speB</italic>, <italic>speC</italic>, <italic>ssa</italic>, and <italic>smeZ</italic></p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr style="background-color:#d8e2f1;">
<th valign="middle" align="left">Superantigen gene</th>
<th valign="middle" align="center">Forward</th>
<th valign="middle" align="center">Reverse</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><italic>speA</italic></td>
<td valign="top" align="left">5&#180;-TAAGAACCAAGAGATGG-3&#180;</td>
<td valign="top" align="left">5&#180;-ATTCTTGAGCAGTTACC-3&#180;</td>
</tr>
<tr style="background-color:#f4f7fc;">
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">Alternative <italic>speA</italic></td>
<td valign="top" align="left">5&#180;-CAAGAACCGAGAGATGT-3&#180;</td>
<td/>
</tr>
<tr>
<td valign="top" align="left"><italic>speB</italic></td>
<td valign="top" align="left">5&#180;-AAGAAGCAAAAGATAGC-3&#180;</td>
<td valign="top" align="left">5&#180;-TGGTAGAAGTTACGTCC-3&#180;</td>
</tr>
<tr style="background-color:#f4f7fc;">
<td valign="top" align="left"><italic>speC</italic></td>
<td valign="top" align="left">5&#180;-GATTTCTACTATTTCACC-3&#180;</td>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">5&#180;-AAATATCTGATCTAGTCCC-3&#180;</td>
</tr>
<tr>
<td valign="top" align="left"><italic>ssa</italic></td>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">5&#180;-GTGTAGAATTGAGGTAATTG-3&#180;</td>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">5&#180;-TAATATAGCCTGTCTCGTAC-3&#180;</td>
</tr>
<tr style="background-color:#f4f7fc;">
<td valign="top" align="left"><italic>smeZ</italic></td>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">5&#180;-TAACTCCTGAAAAGAGGCT-3&#180;</td>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">5&#180;-TTGTAGCTAGAACCAGAAG-3&#180;</td>
</tr>
<tr>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">Alternative <italic>smeZ</italic></td>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">5&#180;-TAGCTCCTGAAAAGAGGCT-3&#180;</td>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">5&#180;-TTGTAGTTAGAACCAGAAG-3&#180;</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T2" position="float">
<label>Table 2</label>
<caption>
<p>Presence or absence of pan-genome orthologous genes according to invasiveness</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr style="background-color:#d8e2f1;">
<th valign="middle" align="left">Gene</th>
<th valign="middle" align="center">Function</th>
<th valign="middle" align="center">Non-invasive (N=24)</th>
<th valign="middle" align="center">Invasive (N=24)</th>
<th valign="middle" align="center"><italic>P</italic></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><italic>IRC3</italic></td>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">ATP-binding, helicase, hydrolase, mitochondrion, nucleotide-binding, putative mitochondrial ATP-dependent helicase irc3</td>
<td valign="top" align="center">Present</td>
<td valign="top" align="center">Absent</td>
<td valign="top" align="center">0.0006</td>
</tr>
<tr style="background-color:#f4f7fc;">
<td valign="top" align="left"><italic>recG</italic></td>
<td valign="top" align="left">DNA helicase</td>
<td valign="top" align="center">Present</td>
<td valign="top" align="center">Absent</td>
<td valign="top" align="center">0.0094</td>
</tr>
<tr>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;"><italic>clpP, CLPP</italic></td>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">Cytoplasm, hydrolase, protease, serine protease, endopeptidase Clp</td>
<td valign="top" align="center">Present</td>
<td valign="top" align="center">Absent</td>
<td valign="top" align="center">0.0392</td>
</tr>
<tr style="background-color:#f4f7fc;">
<td valign="top" align="left"><italic>topB</italic></td>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">DNA-binding, isomerase, magnesium, metal-binding, topoisomerase, DNA topoisomerase</td>
<td valign="top" align="center">Present</td>
<td valign="top" align="center">Absent</td>
<td valign="top" align="center">0.0496</td>
</tr>
<tr>
<td valign="top" align="left"><italic>atoD</italic></td>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">Transferase, acetate CoA-transferase</td>
<td valign="top" align="center">Present</td>
<td valign="top" align="center">Absent</td>
<td valign="top" align="center">0.0496</td>
</tr>
<tr style="background-color:#f4f7fc;">
<td valign="top" align="left"><italic>K02476</italic></td>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">ATP-binding, cell membrane, kinase, membrane, nucleotide-binding, phosphoprotein, transferase, transmembrane, transmembrane helix, two-component regulatory system, histidine kinase</td>
<td valign="top" align="center">Present</td>
<td valign="top" align="center">Absent</td>
<td valign="top" align="center">0.0496</td>
</tr>
<tr>
<td valign="top" align="left"><italic>MTHFS</italic></td>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">5-Formyltetrahydrofolate cyclo-ligase</td>
<td valign="top" align="center">Present</td>
<td valign="top" align="center">Absent</td>
<td valign="top" align="center">0.0496</td>
</tr>
<tr style="background-color:#f4f7fc;">
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;"><italic>csn1</italic>, <italic>cas9</italic></td>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">Antiviral defense, DNA-binding, endonuclease, exonuclease, hydrolase, magnesium, manganese, metal-binding, nuclease, RNA-binding, CRISPR-associated endonuclease Cas9/Csn1</td>
<td valign="top" align="center">Absent</td>
<td valign="top" align="center">Present</td>
<td valign="top" align="center">0.0044</td>
</tr>
<tr>
<td valign="top" align="left"><italic>ispE</italic></td>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">ATP-binding, isoprene biosynthesis, kinase, nucleotide-binding, transferase, 4-(cytidine 5&#180;-diphospho)-2-C-methyl-D-erythritol kinase</td>
<td valign="top" align="center">Absent</td>
<td valign="top" align="center">Present</td>
<td valign="top" align="center">0.0094</td>
</tr>
<tr style="background-color:#f4f7fc;">
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;"><italic>nisK</italic>, <italic>spaK</italic></td>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">ATP-binding, cell membrane, kinase, membrane, nucleotide-binding, phosphoprotein, transferase, transmembrane, transmembrane helix, two-component regulatory system, histidine kinase</td>
<td valign="top" align="center">Absent</td>
<td valign="top" align="center">Present</td>
<td valign="top" align="center">0.0355</td>
</tr>
<tr>
<td valign="top" align="left"><italic>citC</italic></td>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">ATP-binding, ligase, nucleotide-binding, (citrate [pro-3S]-lyase) ligase</td>
<td valign="top" align="center">Absent</td>
<td valign="top" align="center">Present</td>
<td valign="top" align="center">0.0496</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="t2fn1">
<p>Abbreviation: CRISPR, clustered regularly interspaced short palindromic repeats.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T3" position="float">
<label>Table 3</label>
<caption>
<p>Comparison of CDSs among all 87 strains by searching annotated CDSs based on functional annotation pipeline of whole-genome assemblies</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr style="background-color:#d8e2f1;">
<th valign="middle" align="left">CDSs</th>
<th valign="middle" align="center">Non-invasive (N=63)</th>
<th valign="middle" align="center">Invasive (N=24)</th>
<th valign="middle" align="center"><italic>P</italic></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Chitinase</td>
<td valign="top" align="center">27 (42.9%)</td>
<td valign="top" align="center">14 (58.3%)</td>
<td valign="top" align="center">0.293</td>
</tr>
<tr style="background-color:#f4f7fc;">
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;"><bold>Exotoxin type A</bold></td>
<td valign="top" align="center"><bold>13 (20.6%)</bold></td>
<td valign="top" align="center"><bold>12 (50.0%)</bold></td>
<td valign="top" align="center"><bold>0.015</bold></td>
</tr>
<tr>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;"><bold>Procollagen-proline 3-dioxygenase</bold></td>
<td valign="top" align="center"><bold>8 (12.7%)</bold></td>
<td valign="top" align="center"><bold>8 (33.3%)</bold></td>
<td valign="top" align="center"><bold>0.035</bold></td>
</tr>
<tr style="background-color:#f4f7fc;">
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">Platelet binding protein GspB</td>
<td valign="top" align="center">13 (36.1%)</td>
<td valign="top" align="center">4 (16.7%)</td>
<td valign="top" align="center">0.771</td>
</tr>
<tr>
<td valign="top" align="left">C protein alpha-antigen</td>
<td valign="top" align="center">5 (7.9%)</td>
<td valign="top" align="center">3 (12.5%)</td>
<td valign="top" align="center">0.679</td>
</tr>
<tr style="background-color:#f4f7fc;">
<td valign="top" align="left">Glycoprotein-gp2</td>
<td valign="top" align="center">4 (6.3%)</td>
<td valign="top" align="center">3 (12.5%)</td>
<td valign="top" align="center">0.389</td>
</tr>
<tr>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">N-acetylmuramoyl-L-alanine amidase</td>
<td valign="top" align="center">1 (1.6%)</td>
<td valign="top" align="center">0 (0.0%)</td>
<td valign="top" align="center">1.000</td>
</tr>
<tr style="background-color:#f4f7fc;">
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">Trehalose transport system permease protein SugB</td>
<td valign="top" align="center">1 (1.6%)</td>
<td valign="top" align="center">0 (0.0%)</td>
<td valign="top" align="center">1.000</td>
</tr>
<tr>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">Serine-rich adhesin for platelets</td>
<td valign="top" align="center">1 (1.6%)</td>
<td valign="top" align="center">0 (0.0%)</td>
<td valign="top" align="center">1.000</td>
</tr>
<tr style="background-color:#f4f7fc;">
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">Deoxyribonuclease (Yes/No)</td>
<td valign="top" align="center">61 (96.8%)</td>
<td valign="top" align="center">24 (100.0%)</td>
<td valign="top" align="center">1.000</td>
</tr>
<tr>
<td valign="top" align="left" style="padding-left:10px; text-indent:-10px;">Hyaluronan.synthase (Yes/No)</td>
<td valign="top" align="center">49 (77.8%)</td>
<td valign="top" align="center">22 (91.7%)</td>
<td valign="top" align="center">0.216</td>
</tr>
<tr style="background-color:#f4f7fc;">
<td valign="top" align="left">Streptopain (Yes/No)</td>
<td valign="top" align="center">63 (100.0%)</td>
<td valign="top" align="center">23 (95.8%)</td>
<td valign="top" align="center">0.276</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="t3fn1">
<p>The values are presented as N (%). Bold type indicates statistical significance.</p>
</fn>
<fn id="t3fn2">
<p>Abbreviation: CDS, coding DNA sequence.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
</back>
</article>