<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article SYSTEM "http://jats.nlm.nih.gov/archiving/1.2/JATS-archivearticle1.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="1.2" article-type="research-article" xml:lang="en"><?properties open_access?><front><journal-meta><journal-id journal-id-type="publisher-id">439</journal-id><journal-id journal-id-type="doi">10.1007/439.1432-1203</journal-id><journal-title-group><journal-title>Human Genetics</journal-title><abbrev-journal-title abbrev-type="publisher">Hum Genet</abbrev-journal-title></journal-title-group><issn pub-type="ppub">0340-6717</issn><issn pub-type="epub">1432-1203</issn><publisher><publisher-name>Springer Berlin Heidelberg</publisher-name><publisher-loc>Berlin/Heidelberg</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">s00439-020-02132-8</article-id><article-id pub-id-type="manuscript">2132</article-id><article-id pub-id-type="doi">10.1007/s00439-020-02132-8</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Investigation</subject></subj-group></article-categories><title-group><article-title xml:lang="en">Decoding a highly mixed Kazakh genome</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Seidualy</surname><given-names>Madina</given-names></name><xref ref-type="aff" rid="Aff1">1</xref></contrib><contrib contrib-type="author"><name><surname>Blazyte</surname><given-names>Asta</given-names></name><xref ref-type="aff" rid="Aff1">1</xref></contrib><contrib contrib-type="author"><name><surname>Jeon</surname><given-names>Sungwon</given-names></name><xref ref-type="aff" rid="Aff1">1</xref><xref ref-type="aff" rid="Aff2">2</xref></contrib><contrib contrib-type="author"><name><surname>Bhak</surname><given-names>Youngjune</given-names></name><xref ref-type="aff" rid="Aff1">1</xref><xref ref-type="aff" rid="Aff2">2</xref></contrib><contrib contrib-type="author"><name><surname>Jeon</surname><given-names>Yeonsu</given-names></name><xref ref-type="aff" rid="Aff1">1</xref><xref ref-type="aff" rid="Aff2">2</xref></contrib><contrib contrib-type="author"><name><surname>Kim</surname><given-names>Jungeun</given-names></name><xref ref-type="aff" rid="Aff3">3</xref></contrib><contrib contrib-type="author"><name><surname>Eriksson</surname><given-names>Anders</given-names></name><xref ref-type="aff" rid="Aff4">4</xref><xref ref-type="aff" rid="Aff5">5</xref></contrib><contrib contrib-type="author"><name><surname>Bolser</surname><given-names>Dan</given-names></name><xref ref-type="aff" rid="Aff6">6</xref></contrib><contrib contrib-type="author"><name><surname>Yoon</surname><given-names>Changhan</given-names></name><xref ref-type="aff" rid="Aff1">1</xref><xref ref-type="aff" rid="Aff2">2</xref></contrib><contrib contrib-type="author"><name><surname>Manica</surname><given-names>Andrea</given-names></name><xref ref-type="aff" rid="Aff7">7</xref></contrib><contrib contrib-type="author"><name><surname>Lee</surname><given-names>Semin</given-names></name><xref ref-type="aff" rid="Aff1">1</xref><xref ref-type="aff" rid="Aff2">2</xref></contrib><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid">http://orcid.org/0000-0002-4228-1299</contrib-id><name><surname>Bhak</surname><given-names>Jong</given-names></name><xref ref-type="aff" rid="Aff1">1</xref><xref ref-type="aff" rid="Aff2">2</xref><xref ref-type="aff" rid="Aff3">3</xref><xref ref-type="aff" rid="Aff8">8</xref><xref ref-type="corresp" rid="IDs00439020021328_cor12">l</xref></contrib><aff id="Aff1"><label>1</label><institution-wrap><institution-id institution-id-type="GRID">grid.42687.3f</institution-id><institution-id institution-id-type="ISNI">0000 0004 0381 814X</institution-id><institution content-type="org-division">Korean Genomics Center (KOGIC)</institution><institution content-type="org-name">Ulsan National Institute of Science and Technology (UNIST)</institution></institution-wrap><addr-line content-type="postcode">44919</addr-line><addr-line content-type="city">Ulsan</addr-line><country country="KR">Republic of Korea</country></aff><aff id="Aff2"><label>2</label><institution-wrap><institution-id institution-id-type="GRID">grid.42687.3f</institution-id><institution-id institution-id-type="ISNI">0000 0004 0381 814X</institution-id><institution content-type="org-division">Department of Biomedical Engineering, School of Life Sciences</institution><institution content-type="org-name">Ulsan National Institute of Science and Technology (UNIST)</institution></institution-wrap><addr-line content-type="postcode">44919</addr-line><addr-line content-type="city">Ulsan</addr-line><country country="KR">Republic of Korea</country></aff><aff id="Aff3"><label>3</label><institution-wrap><institution-id institution-id-type="GRID">grid.410888.d</institution-id><institution content-type="org-division">Personal Genomics Institute (PGI)</institution><institution content-type="org-name">Genome Research Foundation</institution></institution-wrap><addr-line content-type="postcode">28160</addr-line><addr-line content-type="city">Cheongju</addr-line><country country="KR">Republic of Korea</country></aff><aff id="Aff4"><label>4</label><institution-wrap><institution-id institution-id-type="GRID">grid.13097.3c</institution-id><institution-id institution-id-type="ISNI">0000 0001 2322 6764</institution-id><institution content-type="org-division">Department of Medical and Molecular Genetics</institution><institution content-type="org-name">King’s College London</institution></institution-wrap><addr-line content-type="postcode">SE1 9RT</addr-line><addr-line content-type="city">London</addr-line><country country="GB">UK</country></aff><aff id="Aff5"><label>5</label><institution-wrap><institution-id institution-id-type="GRID">grid.10939.32</institution-id><institution-id institution-id-type="ISNI">0000 0001 0943 7661</institution-id><institution content-type="org-division">cGEM, Institute of Genomics</institution><institution content-type="org-name">University of Tartu</institution></institution-wrap><addr-line content-type="street">Riia 23b</addr-line><addr-line content-type="postcode">51010</addr-line><addr-line content-type="city">Tartu</addr-line><country country="EE">Estonia</country></aff><aff id="Aff6"><label>6</label><institution-wrap><institution content-type="org-name">Geromics Ltd</institution></institution-wrap><addr-line content-type="street">Office 261, 23 Kings Street</addr-line><addr-line content-type="postcode">CB1 1AH</addr-line><addr-line content-type="city">Cambridge</addr-line><country country="GB">UK</country></aff><aff id="Aff7"><label>7</label><institution-wrap><institution-id institution-id-type="GRID">grid.5335.0</institution-id><institution-id institution-id-type="ISNI">0000000121885934</institution-id><institution content-type="org-division">Department of Zoology</institution><institution content-type="org-name">University of Cambridge</institution></institution-wrap><addr-line content-type="street">Downing Street</addr-line><addr-line content-type="postcode">CB2 3EJ</addr-line><addr-line content-type="city">Cambridge</addr-line><country country="GB">UK</country></aff><aff id="Aff8"><label>8</label><institution-wrap><institution-id institution-id-type="GRID">grid.42687.3f</institution-id><institution-id institution-id-type="ISNI">0000 0004 0381 814X</institution-id><institution content-type="org-division">Clinomics LTD</institution><institution content-type="org-name">Ulsan National Institute of Science and Technology (UNIST)</institution></institution-wrap><addr-line content-type="postcode">44919</addr-line><addr-line content-type="city">Ulsan</addr-line><country country="KR">Republic of Korea</country></aff></contrib-group><author-notes><corresp id="IDs00439020021328_cor12"><label>l</label><email>jongbhak@genomics.org</email></corresp></author-notes><pub-date date-type="epub"><day>19</day><month>2</month><year>2020</year></pub-date><pub-date date-type="ppub"><month>5</month><year>2020</year></pub-date><volume>139</volume><issue seq="1">5</issue><fpage>557</fpage><lpage>568</lpage><history><date date-type="registration"><day>5</day><month>2</month><year>2020</year></date><date date-type="received"><day>18</day><month>12</month><year>2019</year></date><date date-type="accepted"><day>5</day><month>2</month><year>2020</year></date><date date-type="online"><day>19</day><month>2</month><year>2020</year></date></history><permissions><copyright-statement>© The Author(s) 2020</copyright-statement><copyright-year>2020</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><license-p><bold>Open Access</bold>This article is licensed under a Creative Commons Attribution 4.0 International License, which permits use, sharing, adaptation, distribution and reproduction in any medium or format, as long as you give appropriate credit to the original author(s) and the source, provide a link to the Creative Commons licence, and indicate if changes were made. The images or other third party material in this article are included in the article's Creative Commons licence, unless indicated otherwise in a credit line to the material. If material is not included in the article's Creative Commons licence and your intended use is not permitted by statutory regulation or exceeds the permitted use, you will need to obtain permission directly from the copyright holder. To view a copy of this licence, visit <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">http://creativecommons.org/licenses/by/4.0/</ext-link>.</license-p></license></permissions><abstract id="Abs1" xml:lang="en"><title>Abstract</title><p id="Par1">We provide a Kazakh whole genome sequence (MJS) and analyses with the largest comparative Kazakh genomic data available to date. We found 102,240 novel SNVs and a high level of heterozygosity. ADMIXTURE analysis confirmed a significant proportion of variations in this individual coming from all continents except Africa and Oceania. A principal component analysis showed neighboring Kalmyk, Uzbek, and Kyrgyz populations to have the strongest resemblance to the MJS genome which reflects fairly recent Kazakh history. MJS’s mitochondrial haplogroup, J1c2, probably represents an early European and Near Eastern influence to Central Asia. This was also supported by the heterozygous SNPs associated with European phenotypic features and strikingly similar Kazakh ancestral composition inferred by ADMIXTURE. Admixture (<italic>f3</italic>) analysis showed that MJS’s genomic signature is best described as a cross between the Neolithic East Asian (Devil’s Gate1) and the Bronze Age European (Halberstadt_LBA1) components rather than a contemporary admixture.</p></abstract><funding-group><award-group><funding-source><institution-wrap><institution>U-K BRAND Research Fund</institution></institution-wrap></funding-source><award-id award-type="FundRef grant">1.190007.01</award-id><principal-award-recipient><name><surname>Bhak</surname><given-names>Jong</given-names></name></principal-award-recipient></award-group><award-group><funding-source><institution-wrap><institution>Ulsan City Research Fund</institution></institution-wrap></funding-source><award-id award-type="FundRef grant">1.190033.01</award-id><principal-award-recipient><name><surname>Bhak</surname><given-names>Jong</given-names></name></principal-award-recipient></award-group></funding-group><custom-meta-group><custom-meta><meta-name>publisher-imprint-name</meta-name><meta-value>Springer</meta-value></custom-meta><custom-meta><meta-name>volume-issue-count</meta-name><meta-value>12</meta-value></custom-meta><custom-meta><meta-name>issue-article-count</meta-name><meta-value>11</meta-value></custom-meta><custom-meta><meta-name>issue-toc-levels</meta-name><meta-value>0</meta-value></custom-meta><custom-meta><meta-name>issue-pricelist-year</meta-name><meta-value>2020</meta-value></custom-meta><custom-meta><meta-name>issue-copyright-holder</meta-name><meta-value>Springer-Verlag GmbH Germany, part of Springer Nature</meta-value></custom-meta><custom-meta><meta-name>issue-copyright-year</meta-name><meta-value>2020</meta-value></custom-meta><custom-meta><meta-name>article-contains-esm</meta-name><meta-value>Yes</meta-value></custom-meta><custom-meta><meta-name>article-numbering-style</meta-name><meta-value>Unnumbered</meta-value></custom-meta><custom-meta><meta-name>article-registration-date-year</meta-name><meta-value>2020</meta-value></custom-meta><custom-meta><meta-name>article-registration-date-month</meta-name><meta-value>2</meta-value></custom-meta><custom-meta><meta-name>article-registration-date-day</meta-name><meta-value>5</meta-value></custom-meta><custom-meta><meta-name>article-toc-levels</meta-name><meta-value>0</meta-value></custom-meta><custom-meta><meta-name>toc-levels</meta-name><meta-value>0</meta-value></custom-meta><custom-meta><meta-name>volume-type</meta-name><meta-value>Regular</meta-value></custom-meta><custom-meta><meta-name>journal-product</meta-name><meta-value>ArchiveJournal</meta-value></custom-meta><custom-meta><meta-name>numbering-style</meta-name><meta-value>Unnumbered</meta-value></custom-meta><custom-meta><meta-name>article-grants-type</meta-name><meta-value>OpenChoice</meta-value></custom-meta><custom-meta><meta-name>metadata-grant</meta-name><meta-value>OpenAccess</meta-value></custom-meta><custom-meta><meta-name>abstract-grant</meta-name><meta-value>OpenAccess</meta-value></custom-meta><custom-meta><meta-name>bodypdf-grant</meta-name><meta-value>OpenAccess</meta-value></custom-meta><custom-meta><meta-name>bodyhtml-grant</meta-name><meta-value>OpenAccess</meta-value></custom-meta><custom-meta><meta-name>bibliography-grant</meta-name><meta-value>OpenAccess</meta-value></custom-meta><custom-meta><meta-name>esm-grant</meta-name><meta-value>OpenAccess</meta-value></custom-meta><custom-meta><meta-name>online-first</meta-name><meta-value>false</meta-value></custom-meta><custom-meta><meta-name>pdf-file-reference</meta-name><meta-value>BodyRef/PDF/439_2020_Article_2132.pdf</meta-value></custom-meta><custom-meta><meta-name>pdf-type</meta-name><meta-value>Typeset</meta-value></custom-meta><custom-meta><meta-name>target-type</meta-name><meta-value>OnlinePDF</meta-value></custom-meta><custom-meta><meta-name>issue-online-date-year</meta-name><meta-value>2020</meta-value></custom-meta><custom-meta><meta-name>issue-online-date-month</meta-name><meta-value>4</meta-value></custom-meta><custom-meta><meta-name>issue-online-date-day</meta-name><meta-value>20</meta-value></custom-meta><custom-meta><meta-name>issue-print-date-year</meta-name><meta-value>2020</meta-value></custom-meta><custom-meta><meta-name>issue-print-date-month</meta-name><meta-value>4</meta-value></custom-meta><custom-meta><meta-name>issue-print-date-day</meta-name><meta-value>20</meta-value></custom-meta><custom-meta><meta-name>issue-type</meta-name><meta-value>Regular</meta-value></custom-meta><custom-meta><meta-name>article-type</meta-name><meta-value>OriginalPaper</meta-value></custom-meta><custom-meta><meta-name>journal-subject-primary</meta-name><meta-value>Biomedicine</meta-value></custom-meta><custom-meta><meta-name>journal-subject-secondary</meta-name><meta-value>Human Genetics</meta-value></custom-meta><custom-meta><meta-name>journal-subject-secondary</meta-name><meta-value>Molecular Medicine</meta-value></custom-meta><custom-meta><meta-name>journal-subject-secondary</meta-name><meta-value>Gene Function</meta-value></custom-meta><custom-meta><meta-name>journal-subject-secondary</meta-name><meta-value>Metabolic Diseases</meta-value></custom-meta><custom-meta><meta-name>journal-subject-collection</meta-name><meta-value>Biomedical and Life Sciences</meta-value></custom-meta><custom-meta><meta-name>open-access</meta-name><meta-value>true</meta-value></custom-meta></custom-meta-group></article-meta><notes notes-type="AuthorContribution"><p>Madina Seidualy, Asta Blazyte and Sungwon Jeon have contributed equally to this work.</p></notes><notes notes-type="Misc"><p>Sequence data from this article have been deposited to NCBI SRA database under accession No. SRS2904218 and NCBI BioSample database under accession No. SAMN08442411.</p></notes><notes notes-type="ESMHint"><title>Electronic supplementary material</title><p>The online version of this article (10.1007/s00439-020-02132-8) contains supplementary material, which is available to authorized users.</p></notes></front><body><sec id="Sec1" sec-type="introduction"><title>Introduction</title><p id="Par2">Recently, a wide variety of genome sequencing technologies have become available heralding a new era of personal genomics (Lander et al. <xref ref-type="bibr" rid="CR42">2001</xref>) and many large population genome projects have been carried out. These include the 1000 Genomes Project (Abecasis et al. <xref ref-type="bibr" rid="CR1">2010</xref>), the UK’s 10,000 and 100,000 Genomes Projects (Walter et al. <xref ref-type="bibr" rid="CR84">2015</xref>; Samuel and Farsides <xref ref-type="bibr" rid="CR71">2017</xref>), the Genome of the Netherlands (Boomsma et al. <xref ref-type="bibr" rid="CR10">2014</xref>), the Estonian Biocentre’s Human Genome Diversity Panel (EGDP) (Pagani et al. <xref ref-type="bibr" rid="CR64">2016</xref>), the Simons Genome Diversity Project (SGDP) (Mallick et al. <xref ref-type="bibr" rid="CR52">2016</xref>), the Genome Russia project (Oleksyk et al. <xref ref-type="bibr" rid="CR63">2015</xref>), 1070 Japanese genomes (Nagasaki et al. <xref ref-type="bibr" rid="CR58">2015</xref>), and the Korean Reference (KOREF) and variome projects (Cho et al. <xref ref-type="bibr" rid="CR11">2016</xref>; Kim et al. <xref ref-type="bibr" rid="CR37">2018</xref>). The Personal Genome Project (PGP) (Ball et al. <xref ref-type="bibr" rid="CR7">2012</xref>) is perhaps the largest genome project in terms of openness and inclusiveness and aims to map all personal and ethnic genomes. However, there remain many practical issues for mapping and accurately analyzing all ethnic groups worldwide. One problem is suitable representation of highly admixed genomes (Medina-Gomez et al. <xref ref-type="bibr" rid="CR55">2015</xref>; Guryev <xref ref-type="bibr" rid="CR25">2017</xref>). Although the 1000 Genomes Project database has been expanding by adding more ethnic representatives, it currently contains only 2504 individuals from 26 populations (phase 3) and lacks much ethnic diversity including an absence of genomes from Central Asian populations (Sudmant et al. <xref ref-type="bibr" rid="CR79">2015</xref>). Other initiatives such as the SGDP and the EGDP include only a small number of Central Asian population representatives (Pagani et al. <xref ref-type="bibr" rid="CR64">2016</xref>; Mallick et al. <xref ref-type="bibr" rid="CR52">2016</xref>). Central Asian populations can be good targets for adding highly admixed samples to our knowledge base of the major and relatively homogeneous ethnic groups. Among many Central Asian countries, Kazakhstan is at the border of ethnically European and Asian nations (Mostafa <xref ref-type="bibr" rid="CR57">2013</xref>). Therefore, demographic inference from Kazakh whole genomes is of special value. We can use Kazakh genomic data as an independent line of evidence that complements inference from archeological and written histories, to understand the roots of the diverse phenotypic features and relationships with other populations. Despite the recently growing scientific interest, there is little Kazakh genomic data available (Pagani et al. <xref ref-type="bibr" rid="CR64">2016</xref>), including, few high coverage sequences with genome-wide genotype array data of 18 individuals (Pagani et al. <xref ref-type="bibr" rid="CR64">2016</xref>; Jeong et al. <xref ref-type="bibr" rid="CR30">2019</xref>). This data paucity causes insufficient description on the complexity of the Kazakh genetic structure and the demographic past of Central Asian populations.</p><p id="Par3">Kazakhstan extends from the Caspian Sea on the west side to the Altai Mountains on the east and shares borders with Russia, China, Uzbekistan, Kyrgyzstan, and Turkmenistan. It is located at the crossroads of the Silk Road trade routes (Comas et al. <xref ref-type="bibr" rid="CR14">1998</xref>). Owing to multiple invasions throughout history, Kazakh territory has been the home to many distinct tribes and clans since the Paleolithic period (Ikawa-Smith <xref ref-type="bibr" rid="CR29">1978</xref>). Indo-European nomadic populations, namely Scythians-Saka started to settle in the Central Asian steppes at the beginning of the first millennium before the Common Era (BCE) (David Llewelyn Snellgrove <xref ref-type="bibr" rid="CR15">2018</xref>). Later on, and until the fourth century CE (370–452), a powerful state formed by the Huns prospered in the region of modern Kazakhstan, causing its inhabitants to move westward (Sinor <xref ref-type="bibr" rid="CR75">1990</xref>). Following the invasion of the Turkic-speaking tribes, Gokturk khanate was formed in the beginning of the sixth century CE (West <xref ref-type="bibr" rid="CR87">2009</xref>). The conquest of southern Kazakhstan by the Arabs followed by the invasion of Mongol tribes into the region led to increased societal complexity (Gibb <xref ref-type="bibr" rid="CR23">1923</xref>; Morgan <xref ref-type="bibr" rid="CR56">2007</xref>).</p><p id="Par4">The current primary ethnic group of Kazakhstan, the Kazakhs, formed from the union of diverse tribal groups, namely Turkic (West <xref ref-type="bibr" rid="CR87">2009</xref>), Mongol (Morgan <xref ref-type="bibr" rid="CR56">2007</xref>), Huns (Sinor <xref ref-type="bibr" rid="CR75">1990</xref>), Nogays (Weissleder <xref ref-type="bibr" rid="CR24">1978</xref>), Iranians (David Llewelyn Snellgrove <xref ref-type="bibr" rid="CR15">2018</xref>), and Arabs (Gibb <xref ref-type="bibr" rid="CR23">1923</xref>). Furthermore, as a consequence of complicated historical events, Kazakhs have been territorially divided into three main hordes (Zhuz) since the fifteenth–sixteenth century CE. The Senior Zhuz tribes settled in eastern and southeastern Kazakhstan (Semirechye), the Middle Zhuz populated central Kazakhstan, and the Junior Zhuz lived primarily in western Kazakhstan. The Senior Zhuz are divided into eleven tribal groups, the Middle Zhuz into seven, and the Junior Zhuz into three (Olcott <xref ref-type="bibr" rid="CR62">1995</xref>). Traditionally marriage within the same tribe is undesirable (Forde <xref ref-type="bibr" rid="CR19">1934</xref>). Consequently, this has led to extensive mixing of Kazakh genomes. In this study, we aimed to increase the Asian PGP repertoire by providing the Kazakh (MJS’s) whole genome sequence; therefore, this study was carried out in the framework of the Pan Asian Population Genomics Initiative (PAPGI, <ext-link ext-link-type="uri" xlink:href="http://papgi.org">http://papgi.org</ext-link>). We use this genome to confirm a high level of heterozygosity and attribute certain genomic components to both ancient and recent admixtures.</p></sec><sec id="Sec2" sec-type="materials|methods"><title>Materials and methods</title><sec id="Sec3"><title>Sample preparation</title><p id="Par5">Our sample donor, MJS, is a healthy Kazakh female, who resides in southern Kazakhstan. MJS belongs genetically to two tribes: her father’s clan is a Middle Zhuz’s tribe, Naiman, people who migrated from Mongolia to settle in the eastern and central part of Kazakhstan in the late twelfth century CE (Akerov <xref ref-type="bibr" rid="CR2">2016</xref>); her mother’s clan is the Senior Zhuz’s tribe, Bayis, descendants of one of the major tribes of Dulat, people who settled in the southeast part (Semirechye) of Kazakhstan during the sixth and seventh centuries CE (Olcott <xref ref-type="bibr" rid="CR62">1995</xref>).</p><p id="Par6">MJS reported her Kazakh ethnicity by providing a genealogical history of four generations. DNA was extracted from MJS’s peripheral blood using DNeasy Blood &amp; Tissue Kit from QIAGEN according to the manufacturer’s protocols. Whole genome sequencing was conducted using the short-read sequencer, Illumina HiSeq X Ten, with 151 bp paired-end reads.</p></sec><sec id="Sec4"><title>Identification of individual variants</title><p id="Par7">We used the GRCh37/hg19 (UCSC's nomenclature) as a reference. Before alignment of the reads to the reference, quality filtering was performed using NGSQC toolkit (v 2.3.3) with default options (Patel and Jain <xref ref-type="bibr" rid="CR66">2012</xref>). We used the Burrows–Wheeler Alignment-MEM (BWA v0.7.8) (Li and Durbin <xref ref-type="bibr" rid="CR46">2009</xref>) with a minimum seed length of 19 bp for mapping against the reference. Quality check of the mapping results was performed with SAMStat (v1.5.1) (Lassmann et al. <xref ref-type="bibr" rid="CR44">2011</xref>). The alignment file was sorted using the SAMtools (v. 0.1.19) (Li et al. <xref ref-type="bibr" rid="CR48">2009</xref>). Reads duplicated in PCR were removed using the MarkDuplicate option in Picard (v1.114) (<ext-link ext-link-type="uri" xlink:href="http://broadinstitute.github.io/picard/">http://broadinstitute.github.io/picard/</ext-link>). Local realignment of reads around indels and recalibration of base quality scores were performed using IndelRealigner and BaseRecalibrator in the Genome Analysis Toolkit (GATK v2.3.9) (McKenna et al. <xref ref-type="bibr" rid="CR54">2010</xref>). We used a GATK Unified Genotyper with the settings ‘-heterozygosity 0.0010-dcov 200-stand_call_conf 30.0-stand_emit_conf 30.0’ to call variants.</p></sec><sec id="Sec5"><title>Annotation of the variants</title><p id="Par8">Single nucleotide variants (SNVs) and small insertions and deletions (indels) ranging from one to 20 bases were identified using GATK (McKenna et al. <xref ref-type="bibr" rid="CR54">2010</xref>). To annotate the type and functional consequences of SNVs and indels we used the snpEff (v4.3) (Cingolani et al. <xref ref-type="bibr" rid="CR13">2012</xref>) and ANNOVAR (v3) software (Wang et al. <xref ref-type="bibr" rid="CR85">2010</xref>). SNVs in the MJS genome were examined for their possible functional effects using computational prediction methods using SIFT (Ng and Henikoff <xref ref-type="bibr" rid="CR60">2003</xref>), Polyphen2 (Jordan et al. <xref ref-type="bibr" rid="CR32">2011</xref>), and PROVEAN (Choi and Chan <xref ref-type="bibr" rid="CR12">2015</xref>). We classified the variants into known and novel SNVs according to their presence in the dbSNP reference collection (<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/snp">https://www.ncbi.nlm.nih.gov/snp</ext-link>) (v147). Known variants were further annotated with possible associations to known diseases or drug responses using the databases OMIM (<ext-link ext-link-type="uri" xlink:href="http://www.omim.org">www.omim.org</ext-link>) and ClinVar (v20170130) (Landrum et al. <xref ref-type="bibr" rid="CR43">2014</xref>). The non-synonymous SNVs in MJS that were predicted to be functionally damaging were also checked in the other 21 publicly available Kazakh genomes (<ext-link ext-link-type="uri" xlink:href="https://www.geenivaramu.ee/en">https://www.geenivaramu.ee/en</ext-link>), (Jeong et al. <xref ref-type="bibr" rid="CR30">2019</xref>). Due to limited SNP genotyping array coverage of the 18 samples, all of the selected SNVs (except for the rs1805124) were reported using only four Kazakh samples (Table S1).</p></sec><sec id="Sec6"><title>Admixture analysis</title><p id="Par9">We examined heterogeneous admixture patterns using the ADMIXTURE (v1.3.0) (Alexander et al. <xref ref-type="bibr" rid="CR3">2009</xref>) program. We collected all the publically available genomic Kazakh data to date which reflected admixed Kazakh individuals in different tribes. Three Kazakh samples were obtained from the Estonian Genome Centers’ biobank (<ext-link ext-link-type="uri" xlink:href="https://www.geenivaramu.ee/en">https://www.geenivaramu.ee/en</ext-link>); one from Central-West, one from Tien-Shan (southeastern Kazakhstan) and one from unidentified location in Kazakhstan, and 18 SNP chip based samples were obtained from the recently published data (Jeong et al. <xref ref-type="bibr" rid="CR30">2019</xref>) deposited in Max Plank digital library. Firstly, we merged the human origin SNP panel (HOSP) data containing 2345 samples from 203 populations worldwide (Lazaridis et al. <xref ref-type="bibr" rid="CR45">2014</xref>) with the sample dataset from the Estonian Genome Centers’ biobank (<ext-link ext-link-type="uri" xlink:href="https://www.geenivaramu.ee/en">https://www.geenivaramu.ee/en</ext-link>), MJS’s genome and the 18 Kazakh dataset using PLINK (v1.90) (Purcell et al. <xref ref-type="bibr" rid="CR69">2007</xref>), utilizing autosomal SNPs. We pruned the panel with linkage disequilibrium (LD) using PLINK (v1.90) using the ‘-indep-pairwise 200 25 0.4’ option. We explored the values of the assumed ancestral populations (<italic>K</italic>) from two to 14. We observed the cross validation error values (Fig. S1) and chose <italic>K</italic> = 2, 4, 6, and 8 to display the increasing Kazakh ancestral complexity along with increasing <italic>K</italic> values. We also used qpGraph of ADMIXtool (Patterson et al. <xref ref-type="bibr" rid="CR67">2012</xref>) and 110 modern human genomes (French, Mongolian, Koryak, Yoruban) from the Human Origin SNP Panel (HOSP) (Lazaridis et al. <xref ref-type="bibr" rid="CR45">2014</xref>) to validate ADMIXTURE graphs.</p></sec><sec id="Sec7"><title>Mitochondrial haplogroup analysis</title><p id="Par10">Variants in the mtDNA sequence were detected by mapping to the rCRS (revised Cambridge Reference Sequence of the human mtDNA) (Andrews et al. <xref ref-type="bibr" rid="CR5">1999</xref>). We used HaploGrep (v 2.1.13) to determine the haplogroup of MJS’s maternal lineage (Weissensteiner et al. <xref ref-type="bibr" rid="CR86">2016</xref>).</p></sec><sec id="Sec8"><title>Principal component analysis (PCA)</title><p id="Par11">MJS genome was projected onto the first two principal components calculated using samples from PAPGI (<ext-link ext-link-type="uri" xlink:href="http://papgi.org">http://papgi.org</ext-link>), HOSP (Lazaridis et al. <xref ref-type="bibr" rid="CR45">2014</xref>), and EGDP (<ext-link ext-link-type="uri" xlink:href="http://evolbio.ut.ee/">http://evolbio.ut.ee/</ext-link>). To optimize the dataset and reduce bias caused by closely linked variants we pruned the merged dataset with LD using PLINK (v1.90) (Purcell et al. <xref ref-type="bibr" rid="CR69">2007</xref>) with the ‘–indep-pairwise 200 25 0.4’, ‘-geno 0.1’, ‘-maf 0.05’ ‘-mind 0.2’ options. Eurasian populations were selected resulting in 938 present-day genome samples for the final visualization. Principal component analysis (PCA) was performed using EIGENSOFT (v6.1.4) (Kang et al. <xref ref-type="bibr" rid="CR35">2010</xref>) with default settings. The output was plotted in the R program (Team <xref ref-type="bibr" rid="CR81">2018</xref>) (v. 3.5.1) using the ggplot2 (Wickham <xref ref-type="bibr" rid="CR89">2009</xref>) (v3.1.0), data.table (v1.11.8) (<ext-link ext-link-type="uri" xlink:href="https://CRAN.R-project.org/package%3ddata.table">https://CRAN.R-project.org/package=data.table</ext-link>), grid (v3.5.0) (<ext-link ext-link-type="uri" xlink:href="https://cran.r-project.org/src/contrib/Archive/grid/">https://cran.r-project.org/src/contrib/Archive/grid/</ext-link>) and gridExtra (v2.3) (<ext-link ext-link-type="uri" xlink:href="https://cran.r-project.org/web/packages/gridExtra/index.html">https://cran.r-project.org/web/packages/gridExtra/index.html</ext-link>) packages.</p><p id="Par12">We performed phylogenetic analysis to check the concordance with the aforementioned PCA analysis. We selected 36 Eurasian ethnic groups with available SNP data from HOSP (Lazaridis et al. <xref ref-type="bibr" rid="CR45">2014</xref>). For phylogenetic tree construction, we calculated pairwise nucleotide distances (pi) and constructed a neighbor-joining tree using Mega 7 (Kumar et al. <xref ref-type="bibr" rid="CR41">2016</xref>). The phylogenetic tree and the map building were conducted using the ggplot2 (v3.1.0) (Wickham <xref ref-type="bibr" rid="CR89">2009</xref>), ggtree (v3.8) (Yu et al. <xref ref-type="bibr" rid="CR91">2016</xref>), ggpubr (v0.2) (<ext-link ext-link-type="uri" xlink:href="https://rpkgs.datanovia.com/ggpubr/index.html">https://rpkgs.datanovia.com/ggpubr/index.html</ext-link>), and data.table (v1.11.8) (<ext-link ext-link-type="uri" xlink:href="https://CRAN.R-project.org/package%3ddata.table">https://CRAN.R-project.org/package=data.table</ext-link>) packages of R (v3.5.1) (Team <xref ref-type="bibr" rid="CR81">2018</xref>). Longitude and latitude information with custom color and shape settings were used to visualize the physical distances between MJS and other population samples.</p></sec><sec id="Sec9"><title>Admixture <italic>f3</italic> statistics based on ancient and present-day genomes</title><p id="Par13">We inferred MJS’s genetic lineage using admixture <italic>f3</italic>-statistics, a method based on measuring the allele frequency correlations between populations (Patterson et al. <xref ref-type="bibr" rid="CR67">2012</xref>). To maximize the comprehensiveness of this analysis and evaluate genetic associations, we selected 2670 present-day and 108 ancient genomes (using published ancient genomes (Lazaridis et al. <xref ref-type="bibr" rid="CR45">2014</xref>; Lipson et al. <xref ref-type="bibr" rid="CR50">2018</xref>; Keller et al. <xref ref-type="bibr" rid="CR36">2012</xref>; Jones et al. <xref ref-type="bibr" rid="CR31">2015</xref>; Siska et al. <xref ref-type="bibr" rid="CR76">2017</xref>; Haak et al. <xref ref-type="bibr" rid="CR26">2015</xref>; Allentoft et al. <xref ref-type="bibr" rid="CR4">2015</xref>; Skoglund et al. <xref ref-type="bibr" rid="CR77">2014</xref>; Fu et al. <xref ref-type="bibr" rid="CR20">2014</xref>; Raghavan et al. <xref ref-type="bibr" rid="CR70">2014</xref>; Seguin-Orlando et al. <xref ref-type="bibr" rid="CR74">2014</xref>; Gamba et al. <xref ref-type="bibr" rid="CR22">2014</xref>)). To measure genetic associations, we used a notation <italic>f3</italic>(A,B;Kazakh), where A and B were ancient and present-day populations in various combinations. We employed the qp3PopTest program (v300) from the ADMIXTOOLS (v3) package to calculate the <italic>f3</italic> statistics. We extracted the combinations which had high significance (|Z| &gt; 3) and a sufficient number (&gt; 1000) of examined SNPs (Table S2). The top 30 genome pairs with the most negative <italic>f3</italic> values are plotted (“<xref rid="Sec11" ref-type="sec">Results and discussion</xref>” section).</p></sec><sec id="Sec10"><title>Sequentially Markovian coalescent analyses</title><p id="Par14">Pairwise sequentially Markovian coalescent (PSMC) analysis was conducted to predict and visualize the level of genome diversity (Li and Durbin <xref ref-type="bibr" rid="CR47">2011</xref>). To estimate the historical effective population size (<italic>N</italic><sub>e</sub>) of the Kazakh genome, we applied the PSMC model, using one diploid genome per population and the MSMC2 program (de Manuel et al. <xref ref-type="bibr" rid="CR16">2016</xref>). We calculated <italic>N</italic><sub>e</sub> for Kazakh (MJS) representing Central Asia and eight additional genomes representing the Northeast and South Asia (Han, Korean, Mongolian, Koryak and Pathan), Africa (Bot San), Europe (French) and the Near East (Turkish), obtained from the PAPGI dataset (<ext-link ext-link-type="uri" xlink:href="http://papgi.org">http://papgi.org</ext-link>).</p><p id="Par15">Multiple sequentially Markovian coalescent (MSMC) analysis was carried out using the MSMC2 program (de Manuel et al. <xref ref-type="bibr" rid="CR16">2016</xref>) to estimate coalescence rates between the haplotypes of the population through time (Schiffels and Durbin <xref ref-type="bibr" rid="CR73">2014</xref>). For comparison, we used a dataset from the PSMC analysis containing one diploid genome per population from the PAPGI data. We estimated the depth of coverage of each chromosome (except for the sex chromosomes and mtDNA) from BAM files and, then, we ran SAMtools (Li et al. <xref ref-type="bibr" rid="CR48">2009</xref>) to generate mask bed files and VCF files. The script generate_multihetsep.py was used to generate input for the MSMC tool, which merges VCF files. Input files denoting chromosome numbers and positions of segregating sites were generated from all the somatic chromosomes. The results were plotted using R (Team <xref ref-type="bibr" rid="CR81">2018</xref>). We adjusted the outcome by setting a mutation rate of 0.5 × 10<sup>−9</sup> bp<sup>−1</sup> year<sup>−1</sup> as suggested by the work on human evolution by Scally (Scally <xref ref-type="bibr" rid="CR72">2016</xref>).</p></sec></sec><sec id="Sec11"><title>Results and discussion</title><sec id="Sec12"><title>Genome sequencing statistics</title><p id="Par16">In total, 82.49 Gbp of nucleotide sequence was generated and mapped to the Human Reference genome (build 37) using BWA (Li and Durbin <xref ref-type="bibr" rid="CR46">2009</xref>). We successfully aligned 99.96% of the reads to the reference genome with an average sequencing depth of 29-fold (Table S3). By comparing the MJS genome with the reference, we observed 5,063,461 short variations that consisted of 4,301,702 single nucleotide variants (SNVs) and 761,759 insertions and deletions (indels). MJS mitochondrial genome was mapped by 100% amplicon coverage (16,569 bp), and contained 32 variants including one novel deletion in the 16S ribosomal DNA and variants associated with Leber’s hereditary optic neuropathy (Table S4).</p><p id="Par17">A total of 4,199,462 SNVs (97.6%, excluding indels) from the whole genome sequence were known variants already deposited in dbSNP (ver. 147), and 102,240 were novel (Fig. <xref rid="Fig1" ref-type="fig">1</xref>). In addition, 231 of the novel SNVs (0.2%) were non-synonymous (nsSNVs).<fig id="Fig1"><label>Fig. 1</label><caption xml:lang="en"><p>Classification of short variants found in the MJS genome</p></caption><graphic specific-use="HTML" mime-subtype="PNG" xlink:href="439_2020_2132_Fig1_HTML.png" id="MO1"/></fig></p><p id="Par18">The number of observed heterozygous SNVs in MJS (2,773,421) was compared with that of 212 whole genomes represented in the PAPGI data to visualize levels of heterozygosity per continental group. We observed a significantly higher level of heterozygosity in MJS than in American, Oceanian, East and Southeast Asian genomes, and moderately higher than an average heterozygosity value of the North, South and Central Asian individuals. Still, heterozygosity was lower than the African genomes used in this analysis (Fig. S2).</p></sec><sec id="Sec13"><title>Functional classification of the variants</title><p id="Par19">There were 13,121 (0.3%) non-synonymous SNVs (nsSNVs) in MJS genome, of which 5001 (38%) nsSNVs were predicted to be deleterious by SIFT (Ng and Henikoff <xref ref-type="bibr" rid="CR60">2003</xref>), Polyphen2 (Jordan et al. <xref ref-type="bibr" rid="CR32">2011</xref>), or PROVEAN analysis (Choi and Chan <xref ref-type="bibr" rid="CR12">2015</xref>). Out of the 5001 nsSNVs, 654 (12.5%) were identified as deleterious by all of the three prediction methods (Table S5 for known SNVs, Table S6 for novel SNVs). Most of these potentially deleterious nsSNVs (572) were heterozygous, and did not result in any known pathological phenotypes in MJS, suggesting they are functionally benign. The ClinVar (Landrum et al. <xref ref-type="bibr" rid="CR43">2014</xref>) (ver. 20170130) analysis of MJS identified 50 SNVs as pathogenic (Table S7). Among them, we found an SNV in the <italic>SCN5A</italic> gene (rs1805124) reported to have a strong link to cardiovascular failure (Mazzaccara et al. <xref ref-type="bibr" rid="CR53">2018</xref>). The leading cause of death in Kazakh population is ischaemic heart disease, which caused 32.5% (51.4 thousand) of all deaths in 2012 (WHO <xref ref-type="bibr" rid="CR88">2015</xref>) and nearly half of the collected Kazakh genomes (10/22) had at least one (rs1805124) risk variant (Table S1).</p><p id="Par20">Among the drug response variants (Table S8) three important mutant alleles (NAT2*5B, NAT2*6A, NAT2*12A) of slow acetylation activity in the liver were identified in <italic>NAT2</italic> gene (Vatsis et al. <xref ref-type="bibr" rid="CR82">1991</xref>) in MJS and one more sample (Table S1). Even though the pathogenicity of these variants isn’t clear (Table S8), these variants were expected to be found in our data set, since they reflect the long history of agriculture in Central Asian populations (Magalon et al. <xref ref-type="bibr" rid="CR51">2008</xref>). Also, in 2008, Magalon et al. (Magalon et al. <xref ref-type="bibr" rid="CR51">2008</xref>) reported that Kazakh population contains 26–35% of slow acetylators and the haplotype NAT2*5B exhibited the highest allele frequency among Kazakhs, while NAT2*6A was found to be approximately three times more commonly in the neighboring Tajik population.</p><p id="Par21">The MJS metabolism of anticoagulant and anticonvulsant medicines, such as mephenytoin, warfarin, tolbutamide, and phenytoin may be compromised due to the pathogenic mutations in <italic>CYP2C19</italic> (rs4244285) (Arici and Özhan <xref ref-type="bibr" rid="CR6">2017</xref>) and antiepileptic carbamazepine metabolism affected by rs1051740 polymorphism (which was predicted to be deleterious by all the three previously described functional impact prediction tools) in the gene <italic>EPHX1</italic> (Zhao et al. <xref ref-type="bibr" rid="CR92">2019</xref>). Moreover, substitution of the valine to alanine found in MJS in position 174 of the <italic>SLCO1B1</italic> gene (rs4149056), which was also confirmed by multiple tools to be deleterious, is known to reduce uptake and transport activity of cholesterol-lowering drugs such as simvastatin, pravastatin, pitavastatin, and fexofenadine (Voora et al. <xref ref-type="bibr" rid="CR83">2009</xref>) which increases the risk of statin-induced myopathy (Link et al. <xref ref-type="bibr" rid="CR49">2008</xref>). This risk variant was found in one more Kazakh sample (2/4, Table S1). Finally, two homozygous variations in MJS in <italic>TAS2R38</italic> (rs10246939 and rs713598), while heterozygous in other Kazakhs, confirmed phenylthiocarbamide taster phenotype (4/4); the ability to taste bitterness in foods, like cabbage, raw broccoli as well as in the drinks like coffee and beer (Perna et al. <xref ref-type="bibr" rid="CR68">2017</xref>). Overall, MJS and the additional genomes presented not only well-established variants in the region but also potential for various pharmacogenomic research directions that could be relevant for Kazakh population.</p></sec><sec id="Sec14"><title>Hints of Caucasian admixture</title><p id="Par22">MJS has both T and C alleles in the <italic>EDAR</italic> gene (Table S1). In Fujimoto’s research in 2008, out of 360 alleles from Japanese and Chinese samples, 87.6% of them were C, whereas the frequency of the C allele occurrence among European samples was 0% (Fujimoto et al. <xref ref-type="bibr" rid="CR21">2008</xref>). The C allele (rs3827760), a hereditary determinant of increased hair thickness, occurred in East Asia, likely in Central China around 30,000 years ago (Kamberov Yana et al. <xref ref-type="bibr" rid="CR34">2013</xref>). Just four out of 22 Kazakh samples demonstrated all possible genotypes suggesting that hair thickness in Kazakh population range from the typical (increased) in East Asia to typical in Europeans (Table S1).</p><p id="Par23">MJS also has heterozygous ancient 111T and 374F alleles (rs1426654 and rs16891982) in <italic>SLC45A2</italic> and <italic>SLC24A5</italic> genes that account for the skin tone, as well as eye and hair color, which are nearly fixed in Europeans and, therefore, are ancestry-informative (Soejima and Koda <xref ref-type="bibr" rid="CR78">2007</xref>). Moreover, the SNV in the <italic>ABCC11</italic> gene resulted in wet earwax in MJS; the same phenotype can be inferred from all the other Kazakh sequences as well (4/4). Commonly, in East Asian populations the homozygous 180Arg allele is associated with dry earwax, whereas all Europeans have the 180Gly allele that results in wet earwax (Yoshiura et al. <xref ref-type="bibr" rid="CR90">2006</xref>). The heterozygosity of the above-mentioned SNVs suggests that the Kazakh MJS genome has genetic overlaps with the European and Asian phenotypes. The assumption of Caucasian admixture was also supported by mtDNA haplogroup J1c2 found in the MJS which is primarily found in Near Eastern and European populations (Hartmann et al. <xref ref-type="bibr" rid="CR27">2009</xref>). Even though Kazakh population contains both East Eurasian (55%) and West Eurasian (41%) mtDNA lineages (Berezina et al. <xref ref-type="bibr" rid="CR8">2011</xref>), J haplogroup is observed in only 3.6% of the Kazakhs (Berezina et al. <xref ref-type="bibr" rid="CR8">2011</xref>). The haplogroup J1c2 is strongly associated with the earliest European farmers and, in fact, has been recently (2017) traced back to the Iron Age Black Sea Scythians (Juras et al. <xref ref-type="bibr" rid="CR33">2017</xref>), which points to a European or Near Eastern maternal lineage as a component of the presumably ancient admixture in MJS. However, such genetic affinities may also be indirect, e.g., involving ancestors of ancient nomadic Turkic people whose direct influx into Kazakh lands occurred much later, in the Common Era (West <xref ref-type="bibr" rid="CR87">2009</xref>).</p></sec><sec id="Sec15"><title>Genetic diversity and structure</title><p id="Par24">We employed the ADMIXTURE (Alexander et al. <xref ref-type="bibr" rid="CR3">2009</xref>) program to estimate possible ancestries of the MJS genome based on autosomal SNPs (Figs. <xref rid="Fig2" ref-type="fig">2</xref>, S3). At <italic>K</italic> = 8, major genetic components were East Asian (yellow) (32.8%), followed by European (dark green) (30.8%) which was also shared by West Asian ancestries as well as Tajik from the Central Asia (Fig. <xref rid="Fig2" ref-type="fig">2</xref>, Table S9). The third major MJS component (orange) was attributed mainly to North Asia (28.9%). Around 6% of the MJS’s genome was associated with South Asians (light green portion). The MJS admixture model was tested by qpGraph and it confirms that Europeans and the mixture of North and East Asians were the best fitting admixture sources (Fig. S4). Furthermore, MJS’s ancestral composition was very similar to that of the other Kazakhs used in the dataset, despite of fairly large geographical distances among the sample origins within the country (Materials and Methods 2.4) and different tribal affiliations. Comparing to other Kazakh samples (Kypchak, Naiman, Argyn and mixed Kazakh), MJS genome showed the highest proportion of the North Asian component, which surprisingly varied little among all Kazakh (by 5.6%). However, the European and East Asian portions varied among the Kazakh samples the most—by approximately 8%, followed by the West Asian component (that varied by 4.5%), where in all cases MJS showed quite average proportions (Table S9). We speculate that a high level of heterozygosity (Fig. S2) with very similar ancestral component composition is a common trait of the Kazakhs. It points to a scenario, where Kazakhs experienced admixtures from various different ethnic groups, since ancient times but in recent times kept admixing mainly between the local (tribes) subpopulations leading to a modern Kazakh genetic identity. However, a large-scale study covering statistically significant number of individuals from different regions and covering all tribal lineages is needed to confirm this hypothesis. <fig id="Fig2"><label>Fig. 2</label><caption xml:lang="en"><p>ADMIXTURE plot showing the MJS genome originating from multiple different artificial ancestral populations. Each modern population is represented by genomes which are depicted as colored bars. Each color indicates a different ancestral group and proportion of possibly shared ancestry within larger subgroups.</p></caption><graphic specific-use="HTML" mime-subtype="PNG" xlink:href="439_2020_2132_Fig2_HTML.png" id="MO2"/></fig></p></sec><sec id="Sec16"><title>Principal component analysis (PCA)</title><p id="Par25">All Kazakh samples, including MJS, cluster together in a PCA plot and the similarities of samples reflect geographic proximities (Fig. <xref rid="Fig3" ref-type="fig">3</xref>). The closest similarity to the Kazakh MJS sample is exhibited by Central Asians (Kyrgyz, Uzbek, Kalmyk) and East Asians (Mongolians), which was confirmed by the phylogenetic tree based on the pairwise nucleotide distances (Fig. S5, Table S10). Moreover, MSMC analysis suggested Kazakh and Mongolian population divergence at around 7000 years ago (Kya) which means relatively recent common ancestry compared to divergence of the Kazakh and Koryak or Kazakh and Han Chinese (around 10 Kya) (Fig. S6). The Near Easterner (Turkish) population separation was estimated to have occurred prior to 13,000 years ago (Kya) (Fig. S6). Although MJS genome has high heterozygosity, all of the methods employed hint higher Kazakh (including MJS) genetic affinity to East Asians (Mongolian) than Caucasians as previously reported (Tarlykov et al. <xref ref-type="bibr" rid="CR80">2013</xref>).<fig id="Fig3"><label>Fig. 3</label><caption xml:lang="en"><p>Plot of the first two principal components with the modern Eurasian populations. Dots represent genomes color-coded by the continental groups: Europe (EUR)—dark green, West Asia (WA)—light green, Central Asia (CA)—blue, North Asia (NA)—peach, South Asia (SA)—grey, Southeast Asia (SEA) and East Asia (EA)—brown, and America (AM)—yellow, Oceania (OC)—coral, serving as outgroups. Separate populations are visualized as different shades and the MJS sample is visualized as a purple dot</p></caption><graphic specific-use="HTML" mime-subtype="PNG" xlink:href="439_2020_2132_Fig3_HTML.png" id="MO3"/></fig></p><p id="Par26">The Kyrgyz appeared to be the closest population to the Kazakhs. It is widely speculated that the Naiman tribe, the paternal line of MJS, has close ethnogenetic ties with the Yenisei Kyrgyz and once coexisted together under the Kyrgyz Khaganate (Akerov <xref ref-type="bibr" rid="CR2">2016</xref>). The split between the two might have occurred around the fifteenth–sixteenth century CE, along with Kyrgyz migration (Heyer et al. <xref ref-type="bibr" rid="CR28">2009</xref>). The Kalmyk similarity to MJS is not surprising as in the seventeenth century CE Kalmyks conquered Western Mongolia as well as Eastern and South Eastern Kazakhstan. Their territories continued to expand at the cost of Kazakh lands until the eighteenth century CE (Olcott <xref ref-type="bibr" rid="CR62">1995</xref>) which might have resulted in the direct admixture of the Kazakh and the Kalmyk despite already shared genetic influences from the Mongolian empire (Nasidze et al. <xref ref-type="bibr" rid="CR59">2005</xref>). As for Uzbek, their genomic similarity can historically be traced back to the Uzbek Khanate in the fifteenth century CE as they have coexisted with many other tribes including the Dulat (ancestors of MJS’s maternal lineage) in various tribal groupings until the two populations split (Paksoy <xref ref-type="bibr" rid="CR65">1992</xref>). Altaians (Tubalars) appeared to be genetically closest ethnic group from the North Asia (Table S10). Besides their close geographic proximity and shared roots (Dulik Matthew et al. <xref ref-type="bibr" rid="CR17">2012</xref>), previous research suggests a significant demographic expansion of Altaian people from the Mongolian Altai territories towards Western steppes after the seventh century CE, based on the Iron Age Western Altaian (Russian and Kazakh territory) and Mongolian Altaian mtDNA similarities (Dulik et al. <xref ref-type="bibr" rid="CR18">2011</xref>). On the other hand, a wave of Kazakh migration around the Altai mountains occurred around the nineteenth–twentieth century CE (Krader <xref ref-type="bibr" rid="CR40">1966</xref>), likely from the Middle Zhuz (Oktyabrskaya <xref ref-type="bibr" rid="CR61">2006</xref>), from which MJS’s paternal line originates. Some of the destinations of this fairly recent migration were as far as Xinjiang in China, and Western Mongolia (Krader <xref ref-type="bibr" rid="CR40">1966</xref>). This migration seems to be surprisingly accurately depicted in the MJS genome as both Mongolian and Xibo tribe (residing in Xinjiang) are two of the most closely related East Asian populations to MJS (Table S10, Fig. S5). A relatively recent Y-chromosome variation analysis proposed shared Kazakh paternal lineages with the Mongolian and attributed their findings to the Mongolian Empire expansion in the thirteenth century CE (Dulik et al. <xref ref-type="bibr" rid="CR18">2011</xref>). The Mongolian genetic influence in the MJS genome also can be traced to her paternal ancestors, members of the Naiman tribe.</p></sec><sec id="Sec17"><title>Admixture <italic>f3</italic> statistics based on ancient and present-day genomes</title><p id="Par27">Knowing the complex Kazakh history which is rich of different admixture sources, we used admixture <italic>f3</italic> statistics to test if it is possible to define a present-day Kazakh individual, MJS, as an admixture product of only two distinct populations (Sudmant et al. <xref ref-type="bibr" rid="CR79">2015</xref>) and what pair of populations would have the highest similarity. Even though the pairwise allele sharing was measured with both present-day and ancient genomes, the highest genetic affinity in nearly all 30 cases was shown by a pair of ancient genomes, where one ancient genome comes from Europe and one from Northeast Asia (Fig. <xref rid="Fig4" ref-type="fig">4</xref>). From the Asian genomes the highest similarity was shown by the Devil’s gate sample Devil'sGate1, an Early Neolithic hunter-gatherer (Siska et al. <xref ref-type="bibr" rid="CR76">2017</xref>); it appeared in more than 10 best representing pairings. So far, the Devil’s gate genomes (1 and 2) are the closest to East Asia genomes available today (Siska et al. <xref ref-type="bibr" rid="CR76">2017</xref>). Moreover, the present-day populations from China (Tujia, Han, Naxi, She, Hezhen, Oroqen, Daur and Yi) and other East Asian territories (Japanese and Korean) consistently showed some of the highest genetic affinities with Kazakh (MJS) supporting the theory of the genetic continuity within the East Asian region (Siska et al. <xref ref-type="bibr" rid="CR76">2017</xref>) and strongly reflecting the East Asian component in MJS genome. The ancient European genomes with the highest proportion of allele sharing were excavated from the Central Europe; the Halberstadt_LBA1 genome dated Late Bronze Age, LBK_EN samples attributed to Early Neolithic and Esperstedt to Middle Neolithic periods (Haak et al. <xref ref-type="bibr" rid="CR26">2015</xref>). The overlap in time frame (Early Neolithic) of the Devil’s gate and Central European genomes suggests the presence of two possible ancient genomic components of different origins present in the MJS. Interestingly, Kazakh effective population size <italic>N</italic><sub>e</sub> prior to the Neolithic era has undergone a radical decrease; around 60,000 years ago (Kya), which would correspond to Middle Paleolithic (Bicho <xref ref-type="bibr" rid="CR9">2013</xref>), Kazakh <italic>N</italic><sub>e</sub> reached its lowest—less than 4000 individuals and around the Upper Paleolithic (Klein <xref ref-type="bibr" rid="CR38">1999</xref>) (40 Kya) recovered to approximately 6000 (Fig. S7). The end of the Last Glacial Maximum and subsequent population size increase suggest it became possible for ancestral MJS’s populations of different origins to migrate and admix at around that time.<fig id="Fig4"><label>Fig. 4</label><caption xml:lang="en"><p>The Admixture <italic>f3</italic> analysis representing MJS genome as a mixture of genomes A and B. The genomes are color-coded by regions; yellow representing ancient genomes found in Europe, red—ancient genomes from Northeast Asia and indigenous (present-day) genomes from Russian Far East, blue—present-day East Asian populations, and green—present-day West Asian populations. Thirty pairs with the lowest <italic>f3</italic> score for the Kazakh (MJS) are presented</p></caption><graphic specific-use="HTML" mime-subtype="PNG" xlink:href="439_2020_2132_Fig4_HTML.png" id="MO4"/></fig></p><p id="Par28">Even though other ancient European genomes in this analysis were attributed to different locations and times; ranging from the Holocene (Lazaridis et al. <xref ref-type="bibr" rid="CR45">2014</xref>; Haak et al. <xref ref-type="bibr" rid="CR26">2015</xref>) to Iron Age (Gamba et al. <xref ref-type="bibr" rid="CR22">2014</xref>), the strong MJS’s allele frequency association with ancient Europeans prevails. Moreover, the only present-day population paired with the East Asian genome (Devil’s gate) is Iranian, which strengthened the evidence of a European/Near Eastern component as inferred by the MJS mtDNA haplogroup. These two components, however, may not be the direct or the only sources of admixture in MJS and other Kazakh as such model does not estimate demographic shifts and complex admixtures from multiple sources.</p></sec></sec><sec id="Sec18" sec-type="conclusion"><title>Conclusions</title><p id="Par29">We present the whole genome sequence and thorough genetic variant and admixture analysis of a Central Asian, Kazakh MJS. We found several SNVs associated with drug toxicity, metabolism, diseases, phenotypic features and identified recent and ancient admixtures. Both PCA and phylogenetic analyses confirm closer MJS and other Kazakh similarity to modern East Asians than Europeans and showed the overall closest genetic affinities are with other Central Asian populations, namely, Kalmyk, Uzbek and Kyrgyz. All populations with significant similarity to MJS genome could be backed up by historic migration events involving the Kazakh population and the major fraction of genomic variation could be attributed to fairly recent admixture with geographically close populations. However, MJS’s mitochondrial DNA haplogroup is of European or Near Eastern (West Asian) ancestry. It corresponds to the heterozygous SNPs associated with European phenotypic features and confirmed by admixture <italic>f3</italic> statistics and all other Kazakh autosomal data showed very similar ancestral compositions to MJS’s. This highly heterozygous and admixed Kazakh genome provides insights into complex admixtures and can serve as a reference for mapping complex heterogeneity in Central Asian populations.</p></sec></body><back><ack><title>Acknowledgements</title><p>This work was supported by the U-K BRAND Research Fund (1.190007.01) of Ulsan National Institute of Science &amp; Technology (UNIST) and by the Research Project Funded by Ulsan City Research Fund (1.190033.01) of Ulsan National Institute of Science &amp; Technology (UNIST). We thank KOGIC members for providing technical assistance and discussions. In addition, the Korea Institute of Science and Technology Information (KISTI) that provided us with access to the Korea Research Environment Open NETwork (KREONET), the internet connection service that enabled efficient information and data transfer.</p></ack><notes notes-type="author-contribution"><title>Authors’ contributions</title><p>MS, JK, SJ, AB, AE, AM, and JB were involved in designing and conceptualizing this study. MS, SJ, and AB were in charge of analysis, data acquisition, and visualization. YB, YJ, and JK contributed in software and pipeline customization. MS and AB wrote the manuscript under supervision of JB and SL. JB, SL, AB, SJ, MS, YB, YJ, CY, AM, DB, and AE all contributed to the manuscript editing process and critical revisions. All authors read and approved the finalized manuscript.</p></notes><notes notes-type="data-availability"><title>Availability of data and material</title><p>The MJS’s whole genome sequence analyzed in this study has been deposited in the NCBI SRA database under accession No. SRS2904218 and NCBI BioSample database under accession No. SAMN08442411. Other datasets are currently available from the corresponding author on reasonable request. Datasets in this study were made using publicly available resources such as PAPGI, EGDP, HOSP and previous studies described in detail in the methods section for each analysis.</p></notes><notes notes-type="ethics"><title>Compliance with ethical standards</title><sec id="FPar1"><title>Conflict of interest</title><p id="Par30">The authors declare they have no conflict of interest.</p></sec><sec id="FPar2"><title>Ethics approval and consent to participate</title><p id="Par31">This study was a part of Korean Personal Genome Project (KPGP) and was approved by the Institutional Review Board at Genome Research Foundation with IRB-REC-20101202-001. MJS also signed a (KPGP) written informed consent to participate in the whole genome sequencing and analysis.</p></sec><sec id="FPar3"><title>Consent for publication</title><p id="Par32">The (KPGP) informed consent included section about data publication, which MJS consented to.</p></sec></notes><ref-list id="Bib1"><title>References</title><ref-list><ref id="CR1"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Abecasis</surname><given-names>GR</given-names></name><name><surname>Altshuler</surname><given-names>D</given-names></name><name><surname>Auton</surname><given-names>A</given-names></name><name><surname>Brooks</surname><given-names>LD</given-names></name><name><surname>Durbin</surname><given-names>RM</given-names></name><name><surname>Gibbs</surname><given-names>RA</given-names></name><etal/></person-group><article-title xml:lang="en">A map of human genome variation from population-scale sequencing</article-title><source>Nature</source><year>2010</year><volume>467</volume><issue>7319</issue><fpage>1061</fpage><lpage>1073</lpage></mixed-citation></ref><ref id="CR2"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Akerov</surname><given-names>TA</given-names></name></person-group><article-title xml:lang="en">On the origin of the Naiman</article-title><source>J Siberian Fed Univ</source><year>2016</year><volume>9</volume><issue>9</issue><fpage>2071</fpage><lpage>2081</lpage><pub-id pub-id-type="doi">10.17516/1997-1370-2016-9-9-2071-2081</pub-id></mixed-citation></ref><ref id="CR3"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Alexander</surname><given-names>DH</given-names></name><name><surname>Novembre</surname><given-names>J</given-names></name><name><surname>Lange</surname><given-names>K</given-names></name></person-group><article-title xml:lang="en">Fast model-based estimation of ancestry in unrelated individuals</article-title><source>Genome Res</source><year>2009</year><volume>19</volume><issue>9</issue><fpage>1655</fpage><lpage>1664</lpage>19648217</mixed-citation></ref><ref id="CR4"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Allentoft</surname><given-names>ME</given-names></name><name><surname>Sikora</surname><given-names>M</given-names></name><name><surname>Sjögren</surname><given-names>K-G</given-names></name><name><surname>Rasmussen</surname><given-names>S</given-names></name><name><surname>Rasmussen</surname><given-names>M</given-names></name><name><surname>Stenderup</surname><given-names>J</given-names></name><etal/></person-group><article-title xml:lang="en">Population genomics of bronze age Eurasia</article-title><source>Nature</source><year>2015</year><volume>522</volume><fpage>167</fpage></mixed-citation></ref><ref id="CR5"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Andrews</surname><given-names>RM</given-names></name><name><surname>Kubacka</surname><given-names>I</given-names></name><name><surname>Chinnery</surname><given-names>PF</given-names></name><name><surname>Lightowlers</surname><given-names>RN</given-names></name><name><surname>Turnbull</surname><given-names>DM</given-names></name><name><surname>Howell</surname><given-names>N</given-names></name></person-group><article-title xml:lang="en">Reanalysis and revision of the Cambridge reference sequence for human mitochondrial DNA</article-title><source>Nat Genet</source><year>1999</year><volume>23</volume><issue>2</issue><fpage>147</fpage></mixed-citation></ref><ref id="CR6"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Arici</surname><given-names>M</given-names></name><name><surname>Özhan</surname><given-names>G</given-names></name></person-group><article-title xml:lang="en">CYP2C9, CYPC19 and CYP2D6 gene profiles and gene susceptibility to drug response and toxicity in Turkish population</article-title><source>Saudi Pharm J.</source><year>2017</year><volume>25</volume><issue>3</issue><fpage>376</fpage><lpage>380</lpage></mixed-citation></ref><ref id="CR7"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ball</surname><given-names>MP</given-names></name><name><surname>Thakuria</surname><given-names>JV</given-names></name><name><surname>Zaranek</surname><given-names>AW</given-names></name><name><surname>Clegg</surname><given-names>T</given-names></name><name><surname>Rosenbaum</surname><given-names>AM</given-names></name><name><surname>Wu</surname><given-names>X</given-names></name><etal/></person-group><article-title xml:lang="en">A public resource facilitating clinical use of genomes</article-title><source>Proc Natl Acad Sci USA</source><year>2012</year><volume>109</volume><issue>30</issue><fpage>11920</fpage><lpage>11927</lpage></mixed-citation></ref><ref id="CR8"><mixed-citation publication-type="other">Berezina G, Svyatova G, Makhmutova Z (2011) The analysis of the genetic structure of the Kazakh population as estimated from mitochondrial DNA polymorphism, pp 2–6</mixed-citation></ref><ref id="CR9"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bicho</surname><given-names>N</given-names></name></person-group><article-title xml:lang="en">Paul Pettitt and Mark White, eds. The British Palaeolithic: Human Societies at the Edge of the Pleistocene World (Routledge: Routledge Archaeology of Northern Europe, Abingdon, 2012, 592 pp., 237 figs., 38 tables, pbk, ISBN 978-0-415-67455-3)</article-title><source>Eur J Archaeol</source><year>2013</year><volume>16</volume><issue>2</issue><fpage>346</fpage><lpage>351</lpage></mixed-citation></ref><ref id="CR10"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Boomsma</surname><given-names>DI</given-names></name><name><surname>Wijmenga</surname><given-names>C</given-names></name><name><surname>Slagboom</surname><given-names>EP</given-names></name><name><surname>Swertz</surname><given-names>MA</given-names></name><name><surname>Karssen</surname><given-names>LC</given-names></name><name><surname>Abdellaoui</surname><given-names>A</given-names></name><etal/></person-group><article-title xml:lang="en">The genome of the Netherlands: design, and project goals</article-title><source>EJHG</source><year>2014</year><volume>22</volume><issue>2</issue><fpage>221</fpage><lpage>227</lpage></mixed-citation></ref><ref id="CR11"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cho</surname><given-names>YS</given-names></name><name><surname>Kim</surname><given-names>H</given-names></name><name><surname>Kim</surname><given-names>HM</given-names></name><name><surname>Jho</surname><given-names>S</given-names></name><name><surname>Jun</surname><given-names>J</given-names></name><name><surname>Lee</surname><given-names>YJ</given-names></name><etal/></person-group><article-title xml:lang="en">An ethnically relevant consensus Korean reference genome is a step towards personal reference genomes</article-title><source>Nat Commun</source><year>2016</year><volume>7</volume><fpage>13637</fpage>5123046</mixed-citation></ref><ref id="CR12"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Choi</surname><given-names>Y</given-names></name><name><surname>Chan</surname><given-names>AP</given-names></name></person-group><article-title xml:lang="en">PROVEAN web server: a tool to predict the functional effect of amino acid substitutions and indels</article-title><source>Bioinformatics (Oxford, England).</source><year>2015</year><volume>31</volume><issue>16</issue><fpage>2745</fpage><lpage>2747</lpage>4528627</mixed-citation></ref><ref id="CR13"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cingolani</surname><given-names>P</given-names></name><name><surname>Platts</surname><given-names>A</given-names></name><name><surname>le Wang</surname><given-names>L</given-names></name><name><surname>Coon</surname><given-names>M</given-names></name><name><surname>Nguyen</surname><given-names>T</given-names></name><name><surname>Wang</surname><given-names>L</given-names></name><etal/></person-group><article-title xml:lang="en">A program for annotating and predicting the effects of single nucleotide polymorphisms, SnpEff: SNPs in the genome of Drosophila melanogaster strain w1118; iso-2; iso-3</article-title><source>Fly.</source><year>2012</year><volume>6</volume><issue>2</issue><fpage>80</fpage><lpage>92</lpage>3679285</mixed-citation></ref><ref id="CR14"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Comas</surname><given-names>D</given-names></name><name><surname>Calafell</surname><given-names>F</given-names></name><name><surname>Mateu</surname><given-names>E</given-names></name><name><surname>Pérez-Lezaun</surname><given-names>A</given-names></name><name><surname>Bosch</surname><given-names>E</given-names></name><name><surname>Martínez-Arias</surname><given-names>R</given-names></name><etal/></person-group><article-title xml:lang="en">Trading genes along the silk road: mtDNA sequences and the origin of Central Asian populations</article-title><source>Am J Hum Genet</source><year>1998</year><volume>63</volume><issue>6</issue><fpage>1824</fpage><lpage>1838</lpage>1377654</mixed-citation></ref><ref id="CR15"><mixed-citation publication-type="other">David Llewelyn Snellgrove JRK (2018) Central Asian arts Encyclopædia Britannica: Encyclopædia Britannica, Inc. <ext-link ext-link-type="uri" xlink:href="https://www.britannica.com/art/Central-Asian-arts/Visual-arts">https://www.britannica.com/art/Central-Asian-arts/Visual-arts</ext-link>. Accessed 4 June 2018</mixed-citation></ref><ref id="CR16"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>de Manuel</surname><given-names>M</given-names></name><name><surname>Kuhlwilm</surname><given-names>M</given-names></name><name><surname>Frandsen</surname><given-names>P</given-names></name><name><surname>Sousa</surname><given-names>VC</given-names></name><name><surname>Desai</surname><given-names>T</given-names></name><name><surname>Prado-Martinez</surname><given-names>J</given-names></name><etal/></person-group><article-title xml:lang="en">Chimpanzee genomic diversity reveals ancient admixture with bonobos</article-title><source>Science</source><year>2016</year><volume>354</volume><issue>6311</issue><fpage>477</fpage><lpage>481</lpage>5546212</mixed-citation></ref><ref id="CR17"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dulik Matthew</surname><given-names>C</given-names></name><name><surname>Zhadanov Sergey</surname><given-names>I</given-names></name><name><surname>Osipova Ludmila</surname><given-names>P</given-names></name><name><surname>Askapuli</surname><given-names>A</given-names></name><name><surname>Gau</surname><given-names>L</given-names></name><name><surname>Gokcumen</surname><given-names>O</given-names></name><etal/></person-group><article-title xml:lang="en">Mitochondrial DNA and Y chromosome variation provides evidence for a recent common ancestry between Native Americans and Indigenous Altaians</article-title><source>Am J Hum Genet</source><year>2012</year><volume>90</volume><issue>2</issue><fpage>229</fpage><lpage>246</lpage>3276666</mixed-citation></ref><ref id="CR18"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dulik</surname><given-names>MC</given-names></name><name><surname>Osipova</surname><given-names>LP</given-names></name><name><surname>Schurr</surname><given-names>TG</given-names></name></person-group><article-title xml:lang="en">Y-chromosome variation in Altaian Kazakhs reveals a common paternal gene pool for Kazakhs and the influence of Mongolian expansions</article-title><source>PLoS ONE</source><year>2011</year><volume>6</volume><issue>3</issue><fpage>e17548</fpage>3055870</mixed-citation></ref><ref id="CR19"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Forde</surname><given-names>CD</given-names></name></person-group><source>Habitat, economy and society: a geographical introduction to ethnology</source><year>1934</year><publisher-loc>Abingdon</publisher-loc><publisher-name>Routledge</publisher-name></mixed-citation></ref><ref id="CR20"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname><given-names>Q</given-names></name><name><surname>Li</surname><given-names>H</given-names></name><name><surname>Moorjani</surname><given-names>P</given-names></name><name><surname>Jay</surname><given-names>F</given-names></name><name><surname>Slepchenko</surname><given-names>SM</given-names></name><name><surname>Bondarev</surname><given-names>AA</given-names></name><etal/></person-group><article-title xml:lang="en">Genome sequence of a 45,000-year-old modern human from western Siberia</article-title><source>Nature</source><year>2014</year><volume>514</volume><issue>7523</issue><fpage>445</fpage><lpage>449</lpage>4753769</mixed-citation></ref><ref id="CR21"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fujimoto</surname><given-names>A</given-names></name><name><surname>Kimura</surname><given-names>R</given-names></name><name><surname>Ohashi</surname><given-names>J</given-names></name><name><surname>Omi</surname><given-names>K</given-names></name><name><surname>Yuliwulandari</surname><given-names>R</given-names></name><name><surname>Batubara</surname><given-names>L</given-names></name><etal/></person-group><article-title xml:lang="en">A scan for genetic determinants of human hair morphology: EDAR is associated with Asian hair thickness</article-title><source>Hum Mol Genet</source><year>2008</year><volume>17</volume><issue>6</issue><fpage>835</fpage><lpage>843</lpage></mixed-citation></ref><ref id="CR22"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gamba</surname><given-names>C</given-names></name><name><surname>Jones</surname><given-names>ER</given-names></name><name><surname>Teasdale</surname><given-names>MD</given-names></name><name><surname>McLaughlin</surname><given-names>RL</given-names></name><name><surname>Gonzalez-Fortes</surname><given-names>G</given-names></name><name><surname>Mattiangeli</surname><given-names>V</given-names></name><etal/></person-group><article-title xml:lang="en">Genome flux and stasis in a five millennium transect of European prehistory</article-title><source>Nature communications.</source><year>2014</year><volume>5</volume><fpage>5257</fpage>4218962</mixed-citation></ref><ref id="CR23"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Gibb</surname><given-names>HAR</given-names></name></person-group><source>The Arab conquests in Central Asia</source><year>2013</year><publisher-loc>New York</publisher-loc><publisher-name>AMS Press</publisher-name></mixed-citation></ref><ref id="CR25"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guryev</surname><given-names>V</given-names></name></person-group><article-title xml:lang="en">Assessment of variant pathogenicity in a highly admixed population</article-title><source>Hum Mutat</source><year>2017</year><volume>38</volume><issue>7</issue><fpage>749</fpage></mixed-citation></ref><ref id="CR26"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Haak</surname><given-names>W</given-names></name><name><surname>Lazaridis</surname><given-names>I</given-names></name><name><surname>Patterson</surname><given-names>N</given-names></name><name><surname>Rohland</surname><given-names>N</given-names></name><name><surname>Mallick</surname><given-names>S</given-names></name><name><surname>Llamas</surname><given-names>B</given-names></name><etal/></person-group><article-title xml:lang="en">Massive migration from the steppe was a source for Indo-European languages in Europe</article-title><source>Nature</source><year>2015</year><volume>522</volume><fpage>207</fpage>5048219</mixed-citation></ref><ref id="CR27"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hartmann</surname><given-names>A</given-names></name><name><surname>Thieme</surname><given-names>M</given-names></name><name><surname>Nanduri</surname><given-names>LK</given-names></name><name><surname>Stempfl</surname><given-names>T</given-names></name><name><surname>Moehle</surname><given-names>C</given-names></name><name><surname>Kivisild</surname><given-names>T</given-names></name><etal/></person-group><article-title xml:lang="en">Validation of microarray-based resequencing of 93 worldwide mitochondrial genomes</article-title><source>Hum Mutat</source><year>2009</year><volume>30</volume><issue>1</issue><fpage>115</fpage><lpage>122</lpage></mixed-citation></ref><ref id="CR28"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Heyer</surname><given-names>E</given-names></name><name><surname>Balaresque</surname><given-names>P</given-names></name><name><surname>Jobling</surname><given-names>MA</given-names></name><name><surname>Quintana-Murci</surname><given-names>L</given-names></name><name><surname>Chaix</surname><given-names>R</given-names></name><name><surname>Segurel</surname><given-names>L</given-names></name><etal/></person-group><article-title xml:lang="en">Genetic diversity and the emergence of ethnic groups in Central Asia</article-title><source>BMC Genet</source><year>2009</year><volume>10</volume><issue>1</issue><fpage>49</fpage>2745423</mixed-citation></ref><ref id="CR29"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Ikawa-Smith</surname><given-names>F</given-names></name></person-group><source>Early paleolithic in South and East Asia</source><year>1978</year><publisher-loc>The Hague</publisher-loc><publisher-name>Mouton</publisher-name></mixed-citation></ref><ref id="CR30"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jeong</surname><given-names>C</given-names></name><name><surname>Balanovsky</surname><given-names>O</given-names></name><name><surname>Lukianova</surname><given-names>E</given-names></name><name><surname>Kahbatkyzy</surname><given-names>N</given-names></name><name><surname>Flegontov</surname><given-names>P</given-names></name><name><surname>Zaporozhchenko</surname><given-names>V</given-names></name><etal/></person-group><article-title xml:lang="en">The genetic history of admixture across inner Eurasia</article-title><source>Nat Ecol Evol</source><year>2019</year><volume>3</volume><issue>6</issue><fpage>966</fpage><lpage>976</lpage>6542712</mixed-citation></ref><ref id="CR31"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jones</surname><given-names>ER</given-names></name><name><surname>Gonzalez-Fortes</surname><given-names>G</given-names></name><name><surname>Connell</surname><given-names>S</given-names></name><name><surname>Siska</surname><given-names>V</given-names></name><name><surname>Eriksson</surname><given-names>A</given-names></name><name><surname>Martiniano</surname><given-names>R</given-names></name><etal/></person-group><article-title xml:lang="en">Upper Palaeolithic genomes reveal deep roots of modern Eurasians</article-title><source>Nature communications.</source><year>2015</year><volume>6</volume><fpage>8912</fpage>4660371</mixed-citation></ref><ref id="CR32"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jordan</surname><given-names>DM</given-names></name><name><surname>Kiezun</surname><given-names>A</given-names></name><name><surname>Baxter</surname><given-names>SM</given-names></name><name><surname>Agarwala</surname><given-names>V</given-names></name><name><surname>Green</surname><given-names>RC</given-names></name><name><surname>Murray</surname><given-names>MF</given-names></name><etal/></person-group><article-title xml:lang="en">Development and validation of a computational method for assessment of missense variants in hypertrophic cardiomyopathy</article-title><source>Am J Hum Genet</source><year>2011</year><volume>88</volume><issue>2</issue><fpage>183</fpage><lpage>192</lpage>3035712</mixed-citation></ref><ref id="CR33"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Juras</surname><given-names>A</given-names></name><name><surname>Krzewińska</surname><given-names>M</given-names></name><name><surname>Nikitin</surname><given-names>AG</given-names></name><name><surname>Ehler</surname><given-names>E</given-names></name><name><surname>Chyleński</surname><given-names>M</given-names></name><name><surname>Łukasik</surname><given-names>S</given-names></name><etal/></person-group><article-title xml:lang="en">Diverse origin of mitochondrial lineages in Iron Age Black Sea Scythians</article-title><source>Sci Rep</source><year>2017</year><volume>7</volume><fpage>43950</fpage>5339713</mixed-citation></ref><ref id="CR34"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kamberov Yana</surname><given-names>G</given-names></name><name><surname>Wang</surname><given-names>S</given-names></name><name><surname>Tan</surname><given-names>J</given-names></name><name><surname>Gerbault</surname><given-names>P</given-names></name><name><surname>Wark</surname><given-names>A</given-names></name><name><surname>Tan</surname><given-names>L</given-names></name><etal/></person-group><article-title xml:lang="en">Modeling recent human evolution in mice by expression of a selected EDAR variant</article-title><source>Cell</source><year>2013</year><volume>152</volume><issue>4</issue><fpage>691</fpage><lpage>702</lpage>3575602</mixed-citation></ref><ref id="CR35"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kang</surname><given-names>HM</given-names></name><name><surname>Sul</surname><given-names>JH</given-names></name><name><surname>Service</surname><given-names>SK</given-names></name><name><surname>Zaitlen</surname><given-names>NA</given-names></name><name><surname>Kong</surname><given-names>S</given-names></name><name><surname>Freimer</surname><given-names>NB</given-names></name><etal/></person-group><article-title xml:lang="en">Variance component model to account for sample structure in genome-wide association studies</article-title><source>Nat Genet</source><year>2010</year><volume>42</volume><fpage>348</fpage>3092069</mixed-citation></ref><ref id="CR36"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Keller</surname><given-names>A</given-names></name><name><surname>Graefen</surname><given-names>A</given-names></name><name><surname>Ball</surname><given-names>M</given-names></name><name><surname>Matzas</surname><given-names>M</given-names></name><name><surname>Boisguerin</surname><given-names>V</given-names></name><name><surname>Maixner</surname><given-names>F</given-names></name><etal/></person-group><article-title xml:lang="en">New insights into the Tyrolean Iceman's origin and phenotype as inferred by whole-genome sequencing</article-title><source>Nature communications.</source><year>2012</year><volume>3</volume><fpage>698</fpage></mixed-citation></ref><ref id="CR37"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname><given-names>J</given-names></name><name><surname>Weber</surname><given-names>JA</given-names></name><name><surname>Jho</surname><given-names>S</given-names></name><name><surname>Jang</surname><given-names>J</given-names></name><name><surname>Jun</surname><given-names>J</given-names></name><name><surname>Cho</surname><given-names>YS</given-names></name><etal/></person-group><article-title xml:lang="en">KoVariome: Korean National Standard Reference Variome database of whole genomes with comprehensive SNV, indel, CNV, and SV analyses</article-title><source>Sci Rep</source><year>2018</year><volume>8</volume><issue>1</issue><fpage>5677</fpage>5885007</mixed-citation></ref><ref id="CR38"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Klein</surname><given-names>RG</given-names></name></person-group><article-title xml:lang="en">Conceptual issues in modern human origins research</article-title><source>Am J Hum Biol</source><year>1999</year><volume>11</volume><issue>1</issue><fpage>79</fpage></mixed-citation></ref><ref id="CR40"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Krader</surname><given-names>L</given-names></name></person-group><article-title xml:lang="en">Social organization of the Mongol-Turkic pastoral nomads</article-title><source>Bull Sch Orient Afr Stud</source><year>1966</year><volume>20</volume><issue>2</issue><fpage>412</fpage></mixed-citation></ref><ref id="CR41"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kumar</surname><given-names>S</given-names></name><name><surname>Stecher</surname><given-names>G</given-names></name><name><surname>Tamura</surname><given-names>K</given-names></name></person-group><article-title xml:lang="en">MEGA7: molecular evolutionary genetics analysis version 7.0 for bigger datasets</article-title><source>Mol Biol Evol</source><year>2016</year><volume>33</volume><issue>7</issue><fpage>1870</fpage><lpage>1874</lpage></mixed-citation></ref><ref id="CR42"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lander</surname><given-names>ES</given-names></name><name><surname>Linton</surname><given-names>LM</given-names></name><name><surname>Birren</surname><given-names>B</given-names></name><name><surname>Nusbaum</surname><given-names>C</given-names></name><name><surname>Zody</surname><given-names>MC</given-names></name><name><surname>Baldwin</surname><given-names>J</given-names></name><etal/></person-group><article-title xml:lang="en">Initial sequencing and analysis of the human genome</article-title><source>Nature</source><year>2001</year><volume>409</volume><issue>6822</issue><fpage>860</fpage><lpage>921</lpage></mixed-citation></ref><ref id="CR43"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Landrum</surname><given-names>MJ</given-names></name><name><surname>Lee</surname><given-names>JM</given-names></name><name><surname>Riley</surname><given-names>GR</given-names></name><name><surname>Jang</surname><given-names>W</given-names></name><name><surname>Rubinstein</surname><given-names>WS</given-names></name><name><surname>Church</surname><given-names>DM</given-names></name><etal/></person-group><article-title xml:lang="en">ClinVar: public archive of relationships among sequence variation and human phenotype</article-title><source>Nucleic Acids Res</source><year>2014</year><volume>42</volume><issue>Database issue</issue><fpage>D980</fpage><lpage>D985</lpage></mixed-citation></ref><ref id="CR44"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lassmann</surname><given-names>T</given-names></name><name><surname>Hayashizaki</surname><given-names>Y</given-names></name><name><surname>Daub</surname><given-names>CO</given-names></name></person-group><article-title xml:lang="en">SAMStat: monitoring biases in next generation sequencing data</article-title><source>Bioinformatics (Oxford, England)</source><year>2011</year><volume>27</volume><issue>1</issue><fpage>130</fpage><lpage>131</lpage></mixed-citation></ref><ref id="CR45"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lazaridis</surname><given-names>I</given-names></name><name><surname>Patterson</surname><given-names>N</given-names></name><name><surname>Mittnik</surname><given-names>A</given-names></name><name><surname>Renaud</surname><given-names>G</given-names></name><name><surname>Mallick</surname><given-names>S</given-names></name><name><surname>Kirsanow</surname><given-names>K</given-names></name><etal/></person-group><article-title xml:lang="en">Ancient human genomes suggest three ancestral populations for present-day Europeans</article-title><source>Nature</source><year>2014</year><volume>513</volume><issue>7518</issue><fpage>409</fpage><lpage>413</lpage>4170574</mixed-citation></ref><ref id="CR46"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>H</given-names></name><name><surname>Durbin</surname><given-names>R</given-names></name></person-group><article-title xml:lang="en">Fast and accurate short read alignment with Burrows-Wheeler transform</article-title><source>Bioinformatics (Oxford, England)</source><year>2009</year><volume>25</volume><issue>14</issue><fpage>1754</fpage><lpage>1760</lpage></mixed-citation></ref><ref id="CR47"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>H</given-names></name><name><surname>Durbin</surname><given-names>R</given-names></name></person-group><article-title xml:lang="en">Inference of human population history from whole genome sequence of a single individual</article-title><source>Nature</source><year>2011</year><volume>475</volume><issue>7357</issue><fpage>493</fpage><lpage>496</lpage><pub-id pub-id-type="doi">10.1038/nature10231</pub-id>3154645</mixed-citation></ref><ref id="CR48"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>H</given-names></name><name><surname>Handsaker</surname><given-names>B</given-names></name><name><surname>Wysoker</surname><given-names>A</given-names></name><name><surname>Fennell</surname><given-names>T</given-names></name><name><surname>Ruan</surname><given-names>J</given-names></name><name><surname>Homer</surname><given-names>N</given-names></name><etal/></person-group><article-title xml:lang="en">The Sequence Alignment/Map format and SAMtools</article-title><source>Bioinformatics (Oxford, England)</source><year>2009</year><volume>25</volume><issue>16</issue><fpage>2078</fpage><lpage>2079</lpage></mixed-citation></ref><ref id="CR49"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Link</surname><given-names>E</given-names></name><name><surname>Parish</surname><given-names>S</given-names></name><name><surname>Armitage</surname><given-names>J</given-names></name><name><surname>Bowman</surname><given-names>L</given-names></name><name><surname>Heath</surname><given-names>S</given-names></name><name><surname>Matsuda</surname><given-names>F</given-names></name><etal/></person-group><article-title xml:lang="en">SLCO1B1 variants and statin-induced myopathy–a genomewide study</article-title><source>The New England journal of medicine.</source><year>2008</year><volume>359</volume><issue>8</issue><fpage>789</fpage><lpage>799</lpage></mixed-citation></ref><ref id="CR50"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lipson</surname><given-names>M</given-names></name><name><surname>Cheronet</surname><given-names>O</given-names></name><name><surname>Mallick</surname><given-names>S</given-names></name><name><surname>Rohland</surname><given-names>N</given-names></name><name><surname>Oxenham</surname><given-names>M</given-names></name><name><surname>Pietrusewsky</surname><given-names>M</given-names></name><etal/></person-group><article-title xml:lang="en">Ancient genomes document multiple waves of migration in Southeast Asian prehistory</article-title><source>Science</source><year>2018</year><volume>361</volume><issue>6397</issue><fpage>92</fpage><lpage>95</lpage>6476732</mixed-citation></ref><ref id="CR51"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Magalon</surname><given-names>H</given-names></name><name><surname>Patin</surname><given-names>E</given-names></name><name><surname>Austerlitz</surname><given-names>F</given-names></name><name><surname>Hegay</surname><given-names>T</given-names></name><name><surname>Aldashev</surname><given-names>A</given-names></name><name><surname>Quintana-Murci</surname><given-names>L</given-names></name><etal/></person-group><article-title xml:lang="en">Population genetic diversity of the NAT2 gene supports a role of acetylation in human adaptation to farming in Central Asia</article-title><source>European journal of human genetics: EJHG.</source><year>2008</year><volume>16</volume><issue>2</issue><fpage>243</fpage><lpage>251</lpage></mixed-citation></ref><ref id="CR52"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mallick</surname><given-names>S</given-names></name><name><surname>Li</surname><given-names>H</given-names></name><name><surname>Lipson</surname><given-names>M</given-names></name><name><surname>Mathieson</surname><given-names>I</given-names></name><name><surname>Gymrek</surname><given-names>M</given-names></name><name><surname>Racimo</surname><given-names>F</given-names></name><etal/></person-group><article-title xml:lang="en">The Simons genome diversity project: 300 genomes from 142 diverse populations</article-title><source>Nature</source><year>2016</year><volume>538</volume><fpage>201</fpage>5161557</mixed-citation></ref><ref id="CR53"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mazzaccara</surname><given-names>C</given-names></name><name><surname>Limongelli</surname><given-names>G</given-names></name><name><surname>Petretta</surname><given-names>M</given-names></name><name><surname>Vastarella</surname><given-names>R</given-names></name><name><surname>Pacileo</surname><given-names>G</given-names></name><name><surname>Bonaduce</surname><given-names>D</given-names></name><etal/></person-group><article-title xml:lang="en">A common polymorphism in the SCN5A gene is associated with dilated cardiomyopathy</article-title><source>J Cardiovasc Med (Hagerstown).</source><year>2018</year><volume>19</volume><issue>7</issue><fpage>344</fpage><lpage>350</lpage>6012048</mixed-citation></ref><ref id="CR54"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McKenna</surname><given-names>A</given-names></name><name><surname>Hanna</surname><given-names>M</given-names></name><name><surname>Banks</surname><given-names>E</given-names></name><name><surname>Sivachenko</surname><given-names>A</given-names></name><name><surname>Cibulskis</surname><given-names>K</given-names></name><name><surname>Kernytsky</surname><given-names>A</given-names></name><etal/></person-group><article-title xml:lang="en">The genome analysis toolkit: a MapReduce framework for analyzing next-generation DNA sequencing data</article-title><source>Genome Res</source><year>2010</year><volume>20</volume><issue>9</issue><fpage>1297</fpage><lpage>1303</lpage>2928508</mixed-citation></ref><ref id="CR55"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Medina-Gomez</surname><given-names>C</given-names></name><name><surname>Felix</surname><given-names>JF</given-names></name><name><surname>Estrada</surname><given-names>K</given-names></name><name><surname>Peters</surname><given-names>MJ</given-names></name><name><surname>Herrera</surname><given-names>L</given-names></name><name><surname>Kruithof</surname><given-names>CJ</given-names></name><etal/></person-group><article-title xml:lang="en">Challenges in conducting genome-wide association studies in highly admixed multi-ethnic populations: the Generation R Study</article-title><source>Eur J Epidemiol</source><year>2015</year><volume>30</volume><issue>4</issue><fpage>317</fpage><lpage>330</lpage>4385148</mixed-citation></ref><ref id="CR56"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Morgan</surname><given-names>D</given-names></name></person-group><source>The Mongols</source><year>2007</year><edition>2</edition><publisher-loc>Oxford</publisher-loc><publisher-name>Blackwell</publisher-name></mixed-citation></ref><ref id="CR57"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mostafa</surname><given-names>G</given-names></name></person-group><article-title xml:lang="en">The Concept of ‘Eurasia’: Kazakhstan's Eurasian Policy and its Implications</article-title><source>J Eur Stud</source><year>2013</year><volume>4</volume><issue>2</issue><fpage>160</fpage><lpage>170</lpage></mixed-citation></ref><ref id="CR58"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nagasaki</surname><given-names>M</given-names></name><name><surname>Yasuda</surname><given-names>J</given-names></name><name><surname>Katsuoka</surname><given-names>F</given-names></name><name><surname>Nariai</surname><given-names>N</given-names></name><name><surname>Kojima</surname><given-names>K</given-names></name><name><surname>Kawai</surname><given-names>Y</given-names></name><etal/></person-group><article-title xml:lang="en">Rare variant discovery by deep whole-genome sequencing of 1,070 Japanese individuals</article-title><source>Nat Commun</source><year>2015</year><volume>6</volume><fpage>8018</fpage>4560751</mixed-citation></ref><ref id="CR59"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nasidze</surname><given-names>I</given-names></name><name><surname>Quinque</surname><given-names>D</given-names></name><name><surname>Dupanloup</surname><given-names>I</given-names></name><name><surname>Cordaux</surname><given-names>R</given-names></name><name><surname>Kokshunova</surname><given-names>L</given-names></name><name><surname>Stoneking</surname><given-names>M</given-names></name></person-group><article-title xml:lang="en">Genetic evidence for the Mongolian ancestry of Kalmyks</article-title><source>Am J Phys Anthropol</source><year>2005</year><volume>128</volume><issue>4</issue><fpage>846</fpage><lpage>854</lpage></mixed-citation></ref><ref id="CR60"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ng</surname><given-names>PC</given-names></name><name><surname>Henikoff</surname><given-names>S</given-names></name></person-group><article-title xml:lang="en">SIFT: Predicting amino acid changes that affect protein function</article-title><source>Nucleic Acids Res</source><year>2003</year><volume>31</volume><issue>13</issue><fpage>3812</fpage><lpage>3814</lpage>12824425</mixed-citation></ref><ref id="CR61"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Oktyabrskaya</surname><given-names>IV</given-names></name></person-group><article-title xml:lang="en">Winter history of the Turata Kazakhs</article-title><source>Archaeol Ethnol Anthropol Eurasia</source><year>2006</year><volume>25</volume><issue>1</issue><fpage>132</fpage><lpage>144</lpage></mixed-citation></ref><ref id="CR62"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Olcott</surname><given-names>MB</given-names></name></person-group><source>The Kazakhs</source><year>1995</year><edition>2</edition><publisher-loc>Stanford</publisher-loc><publisher-name>Hoover Institution Press</publisher-name></mixed-citation></ref><ref id="CR63"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Oleksyk</surname><given-names>TK</given-names></name><name><surname>Brukhin</surname><given-names>V</given-names></name><name><surname>O'Brien</surname><given-names>SJ</given-names></name></person-group><article-title xml:lang="en">The Genome Russia project: closing the largest remaining omission on the world Genome map</article-title><source>GigaScience.</source><year>2015</year><volume>4</volume><fpage>53</fpage>4644275</mixed-citation></ref><ref id="CR64"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pagani</surname><given-names>L</given-names></name><name><surname>Lawson</surname><given-names>DJ</given-names></name><name><surname>Jagoda</surname><given-names>E</given-names></name><name><surname>Mörseburg</surname><given-names>A</given-names></name><name><surname>Eriksson</surname><given-names>A</given-names></name><name><surname>Mitt</surname><given-names>M</given-names></name><etal/></person-group><article-title xml:lang="en">Genomic analyses inform on migration events during the peopling of Eurasia</article-title><source>Nature</source><year>2016</year><volume>538</volume><fpage>238</fpage>5164938</mixed-citation></ref><ref id="CR65"><mixed-citation publication-type="other">Paksoy HZV (1992) Togan: The Origins of the Kazaks and the Özbeks, pp 83–100</mixed-citation></ref><ref id="CR66"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Patel</surname><given-names>RK</given-names></name><name><surname>Jain</surname><given-names>M</given-names></name></person-group><article-title xml:lang="en">NGS QC Toolkit: a toolkit for quality control of next generation sequencing data</article-title><source>PLoS ONE</source><year>2012</year><volume>7</volume><issue>2</issue><fpage>e30619-e</fpage></mixed-citation></ref><ref id="CR67"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Patterson</surname><given-names>N</given-names></name><name><surname>Moorjani</surname><given-names>P</given-names></name><name><surname>Luo</surname><given-names>Y</given-names></name><name><surname>Mallick</surname><given-names>S</given-names></name><name><surname>Rohland</surname><given-names>N</given-names></name><name><surname>Zhan</surname><given-names>Y</given-names></name><etal/></person-group><article-title xml:lang="en">Ancient admixture in human history</article-title><source>Genetics</source><year>2012</year><volume>192</volume><issue>3</issue><fpage>1065</fpage>3522152</mixed-citation></ref><ref id="CR68"><mixed-citation publication-type="other">Perna S, Riva A, Nicosanti G, Carrai M, Barale R, Vigo B, et al (2017) Association of the bitter taste receptor gene TAS2R38 (polymorphism RS713598) with sensory responsiveness, food preferences, biochemical parameters and body-composition markers. A cross-sectional study in Italy, pp 1–8</mixed-citation></ref><ref id="CR69"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Purcell</surname><given-names>S</given-names></name><name><surname>Neale</surname><given-names>B</given-names></name><name><surname>Todd-Brown</surname><given-names>K</given-names></name><name><surname>Thomas</surname><given-names>L</given-names></name><name><surname>Ferreira</surname><given-names>MA</given-names></name><name><surname>Bender</surname><given-names>D</given-names></name><etal/></person-group><article-title xml:lang="en">PLINK: a tool set for whole-genome association and population-based linkage analyses</article-title><source>Am J Hum Genet</source><year>2007</year><volume>81</volume><issue>3</issue><fpage>559</fpage><lpage>575</lpage>17701901</mixed-citation></ref><ref id="CR70"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Raghavan</surname><given-names>M</given-names></name><name><surname>Skoglund</surname><given-names>P</given-names></name><name><surname>Graf</surname><given-names>KE</given-names></name><name><surname>Metspalu</surname><given-names>M</given-names></name><name><surname>Albrechtsen</surname><given-names>A</given-names></name><name><surname>Moltke</surname><given-names>I</given-names></name><etal/></person-group><article-title xml:lang="en">Upper Palaeolithic Siberian genome reveals dual ancestry of Native Americans</article-title><source>Nature</source><year>2014</year><volume>505</volume><issue>7481</issue><fpage>87</fpage><lpage>91</lpage></mixed-citation></ref><ref id="CR71"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Samuel</surname><given-names>GN</given-names></name><name><surname>Farsides</surname><given-names>B</given-names></name></person-group><article-title xml:lang="en">The UK's 100,000 genomes project: manifesting policymakers’ expectations</article-title><source>New genetics and society.</source><year>2017</year><volume>36</volume><issue>4</issue><fpage>336</fpage><lpage>353</lpage>5706982</mixed-citation></ref><ref id="CR72"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Scally</surname><given-names>A</given-names></name></person-group><article-title xml:lang="en">The mutation rate in human evolution and demographic inference</article-title><source>Curr Opin Genet Dev</source><year>2016</year><volume>41</volume><fpage>36</fpage><lpage>43</lpage></mixed-citation></ref><ref id="CR73"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schiffels</surname><given-names>S</given-names></name><name><surname>Durbin</surname><given-names>R</given-names></name></person-group><article-title xml:lang="en">Inferring human population size and separation history from multiple genome sequences</article-title><source>Nat Genet</source><year>2014</year><volume>46</volume><fpage>919</fpage>4116295</mixed-citation></ref><ref id="CR74"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Seguin-Orlando</surname><given-names>A</given-names></name><name><surname>Korneliussen</surname><given-names>TS</given-names></name><name><surname>Sikora</surname><given-names>M</given-names></name><name><surname>Malaspinas</surname><given-names>AS</given-names></name><name><surname>Manica</surname><given-names>A</given-names></name><name><surname>Moltke</surname><given-names>I</given-names></name><etal/></person-group><article-title xml:lang="en">Paleogenomics. Genomic structure in Europeans dating back at least 36,200 years</article-title><source>Science</source><year>2014</year><volume>346</volume><issue>6213</issue><fpage>1113</fpage><lpage>1118</lpage></mixed-citation></ref><ref id="CR75"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Sinor</surname><given-names>D</given-names></name></person-group><person-group person-group-type="editor"><name><surname>Sinor</surname><given-names>D</given-names></name></person-group><article-title xml:lang="en">The Hun period</article-title><source>The Cambridge history of early inner Asia</source><year>1990</year><publisher-loc>Cambridge</publisher-loc><publisher-name>Cambridge University Press</publisher-name><fpage>177</fpage><lpage>205</lpage></mixed-citation></ref><ref id="CR76"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Siska</surname><given-names>V</given-names></name><name><surname>Jones</surname><given-names>ER</given-names></name><name><surname>Jeon</surname><given-names>S</given-names></name><name><surname>Bhak</surname><given-names>Y</given-names></name><name><surname>Kim</surname><given-names>H-M</given-names></name><name><surname>Cho</surname><given-names>YS</given-names></name><etal/></person-group><article-title xml:lang="en">Genome-wide data from two early Neolithic East Asian individuals dating to 7700 years ago</article-title><source>Science Advances.</source><year>2017</year><volume>3</volume><issue>2</issue><fpage>e1601877</fpage>5287702</mixed-citation></ref><ref id="CR77"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Skoglund</surname><given-names>P</given-names></name><name><surname>Malmstrom</surname><given-names>H</given-names></name><name><surname>Omrak</surname><given-names>A</given-names></name><name><surname>Raghavan</surname><given-names>M</given-names></name><name><surname>Valdiosera</surname><given-names>C</given-names></name><name><surname>Gunther</surname><given-names>T</given-names></name><etal/></person-group><article-title xml:lang="en">Genomic diversity and admixture differs for Stone-Age Scandinavian foragers and farmers</article-title><source>Science</source><year>2014</year><volume>344</volume><issue>6185</issue><fpage>747</fpage><lpage>750</lpage></mixed-citation></ref><ref id="CR78"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Soejima</surname><given-names>M</given-names></name><name><surname>Koda</surname><given-names>Y</given-names></name></person-group><article-title xml:lang="en">Population differences of two coding SNPs in pigmentation-related genes SLC24A5 and SLC45A2</article-title><source>Int J Legal Med</source><year>2007</year><volume>121</volume><issue>1</issue><fpage>36</fpage><lpage>39</lpage></mixed-citation></ref><ref id="CR79"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sudmant</surname><given-names>PH</given-names></name><name><surname>Rausch</surname><given-names>T</given-names></name><name><surname>Gardner</surname><given-names>EJ</given-names></name><name><surname>Handsaker</surname><given-names>RE</given-names></name><name><surname>Abyzov</surname><given-names>A</given-names></name><name><surname>Huddleston</surname><given-names>J</given-names></name><etal/></person-group><article-title xml:lang="en">An integrated map of structural variation in 2,504 human genomes</article-title><source>Nature</source><year>2015</year><volume>526</volume><fpage>75</fpage>4617611</mixed-citation></ref><ref id="CR80"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tarlykov</surname><given-names>PV</given-names></name><name><surname>Zholdybayeva</surname><given-names>EV</given-names></name><name><surname>Akilzhanova</surname><given-names>AR</given-names></name><name><surname>Nurkina</surname><given-names>ZM</given-names></name><name><surname>Sabitov</surname><given-names>ZM</given-names></name><name><surname>Rakhypbekov</surname><given-names>TK</given-names></name><etal/></person-group><article-title xml:lang="en">Mitochondrial and Y-chromosomal profile of the Kazakh population from East Kazakhstan</article-title><source>Croat Med J</source><year>2013</year><volume>54</volume><issue>1</issue><fpage>17</fpage><lpage>24</lpage>3583390</mixed-citation></ref><ref id="CR81"><mixed-citation publication-type="other">Team RC (2018) R: a language and environment for statistical computing. R Foundation for Statistical Computing</mixed-citation></ref><ref id="CR82"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vatsis</surname><given-names>KP</given-names></name><name><surname>Martell</surname><given-names>KJ</given-names></name><name><surname>Weber</surname><given-names>WW</given-names></name></person-group><article-title xml:lang="en">Diverse point mutations in the human gene for polymorphic N-acetyltransferase</article-title><source>Proc Natl Acad Sci USA</source><year>1991</year><volume>88</volume><issue>14</issue><fpage>6333</fpage><lpage>6337</lpage></mixed-citation></ref><ref id="CR83"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Voora</surname><given-names>D</given-names></name><name><surname>Shah</surname><given-names>SH</given-names></name><name><surname>Spasojevic</surname><given-names>I</given-names></name><name><surname>Ali</surname><given-names>S</given-names></name><name><surname>Reed</surname><given-names>CR</given-names></name><name><surname>Salisbury</surname><given-names>BA</given-names></name><etal/></person-group><article-title xml:lang="en">The SLCO1B1*5 genetic variant is associated with statin-induced side effects</article-title><source>J Am Coll Cardiol</source><year>2009</year><volume>54</volume><issue>17</issue><fpage>1609</fpage><lpage>1616</lpage>3417133</mixed-citation></ref><ref id="CR84"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Walter</surname><given-names>K</given-names></name><name><surname>Min</surname><given-names>JL</given-names></name><name><surname>Huang</surname><given-names>J</given-names></name><name><surname>Crooks</surname><given-names>L</given-names></name><name><surname>Memari</surname><given-names>Y</given-names></name><name><surname>McCarthy</surname><given-names>S</given-names></name><etal/></person-group><article-title xml:lang="en">The UK10K project identifies rare variants in health and disease</article-title><source>Nature</source><year>2015</year><volume>526</volume><issue>7571</issue><fpage>82</fpage><lpage>90</lpage></mixed-citation></ref><ref id="CR85"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>K</given-names></name><name><surname>Li</surname><given-names>M</given-names></name><name><surname>Hakonarson</surname><given-names>H</given-names></name></person-group><article-title xml:lang="en">ANNOVAR: functional annotation of genetic variants from high-throughput sequencing data</article-title><source>Nucleic Acids Res</source><year>2010</year><volume>38</volume><issue>16</issue><fpage>e164</fpage>20601685</mixed-citation></ref><ref id="CR86"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Weissensteiner</surname><given-names>H</given-names></name><name><surname>Pacher</surname><given-names>D</given-names></name><name><surname>Kloss-Brandstatter</surname><given-names>A</given-names></name><name><surname>Forer</surname><given-names>L</given-names></name><name><surname>Specht</surname><given-names>G</given-names></name><name><surname>Bandelt</surname><given-names>HJ</given-names></name><etal/></person-group><article-title xml:lang="en">HaploGrep 2: mitochondrial haplogroup classification in the era of high-throughput sequencing</article-title><source>Nucleic Acids Res</source><year>2016</year><volume>44</volume><issue>W1</issue><fpage>W58</fpage><lpage>W63</lpage>4987869</mixed-citation></ref><ref id="CR24"><mixed-citation publication-type="other">Weissleder W (1978) The nomadic alternative: modes and models of interaction in the African-Asian deserts and steppes, p 157–163</mixed-citation></ref><ref id="CR87"><mixed-citation publication-type="other">West BA (2009) Turkic peoples. In: Encyclopedia of the peoples of Asia and Oceania. Facts On File Inc., New York, p 829</mixed-citation></ref><ref id="CR88"><mixed-citation publication-type="other">WHO (2015) Kazakhstan: WHO statistical profile 2015 January. <ext-link ext-link-type="uri" xlink:href="http://www.who.int/countries/kaz/en/">http://www.who.int/countries/kaz/en/</ext-link></mixed-citation></ref><ref id="CR89"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Wickham</surname><given-names>H</given-names></name></person-group><source>Ggplot2: elegant graphics for data analysis</source><year>2009</year><edition>1</edition><publisher-loc>New York</publisher-loc><publisher-name>Springer</publisher-name></mixed-citation></ref><ref id="CR90"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yoshiura</surname><given-names>K</given-names></name><name><surname>Kinoshita</surname><given-names>A</given-names></name><name><surname>Ishida</surname><given-names>T</given-names></name><name><surname>Ninokata</surname><given-names>A</given-names></name><name><surname>Ishikawa</surname><given-names>T</given-names></name><name><surname>Kaname</surname><given-names>T</given-names></name><etal/></person-group><article-title xml:lang="en">A SNP in the ABCC11 gene is the determinant of human earwax type</article-title><source>Nat Genet</source><year>2006</year><volume>38</volume><issue>3</issue><fpage>324</fpage><lpage>330</lpage></mixed-citation></ref><ref id="CR91"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname><given-names>GK</given-names></name><name><surname>Smith</surname><given-names>D</given-names></name><name><surname>Zhu</surname><given-names>H</given-names></name><name><surname>Guan</surname><given-names>Y</given-names></name><name><surname>Tsan-Yuk Lam</surname><given-names>T</given-names></name></person-group><article-title xml:lang="en">GGTREE : an R package for visualization and annotation of phylogenetic trees with their covariates and other associated data</article-title><source>Methods Ecol Evol</source><year>2017</year><volume>8</volume><fpage>28</fpage><lpage>36</lpage><pub-id pub-id-type="doi">10.1111/2041-210X.12628</pub-id></mixed-citation></ref><ref id="CR92"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname><given-names>G-X</given-names></name><name><surname>Shen</surname><given-names>M-L</given-names></name><name><surname>Zhang</surname><given-names>Z</given-names></name><name><surname>Wang</surname><given-names>P</given-names></name><name><surname>Xie</surname><given-names>C-X</given-names></name><name><surname>He</surname><given-names>G-H</given-names></name></person-group><article-title xml:lang="en">Association between EPHX1 polymorphisms and carbamazepine metabolism in epilepsy: a meta-analysis</article-title><source>Int J Clin Pharm</source><year>2019</year><volume>41</volume><fpage>1414</fpage><lpage>1428</lpage></mixed-citation></ref></ref-list></ref-list><app-group><app id="App1"><sec id="Sec19"><title>Electronic supplementary material</title><p id="Par33">Below is the link to the electronic supplementary material. <supplementary-material content-type="local-data" id="MOESM1" xlink:title="Electronic supplementary material"><media xlink:href="439_2020_2132_MOESM1_ESM.tif" mimetype="image" mime-subtype="tiff"><caption xml:lang="en"><p><bold>Fig. S1.</bold> The cross validation error values in ADMIXTURE in relation to <italic>K.</italic> (TIFF 43 kb)</p></caption></media></supplementary-material><supplementary-material content-type="local-data" id="MOESM2" xlink:title="Electronic supplementary material"><media xlink:href="439_2020_2132_MOESM2_ESM.tif" mimetype="image" mime-subtype="tiff"><caption xml:lang="en"><p><bold>Fig. S2.</bold> Heterozygous to homozygous SNP ratios of genomes from the PAPGI dataset. Genomes were grouped into boxplots by their continents. Red line indicates the ratio of MJS. (TIFF 389 kb)</p></caption></media></supplementary-material><supplementary-material content-type="local-data" id="MOESM3" xlink:title="Electronic supplementary material"><media xlink:href="439_2020_2132_MOESM3_ESM.pdf" mimetype="application" mime-subtype="pdf"><caption xml:lang="en"><p><bold>Fig. S3.</bold> An ADMIXTURE plot showing the increasing complexity of MJS genome as the number of artificial ancestral groups increases from <italic>K</italic> = 2 to K = 8. Each sample is represented by a colored bar. The colors within the bars indicate possible ancestral groups. Shared colored fractions among the samples indicate shared artificial ancestry. (PDF 15 kb)</p></caption></media></supplementary-material><supplementary-material content-type="local-data" id="MOESM4" xlink:title="Electronic supplementary material"><media xlink:href="439_2020_2132_MOESM4_ESM.jpg" mimetype="image" mime-subtype="jpeg"><caption xml:lang="en"><p><bold>Fig. S4.</bold> The qpGraph results of four Kazakh admixture models. The dashed lines represent admixture events tested and the percentages denote the proportions of admixture relative to the two admixture sources. Units along the solid lines represent the measure of drift. The Z-scores are − 2.918, − 2.794 for panels A, B, respectively, and − 2.617 for panels C and D. (JPEG 3610 kb)</p></caption></media></supplementary-material><supplementary-material content-type="local-data" id="MOESM5" xlink:title="Electronic supplementary material"><media xlink:href="439_2020_2132_MOESM5_ESM.tif" mimetype="image" mime-subtype="tiff"><caption xml:lang="en"><p><bold>Fig. S5.</bold> The phylogenetic relationship of MJS and various populations. A) Phylogenetic tree based on pairwise nucleotide distances between the MJS and other population samples. B) Geographical location of samples used in the phylogenetic tree construction (TIFF 2040 kb)</p></caption></media></supplementary-material><supplementary-material content-type="local-data" id="MOESM6" xlink:title="Electronic supplementary material"><media xlink:href="439_2020_2132_MOESM6_ESM.tif" mimetype="image" mime-subtype="tiff"><caption xml:lang="en"><p><bold>Fig. S6.</bold> Relative cross-coalescence rate over time showing the genomic diversification history of the Kazakh individual (MJS) (TIFF 1089 kb)</p></caption></media></supplementary-material><supplementary-material content-type="local-data" id="MOESM7" xlink:title="Electronic supplementary material"><media xlink:href="439_2020_2132_MOESM7_ESM.tif" mimetype="image" mime-subtype="tiff"><caption xml:lang="en"><p><bold>Fig. S7.</bold> The PSMC analysis showing retrospective changes in effective population size <italic>N</italic><sub><italic>e</italic></sub> of the Kazakh (MJS). The effective population size <italic>N</italic><sub><italic>e</italic></sub> of the Kazakh is based on MJS genome, plotted together with <italic>N</italic><sub><italic>e</italic></sub> of African (Bot San), Northeast and South Asian (Han, Korean, Mongolian, Koryak, and Pathan), European (French), and Middle Easterner (Turkish) genomes. (TIFF 1146 kb)</p></caption></media></supplementary-material><supplementary-material content-type="local-data" id="MOESM8" xlink:title="Electronic supplementary material"><media xlink:href="439_2020_2132_MOESM8_ESM.xlsx" mimetype="application" mime-subtype="vnd.ms-excel"><caption xml:lang="en"><p><bold>Table S1:</bold> Pathogenic and phenotypic variant analysis in the Kazakh based on the findings in MJS genome. <bold>Table S2:</bold> Z and <italic>f3</italic> values of MJS in an admixture <italic>f3</italic> test. <bold>Table S3:</bold> Mapping statistics of MJS genome. <bold>Table S4:</bold> A list of the variants in MJS's mtDNA. <bold>Table S5:</bold> A list of non-synonymous known deleterious SNVs confirmed by SIFT, Polyphen2, and PROVEAN prediction methods. <bold>Table S6:</bold> A list of non-synonymous novel deleterious SNVs confirmed by SIFT, Polyphen2, and PROVEAN prediction methods. <bold>Table S7:</bold> SNVs predicted to be pathogenic in ClinVar database (v20170130). <bold>Table S8:</bold> SNVs related to drug response in ClinVar (v20170130). <bold>Table S9:</bold> Kazakh genomes ADMIXTURE components. <bold>Table S10:</bold> Phylogenetic distance score matrix of genomes (XLSX 1104 kb)</p></caption></media></supplementary-material></p></sec></app></app-group><notes notes-type="Misc"><title>Publisher's Note</title><p>Springer Nature remains neutral with regard to jurisdictional claims in published maps and institutional affiliations.</p></notes></back></article>