<?xml version="1.0" encoding="UTF-8"?>

<?xml-stylesheet type="text/xsl" href="/static/oaitohtml.xsl"?>

<!--
<?xml-stylesheet type="text/xsl" href="/oaitohtml.xsl"?>
-->

<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
    <responseDate>2026-10-12T06:14:00Z</responseDate>
    <request verb="GetRecord" metadataPrefix="oai_dc" identifier="10.11922/sciencedb.j00001.00340" >https://www.scidb.cn/oai</request>
<GetRecord>
    <record>
    <header >
    <identifier>10.11922/sciencedb.j00001.00340</identifier>
    <datestamp>2022-01-28T14:59:01Z</datestamp>
</header>
    <metadata>
        
<oai_dc:dc xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
  <dc:date>2022-01-28</dc:date>
  <dc:title>Evaluation data set of Word Segmentation technology in Minority Languages (MLWS2021)</dc:title>
  <dc:identifier>doi:10.11922/sciencedb.j00001.00340</dc:identifier>
  <dc:language>en</dc:language>
  <dc:description>MLWS2021 word segmentation evaluation data set includes Mongolian, Tibetan and Uyghur languages. The evaluation object is the core technology of automatic word segmentation in Mongolian, Uyghur and Tibetan Languages. On the basis of MLWS2017, the data set is expanded from the previous news field to news, economy, law, entertainment and other fields; The data scale has also expanded from more than 30000 sentences before to 155000 sentences at present. The dataset includes three data files, including: (1) Tibetan.Zip is Tibetan word segmentation and annotation data, with a data volume of 25000 sentences and a file size of 1.52MB; (2) Mongolian.Zip is Mongolian word segmentation and annotation data, with a data volume of 65000 sentences and a file size of 3.16MB; (3)Uyghur.Zip is Uighur word segmentation and annotation data, with a data volume of 65000 sentences and a file size of 5.12MB.&amp;nbsp;</dc:description>
  <dc:subject>minority language; word segmentation; evaluation dataset; standard specification for word segmentation</dc:subject>
  <dc:creator>Zhao Xiaobing</dc:creator>
  <dc:creator>Gao Lu</dc:creator>
  <dc:creator>Gao Dingguo</dc:creator>
  <dc:creator>Bao Wugedele</dc:creator>
  <dc:creator>Mieradilijiang Maimaiti</dc:creator>
  <dc:creator>Liu Yang</dc:creator>
  <dc:creator>Cai Zhijie</dc:creator>
  <dc:creator>Sun Yuan</dc:creator>
  <dc:rights>PUBLIC</dc:rights>
  <dc:rights>https://creativecommons.org/licenses/by/4.0/</dc:rights>
  <dc:type>dataset</dc:type>
  <dc:relation>http://www.doi.org/10.11922/11-6035.csd.2021.0091.zh</dc:relation>
  <dc:publisher>Science Data Bank</dc:publisher>
</oai_dc:dc>

    </metadata>
</record>
</GetRecord>
</OAI-PMH>