<?xml version="1.0" encoding="UTF-8"?>

<?xml-stylesheet type="text/xsl" href="/static/oaitohtml.xsl"?>

<!--
<?xml-stylesheet type="text/xsl" href="/oaitohtml.xsl"?>
-->

<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
    <responseDate>2026-10-12T07:01:23Z</responseDate>
    <request verb="GetRecord" metadataPrefix="oai_dc" identifier="10.57760/sciencedb.29160" >https://www.scidb.cn/oai</request>
<GetRecord>
    <record>
    <header >
    <identifier>10.57760/sciencedb.29160</identifier>
    <datestamp>2025-10-09T16:08:29Z</datestamp>
</header>
    <metadata>
        
<oai_dc:dc xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
  <dc:date>2025-10-09</dc:date>
  <dc:title>CMiLBench: A Hierarchical Multitask Benchmark for Low-Resource Minority Languages in China</dc:title>
  <dc:identifier>doi:10.57760/sciencedb.29160</dc:identifier>
  <dc:language>en</dc:language>
  <dc:description>As large language models (LLMs) continue to advance, they achieve strong performance on high-resource language tasks and show promising potential for low-resource language processing. However, existing benchmarks primarily focus on high-resource languages, with limited coverage of Chinese minority languages. To address this gap, we introduce CMiLBench (Chinese Minority Language Benchmark), a comprehensive evaluation framework targeting three key Chinese minority languages: Tibetan, Mongolian, and Uyghur. CMiLBench comprises 17 task categories and 24,663 samples, covering both language understanding and generation. Several tasks are derived from native corpora and culturally grounded content, offering a realistic assessment of model performance in authentic minority language scenarios. Tasks are stratified into five difficulty levels, and evaluation is conducted using both automatic metrics and LLM-as-a-Judge scoring. We evaluate 14 leading commercial and open-source LLMs, demonstrating that CMiLBench serves as a reliable benchmark and is broadly applicable for evaluating LLMs&amp;rsquo; capabilities in Chinese minority languages, thereby calibrating current technological progress and advancing research and application development of multilingual models for low-resource languages.</dc:description>
  <dc:subject>Chinese minority languages; LLMs; Benchmark</dc:subject>
  <dc:creator>Yijie Li</dc:creator>
  <dc:creator>Yuan Sun</dc:creator>
  <dc:rights>PUBLIC</dc:rights>
  <dc:rights>https://creativecommons.org/licenses/by-nc-sa/4.0/</dc:rights>
  <dc:type>dataset</dc:type>
  <dc:publisher>Science Data Bank</dc:publisher>
</oai_dc:dc>

    </metadata>
</record>
</GetRecord>
</OAI-PMH>