<?xml version="1.0" encoding="UTF-8"?>

<?xml-stylesheet type="text/xsl" href="/static/oaitohtml.xsl"?>

<!--
<?xml-stylesheet type="text/xsl" href="/oaitohtml.xsl"?>
-->

<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
    <responseDate>2026-10-11T19:57:36Z</responseDate>
    <request verb="GetRecord" metadataPrefix="oai_dc" identifier="10.11922/sciencedb.j00001.00355" >https://www.scidb.cn/oai</request>
<GetRecord>
    <record>
    <header >
    <identifier>10.11922/sciencedb.j00001.00355</identifier>
    <datestamp>2022-02-07T13:41:46Z</datestamp>
</header>
    <metadata>
        
<oai_dc:dc xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
  <dc:date>2022-02-07</dc:date>
  <dc:title>A dataset of Tibetan-Chinese cross-language text plagiarism detection</dc:title>
  <dc:identifier>doi:10.11922/sciencedb.j00001.00355</dc:identifier>
  <dc:language>en</dc:language>
  <dc:description>Proceeding from the actual problems of Chinese minority language information processing, this paper establishes a Tibetan-Chinese cross-language text plagiarism detection corpus containing 150,000 sentence pairs, based on SemEval 2014 English evaluation corpus and data enhancement method, to solve the problem of lack of corpus in Tibetan-Chinese cross-language text plagiarism detection. This dataset provides fundamental basis for Tibetan-Chinese cross-language text plagiarism detection. Also the dataset can be used in Tibetan-Chinese semantic computing and other natural language processing tasks. In addition, data enhancement method in the process of data set construction also provides a solution for other low-resource languages to solve the problem of lack of corpus in natural language processing tasks.</dc:description>
  <dc:subject>Text plagiarism detection; Tibetan-Chinese cross-language; cross-language corpus; Low resource</dc:subject>
  <dc:creator>BAO wei</dc:creator>
  <dc:creator>DONG Jian</dc:creator>
  <dc:creator>XU Yang</dc:creator>
  <dc:creator>SHEN Yingli</dc:creator>
  <dc:creator>QI Xiaoke</dc:creator>
  <dc:rights>PUBLIC</dc:rights>
  <dc:rights>https://creativecommons.org/licenses/by/4.0/</dc:rights>
  <dc:type>dataset</dc:type>
  <dc:relation>http://www.doi.org/10.11922/11-6035.csd.2021.0100.zh</dc:relation>
  <dc:publisher>Science Data Bank</dc:publisher>
</oai_dc:dc>

    </metadata>
</record>
</GetRecord>
</OAI-PMH>