<?xml version="1.0" encoding="UTF-8"?>

<?xml-stylesheet type="text/xsl" href="/static/oaitohtml.xsl"?>

<!--
<?xml-stylesheet type="text/xsl" href="/oaitohtml.xsl"?>
-->

<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
    <responseDate>2026-10-11T19:35:27Z</responseDate>
    <request verb="GetRecord" metadataPrefix="oai_dc" identifier="10.57760/sciencedb.27244" >https://www.scidb.cn/oai</request>
<GetRecord>
    <record>
    <header >
    <identifier>10.57760/sciencedb.27244</identifier>
    <datestamp>2025-12-02T16:50:20Z</datestamp>
</header>
    <metadata>
        
<oai_dc:dc xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
  <dc:date>2025-12-02</dc:date>
  <dc:title>Explainable Detecting Noncompliance in Privacy Agreement based on Large Language Model Fine-tune-dataset</dc:title>
  <dc:identifier>doi:10.57760/sciencedb.27244</dc:identifier>
  <dc:language>en</dc:language>
  <dc:description>The dataset includes three sub datasets of the article, all of which are datasets used for model training in the article &amp;quot;Explanatory Detection of Privacy Protocol Violations Based on Large Language Model Fine tuning&amp;quot;. This includes three parts: a dataset for pre classifying privacy protocol texts, a named entity recognition corpus for identifying key content of privacy protocols, and a corpus dataset for detecting violations using a large language model.&amp;nbsp;The dataset for pre classification of privacy protocol text contains a total of 15627 data items, including classification labels and corresponding text data that have been classified according to privacy protocols.&amp;nbsp;The original annotated corpus for named entity recognition of key content in privacy protocols includes data annotated with BOEM for named entity recognition of privacy protocols for 40 apps.&amp;nbsp;The corpus dataset for violation detection using the big language model consists of three parts. The first part is the common knowledge of regulations, which includes a total of 147 basic contents from the Personal Information Security Specification; The second part is compliance labeling, which includes a training dataset of 1488 data sets obtained by balancing the core content of the privacy agreement with violation labeling corpora; The third part is the semantic interpretation text of the regulations, which includes a dataset of 139 articles obtained by manually interpreting some of the content in the second part of the data.&amp;nbsp;</dc:description>
  <dc:subject>Privacy Agreement; large language model; Compliance</dc:subject>
  <dc:creator>Zhu Hou</dc:creator>
  <dc:creator>Tan Yawen</dc:creator>
  <dc:creator>Wu Zishuai</dc:creator>
  <dc:rights>PUBLIC</dc:rights>
  <dc:rights>https://creativecommons.org/licenses/by-nc-nd/4.0/</dc:rights>
  <dc:type>dataset</dc:type>
  <dc:publisher>Science Data Bank</dc:publisher>
</oai_dc:dc>

    </metadata>
</record>
</GetRecord>
</OAI-PMH>