<?xml version="1.0" encoding="UTF-8"?>

<?xml-stylesheet type="text/xsl" href="/static/oaitohtml.xsl"?>

<!--
<?xml-stylesheet type="text/xsl" href="/oaitohtml.xsl"?>
-->

<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
    <responseDate>2026-10-12T04:43:19Z</responseDate>
    <request verb="GetRecord" metadataPrefix="oai_dc" identifier="10.57760/sciencedb.43207" >https://www.scidb.cn/oai</request>
<GetRecord>
    <record>
    <header >
    <identifier>10.57760/sciencedb.43207</identifier>
    <datestamp>2026-08-31T10:55:25Z</datestamp>
</header>
    <metadata>
        
<oai_dc:dc xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
  <dc:date>2026-08-31</dc:date>
  <dc:title>Vulnerability Detection Datasets for Large Model Fine-tuning</dc:title>
  <dc:identifier>doi:10.57760/sciencedb.43207</dc:identifier>
  <dc:language>en</dc:language>
  <dc:description>This dataset is a specialized vulnerability detection corpus built specifically for fine-tuning large language models. It integrates multiple well-known benchmark datasets for code vulnerability detection, aiming to train models to identify security vulnerabilities in code through instruction fine-tuning. The dataset converts raw code snippets into a standard format of &amp;quot;instruction input output&amp;quot;, suitable for supervised fine-tuning tasks, helping models learn code auditing and secure coding standards.&amp;nbsp;The dataset consists of multiple subsets, each corresponding to a specific vulnerability detection benchmark or source, covering different types of vulnerabilities and code scenarios:&amp;nbsp;BigVul: Contains real C/C++vulnerability code and its patched versions.&amp;nbsp;CVEpixels: Based on the CVE database, it includes specific repair code for publicly available vulnerabilities.&amp;nbsp;Devign: Focused on vulnerability detection in large-scale real-world projects.&amp;nbsp;DiverseVul: emphasizes the diversity of vulnerability types.&amp;nbsp;Juliet: The testing suite provided by NIST contains a large number of manually constructed vulnerability samples.&amp;nbsp;MixVul: A publicly available dataset for software vulnerability detection, widely used in deep learning based vulnerability detection tasks&amp;nbsp;ReVeal: A dataset built using Chromium and Debian repair commits&amp;nbsp;The data files are stored in JSON List format. The sample structure follows the standard instruction fine-tuning format and includes the following four core fields: instruction, input, output, and idx&amp;nbsp;</dc:description>
  <dc:subject>漏洞数据集; 大模型; 微调</dc:subject>
  <dc:creator>Cao Mingsheng</dc:creator>
  <dc:creator>Ding Qiaolong</dc:creator>
  <dc:creator>Xu Hui</dc:creator>
  <dc:rights>RESTRICTED</dc:rights>
  <dc:type>dataset</dc:type>
  <dc:publisher>Science Data Bank</dc:publisher>
</oai_dc:dc>

    </metadata>
</record>
</GetRecord>
</OAI-PMH>