<?xml version="1.0" encoding="UTF-8"?>

<?xml-stylesheet type="text/xsl" href="/static/oaitohtml.xsl"?>

<!--
<?xml-stylesheet type="text/xsl" href="/oaitohtml.xsl"?>
-->

<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
    <responseDate>2026-10-12T01:21:58Z</responseDate>
    <request verb="GetRecord" metadataPrefix="oai_dc" identifier="10.57760/sciencedb.j00133.00626" >https://www.scidb.cn/oai</request>
<GetRecord>
    <record>
    <header >
    <identifier>10.57760/sciencedb.j00133.00626</identifier>
    <datestamp>2026-06-29T08:53:24Z</datestamp>
</header>
    <metadata>
        
<oai_dc:dc xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
  <dc:date>2026-06-29</dc:date>
  <dc:title>Dataset for Measuring Multimodal Research Paper Ideational Similarity</dc:title>
  <dc:identifier>doi:10.57760/sciencedb.j00133.00626</dc:identifier>
  <dc:language>en</dc:language>
  <dc:description>This dataset takes the Chinese journals &amp;quot;Data Analysis and Knowledge Discovery&amp;quot;, &amp;quot;Journal of Information Science&amp;quot;, and &amp;quot;Information Magazine&amp;quot; under the discipline of Information Resources Management as data sources, collecting a total of 4,500 academic papers published from 2018 to 2024. Research idea diagrams and their context text descriptions are extracted from the papers. After manual screening to eliminate unclear and messy research idea diagrams, 1,750 high-quality multimodal samples are obtained. Additionally, due to the large workload of retrieving similar samples from real papers, this paper constructs similar samples by itself for experiments.&amp;nbsp;The similarity in the research ideas of the papers is not only reflected in the replacement of synonyms but also in the reconstruction of the logical process. Therefore, this study constructs a dataset of similar samples based on the above scenarios. Specifically, in the context of synonym replacement, the overall layout and structural framework of the original graph remain unchanged, only the text content in the graph is modified. Some similar samples are constructed by the method of concept substitution, but the general research idea remains unchanged. In the context of process reconstruction, the construction mainly focuses on the reorganization of the process, keeping the text content of the original graph roughly unchanged, while modifying the shape, sequence, orientation, etc. of the structural framework in the graph, but the general research idea remains unchanged.&amp;nbsp;Based on the above-mentioned method for constructing similar samples, this dataset takes the original research idea graphs obtained as the original samples and builds similar samples including operation types such as synonym replacement and process reconstruction, totaling 5,020.</dc:description>
  <dc:subject>Research Approach of the Paper; Similarity Measurement; Multimodal Fusion Enhancement; Feature Alignment</dc:subject>
  <dc:creator>qiu xin peng</dc:creator>
  <dc:rights>PUBLIC</dc:rights>
  <dc:rights>https://creativecommons.org/licenses/by/4.0/</dc:rights>
  <dc:type>dataset</dc:type>
  <dc:publisher>Science Data Bank</dc:publisher>
</oai_dc:dc>

    </metadata>
</record>
</GetRecord>
</OAI-PMH>