<?xml version="1.0" encoding="UTF-8"?>

<?xml-stylesheet type="text/xsl" href="/static/oaitohtml.xsl"?>

<!--
<?xml-stylesheet type="text/xsl" href="/oaitohtml.xsl"?>
-->

<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
    <responseDate>2026-10-12T06:04:05Z</responseDate>
    <request verb="GetRecord" metadataPrefix="oai_dc" identifier="10.57760/sciencedb.011xb" >https://www.scidb.cn/oai</request>
<GetRecord>
    <record>
    <header >
    <identifier>10.57760/sciencedb.011xb</identifier>
    <datestamp>2026-09-23T19:49:14Z</datestamp>
</header>
    <metadata>
        
<oai_dc:dc xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
  <dc:date>2026-09-23</dc:date>
  <dc:title>Escherichia coli CRISPR-Cas9 sgRNA On-target Activity Dataset (Wild-type, eSpCas9, and recA-Knockout Backgrounds)</dc:title>
  <dc:identifier>doi:10.57760/sciencedb.011xb</dc:identifier>
  <dc:language>en</dc:language>
  <dc:description>This dataset is the supporting data for the paper &amp;quot;Prediction Model for Prokaryotic sgRNA Cleavage Activity Fusing Local Multi-Scale Convolution and Global RNA Semantics&amp;quot; (submitted to《Progress in Biochemistry and Biophysics》, manuscript No.20260295). The data were mainly derived from the construction of a prokaryotic sgRNA cleavage-activity prediction model and related supplementary experiments conducted at the College of Science, Dalian Maritime University, from December 2025 to September 2026. The dataset integrates three prokaryotic benchmark datasets&amp;mdash;wild-type Cas9 and its high-fidelity variants eSpCas9 and knoRecA_Cas9&amp;mdash;sourced from an E. coli genome-scale sgRNA library (Wang et al., Nucleic Acids Research, 2018), comprising more than 137,000 sgRNA sequences of 43 nt together with their cleavage activities measured by high-throughput sequencing, and balanced binary labels (high-/low-efficiency) generated using the median editing efficiency of each dataset as the threshold. It also records the 640-dimensional global semantic features extracted by the pretrained RNA foundation model RNA-FM, the source code and trained weights of the dual-channel fusion model (local multi-scale convolution + global RNA semantics), and the results of supplementary experiments including cross-species few-shot fine-tuning, label-noise robustness, and dual-channel fusion-weight attribution. This dataset lays the foundation for the precise preliminary screening of high-activity sgRNAs and the development of new methods for sgRNA cleavage-activity prediction in prokaryotic CRISPR-Cas9 genome editing.</dc:description>
  <dc:subject>sgRNA; CRISPR-Cas9; cleavage activity prediction</dc:subject>
  <dc:creator>yao yue</dc:creator>
  <dc:rights>PUBLIC</dc:rights>
  <dc:rights>https://creativecommons.org/licenses/by/4.0/</dc:rights>
  <dc:type>dataset</dc:type>
  <dc:publisher>Science Data Bank</dc:publisher>
</oai_dc:dc>

    </metadata>
</record>
</GetRecord>
</OAI-PMH>