<?xml version="1.0" encoding="UTF-8"?>

<?xml-stylesheet type="text/xsl" href="/static/oaitohtml.xsl"?>

<!--
<?xml-stylesheet type="text/xsl" href="/oaitohtml.xsl"?>
-->

<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
    <responseDate>2026-10-11T18:27:03Z</responseDate>
    <request verb="GetRecord" metadataPrefix="oai_dc" identifier="10.57760/sciencedb.j00133.00708" >https://www.scidb.cn/oai</request>
<GetRecord>
    <record>
    <header >
    <identifier>10.57760/sciencedb.j00133.00708</identifier>
    <datestamp>2026-09-20T09:58:17Z</datestamp>
</header>
    <metadata>
        
<oai_dc:dc xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
  <dc:date>2026-09-20</dc:date>
  <dc:title>Dataset on MD&amp;amp;A Texts and Financial Ratios for Predicting Financial Distress in Chinese A-share Listed Companies (2013&amp;ndash;2020)</dc:title>
  <dc:identifier>doi:10.57760/sciencedb.j00133.00708</dc:identifier>
  <dc:language>en</dc:language>
  <dc:description>		The dataset comprises two types of data: structured financial data sourced from the Guotai-An CSMAR database, which has been truncated to the 1st percentile and standardised, ultimately retaining 21 core financial indicators across five major categories; MD&amp;amp;A text data was crawled from publicly available annual reports on the Juchao Information Network. Following text cleaning, expansion using the FinBert domain-specific vocabulary, and chi-squared feature selection, it was quantified into four features: text sentiment, innovation/risk information content, and text similarity. All corpus statistics were derived solely from the training set to strictly avoid the disclosure of future information.		The dataset is uniquely identified by &amp;lsquo;stock code &amp;ndash; year&amp;rsquo;; each row corresponds to a single annual observation for a company, whilst each column contains the company identifier, 25 feature indicators and a binary &amp;lsquo;ST/*ST&amp;rsquo; label indicating financial distress.</dc:description>
  <dc:subject>time-series evolution features; MD&amp;A; deep learning; text features; financial distress prediction</dc:subject>
  <dc:creator>jia wei feng</dc:creator>
  <dc:rights>PUBLIC</dc:rights>
  <dc:rights>https://creativecommons.org/licenses/by-nc-sa/4.0/</dc:rights>
  <dc:type>dataset</dc:type>
  <dc:publisher>Science Data Bank</dc:publisher>
</oai_dc:dc>

    </metadata>
</record>
</GetRecord>
</OAI-PMH>