<?xml version="1.0" encoding="UTF-8"?>

<?xml-stylesheet type="text/xsl" href="/static/oaitohtml.xsl"?>

<!--
<?xml-stylesheet type="text/xsl" href="/oaitohtml.xsl"?>
-->

<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
    <responseDate>2026-10-12T05:11:32Z</responseDate>
    <request verb="GetRecord" metadataPrefix="oai_dc" identifier="10.57760/sciencedb.j00001.01876" >https://www.scidb.cn/oai</request>
<GetRecord>
    <record>
    <header >
    <identifier>10.57760/sciencedb.j00001.01876</identifier>
    <datestamp>2026-07-31T16:52:47Z</datestamp>
</header>
    <metadata>
        
<oai_dc:dc xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
  <dc:date>2026-07-31</dc:date>
  <dc:title>A Tibetan OCR Dataset for Complex-Layout Documents（Ti-OCR）</dc:title>
  <dc:identifier>doi:10.57760/sciencedb.j00001.01876</dc:identifier>
  <dc:language>en</dc:language>
  <dc:description>&amp;ldquo;A Tibetan OCR Dataset for Complex-Layout Documents&amp;rdquo; is constructed for Tibetan text detection, text recognition, and complex document layout understanding in real publication and paper-document scenarios. The data are mainly collected from books, newspapers, and various paper documents, covering body text, titles, tables of contents, notes, tables, formulas, text embedded in figures, publication information, and multilingual or multimodal mixed-layout elements. The original images were captured using smartphones and digital cameras under natural light or indoor lighting conditions. No additional layout rectification or manual image enhancement was applied during data collection, so as to preserve common characteristics of real document images, including perspective distortion, slight page bending, uneven illumination, page shadows, show-through, edge cropping, table-line interference, and image-text mixed layouts. This dataset is a document image dataset and does not contain continuous temporal observations or geospatial observation variables; therefore, acquisition time and fixed geographic spatial coverage are not used as organizational dimensions of the dataset.The dataset has a total volume of 3.08 GB and consists of 1,080 JPG images and 1,080 corresponding JSON annotation files, with 41,098 annotated text instances in total. According to page structure and layout complexity, the dataset is divided into four layout categories: full-text pages (FTS), table-text mixed pages (TTM), image-text mixed pages (TIM), and multi-element mixed pages (MEM). The numbers of samples in these four categories are 452, 99, 321, and 208, accounting for 41.85%, 9.17%, 29.72%, and 19.26%, respectively. At the instance level, the dataset provides five language or content-type labels, including Tibetan (TI), Chinese (CH), English (EN), digits (DI), and formulas (FM), with 34,818, 1,110, 1,292, 3,609, and 269 instances, respectively. Each image contains 38.05 annotated text instances on average, with the number of instances per image ranging from 1 to 497. The dataset therefore covers multiple levels of page complexity, from simple text pages to dense tables, image-text mixed pages, and multi-element mixed layouts.This dataset can be used for Tibetan complex-layout document text detection, text recognition, end-to-end OCR, table text parsing, image-text mixed page analysis, multilingual document OCR, publication digitization, and low-resource language document intelligence research. The data are organized in common JPG image and JSON annotation formats and can be read, parsed, converted, trained, and evaluated using Python together with tools such as OpenCV, Pillow, json, and PaddleOCR. For text detection tasks, the polygon coordinates in the points field can be directly used as supervision signals. For text recognition tasks, text regions can be cropped according to the annotated coordinates, and the text field can be used as the recognition label. The samples are mainly derived from publicly available publications and various paper document images and do not involve personal sensitive information. The dataset is recommended for academic research, algorithm training, and performance evaluation, subject to the usage rules and citation requirements of the data publishing platform.</dc:description>
  <dc:subject>Tibetan OCR; complex-layout document; dataset; text detection and recognition; low-resource language</dc:subject>
  <dc:creator>jia yang ji</dc:creator>
  <dc:creator>Zhuo ma cuo</dc:creator>
  <dc:creator>Qing cuo</dc:creator>
  <dc:creator>Ze rang cuo</dc:creator>
  <dc:creator>Le mao cai rang</dc:creator>
  <dc:rights>PUBLIC</dc:rights>
  <dc:rights>https://creativecommons.org/licenses/by-nc-sa/4.0/</dc:rights>
  <dc:type>dataset</dc:type>
  <dc:publisher>Science Data Bank</dc:publisher>
</oai_dc:dc>

    </metadata>
</record>
</GetRecord>
</OAI-PMH>