<?xml version="1.0" encoding="UTF-8" standalone="yes"?><?xml-stylesheet type="text/xsl" href="/oai-pmh-repository/static/oai2.xsl"?>
<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd">
    <responseDate>2026-10-09T01:59:39Z</responseDate>
    <request verb="GetRecord" identifier="oai:data.mendeley.com/n9xttk45df.1" metadataPrefix="oai_dc">https://data.mendeley.com/oai</request>
    <GetRecord>
        <record>
            <header>
                <identifier>oai:data.mendeley.com/n9xttk45df.1</identifier>
                <datestamp>2026-09-09T09:07:15Z</datestamp>
            </header>
            <metadata><oai_dc:dc xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd" xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
    <dc:creator>saha, Dipto</dc:creator>
    <dc:title>A Synchronized Chittagonian Speech Corpus for Dialect ASR and Speech-to-Speech Translation</dc:title>
    <dc:publisher>Mendeley Data</dc:publisher>
    <dc:description>This dataset presents a synchronized Chittagonian speech corpus developed for low-resource speech processing and dialect translation. The corpus contains 8,004 Chittagonian speech samples paired with corresponding Chittagonian transcripts and Standard Bangla translations. The textual component of the dataset is taken from the publicly available Kothon dataset, a Chittagonian–Standard Bangla parallel text resource. In this work, the original text pairs were extended with newly collected speech recordings from native Chittagonian speakers, creating a synchronized speech-text corpus for automatic speech recognition (ASR), speech-to-speech translation, machine translation, dialect normalization, and text-to-speech (TTS) . The dataset is organized into predefined training, validation, and test splits containing 6,403, 801, and 800 samples, respectively. Each audio recording is provided in WAV format. 
This resource aims to support reproducible research on Chittagonian, an under-resourced regional variety of Bangla, by enabling the development and evaluation of machine learning models for dialect-aware speech and language technologies.</dc:description>
    <dc:subject>Natural Language Processing</dc:subject>
    <dc:subject>Machine Translation</dc:subject>
    <dc:subject>Speech Recognition</dc:subject>
    <dc:subject>Dialect</dc:subject>
    <dc:subject>Bengali Language</dc:subject>
    <dc:contributor id="https://orcid.org/0009-0007-9588-955X">Islam, Md Ariful</dc:contributor>
    <dc:contributor id="https://orcid.org/0000-0002-4535-5978">Islam, Md. Milon</dc:contributor>
    <dc:type>Dataset</dc:type>
    <dc:identifier>doi:10.17632/n9xttk45df.1</dc:identifier>
    <dc:identifier>oai:data.mendeley.com/n9xttk45df.1</dc:identifier>
    <dc:rights>Creative Commons Attribution 4.0 International</dc:rights>
    <dc:rights>http://creativecommons.org/licenses/by/4.0</dc:rights>
    <dc:relation>https://data.mendeley.com/datasets/n9xttk45df</dc:relation>
    <dc:date>2026-09-09T09:07:15Z</dc:date>
</oai_dc:dc></metadata>
        </record>
    </GetRecord>
</OAI-PMH>
