<?xml version="1.0" encoding="UTF-8"?>
<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/ http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd">
  <responseDate>2026-07-28T23:36:30Z</responseDate>
  <request identifier="26701" metadataPrefix="oai_dc" verb="GetRecord">https://drops.dagstuhl.de/oai</request>
  <GetRecord>
    <record>
      <header>
        <identifier>oai:drops-oai.dagstuhl.de:26701</identifier>
        <datestamp>2026-07-28T09:32:41Z</datestamp>
        <setSpec>ddc:004</setSpec>
        <setSpec>open_access</setSpec>
      </header>
      <metadata>
        <oai_dc:dc xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
          <dc:title>Cross-Language Text Readability Assessment: Leveraging Multilingual Models for Improved Performance in CEFR-Level Classification</dc:title>
          <dc:creator>Ribeiro, Eugénio</dc:creator>
          <dc:creator>Baptista, Jorge</dc:creator>
          <dc:subject>Readability</dc:subject>
          <dc:subject>Text Complexity</dc:subject>
          <dc:subject>CEFR</dc:subject>
          <dc:subject>Multilinguality</dc:subject>
          <dc:description>The automatic assessment of text readability and the classification of texts by levels is essential for language education and language industries that rely on effective communication. This study explores cross-language automatic readability level classification using the levels defined by the Common European Framework of Reference for Languages (CEFR). We investigate the potential of using data in one language to improve classification performance in different languages and, thus, optimize the utilization of the limited labeled resources available for each language. We rely on a pre-trained multilingual Transformer-based language model, by fine-tuning it on annotated data in one language or in a combination of languages, and then assessing its ability to generalize even to unseen languages. In an additional scenario, we further fine-tune the models on data in the target language, to assess whether the models trained on data in different languages can capture generic information regarding text readability and then be further specialized to capture the specific characteristics of the target language. Our experiments covering the English, Dutch, and German languages revealed that direct generalization to unseen languages is challenging. However, when paired with data in the target language, multilingual data can be leveraged to capture cross-language aspects of text readability, leading to more robust and better-performing models.</dc:description>
          <dc:publisher>Schloss Dagstuhl – Leibniz-Zentrum für Informatik</dc:publisher>
          <dc:contributor>Eugénio Ribeiro and Jorge Baptista</dc:contributor>
          <dc:date>2026</dc:date>
          <dc:relation>Is Part Of OASIcs, Volume 144, 15th Symposium on Languages, Applications and Technologies (SLATE 2026)</dc:relation>
          <dc:type>InProceedings</dc:type>
          <dc:type>Text</dc:type>
          <dc:type>doc-type:ResearchArticle</dc:type>
          <dc:type>publishedVersion</dc:type>
          <dc:format>application/pdf</dc:format>
          <dc:identifier>doi:10.4230/OASIcs.SLATE.2026.3</dc:identifier>
          <dc:identifier>urn:nbn:de:0030-drops-267016</dc:identifier>
          <dc:identifier>https://drops.dagstuhl.de/entities/document/10.4230/OASIcs.SLATE.2026.3</dc:identifier>
          <dc:language>eng</dc:language>
          <dc:rights>https://creativecommons.org/licenses/by/4.0/legalcode</dc:rights>
        </oai_dc:dc>
      </metadata>
    </record>
  </GetRecord>
</OAI-PMH>
