<oai_dc:dc xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
  <dc:creator>Rossini, Diego</dc:creator>
  <dc:creator>van der Plas, Marie Louise Elizabeth</dc:creator>
  <dc:date>2026</dc:date>
  <dc:description xmlns:ns0="xml" ns0:lang="en">We present a comprehensive approach for multiword expression (MWE) identification that combines binary token-level classification, linguistic feature integration, and data augmentation. Our DeBERTa-v3-large model achieves 69.8% F1 on the CoAM dataset, surpassing the best results (Qwen-72B, 57.8% F1) on this dataset by 12 points while using 165x fewer parameters. We achieve this performance by (1) reformulating detection as binary token-level START/END/INSIDE classification rather than span-based prediction, (2) incorporating NP chunking and dependency features that help discontinuous and NOUN-type MWEs identification, and (3) applying oversampling that addresses severe class imbalance in the training data. We confirm the generalization of our method on the STREUSLE dataset, achieving 78.9% F1. These results demonstrate that carefully designed smaller models can substantially outperform LLMs on structured NLP tasks, with important implications for resource-constrained deployments.</dc:description>
  <dc:format>application/pdf</dc:format>
  <dc:identifier>https://localhost:5000/ark:/12658/srd1335030</dc:identifier>
  <dc:identifier>https://susi.usi.ch/global/documents/335030</dc:identifier>
  <dc:identifier>https://susi.usi.ch/documents/335030/files/Rossini_vanderPlas_539.pdf</dc:identifier>
  <dc:language>eng</dc:language>
  <dc:relation>https://aclanthology.org/2026.findings-eacl.135.pdf</dc:relation>
  <dc:relation>info:eu-repo/semantics/altIdentifier/doi/10.18653/v1/2026.findings-eacl.135</dc:relation>
  <dc:relation>info:eu-repo/semantics/altIdentifier/ark/12658/srd1335030</dc:relation>
  <dc:rights>info:eu-repo/semantics/openAccess</dc:rights>
  <dc:rights>CC BY</dc:rights>
  <dc:source>Findings of the Association for Computational Linguistics (EACL 2026). - 2026</dc:source>
  <dc:subject xmlns:ns1="xml" ns1:lang="en">MWE</dc:subject>
  <dc:subject xmlns:ns2="xml" ns2:lang="en">NLP</dc:subject>
  <dc:subject xmlns:ns3="xml" ns3:lang="en">Token Classification</dc:subject>
  <dc:subject>info:eu-repo/classification/udc/81</dc:subject>
  <dc:title xmlns:ns4="xml" ns4:lang="en">Binary token-level classification with DeBERTa for all-type MWE identification : a lightweight approach with linguistic enhancement</dc:title>
  <dc:type>http://purl.org/coar/resource_type/c_5794</dc:type>
</oai_dc:dc>
