<oai_dc:dc xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
  <dc:creator>Rossini, Diego</dc:creator>
  <dc:creator>van der Plas, Lonneke</dc:creator>
  <dc:date>2026</dc:date>
  <dc:description xmlns:ns0="xml" ns0:lang="en">We present a scalable, modular pipeline for automatic neologism detection that combines rule-based filtering with LLM classification. The pipeline is grounded in two complementary word-formation frameworks, grammatical and extra-grammatical morphology, which jointly define the scope of what counts as a neologism and inform a four-class classification scheme (NEOLOGISM, ENTITY, FOREIGN, NONE). While designed to be modular and transferable at the architectural level, the pipeline is instantiated on 527 million English-language Reddit posts spanning 2005–2024. From this corpus, we extract 124.6 million unique tokens and reduce them by over 99.99% to yield 1,021 neologism candidates, a set small enough for manual expert verification. Multiple LLMs independently classify each candidate via majority vote, with a final verification step, revealing substantial cross-model disagreement and highlighting the challenge of operationalizing neologism detection at scale. Manual annotation of all 1,021 candidates confirms that 599 (58.7%) are genuine lexical innovations.</dc:description>
  <dc:format>application/pdf</dc:format>
  <dc:identifier>https://susi.usi.ch/global/documents/336329</dc:identifier>
  <dc:identifier>https://localhost:5000/ark:/12658/srd1336329</dc:identifier>
  <dc:identifier>https://susi.usi.ch/documents/336329/files/LREC_2026_NeoLLM_Workshop_NeoPipe_Rossini_vanderPlas.pdf</dc:identifier>
  <dc:language>eng</dc:language>
  <dc:relation>info:eu-repo/semantics/altIdentifier/issn/2522-2686</dc:relation>
  <dc:relation>info:eu-repo/semantics/altIdentifier/doi/10.63317/4o6ks86o293r</dc:relation>
  <dc:relation>info:eu-repo/semantics/altIdentifier/ark/12658/srd1336329</dc:relation>
  <dc:rights>info:eu-repo/semantics/openAccess</dc:rights>
  <dc:rights>CC BY-NC</dc:rights>
  <dc:source>Proceedings of the Workshop Neology and Large Language Models. - 2026, p. 1-15</dc:source>
  <dc:subject xmlns:ns1="xml" ns1:lang="en">Neologism detection</dc:subject>
  <dc:subject xmlns:ns2="xml" ns2:lang="en">Lexical innovation</dc:subject>
  <dc:subject xmlns:ns3="xml" ns3:lang="en">Reddit</dc:subject>
  <dc:subject xmlns:ns4="xml" ns4:lang="en">Large language models</dc:subject>
  <dc:subject xmlns:ns5="xml" ns5:lang="en">Rule-based filtering</dc:subject>
  <dc:subject xmlns:ns6="xml" ns6:lang="en">Computational neology</dc:subject>
  <dc:subject>info:eu-repo/classification/udc/004</dc:subject>
  <dc:title xmlns:ns7="xml" ns7:lang="en">From 124 million tokens to 1,021 neologisms : a large-scale pipeline for automatic neologism detection</dc:title>
  <dc:type>http://purl.org/coar/resource_type/c_5794</dc:type>
</oai_dc:dc>
