We are pleased to inform you about the acceptance of papers at the Workshop on Open-Source Arabic Corpora and Processing Tools (OSACT7) as well as the Workshop on Structured Linguistic Data and Evaluation (SLiDE), co-located with the Language Resources and Evaluation Conference (LREC 2026)
TTLab at AraSentEval: SARF (صرف) Sentiment Analysis via Root-based Fusion for Multi-Dialectal Arabic
May, 2026.
TTLab at AraSentEval: SARF( صرف) Sentiment Analysis via Root-based
Fusion for Multi-Dialectal Arabic. The 7th Workshop on Open-Source Arabic Corpora and Processing
Tools (OSACT7) with 5 Shared Tasks, 262–268.
BibTeX
@inproceedings{Abusaleh:et:al:2026:sarf,
title = {TTLab at AraSentEval: SARF( صرف) Sentiment Analysis via Root-based
Fusion for Multi-Dialectal Arabic},
author = {Abusaleh, Ali and Verma, Bhuvanesh and Mehler, Alexander},
booktitle = {The 7th Workshop on Open-Source Arabic Corpora and Processing
Tools (OSACT7) with 5 Shared Tasks},
month = {May},
year = {2026},
pages = {262--268},
address = {Palma, Mallorca, Spain},
publisher = {European Language Resources Association (ELRA)},
editor = {Al-Khalifa, Hend and El-Haj, Mo and Ezzini, Saad},
doi = {10.63317/4wj6s3ys5osk},
keywords = {NLP, Sentiment Analysis, Arabic analysis, new-data-spaces, circlet, satek},
abstract = {Arabic sentiment analysis is challenged by morphological complexity
and lexical variation across Arabic dialects, compounded by subjectivity
in how speakers and writers express sentiment. In this paper,
we present our submission for the AraSentEval 2026 Shared Task
on Arabic Dialect Sentiment Analysis. We propose SARF (صرف) a
multi-view architectural framework that integrates surface-level
context with stemmed and rooted morphological perspectives using
a shared MARBERTv2 encoder. Our system employs a hybrid BERT-CNN-BiLSTM-Attention
architecture to capture both local sentiment n-grams and global
sequential dependencies. Experimental results show that while
individual morphological normalization strategies (stemming or
rooting) may degrade performance, their joint integration via
cross-morphological attention provides robust features across
diverse dialects. Our final system achieved a competitive macro-F1-score
of 0.9263, ranking 2nd out of 15 participating teams.}
}
Gutenberg+: A More Temporally Faithful Corpus for Diachronic NLP
2026.
Gutenberg+: A More Temporally Faithful Corpus for Diachronic NLP. Proceedings Workshop on Structured Linguistic Data and Evaluation
(SLiDE 2026), co-located with the Language Resources and Evaluation
Conference (LREC 2026), 86–92.
BibTeX
@inproceedings{Hammerla:Mehler:2026:a,
title = {{Gutenberg+}: A More Temporally Faithful Corpus for Diachronic {NLP}},
author = {Leon Hammerla and Alexander Mehler},
booktitle = {Proceedings Workshop on Structured Linguistic Data and Evaluation
(SLiDE 2026), co-located with the Language Resources and Evaluation
Conference (LREC 2026)},
year = {2026},
keywords = {neglab},
pages = {86--92},
address = {Palma, Mallorca, Spain},
publisher = {European Language Resources Association (ELRA)},
editor = {Erhard Hinrichs (Tübingen University, Germany) and Joakim Nivre (Uppsala University, Sweden)
and Petya Osenova (Sofia University, Bulgaria) and James Pustejovsky (Brandeis University, USA)
and Claus Zinn (Tübingen University, Germany)},
doi = {10.63317/2kjofgrkkbt9},
abstract = {We introduce Gutenberg+, a temporally more faithful version of
the Project Gutenberg (PG) corpus, one of the most widely used
resources for diachronic text analysis. Despite its popularity,
the PG corpus contains a major yet overlooked flaw: around 15%
of its entries are collections (e.g., anthologies of books, letters,
or poems) rather than atomic works, which distorts temporal analyses
since such collections may span multiple decades. We present an
automatic method to detect and split these collections into their
constituent works, producing a finer-grained and temporally consistent
corpus. We further re-annotate publication years using LLM-based
retrieval-augmented generative methods, demonstrating the potential
of LLMs to enhance structured linguistic resources. To illustrate
the utility of Gutenberg+, we conduct a small-scale diachronic
case study on negation, showing that our refined corpus captures
more nuanced cross-linguistic variation than the original PG data.
Finally, we release the corpus in UIMA format with full metadata
and linguistic annotations, providing a standardized resource
for future research on diachronic language change.}
}
