2026
Model-Based Quality Assessment for Massively Multilingual Parallel Data
arXiv preprint arXiv:2606.00285, 2026
@article{ibrahim2026model,
title = {Model-Based Quality Assessment for Massively Multilingual Parallel Data},
author = {Ibrahim, Abdelaziz and Li, Zihao and Tiedemann, J{\"o}rg and Ji, Shaoxiong},
journal = {arXiv preprint arXiv:2606.00285},
year = {2026}
}
Data-Centric Continual Pre-training for 500+ Languages: A New Bilingual Translation Corpus and Multilingual Models
Findings of the Association for Computational Linguistics: ACL 2026, pp. 18776–18807, 2026
@inproceedings{ji-etal-2026-data,
title = {Data-Centric Continual Pre-training for 500+ Languages: A New Bilingual Translation Corpus and Multilingual Models},
author = {Ji, Shaoxiong and
Li, Zihao and
Paavola, Jaakko and
Luo, Hengyu and
Tiedemann, J{\"o}rg},
editor = {Liakata, Maria and
Moreira, Viviane P. and
Zhang, Jiajun and
Jurgens, David},
booktitle = {Findings of the {A}ssociation for {C}omputational {L}inguistics: {ACL} 2026},
month = jul,
year = {2026},
address = {San Diego, California, United States},
publisher = {Association for Computational Linguistics},
url = {https://aclanthology.org/2026.findings-acl.937/},
doi = {10.18653/v1/2026.findings-acl.937},
pages = {18776--18807},
isbn = {979-8-89176-395-1}
}
MaLA: A Corpus and Data Mix for Massive Language Adaptation of Large Language Models
Proceedings of Conference on Language Modeling (COLM 2026), 2026
@inproceedings{ji2026mala,
title = {MaLA: A Corpus and Data Mix for Massive Language Adaptation of Large Language Models},
author = {Ji, Shaoxiong and Li, Zihao and Paavola, Jaakko and Lin, Peiqin and Chen, Pinzhen and O'Brien, Dayy{\'a}n and Luo, Hengyu and Sch{\"u}tze, Hinrich and Tiedemann, J{\"o}rg and Haddow, Barry},
booktitle = {Proceedings of Conference on Language Modeling (COLM 2026)},
year = {2026},
url = {https://www.olaresearch.org/MaLA/}
}
XCR-Bench: A Multi-Task Benchmark for Evaluating Cultural Reasoning in LLMs
arXiv preprint arXiv:2601.14063, 2026
@article{kabir2026xcr,
title = {XCR-Bench: A Multi-Task Benchmark for Evaluating Cultural Reasoning in LLMs},
author = {Kabir, Mohsinul and Ahmed, Tasnim and Rahman, Md Mezbaur and Ji, Shaoxiong and Alhuzali, Hassan and Ananiadou, Sophia},
journal = {arXiv preprint arXiv:2601.14063},
year = {2026}
}
Test-Time Scaling of Reasoning Models for Machine Translation
Proceedings of the 19th Conference of the European Chapter of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 2902–2917, 2026
@inproceedings{li-etal-2026-test,
title = {Test-Time Scaling of Reasoning Models for Machine Translation},
author = {Li, Zihao and
Ji, Shaoxiong and
Tiedemann, J{\"o}rg},
editor = {Demberg, Vera and
Inui, Kentaro and
Marquez, Llu{\'i}s},
booktitle = {Proceedings of the 19th Conference of the {E}uropean Chapter of the {A}ssociation for {C}omputational {L}inguistics (Volume 1: Long Papers)},
month = mar,
year = {2026},
address = {Rabat, Morocco},
publisher = {Association for Computational Linguistics},
url = {https://aclanthology.org/2026.eacl-long.133/},
doi = {10.18653/v1/2026.eacl-long.133},
pages = {2902--2917},
isbn = {979-8-89176-380-7}
}
Overview of the PsyDefDetect Shared Task at BioNLP 2026: Detecting Levels of Psychological Defense Mechanisms in Supportive Conversations
BioNLP 2026, pp. 932–943, 2026
@inproceedings{na-etal-2026-overview,
title = {Overview of the {P}sy{D}ef{D}etect Shared Task at {B}io{NLP} 2026: Detecting Levels of Psychological Defense Mechanisms in Supportive Conversations},
author = {Na, Hongbin and
Wang, Zimu and
Chen, Zhaoming and
Hua, Yining and
Gao, Rena and
Yang, Kailai and
Chen, Ling and
Wang, Wei and
Ji, Shaoxiong and
Torous, John and
Ananiadou, Sophia},
editor = {Demner-Fushman, Dina and
Ananiadou, Sophia and
Roberts, Kirk and
Tsujii, Junichi},
booktitle = {{B}io{NLP} 2026},
month = jul,
year = {2026},
address = {San Diego, California},
publisher = {Association for Computational Linguistics},
url = {https://aclanthology.org/2026.bionlp-1.75/},
doi = {10.18653/v1/2026.bionlp-1.75},
pages = {932--943},
isbn = {979-8-89176-434-7}
}
You Never Know a Person, You Only Know Their Defenses: Detecting Levels of Psychological Defense Mechanisms in Supportive Conversations
Findings of the Association for Computational Linguistics: ACL 2026, pp. 14428–14448, 2026
@inproceedings{na-etal-2026-never,
title = {You Never Know a Person, You Only Know Their Defenses: Detecting Levels of Psychological Defense Mechanisms in Supportive Conversations},
author = {Na, Hongbin and
Wang, Zimu and
Chen, Zhaoming and
Zhou, Peilin and
Hua, Yining and
Zhou, Grace Ziqi and
Zhang, Haiyang and
Shen, Tao and
Wang, Wei and
Torous, John and
Ji, Shaoxiong and
Chen, Ling},
editor = {Liakata, Maria and
Moreira, Viviane P. and
Zhang, Jiajun and
Jurgens, David},
booktitle = {Findings of the {A}ssociation for {C}omputational {L}inguistics: {ACL} 2026},
month = jul,
year = {2026},
address = {San Diego, California, United States},
publisher = {Association for Computational Linguistics},
url = {https://aclanthology.org/2026.findings-acl.708/},
doi = {10.18653/v1/2026.findings-acl.708},
pages = {14428--14448},
isbn = {979-8-89176-395-1}
}
Reasoning over Grammar: Can Synthetic Linguistic Reasoning Traces Enhance Low-Resource Machine Translation?
arXiv preprint arXiv:2606.03782, 2026
@article{pei2026reasoning,
title = {Reasoning over Grammar: Can Synthetic Linguistic Reasoning Traces Enhance Low-Resource Machine Translation?},
author = {Pei, Renhao and Liu, Yihong and Pyysalo, Sampo and Sch{\"u}tze, Hinrich and Ji, Shaoxiong},
journal = {arXiv preprint arXiv:2606.03782},
year = {2026}
}
Are LLMs Ready to Assist Physicians? PhysAssistBench for Interactive Doctor-Patient-EHR Assistance
arXiv preprint arXiv:2606.18613, 2026
@article{du2026llms,
title = {Are LLMs Ready to Assist Physicians? PhysAssistBench for Interactive Doctor-Patient-EHR Assistance},
author = {Tianming Du and Peijie Yu and Sihan Shang and Danli Shi and My Linh Nguyen and Shengbo Gao and Guangyuan Li and Yinghong Yu and Yan Jiang and Qianlong Zhao and Behzad Bozorgtabar and Shaoxiong Ji and Jiazhen Pan and Daniel Rueckert and Jiancheng Yang},
journal = {arXiv preprint arXiv:2606.18613},
year = {2026}
}
A Parallel Cross-Lingual Benchmark for Multimodal Idiomaticity Understanding
Proceedings of the Fifteenth Language Resources and Evaluation Conference (LREC 2026), pp. 9434–9448, 2026
@inproceedings{torunoluselamet-etal-2026-parallel,
title = {A Parallel Cross-Lingual Benchmark for Multimodal Idiomaticity Understanding},
author = {Torunoğlu-Selamet, Dilara and Arslan, Doğukan and Wilkens, Rodrigo and He, Wei and Eryiğit, Doruk and Pickard, Thomas and Pagano, Adriana S. and Villavicencio, Aline and Eryiğit, Gülşen and Abuczki, Ágnes and Cardoso, Aida and Lazarenka, Alesia and Almassova, Dina and Mendes, Amália and Kanellopoulou, Anna and Brosa-Rodriguez, Antoni and Valkovska, Baiba and Wojtowicz, Beata and Pedersen, Bolette and Hidalgo-Ternero, Carlos Manuel and Liebeskind, Chaya and Jokić, Danka and Alves, Diego and Triantafyllidi, Eleni and Velldal, Erik and Philippy, Fred and Oleskeviciene, Giedre Valunaite and Rizgeliene, Ieva and Skadina, Inguna and Lobzhanidze, Irina and Haugen, Isabell Stinessen and Krito, Jauza Akbar and Marković, Jelena M. and Monti, Johanna and Sauca, Josue Alejandro and Dobrovoljc, Kaja and Ugwuanyi, Kingsley O. and Rituma, Laura and Øvrelid, Lilja and Agro, Maha Tufail and Abjalova, Manzura and Chatzigrigoriou, Maria and Ramos, María del Mar Sánchez and Pendevska, Marija and Seyyedrezaei, Masoumeh and Shamsfard, Mehrnoush and Ahsan, Momina and Khan, Muhammad Ahsan Riaz and Norman, Nathalie Carmen Hau and Ayyıldız, Nilay Erdem and Hosseini-Kivanani, Nina and Ligeti-Nagy, Noémi and Naeem, Numaan and Kanishcheva, Olha and Yatsyshyna, Olha and Orel, Daniil and Giommarelli, Petra and Osenova, Petya and Garabik, Radovan and Semou, Regina E. and Rebechi, Rozane and Pranida, Salsabila Zahirah and Touileb, Samia and Nimb, Sanni and Ahmad, Sarfraz and Sharipova, Sarvinoz and Golan, Shahar and Ji, Shaoxiong and Aboh, Sopuruchi Christian and Sucur, Srdjan and Markantonatou, Stella and Olsen, Sussi and Tajalli, Vahide and Lipp, Veronika and Giouli, Voula and Eraydın, Yelda Yeşildal and Saaberi, Zahra and Xie, Zhuohan},
booktitle = {Proceedings of the Fifteenth Language Resources and Evaluation Conference (LREC 2026)},
month = {May},
year = {2026},
pages = {9434--9448},
address = {Palma, Mallorca, Spain},
publisher = {European Language Resources Association (ELRA)},
editor = {Piperidis, Stelios and Bel, Núria and van den Heuvel, Henk and Ide, Nancy and Krek, Simon and Toral, Antonio},
doi = {10.63317/5cvnbcoktfo2},
abstract = {Potentially idiomatic expressions (PIEs) carry meanings inherently tied to the everyday experience of a given language community. As such, they constitute an interesting challenge for assessing the linguistic (and to some extent cultural) capabilities of NLP systems. In this paper, we present XMPIE, a parallel multilingual and multimodal dataset of potentially idiomatic expressions. The dataset, containing 34 languages and over ten thousand items, allows comparative analyses of idiomatic patterns among language-specific realisations and preferences in order to gather insights about shared cultural aspects. This parallel dataset allows evaluation of language model performance for a given PIE in different languages and whether idiomatic understanding in one language can be transferred to another. Moreover, the dataset supports the study of PIEs across textual and visual modalities, to measure to what extent PIE understanding in one modality transfers or implies in understanding in another modality (text vs. image). The data was created by language experts, with both textual and visual components crafted under multilingual guidelines, and each PIE is accompanied by five images representing a spectrum from idiomatic to literal meanings, including semantically related and random distractors. The result is a high-quality benchmark for evaluating multilingual and multimodal idiomatic language understanding.}
}
Psychologically-Grounded Graph Modeling for Interpretable Depression Detection
arXiv preprint arXiv:2604.24126, 2026
@article{vyalla2026psychologically,
title = {Psychologically-Grounded Graph Modeling for Interpretable Depression Detection},
author = {Vyalla, Rishitej Reddy and Prasad, Kritarth and Anand, Avinash and Cambria, Erik and Ji, Shaoxiong and Alamri, Faten S and Wang, Zhengkui},
journal = {arXiv preprint arXiv:2604.24126},
year = {2026}
}
Graph2text or Graph2token: A Perspective of Large Language Models for Graph Learning
ACM Trans. Inf. Syst., 2026
@article{yu2025graph2text,
author = {Yu, Shuo and Wang, Yingbo and Li, Ruolin and Liu, Guchun and Shen, Yanming and Ji, Shaoxiong and Li, Bowen and Han, Fengling and Zhang, Xiuzhen and Xia, Feng},
title = {Graph2text or Graph2token: A Perspective of Large Language Models for Graph Learning},
year = {2026},
issue_date = {March 2026},
publisher = {Association for Computing Machinery},
address = {New York, NY, USA},
volume = {44},
number = {3},
issn = {1046-8188},
url = {https://doi.org/10.1145/3786600},
doi = {10.1145/3786600},
abstract = {Graphs are prevalent in numerous real-world applications. Previous methods directly model graph structures and achieve significant success. However, these methods encounter bottlenecks due to the inherent irregularity of graphs. An innovative solution is converting graphs into textual representations, thereby harnessing the powerful capabilities of Large Language Models (LLMs) to process and comprehend graphs. In this article, we present a comprehensive review of methodologies for applying LLMs to graphs, termed LLM4graph. The core of LLM4graph lies in transforming graphs into texts for LLMs to understand and analyze. Thus, we propose a novel taxonomy of LLM4graph methods from the view of the transformation. Specifically, existing methods can be divided into two paradigms: Graph2text and Graph2token, which transform graphs into texts or tokens as the input of LLMs, respectively. We point out four challenges during the transformation to systematically present existing methods from a problem-oriented perspective. For practical concerns, we provide a guideline for researchers on selecting appropriate models and LLMs for different graphs and hardware constraints. To empirically evaluate our taxonomy and different technical choices, we conduct experiments with representative methods in Graph2text and Graph2token. We also identify five future research directions for LLM4graph.},
journal = {ACM Trans. Inf. Syst.},
month = feb,
articleno = {57},
numpages = {49},
keywords = {Graph data, Large language model, Graph to text, Graph to token, Graph learning}
}
2025
Roleplaying with Structure: Synthetic Therapist-Client Conversation Generation from Questionnaires
arXiv preprint arXiv:2510.25384, 2025
@article{vu2025roleplaying,
title = {Roleplaying with Structure: Synthetic Therapist-Client Conversation Generation from Questionnaires},
author = {Doan Nam Long Vu and Rui Tan and Lena Moench and Svenja Jule Francke and Daniel Woiwod and Florian Thomas-Odenthal and Sanna Stroth and Tilo Kircher and Christiane Hermann and Udo Dannlowski and Hamidreza Jamalabadi and Shaoxiong Ji},
journal = {arXiv preprint arXiv:2510.25384},
year = {2025}
}
Rethinking Multilingual Continual Pretraining: Data Mixing for Adapting LLMs Across Languages and Resources
Conference on Language Modeling (COLM), 2025
@inproceedings{li2025rethinking,
title = {Rethinking Multilingual Continual Pretraining: Data Mixing for Adapting {LLMs} Across Languages and Resources},
author = {Li, Zihao and Ji, Shaoxiong and Luo, Hengyu and Tiedemann, J{\"o}rg},
booktitle = {Conference on Language Modeling ({COLM})},
year = {2025},
url = {https://openreview.net/pdf?id=mpTIzK4Zca}
}
GlotEval: A Test Suite for Massively Multilingual Evaluation of Large Language Models
Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing: System Demonstrations, pp. 602–614, 2025
@inproceedings{luo2025gloteval,
title = {{G}lot{E}val: A Test Suite for Massively Multilingual Evaluation of Large Language Models},
author = {Luo, Hengyu and Li, Zihao and Attieh, Joseph and Devkota, Sawal and de Gibert, Ona and Huang, Xu and Ji, Shaoxiong and Lin, Peiqin and Mantina, Bhavani Sai Praneeth Varma and Sreenidhi, Ananda and V{\'a}zquez, Ra{\'u}l and Wang, Mengjie and Yusofi, Samea and Yuan, Fei and Tiedemann, J{\"o}rg},
booktitle = {Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing: System Demonstrations},
pages = {602--614},
year = {2025},
doi = {10.18653/v1/2025.emnlp-demos.43},
url = {https://aclanthology.org/2025.emnlp-demos.43/}
}