Publications

Research articles, conference proceedings, and preprints by Omni Language AI Research.

2026

Model-Based Quality Assessment for Massively Multilingual Parallel Data
Abdelaziz Ibrahim, Zihao Li, Jörg Tiedemann, and Shaoxiong Ji
arXiv preprint arXiv:2606.00285, 2026
@article{ibrahim2026model,
  title   = {Model-Based Quality Assessment for Massively Multilingual Parallel Data},
  author  = {Ibrahim, Abdelaziz and Li, Zihao and Tiedemann, J{\"o}rg and Ji, Shaoxiong},
  journal = {arXiv preprint arXiv:2606.00285},
  year    = {2026}
}
Data-Centric Continual Pre-training for 500+ Languages: A New Bilingual Translation Corpus and Multilingual Models
Shaoxiong Ji, Zihao Li, Jaakko Paavola, Hengyu Luo, and Jörg Tiedemann
Findings of the Association for Computational Linguistics: ACL 2026, pp. 18776–18807, 2026
Paper
@inproceedings{ji-etal-2026-data,
  title     = {Data-Centric Continual Pre-training for 500+ Languages: A New Bilingual Translation Corpus and Multilingual Models},
  author    = {Ji, Shaoxiong  and
               Li, Zihao  and
               Paavola, Jaakko  and
               Luo, Hengyu  and
               Tiedemann, J{\"o}rg},
  editor    = {Liakata, Maria  and
               Moreira, Viviane P.  and
               Zhang, Jiajun  and
               Jurgens, David},
  booktitle = {Findings of the {A}ssociation for {C}omputational {L}inguistics: {ACL} 2026},
  month     = jul,
  year      = {2026},
  address   = {San Diego, California, United States},
  publisher = {Association for Computational Linguistics},
  url       = {https://aclanthology.org/2026.findings-acl.937/},
  doi       = {10.18653/v1/2026.findings-acl.937},
  pages     = {18776--18807},
  isbn      = {979-8-89176-395-1}
}
MaLA: A Corpus and Data Mix for Massive Language Adaptation of Large Language Models
Shaoxiong Ji, Zihao Li, Jaakko Paavola, Peiqin Lin, Pinzhen Chen, Dayyán O'Brien, Hengyu Luo, Hinrich Schütze, Jörg Tiedemann, and Barry Haddow
Proceedings of Conference on Language Modeling (COLM 2026), 2026
Paper
@inproceedings{ji2026mala,
  title     = {MaLA: A Corpus and Data Mix for Massive Language Adaptation of Large Language Models},
  author    = {Ji, Shaoxiong and Li, Zihao and Paavola, Jaakko and Lin, Peiqin and Chen, Pinzhen and O'Brien, Dayy{\'a}n and Luo, Hengyu and Sch{\"u}tze, Hinrich and Tiedemann, J{\"o}rg and Haddow, Barry},
  booktitle = {Proceedings of Conference on Language Modeling (COLM 2026)},
  year      = {2026},
  url       = {https://www.olaresearch.org/MaLA/}
}
XCR-Bench: A Multi-Task Benchmark for Evaluating Cultural Reasoning in LLMs
Mohsinul Kabir, Tasnim Ahmed, Md Mezbaur Rahman, Shaoxiong Ji, Hassan Alhuzali, and Sophia Ananiadou
arXiv preprint arXiv:2601.14063, 2026
@article{kabir2026xcr,
  title   = {XCR-Bench: A Multi-Task Benchmark for Evaluating Cultural Reasoning in LLMs},
  author  = {Kabir, Mohsinul and Ahmed, Tasnim and Rahman, Md Mezbaur and Ji, Shaoxiong and Alhuzali, Hassan and Ananiadou, Sophia},
  journal = {arXiv preprint arXiv:2601.14063},
  year    = {2026}
}
Test-Time Scaling of Reasoning Models for Machine Translation
Zihao Li, Shaoxiong Ji, and Jörg Tiedemann
Proceedings of the 19th Conference of the European Chapter of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 2902–2917, 2026
Paper
@inproceedings{li-etal-2026-test,
  title     = {Test-Time Scaling of Reasoning Models for Machine Translation},
  author    = {Li, Zihao  and
               Ji, Shaoxiong  and
               Tiedemann, J{\"o}rg},
  editor    = {Demberg, Vera  and
               Inui, Kentaro  and
               Marquez, Llu{\'i}s},
  booktitle = {Proceedings of the 19th Conference of the {E}uropean Chapter of the {A}ssociation for {C}omputational {L}inguistics (Volume 1: Long Papers)},
  month     = mar,
  year      = {2026},
  address   = {Rabat, Morocco},
  publisher = {Association for Computational Linguistics},
  url       = {https://aclanthology.org/2026.eacl-long.133/},
  doi       = {10.18653/v1/2026.eacl-long.133},
  pages     = {2902--2917},
  isbn      = {979-8-89176-380-7}
}
Overview of the PsyDefDetect Shared Task at BioNLP 2026: Detecting Levels of Psychological Defense Mechanisms in Supportive Conversations
Hongbin Na, Zimu Wang, Zhaoming Chen, Yining Hua, Rena Gao, Kailai Yang, Ling Chen, Wei Wang, Shaoxiong Ji, John Torous, and Sophia Ananiadou
BioNLP 2026, pp. 932–943, 2026
Paper
@inproceedings{na-etal-2026-overview,
  title     = {Overview of the {P}sy{D}ef{D}etect Shared Task at {B}io{NLP} 2026: Detecting Levels of Psychological Defense Mechanisms in Supportive Conversations},
  author    = {Na, Hongbin  and
               Wang, Zimu  and
               Chen, Zhaoming  and
               Hua, Yining  and
               Gao, Rena  and
               Yang, Kailai  and
               Chen, Ling  and
               Wang, Wei  and
               Ji, Shaoxiong  and
               Torous, John  and
               Ananiadou, Sophia},
  editor    = {Demner-Fushman, Dina  and
               Ananiadou, Sophia  and
               Roberts, Kirk  and
               Tsujii, Junichi},
  booktitle = {{B}io{NLP} 2026},
  month     = jul,
  year      = {2026},
  address   = {San Diego, California},
  publisher = {Association for Computational Linguistics},
  url       = {https://aclanthology.org/2026.bionlp-1.75/},
  doi       = {10.18653/v1/2026.bionlp-1.75},
  pages     = {932--943},
  isbn      = {979-8-89176-434-7}
}
You Never Know a Person, You Only Know Their Defenses: Detecting Levels of Psychological Defense Mechanisms in Supportive Conversations
Hongbin Na, Zimu Wang, Zhaoming Chen, Peilin Zhou, Yining Hua, Grace Ziqi Zhou, Haiyang Zhang, Tao Shen, Wei Wang, John Torous, Shaoxiong Ji, and Ling Chen
Findings of the Association for Computational Linguistics: ACL 2026, pp. 14428–14448, 2026
Paper
@inproceedings{na-etal-2026-never,
  title     = {You Never Know a Person, You Only Know Their Defenses: Detecting Levels of Psychological Defense Mechanisms in Supportive Conversations},
  author    = {Na, Hongbin  and
               Wang, Zimu  and
               Chen, Zhaoming  and
               Zhou, Peilin  and
               Hua, Yining  and
               Zhou, Grace Ziqi  and
               Zhang, Haiyang  and
               Shen, Tao  and
               Wang, Wei  and
               Torous, John  and
               Ji, Shaoxiong  and
               Chen, Ling},
  editor    = {Liakata, Maria  and
               Moreira, Viviane P.  and
               Zhang, Jiajun  and
               Jurgens, David},
  booktitle = {Findings of the {A}ssociation for {C}omputational {L}inguistics: {ACL} 2026},
  month     = jul,
  year      = {2026},
  address   = {San Diego, California, United States},
  publisher = {Association for Computational Linguistics},
  url       = {https://aclanthology.org/2026.findings-acl.708/},
  doi       = {10.18653/v1/2026.findings-acl.708},
  pages     = {14428--14448},
  isbn      = {979-8-89176-395-1}
}
Reasoning over Grammar: Can Synthetic Linguistic Reasoning Traces Enhance Low-Resource Machine Translation?
Renhao Pei, Yihong Liu, Sampo Pyysalo, Hinrich Schütze, and Shaoxiong Ji
arXiv preprint arXiv:2606.03782, 2026
@article{pei2026reasoning,
  title   = {Reasoning over Grammar: Can Synthetic Linguistic Reasoning Traces Enhance Low-Resource Machine Translation?},
  author  = {Pei, Renhao and Liu, Yihong and Pyysalo, Sampo and Sch{\"u}tze, Hinrich and Ji, Shaoxiong},
  journal = {arXiv preprint arXiv:2606.03782},
  year    = {2026}
}
Are LLMs Ready to Assist Physicians? PhysAssistBench for Interactive Doctor-Patient-EHR Assistance
Tianming Du, Peijie Yu, Sihan Shang, Danli Shi, My Linh Nguyen, Shengbo Gao, Guangyuan Li, Yinghong Yu, Yan Jiang, Qianlong Zhao, Behzad Bozorgtabar, Shaoxiong Ji, Jiazhen Pan, Daniel Rueckert, and Jiancheng Yang
arXiv preprint arXiv:2606.18613, 2026
@article{du2026llms,
  title   = {Are LLMs Ready to Assist Physicians? PhysAssistBench for Interactive Doctor-Patient-EHR Assistance},
  author  = {Tianming Du and Peijie Yu and Sihan Shang and Danli Shi and My Linh Nguyen and Shengbo Gao and Guangyuan Li and Yinghong Yu and Yan Jiang and Qianlong Zhao and Behzad Bozorgtabar and Shaoxiong Ji and Jiazhen Pan and Daniel Rueckert and Jiancheng Yang},
  journal = {arXiv preprint arXiv:2606.18613},
  year    = {2026}
}
A Parallel Cross-Lingual Benchmark for Multimodal Idiomaticity Understanding
Dilara Torunoğlu-Selamet, Doğukan Arslan, Rodrigo Wilkens, Wei He, Doruk Eryiğit, Thomas Pickard, Adriana S. Pagano, Aline Villavicencio, Gülşen Eryiğit, Ágnes Abuczki, Aida Cardoso, Alesia Lazarenka, Dina Almassova, Amália Mendes, Anna Kanellopoulou, Antoni Brosa-Rodriguez, Baiba Valkovska, Beata Wojtowicz, Bolette Pedersen, Carlos Manuel Hidalgo-Ternero, Chaya Liebeskind, Danka Jokić, Diego Alves, Eleni Triantafyllidi, Erik Velldal, Fred Philippy, Giedre Valunaite Oleskeviciene, Ieva Rizgeliene, Inguna Skadina, Irina Lobzhanidze, Isabell Stinessen Haugen, Jauza Akbar Krito, Jelena M. Marković, Johanna Monti, Josue Alejandro Sauca, Kaja Dobrovoljc, Kingsley O. Ugwuanyi, Laura Rituma, Lilja Øvrelid, Maha Tufail Agro, Manzura Abjalova, Maria Chatzigrigoriou, María del Mar Sánchez Ramos, Marija Pendevska, Masoumeh Seyyedrezaei, Mehrnoush Shamsfard, Momina Ahsan, Muhammad Ahsan Riaz Khan, Nathalie Carmen Hau Norman, Nilay Erdem Ayyıldız, Nina Hosseini-Kivanani, Noémi Ligeti-Nagy, Numaan Naeem, Olha Kanishcheva, Olha Yatsyshyna, Daniil Orel, Petra Giommarelli, Petya Osenova, Radovan Garabik, Regina E. Semou, Rozane Rebechi, Salsabila Zahirah Pranida, Samia Touileb, Sanni Nimb, Sarfraz Ahmad, Sarvinoz Sharipova, Shahar Golan, Shaoxiong Ji, Sopuruchi Christian Aboh, Srdjan Sucur, Stella Markantonatou, Sussi Olsen, Vahide Tajalli, Veronika Lipp, Voula Giouli, Yelda Yeşildal Eraydın, Zahra Saaberi, and Zhuohan Xie
Proceedings of the Fifteenth Language Resources and Evaluation Conference (LREC 2026), pp. 9434–9448, 2026
@inproceedings{torunoluselamet-etal-2026-parallel,
  title     = {A Parallel Cross-Lingual Benchmark for Multimodal Idiomaticity Understanding},
  author    = {Torunoğlu-Selamet, Dilara and Arslan, Doğukan and Wilkens, Rodrigo and He, Wei and Eryiğit, Doruk and Pickard, Thomas and Pagano, Adriana S. and Villavicencio, Aline and Eryiğit, Gülşen and Abuczki, Ágnes and Cardoso, Aida and Lazarenka, Alesia and Almassova, Dina and Mendes, Amália and Kanellopoulou, Anna and Brosa-Rodriguez, Antoni and Valkovska, Baiba and Wojtowicz, Beata and Pedersen, Bolette and Hidalgo-Ternero, Carlos Manuel and Liebeskind, Chaya and Jokić, Danka and Alves, Diego and Triantafyllidi, Eleni and Velldal, Erik and Philippy, Fred and Oleskeviciene, Giedre Valunaite and Rizgeliene, Ieva and Skadina, Inguna and Lobzhanidze, Irina and Haugen, Isabell Stinessen and Krito, Jauza Akbar and Marković, Jelena M. and Monti, Johanna and Sauca, Josue Alejandro and Dobrovoljc, Kaja and Ugwuanyi, Kingsley O. and Rituma, Laura and Øvrelid, Lilja and Agro, Maha Tufail and Abjalova, Manzura and Chatzigrigoriou, Maria and Ramos, María del Mar Sánchez and Pendevska, Marija and Seyyedrezaei, Masoumeh and Shamsfard, Mehrnoush and Ahsan, Momina and Khan, Muhammad Ahsan Riaz and Norman, Nathalie Carmen Hau and Ayyıldız, Nilay Erdem and Hosseini-Kivanani, Nina and Ligeti-Nagy, Noémi and Naeem, Numaan and Kanishcheva, Olha and Yatsyshyna, Olha and Orel, Daniil and Giommarelli, Petra and Osenova, Petya and Garabik, Radovan and Semou, Regina E. and Rebechi, Rozane and Pranida, Salsabila Zahirah and Touileb, Samia and Nimb, Sanni and Ahmad, Sarfraz and Sharipova, Sarvinoz and Golan, Shahar and Ji, Shaoxiong and Aboh, Sopuruchi Christian and Sucur, Srdjan and Markantonatou, Stella and Olsen, Sussi and Tajalli, Vahide and Lipp, Veronika and Giouli, Voula and Eraydın, Yelda Yeşildal and Saaberi, Zahra and Xie, Zhuohan},
  booktitle = {Proceedings of the Fifteenth Language Resources and Evaluation Conference (LREC 2026)},
  month     = {May},
  year      = {2026},
  pages     = {9434--9448},
  address   = {Palma, Mallorca, Spain},
  publisher = {European Language Resources Association (ELRA)},
  editor    = {Piperidis, Stelios and Bel, Núria and van den Heuvel, Henk and Ide, Nancy and Krek, Simon and Toral, Antonio},
  doi       = {10.63317/5cvnbcoktfo2},
  abstract  = {Potentially idiomatic expressions (PIEs) carry meanings inherently tied to the everyday experience of a given language community. As such, they constitute an interesting challenge for assessing the linguistic (and to some extent cultural) capabilities of NLP systems. In this paper, we present XMPIE, a parallel multilingual and multimodal dataset of potentially idiomatic expressions. The dataset, containing 34 languages and over ten thousand items, allows comparative analyses of idiomatic patterns among language-specific realisations and preferences in order to gather insights about shared cultural aspects. This parallel dataset allows evaluation of language model performance for a given PIE in different languages and whether idiomatic understanding in one language can be transferred to another. Moreover, the dataset supports the study of PIEs across textual and visual modalities, to measure to what extent PIE understanding in one modality transfers or implies in understanding in another modality (text vs. image). The data was created by language experts, with both textual and visual components crafted under multilingual guidelines, and each PIE is accompanied by five images representing a spectrum from idiomatic to literal meanings, including semantically related and random distractors. The result is a high-quality benchmark for evaluating multilingual and multimodal idiomatic language understanding.}
}
Psychologically-Grounded Graph Modeling for Interpretable Depression Detection
Rishitej Reddy Vyalla, Kritarth Prasad, Avinash Anand, Erik Cambria, Shaoxiong Ji, Faten S Alamri, and Zhengkui Wang
arXiv preprint arXiv:2604.24126, 2026
@article{vyalla2026psychologically,
  title   = {Psychologically-Grounded Graph Modeling for Interpretable Depression Detection},
  author  = {Vyalla, Rishitej Reddy and Prasad, Kritarth and Anand, Avinash and Cambria, Erik and Ji, Shaoxiong and Alamri, Faten S and Wang, Zhengkui},
  journal = {arXiv preprint arXiv:2604.24126},
  year    = {2026}
}
Graph2text or Graph2token: A Perspective of Large Language Models for Graph Learning
Shuo Yu, Yingbo Wang, Ruolin Li, Guchun Liu, Yanming Shen, Shaoxiong Ji, Bowen Li, Fengling Han, Xiuzhen Zhang, and Feng Xia
ACM Trans. Inf. Syst., 2026
Paper
@article{yu2025graph2text,
  author     = {Yu, Shuo and Wang, Yingbo and Li, Ruolin and Liu, Guchun and Shen, Yanming and Ji, Shaoxiong and Li, Bowen and Han, Fengling and Zhang, Xiuzhen and Xia, Feng},
  title      = {Graph2text or Graph2token: A Perspective of Large Language Models for Graph Learning},
  year       = {2026},
  issue_date = {March 2026},
  publisher  = {Association for Computing Machinery},
  address    = {New York, NY, USA},
  volume     = {44},
  number     = {3},
  issn       = {1046-8188},
  url        = {https://doi.org/10.1145/3786600},
  doi        = {10.1145/3786600},
  abstract   = {Graphs are prevalent in numerous real-world applications. Previous methods directly model graph structures and achieve significant success. However, these methods encounter bottlenecks due to the inherent irregularity of graphs. An innovative solution is converting graphs into textual representations, thereby harnessing the powerful capabilities of Large Language Models (LLMs) to process and comprehend graphs. In this article, we present a comprehensive review of methodologies for applying LLMs to graphs, termed LLM4graph. The core of LLM4graph lies in transforming graphs into texts for LLMs to understand and analyze. Thus, we propose a novel taxonomy of LLM4graph methods from the view of the transformation. Specifically, existing methods can be divided into two paradigms: Graph2text and Graph2token, which transform graphs into texts or tokens as the input of LLMs, respectively. We point out four challenges during the transformation to systematically present existing methods from a problem-oriented perspective. For practical concerns, we provide a guideline for researchers on selecting appropriate models and LLMs for different graphs and hardware constraints. To empirically evaluate our taxonomy and different technical choices, we conduct experiments with representative methods in Graph2text and Graph2token. We also identify five future research directions for LLM4graph.},
  journal    = {ACM Trans. Inf. Syst.},
  month      = feb,
  articleno  = {57},
  numpages   = {49},
  keywords   = {Graph data, Large language model, Graph to text, Graph to token, Graph learning}
}

2025

Roleplaying with Structure: Synthetic Therapist-Client Conversation Generation from Questionnaires
Doan Nam Long Vu, Rui Tan, Lena Moench, Svenja Jule Francke, Daniel Woiwod, Florian Thomas-Odenthal, Sanna Stroth, Tilo Kircher, Christiane Hermann, Udo Dannlowski, Hamidreza Jamalabadi, and Shaoxiong Ji
arXiv preprint arXiv:2510.25384, 2025
@article{vu2025roleplaying,
  title   = {Roleplaying with Structure: Synthetic Therapist-Client Conversation Generation from Questionnaires},
  author  = {Doan Nam Long Vu and Rui Tan and Lena Moench and Svenja Jule Francke and Daniel Woiwod and Florian Thomas-Odenthal and Sanna Stroth and Tilo Kircher and Christiane Hermann and Udo Dannlowski and Hamidreza Jamalabadi and Shaoxiong Ji},
  journal = {arXiv preprint arXiv:2510.25384},
  year    = {2025}
}
Rethinking Multilingual Continual Pretraining: Data Mixing for Adapting LLMs Across Languages and Resources
Zihao Li, Shaoxiong Ji, Hengyu Luo, and Jörg Tiedemann
Conference on Language Modeling (COLM), 2025
Paper
@inproceedings{li2025rethinking,
  title     = {Rethinking Multilingual Continual Pretraining: Data Mixing for Adapting {LLMs} Across Languages and Resources},
  author    = {Li, Zihao and Ji, Shaoxiong and Luo, Hengyu and Tiedemann, J{\"o}rg},
  booktitle = {Conference on Language Modeling ({COLM})},
  year      = {2025},
  url       = {https://openreview.net/pdf?id=mpTIzK4Zca}
}
GlotEval: A Test Suite for Massively Multilingual Evaluation of Large Language Models
Hengyu Luo, Zihao Li, Joseph Attieh, Sawal Devkota, Ona de Gibert, Xu Huang, Shaoxiong Ji, Peiqin Lin, Bhavani Sai Praneeth Varma Mantina, Ananda Sreenidhi, Raúl Vázquez, Mengjie Wang, Samea Yusofi, Fei Yuan, and Jörg Tiedemann
Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing: System Demonstrations, pp. 602–614, 2025
Paper
@inproceedings{luo2025gloteval,
  title     = {{G}lot{E}val: A Test Suite for Massively Multilingual Evaluation of Large Language Models},
  author    = {Luo, Hengyu and Li, Zihao and Attieh, Joseph and Devkota, Sawal and de Gibert, Ona and Huang, Xu and Ji, Shaoxiong and Lin, Peiqin and Mantina, Bhavani Sai Praneeth Varma and Sreenidhi, Ananda and V{\'a}zquez, Ra{\'u}l and Wang, Mengjie and Yusofi, Samea and Yuan, Fei and Tiedemann, J{\"o}rg},
  booktitle = {Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing: System Demonstrations},
  pages     = {602--614},
  year      = {2025},
  doi       = {10.18653/v1/2025.emnlp-demos.43},
  url       = {https://aclanthology.org/2025.emnlp-demos.43/}
}