@misc{gumma2026ideagentagenticqualitydiversitysearch,title={IDEAgent: Agentic Quality-Diversity Search for Research Idea Generation},author={Gumma, Varun and Majumder, Navonil and Sinhahajari, Soumitra and Poria, Soujanya},year={2026},archiveprefix={arXiv},primaryclass={cs.AI},url={https://arxiv.org/abs/2607.22375}}
ACL
UPDESH: Synthesizing Grounded Instruction Tuning Data for 13 Indic Languages
Pranjal A Chitale, Varun Gumma, Sanchit Ahuja, Prashant Kodali, Manan Uppadhyay, Deepthi Sudharsan, and Sunayana Sitaram
In Proceedings of the 64th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), Jul 2026
@inproceedings{chitale-etal-2026-updesh,title={{UPDESH}: Synthesizing Grounded Instruction Tuning Data for 13 {I}ndic Languages},author={Chitale, Pranjal A and Gumma, Varun and Ahuja, Sanchit and Kodali, Prashant and Uppadhyay, Manan and Sudharsan, Deepthi and Sitaram, Sunayana},editor={Liakata, Maria and Moreira, Viviane P. and Zhang, Jiajun and Jurgens, David},booktitle={Proceedings of the 64th Annual Meeting of the {A}ssociation for {C}omputational {L}inguistics (Volume 1: Long Papers)},month=jul,year={2026},address={San Diego, California, United States},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2026.acl-long.1763/},doi={10.18653/v1/2026.acl-long.1763},pages={37997--38041},isbn={979-8-89176-390-6}}
ICLR
OffTopicEval: When Large Language Models Enter the Wrong Chat, Almost Always!
Jingdi Lei*, Varun Gumma*, Rishabh Bhardwaj*, Seok Min Lim, Chuan Li, Amir Zadeh, and Soujanya Poria
In The Fourteenth International Conference on Learning Representations, 2026
@inproceedings{lei2026offtopiceval,title={OffTopicEval: When Large Language Models Enter the Wrong Chat, Almost Always!},author={Lei, Jingdi and Gumma, Varun and Bhardwaj, Rishabh and Lim, Seok Min and Li, Chuan and Zadeh, Amir and Poria, Soujanya},booktitle={The Fourteenth International Conference on Learning Representations},year={2026},url={https://openreview.net/forum?id=EcIyiJrajc}}
2025
Preprint
HEALTH-PARIKSHA: Assessing RAG Models for Health Chatbots in Real-World Multilingual Settings
Varun Gumma, Ananditha Raghunath, Mohit Jain†, and Sunayana Sitaram†
@misc{gumma2025healthparikshaassessingragmodels,title={HEALTH-PARIKSHA: Assessing RAG Models for Health Chatbots in Real-World Multilingual Settings},author={Gumma, Varun and Raghunath, Ananditha and Jain, Mohit and Sitaram, Sunayana},year={2025},archiveprefix={arXiv},primaryclass={cs.CL},url={https://arxiv.org/abs/2410.13671}}
NAACL
Towards Inducing Long-Context Abilities in Multilingual Neural Machine Translation Models
Varun Gumma*, Pranjal A Chitale*, and Kalika Bali
In Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), Apr 2025
@inproceedings{gumma-etal-2025-towards,title={Towards Inducing Long-Context Abilities in Multilingual Neural Machine Translation Models},author={Gumma, Varun and Chitale, Pranjal A and Bali, Kalika},editor={Chiruzzo, Luis and Ritter, Alan and Wang, Lu},booktitle={Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers)},month=apr,year={2025},address={Albuquerque, New Mexico},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2025.naacl-long.366/},doi={10.18653/v1/2025.naacl-long.366},pages={7158--7170},isbn={979-8-89176-189-6}}
@inproceedings{ochieng-etal-2025-beyond,title={Beyond Metrics: Evaluating {LLM}s Effectiveness in Culturally Nuanced, Low-Resource Real-World Scenarios},author={Ochieng, Millicent and Gumma, Varun and Sitaram, Sunayana and Wang, Jindong and Chaudhary, Vishrav and Ronen, Keshet and Bali, Kalika and O{'}Neill, Jacki},editor={Lignos, Constantine and Abdulmumin, Idris and Adelani, David},booktitle={Proceedings of the Sixth Workshop on African Natural Language Processing (AfricaNLP 2025)},month=jul,year={2025},address={Vienna, Austria},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2025.africanlp-1.33/},doi={10.18653/v1/2025.africanlp-1.33},pages={230--247},isbn={979-8-89176-257-2}}
2024
EvalEval
Contamination Report for Multilingual Benchmarks
Sanchit Ahuja*, Varun Gumma*, and Sunayana Sitaram
@misc{ahuja2024contaminationreportmultilingualbenchmarks,title={Contamination Report for Multilingual Benchmarks},author={Ahuja, Sanchit and Gumma, Varun and Sitaram, Sunayana},year={2024},archiveprefix={arXiv},primaryclass={cs.CL},url={https://arxiv.org/abs/2410.16186}}
EMNLP
PARIKSHA: A Large-Scale Investigation of Human-LLM Evaluator Agreement on Multilingual and Multi-Cultural Data
@inproceedings{watts-etal-2024-pariksha,title={{PARIKSHA}: A Large-Scale Investigation of Human-{LLM} Evaluator Agreement on Multilingual and Multi-Cultural Data},author={Watts, Ishaan and Gumma, Varun and Yadavalli, Aditya and Seshadri, Vivek and Swaminathan, Manohar and Sitaram, Sunayana},editor={Al-Onaizan, Yaser and Bansal, Mohit and Chen, Yun-Nung},booktitle={Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing},month=nov,year={2024},address={Miami, Florida, USA},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2024.emnlp-main.451/},doi={10.18653/v1/2024.emnlp-main.451},pages={7900--7932}}
FAccT
Akal Badi ya Bias: An Exploratory Study of Gender Bias in Hindi Language Technology
Existing research in measuring and mitigating gender bias predominantly centers on English, overlooking the intricate challenges posed by non-English languages and the Global South. This paper presents the first comprehensive study delving into the nuanced landscape of gender bias in Hindi, the third most spoken language globally. Our study employs diverse mining techniques, computational models, field studies and sheds light on the limitations of current methodologies. Given the challenges faced with mining gender biased statements in Hindi using existing methods, we conducted field studies to bootstrap the collection of such sentences. Through field studies involving rural and low-income community women, we uncover diverse perceptions of gender bias, underscoring the necessity for context-specific approaches. This paper advocates for a community-centric research design, amplifying voices often marginalized in previous studies. Our findings not only contribute to the understanding of gender bias in Hindi but also establish a foundation for further exploration of Indic languages. By exploring the intricacies of this understudied context, we call for thoughtful engagement with gender bias, promoting inclusivity and equity in linguistic and cultural contexts beyond the Global North.
@inproceedings{10.1145/3630106.3659017,author={Hada, Rishav and Husain, Safiya and Gumma, Varun and Diddee, Harshita and Yadavalli, Aditya and Seth, Agrima and Kulkarni, Nidhi and Gadiraju, Ujwal and Vashistha, Aditya and Seshadri, Vivek and Bali, Kalika},title={Akal Badi ya Bias: An Exploratory Study of Gender Bias in Hindi Language Technology},year={2024},isbn={9798400704505},publisher={Association for Computing Machinery},address={New York, NY, USA},url={https://doi.org/10.1145/3630106.3659017},doi={10.1145/3630106.3659017},booktitle={Proceedings of the 2024 ACM Conference on Fairness, Accountability, and Transparency},pages={1926–1939},numpages={14},keywords={Community centric, Gender bias, Global South, Hindi, India, Indic languages},location={Rio de Janeiro, Brazil},series={FAccT '24}}
@inproceedings{hada-etal-2024-metal,title={{METAL}: Towards Multilingual Meta-Evaluation},author={Hada, Rishav and Gumma, Varun and Ahmed, Mohamed and Bali, Kalika and Sitaram, Sunayana},editor={Duh, Kevin and Gomez, Helena and Bethard, Steven},booktitle={Findings of the Association for Computational Linguistics: NAACL 2024},month=jun,year={2024},address={Mexico City, Mexico},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2024.findings-naacl.148/},doi={10.18653/v1/2024.findings-naacl.148},pages={2280--2298}}
NAACL
MEGAVERSE: Benchmarking Large Language Models Across Languages, Modalities, Models and Tasks
In Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), Jun 2024
@inproceedings{ahuja-etal-2024-megaverse,title={{MEGAVERSE}: Benchmarking Large Language Models Across Languages, Modalities, Models and Tasks},author={Ahuja, Sanchit and Aggarwal, Divyanshu and Gumma, Varun and Watts, Ishaan and Sathe, Ashutosh and Ochieng, Millicent and Hada, Rishav and Jain, Prachi and Ahmed, Mohamed and Bali, Kalika and Sitaram, Sunayana},editor={Duh, Kevin and Gomez, Helena and Bethard, Steven},booktitle={Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers)},month=jun,year={2024},address={Mexico City, Mexico},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2024.naacl-long.143/},doi={10.18653/v1/2024.naacl-long.143},pages={2598--2637}}
EACL
Are Large Language Model-based Evaluators the Solution to Scaling Up Multilingual Evaluation?
Rishav Hada, Varun Gumma, Adrian Wynter, Harshita Diddee, Mohamed Ahmed, Monojit Choudhury, Kalika Bali, and Sunayana Sitaram
In Findings of the Association for Computational Linguistics: EACL 2024, Mar 2024
@inproceedings{hada-etal-2024-large,title={Are Large Language Model-based Evaluators the Solution to Scaling Up Multilingual Evaluation?},author={Hada, Rishav and Gumma, Varun and de Wynter, Adrian and Diddee, Harshita and Ahmed, Mohamed and Choudhury, Monojit and Bali, Kalika and Sitaram, Sunayana},editor={Graham, Yvette and Purver, Matthew},booktitle={Findings of the Association for Computational Linguistics: EACL 2024},month=mar,year={2024},address={St. Julian{'}s, Malta},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2024.findings-eacl.71/},doi={10.18653/v1/2024.findings-eacl.71},pages={1051--1070}}
EACL
MAFIA: Multi-Adapter Fused Inclusive Language Models
@inproceedings{jain-etal-2024-mafia,title={{MAFIA}: Multi-Adapter Fused Inclusive Language Models},author={Jain, Prachi and Sathe, Ashutosh and Gumma, Varun and Ahuja, Kabir and Sitaram, Sunayana},editor={Graham, Yvette and Purver, Matthew},booktitle={Proceedings of the 18th Conference of the European Chapter of the Association for Computational Linguistics (Volume 1: Long Papers)},month=mar,year={2024},address={St. Julian{'}s, Malta},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2024.eacl-long.37/},doi={10.18653/v1/2024.eacl-long.37},pages={627--645}}
ComputEL
MunTTS: A Text-to-Speech System for Mundari
Varun Gumma, Rishav Hada, Aditya Yadavalli, Pamir Gogoi, Ishani Mondal, Vivek Seshadri, and Kalika Bali
In Proceedings of the Seventh Workshop on the Use of Computational Methods in the Study of Endangered Languages, Mar 2024
@inproceedings{gumma-etal-2024-muntts,title={{M}un{TTS}: A Text-to-Speech System for {M}undari},author={Gumma, Varun and Hada, Rishav and Yadavalli, Aditya and Gogoi, Pamir and Mondal, Ishani and Seshadri, Vivek and Bali, Kalika},editor={Moeller, Sarah and Agyapong, Godfred and Arppe, Antti and Chaudhary, Aditi and Rijhwani, Shruti and Cox, Christopher and Henke, Ryan and Palmer, Alexis and Rosenblum, Daisy and Schwartz, Lane},booktitle={Proceedings of the Seventh Workshop on the Use of Computational Methods in the Study of Endangered Languages},month=mar,year={2024},address={St. Julians, Malta},publisher={Association for Computational Linguistics},url={https://aclanthology.org/2024.computel-1.11/},doi={10.18653/v1/2024.computel-1.11},pages={76--82}}
2023
TMLR
IndicTrans2: Towards High-Quality and Accessible Machine Translation Models for all 22 Scheduled Indian Languages
Jay Gala*, Pranjal A Chitale*, A K Raghavan, Varun Gumma, Sumanth Doddapaneni, Aswanth Kumar M, Janki Atul Nawale, Anupama Sujatha, Ratish Puduppully, Vivek Raghavan, Pratyush Kumar, Mitesh M Khapra, Raj Dabre, and Anoop Kunchukuttan
@article{gala2023indictrans,title={IndicTrans2: Towards High-Quality and Accessible Machine Translation Models for all 22 Scheduled Indian Languages},author={Gala, Jay and Chitale, Pranjal A and Raghavan, A K and Gumma, Varun and Doddapaneni, Sumanth and M, Aswanth Kumar and Nawale, Janki Atul and Sujatha, Anupama and Puduppully, Ratish and Raghavan, Vivek and Kumar, Pratyush and Khapra, Mitesh M and Dabre, Raj and Kunchukuttan, Anoop},journal={Transactions on Machine Learning Research},issn={2835-8856},year={2023},url={https://openreview.net/forum?id=vfT4YuzAYA},note={}}
EAMT
An Empirical Study of Leveraging Knowledge Distillation for Compressing Multilingual Neural Machine Translation Models
Varun Gumma, Raj Dabre, and Pratyush Kumar
In Proceedings of the 24th Annual Conference of the European Association for Machine Translation, Jun 2023
@inproceedings{gumma-etal-2023-empirical,title={An Empirical Study of Leveraging Knowledge Distillation for Compressing Multilingual Neural Machine Translation Models},author={Gumma, Varun and Dabre, Raj and Kumar, Pratyush},editor={Nurminen, Mary and Brenner, Judith and Koponen, Maarit and Latomaa, Sirkku and Mikhailov, Mikhail and Schierl, Frederike and Ranasinghe, Tharindu and Vanmassenhove, Eva and Vidal, Sergi Alvarez and Aranberri, Nora and Nunziatini, Mara and Escart{\'i}n, Carla Parra and Forcada, Mikel and Popovic, Maja and Scarton, Carolina and Moniz, Helena},booktitle={Proceedings of the 24th Annual Conference of the European Association for Machine Translation},month=jun,year={2023},address={Tampere, Finland},publisher={European Association for Machine Translation},url={https://aclanthology.org/2023.eamt-1.11/},pages={103--114}}
2022
SECRYPT
PAMMELA: Policy Administration Methodology using Machine Learning
@inproceedings{Gumma_2022,title={PAMMELA: Policy Administration Methodology using Machine Learning},url={http://dx.doi.org/10.5220/0011272400003283},doi={10.5220/0011272400003283},booktitle={Proceedings of the 19th International Conference on Security and Cryptography},publisher={SCITEPRESS - Science and Technology Publications},author={Gumma, Varun and Mitra, Barsha and Dey, Soumyadeep and Patel, Pratik and Suman, Sourabh and Das, Saptarshi and Vaidya, Jaideep},year={2022},pages={147–157}}