Publications
My work spans LLM serving, reproducibility, trustworthy AI, and evaluation.
Published & accepted
-
One Threshold Does Not Fit All Languages: Language-Conditional Deferral for Reliable and Efficient Low-Resource Text Classification
@misc{vangala2026thresholddoesfitlanguages, title={One Threshold Does Not Fit All Languages: Language-Conditional Deferral for Reliable and Efficient Low-Resource Text Classification}, author={Bhanu Prakash Vangala and Vangala Navya}, year={2026}, eprint={2609.37861}, archivePrefix={arXiv}, primaryClass={cs.CL}, url={https://arxiv.org/abs/2609.37861}, } -
AI-Generated Code Is Not Reproducible (Yet): An Empirical Study of Execution Reliability in LLM-Based Coding Agents
@inproceedings{vangala2026aigeneratedcodeexecution, author = {Vangala, Bhanu Prakash and Gehani, Ashish and Malik, Tanu}, title = {AI-Generated Code Is Not Reproducible (Yet): An Empirical Study of Execution Reliability in LLM-Based Coding Agents}, booktitle = {Proceedings of the 4th ACM Conference on Reproducibility and Replicability}, series = {ACM REP '26}, publisher = {ACM}, year = {2026}, month = jul, pages = {33--46}, doi = {10.1145/3820002.3828581}, url = {https://doi.org/10.1145/3820002.3828581} } -
Pick and Spin: Cold-Start-Aware Routing for Self-Hosted LLM Serving
@inproceedings{vangala2026pickspin, author = {Vangala, Bhanu Prakash and Malik, Tanu}, title = {Pick and Spin: Cold-Start-Aware Routing for Self-Hosted LLM Serving}, booktitle = {2026 IEEE 19th International Conference on Cloud Computing (CLOUD)}, publisher = {IEEE}, year = {2026}, month = jul, pages = {388--394}, doi = {10.1109/cloud72782.2026.00050}, url = {https://doi.org/10.1109/cloud72782.2026.00050} } -
-
AI-Generated Code Is Not Reproducible (Yet): An Empirical Study of Dependency Gaps in LLM-Based Coding Agents
@misc{vangala2026aigeneratedcodereproducibleyet, title={AI-Generated Code Is Not Reproducible (Yet): An Empirical Study of Dependency Gaps in LLM-Based Coding Agents}, author={Bhanu Prakash Vangala and Ali Adibifar and Ashish Gehani and Tanu Malik}, year={2026}, eprint={2512.22387}, archivePrefix={arXiv}, primaryClass={cs.SE}, url={https://arxiv.org/abs/2512.22387}, } -
Efficient Multi-Model Orchestration for Self-Hosted Large Language Models
@misc{vangala2025efficientmultimodelorchestrationselfhosted, title={Efficient Multi-Model Orchestration for Self-Hosted Large Language Models}, author={Bhanu Prakash Vangala and Tanu Malik}, year={2025}, eprint={2512.22402}, archivePrefix={arXiv}, primaryClass={cs.DC}, url={https://arxiv.org/abs/2512.22402}, } -
HalluMat: Detecting Hallucinations in LLM-Generated Materials Science Content Through Multi-Stage Verification
@misc{vangala2025hallumatdetectinghallucinationsllmgenerated, title={HalluMat: Detecting Hallucinations in LLM-Generated Materials Science Content Through Multi-Stage Verification}, author={Bhanu Prakash Vangala and Sajid Mahmud and Pawan Neupane and Joel Selvaraj and Jianlin Cheng}, year={2025}, eprint={2512.22396}, archivePrefix={arXiv}, primaryClass={cs.AI}, url={https://arxiv.org/abs/2512.22396}, } -
-
Submitted & under review
Submitted manuscripts; these have not yet been accepted.
-
Code That Works, Environments That Don’t: Measuring Environment Reproducibility in AI-Generated Software
Next version in progress: a public benchmark that tests how reliable code from AI coding agents really is, covering reproducibility, environment instability and dependency security. View the benchmark
@misc{vangala2026codeworksenvironmentsdont, title={Code That Works, Environments That Don't: Measuring Environment Reproducibility in AI-Generated Software}, author={Bhanu Prakash Vangala and Tanu Malik}, year={2026}, eprint={2610.00425}, archivePrefix={arXiv}, primaryClass={cs.SE}, url={https://arxiv.org/abs/2610.00425}, } -
How Many Labels Does a Language Need? Annotation Budgets and Cross-Lingual Pooling for African-Language Text Classification
@misc{vangala2026labelsdoeslanguageneed, title={How Many Labels Does a Language Need? Annotation Budgets and Cross-Lingual Pooling for African-Language Text Classification}, author={Bhanu Prakash Vangala and Sowmya Guda and Navya Vangala}, year={2026}, eprint={2609.37882}, archivePrefix={arXiv}, primaryClass={cs.CL}, url={https://arxiv.org/abs/2609.37882}, } -
Evaluation Choices Shape Biomedical ML Claims: A Pediatric Pneumonia Benchmark Case Study
@misc{vangala2026evaluationchoicesshapebiomedical, title={Evaluation Choices Shape Biomedical ML Claims: A Pediatric Pneumonia Benchmark Case Study}, author={Bhanu Prakash Vangala and Sowmya Guda and Latha Peddi and Navya Vangala}, year={2026}, eprint={2609.37848}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2609.37848}, } -
AI Sees, XAI Explains? Evaluating Explanation Reliability in Automated Seed Quality Inspection
@misc{vangala2026aiseesxaiexplains, author = {Vangala, Bhanu Prakash and Vangala, Navya}, title = {AI Sees, XAI Explains? Evaluating Explanation Reliability in Automated Seed Quality Inspection}, year = {2026}, howpublished = {SSRN}, note = {Available at SSRN: https://ssrn.com/abstract=7544801}, doi = {10.2139/ssrn.7544801}, url = {https://doi.org/10.2139/ssrn.7544801} } -
-
-
Scores That Hold, Benchmarks That Leak: Measuring Dataset Contamination in Public Brain-Tumor MRI Classification
@misc{vangala2026scoresholdbenchmarksleak, title={Scores That Hold, Benchmarks That Leak: Measuring Dataset Contamination in Public Brain-Tumor MRI Classification}, author={Bhanu Prakash Vangala and Sowmya Guda and Latha Peddi and Navya Vangala}, year={2026}, eprint={2610.00421}, archivePrefix={arXiv}, primaryClass={cs.CV}, url={https://arxiv.org/abs/2610.00421}, }




