@misc{das2026edgeai,title={Large Models for Small Devices: Recent Advances and Empirical Analysis of Edge {AI} Deployment},author={Das, Subhransu and Cheng, Jiaming and Kumar, Arnav and Afrose, Sadia and Han, Mingzhe and Silagy, Michael and Palande, Shreya and Soni, Brijesh and Ramnath, Rajiv},year={2026},archiveprefix={arXiv},primaryclass={cs.AI},note={Preprint}}
Multi-agent LLM systems relay key-value caches instead of text and credit their gains to exchanged "latent thoughts". That credit is a claim about which example’s cache is relayed, not merely that one is. We audit it causally in released systems, replacing the cache with deranged (mismatched-example), zeroed, and moment-matched random counterparts under two regimes defined by whether the receiver needs the sender’s private information. Where it does, the battery reads ceiling: 100% against 23-25% for answer-irrelevant relays on the primary backbone, replicated across three families, five checkpoints, and a prose document-QA surface. Where it does not, a pre-registered five-seed protocol establishes equivalence within 2.8 points under Holm-corrected TOST. Benchmark deltas do not by themselves establish latent-thought transmission; establishing it takes a mismatched-cache audit, which we release.
@misc{cheng2026latentaudit,title={When Does Latent Communication Pay? A Causal Audit of Relayed {KV} Caches in Multi-Agent {LLMs}},author={Cheng, Jiaming and Das, Subhransu and Ramnath, Rajiv},year={2026},archiveprefix={arXiv},primaryclass={cs.CR},note={Preprint},}
@inproceedings{das2026phasewise,title={Phase-Wise Analysis of {LLM} Inference Acceleration on {GPU}, {CPU}, and Edge Device},author={Das, Subhransu and Cheng, Jiaming and Vallabhajosyula, Swathi and Soni, Brijesh and Ramnath, Rajiv},booktitle={Practice and Experience in Advanced Research Computing (PEARC '26)},year={2026},note={To appear},}
@inproceedings{das2026spice,title={{SPICE}: Structured Pruning for Inference on Constrained Edge Devices},author={Das, Subhransu and Cheng, Jiaming and Rakshit, Aniruddha and Soni, Brijesh and Boubin, Jayson and Ramnath, Rajiv},booktitle={IEEE Consumer Communications \& Networking Conference (CCNC)},year={2026},}
@inproceedings{das2025epic,title={{EPIC}: Efficient Pruning for Inference on Constrained Devices},author={Das, Subhransu and Cheng, Jiaming and Rakshit, Aniruddha and Boubin, Jayson and Ramnath, Rajiv},booktitle={Practice and Experience in Advanced Research Computing (PEARC '25)},year={2025},}