Xiurui Pan, Endian Li, Qiao Li, Shengwen Liang, Yizhou Shan, Ke Zhou, Yingwei Luo, Xiaolin Wang, Jie Zhang. InstAttention: In-Storage Attention Offloading for Cost-Effective Long-Context LLM Inference. In IEEE International Symposium on High Performance Computer Architecture, HPCA 2025, Las Vegas, NV, USA, March 1-5, 2025. pages 1510-1525, IEEE, 2025. [doi]
@inproceedings{PanLLLSZLWZ25,
title = {InstAttention: In-Storage Attention Offloading for Cost-Effective Long-Context LLM Inference},
author = {Xiurui Pan and Endian Li and Qiao Li and Shengwen Liang and Yizhou Shan and Ke Zhou and Yingwei Luo and Xiaolin Wang and Jie Zhang},
year = {2025},
doi = {10.1109/HPCA61900.2025.00113},
url = {https://doi.org/10.1109/HPCA61900.2025.00113},
researchr = {https://researchr.org/publication/PanLLLSZLWZ25},
cites = {0},
citedby = {0},
pages = {1510-1525},
booktitle = {IEEE International Symposium on High Performance Computer Architecture, HPCA 2025, Las Vegas, NV, USA, March 1-5, 2025},
publisher = {IEEE},
isbn = {979-8-3315-0647-6},
}