<?xml version="1.0" encoding="US-ASCII"?>
<dblp>
<inproceedings key="conf/fast/QinLHCRZ0ZX25" mdate="2025-03-07">
<author>Ruoyu Qin</author>
<author>Zheming Li</author>
<author>Weiran He</author>
<author>Jialei Cui</author>
<author>Feng Ren</author>
<author>Mingxing Zhang</author>
<author>Yongwei Wu 0001</author>
<author>Weimin Zheng</author>
<author>Xinran Xu</author>
<title>Mooncake: Trading More Storage for Less Computation - A KVCache-centric Architecture for Serving LLM Chatbot.</title>
<pages>155-170</pages>
<year>2025</year>
<booktitle>FAST</booktitle>
<ee type="oa">https://www.usenix.org/conference/fast25/presentation/qin</ee>
<crossref>conf/fast/2025</crossref>
<url>db/conf/fast/fast2025.html#QinLHCRZ0ZX25</url>
</inproceedings></dblp>
