{"type": "article", "title": "Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve", "publisher": "Web Pulse", "url": "https://wpnews.pro/news/taming-throughput-latency-tradeoff-in-llm-inference-with-sarathi-serve", "original_source": "https://arxiv.org/abs/2403.02310", "published": "2026-09-07T09:00:00+00:00", "accessed": "2026-09-07", "id": "taming-throughput-latency-tradeoff-in-llm-inference-with-sarathi-serve"}