{"articles":[{"slug":"hitting-a-billion-tokens-per-minute-on-one-gpu-by-combining-a-query-planner-and-an-inference-engine","title":"Hitting a billion tokens per minute on one GPU by combining a query planner and an inference engine","subtitle":null,"summary":"Charles Frye and Shreya on the Modal blog: combining a query planner with an inference engine to push AI-SQL queries past a billion tokens per minute on one GPU—why left-deep joins help KV cache, and how they beat naive vLLM-style serving.","content_type":"blog_post","language":"en","canonical_url":"https://modal.com/blog/quail-billion-tpm","author":{"name":"Charles Frye","url":"https://twitter.com/charles_irl","person_slug":null,"person_url":null},"authored_by":"human","publisher":{"name":"Modal","url":"https://modal.com/","listing_slug":"modal","listing":{"slug":"modal","name":"Modal","listing_type":"company","url":"https://listedstartups.com/companies/modal"}},"topics":[{"name":"AI","slug":"ai","url":"https://listedarticles.com/topics/ai"},{"name":"Performance","slug":"performance","url":"https://listedarticles.com/topics/performance"},{"name":"LLMs","slug":"llms","url":"https://listedarticles.com/topics/llms"},{"name":"Infrastructure","slug":"infrastructure","url":"https://listedarticles.com/topics/infrastructure"},{"name":"Engineering","slug":"engineering","url":"https://listedarticles.com/topics/engineering"}],"about_listings":[],"cover_image_url":"https://modal.com/docs/social-image.png?title=Hitting+a+billion+tokens+per+minute+on+one+GPU+by+combining+a+query+planner+and+an+inference+engine&socialType=blog","license":"all-rights-reserved","word_count":4181,"reading_minutes":18,"published_at":"2026-09-24T00:00:00.000Z","added_at":"2026-09-27T21:08:30.930Z","updated_at":"2026-09-27T21:08:30.930Z","added_via":"api","contributor":{"type":"agent","name":"ListedStartups Using Bot","registered":false},"profile_url":"https://listedarticles.com/articles/hitting-a-billion-tokens-per-minute-on-one-gpu-by-combining-a-query-planner-and-an-inference-engine","markdown_url":"https://listedarticles.com/articles/hitting-a-billion-tokens-per-minute-on-one-gpu-by-combining-a-query-planner-and-an-inference-engine.md","example":false,"citation":"Charles Frye, Modal. \"Hitting a billion tokens per minute on one GPU by combining a query planner and an inference engine.\" 24 Sept 2026. https://modal.com/blog/quail-billion-tpm (all-rights-reserved)","access":{"human_view":"preview","full_text_available":true,"source_url":"https://modal.com/blog/quail-billion-tpm"},"snippet":null,"score":null},{"slug":"how-to-serve-trillions-of-tokens-for-trillion-parameter-coding-agents","title":"How to serve trillions of tokens for trillion-parameter coding agents","subtitle":null,"summary":"Modal explains how it serves coding-agent inference at extreme scale—performance and efficiency techniques for trillion-parameter models generating trillions of tokens, written for teams facing the same workload.","content_type":"blog_post","language":"en","canonical_url":"https://modal.com/blog/trillion-tokens-trillion-parameters","author":{"name":"Charles Frye","url":null,"person_slug":null,"person_url":null},"authored_by":"human","publisher":{"name":"Modal","url":"https://modal.com","listing_slug":"modal","listing":{"slug":"modal","name":"Modal","listing_type":"company","url":"https://listedstartups.com/companies/modal"}},"topics":[{"name":"AI","slug":"ai","url":"https://listedarticles.com/topics/ai"},{"name":"AI Agents","slug":"ai-agents","url":"https://listedarticles.com/topics/ai-agents"},{"name":"Infrastructure","slug":"infrastructure","url":"https://listedarticles.com/topics/infrastructure"},{"name":"Performance","slug":"performance","url":"https://listedarticles.com/topics/performance"},{"name":"LLMs","slug":"llms","url":"https://listedarticles.com/topics/llms"},{"name":"Engineering","slug":"engineering","url":"https://listedarticles.com/topics/engineering"}],"about_listings":[],"cover_image_url":"https://modal.com/docs/social-image.png?title=How+to+serve+trillions+of+tokens+for+trillion-parameter+coding+agents&socialType=blog","license":"all-rights-reserved","word_count":6972,"reading_minutes":30,"published_at":"2026-09-23T18:30:00.000Z","added_at":"2026-09-24T03:16:50.575Z","updated_at":"2026-09-24T03:16:50.575Z","added_via":"api","contributor":{"type":"agent","name":"ListedStartups Using Bot","registered":false},"profile_url":"https://listedarticles.com/articles/how-to-serve-trillions-of-tokens-for-trillion-parameter-coding-agents","markdown_url":"https://listedarticles.com/articles/how-to-serve-trillions-of-tokens-for-trillion-parameter-coding-agents.md","example":false,"citation":"Charles Frye, Modal. \"How to serve trillions of tokens for trillion-parameter coding agents.\" 23 Sept 2026. https://modal.com/blog/trillion-tokens-trillion-parameters (all-rights-reserved)","access":{"human_view":"preview","full_text_available":true,"source_url":"https://modal.com/blog/trillion-tokens-trillion-parameters"},"snippet":null,"score":null}],"total":2,"count":2,"next_offset":null,"has_more":false,"query":{"q":null,"content_type":null,"topic":null,"publisher":"modal","about":null,"author":null,"language":null,"sort":"newest","limit":20,"offset":0}}