{"exhaustive":{"nbHits":true,"typo":true},"exhaustiveNbHits":true,"exhaustiveTypo":true,"hits":[{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"wskwon"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"vLLM: Easy, Fast, and Cheap LLM Serving with PagedAttention"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://vllm.ai/"}},"_tags":["story","author_wskwon","story_36409082"],"author":"wskwon","children":[36409083,36409422,36410971,36410974,36411090,36411111,36411257,36411375,36411560,36411717,36411778,36412091,36412199,36413305,36413909],"created_at":"2023-06-20T19:17:32Z","created_at_i":1687288652,"num_comments":42,"objectID":"36409082","points":295,"story_id":36409082,"title":"vLLM: Easy, Fast, and Cheap LLM Serving with PagedAttention","updated_at":"2026-02-11T02:46:10Z","url":"https://vllm.ai/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"yz-yu"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Nano-vLLM: How a vLLM-style inference engine works"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://neutree.ai/blog/nano-vllm-part-1"}},"_tags":["story","author_yz-yu","story_46855447"],"author":"yz-yu","children":[46856317,46857253,46859173,46860227,46869493,46869877],"created_at":"2026-02-02T12:52:35Z","created_at_i":1770036755,"num_comments":27,"objectID":"46855447","points":271,"story_id":46855447,"title":"Nano-vLLM: How a vLLM-style inference engine works","updated_at":"2026-03-06T03:02:42Z","url":"https://neutree.ai/blog/nano-vllm-part-1"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"yu3zhou4"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Show HN: Tiny-vLLM \u2013 high performance LLM inference engine in C++ and CUDA"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://github.com/jmaczan/tiny-vllm"}},"_tags":["story","author_yu3zhou4","story_48328184","show_hn"],"author":"yu3zhou4","children":[48328913,48328953,48329707,48329980,48329982,48330003,48330136,48331721,48332029,48332174,48333392,48333867,48334472,48337813,48342660,48343580,48348971,48363579],"created_at":"2026-05-29T19:38:27Z","created_at_i":1780083507,"num_comments":18,"objectID":"48328184","points":205,"story_id":48328184,"title":"Show HN: Tiny-vLLM \u2013 high performance LLM inference engine in C++ and CUDA","updated_at":"2026-07-26T18:58:14Z","url":"https://github.com/jmaczan/tiny-vllm"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"samaysharma"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Life of an inference request (vLLM V1): How LLMs are served efficiently at scale"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://www.ubicloud.com/blog/life-of-an-inference-request-vllm-v1"}},"_tags":["story","author_samaysharma","story_44407058"],"author":"samaysharma","children":[44408397,44409281,44409637,44410063,44410575],"created_at":"2025-06-28T18:42:05Z","created_at_i":1751136125,"num_comments":21,"objectID":"44407058","points":175,"story_id":44407058,"title":"Life of an inference request (vLLM V1): How LLMs are served efficiently at scale","updated_at":"2026-08-23T11:49:41Z","url":"https://www.ubicloud.com/blog/life-of-an-inference-request-vllm-v1"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"sebg"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Inside vLLM: Anatomy of a High-Throughput LLM Inference System (2025)"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://www.aleksagordic.com/blog/vllm"}},"_tags":["story","author_sebg","story_49202852"],"author":"sebg","children":[49203643,49204082,49205405,49207633],"created_at":"2026-08-06T21:30:21Z","created_at_i":1786051821,"num_comments":10,"objectID":"49202852","points":151,"story_id":49202852,"title":"Inside vLLM: Anatomy of a High-Throughput LLM Inference System (2025)","updated_at":"2026-08-12T22:03:32Z","url":"https://www.aleksagordic.com/blog/vllm"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"robertnishihara"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"vLLM large scale serving: DeepSeek 2.2k tok/s/h200 with wide-ep"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://blog.vllm.ai/2025/12/17/large-scale-serving.html"}},"_tags":["story","author_robertnishihara","story_46602737"],"author":"robertnishihara","children":[46611750,46611762,46612042,46612045,46612532,46613438,46613887,46614079,46614173,46615087],"created_at":"2026-01-13T15:59:59Z","created_at_i":1768319999,"num_comments":54,"objectID":"46602737","points":147,"story_id":46602737,"title":"vLLM large scale serving: DeepSeek 2.2k tok/s/h200 with wide-ep","updated_at":"2026-03-05T23:22:30Z","url":"https://blog.vllm.ai/2025/12/17/large-scale-serving.html"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"theanonymousone"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"KVarN: Native vLLM backend for KV-cache quantization by Huawei"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://github.com/huawei-csl/KVarN"}},"_tags":["story","author_theanonymousone","story_48399974"],"author":"theanonymousone","children":[48400484,48400498,48401659,48405168,48409738,48411026,48413673],"created_at":"2026-06-04T15:18:00Z","created_at_i":1780586280,"num_comments":16,"objectID":"48399974","points":143,"story_id":48399974,"title":"KVarN: Native vLLM backend for KV-cache quantization by Huawei","updated_at":"2026-06-07T02:03:25Z","url":"https://github.com/huawei-csl/KVarN"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"simonpure"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Nano-Vllm: Lightweight vLLM implementation built from scratch"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://github.com/GeeeekExplorer/nano-vllm"}},"_tags":["story","author_simonpure","story_44352615"],"author":"simonpure","children":[44353614,44354322,44354480,44354627,44354666,44354802,44354872,44354978,44355487,44355956,44357159],"created_at":"2025-06-23T05:10:44Z","created_at_i":1750655444,"num_comments":16,"objectID":"44352615","points":125,"story_id":44352615,"title":"Nano-Vllm: Lightweight vLLM implementation built from scratch","updated_at":"2025-11-05T05:43:16Z","url":"https://github.com/GeeeekExplorer/nano-vllm"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"mrrrcs"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"vLLM v0.28.0"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://github.com/vllm-project/vllm/releases/tag/v0.28.0"}},"_tags":["story","author_mrrrcs","story_49492067"],"author":"mrrrcs","children":[49493106,49493912,49493923,49494065,49494229],"created_at":"2026-08-29T18:22:00Z","created_at_i":1788027720,"num_comments":42,"objectID":"49492067","points":107,"story_id":49492067,"title":"vLLM v0.28.0","updated_at":"2026-08-31T05:00:37Z","url":"https://github.com/vllm-project/vllm/releases/tag/v0.28.0"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"berlianta"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Eagle 3.1: Collaboration Between the EAGLE Team, vLLM Team, and TorchSpec Team"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://vllm.ai/blog/2026-05-26-eagle-3-1"}},"_tags":["story","author_berlianta","story_48278407"],"author":"berlianta","children":[48278814,48279042,48280011,48280355,48282914],"created_at":"2026-05-26T11:46:10Z","created_at_i":1779795970,"num_comments":24,"objectID":"48278407","points":69,"story_id":48278407,"title":"Eagle 3.1: Collaboration Between the EAGLE Team, vLLM Team, and TorchSpec Team","updated_at":"2026-06-02T04:53:06Z","url":"https://vllm.ai/blog/2026-05-26-eagle-3-1"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"lukebechtel"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Surpassing vLLM with a Generated Inference Stack"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://infinity.inc/case-studies/qwen3-optimization"}},"_tags":["story","author_lukebechtel","story_47324364"],"author":"lukebechtel","children":[47327386,47327411,47328308,47329186,47330332,47332192,47333926,47334028],"created_at":"2026-03-10T15:12:52Z","created_at_i":1773155572,"num_comments":22,"objectID":"47324364","points":62,"story_id":47324364,"title":"Surpassing vLLM with a Generated Inference Stack","updated_at":"2026-03-15T00:29:45Z","url":"https://infinity.inc/case-studies/qwen3-optimization"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"waybarrios"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"vLLM-MLX \u2013 Run LLMs on Mac at 464 tok/s"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://github.com/waybarrios/vllm-mlx"}},"_tags":["story","author_waybarrios","story_46642846"],"author":"waybarrios","children":[46642847],"created_at":"2026-01-16T03:58:21Z","created_at_i":1768535901,"num_comments":3,"objectID":"46642846","points":33,"story_id":46642846,"title":"vLLM-MLX \u2013 Run LLMs on Mac at 464 tok/s","updated_at":"2026-05-12T10:58:16Z","url":"https://github.com/waybarrios/vllm-mlx"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"jxmorris12"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"VLLM: Easy, Fast, and Cheap LLM Serving with PagedAttention"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://blog.vllm.ai/2023/06/20/vllm.html"}},"_tags":["story","author_jxmorris12","story_44446280"],"author":"jxmorris12","children":[44468079,44468106,44469560],"created_at":"2025-07-02T17:16:20Z","created_at_i":1751476580,"num_comments":5,"objectID":"44446280","points":20,"story_id":44446280,"title":"VLLM: Easy, Fast, and Cheap LLM Serving with PagedAttention","updated_at":"2025-07-05T10:29:47Z","url":"https://blog.vllm.ai/2023/06/20/vllm.html"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"helloericsf"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Is LMDeploy the Ultimate Solution? Why It Outshines VLLM, TRT-LLM, TGI, and MLC"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://bentoml.com/blog/benchmarking-llm-inference-backends"}},"_tags":["story","author_helloericsf","story_40740065"],"author":"helloericsf","children":[40740092,40740252,40740260,40740508],"created_at":"2024-06-20T15:48:34Z","created_at_i":1718898514,"num_comments":8,"objectID":"40740065","points":16,"story_id":40740065,"title":"Is LMDeploy the Ultimate Solution? Why It Outshines VLLM, TRT-LLM, TGI, and MLC","updated_at":"2024-09-20T17:15:45Z","url":"https://bentoml.com/blog/benchmarking-llm-inference-backends"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"chaoyu"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Benchmarking LLM Inference Back Ends: VLLM, LMDeploy, MLC-LLM, TensorRT-LLM, TGI"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://www.bentoml.com/blog/benchmarking-llm-inference-backends"}},"_tags":["story","author_chaoyu","story_40886218"],"author":"chaoyu","children":[40892750,40894829],"created_at":"2024-07-05T21:32:55Z","created_at_i":1720215175,"num_comments":1,"objectID":"40886218","points":15,"story_id":40886218,"title":"Benchmarking LLM Inference Back Ends: VLLM, LMDeploy, MLC-LLM, TensorRT-LLM, TGI","updated_at":"2025-05-01T04:07:04Z","url":"https://www.bentoml.com/blog/benchmarking-llm-inference-backends"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"skeptrune"},"story_text":{"matchLevel":"none","matchedWords":[],"value":"If you just want to use it, try here - https://pdf2md.trieve.ai . I think the LLM's are astoundingly good at converting complex powerpoint style infographics.
I wouldn't normally think folks on HN would find this interesting as the general concept has been posted about already in the past few months. We were heavily inspired by Zerox[1].
However, the stack we went with was fun and over-engineered which is more likely to create interesting discussion. We use all the same tools at Trieve (our main product), but wanted to see if they would be a good fit for something that needed to get built in a tighter timeline and we think they were!
Took us 2 weeks to get this setup end-to-end and it's by no means complete (see roadmap in linked README). However, it's cool that a relatively cookie cutter web service like this can be created with pure open-source dependencies and non-standard Rust tooling so quickly. Rust won't kill your startup!
- Minijinja templates for the UI[2]
- PDFObject for doc display in-browser[3]
- actix/actix-web HTTP server framework[4]
- Redis queue macro for worker async processing[5]
- Clickhouse for task storage[6]
- chm CLI to handle Clickhouse migrations[7]
- MinIO S3 for object storage[8]
[1]: https://news.ycombinator.com/item?id=41048194
[2]: https://github.com/mitsuhiko/minijinja
[3]: https://github.com/pipwerks/pdfobject
[4]: https://github.com/actix/actix-web
[5]: https://github.com/devflowinc/trieve/blob/main/pdf2md/server...
[6]: https://github.com/ClickHouse/ClickHouse
[7]: https://docs.rs/chm/latest/chm/index.html
[8]: https://github.com/minio/minio"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Show HN: PDF2MD \u2013 Rust+Redis+ClickHouse+VLLM conversion pipeline for PDFs"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://github.com/devflowinc/trieve/tree/main/pdf2md"}},"_tags":["story","author_skeptrune","story_42208648","show_hn"],"author":"skeptrune","created_at":"2024-11-21T21:05:41Z","created_at_i":1732223141,"num_comments":0,"objectID":"42208648","points":13,"story_id":42208648,"story_text":"If you just want to use it, try here - https://pdf2md.trieve.ai . I think the LLM's are astoundingly good at converting complex powerpoint style infographics.
I wouldn't normally think folks on HN would find this interesting as the general concept has been posted about already in the past few months. We were heavily inspired by Zerox[1].
However, the stack we went with was fun and over-engineered which is more likely to create interesting discussion. We use all the same tools at Trieve (our main product), but wanted to see if they would be a good fit for something that needed to get built in a tighter timeline and we think they were!
Took us 2 weeks to get this setup end-to-end and it's by no means complete (see roadmap in linked README). However, it's cool that a relatively cookie cutter web service like this can be created with pure open-source dependencies and non-standard Rust tooling so quickly. Rust won't kill your startup!
- Minijinja templates for the UI[2]
- PDFObject for doc display in-browser[3]
- actix/actix-web HTTP server framework[4]
- Redis queue macro for worker async processing[5]
- Clickhouse for task storage[6]
- chm CLI to handle Clickhouse migrations[7]
- MinIO S3 for object storage[8]
[1]: https://news.ycombinator.com/item?id=41048194
[2]: https://github.com/mitsuhiko/minijinja
[3]: https://github.com/pipwerks/pdfobject
[4]: https://github.com/actix/actix-web
[5]: https://github.com/devflowinc/trieve/blob/main/pdf2md/server...
[6]: https://github.com/ClickHouse/ClickHouse
[7]: https://docs.rs/chm/latest/chm/index.html
[8]: https://github.com/minio/minio","title":"Show HN: PDF2MD \u2013 Rust+Redis+ClickHouse+VLLM conversion pipeline for PDFs","updated_at":"2024-12-30T23:52:44Z","url":"https://github.com/devflowinc/trieve/tree/main/pdf2md"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"sherlockxu"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Benchmarking LLM Inference Back Ends: VLLM, LMDeploy, MLC-LLM, TRT-LLM, and TGI"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://bentoml.com/blog/benchmarking-llm-inference-backends"}},"_tags":["story","author_sherlockxu","story_40601794"],"author":"sherlockxu","children":[40601960,40604730,40605032],"created_at":"2024-06-06T20:08:00Z","created_at_i":1717704480,"num_comments":2,"objectID":"40601794","points":12,"story_id":40601794,"title":"Benchmarking LLM Inference Back Ends: VLLM, LMDeploy, MLC-LLM, TRT-LLM, and TGI","updated_at":"2024-09-20T17:09:35Z","url":"https://bentoml.com/blog/benchmarking-llm-inference-backends"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"daqulalin"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Show HN: MemStitch \u2013 Zero-copy context bridging for vLLM (25x TTFT speedup)"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://github.com/DaqulaLin/MemStitch"}},"_tags":["story","author_daqulalin","story_48901051","show_hn"],"author":"daqulalin","children":[48901071,48901664,48901978,48903399],"created_at":"2026-07-14T01:04:02Z","created_at_i":1783991042,"num_comments":1,"objectID":"48901051","points":12,"story_id":48901051,"title":"Show HN: MemStitch \u2013 Zero-copy context bridging for vLLM (25x TTFT speedup)","updated_at":"2026-07-16T10:53:09Z","url":"https://github.com/DaqulaLin/MemStitch"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"zhwu"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Serving LLM 24x Faster on the Cloud with VLLM and SkyPilot"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://blog.skypilot.co/serving-llm-24x-faster-on-the-cloud-with-vllm-and-skypilot/"}},"_tags":["story","author_zhwu","story_36523357"],"author":"zhwu","children":[36523632],"created_at":"2023-06-29T17:11:17Z","created_at_i":1688058677,"num_comments":1,"objectID":"36523357","points":12,"story_id":36523357,"title":"Serving LLM 24x Faster on the Cloud with VLLM and SkyPilot","updated_at":"2024-09-20T14:23:54Z","url":"https://blog.skypilot.co/serving-llm-24x-faster-on-the-cloud-with-vllm-and-skypilot/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"btwillard"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Our project, Outlines, now offers guided/constrained generation (e.g. according to a JSON schema) via the VLLM library.
My colleague, R\u00e9mi, created some patches that allow one to pass vLLM a JSON schema along with the prompt, which dramatically simplifies deployment of JSON-guided generation. He also added a new `serve` interface that puts it all together and makes serving such models a 2-3 line process.
Check it out and tell us what you think!"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Show HN: VLLM with JSON Guided Generation"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://outlines-dev.github.io/outlines/reference/vllm/"}},"_tags":["story","author_btwillard","story_38735503","show_hn"],"author":"btwillard","children":[38735564],"created_at":"2023-12-22T16:12:11Z","created_at_i":1703261531,"num_comments":3,"objectID":"38735503","points":11,"story_id":38735503,"story_text":"Our project, Outlines, now offers guided/constrained generation (e.g. according to a JSON schema) via the VLLM library.
My colleague, R\u00e9mi, created some patches that allow one to pass vLLM a JSON schema along with the prompt, which dramatically simplifies deployment of JSON-guided generation. He also added a new `serve` interface that puts it all together and makes serving such models a 2-3 line process.
Check it out and tell us what you think!","title":"Show HN: VLLM with JSON Guided Generation","updated_at":"2024-09-20T15:58:59Z","url":"https://outlines-dev.github.io/outlines/reference/vllm/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"twelvenmonkeys"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Free vLLM Course: Inference, Compression, Benchmarks"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://www.deeplearning.ai/courses/fast-and-efficient-llm-inference-with-vllm"}},"_tags":["story","author_twelvenmonkeys","story_48386932"],"author":"twelvenmonkeys","children":[48386933],"created_at":"2026-06-03T17:27:14Z","created_at_i":1780507634,"num_comments":0,"objectID":"48386932","points":8,"story_id":48386932,"title":"Free vLLM Course: Inference, Compression, Benchmarks","updated_at":"2026-06-05T14:16:48Z","url":"https://www.deeplearning.ai/courses/fast-and-efficient-llm-inference-with-vllm"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"pember"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Heaps do lie: debugging a memory leak in vLLM"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://mistral.ai/news/debugging-memory-leak-in-vllm"}},"_tags":["story","author_pember","story_46707015"],"author":"pember","children":[46824814],"created_at":"2026-01-21T15:27:39Z","created_at_i":1769009259,"num_comments":1,"objectID":"46707015","points":7,"story_id":46707015,"title":"Heaps do lie: debugging a memory leak in vLLM","updated_at":"2026-03-05T23:22:30Z","url":"https://mistral.ai/news/debugging-memory-leak-in-vllm"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"ericcurtin"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Hi HN, I\u2019m one of the authors of this post.
We\u2019ve updated Docker Model Runner to support vLLM alongside the existing llama.cpp backend. The goal is to bridge the gap between local prototyping (often done with GGUF/llama.cpp) and high-throughput production (often done with Safetensors/vLLM) using a consistent Docker workflow.
Key technical details:
Auto-routing: The tool detects the model format. If you pull a GGUF model, it routes to llama.cpp. If you pull a Safetensors model, it routes to vLLM.
API: It exposes an OpenAI-compatible API (/v1/chat/completions), so the client code doesn't need to change based on the backend.
Usage: It\u2019s just docker model run ai/smollm2-vllm.
Current Limitations:
Right now, the vLLM backend is optimized for x86_64 with Nvidia GPUs.
We are actively working on WSL2 support for Windows users and DGX Spark compatibility.
Happy to answer any questions about the integration or the roadmap!
https://www.docker.com/blog/docker-model-runner-integrates-v..."},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Show HN: Docker Model Runner Integrates vLLM for High-Throughput Inference"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://github.com/docker/model-runner"}},"_tags":["story","author_ericcurtin","story_45996081","show_hn"],"author":"ericcurtin","children":[45996088],"created_at":"2025-11-20T18:41:51Z","created_at_i":1763664111,"num_comments":1,"objectID":"45996081","points":7,"story_id":45996081,"story_text":"Hi HN, I\u2019m one of the authors of this post.
We\u2019ve updated Docker Model Runner to support vLLM alongside the existing llama.cpp backend. The goal is to bridge the gap between local prototyping (often done with GGUF/llama.cpp) and high-throughput production (often done with Safetensors/vLLM) using a consistent Docker workflow.
Key technical details:
Auto-routing: The tool detects the model format. If you pull a GGUF model, it routes to llama.cpp. If you pull a Safetensors model, it routes to vLLM.
API: It exposes an OpenAI-compatible API (/v1/chat/completions), so the client code doesn't need to change based on the backend.
Usage: It\u2019s just docker model run ai/smollm2-vllm.
Current Limitations:
Right now, the vLLM backend is optimized for x86_64 with Nvidia GPUs.
We are actively working on WSL2 support for Windows users and DGX Spark compatibility.
Happy to answer any questions about the integration or the roadmap!
https://www.docker.com/blog/docker-model-runner-integrates-v...","title":"Show HN: Docker Model Runner Integrates vLLM for High-Throughput Inference","updated_at":"2026-04-01T17:11:45Z","url":"https://github.com/docker/model-runner"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"wskwon"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Kimi K3 on vLLM: Up to 370 Tokens/sec"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://vllm.ai/blog/2026-07-27-k3"}},"_tags":["story","author_wskwon","story_49071233"],"author":"wskwon","created_at":"2026-07-27T15:44:32Z","created_at_i":1785167072,"num_comments":0,"objectID":"49071233","points":7,"story_id":49071233,"title":"Kimi K3 on vLLM: Up to 370 Tokens/sec","updated_at":"2026-07-29T20:41:26Z","url":"https://vllm.ai/blog/2026-07-27-k3"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"mtrofficus"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"A production-grade OCR pipeline on Kubernetes with vLLM and Rust"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://github.com/neural-maze/production-ocr-course"}},"_tags":["story","author_mtrofficus","story_49037050"],"author":"mtrofficus","children":[49037051],"created_at":"2026-07-24T15:25:16Z","created_at_i":1784906716,"num_comments":0,"objectID":"49037050","points":7,"story_id":49037050,"title":"A production-grade OCR pipeline on Kubernetes with vLLM and Rust","updated_at":"2026-07-25T14:32:40Z","url":"https://github.com/neural-maze/production-ocr-course"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"ubermenchh"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"I built this to understand how vLLM works internally."},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Show HN: Mini-vLLM in ~500 lines of Python"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://github.com/ubermenchh/mini-vllm"}},"_tags":["story","author_ubermenchh","story_46415045","show_hn"],"author":"ubermenchh","children":[46419828,46425840],"created_at":"2025-12-28T22:13:01Z","created_at_i":1766959981,"num_comments":4,"objectID":"46415045","points":5,"story_id":46415045,"story_text":"I built this to understand how vLLM works internally.","title":"Show HN: Mini-vLLM in ~500 lines of Python","updated_at":"2026-03-05T23:14:51Z","url":"https://github.com/ubermenchh/mini-vllm"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"mips_avatar"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"vLLM-iOS: 88% Faster Multi-Agent Inference on iOS"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://jonready.com/blog/posts/continuous-batching-on-an-iphone.html"}},"_tags":["story","author_mips_avatar","story_49440382"],"author":"mips_avatar","children":[49440389,49440399,49440471],"created_at":"2026-08-25T20:47:00Z","created_at_i":1787690820,"num_comments":3,"objectID":"49440382","points":5,"story_id":49440382,"title":"vLLM-iOS: 88% Faster Multi-Agent Inference on iOS","updated_at":"2026-08-26T21:35:56Z","url":"https://jonready.com/blog/posts/continuous-batching-on-an-iphone.html"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"addisud"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"vLLM introduces memory optimizations for long-context inference"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://github.com/vllm-project/vllm/releases"}},"_tags":["story","author_addisud","story_47643924"],"author":"addisud","children":[47643925],"created_at":"2026-04-04T21:59:21Z","created_at_i":1775339961,"num_comments":0,"objectID":"47643924","points":5,"story_id":47643924,"title":"vLLM introduces memory optimizations for long-context inference","updated_at":"2026-04-05T03:14:43Z","url":"https://github.com/vllm-project/vllm/releases"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"rpotluri"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"I work on inference scheduling \u2014 KV cache-aware routing, load balancing across GPU workers, that kind of thing. I wanted something like k9s but for my inference stack. Nothing existed, so I built it.
llmtop is a real-time terminal dashboard for LLM inference workers. It scrapes the Prometheus /metrics endpoints that vLLM, SGLang, and LMCache already expose and shows everything in one view: KV cache usage, queue depth, TTFT/ITL latencies (P50/P99 from histogram buckets), token throughput, prefix cache hit rates. Color-coded \u2014 red means go fix it.
```\nbrew install InfraWhisperer/tap/llmtop\nOr go install github.com/InfraWhisperer/llmtop/cmd/llmtop@latest.\n```
Single binary, no Prometheus server needed, no Grafana, no config. Just run llmtop and it auto-discovers local workers.
Written in Go with Bubbletea. Working on Kubernetes pod auto-discovery and a GPU metrics view next."},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Show HN: Llmtop \u2013 Htop for LLM Inference Clusters (vLLM, SGLang, Ollama, llama)"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://github.com/InfraWhisperer/llmtop"}},"_tags":["story","author_rpotluri","story_47421736","show_hn"],"author":"rpotluri","created_at":"2026-03-18T04:58:49Z","created_at_i":1773809929,"num_comments":0,"objectID":"47421736","points":5,"story_id":47421736,"story_text":"I work on inference scheduling \u2014 KV cache-aware routing, load balancing across GPU workers, that kind of thing. I wanted something like k9s but for my inference stack. Nothing existed, so I built it.
llmtop is a real-time terminal dashboard for LLM inference workers. It scrapes the Prometheus /metrics endpoints that vLLM, SGLang, and LMCache already expose and shows everything in one view: KV cache usage, queue depth, TTFT/ITL latencies (P50/P99 from histogram buckets), token throughput, prefix cache hit rates. Color-coded \u2014 red means go fix it.
```\nbrew install InfraWhisperer/tap/llmtop\nOr go install github.com/InfraWhisperer/llmtop/cmd/llmtop@latest.\n```
Single binary, no Prometheus server needed, no Grafana, no config. Just run llmtop and it auto-discovers local workers.
Written in Go with Bubbletea. Working on Kubernetes pod auto-discovery and a GPU metrics view next.","title":"Show HN: Llmtop \u2013 Htop for LLM Inference Clusters (vLLM, SGLang, Ollama, llama)","updated_at":"2026-03-18T07:06:59Z","url":"https://github.com/InfraWhisperer/llmtop"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"platers"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Nano-vLLM: A lightweight vLLM implementation built from scratch"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://github.com/GeeeekExplorer/nano-vllm"}},"_tags":["story","author_platers","story_44334365"],"author":"platers","created_at":"2025-06-21T03:57:21Z","created_at_i":1750478241,"num_comments":0,"objectID":"44334365","points":5,"story_id":44334365,"title":"Nano-vLLM: A lightweight vLLM implementation built from scratch","updated_at":"2025-06-21T08:48:53Z","url":"https://github.com/GeeeekExplorer/nano-vllm"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"xmo"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"vLLM V1: A Major Upgrade to vLLM's Core Architecture"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://blog.vllm.ai/2025/01/27/v1-alpha-release.html"}},"_tags":["story","author_xmo","story_42844059"],"author":"xmo","created_at":"2025-01-27T18:24:36Z","created_at_i":1738002276,"num_comments":0,"objectID":"42844059","points":5,"story_id":42844059,"title":"vLLM V1: A Major Upgrade to vLLM's Core Architecture","updated_at":"2025-06-23T14:38:38Z","url":"https://blog.vllm.ai/2025/01/27/v1-alpha-release.html"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"ahmedriad1"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Show HN: vLLM Studio \u2013 A macOS app for using vLLM models"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://github.com/agentset-ai/vllm-studio"}},"_tags":["story","author_ahmedriad1","story_47470729","show_hn"],"author":"ahmedriad1","children":[47470766,47471519],"created_at":"2026-03-21T20:03:20Z","created_at_i":1774123400,"num_comments":4,"objectID":"47470729","points":4,"story_id":47470729,"title":"Show HN: vLLM Studio \u2013 A macOS app for using vLLM models","updated_at":"2026-03-21T21:47:45Z","url":"https://github.com/agentset-ai/vllm-studio"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"tim_sw"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"VLLM: Anatomy of a High-Throughput LLM Inference System"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://www.aleksagordic.com/blog/vllm"}},"_tags":["story","author_tim_sw","story_45094003"],"author":"tim_sw","children":[45096244],"created_at":"2025-09-01T16:18:09Z","created_at_i":1756743489,"num_comments":1,"objectID":"45094003","points":4,"story_id":45094003,"title":"VLLM: Anatomy of a High-Throughput LLM Inference System","updated_at":"2026-03-05T22:36:07Z","url":"https://www.aleksagordic.com/blog/vllm"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"nesh23"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Show HN: Kvcachescope \u2013 Why Nvidia-smi is blind to vLLM KV cache leaks"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://github.com/brian-mwirigi/kvcachescope"}},"_tags":["story","author_nesh23","story_49305702","show_hn"],"author":"nesh23","children":[49305710,49406864],"created_at":"2026-08-14T23:11:14Z","created_at_i":1786749074,"num_comments":0,"objectID":"49305702","points":4,"story_id":49305702,"title":"Show HN: Kvcachescope \u2013 Why Nvidia-smi is blind to vLLM KV cache leaks","updated_at":"2026-08-23T07:50:36Z","url":"https://github.com/brian-mwirigi/kvcachescope"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"kristianpaul"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"vLLM Recipes"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://recipes.vllm.ai"}},"_tags":["story","author_kristianpaul","story_49168522"],"author":"kristianpaul","created_at":"2026-08-04T13:15:47Z","created_at_i":1785849347,"num_comments":0,"objectID":"49168522","points":4,"story_id":49168522,"title":"vLLM Recipes","updated_at":"2026-08-04T13:30:01Z","url":"https://recipes.vllm.ai"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"robertnishihara"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"67% Cost Savings with PD Disaggregation Using Ray and vLLM on AMD MI325X"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://www.anyscale.com/blog/ray-vllm-prefill-decode-disaggregation-amd-mi325x-67-percent-savings"}},"_tags":["story","author_robertnishihara","story_48551177"],"author":"robertnishihara","created_at":"2026-06-16T06:07:19Z","created_at_i":1781590039,"num_comments":0,"objectID":"48551177","points":4,"story_id":48551177,"title":"67% Cost Savings with PD Disaggregation Using Ray and vLLM on AMD MI325X","updated_at":"2026-06-16T16:32:51Z","url":"https://www.anyscale.com/blog/ray-vllm-prefill-decode-disaggregation-amd-mi325x-67-percent-savings"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"matt_d"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Breaking the Ice: Analyzing Cold Start Latency in vLLM"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://arxiv.org/abs/2606.07362"}},"_tags":["story","author_matt_d","story_48482230"],"author":"matt_d","created_at":"2026-06-10T20:31:09Z","created_at_i":1781123469,"num_comments":0,"objectID":"48482230","points":4,"story_id":48482230,"title":"Breaking the Ice: Analyzing Cold Start Latency in vLLM","updated_at":"2026-06-11T00:28:56Z","url":"https://arxiv.org/abs/2606.07362"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"everlier"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Show HN: Harbor v0.4.19 \u2013 harbor launch \u2013back end vLLM \u2013web codex"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://github.com/av/harbor/releases/tag/v0.4.19"}},"_tags":["story","author_everlier","story_48280543","show_hn"],"author":"everlier","created_at":"2026-05-26T14:43:03Z","created_at_i":1779806583,"num_comments":0,"objectID":"48280543","points":4,"story_id":48280543,"title":"Show HN: Harbor v0.4.19 \u2013 harbor launch \u2013back end vLLM \u2013web codex","updated_at":"2026-05-26T14:53:55Z","url":"https://github.com/av/harbor/releases/tag/v0.4.19"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"matt_d"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"vLLM IR: A Functional Intermediate Representation for vLLM"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://github.com/vllm-project/vllm/issues/32358"}},"_tags":["story","author_matt_d","story_47681076"],"author":"matt_d","created_at":"2026-04-07T20:40:17Z","created_at_i":1775594417,"num_comments":0,"objectID":"47681076","points":4,"story_id":47681076,"title":"vLLM IR: A Functional Intermediate Representation for vLLM","updated_at":"2026-04-07T23:18:41Z","url":"https://github.com/vllm-project/vllm/issues/32358"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"shenli3514"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Inferact: A New Company from the Creators of vLLM ($150M Seed)"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://inferact.ai/"}},"_tags":["story","author_shenli3514","story_46723177"],"author":"shenli3514","created_at":"2026-01-22T18:25:08Z","created_at_i":1769106308,"num_comments":0,"objectID":"46723177","points":4,"story_id":46723177,"title":"Inferact: A New Company from the Creators of vLLM ($150M Seed)","updated_at":"2026-03-05T23:23:32Z","url":"https://inferact.ai/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"yvbbrjdr"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Faster Open-Source Llama3 Serving with SGLang Runtime (vs. TensorRT-LLM, VLLM)"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://lmsys.org/blog/2024-07-25-sglang-llama3/"}},"_tags":["story","author_yvbbrjdr","story_41073838"],"author":"yvbbrjdr","created_at":"2024-07-25T22:01:11Z","created_at_i":1721944871,"num_comments":0,"objectID":"41073838","points":4,"story_id":41073838,"title":"Faster Open-Source Llama3 Serving with SGLang Runtime (vs. TensorRT-LLM, VLLM)","updated_at":"2024-09-24T02:46:57Z","url":"https://lmsys.org/blog/2024-07-25-sglang-llama3/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"paulcjh"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"VLLM with Mistral 7B guide and benchmarks (1.8k+ tokens/s)"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://docs.mystic.ai/docs/mistral-ai-7b-vllm-fast-inference-guide"}},"_tags":["story","author_paulcjh","story_37697751"],"author":"paulcjh","children":[37697765],"created_at":"2023-09-29T00:18:49Z","created_at_i":1695946729,"num_comments":3,"objectID":"37697751","points":3,"story_id":37697751,"title":"VLLM with Mistral 7B guide and benchmarks (1.8k+ tokens/s)","updated_at":"2024-09-20T15:18:24Z","url":"https://docs.mystic.ai/docs/mistral-ai-7b-vllm-fast-inference-guide"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"pythongiant"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"hey everyone, i decided to make a vLLM plugin that implements the Star-KV paper. the results are quiet promising with a decode kernel thats faster than FA2 in higher batch sizes. would love any thoughts and recommendations"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Show HN: Proxima serves 4x more requests with no hardware change on vLLM"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://github.com/Tenosra/Proxima"}},"_tags":["story","author_pythongiant","story_49259550","show_hn"],"author":"pythongiant","children":[49259582],"created_at":"2026-08-11T15:04:01Z","created_at_i":1786460641,"num_comments":1,"objectID":"49259550","points":3,"story_id":49259550,"story_text":"hey everyone, i decided to make a vLLM plugin that implements the Star-KV paper. the results are quiet promising with a decode kernel thats faster than FA2 in higher batch sizes. would love any thoughts and recommendations","title":"Show HN: Proxima serves 4x more requests with no hardware change on vLLM","updated_at":"2026-08-11T17:48:12Z","url":"https://github.com/Tenosra/Proxima"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"IngessLabs"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"How vLLM Works"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://avkcode.github.io/blog/how-vllm-works.html"}},"_tags":["story","author_IngessLabs","story_47996662"],"author":"IngessLabs","children":[47996663,47997473],"created_at":"2026-05-03T13:15:50Z","created_at_i":1777814150,"num_comments":1,"objectID":"47996662","points":3,"story_id":47996662,"title":"How vLLM Works","updated_at":"2026-05-04T14:31:20Z","url":"https://avkcode.github.io/blog/how-vllm-works.html"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"raullen"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"vLLM-mlx \u2013 65 tok/s LLM inference on Mac with tool calling and prompt caching"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://github.com/raullenchai/vllm-mlx"}},"_tags":["story","author_raullen","story_47162364"],"author":"raullen","children":[47162365],"created_at":"2026-02-26T05:52:47Z","created_at_i":1772085167,"num_comments":1,"objectID":"47162364","points":3,"story_id":47162364,"title":"vLLM-mlx \u2013 65 tok/s LLM inference on Mac with tool calling and prompt caching","updated_at":"2026-03-23T13:18:21Z","url":"https://github.com/raullenchai/vllm-mlx"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"EtaoinWu"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Lumine AI from ByteDance: A VLLM Agent That Plays Genshin Impact"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://www.lumine-ai.org/"}},"_tags":["story","author_EtaoinWu","story_45914370"],"author":"EtaoinWu","children":[45914371],"created_at":"2025-11-13T12:58:17Z","created_at_i":1763038697,"num_comments":1,"objectID":"45914370","points":3,"story_id":45914370,"title":"Lumine AI from ByteDance: A VLLM Agent That Plays Genshin Impact","updated_at":"2026-03-05T22:58:23Z","url":"https://www.lumine-ai.org/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"MDK8888"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"A Minimal Implementation of Vllm"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://github.com/MDK8888/vllmini"}},"_tags":["story","author_MDK8888","story_41022766"],"author":"MDK8888","children":[41022767],"created_at":"2024-07-21T05:58:49Z","created_at_i":1721541529,"num_comments":1,"objectID":"41022766","points":3,"story_id":41022766,"title":"A Minimal Implementation of Vllm","updated_at":"2024-12-09T21:48:36Z","url":"https://github.com/MDK8888/vllmini"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"sherlockxu"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Self-Host LLMs with VLLM and BentoML"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://github.com/bentoml/BentoVLLM"}},"_tags":["story","author_sherlockxu","story_40141938"],"author":"sherlockxu","children":[40182171],"created_at":"2024-04-24T08:26:46Z","created_at_i":1713947206,"num_comments":1,"objectID":"40141938","points":3,"story_id":40141938,"title":"Self-Host LLMs with VLLM and BentoML","updated_at":"2024-09-20T16:52:30Z","url":"https://github.com/bentoml/BentoVLLM"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"simonpure"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"Native-speed vLLM transformers modeling back end"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://huggingface.co/blog/native-speed-vllm-transformers-backend"}},"_tags":["story","author_simonpure","story_48838117"],"author":"simonpure","created_at":"2026-07-08T22:16:44Z","created_at_i":1783549004,"num_comments":0,"objectID":"48838117","points":3,"story_id":48838117,"title":"Native-speed vLLM transformers modeling back end","updated_at":"2026-07-09T18:30:19Z","url":"https://huggingface.co/blog/native-speed-vllm-transformers-backend"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"kristianpaul"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"vLLM Recipes"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["vllm"],"value":"https://recipes.vllm.ai"}},"_tags":["story","author_kristianpaul","story_48648305"],"author":"kristianpaul","created_at":"2026-06-23T17:26:41Z","created_at_i":1782235601,"num_comments":0,"objectID":"48648305","points":3,"story_id":48648305,"title":"vLLM Recipes","updated_at":"2026-06-25T05:58:50Z","url":"https://recipes.vllm.ai"}],"hitsPerPage":50,"nbHits":29913,"nbPages":20,"page":0,"params":"query=vLLM&tags=story&hitsPerPage=50&advancedSyntax=true&analyticsTags=backend","processingTimeMS":25,"processingTimingsMS":{"_request":{"roundTrip":28},"afterFetch":{"format":{"highlighting":1,"total":1},"merge":{"mergeLoop":{"prepareNextHit":8,"total":8},"total":10},"total":10},"fetch":{"query":3,"scanning":10,"total":14},"total":25},"query":"vLLM","serverTimeMS":28}