{"exhaustive":{"nbHits":false,"typo":false},"exhaustiveNbHits":false,"exhaustiveTypo":false,"hits":[{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"jumploops"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"LMArena is a cancer on AI"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://surgehq.ai/blog/lmarena-is-a-plague-on-ai"}},"_tags":["story","author_jumploops","story_46522632"],"author":"jumploops","children":[46533262,46533342,46533570,46533584,46533605,46533616,46533766,46533772,46533788,46533801,46533826,46533969,46534126,46534137,46534352,46534501,46534602,46534606,46534794,46534801,46535178,46535192,46535361,46535419,46535551,46535663,46535906,46536032,46536429,46536468,46536631,46536882,46538513,46539529,46539689,46544573],"created_at":"2026-01-07T04:40:48Z","created_at_i":1767760848,"num_comments":100,"objectID":"46522632","points":246,"story_id":46522632,"title":"LMArena is a cancer on AI","updated_at":"2026-03-05T23:17:57Z","url":"https://surgehq.ai/blog/lmarena-is-a-plague-on-ai"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"reed1234"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"GPT-5.2-high LMArena scores released, OpenAI falls from #6 to #13"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://lmarena.ai/leaderboard"}},"_tags":["story","author_reed1234","story_46298597"],"author":"reed1234","children":[46298598,46299317],"created_at":"2025-12-17T05:26:27Z","created_at_i":1765949187,"num_comments":3,"objectID":"46298597","points":13,"story_id":46298597,"title":"GPT-5.2-high LMArena scores released, OpenAI falls from #6 to #13","updated_at":"2026-03-05T23:14:04Z","url":"https://lmarena.ai/leaderboard"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"weavedfreedunes"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"Meta got caught gaming LMArena"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://www.theverge.com/meta/645012/meta-llama-4-maverick-benchmarks-gaming"}},"_tags":["story","author_weavedfreedunes","story_43617660"],"author":"weavedfreedunes","children":[43617702,43617881],"created_at":"2025-04-08T01:54:50Z","created_at_i":1744077290,"num_comments":2,"objectID":"43617660","points":11,"story_id":43617660,"title":"Meta got caught gaming LMArena","updated_at":"2025-04-08T17:15:25Z","url":"https://www.theverge.com/meta/645012/meta-llama-4-maverick-benchmarks-gaming"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"moteo_dev"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"A motion-graphic comparison website in the vein of LMArena. The videos are rendered via Remotion.
We hope that AI will be used in interesting ways to help with video production, so we wanted to give some of the models available today a shot at some basic graphics."},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"Show HN: I built LMArena for Motion Graphics"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://graphicarena-1.onrender.com/"}},"_tags":["story","author_moteo_dev","story_44879451","show_hn"],"author":"moteo_dev","children":[44880261,44880266,44882381],"created_at":"2025-08-12T17:34:27Z","created_at_i":1755020067,"num_comments":3,"objectID":"44879451","points":8,"story_id":44879451,"story_text":"A motion-graphic comparison website in the vein of LMArena. The videos are rendered via Remotion.
We hope that AI will be used in interesting ways to help with video production, so we wanted to give some of the models available today a shot at some basic graphics.","title":"Show HN: I built LMArena for Motion Graphics","updated_at":"2026-03-05T22:34:15Z","url":"https://graphicarena-1.onrender.com/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"gk1"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"LMArena Is a Cancer on AI"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://surgehq.ai/blog/lmarena-is-a-plague-on-ai?r=greg"}},"_tags":["story","author_gk1","story_46514691"],"author":"gk1","children":[46515167],"created_at":"2026-01-06T16:46:12Z","created_at_i":1767717972,"num_comments":1,"objectID":"46514691","points":6,"story_id":46514691,"title":"LMArena Is a Cancer on AI","updated_at":"2026-03-05T23:17:25Z","url":"https://surgehq.ai/blog/lmarena-is-a-plague-on-ai?r=greg"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"EvgeniyZh"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"LMArena Is a Cancer on AI"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://surgehq.ai/blog/lmarena-is-a-plague-on-ai"}},"_tags":["story","author_EvgeniyZh","story_46240450"],"author":"EvgeniyZh","children":[46241374],"created_at":"2025-12-12T03:08:46Z","created_at_i":1765508926,"num_comments":1,"objectID":"46240450","points":5,"story_id":46240450,"title":"LMArena Is a Cancer on AI","updated_at":"2026-03-05T23:10:03Z","url":"https://surgehq.ai/blog/lmarena-is-a-plague-on-ai"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"holdingunsteady"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"LMArena is a cancer on AI"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://surgehq.ai/blog/lmarena-is-a-plague-on-ai"}},"_tags":["story","author_holdingunsteady","story_46206339"],"author":"holdingunsteady","children":[46206612],"created_at":"2025-12-09T15:55:47Z","created_at_i":1765295747,"num_comments":1,"objectID":"46206339","points":4,"story_id":46206339,"title":"LMArena is a cancer on AI","updated_at":"2026-03-05T23:07:21Z","url":"https://surgehq.ai/blog/lmarena-is-a-plague-on-ai"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"joaogui1"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"Released Llama 4 Maverick places 32nd in LMArena"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://lmarena.ai/?gad_source=1&gclid=CjwKCAjw--K_BhB5EiwAuwYoyopi5o5QOVDWwcwwpf8-sKX0LqEb4LLqwEBIXPkTri3wrvi5TdwQ2xoCabUQAvD_BwE"}},"_tags":["story","author_joaogui1","story_43652957"],"author":"joaogui1","children":[43652958],"created_at":"2025-04-11T12:17:25Z","created_at_i":1744373845,"num_comments":1,"objectID":"43652957","points":4,"story_id":43652957,"title":"Released Llama 4 Maverick places 32nd in LMArena","updated_at":"2025-04-11T15:05:09Z","url":"https://lmarena.ai/?gad_source=1&gclid=CjwKCAjw--K_BhB5EiwAuwYoyopi5o5QOVDWwcwwpf8-sKX0LqEb4LLqwEBIXPkTri3wrvi5TdwQ2xoCabUQAvD_BwE"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"1024core"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"Google's Gemini ranked #1 on LMArena across all categories"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://twitter.com/JeffDean/status/1865081640546156993"}},"_tags":["story","author_1024core","story_42344605"],"author":"1024core","children":[42344848],"created_at":"2024-12-06T21:23:02Z","created_at_i":1733520182,"num_comments":1,"objectID":"42344605","points":4,"story_id":42344605,"title":"Google's Gemini ranked #1 on LMArena across all categories","updated_at":"2024-12-07T00:04:40Z","url":"https://twitter.com/JeffDean/status/1865081640546156993"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"parris"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"LMArena launches new video arena"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://twitter.com/lmarena_ai/status/1950593176009662698"}},"_tags":["story","author_parris","story_44736586"],"author":"parris","created_at":"2025-07-30T16:54:12Z","created_at_i":1753894452,"num_comments":0,"objectID":"44736586","points":4,"story_id":44736586,"title":"LMArena launches new video arena","updated_at":"2025-07-30T17:20:26Z","url":"https://twitter.com/lmarena_ai/status/1950593176009662698"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"cui"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"LMArena Is a Plague on AI"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://surgehq.ai/blog/lmarena-is-a-plague-on-ai"}},"_tags":["story","author_cui","story_46198504"],"author":"cui","children":[46198505,46198820],"created_at":"2025-12-08T22:28:03Z","created_at_i":1765232883,"num_comments":1,"objectID":"46198504","points":3,"story_id":46198504,"title":"LMArena Is a Plague on AI","updated_at":"2026-03-05T23:10:31Z","url":"https://surgehq.ai/blog/lmarena-is-a-plague-on-ai"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"peterdavehello"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"The latest ChatGPT-4o (2025-03-26) jumps to #2 on LMArena, surpassing GPT-4.5"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://twitter.com/lmarena_ai/status/1905340075225043057"}},"_tags":["story","author_peterdavehello","story_43497433"],"author":"peterdavehello","children":[43497634],"created_at":"2025-03-27T19:52:54Z","created_at_i":1743105174,"num_comments":1,"objectID":"43497433","points":3,"story_id":43497433,"title":"The latest ChatGPT-4o (2025-03-26) jumps to #2 on LMArena, surpassing GPT-4.5","updated_at":"2025-03-27T22:34:58Z","url":"https://twitter.com/lmarena_ai/status/1905340075225043057"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"NavinF"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"Grok-3-Preview-02-24 tops lmarena Leaderboard; GPT-4.5-Preview within 95% CI"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://lmarena.ai/?leaderboard"}},"_tags":["story","author_NavinF","story_43259983"],"author":"NavinF","children":[43259998,43260019],"created_at":"2025-03-04T21:30:51Z","created_at_i":1741123851,"num_comments":1,"objectID":"43259983","points":3,"story_id":43259983,"title":"Grok-3-Preview-02-24 tops lmarena Leaderboard; GPT-4.5-Preview within 95% CI","updated_at":"2025-03-04T22:54:02Z","url":"https://lmarena.ai/?leaderboard"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"thunderbong"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"MAI-Image-1, debuting in the top on LMArena"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://microsoft.ai/news/introducing-mai-image-1-debuting-in-the-top-10-on-lmarena/"}},"_tags":["story","author_thunderbong","story_45590323"],"author":"thunderbong","created_at":"2025-10-15T10:22:36Z","created_at_i":1760523756,"num_comments":0,"objectID":"45590323","points":3,"story_id":45590323,"title":"MAI-Image-1, debuting in the top on LMArena","updated_at":"2026-03-05T22:53:04Z","url":"https://microsoft.ai/news/introducing-mai-image-1-debuting-in-the-top-10-on-lmarena/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"RyanShook"},"title":{"fullyHighlighted":true,"matchLevel":"full","matchedWords":["lmarena"],"value":"LMArena"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://beta.lmarena.ai/"}},"_tags":["story","author_RyanShook","story_44055145"],"author":"RyanShook","created_at":"2025-05-21T19:16:52Z","created_at_i":1747855012,"num_comments":0,"objectID":"44055145","points":3,"story_id":44055145,"title":"LMArena","updated_at":"2025-05-21T19:29:41Z","url":"https://beta.lmarena.ai/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"lostmsu"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"LM arena public voting is not objective for LLM evaluation"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://old.reddit.com/r/MachineLearning/comments/1i83mhj/lm_arena_public_voting_is_not_objective_for_llm/"}},"_tags":["story","author_lostmsu","story_42809061"],"author":"lostmsu","children":[42809119,42809292],"created_at":"2025-01-23T23:17:59Z","created_at_i":1737674279,"num_comments":2,"objectID":"42809061","points":2,"story_id":42809061,"title":"LM arena public voting is not objective for LLM evaluation","updated_at":"2025-02-18T10:49:03Z","url":"https://old.reddit.com/r/MachineLearning/comments/1i83mhj/lm_arena_public_voting_is_not_objective_for_llm/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"sreenathmenon"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"Llmswap: Avoid LLM vendor lock-in \u2013 10 providers with top LMArena models"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://github.com/sreenathmmenon/llmswap"}},"_tags":["story","author_sreenathmenon","story_45476581"],"author":"sreenathmenon","children":[45476582],"created_at":"2025-10-04T20:50:30Z","created_at_i":1759611030,"num_comments":1,"objectID":"45476581","points":2,"story_id":45476581,"title":"Llmswap: Avoid LLM vendor lock-in \u2013 10 providers with top LMArena models","updated_at":"2026-03-05T22:50:21Z","url":"https://github.com/sreenathmmenon/llmswap"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"toomuchtodo"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"LMArena Goes from Academic Project to $600M Startup"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://www.bloomberg.com/news/articles/2025-05-21/lmarena-goes-from-academic-project-to-600-million-startup"}},"_tags":["story","author_toomuchtodo","story_44052584"],"author":"toomuchtodo","children":[44052609],"created_at":"2025-05-21T15:32:28Z","created_at_i":1747841548,"num_comments":1,"objectID":"44052584","points":2,"story_id":44052584,"title":"LMArena Goes from Academic Project to $600M Startup","updated_at":"2025-05-21T18:23:56Z","url":"https://www.bloomberg.com/news/articles/2025-05-21/lmarena-goes-from-academic-project-to-600-million-startup"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"chandureddyvari"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"The latest Gemini 2.5 Pro reflects a 24-point Elo score jump on LMArena"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://blog.google/products/gemini/gemini-2-5-pro-latest-preview/"}},"_tags":["story","author_chandureddyvari","story_44202386"],"author":"chandureddyvari","created_at":"2025-06-06T16:19:31Z","created_at_i":1749226771,"num_comments":0,"objectID":"44202386","points":2,"story_id":44202386,"title":"The latest Gemini 2.5 Pro reflects a 24-point Elo score jump on LMArena","updated_at":"2025-06-06T17:16:57Z","url":"https://blog.google/products/gemini/gemini-2-5-pro-latest-preview/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"doener"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"Llama 4 has been added to LMArena after it was found out they cheated"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://twitter.com/pigeon__s/status/1910705956486336586"}},"_tags":["story","author_doener","story_43686287"],"author":"doener","created_at":"2025-04-14T21:00:35Z","created_at_i":1744664435,"num_comments":0,"objectID":"43686287","points":2,"story_id":43686287,"title":"Llama 4 has been added to LMArena after it was found out they cheated","updated_at":"2025-04-15T03:19:25Z","url":"https://twitter.com/pigeon__s/status/1910705956486336586"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"maxloh"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"Meta accused of Llama 4 bait-n-switch to juice LMArena rank"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://www.theregister.com/2025/04/08/meta_llama4_cheating/"}},"_tags":["story","author_maxloh","story_43630528"],"author":"maxloh","created_at":"2025-04-09T10:05:34Z","created_at_i":1744193134,"num_comments":0,"objectID":"43630528","points":2,"story_id":43630528,"title":"Meta accused of Llama 4 bait-n-switch to juice LMArena rank","updated_at":"2025-04-09T10:14:57Z","url":"https://www.theregister.com/2025/04/08/meta_llama4_cheating/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"gmays"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"OpenAI testing new Image-2 models on LM Arena"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://www.testingcatalog.com/openai-testing-new-image-2-models-on-lm-arena/"}},"_tags":["story","author_gmays","story_46218844"],"author":"gmays","created_at":"2025-12-10T15:33:38Z","created_at_i":1765380818,"num_comments":0,"objectID":"46218844","points":1,"story_id":46218844,"title":"OpenAI testing new Image-2 models on LM Arena","updated_at":"2026-03-05T23:08:22Z","url":"https://www.testingcatalog.com/openai-testing-new-image-2-models-on-lm-arena/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"Jasondells"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"Alledged Gemini 3.0 Pre-Release Models Lithiumflow and Orionmist on LMArena"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://winbuzzer.com/2025/10/20/alledged-google-gemini-3-0-pre-release-models-lithiumflow-and-orionmist-surface-on-lmarena-xcxwbn/"}},"_tags":["story","author_Jasondells","story_45641961"],"author":"Jasondells","created_at":"2025-10-20T09:45:42Z","created_at_i":1760953542,"num_comments":0,"objectID":"45641961","points":1,"story_id":45641961,"title":"Alledged Gemini 3.0 Pre-Release Models Lithiumflow and Orionmist on LMArena","updated_at":"2026-03-05T22:50:55Z","url":"https://winbuzzer.com/2025/10/20/alledged-google-gemini-3-0-pre-release-models-lithiumflow-and-orionmist-surface-on-lmarena-xcxwbn/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"consumer451"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"This must be a bug, or maybe something more interesting?
On any sub, if you post a link to LLM industry fave site Chatbot Arena (LMSYS) [0], it will get [removed by Reddit].
I just listened to the latest Gradient Dissent pod, and the creator of lmarena.ai was hoping to get more non-industry users to use their site. What is going on here?
[0] https://lmarena.ai"},"title":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"Tell HN: Linking to lmarena.ai is banned, site-wide on Reddit"}},"_tags":["story","author_consumer451","story_42551846","ask_hn"],"author":"consumer451","created_at":"2024-12-30T18:07:42Z","created_at_i":1735582062,"num_comments":0,"objectID":"42551846","points":1,"story_id":42551846,"story_text":"This must be a bug, or maybe something more interesting?
On any sub, if you post a link to LLM industry fave site Chatbot Arena (LMSYS) [0], it will get [removed by Reddit].
I just listened to the latest Gradient Dissent pod, and the creator of lmarena.ai was hoping to get more non-industry users to use their site. What is going on here?
[0] https://lmarena.ai","title":"Tell HN: Linking to lmarena.ai is banned, site-wide on Reddit","updated_at":"2024-12-30T18:45:03Z"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"jarodrh"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"I started leaning in on AI heavily this year, as I wanted to get more done autonomously, but then my token usage climbed dramatically to the point where my weekly quota would run out before the end of the week, sometimes a couple of days into the week.
I realised I had to do something about it else I'd have to double my spend. So I decided to start tracking my cost per task type. This revealed that a lot of my spend went to searches/scans or simple things like scouting tasks.
I then decided to turn this into a simple CLI tool that can be used to read your OpenAI-style logs locally, and analyze the cost and compare this spend to other models, then show you how much you could potentially save by switching those calls to a cheaper model.
When you run analyze you get an offline estimate priced against LiteLLM and gated by LMArena tiers. The general savings bands come from the research published by RouteLLM; but you can confirm this yourself using 2 commands --measure (shows the prompt-response output side by side) and --judge (a model chosen to do the comparisons). These send a sample of the prompts from the logs to the candidate models - either the default choice or set by you. This call goes directly to the model provider (never through me) as any normal LLM call would, and the response is shown and judged to either be better or worse or a tie.
It's deliberately small, because I tend to over complicate/think things sometimes: analyze + capture + a few commands, doing three jobs. Cost, quality visibility, routing recommendation.
Nothing is hosted. capture is an optional local proxy on your own machine, and there's no endpoint in the path of your data. You can confirm this by checking the source.
I included a demo so you can check out the output. It has a synthetic 56k call log (a month's worth) showing how costs can drop from $549.46 to $343.91 a month. A 37.4% saving.
Try it:
uvx frugon analyze --demo\n\nor uv tool install frugon\n\nThen point it at your own logs.All feedback is welcome, especially any on the routing/quality logic, or anything else, good or bad."},"title":{"matchLevel":"none","matchedWords":[],"value":"Show HN: Frugon \u2013 Find which LLM calls a cheaper model could handle (local, MIT)"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://github.com/Rodiun/frugon"}},"_tags":["story","author_jarodrh","story_48816724","show_hn"],"author":"jarodrh","children":[48829495,48837992,48839545,48839736,48841498,48843451,48843538,48855432,48867590,48870323,48872619,48873987,48876994,48881397,48968057],"created_at":"2026-07-07T12:20:54Z","created_at_i":1783426854,"num_comments":24,"objectID":"48816724","points":67,"story_id":48816724,"story_text":"I started leaning in on AI heavily this year, as I wanted to get more done autonomously, but then my token usage climbed dramatically to the point where my weekly quota would run out before the end of the week, sometimes a couple of days into the week.
I realised I had to do something about it else I'd have to double my spend. So I decided to start tracking my cost per task type. This revealed that a lot of my spend went to searches/scans or simple things like scouting tasks.
I then decided to turn this into a simple CLI tool that can be used to read your OpenAI-style logs locally, and analyze the cost and compare this spend to other models, then show you how much you could potentially save by switching those calls to a cheaper model.
When you run analyze you get an offline estimate priced against LiteLLM and gated by LMArena tiers. The general savings bands come from the research published by RouteLLM; but you can confirm this yourself using 2 commands --measure (shows the prompt-response output side by side) and --judge (a model chosen to do the comparisons). These send a sample of the prompts from the logs to the candidate models - either the default choice or set by you. This call goes directly to the model provider (never through me) as any normal LLM call would, and the response is shown and judged to either be better or worse or a tie.
It's deliberately small, because I tend to over complicate/think things sometimes: analyze + capture + a few commands, doing three jobs. Cost, quality visibility, routing recommendation.
Nothing is hosted. capture is an optional local proxy on your own machine, and there's no endpoint in the path of your data. You can confirm this by checking the source.
I included a demo so you can check out the output. It has a synthetic 56k call log (a month's worth) showing how costs can drop from $549.46 to $343.91 a month. A 37.4% saving.
Try it:
uvx frugon analyze --demo\n\nor uv tool install frugon\n\nThen point it at your own logs.All feedback is welcome, especially any on the routing/quality logic, or anything else, good or bad.","title":"Show HN: Frugon \u2013 Find which LLM calls a cheaper model could handle (local, MIT)","updated_at":"2026-07-19T17:46:20Z","url":"https://github.com/Rodiun/frugon"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"ieuanking"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"Hi HN,
I'm one of the co-founders of PreCog AI, a project my friend and I started to make the best AI models more accessible. PreCog AI is a chatbot that automatically picks and answers with the best AI model for whatever task you throw at it.
We made PreCog public on Monday and are getting great feedback. Originally built as an internal tool to help our small team reduce costs (paying for various chatbots) and get better AI output, PreCog has helped us so much with our workflow and ideation that we just had to share it.
Key Features of PreCog \n- AI Model Matchmaking: With access to 18 models, PreCog automatically matches your questions with the most fitting AI model based on the task.
- Versatile Adaptation: Works with any task, from coding to creative writing, giving you the right tool for the job.
-Ongoing Updates: Stay current with AI advancements using the latest LLM leaderboard data (we are constantly adding and changing our leaderboard). See the leaderboard here - https://precog.ubik.studio/leaderboard
-Preferred Model Selection: If you have a preferred model, choose it, and PreCog will use that model exclusively to respond.
How PreCog Works:
PreCog analyzes your query, references the model leaderboard, and then matches your query with the highest-ranked AI for that niche task. Delivering high-quality, task-specific output. PreCog's Model Leaderboard ranks AI models through over a million human comparisons, evaluated and presented on an Elo-scale. The dataset used to build the PreCog Leaderboard is from ChatBot Arena by https://lmarena.ai/. Researchers from UC Berkeley SkyLab and LMSYS developed the battle framework to produce the dataset.
I love feedback, questions, and critiques it helps me and my friend develop with the user in mind.
You can reach me at anytime at:\ninfo@ubik.studio"},"title":{"matchLevel":"none","matchedWords":[],"value":"Show HN: PreCog AI \u2013 Automatic AI Model Selection for Any Task"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://precog.ubik.studio/"}},"_tags":["story","author_ieuanking","story_41937572","show_hn"],"author":"ieuanking","children":[41937938,41938210,41938502,41938769],"created_at":"2024-10-24T17:25:42Z","created_at_i":1729790742,"num_comments":18,"objectID":"41937572","points":61,"story_id":41937572,"story_text":"Hi HN,
I'm one of the co-founders of PreCog AI, a project my friend and I started to make the best AI models more accessible. PreCog AI is a chatbot that automatically picks and answers with the best AI model for whatever task you throw at it.
We made PreCog public on Monday and are getting great feedback. Originally built as an internal tool to help our small team reduce costs (paying for various chatbots) and get better AI output, PreCog has helped us so much with our workflow and ideation that we just had to share it.
Key Features of PreCog \n- AI Model Matchmaking: With access to 18 models, PreCog automatically matches your questions with the most fitting AI model based on the task.
- Versatile Adaptation: Works with any task, from coding to creative writing, giving you the right tool for the job.
-Ongoing Updates: Stay current with AI advancements using the latest LLM leaderboard data (we are constantly adding and changing our leaderboard). See the leaderboard here - https://precog.ubik.studio/leaderboard
-Preferred Model Selection: If you have a preferred model, choose it, and PreCog will use that model exclusively to respond.
How PreCog Works:
PreCog analyzes your query, references the model leaderboard, and then matches your query with the highest-ranked AI for that niche task. Delivering high-quality, task-specific output. PreCog's Model Leaderboard ranks AI models through over a million human comparisons, evaluated and presented on an Elo-scale. The dataset used to build the PreCog Leaderboard is from ChatBot Arena by https://lmarena.ai/. Researchers from UC Berkeley SkyLab and LMSYS developed the battle framework to produce the dataset.
I love feedback, questions, and critiques it helps me and my friend develop with the user in mind.
You can reach me at anytime at:\ninfo@ubik.studio","title":"Show HN: PreCog AI \u2013 Automatic AI Model Selection for Any Task","updated_at":"2024-10-30T12:28:02Z","url":"https://precog.ubik.studio/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"codelensai"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"I built CodeLens.AI - a tool that compares how 6 top LLMs (GPT-5, Claude Opus 4.1, Claude Sonnet 4.5, Grok 4, Gemini 2.5 Pro, o3) handle your actual code tasks.
How it works:\n- Upload code + describe task (refactoring, security review, architecture, etc.)
- All 6 models run in parallel (~2-5 min)
- See side-by-side comparison with AI judge scores
- Community votes on winners (blind voting)
- Each evaluation gets reflected in the overall AI model leaderboard, showing us best ones
Why I built this: Existing benchmarks (HumanEval, SWE-Bench) don't reflect real-world developer tasks. I wanted to know which model actually solves MY specific problems - refactoring legacy TypeScript, reviewing React components, etc. It's also similar to LMArena, but their evaluations are not entirely transparent.
Current status:
- Live at https://codelens.ai
- 23 evaluations so far (small sample, I know!)
- Free tier processes 3 evals per day (first-come, first-served queue)
- Looking for real tasks to make the benchmark meaningful
- Happy to answer questions about the tech stack, cost structure, or methodology.
Currently in validation stage. What are your first impressions?"},"title":{"matchLevel":"none","matchedWords":[],"value":"Show HN: Benchmark AI on your actual code (GPT-5, Claude, Grok, Gemini, o3)"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://codelens.ai"}},"_tags":["story","author_codelensai","story_45553152","show_hn"],"author":"codelensai","created_at":"2025-10-11T22:19:31Z","created_at_i":1760221171,"num_comments":0,"objectID":"45553152","points":7,"story_id":45553152,"story_text":"I built CodeLens.AI - a tool that compares how 6 top LLMs (GPT-5, Claude Opus 4.1, Claude Sonnet 4.5, Grok 4, Gemini 2.5 Pro, o3) handle your actual code tasks.
How it works:\n- Upload code + describe task (refactoring, security review, architecture, etc.)
- All 6 models run in parallel (~2-5 min)
- See side-by-side comparison with AI judge scores
- Community votes on winners (blind voting)
- Each evaluation gets reflected in the overall AI model leaderboard, showing us best ones
Why I built this: Existing benchmarks (HumanEval, SWE-Bench) don't reflect real-world developer tasks. I wanted to know which model actually solves MY specific problems - refactoring legacy TypeScript, reviewing React components, etc. It's also similar to LMArena, but their evaluations are not entirely transparent.
Current status:
- Live at https://codelens.ai
- 23 evaluations so far (small sample, I know!)
- Free tier processes 3 evals per day (first-come, first-served queue)
- Looking for real tasks to make the benchmark meaningful
- Happy to answer questions about the tech stack, cost structure, or methodology.
Currently in validation stage. What are your first impressions?","title":"Show HN: Benchmark AI on your actual code (GPT-5, Claude, Grok, Gemini, o3)","updated_at":"2026-03-05T22:50:30Z","url":"https://codelens.ai"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"foke82"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"Hi HN,
I'm a software engineer working on C++/python/robotics by day, dabbing into web apps by night. I built The Frontier (https://the-frontier.app) because the LLM market is moving so fast it's hard to tell if you're overpaying for performance.
Pricing is easy to find, but it's hard to tell if you're missing a similarly priced or even cheaper model with better performance. So I built a visualization that maps LM Arena\u2019s Elo scores against OpenRouter\u2019s pricing.
The main thing it does is calculate the Pareto frontier. It highlights the optimal models at each price point, so you can easily spot when a model is technically a "bad deal" compared to its peers.
The hard part:\nThe real headache wasn't the UI, it was the messy data. LMArena names models one way (e.g. "qwen3-coder-480b-a35b-instruct"), OpenRouter another ("qwen/qwen3-coder"), and you have to deal with a mess of variants like "thinking", "instruct", "fast", or "v1.0" vs "v1". I ended up building an automated scoring system to match these models automatically so the chart stays clean without manual mapping.
I'm pretty happy with the result, I find myself surfing the frontier (literally), going up and down the frontier to find the best model for my use case and budget.
The Tech:\n- React + Vite\n- ECharts for the visualization\n- A daily sync to keep the chart up-to-date with new releases
I also just added Latency and Throughput metrics because sometimes latency or throughput is just as important as intelligence.
I\u2019d love to hear what you think, especially if you spot any weird model matches (Unfortunately they still happen) or have ideas of what to add next ! I have a few ideas, like combining latency and throughput into one, or even intelligence, latency and throughput, I'll call it Wisdom :)
URL: https://the-frontier.app/
Thanks!"},"title":{"matchLevel":"none","matchedWords":[],"value":"Show HN: The Frontier, Tracking the LLM Pareto Frontier"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://the-frontier.app/"}},"_tags":["story","author_foke82","story_46960202","show_hn"],"author":"foke82","created_at":"2026-02-10T14:35:24Z","created_at_i":1770734124,"num_comments":0,"objectID":"46960202","points":3,"story_id":46960202,"story_text":"Hi HN,
I'm a software engineer working on C++/python/robotics by day, dabbing into web apps by night. I built The Frontier (https://the-frontier.app) because the LLM market is moving so fast it's hard to tell if you're overpaying for performance.
Pricing is easy to find, but it's hard to tell if you're missing a similarly priced or even cheaper model with better performance. So I built a visualization that maps LM Arena\u2019s Elo scores against OpenRouter\u2019s pricing.
The main thing it does is calculate the Pareto frontier. It highlights the optimal models at each price point, so you can easily spot when a model is technically a "bad deal" compared to its peers.
The hard part:\nThe real headache wasn't the UI, it was the messy data. LMArena names models one way (e.g. "qwen3-coder-480b-a35b-instruct"), OpenRouter another ("qwen/qwen3-coder"), and you have to deal with a mess of variants like "thinking", "instruct", "fast", or "v1.0" vs "v1". I ended up building an automated scoring system to match these models automatically so the chart stays clean without manual mapping.
I'm pretty happy with the result, I find myself surfing the frontier (literally), going up and down the frontier to find the best model for my use case and budget.
The Tech:\n- React + Vite\n- ECharts for the visualization\n- A daily sync to keep the chart up-to-date with new releases
I also just added Latency and Throughput metrics because sometimes latency or throughput is just as important as intelligence.
I\u2019d love to hear what you think, especially if you spot any weird model matches (Unfortunately they still happen) or have ideas of what to add next ! I have a few ideas, like combining latency and throughput into one, or even intelligence, latency and throughput, I'll call it Wisdom :)
URL: https://the-frontier.app/
Thanks!","title":"Show HN: The Frontier, Tracking the LLM Pareto Frontier","updated_at":"2026-03-05T23:33:58Z","url":"https://the-frontier.app/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"pllu"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"Hi HN. AI benchmarks tell you performance metrics, but not how models actually behave in practice. Sites like LMArena and OpenRouter are great for testing prompts, but exploring multiple models takes time and effort.
Hard Prompts is a curated gallery of AI model responses to interesting questions. The goal is to help you quickly build an intuition for model personalities and capabilities. Seeing answers to questions like "What would you do if you were conscious?" or "Write a haiku that could not have been written by a human" really helps to illustrate the similarities and differences between model behaviour.
Four responses are currently generated per model to show how answers vary due to sampling randomness. Varied temperatures, extended thinking etc coming soon.
I'll be posting updates on X (@hardprompts) and have a roadmap for upcoming features at hardprompts.ai/roadmap.
Would love your feedback on:
- The overall concept and execution\n- Specific prompts you'd like to see added\n- Which models you'd prioritise for inclusion\n- New features
Thanks for taking a look!"},"title":{"matchLevel":"none","matchedWords":[],"value":"Show HN: Hard Prompts \u2013 Compare how AI models respond to interesting questions"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://hardprompts.ai/"}},"_tags":["story","author_pllu","story_45746893","show_hn"],"author":"pllu","created_at":"2025-10-29T13:56:33Z","created_at_i":1761746193,"num_comments":0,"objectID":"45746893","points":3,"story_id":45746893,"story_text":"Hi HN. AI benchmarks tell you performance metrics, but not how models actually behave in practice. Sites like LMArena and OpenRouter are great for testing prompts, but exploring multiple models takes time and effort.
Hard Prompts is a curated gallery of AI model responses to interesting questions. The goal is to help you quickly build an intuition for model personalities and capabilities. Seeing answers to questions like "What would you do if you were conscious?" or "Write a haiku that could not have been written by a human" really helps to illustrate the similarities and differences between model behaviour.
Four responses are currently generated per model to show how answers vary due to sampling randomness. Varied temperatures, extended thinking etc coming soon.
I'll be posting updates on X (@hardprompts) and have a roadmap for upcoming features at hardprompts.ai/roadmap.
Would love your feedback on:
- The overall concept and execution\n- Specific prompts you'd like to see added\n- Which models you'd prioritise for inclusion\n- New features
Thanks for taking a look!","title":"Show HN: Hard Prompts \u2013 Compare how AI models respond to interesting questions","updated_at":"2026-03-05T22:57:45Z","url":"https://hardprompts.ai/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"xjconlyme"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"I've iterated YPerf since last submission with features to make LLM selection more data-driven:
- Added benchmark rankings from lmarena.ai to quantify model intelligence\n- Built-in cost estimator to forecast average usage expenses\n- Real-time uptime monitoring across providers\n- Side-by-side provider comparisons for identical models
The goal is to help teams make more informed decisions when choosing LLMs by providing transparency around performance, reliability and costs."},"title":{"matchLevel":"none","matchedWords":[],"value":"Show HN: YPerf \u2013 Compare LLM models with performance/cost/uptime metrics"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://yperf.com"}},"_tags":["story","author_xjconlyme","story_42574040","show_hn"],"author":"xjconlyme","children":[42577198],"created_at":"2025-01-02T13:00:53Z","created_at_i":1735822853,"num_comments":2,"objectID":"42574040","points":2,"story_id":42574040,"story_text":"I've iterated YPerf since last submission with features to make LLM selection more data-driven:
- Added benchmark rankings from lmarena.ai to quantify model intelligence\n- Built-in cost estimator to forecast average usage expenses\n- Real-time uptime monitoring across providers\n- Side-by-side provider comparisons for identical models
The goal is to help teams make more informed decisions when choosing LLMs by providing transparency around performance, reliability and costs.","title":"Show HN: YPerf \u2013 Compare LLM models with performance/cost/uptime metrics","updated_at":"2025-01-03T02:05:14Z","url":"https://yperf.com"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"robinbanner"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"I got frustrated paying $60/M tokens for reasoning queries when a $0.80/M model gives comparable results for most of them. So I built Komilion \u2014 a model router that classifies each API request and routes it to a cheaper model that fits.
- Drop-in replacement for the OpenAI SDK (change one line: base_url)\n- Each query gets classified (regex fast path + lightweight LLM classifier) and matched against ~390 models\n- Three tiers (Frugal/Balanced/Premium) to control the quality-cost tradeoff\n- Automatic failover if a provider goes down\n- Cost metadata in every response
The routing logic is benchmark-driven (LMArena, Artificial Analysis), not ML-based \u2014 simpler to debug and reason about. The regex fast path handles ~60% of requests in under 5ms with zero API calls.
Example: a customer support bot doing 10K conversations/month went from ~$250/mo (everything pinned to Opus 4.6) to ~$40/mo with routing. Most conversations were FAQ-level questions that a smaller model handled fine.
Stack: Next.js, Vercel, Neon PostgreSQL, OpenRouter upstream. Hosting cost: ~$20/month.
We ran a head-to-head benchmark: same 15 prompts through Opus, GPT-4o, Gemini Pro, and the router. Simple tasks cost 66% less with routing. Complex tasks produced 2x more detailed output because the router picked specialized models per task type. Full data: https://dev.to/robinbanner/we-benchmarked-4-ai-api-strategie...
Architecture writeup: https://dev.to/robinbanner/inside-komilions-architecture-how... \u2014 there's a free tier if you want to try it."},"title":{"matchLevel":"none","matchedWords":[],"value":"Show HN: API router that picks the cheapest model that fits each query"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://www.komilion.com/"}},"_tags":["story","author_robinbanner","story_47036011","show_hn"],"author":"robinbanner","children":[47036018,47043320],"created_at":"2026-02-16T15:11:25Z","created_at_i":1771254685,"num_comments":1,"objectID":"47036011","points":1,"story_id":47036011,"story_text":"I got frustrated paying $60/M tokens for reasoning queries when a $0.80/M model gives comparable results for most of them. So I built Komilion \u2014 a model router that classifies each API request and routes it to a cheaper model that fits.
- Drop-in replacement for the OpenAI SDK (change one line: base_url)\n- Each query gets classified (regex fast path + lightweight LLM classifier) and matched against ~390 models\n- Three tiers (Frugal/Balanced/Premium) to control the quality-cost tradeoff\n- Automatic failover if a provider goes down\n- Cost metadata in every response
The routing logic is benchmark-driven (LMArena, Artificial Analysis), not ML-based \u2014 simpler to debug and reason about. The regex fast path handles ~60% of requests in under 5ms with zero API calls.
Example: a customer support bot doing 10K conversations/month went from ~$250/mo (everything pinned to Opus 4.6) to ~$40/mo with routing. Most conversations were FAQ-level questions that a smaller model handled fine.
Stack: Next.js, Vercel, Neon PostgreSQL, OpenRouter upstream. Hosting cost: ~$20/month.
We ran a head-to-head benchmark: same 15 prompts through Opus, GPT-4o, Gemini Pro, and the router. Simple tasks cost 66% less with routing. Complex tasks produced 2x more detailed output because the router picked specialized models per task type. Full data: https://dev.to/robinbanner/we-benchmarked-4-ai-api-strategie...
Architecture writeup: https://dev.to/robinbanner/inside-komilions-architecture-how... \u2014 there's a free tier if you want to try it.","title":"Show HN: API router that picks the cheapest model that fits each query","updated_at":"2026-03-05T23:33:47Z","url":"https://www.komilion.com/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"gptbased"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"gptbased joins LMArena rankings with live OpenRouter pricing. Daily snapshots.
Features:
- 8 LMArena categories: text, webdev, vision, image-gen, image-edit, and three video subsets\n- "Best value" picks via Pareto frontier in (cost, Elo) space, knee of the curve\n- Side-by-side compare\n- Email alerts when a new model enters\n- Free RapidAPI tier
What else do you want to see?"},"title":{"matchLevel":"none","matchedWords":[],"value":"Show HN: Gptbased \u2013 LLM leaderboard that emails you when to switch"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://gptbased.com"}},"_tags":["story","author_gptbased","story_48579719","show_hn"],"author":"gptbased","created_at":"2026-06-18T02:09:22Z","created_at_i":1781748562,"num_comments":0,"objectID":"48579719","points":1,"story_id":48579719,"story_text":"gptbased joins LMArena rankings with live OpenRouter pricing. Daily snapshots.
Features:
- 8 LMArena categories: text, webdev, vision, image-gen, image-edit, and three video subsets\n- "Best value" picks via Pareto frontier in (cost, Elo) space, knee of the curve\n- Side-by-side compare\n- Email alerts when a new model enters\n- Free RapidAPI tier
What else do you want to see?","title":"Show HN: Gptbased \u2013 LLM leaderboard that emails you when to switch","updated_at":"2026-06-18T02:26:26Z","url":"https://gptbased.com"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"countWSS"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"Like LM arena where users get two hidden LLMs(so e.g. AI song names)\nand decide if A or B is better(or tie).\nWhen user chooses A/B the songs names are revealed,\ntheir rating are updated and Top N songs are\ndisplayed on leaderboard."},"title":{"matchLevel":"none","matchedWords":[],"value":"Is there \"Song Arena\" type website?"}},"_tags":["story","author_countWSS","story_44020351","ask_hn"],"author":"countWSS","created_at":"2025-05-18T10:36:24Z","created_at_i":1747564584,"num_comments":0,"objectID":"44020351","points":1,"story_id":44020351,"story_text":"Like LM arena where users get two hidden LLMs(so e.g. AI song names)\nand decide if A or B is better(or tie).\nWhen user chooses A/B the songs names are revealed,\ntheir rating are updated and Top N songs are\ndisplayed on leaderboard.","title":"Is there \"Song Arena\" type website?","updated_at":"2025-05-18T10:41:42Z"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"wspittman"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"This simple game challenges you drag-and-drop a set of LLM models into the correct order (release date, input cost at launch, LM Arena score) using just their name.
It's no secret that LLM model names are a bit of a mess, but when OpenAI decided to backtrack from 4.5 to 4.1, well, I just couldn't let it go.
This was my first attempt at vibe coding something with VSCode Copilot Agent (w/ Claude 3.7 Sonnet). The agent got me like 95% functionality, but the moment I needed to crack open the code myself I couldn't handle the mess and had to clean it up. I am bad at letting go control enough for vibe coding. If anyone else has had that problem and gotten over it, I would love some helpful tips.
I collected the data primarily by digging through Simon Willison's blog archives and snapshotting LMArena's leaderboard. If you spot any inaccuracies in the model data, I'd appreciate corrections with sources!
Nothing fancy in the tech here, just a vanilla HTML/CSS/JS site hosted on GitHub Pages.\nGitHub link (MIT License): https://github.com/wspittman/BetterThan4ButNot5"},"title":{"matchLevel":"none","matchedWords":[],"value":"Show HN: Better Than 4 But Not 5 \u2013 An LLM model ordering challenge"},"url":{"matchLevel":"none","matchedWords":[],"value":"https://wspittman.github.io/BetterThan4ButNot5/"}},"_tags":["story","author_wspittman","story_43732009","show_hn"],"author":"wspittman","created_at":"2025-04-18T21:27:21Z","created_at_i":1745011641,"num_comments":0,"objectID":"43732009","points":1,"story_id":43732009,"story_text":"This simple game challenges you drag-and-drop a set of LLM models into the correct order (release date, input cost at launch, LM Arena score) using just their name.
It's no secret that LLM model names are a bit of a mess, but when OpenAI decided to backtrack from 4.5 to 4.1, well, I just couldn't let it go.
This was my first attempt at vibe coding something with VSCode Copilot Agent (w/ Claude 3.7 Sonnet). The agent got me like 95% functionality, but the moment I needed to crack open the code myself I couldn't handle the mess and had to clean it up. I am bad at letting go control enough for vibe coding. If anyone else has had that problem and gotten over it, I would love some helpful tips.
I collected the data primarily by digging through Simon Willison's blog archives and snapshotting LMArena's leaderboard. If you spot any inaccuracies in the model data, I'd appreciate corrections with sources!
Nothing fancy in the tech here, just a vanilla HTML/CSS/JS site hosted on GitHub Pages.\nGitHub link (MIT License): https://github.com/wspittman/BetterThan4ButNot5","title":"Show HN: Better Than 4 But Not 5 \u2013 An LLM model ordering challenge","updated_at":"2025-04-18T21:30:04Z","url":"https://wspittman.github.io/BetterThan4ButNot5/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"thomasbrd"},"story_text":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"Hi everyone,
Inspired by ChatBot Arena (https://lmarena.ai/), we've created a background removal comparator and need your honest feedback to create a community-driven benchmark.
Current Rankings:
PhotoRoom: Elo 1025
RemoveBG: Elo 1006
BRIA RMBG 2.0: Elo 969
This ranking is currently based on approximately 1,500 votes collected between December 1-10, 2024.
How It Works: Visit our Background Removal Arena <https://huggingface.co/spaces/bgsys/background-removal-arena> to see the tools in action and cast your votes. Your participation helps refine the Elo rankings and ensures they become more and more reliable.
How You Can Help:
Vote: Compare the tools and vote for your preferred option.\nFeedback: Share your experience with the voting process and the tool rankings.
Was voting easy?\nDo the rankings reflect your usage?\nAny suggestions for improvement?
Transparency: Our process is fully public and auditable. The code is available on Hugging Face <https://huggingface.co/spaces/bgsys/background-removal-arena>. Any feedback is welcome!
Your input is crucial in making this a valuable resource for everyone. Thank you for your support and participation!"},"title":{"matchLevel":"none","matchedWords":[],"value":"Help Us Rank the Best Background Removal Tools"}},"_tags":["story","author_thomasbrd","story_42386568","ask_hn"],"author":"thomasbrd","created_at":"2024-12-11T10:39:21Z","created_at_i":1733913561,"num_comments":0,"objectID":"42386568","points":1,"story_id":42386568,"story_text":"Hi everyone,
Inspired by ChatBot Arena (https://lmarena.ai/), we've created a background removal comparator and need your honest feedback to create a community-driven benchmark.
Current Rankings:
PhotoRoom: Elo 1025
RemoveBG: Elo 1006
BRIA RMBG 2.0: Elo 969
This ranking is currently based on approximately 1,500 votes collected between December 1-10, 2024.
How It Works: Visit our Background Removal Arena <https://huggingface.co/spaces/bgsys/background-removal-arena> to see the tools in action and cast your votes. Your participation helps refine the Elo rankings and ensures they become more and more reliable.
How You Can Help:
Vote: Compare the tools and vote for your preferred option.\nFeedback: Share your experience with the voting process and the tool rankings.
Was voting easy?\nDo the rankings reflect your usage?\nAny suggestions for improvement?
Transparency: Our process is fully public and auditable. The code is available on Hugging Face <https://huggingface.co/spaces/bgsys/background-removal-arena>. Any feedback is welcome!
Your input is crucial in making this a valuable resource for everyone. Thank you for your support and participation!","title":"Help Us Rank the Best Background Removal Tools","updated_at":"2024-12-11T10:44:40Z"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"zopper"},"title":{"matchLevel":"none","matchedWords":[],"value":"New Gemini model significantly outperforms others on Chatbot Arena (LMSYS)"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://lmarena.ai/"}},"_tags":["story","author_zopper","story_42341788"],"author":"zopper","children":[42342022,42344623,42344895,42345193,42350377,42354507],"created_at":"2024-12-06T17:18:25Z","created_at_i":1733505505,"num_comments":18,"objectID":"42341788","points":110,"story_id":42341788,"title":"New Gemini model significantly outperforms others on Chatbot Arena (LMSYS)","updated_at":"2025-11-10T20:40:58Z","url":"https://lmarena.ai/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"cynicalpeace"},"title":{"matchLevel":"none","matchedWords":[],"value":"Nvidia Outperforms GPT-4o with Open Source Model"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://github.com/lmarena/arena-hard-auto"}},"_tags":["story","author_cynicalpeace","story_41861570"],"author":"cynicalpeace","children":[41861571,41861613],"created_at":"2024-10-16T17:29:55Z","created_at_i":1729099795,"num_comments":3,"objectID":"41861570","points":23,"story_id":41861570,"title":"Nvidia Outperforms GPT-4o with Open Source Model","updated_at":"2024-11-04T18:21:56Z","url":"https://github.com/lmarena/arena-hard-auto"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"handfuloflight"},"title":{"matchLevel":"none","matchedWords":[],"value":"New Gemini model beats GPT 4o and Claude 3.5 Sonnet"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://lmarena.ai/?leaderboard"}},"_tags":["story","author_handfuloflight","story_42141241"],"author":"handfuloflight","children":[42153465],"created_at":"2024-11-14T21:08:02Z","created_at_i":1731618482,"num_comments":0,"objectID":"42141241","points":8,"story_id":42141241,"title":"New Gemini model beats GPT 4o and Claude 3.5 Sonnet","updated_at":"2024-11-17T23:13:00Z","url":"https://lmarena.ai/?leaderboard"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"leanvector"},"title":{"matchLevel":"none","matchedWords":[],"value":"Comparing 15 AI video models side-by-side using identical prompts"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://lmarena.ai/?chat-modality=video"}},"_tags":["story","author_leanvector","story_46711411"],"author":"leanvector","children":[46711412,46712771],"created_at":"2026-01-21T20:54:00Z","created_at_i":1769028840,"num_comments":3,"objectID":"46711411","points":6,"story_id":46711411,"title":"Comparing 15 AI video models side-by-side using identical prompts","updated_at":"2026-03-05T23:22:47Z","url":"https://lmarena.ai/?chat-modality=video"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"anjneymidha"},"title":{"matchLevel":"none","matchedWords":[],"value":"Grok3 is first model to surpass 1400 on the Chat Arena benchmark"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://twitter.com/lmarena_ai/status/1891706264800936307"}},"_tags":["story","author_anjneymidha","story_43086745"],"author":"anjneymidha","children":[43086794],"created_at":"2025-02-18T06:46:29Z","created_at_i":1739861189,"num_comments":3,"objectID":"43086745","points":5,"story_id":43086745,"title":"Grok3 is first model to surpass 1400 on the Chat Arena benchmark","updated_at":"2025-02-18T22:03:20Z","url":"https://twitter.com/lmarena_ai/status/1891706264800936307"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"tizkovatereza"},"title":{"matchLevel":"none","matchedWords":[],"value":"Compare sota LLMs on web dev tasks"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://web.lmarena.ai/"}},"_tags":["story","author_tizkovatereza","story_42388473"],"author":"tizkovatereza","children":[42388474],"created_at":"2024-12-11T15:08:07Z","created_at_i":1733929687,"num_comments":1,"objectID":"42388473","points":5,"story_id":42388473,"title":"Compare sota LLMs on web dev tasks","updated_at":"2024-12-11T16:14:16Z","url":"https://web.lmarena.ai/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"amrrs"},"title":{"matchLevel":"none","matchedWords":[],"value":"OpenAI reclaims the #1 spot, surpassing Gemini-Exp-1114"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://twitter.com/lmarena_ai/status/1859307979184689269"}},"_tags":["story","author_amrrs","story_42197143"],"author":"amrrs","children":[42197162],"created_at":"2024-11-20T19:13:46Z","created_at_i":1732130026,"num_comments":0,"objectID":"42197143","points":3,"story_id":42197143,"title":"OpenAI reclaims the #1 spot, surpassing Gemini-Exp-1114","updated_at":"2024-11-21T13:48:28Z","url":"https://twitter.com/lmarena_ai/status/1859307979184689269"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"robertwt7"},"title":{"matchLevel":"none","matchedWords":[],"value":"Gemini 2.5 Pro Still Tops Text and Vision Benchmarks"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://lmarena.ai/leaderboard"}},"_tags":["story","author_robertwt7","story_45650645"],"author":"robertwt7","created_at":"2025-10-20T23:21:26Z","created_at_i":1761002486,"num_comments":0,"objectID":"45650645","points":2,"story_id":45650645,"title":"Gemini 2.5 Pro Still Tops Text and Vision Benchmarks","updated_at":"2026-03-05T22:51:30Z","url":"https://lmarena.ai/leaderboard"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"doener"},"title":{"matchLevel":"none","matchedWords":[],"value":"AI Leaderboard Overview"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://lmarena.ai/leaderboard"}},"_tags":["story","author_doener","story_44861666"],"author":"doener","created_at":"2025-08-11T07:41:01Z","created_at_i":1754898061,"num_comments":0,"objectID":"44861666","points":2,"story_id":44861666,"title":"AI Leaderboard Overview","updated_at":"2026-03-05T22:32:48Z","url":"https://lmarena.ai/leaderboard"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"MukundMohanK"},"title":{"matchLevel":"none","matchedWords":[],"value":"Academic project turns into a $600M crowdsourced LLM benchmarking startup"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://www.bloomberg.com/news/articles/2025-05-21/lmarena-goes-from-academic-project-to-600-million-startup"}},"_tags":["story","author_MukundMohanK","story_44061708"],"author":"MukundMohanK","created_at":"2025-05-22T13:10:27Z","created_at_i":1747919427,"num_comments":0,"objectID":"44061708","points":2,"story_id":44061708,"title":"Academic project turns into a $600M crowdsourced LLM benchmarking startup","updated_at":"2025-05-22T14:17:58Z","url":"https://www.bloomberg.com/news/articles/2025-05-21/lmarena-goes-from-academic-project-to-600-million-startup"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"yenniejun111"},"title":{"matchLevel":"none","matchedWords":[],"value":"The Leaderboard Illusion"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://cohere.com/research/lmarena"}},"_tags":["story","author_yenniejun111","story_43968374"],"author":"yenniejun111","created_at":"2025-05-12T23:23:13Z","created_at_i":1747092193,"num_comments":0,"objectID":"43968374","points":2,"story_id":43968374,"title":"The Leaderboard Illusion","updated_at":"2025-05-12T23:28:03Z","url":"https://cohere.com/research/lmarena"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"croemer"},"title":{"matchLevel":"none","matchedWords":[],"value":"Copilot Arena"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://github.com/lmarena/copilot-arena"}},"_tags":["story","author_croemer","story_43848127"],"author":"croemer","created_at":"2025-04-30T17:11:30Z","created_at_i":1746033090,"num_comments":0,"objectID":"43848127","points":2,"story_id":43848127,"title":"Copilot Arena","updated_at":"2025-04-30T18:02:50Z","url":"https://github.com/lmarena/copilot-arena"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"ijidak"},"title":{"matchLevel":"none","matchedWords":[],"value":"AI Chatbot Leaderboard"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://lmarena.ai/?leaderboard"}},"_tags":["story","author_ijidak","story_43848113"],"author":"ijidak","created_at":"2025-04-30T17:10:40Z","created_at_i":1746033040,"num_comments":0,"objectID":"43848113","points":2,"story_id":43848113,"title":"AI Chatbot Leaderboard","updated_at":"2025-04-30T18:03:31Z","url":"https://lmarena.ai/?leaderboard"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"tosh"},"title":{"matchLevel":"none","matchedWords":[],"value":"WebDev Arena"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://blog.lmarena.ai/blog/2025/webdev-arena/"}},"_tags":["story","author_tosh","story_43325292"],"author":"tosh","created_at":"2025-03-10T19:58:24Z","created_at_i":1741636704,"num_comments":0,"objectID":"43325292","points":2,"story_id":43325292,"title":"WebDev Arena","updated_at":"2025-03-10T20:54:59Z","url":"https://blog.lmarena.ai/blog/2025/webdev-arena/"},{"_highlightResult":{"author":{"matchLevel":"none","matchedWords":[],"value":"simonpure"},"title":{"matchLevel":"none","matchedWords":[],"value":"P2L: Prompt-Based LLM Routing Wins Chatbot Arena #1"},"url":{"fullyHighlighted":false,"matchLevel":"full","matchedWords":["lmarena"],"value":"https://github.com/lmarena/p2l"}},"_tags":["story","author_simonpure","story_43254250"],"author":"simonpure","created_at":"2025-03-04T13:26:51Z","created_at_i":1741094811,"num_comments":0,"objectID":"43254250","points":2,"story_id":43254250,"title":"P2L: Prompt-Based LLM Routing Wins Chatbot Arena #1","updated_at":"2025-03-13T10:05:05Z","url":"https://github.com/lmarena/p2l"}],"hitsPerPage":50,"nbHits":458,"nbPages":10,"page":0,"params":"query=LMArena&tags=story&hitsPerPage=50&advancedSyntax=true&analyticsTags=backend","processingTimeMS":10,"processingTimingsMS":{"_request":{"roundTrip":18},"afterFetch":{"format":{"highlighting":1,"total":1},"merge":{"mergeLoop":{"prepareNextHit":2,"total":2},"total":3},"total":3},"fetch":{"query":5,"total":6},"total":10},"query":"LMArena","serverTimeMS":12}