[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"tool-ori-eval":3},{"tool":4,"categoryPool":39},{"id":5,"name":6,"slug":7,"description":8,"url":9,"githubUrl":-1,"logoUrl":10,"openSource":11,"categories":12,"repo":-1,"company":17,"articles":24,"createdAt":38},"entity_01kz4f26atesj8vra4778gn68r","Ori Eval","ori-eval","Find the best model for your project. Ori Eval runs your agent and model on your prompts, asserts on the tools it called, and grades the answers with an LLM judge, so you catch regressions and pick the best model before you ship.","https:\u002F\u002Fopenrouter.ai\u002Fori\u002Feval","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fglyph-9686a0ef.png",false,[13],{"id":-1,"name":14,"slug":15,"count":16},"Evals","evals",0,{"id":18,"name":19,"slug":20,"url":21,"bio":22,"githubOrg":23},"entity_01kynhnpt8ecztermr3mazqbw9","OpenRouter","openrouter","https:\u002F\u002Fopenrouter.ai\u002F","The unified interface for LLMs. Find the best models & prices for your prompts","OpenRouterTeam",[25],{"id":26,"name":27,"slug":28,"summary":29,"url":30,"kind":31,"platform":32,"author":33,"authorHandle":-1,"publisher":34,"publishedAt":35,"about":36,"writtenBy":37,"createdAt":-1},"entity_01kz4fb22fesj8vrb2x82ymqfv","Ori Eval: Find the Best Model for What You're Building","ori-eval-find-the-best-model-for-what-you-re-building","Ori Eval runs your agent on your own prompts, asserts on the tools it called, and grades answers with an LLM judge. Find the best model for what you're building.","https:\u002F\u002Fopenrouter.ai\u002Fblog\u002Fannouncements\u002Fori-eval\u002F","announcement","web","Jacky Liang","OpenRouter Blog","2026-08-03",[],[],1785782671,[40,72,93,118],{"id":41,"name":42,"slug":43,"description":44,"url":45,"githubUrl":45,"logoUrl":46,"openSource":47,"categories":48,"repo":50,"company":64,"articles":70,"createdAt":71},"entity_01kzhe2n9gfagsad1ydt1e7gkn","AACR-Bench","aacr-bench","An Alibaba open-source multi-language benchmark for evaluating LLMs in repository-level automatic code review, featuring an AI-assisted and expert-verified dataset.","https:\u002F\u002Fgithub.com\u002Falibaba\u002Faacr-bench","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Falibaba-0f6c077a.png",true,[49],{"id":-1,"name":14,"slug":15,"count":16},{"fullName":51,"url":45,"stars":52,"forks":53,"openIssues":54,"language":55,"license":56,"topics":57,"pushedAt":63,"archived":11},"alibaba\u002Faacr-bench",206,16,6,"Python","Apache-2.0",[58,59,60,61,62],"benchmark","code-review","multi-language","repository-level-context","software-engineering","2026-08-04T05:36:38Z",{"id":65,"name":66,"slug":67,"url":68,"bio":69,"githubOrg":67},"entity_01kzmeveq2f22bbn6jcepyspwk","Alibaba","alibaba","https:\u002F\u002Fwww.alibabagroup.com","Alibaba Open Source",[],1786217846,{"id":73,"name":74,"slug":75,"description":76,"url":77,"githubUrl":78,"logoUrl":79,"openSource":47,"categories":80,"repo":82,"company":-1,"articles":91,"createdAt":92},"entity_01kzpjy8r7fvn8q0rcmcw8xg56","BenchLocal","benchlocal","Test LLMs on real tasks. Compare models side-by-side.","https:\u002F\u002Fbenchlocal.com\u002F","https:\u002F\u002Fgithub.com\u002Fstevibe\u002FBenchLocal","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fstevibe-cd02357b.png",[81],{"id":-1,"name":14,"slug":15,"count":16},{"fullName":83,"url":78,"stars":84,"forks":85,"openIssues":86,"language":87,"license":88,"topics":89,"pushedAt":90,"archived":11},"stevibe\u002FBenchLocal",402,45,12,"TypeScript","MIT",[],"2026-08-10T15:10:36Z",[],1786390717,{"id":94,"name":95,"slug":96,"description":97,"url":98,"githubUrl":-1,"logoUrl":99,"openSource":11,"categories":100,"repo":-1,"company":-1,"articles":102,"createdAt":117},"entity_01kyb2mdndeshatkngff89rdmc","Braintrust","braintrust","Ship quality agents at scale. Braintrust is the AI observability platform for tracing production, running evals, and catching regressions before they reach users.","https:\u002F\u002Fwww.braintrust.dev\u002F","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ficon180-ef918139.png",[101],{"id":-1,"name":14,"slug":15,"count":16},[103],{"id":104,"name":105,"slug":106,"summary":107,"url":108,"kind":109,"platform":110,"author":111,"authorHandle":112,"publisher":113,"publishedAt":114,"about":115,"writtenBy":116,"createdAt":-1},"entity_01kyb1ndrmf6esz3naek99dvk8","The context gold rush: Why everyone is building the same thing.","the-context-gold-rush-why-everyone-is-building-the-same-thing","You either die building product or live long enough to do context management.","https:\u002F\u002Fx.com\u002Fsamzliu\u002Fstatus\u002F2080210797465379147","blog","x","Sam Z Liu","samzliu","X","2026-07-23",[],[],1784930776,{"id":119,"name":120,"slug":121,"description":122,"url":123,"githubUrl":124,"logoUrl":125,"openSource":47,"categories":126,"repo":128,"company":-1,"articles":141,"createdAt":154},"entity_01kzsa6ykgextv2krhnxvx212d","Dynobox","dynobox","Cross-harness testing for multi-step agent flows","https:\u002F\u002Fdynobox.xyz","https:\u002F\u002Fgithub.com\u002Fdynobox\u002Fdynobox","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon-6008b356.svg",[127],{"id":-1,"name":14,"slug":15,"count":16},{"fullName":129,"url":124,"stars":54,"forks":16,"openIssues":16,"language":87,"license":56,"topics":130,"pushedAt":140,"archived":11},"dynobox\u002Fdynobox",[131,132,133,134,135,136,15,137,138,139],"agent","agent-skills","claude-code","cli","codex","devtools","llm","opencode","testing","2026-08-12T16:11:17Z",[142],{"id":143,"name":144,"slug":145,"summary":146,"url":147,"kind":148,"platform":110,"author":149,"authorHandle":150,"publisher":113,"publishedAt":151,"about":152,"writtenBy":153,"createdAt":-1},"entity_01kzsa7tgbextv2kt75xknj5mg","Over engineering a pipeline to email me about strangers’ skills for dynobox","over-engineering-a-pipeline-to-email-me-about-strangers-skills-for-dynobox","For the past few weeks I've been building Dynobox - a local test runner that records what an agent does inside a harness (Claude Code, Codex, OpenCode etc.) and lets you write deterministic assertions against its non-deterministic behavior: tool calls, shell commands, files created or changed and so on.","https:\u002F\u002Fx.com\u002Fbhkdotdev\u002Fstatus\u002F2087281766440546814","article","bhk","bhkdotdev","2026-08-11",[],[],1786482227]