[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"tool-benchlocal":3},{"tool":4,"categoryPool":30},{"id":5,"name":6,"slug":7,"description":8,"url":9,"githubUrl":10,"logoUrl":11,"openSource":12,"categories":13,"repo":18,"company":-1,"articles":28,"createdAt":29},"entity_01kzpjy8r7fvn8q0rcmcw8xg56","BenchLocal","benchlocal","Test LLMs on real tasks. Compare models side-by-side.","https:\u002F\u002Fbenchlocal.com\u002F","https:\u002F\u002Fgithub.com\u002Fstevibe\u002FBenchLocal","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fstevibe-cd02357b.png",true,[14],{"id":-1,"name":15,"slug":16,"count":17},"Evals","evals",0,{"fullName":19,"url":10,"stars":20,"forks":21,"openIssues":22,"language":23,"license":24,"topics":25,"pushedAt":26,"archived":27},"stevibe\u002FBenchLocal",402,45,12,"TypeScript","MIT",[],"2026-08-10T15:10:36Z",false,[],1786390717,[31,62,68,93],{"id":32,"name":33,"slug":34,"description":35,"url":36,"githubUrl":36,"logoUrl":37,"openSource":12,"categories":38,"repo":40,"company":54,"articles":60,"createdAt":61},"entity_01kzhe2n9gfagsad1ydt1e7gkn","AACR-Bench","aacr-bench","An Alibaba open-source multi-language benchmark for evaluating LLMs in repository-level automatic code review, featuring an AI-assisted and expert-verified dataset.","https:\u002F\u002Fgithub.com\u002Falibaba\u002Faacr-bench","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Falibaba-0f6c077a.png",[39],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":41,"url":36,"stars":42,"forks":43,"openIssues":44,"language":45,"license":46,"topics":47,"pushedAt":53,"archived":27},"alibaba\u002Faacr-bench",206,16,6,"Python","Apache-2.0",[48,49,50,51,52],"benchmark","code-review","multi-language","repository-level-context","software-engineering","2026-08-04T05:36:38Z",{"id":55,"name":56,"slug":57,"url":58,"bio":59,"githubOrg":57},"entity_01kzmeveq2f22bbn6jcepyspwk","Alibaba","alibaba","https:\u002F\u002Fwww.alibabagroup.com","Alibaba Open Source",[],1786217846,{"id":5,"name":6,"slug":7,"description":8,"url":9,"githubUrl":10,"logoUrl":11,"openSource":12,"categories":63,"repo":65,"company":-1,"articles":67,"createdAt":29},[64],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":19,"url":10,"stars":20,"forks":21,"openIssues":22,"language":23,"license":24,"topics":66,"pushedAt":26,"archived":27},[],[],{"id":69,"name":70,"slug":71,"description":72,"url":73,"githubUrl":-1,"logoUrl":74,"openSource":27,"categories":75,"repo":-1,"company":-1,"articles":77,"createdAt":92},"entity_01kyb2mdndeshatkngff89rdmc","Braintrust","braintrust","Ship quality agents at scale. Braintrust is the AI observability platform for tracing production, running evals, and catching regressions before they reach users.","https:\u002F\u002Fwww.braintrust.dev\u002F","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ficon180-ef918139.png",[76],{"id":-1,"name":15,"slug":16,"count":17},[78],{"id":79,"name":80,"slug":81,"summary":82,"url":83,"kind":84,"platform":85,"author":86,"authorHandle":87,"publisher":88,"publishedAt":89,"about":90,"writtenBy":91,"createdAt":-1},"entity_01kyb1ndrmf6esz3naek99dvk8","The context gold rush: Why everyone is building the same thing.","the-context-gold-rush-why-everyone-is-building-the-same-thing","You either die building product or live long enough to do context management.","https:\u002F\u002Fx.com\u002Fsamzliu\u002Fstatus\u002F2080210797465379147","blog","x","Sam Z Liu","samzliu","X","2026-07-23",[],[],1784930776,{"id":94,"name":95,"slug":96,"description":97,"url":98,"githubUrl":99,"logoUrl":100,"openSource":12,"categories":101,"repo":103,"company":-1,"articles":116,"createdAt":129},"entity_01kzsa6ykgextv2krhnxvx212d","Dynobox","dynobox","Cross-harness testing for multi-step agent flows","https:\u002F\u002Fdynobox.xyz","https:\u002F\u002Fgithub.com\u002Fdynobox\u002Fdynobox","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon-6008b356.svg",[102],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":104,"url":99,"stars":44,"forks":17,"openIssues":17,"language":23,"license":46,"topics":105,"pushedAt":115,"archived":27},"dynobox\u002Fdynobox",[106,107,108,109,110,111,16,112,113,114],"agent","agent-skills","claude-code","cli","codex","devtools","llm","opencode","testing","2026-08-12T16:11:17Z",[117],{"id":118,"name":119,"slug":120,"summary":121,"url":122,"kind":123,"platform":85,"author":124,"authorHandle":125,"publisher":88,"publishedAt":126,"about":127,"writtenBy":128,"createdAt":-1},"entity_01kzsa7tgbextv2kt75xknj5mg","Over engineering a pipeline to email me about strangers’ skills for dynobox","over-engineering-a-pipeline-to-email-me-about-strangers-skills-for-dynobox","For the past few weeks I've been building Dynobox - a local test runner that records what an agent does inside a harness (Claude Code, Codex, OpenCode etc.) and lets you write deterministic assertions against its non-deterministic behavior: tool calls, shell commands, files created or changed and so on.","https:\u002F\u002Fx.com\u002Fbhkdotdev\u002Fstatus\u002F2087281766440546814","article","bhk","bhkdotdev","2026-08-11",[],[],1786482227]