[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"tool-evalite":3},{"tool":4,"categoryPool":32},{"id":5,"name":6,"slug":7,"description":8,"url":9,"githubUrl":10,"logoUrl":11,"openSource":12,"categories":13,"repo":18,"company":-1,"articles":30,"createdAt":31},"entity_01kwn4p7mgf25rdkzn3jks64q9","Evalite","evalite","Evalite makes evals simple. Test your AI-powered apps with a local dev server.","https:\u002F\u002Fwww.evalite.dev\u002F","https:\u002F\u002Fgithub.com\u002Fmattpocock\u002Fevalite","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Flogo-dark.-5xXVk5Z-d08b964a.svg",true,[14],{"id":-1,"name":15,"slug":16,"count":17},"Evals","evals",0,{"fullName":19,"url":10,"stars":20,"forks":21,"openIssues":22,"language":23,"license":24,"topics":25,"pushedAt":28,"archived":29},"mattpocock\u002Fevalite",1651,101,63,"TypeScript","MIT",[26,16,27],"ai","typescript","2026-04-28T18:31:29Z",false,[],1783120993,[33,64,83,108],{"id":34,"name":35,"slug":36,"description":37,"url":38,"githubUrl":38,"logoUrl":39,"openSource":12,"categories":40,"repo":42,"company":56,"articles":62,"createdAt":63},"entity_01kzhe2n9gfagsad1ydt1e7gkn","AACR-Bench","aacr-bench","An Alibaba open-source multi-language benchmark for evaluating LLMs in repository-level automatic code review, featuring an AI-assisted and expert-verified dataset.","https:\u002F\u002Fgithub.com\u002Falibaba\u002Faacr-bench","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Falibaba-0f6c077a.png",[41],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":43,"url":38,"stars":44,"forks":45,"openIssues":46,"language":47,"license":48,"topics":49,"pushedAt":55,"archived":29},"alibaba\u002Faacr-bench",206,16,6,"Python","Apache-2.0",[50,51,52,53,54],"benchmark","code-review","multi-language","repository-level-context","software-engineering","2026-08-04T05:36:38Z",{"id":57,"name":58,"slug":59,"url":60,"bio":61,"githubOrg":59},"entity_01kzmeveq2f22bbn6jcepyspwk","Alibaba","alibaba","https:\u002F\u002Fwww.alibabagroup.com","Alibaba Open Source",[],1786217846,{"id":65,"name":66,"slug":67,"description":68,"url":69,"githubUrl":70,"logoUrl":71,"openSource":12,"categories":72,"repo":74,"company":-1,"articles":81,"createdAt":82},"entity_01kzpjy8r7fvn8q0rcmcw8xg56","BenchLocal","benchlocal","Test LLMs on real tasks. Compare models side-by-side.","https:\u002F\u002Fbenchlocal.com\u002F","https:\u002F\u002Fgithub.com\u002Fstevibe\u002FBenchLocal","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fstevibe-cd02357b.png",[73],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":75,"url":70,"stars":76,"forks":77,"openIssues":78,"language":23,"license":24,"topics":79,"pushedAt":80,"archived":29},"stevibe\u002FBenchLocal",402,45,12,[],"2026-08-10T15:10:36Z",[],1786390717,{"id":84,"name":85,"slug":86,"description":87,"url":88,"githubUrl":-1,"logoUrl":89,"openSource":29,"categories":90,"repo":-1,"company":-1,"articles":92,"createdAt":107},"entity_01kyb2mdndeshatkngff89rdmc","Braintrust","braintrust","Ship quality agents at scale. Braintrust is the AI observability platform for tracing production, running evals, and catching regressions before they reach users.","https:\u002F\u002Fwww.braintrust.dev\u002F","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ficon180-ef918139.png",[91],{"id":-1,"name":15,"slug":16,"count":17},[93],{"id":94,"name":95,"slug":96,"summary":97,"url":98,"kind":99,"platform":100,"author":101,"authorHandle":102,"publisher":103,"publishedAt":104,"about":105,"writtenBy":106,"createdAt":-1},"entity_01kyb1ndrmf6esz3naek99dvk8","The context gold rush: Why everyone is building the same thing.","the-context-gold-rush-why-everyone-is-building-the-same-thing","You either die building product or live long enough to do context management.","https:\u002F\u002Fx.com\u002Fsamzliu\u002Fstatus\u002F2080210797465379147","blog","x","Sam Z Liu","samzliu","X","2026-07-23",[],[],1784930776,{"id":109,"name":110,"slug":111,"description":112,"url":113,"githubUrl":114,"logoUrl":115,"openSource":12,"categories":116,"repo":118,"company":-1,"articles":131,"createdAt":144},"entity_01kzsa6ykgextv2krhnxvx212d","Dynobox","dynobox","Cross-harness testing for multi-step agent flows","https:\u002F\u002Fdynobox.xyz","https:\u002F\u002Fgithub.com\u002Fdynobox\u002Fdynobox","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon-6008b356.svg",[117],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":119,"url":114,"stars":46,"forks":17,"openIssues":17,"language":23,"license":48,"topics":120,"pushedAt":130,"archived":29},"dynobox\u002Fdynobox",[121,122,123,124,125,126,16,127,128,129],"agent","agent-skills","claude-code","cli","codex","devtools","llm","opencode","testing","2026-08-12T16:11:17Z",[132],{"id":133,"name":134,"slug":135,"summary":136,"url":137,"kind":138,"platform":100,"author":139,"authorHandle":140,"publisher":103,"publishedAt":141,"about":142,"writtenBy":143,"createdAt":-1},"entity_01kzsa7tgbextv2kt75xknj5mg","Over engineering a pipeline to email me about strangers’ skills for dynobox","over-engineering-a-pipeline-to-email-me-about-strangers-skills-for-dynobox","For the past few weeks I've been building Dynobox - a local test runner that records what an agent does inside a harness (Claude Code, Codex, OpenCode etc.) and lets you write deterministic assertions against its non-deterministic behavior: tool calls, shell commands, files created or changed and so on.","https:\u002F\u002Fx.com\u002Fbhkdotdev\u002Fstatus\u002F2087281766440546814","article","bhk","bhkdotdev","2026-08-11",[],[],1786482227]