[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"tool-dynobox":3},{"tool":4,"categoryPool":51},{"id":5,"name":6,"slug":7,"description":8,"url":9,"githubUrl":10,"logoUrl":11,"openSource":12,"categories":13,"repo":18,"company":-1,"articles":35,"createdAt":50},"entity_01kzsa6ykgextv2krhnxvx212d","Dynobox","dynobox","Cross-harness testing for multi-step agent flows","https:\u002F\u002Fdynobox.xyz","https:\u002F\u002Fgithub.com\u002Fdynobox\u002Fdynobox","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon-6008b356.svg",true,[14],{"id":-1,"name":15,"slug":16,"count":17},"Evals","evals",0,{"fullName":19,"url":10,"stars":20,"forks":17,"openIssues":17,"language":21,"license":22,"topics":23,"pushedAt":33,"archived":34},"dynobox\u002Fdynobox",6,"TypeScript","Apache-2.0",[24,25,26,27,28,29,16,30,31,32],"agent","agent-skills","claude-code","cli","codex","devtools","llm","opencode","testing","2026-08-12T16:11:17Z",false,[36],{"id":37,"name":38,"slug":39,"summary":40,"url":41,"kind":42,"platform":43,"author":44,"authorHandle":45,"publisher":46,"publishedAt":47,"about":48,"writtenBy":49,"createdAt":-1},"entity_01kzsa7tgbextv2kt75xknj5mg","Over engineering a pipeline to email me about strangers’ skills for dynobox","over-engineering-a-pipeline-to-email-me-about-strangers-skills-for-dynobox","For the past few weeks I've been building Dynobox - a local test runner that records what an agent does inside a harness (Claude Code, Codex, OpenCode etc.) and lets you write deterministic assertions against its non-deterministic behavior: tool calls, shell commands, files created or changed and so on.","https:\u002F\u002Fx.com\u002Fbhkdotdev\u002Fstatus\u002F2087281766440546814","article","x","bhk","bhkdotdev","X","2026-08-11",[],[],1786482227,[52,81,101,124],{"id":53,"name":54,"slug":55,"description":56,"url":57,"githubUrl":57,"logoUrl":58,"openSource":12,"categories":59,"repo":61,"company":73,"articles":79,"createdAt":80},"entity_01kzhe2n9gfagsad1ydt1e7gkn","AACR-Bench","aacr-bench","An Alibaba open-source multi-language benchmark for evaluating LLMs in repository-level automatic code review, featuring an AI-assisted and expert-verified dataset.","https:\u002F\u002Fgithub.com\u002Falibaba\u002Faacr-bench","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Falibaba-0f6c077a.png",[60],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":62,"url":57,"stars":63,"forks":64,"openIssues":20,"language":65,"license":22,"topics":66,"pushedAt":72,"archived":34},"alibaba\u002Faacr-bench",206,16,"Python",[67,68,69,70,71],"benchmark","code-review","multi-language","repository-level-context","software-engineering","2026-08-04T05:36:38Z",{"id":74,"name":75,"slug":76,"url":77,"bio":78,"githubOrg":76},"entity_01kzmeveq2f22bbn6jcepyspwk","Alibaba","alibaba","https:\u002F\u002Fwww.alibabagroup.com","Alibaba Open Source",[],1786217846,{"id":82,"name":83,"slug":84,"description":85,"url":86,"githubUrl":87,"logoUrl":88,"openSource":12,"categories":89,"repo":91,"company":-1,"articles":99,"createdAt":100},"entity_01kzpjy8r7fvn8q0rcmcw8xg56","BenchLocal","benchlocal","Test LLMs on real tasks. Compare models side-by-side.","https:\u002F\u002Fbenchlocal.com\u002F","https:\u002F\u002Fgithub.com\u002Fstevibe\u002FBenchLocal","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fstevibe-cd02357b.png",[90],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":92,"url":87,"stars":93,"forks":94,"openIssues":95,"language":21,"license":96,"topics":97,"pushedAt":98,"archived":34},"stevibe\u002FBenchLocal",402,45,12,"MIT",[],"2026-08-10T15:10:36Z",[],1786390717,{"id":102,"name":103,"slug":104,"description":105,"url":106,"githubUrl":-1,"logoUrl":107,"openSource":34,"categories":108,"repo":-1,"company":-1,"articles":110,"createdAt":123},"entity_01kyb2mdndeshatkngff89rdmc","Braintrust","braintrust","Ship quality agents at scale. Braintrust is the AI observability platform for tracing production, running evals, and catching regressions before they reach users.","https:\u002F\u002Fwww.braintrust.dev\u002F","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ficon180-ef918139.png",[109],{"id":-1,"name":15,"slug":16,"count":17},[111],{"id":112,"name":113,"slug":114,"summary":115,"url":116,"kind":117,"platform":43,"author":118,"authorHandle":119,"publisher":46,"publishedAt":120,"about":121,"writtenBy":122,"createdAt":-1},"entity_01kyb1ndrmf6esz3naek99dvk8","The context gold rush: Why everyone is building the same thing.","the-context-gold-rush-why-everyone-is-building-the-same-thing","You either die building product or live long enough to do context management.","https:\u002F\u002Fx.com\u002Fsamzliu\u002Fstatus\u002F2080210797465379147","blog","Sam Z Liu","samzliu","2026-07-23",[],[],1784930776,{"id":5,"name":6,"slug":7,"description":8,"url":9,"githubUrl":10,"logoUrl":11,"openSource":12,"categories":125,"repo":127,"company":-1,"articles":129,"createdAt":50},[126],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":19,"url":10,"stars":20,"forks":17,"openIssues":17,"language":21,"license":22,"topics":128,"pushedAt":33,"archived":34},[24,25,26,27,28,29,16,30,31,32],[130],{"id":37,"name":38,"slug":39,"summary":40,"url":41,"kind":42,"platform":43,"author":44,"authorHandle":45,"publisher":46,"publishedAt":47,"about":131,"writtenBy":132,"createdAt":-1},[],[]]