[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"tool-r0b0bench":3},{"tool":4,"categoryPool":37},{"id":5,"name":6,"slug":6,"description":7,"url":8,"githubUrl":8,"logoUrl":-1,"openSource":9,"categories":10,"repo":15,"company":29,"articles":35,"createdAt":36},"entity_01kz27j47xe03rzbrt16373wbn","r0b0bench","Specification for a reproducible, provenance-bound multi-lane LLM benchmark suite.","https:\u002F\u002Fgithub.com\u002Fr0b0tlab\u002Fr0b0bench",true,[11],{"id":-1,"name":12,"slug":13,"count":14},"Evals","evals",0,{"fullName":16,"url":8,"stars":17,"forks":14,"openIssues":14,"language":18,"license":19,"topics":20,"pushedAt":27,"archived":28},"r0b0tlab\u002Fr0b0bench",10,"Python","MIT",[21,22,23,24,25,26],"benchmarking","bfcl","humaneval","ifeval","llm-benchmark","reproducibility","2026-08-11T13:40:16Z",false,{"id":30,"name":31,"slug":32,"url":8,"bio":33,"githubOrg":34},"entity_01kzmew0g8e8krkm5syb752j5n","mr-r0b0t - r0b0tlab","mr-r0b0t-r0b0tlab","OSS r0b0tlab projects for the community.","r0b0tlab",[],1785707696,[38,68,88,113],{"id":39,"name":40,"slug":41,"description":42,"url":43,"githubUrl":43,"logoUrl":44,"openSource":9,"categories":45,"repo":47,"company":60,"articles":66,"createdAt":67},"entity_01kzhe2n9gfagsad1ydt1e7gkn","AACR-Bench","aacr-bench","An Alibaba open-source multi-language benchmark for evaluating LLMs in repository-level automatic code review, featuring an AI-assisted and expert-verified dataset.","https:\u002F\u002Fgithub.com\u002Falibaba\u002Faacr-bench","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Falibaba-0f6c077a.png",[46],{"id":-1,"name":12,"slug":13,"count":14},{"fullName":48,"url":43,"stars":49,"forks":50,"openIssues":51,"language":18,"license":52,"topics":53,"pushedAt":59,"archived":28},"alibaba\u002Faacr-bench",206,16,6,"Apache-2.0",[54,55,56,57,58],"benchmark","code-review","multi-language","repository-level-context","software-engineering","2026-08-04T05:36:38Z",{"id":61,"name":62,"slug":63,"url":64,"bio":65,"githubOrg":63},"entity_01kzmeveq2f22bbn6jcepyspwk","Alibaba","alibaba","https:\u002F\u002Fwww.alibabagroup.com","Alibaba Open Source",[],1786217846,{"id":69,"name":70,"slug":71,"description":72,"url":73,"githubUrl":74,"logoUrl":75,"openSource":9,"categories":76,"repo":78,"company":-1,"articles":86,"createdAt":87},"entity_01kzpjy8r7fvn8q0rcmcw8xg56","BenchLocal","benchlocal","Test LLMs on real tasks. Compare models side-by-side.","https:\u002F\u002Fbenchlocal.com\u002F","https:\u002F\u002Fgithub.com\u002Fstevibe\u002FBenchLocal","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fstevibe-cd02357b.png",[77],{"id":-1,"name":12,"slug":13,"count":14},{"fullName":79,"url":74,"stars":80,"forks":81,"openIssues":82,"language":83,"license":19,"topics":84,"pushedAt":85,"archived":28},"stevibe\u002FBenchLocal",402,45,12,"TypeScript",[],"2026-08-10T15:10:36Z",[],1786390717,{"id":89,"name":90,"slug":91,"description":92,"url":93,"githubUrl":-1,"logoUrl":94,"openSource":28,"categories":95,"repo":-1,"company":-1,"articles":97,"createdAt":112},"entity_01kyb2mdndeshatkngff89rdmc","Braintrust","braintrust","Ship quality agents at scale. Braintrust is the AI observability platform for tracing production, running evals, and catching regressions before they reach users.","https:\u002F\u002Fwww.braintrust.dev\u002F","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ficon180-ef918139.png",[96],{"id":-1,"name":12,"slug":13,"count":14},[98],{"id":99,"name":100,"slug":101,"summary":102,"url":103,"kind":104,"platform":105,"author":106,"authorHandle":107,"publisher":108,"publishedAt":109,"about":110,"writtenBy":111,"createdAt":-1},"entity_01kyb1ndrmf6esz3naek99dvk8","The context gold rush: Why everyone is building the same thing.","the-context-gold-rush-why-everyone-is-building-the-same-thing","You either die building product or live long enough to do context management.","https:\u002F\u002Fx.com\u002Fsamzliu\u002Fstatus\u002F2080210797465379147","blog","x","Sam Z Liu","samzliu","X","2026-07-23",[],[],1784930776,{"id":114,"name":115,"slug":116,"description":117,"url":118,"githubUrl":119,"logoUrl":120,"openSource":9,"categories":121,"repo":123,"company":-1,"articles":136,"createdAt":149},"entity_01kzsa6ykgextv2krhnxvx212d","Dynobox","dynobox","Cross-harness testing for multi-step agent flows","https:\u002F\u002Fdynobox.xyz","https:\u002F\u002Fgithub.com\u002Fdynobox\u002Fdynobox","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon-6008b356.svg",[122],{"id":-1,"name":12,"slug":13,"count":14},{"fullName":124,"url":119,"stars":51,"forks":14,"openIssues":14,"language":83,"license":52,"topics":125,"pushedAt":135,"archived":28},"dynobox\u002Fdynobox",[126,127,128,129,130,131,13,132,133,134],"agent","agent-skills","claude-code","cli","codex","devtools","llm","opencode","testing","2026-08-12T16:11:17Z",[137],{"id":138,"name":139,"slug":140,"summary":141,"url":142,"kind":143,"platform":105,"author":144,"authorHandle":145,"publisher":108,"publishedAt":146,"about":147,"writtenBy":148,"createdAt":-1},"entity_01kzsa7tgbextv2kt75xknj5mg","Over engineering a pipeline to email me about strangers’ skills for dynobox","over-engineering-a-pipeline-to-email-me-about-strangers-skills-for-dynobox","For the past few weeks I've been building Dynobox - a local test runner that records what an agent does inside a harness (Claude Code, Codex, OpenCode etc.) and lets you write deterministic assertions against its non-deterministic behavior: tool calls, shell commands, files created or changed and so on.","https:\u002F\u002Fx.com\u002Fbhkdotdev\u002Fstatus\u002F2087281766440546814","article","bhk","bhkdotdev","2026-08-11",[],[],1786482227]