[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"tool-vulcanbench":3},{"tool":4,"categoryPool":28},{"id":5,"name":6,"slug":7,"description":8,"url":9,"githubUrl":9,"logoUrl":-1,"openSource":10,"categories":11,"repo":16,"company":-1,"articles":26,"createdAt":27},"entity_01kye8h19gfaj889yd04zx43dn","VulcanBench","vulcanbench","Open source, clear, transparent, real world llm benchmarks","https:\u002F\u002Fgithub.com\u002Fmorganlinton\u002FVulcanBench",true,[12],{"id":-1,"name":13,"slug":14,"count":15},"Evals","evals",0,{"fullName":17,"url":9,"stars":18,"forks":19,"openIssues":20,"language":21,"license":22,"topics":23,"pushedAt":24,"archived":25},"morganlinton\u002FVulcanBench",57,5,11,"Python","Apache-2.0",[],"2026-08-10T03:02:19Z",false,[],1785037620,[29,58,79,104],{"id":30,"name":31,"slug":32,"description":33,"url":34,"githubUrl":34,"logoUrl":35,"openSource":10,"categories":36,"repo":38,"company":50,"articles":56,"createdAt":57},"entity_01kzhe2n9gfagsad1ydt1e7gkn","AACR-Bench","aacr-bench","An Alibaba open-source multi-language benchmark for evaluating LLMs in repository-level automatic code review, featuring an AI-assisted and expert-verified dataset.","https:\u002F\u002Fgithub.com\u002Falibaba\u002Faacr-bench","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Falibaba-0f6c077a.png",[37],{"id":-1,"name":13,"slug":14,"count":15},{"fullName":39,"url":34,"stars":40,"forks":41,"openIssues":42,"language":21,"license":22,"topics":43,"pushedAt":49,"archived":25},"alibaba\u002Faacr-bench",206,16,6,[44,45,46,47,48],"benchmark","code-review","multi-language","repository-level-context","software-engineering","2026-08-04T05:36:38Z",{"id":51,"name":52,"slug":53,"url":54,"bio":55,"githubOrg":53},"entity_01kzmeveq2f22bbn6jcepyspwk","Alibaba","alibaba","https:\u002F\u002Fwww.alibabagroup.com","Alibaba Open Source",[],1786217846,{"id":59,"name":60,"slug":61,"description":62,"url":63,"githubUrl":64,"logoUrl":65,"openSource":10,"categories":66,"repo":68,"company":-1,"articles":77,"createdAt":78},"entity_01kzpjy8r7fvn8q0rcmcw8xg56","BenchLocal","benchlocal","Test LLMs on real tasks. Compare models side-by-side.","https:\u002F\u002Fbenchlocal.com\u002F","https:\u002F\u002Fgithub.com\u002Fstevibe\u002FBenchLocal","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fstevibe-cd02357b.png",[67],{"id":-1,"name":13,"slug":14,"count":15},{"fullName":69,"url":64,"stars":70,"forks":71,"openIssues":72,"language":73,"license":74,"topics":75,"pushedAt":76,"archived":25},"stevibe\u002FBenchLocal",402,45,12,"TypeScript","MIT",[],"2026-08-10T15:10:36Z",[],1786390717,{"id":80,"name":81,"slug":82,"description":83,"url":84,"githubUrl":-1,"logoUrl":85,"openSource":25,"categories":86,"repo":-1,"company":-1,"articles":88,"createdAt":103},"entity_01kyb2mdndeshatkngff89rdmc","Braintrust","braintrust","Ship quality agents at scale. Braintrust is the AI observability platform for tracing production, running evals, and catching regressions before they reach users.","https:\u002F\u002Fwww.braintrust.dev\u002F","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ficon180-ef918139.png",[87],{"id":-1,"name":13,"slug":14,"count":15},[89],{"id":90,"name":91,"slug":92,"summary":93,"url":94,"kind":95,"platform":96,"author":97,"authorHandle":98,"publisher":99,"publishedAt":100,"about":101,"writtenBy":102,"createdAt":-1},"entity_01kyb1ndrmf6esz3naek99dvk8","The context gold rush: Why everyone is building the same thing.","the-context-gold-rush-why-everyone-is-building-the-same-thing","You either die building product or live long enough to do context management.","https:\u002F\u002Fx.com\u002Fsamzliu\u002Fstatus\u002F2080210797465379147","blog","x","Sam Z Liu","samzliu","X","2026-07-23",[],[],1784930776,{"id":105,"name":106,"slug":107,"description":108,"url":109,"githubUrl":110,"logoUrl":111,"openSource":10,"categories":112,"repo":114,"company":-1,"articles":127,"createdAt":140},"entity_01kzsa6ykgextv2krhnxvx212d","Dynobox","dynobox","Cross-harness testing for multi-step agent flows","https:\u002F\u002Fdynobox.xyz","https:\u002F\u002Fgithub.com\u002Fdynobox\u002Fdynobox","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon-6008b356.svg",[113],{"id":-1,"name":13,"slug":14,"count":15},{"fullName":115,"url":110,"stars":42,"forks":15,"openIssues":15,"language":73,"license":22,"topics":116,"pushedAt":126,"archived":25},"dynobox\u002Fdynobox",[117,118,119,120,121,122,14,123,124,125],"agent","agent-skills","claude-code","cli","codex","devtools","llm","opencode","testing","2026-08-12T16:11:17Z",[128],{"id":129,"name":130,"slug":131,"summary":132,"url":133,"kind":134,"platform":96,"author":135,"authorHandle":136,"publisher":99,"publishedAt":137,"about":138,"writtenBy":139,"createdAt":-1},"entity_01kzsa7tgbextv2kt75xknj5mg","Over engineering a pipeline to email me about strangers’ skills for dynobox","over-engineering-a-pipeline-to-email-me-about-strangers-skills-for-dynobox","For the past few weeks I've been building Dynobox - a local test runner that records what an agent does inside a harness (Claude Code, Codex, OpenCode etc.) and lets you write deterministic assertions against its non-deterministic behavior: tool calls, shell commands, files created or changed and so on.","https:\u002F\u002Fx.com\u002Fbhkdotdev\u002Fstatus\u002F2087281766440546814","article","bhk","bhkdotdev","2026-08-11",[],[],1786482227]