[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"tool-terminal-bench":3},{"tool":4,"categoryPool":44},{"id":5,"name":6,"slug":7,"description":8,"url":9,"githubUrl":10,"logoUrl":11,"openSource":12,"categories":13,"repo":18,"company":-1,"articles":28,"createdAt":43},"entity_01kz0hp01wfk3akadn0cmg1k41","Terminal-Bench","terminal-bench","A benchmark for LLMs on complicated tasks in the terminal","https:\u002F\u002Fwww.tbench.ai","https:\u002F\u002Fgithub.com\u002Fharbor-framework\u002Fterminal-bench","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon-f3479231.ico",true,[14],{"id":-1,"name":15,"slug":16,"count":17},"Evals","evals",0,{"fullName":19,"url":10,"stars":20,"forks":21,"openIssues":22,"language":23,"license":24,"topics":25,"pushedAt":26,"archived":27},"harbor-framework\u002Fterminal-bench",486,371,123,"Python","Apache-2.0",[],"2026-08-13T07:37:10Z",false,[29],{"id":30,"name":31,"slug":32,"summary":33,"url":34,"kind":35,"platform":36,"author":37,"authorHandle":38,"publisher":39,"publishedAt":40,"about":41,"writtenBy":42,"createdAt":-1},"entity_01kz0hfjg5e8jae1jdrae7xnw4","Good Benchmarks","good-benchmarks","Benchmarks are where SOTA has to earn its name. This post is about designing good tasks.","https:\u002F\u002Fx.com\u002Fneversupervised\u002Fstatus\u002F2075432858270003462","article","x","Ivan Bercovich","neversupervised","X","2026-07-10",[],[],1785651200,[45,74,95,118],{"id":46,"name":47,"slug":48,"description":49,"url":50,"githubUrl":50,"logoUrl":51,"openSource":12,"categories":52,"repo":54,"company":66,"articles":72,"createdAt":73},"entity_01kzhe2n9gfagsad1ydt1e7gkn","AACR-Bench","aacr-bench","An Alibaba open-source multi-language benchmark for evaluating LLMs in repository-level automatic code review, featuring an AI-assisted and expert-verified dataset.","https:\u002F\u002Fgithub.com\u002Falibaba\u002Faacr-bench","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Falibaba-0f6c077a.png",[53],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":55,"url":50,"stars":56,"forks":57,"openIssues":58,"language":23,"license":24,"topics":59,"pushedAt":65,"archived":27},"alibaba\u002Faacr-bench",206,16,6,[60,61,62,63,64],"benchmark","code-review","multi-language","repository-level-context","software-engineering","2026-08-04T05:36:38Z",{"id":67,"name":68,"slug":69,"url":70,"bio":71,"githubOrg":69},"entity_01kzmeveq2f22bbn6jcepyspwk","Alibaba","alibaba","https:\u002F\u002Fwww.alibabagroup.com","Alibaba Open Source",[],1786217846,{"id":75,"name":76,"slug":77,"description":78,"url":79,"githubUrl":80,"logoUrl":81,"openSource":12,"categories":82,"repo":84,"company":-1,"articles":93,"createdAt":94},"entity_01kzpjy8r7fvn8q0rcmcw8xg56","BenchLocal","benchlocal","Test LLMs on real tasks. Compare models side-by-side.","https:\u002F\u002Fbenchlocal.com\u002F","https:\u002F\u002Fgithub.com\u002Fstevibe\u002FBenchLocal","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fstevibe-cd02357b.png",[83],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":85,"url":80,"stars":86,"forks":87,"openIssues":88,"language":89,"license":90,"topics":91,"pushedAt":92,"archived":27},"stevibe\u002FBenchLocal",402,45,12,"TypeScript","MIT",[],"2026-08-10T15:10:36Z",[],1786390717,{"id":96,"name":97,"slug":98,"description":99,"url":100,"githubUrl":-1,"logoUrl":101,"openSource":27,"categories":102,"repo":-1,"company":-1,"articles":104,"createdAt":117},"entity_01kyb2mdndeshatkngff89rdmc","Braintrust","braintrust","Ship quality agents at scale. Braintrust is the AI observability platform for tracing production, running evals, and catching regressions before they reach users.","https:\u002F\u002Fwww.braintrust.dev\u002F","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ficon180-ef918139.png",[103],{"id":-1,"name":15,"slug":16,"count":17},[105],{"id":106,"name":107,"slug":108,"summary":109,"url":110,"kind":111,"platform":36,"author":112,"authorHandle":113,"publisher":39,"publishedAt":114,"about":115,"writtenBy":116,"createdAt":-1},"entity_01kyb1ndrmf6esz3naek99dvk8","The context gold rush: Why everyone is building the same thing.","the-context-gold-rush-why-everyone-is-building-the-same-thing","You either die building product or live long enough to do context management.","https:\u002F\u002Fx.com\u002Fsamzliu\u002Fstatus\u002F2080210797465379147","blog","Sam Z Liu","samzliu","2026-07-23",[],[],1784930776,{"id":119,"name":120,"slug":121,"description":122,"url":123,"githubUrl":124,"logoUrl":125,"openSource":12,"categories":126,"repo":128,"company":-1,"articles":141,"createdAt":153},"entity_01kzsa6ykgextv2krhnxvx212d","Dynobox","dynobox","Cross-harness testing for multi-step agent flows","https:\u002F\u002Fdynobox.xyz","https:\u002F\u002Fgithub.com\u002Fdynobox\u002Fdynobox","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon-6008b356.svg",[127],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":129,"url":124,"stars":58,"forks":17,"openIssues":17,"language":89,"license":24,"topics":130,"pushedAt":140,"archived":27},"dynobox\u002Fdynobox",[131,132,133,134,135,136,16,137,138,139],"agent","agent-skills","claude-code","cli","codex","devtools","llm","opencode","testing","2026-08-12T16:11:17Z",[142],{"id":143,"name":144,"slug":145,"summary":146,"url":147,"kind":35,"platform":36,"author":148,"authorHandle":149,"publisher":39,"publishedAt":150,"about":151,"writtenBy":152,"createdAt":-1},"entity_01kzsa7tgbextv2kt75xknj5mg","Over engineering a pipeline to email me about strangers’ skills for dynobox","over-engineering-a-pipeline-to-email-me-about-strangers-skills-for-dynobox","For the past few weeks I've been building Dynobox - a local test runner that records what an agent does inside a harness (Claude Code, Codex, OpenCode etc.) and lets you write deterministic assertions against its non-deterministic behavior: tool calls, shell commands, files created or changed and so on.","https:\u002F\u002Fx.com\u002Fbhkdotdev\u002Fstatus\u002F2087281766440546814","bhk","bhkdotdev","2026-08-11",[],[],1786482227]