[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"tool-posttrainbench":3},{"tool":4,"categoryPool":48},{"id":5,"name":6,"slug":7,"description":8,"url":9,"githubUrl":10,"logoUrl":11,"openSource":12,"categories":13,"repo":18,"company":-1,"articles":34,"createdAt":47},"entity_01kynbyfz5esp8c5gps7nzxs7h","PostTrainBench","posttrainbench","Measuring how well CLI agents like Claude Code or Codex CLI can post-train base LLMs on a single H100 GPU in 10 hours","https:\u002F\u002Fposttrainbench.com\u002F","https:\u002F\u002Fgithub.com\u002Faisa-group\u002FPostTrainBench","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon-41ad5bc5.svg",true,[14],{"id":-1,"name":15,"slug":16,"count":17},"Evals","evals",0,{"fullName":19,"url":10,"stars":20,"forks":21,"openIssues":22,"language":23,"license":24,"topics":25,"pushedAt":32,"archived":33},"aisa-group\u002FPostTrainBench",511,58,20,"Python","MIT",[26,27,28,29,30,31],"ai-research-automation","ai-safety","claude-code","codex-cli","gemini-cli","post-training","2026-08-13T06:35:44Z",false,[35],{"id":36,"name":37,"slug":38,"summary":39,"url":40,"kind":41,"platform":42,"author":43,"authorHandle":-1,"publisher":6,"publishedAt":44,"about":45,"writtenBy":46,"createdAt":-1},"entity_01kynbs8kce4frj5kted3gdmsr","PostTrainBench v1.1: Hardening the benchmark against reward hacking","posttrainbench-v1-1-hardening-the-benchmark-against-reward-hacking","PostTrainBench v1.1 clarifies the boundary between legitimate benchmark hill climbing and item specific contamination, with specialized checks for external LLM API use, model substitution, and direct lookup.","https:\u002F\u002Fposttrainbench.com\u002Fblog\u002Fposttrainbench-1-1\u002F","announcement","web","PostTrainBench team","2026-07-28",[],[],1785276088,[49,79,99,124],{"id":50,"name":51,"slug":52,"description":53,"url":54,"githubUrl":54,"logoUrl":55,"openSource":12,"categories":56,"repo":58,"company":71,"articles":77,"createdAt":78},"entity_01kzhe2n9gfagsad1ydt1e7gkn","AACR-Bench","aacr-bench","An Alibaba open-source multi-language benchmark for evaluating LLMs in repository-level automatic code review, featuring an AI-assisted and expert-verified dataset.","https:\u002F\u002Fgithub.com\u002Falibaba\u002Faacr-bench","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Falibaba-0f6c077a.png",[57],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":59,"url":54,"stars":60,"forks":61,"openIssues":62,"language":23,"license":63,"topics":64,"pushedAt":70,"archived":33},"alibaba\u002Faacr-bench",206,16,6,"Apache-2.0",[65,66,67,68,69],"benchmark","code-review","multi-language","repository-level-context","software-engineering","2026-08-04T05:36:38Z",{"id":72,"name":73,"slug":74,"url":75,"bio":76,"githubOrg":74},"entity_01kzmeveq2f22bbn6jcepyspwk","Alibaba","alibaba","https:\u002F\u002Fwww.alibabagroup.com","Alibaba Open Source",[],1786217846,{"id":80,"name":81,"slug":82,"description":83,"url":84,"githubUrl":85,"logoUrl":86,"openSource":12,"categories":87,"repo":89,"company":-1,"articles":97,"createdAt":98},"entity_01kzpjy8r7fvn8q0rcmcw8xg56","BenchLocal","benchlocal","Test LLMs on real tasks. Compare models side-by-side.","https:\u002F\u002Fbenchlocal.com\u002F","https:\u002F\u002Fgithub.com\u002Fstevibe\u002FBenchLocal","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fstevibe-cd02357b.png",[88],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":90,"url":85,"stars":91,"forks":92,"openIssues":93,"language":94,"license":24,"topics":95,"pushedAt":96,"archived":33},"stevibe\u002FBenchLocal",402,45,12,"TypeScript",[],"2026-08-10T15:10:36Z",[],1786390717,{"id":100,"name":101,"slug":102,"description":103,"url":104,"githubUrl":-1,"logoUrl":105,"openSource":33,"categories":106,"repo":-1,"company":-1,"articles":108,"createdAt":123},"entity_01kyb2mdndeshatkngff89rdmc","Braintrust","braintrust","Ship quality agents at scale. Braintrust is the AI observability platform for tracing production, running evals, and catching regressions before they reach users.","https:\u002F\u002Fwww.braintrust.dev\u002F","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ficon180-ef918139.png",[107],{"id":-1,"name":15,"slug":16,"count":17},[109],{"id":110,"name":111,"slug":112,"summary":113,"url":114,"kind":115,"platform":116,"author":117,"authorHandle":118,"publisher":119,"publishedAt":120,"about":121,"writtenBy":122,"createdAt":-1},"entity_01kyb1ndrmf6esz3naek99dvk8","The context gold rush: Why everyone is building the same thing.","the-context-gold-rush-why-everyone-is-building-the-same-thing","You either die building product or live long enough to do context management.","https:\u002F\u002Fx.com\u002Fsamzliu\u002Fstatus\u002F2080210797465379147","blog","x","Sam Z Liu","samzliu","X","2026-07-23",[],[],1784930776,{"id":125,"name":126,"slug":127,"description":128,"url":129,"githubUrl":130,"logoUrl":131,"openSource":12,"categories":132,"repo":134,"company":-1,"articles":146,"createdAt":159},"entity_01kzsa6ykgextv2krhnxvx212d","Dynobox","dynobox","Cross-harness testing for multi-step agent flows","https:\u002F\u002Fdynobox.xyz","https:\u002F\u002Fgithub.com\u002Fdynobox\u002Fdynobox","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon-6008b356.svg",[133],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":135,"url":130,"stars":62,"forks":17,"openIssues":17,"language":94,"license":63,"topics":136,"pushedAt":145,"archived":33},"dynobox\u002Fdynobox",[137,138,28,139,140,141,16,142,143,144],"agent","agent-skills","cli","codex","devtools","llm","opencode","testing","2026-08-12T16:11:17Z",[147],{"id":148,"name":149,"slug":150,"summary":151,"url":152,"kind":153,"platform":116,"author":154,"authorHandle":155,"publisher":119,"publishedAt":156,"about":157,"writtenBy":158,"createdAt":-1},"entity_01kzsa7tgbextv2kt75xknj5mg","Over engineering a pipeline to email me about strangers’ skills for dynobox","over-engineering-a-pipeline-to-email-me-about-strangers-skills-for-dynobox","For the past few weeks I've been building Dynobox - a local test runner that records what an agent does inside a harness (Claude Code, Codex, OpenCode etc.) and lets you write deterministic assertions against its non-deterministic behavior: tool calls, shell commands, files created or changed and so on.","https:\u002F\u002Fx.com\u002Fbhkdotdev\u002Fstatus\u002F2087281766440546814","article","bhk","bhkdotdev","2026-08-11",[],[],1786482227]