[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"tool-vitest-evals":3},{"tool":4,"categoryPool":36},{"id":5,"name":6,"slug":6,"description":7,"url":8,"githubUrl":8,"logoUrl":9,"openSource":10,"categories":11,"repo":16,"company":27,"articles":34,"createdAt":35},"entity_01kwn4nwxyeh19ap78bwvfbkse","vitest-evals","A vitest extension for running evals.","https:\u002F\u002Fgithub.com\u002Fgetsentry\u002Fvitest-evals","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fgetsentry-ef890959.png",true,[12],{"id":-1,"name":13,"slug":14,"count":15},"Evals","evals",0,{"fullName":17,"url":8,"stars":18,"forks":19,"openIssues":20,"language":21,"license":22,"topics":23,"pushedAt":25,"archived":26},"getsentry\u002Fvitest-evals",328,6,14,"TypeScript","Apache-2.0",[24],"tag-non-production","2026-08-07T21:00:42Z",false,{"id":28,"name":29,"slug":30,"url":31,"bio":32,"githubOrg":33},"entity_01kynhndm1eczterchpn9vxe47","Sentry","sentry","https:\u002F\u002Fsentry.io\u002F","Application performance monitoring for developers & software teams to see errors clearer, solve issues faster & continue learning continuously. Get started at sentry.io.","getsentry",[],1783120982,[37,66,86,111],{"id":38,"name":39,"slug":40,"description":41,"url":42,"githubUrl":42,"logoUrl":43,"openSource":10,"categories":44,"repo":46,"company":58,"articles":64,"createdAt":65},"entity_01kzhe2n9gfagsad1ydt1e7gkn","AACR-Bench","aacr-bench","An Alibaba open-source multi-language benchmark for evaluating LLMs in repository-level automatic code review, featuring an AI-assisted and expert-verified dataset.","https:\u002F\u002Fgithub.com\u002Falibaba\u002Faacr-bench","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Falibaba-0f6c077a.png",[45],{"id":-1,"name":13,"slug":14,"count":15},{"fullName":47,"url":42,"stars":48,"forks":49,"openIssues":19,"language":50,"license":22,"topics":51,"pushedAt":57,"archived":26},"alibaba\u002Faacr-bench",206,16,"Python",[52,53,54,55,56],"benchmark","code-review","multi-language","repository-level-context","software-engineering","2026-08-04T05:36:38Z",{"id":59,"name":60,"slug":61,"url":62,"bio":63,"githubOrg":61},"entity_01kzmeveq2f22bbn6jcepyspwk","Alibaba","alibaba","https:\u002F\u002Fwww.alibabagroup.com","Alibaba Open Source",[],1786217846,{"id":67,"name":68,"slug":69,"description":70,"url":71,"githubUrl":72,"logoUrl":73,"openSource":10,"categories":74,"repo":76,"company":-1,"articles":84,"createdAt":85},"entity_01kzpjy8r7fvn8q0rcmcw8xg56","BenchLocal","benchlocal","Test LLMs on real tasks. Compare models side-by-side.","https:\u002F\u002Fbenchlocal.com\u002F","https:\u002F\u002Fgithub.com\u002Fstevibe\u002FBenchLocal","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fstevibe-cd02357b.png",[75],{"id":-1,"name":13,"slug":14,"count":15},{"fullName":77,"url":72,"stars":78,"forks":79,"openIssues":80,"language":21,"license":81,"topics":82,"pushedAt":83,"archived":26},"stevibe\u002FBenchLocal",402,45,12,"MIT",[],"2026-08-10T15:10:36Z",[],1786390717,{"id":87,"name":88,"slug":89,"description":90,"url":91,"githubUrl":-1,"logoUrl":92,"openSource":26,"categories":93,"repo":-1,"company":-1,"articles":95,"createdAt":110},"entity_01kyb2mdndeshatkngff89rdmc","Braintrust","braintrust","Ship quality agents at scale. Braintrust is the AI observability platform for tracing production, running evals, and catching regressions before they reach users.","https:\u002F\u002Fwww.braintrust.dev\u002F","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ficon180-ef918139.png",[94],{"id":-1,"name":13,"slug":14,"count":15},[96],{"id":97,"name":98,"slug":99,"summary":100,"url":101,"kind":102,"platform":103,"author":104,"authorHandle":105,"publisher":106,"publishedAt":107,"about":108,"writtenBy":109,"createdAt":-1},"entity_01kyb1ndrmf6esz3naek99dvk8","The context gold rush: Why everyone is building the same thing.","the-context-gold-rush-why-everyone-is-building-the-same-thing","You either die building product or live long enough to do context management.","https:\u002F\u002Fx.com\u002Fsamzliu\u002Fstatus\u002F2080210797465379147","blog","x","Sam Z Liu","samzliu","X","2026-07-23",[],[],1784930776,{"id":112,"name":113,"slug":114,"description":115,"url":116,"githubUrl":117,"logoUrl":118,"openSource":10,"categories":119,"repo":121,"company":-1,"articles":134,"createdAt":147},"entity_01kzsa6ykgextv2krhnxvx212d","Dynobox","dynobox","Cross-harness testing for multi-step agent flows","https:\u002F\u002Fdynobox.xyz","https:\u002F\u002Fgithub.com\u002Fdynobox\u002Fdynobox","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon-6008b356.svg",[120],{"id":-1,"name":13,"slug":14,"count":15},{"fullName":122,"url":117,"stars":19,"forks":15,"openIssues":15,"language":21,"license":22,"topics":123,"pushedAt":133,"archived":26},"dynobox\u002Fdynobox",[124,125,126,127,128,129,14,130,131,132],"agent","agent-skills","claude-code","cli","codex","devtools","llm","opencode","testing","2026-08-12T16:11:17Z",[135],{"id":136,"name":137,"slug":138,"summary":139,"url":140,"kind":141,"platform":103,"author":142,"authorHandle":143,"publisher":106,"publishedAt":144,"about":145,"writtenBy":146,"createdAt":-1},"entity_01kzsa7tgbextv2kt75xknj5mg","Over engineering a pipeline to email me about strangers’ skills for dynobox","over-engineering-a-pipeline-to-email-me-about-strangers-skills-for-dynobox","For the past few weeks I've been building Dynobox - a local test runner that records what an agent does inside a harness (Claude Code, Codex, OpenCode etc.) and lets you write deterministic assertions against its non-deterministic behavior: tool calls, shell commands, files created or changed and so on.","https:\u002F\u002Fx.com\u002Fbhkdotdev\u002Fstatus\u002F2087281766440546814","article","bhk","bhkdotdev","2026-08-11",[],[],1786482227]