[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"tool-memconflict":3},{"tool":4,"categoryPool":24},{"id":5,"name":6,"slug":7,"description":8,"url":9,"githubUrl":9,"logoUrl":-1,"openSource":10,"categories":11,"repo":16,"company":-1,"articles":22,"createdAt":23},"entity_01kzktrm3jecxrcgyq49hf0gwm","MemConflict","memconflict","Benchmark and evaluation toolkit for long-term memory systems under memory conflicts.","https:\u002F\u002Fgithub.com\u002FTaoZhen1110\u002FMemConflict",false,[12],{"id":-1,"name":13,"slug":14,"count":15},"Evals","evals",0,{"fullName":17,"url":9,"stars":18,"forks":15,"openIssues":15,"language":19,"license":-1,"topics":20,"pushedAt":21,"archived":10},"TaoZhen1110\u002FMemConflict",10,"Python",[],"2026-06-27T03:10:38Z",[],1786298257,[25,56,77,102],{"id":26,"name":27,"slug":28,"description":29,"url":30,"githubUrl":30,"logoUrl":31,"openSource":32,"categories":33,"repo":35,"company":48,"articles":54,"createdAt":55},"entity_01kzhe2n9gfagsad1ydt1e7gkn","AACR-Bench","aacr-bench","An Alibaba open-source multi-language benchmark for evaluating LLMs in repository-level automatic code review, featuring an AI-assisted and expert-verified dataset.","https:\u002F\u002Fgithub.com\u002Falibaba\u002Faacr-bench","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Falibaba-0f6c077a.png",true,[34],{"id":-1,"name":13,"slug":14,"count":15},{"fullName":36,"url":30,"stars":37,"forks":38,"openIssues":39,"language":19,"license":40,"topics":41,"pushedAt":47,"archived":10},"alibaba\u002Faacr-bench",206,16,6,"Apache-2.0",[42,43,44,45,46],"benchmark","code-review","multi-language","repository-level-context","software-engineering","2026-08-04T05:36:38Z",{"id":49,"name":50,"slug":51,"url":52,"bio":53,"githubOrg":51},"entity_01kzmeveq2f22bbn6jcepyspwk","Alibaba","alibaba","https:\u002F\u002Fwww.alibabagroup.com","Alibaba Open Source",[],1786217846,{"id":57,"name":58,"slug":59,"description":60,"url":61,"githubUrl":62,"logoUrl":63,"openSource":32,"categories":64,"repo":66,"company":-1,"articles":75,"createdAt":76},"entity_01kzpjy8r7fvn8q0rcmcw8xg56","BenchLocal","benchlocal","Test LLMs on real tasks. Compare models side-by-side.","https:\u002F\u002Fbenchlocal.com\u002F","https:\u002F\u002Fgithub.com\u002Fstevibe\u002FBenchLocal","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fstevibe-cd02357b.png",[65],{"id":-1,"name":13,"slug":14,"count":15},{"fullName":67,"url":62,"stars":68,"forks":69,"openIssues":70,"language":71,"license":72,"topics":73,"pushedAt":74,"archived":10},"stevibe\u002FBenchLocal",402,45,12,"TypeScript","MIT",[],"2026-08-10T15:10:36Z",[],1786390717,{"id":78,"name":79,"slug":80,"description":81,"url":82,"githubUrl":-1,"logoUrl":83,"openSource":10,"categories":84,"repo":-1,"company":-1,"articles":86,"createdAt":101},"entity_01kyb2mdndeshatkngff89rdmc","Braintrust","braintrust","Ship quality agents at scale. Braintrust is the AI observability platform for tracing production, running evals, and catching regressions before they reach users.","https:\u002F\u002Fwww.braintrust.dev\u002F","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ficon180-ef918139.png",[85],{"id":-1,"name":13,"slug":14,"count":15},[87],{"id":88,"name":89,"slug":90,"summary":91,"url":92,"kind":93,"platform":94,"author":95,"authorHandle":96,"publisher":97,"publishedAt":98,"about":99,"writtenBy":100,"createdAt":-1},"entity_01kyb1ndrmf6esz3naek99dvk8","The context gold rush: Why everyone is building the same thing.","the-context-gold-rush-why-everyone-is-building-the-same-thing","You either die building product or live long enough to do context management.","https:\u002F\u002Fx.com\u002Fsamzliu\u002Fstatus\u002F2080210797465379147","blog","x","Sam Z Liu","samzliu","X","2026-07-23",[],[],1784930776,{"id":103,"name":104,"slug":105,"description":106,"url":107,"githubUrl":108,"logoUrl":109,"openSource":32,"categories":110,"repo":112,"company":-1,"articles":125,"createdAt":138},"entity_01kzsa6ykgextv2krhnxvx212d","Dynobox","dynobox","Cross-harness testing for multi-step agent flows","https:\u002F\u002Fdynobox.xyz","https:\u002F\u002Fgithub.com\u002Fdynobox\u002Fdynobox","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon-6008b356.svg",[111],{"id":-1,"name":13,"slug":14,"count":15},{"fullName":113,"url":108,"stars":39,"forks":15,"openIssues":15,"language":71,"license":40,"topics":114,"pushedAt":124,"archived":10},"dynobox\u002Fdynobox",[115,116,117,118,119,120,14,121,122,123],"agent","agent-skills","claude-code","cli","codex","devtools","llm","opencode","testing","2026-08-12T16:11:17Z",[126],{"id":127,"name":128,"slug":129,"summary":130,"url":131,"kind":132,"platform":94,"author":133,"authorHandle":134,"publisher":97,"publishedAt":135,"about":136,"writtenBy":137,"createdAt":-1},"entity_01kzsa7tgbextv2kt75xknj5mg","Over engineering a pipeline to email me about strangers’ skills for dynobox","over-engineering-a-pipeline-to-email-me-about-strangers-skills-for-dynobox","For the past few weeks I've been building Dynobox - a local test runner that records what an agent does inside a harness (Claude Code, Codex, OpenCode etc.) and lets you write deterministic assertions against its non-deterministic behavior: tool calls, shell commands, files created or changed and so on.","https:\u002F\u002Fx.com\u002Fbhkdotdev\u002Fstatus\u002F2087281766440546814","article","bhk","bhkdotdev","2026-08-11",[],[],1786482227]