[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"tool-phoenix":3},{"tool":4,"categoryPool":50},{"id":5,"name":6,"slug":7,"description":8,"url":9,"githubUrl":10,"logoUrl":11,"openSource":12,"categories":13,"repo":18,"company":41,"articles":48,"createdAt":49},"entity_01ky680xmnf24s833f2ya485vp","Phoenix","phoenix","AI Observability & Evaluation","https:\u002F\u002Farize.com\u002Fdocs\u002Fphoenix","https:\u002F\u002Fgithub.com\u002FArize-ai\u002Fphoenix","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fapple-touch-icon-31c8aa5d.png",false,[14],{"id":-1,"name":15,"slug":16,"count":17},"Evals","evals",0,{"fullName":19,"url":10,"stars":20,"forks":21,"openIssues":22,"language":23,"license":-1,"topics":24,"pushedAt":40,"archived":12},"Arize-ai\u002Fphoenix",11025,1054,913,"Python",[25,26,27,28,29,30,16,31,32,33,34,35,36,37,38,39],"agents","ai-monitoring","ai-observability","aiengineering","anthropic","datasets","langchain","llamaindex","llm-eval","llm-evaluation","llmops","llms","openai","prompt-engineering","smolagents","2026-08-13T05:57:44Z",{"id":42,"name":43,"slug":44,"url":45,"bio":46,"githubOrg":47},"entity_01kzmezygjf22bbnhcgwycy86m","Arize AI","arize-ai","https:\u002F\u002Fwww.arize.com\u002F","Continuously improve AI agents with agent observability, evaluation, tracing, and experimentation.","Arize-ai",[],1784768657,[51,82,103,128],{"id":52,"name":53,"slug":54,"description":55,"url":56,"githubUrl":56,"logoUrl":57,"openSource":58,"categories":59,"repo":61,"company":74,"articles":80,"createdAt":81},"entity_01kzhe2n9gfagsad1ydt1e7gkn","AACR-Bench","aacr-bench","An Alibaba open-source multi-language benchmark for evaluating LLMs in repository-level automatic code review, featuring an AI-assisted and expert-verified dataset.","https:\u002F\u002Fgithub.com\u002Falibaba\u002Faacr-bench","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Falibaba-0f6c077a.png",true,[60],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":62,"url":56,"stars":63,"forks":64,"openIssues":65,"language":23,"license":66,"topics":67,"pushedAt":73,"archived":12},"alibaba\u002Faacr-bench",206,16,6,"Apache-2.0",[68,69,70,71,72],"benchmark","code-review","multi-language","repository-level-context","software-engineering","2026-08-04T05:36:38Z",{"id":75,"name":76,"slug":77,"url":78,"bio":79,"githubOrg":77},"entity_01kzmeveq2f22bbn6jcepyspwk","Alibaba","alibaba","https:\u002F\u002Fwww.alibabagroup.com","Alibaba Open Source",[],1786217846,{"id":83,"name":84,"slug":85,"description":86,"url":87,"githubUrl":88,"logoUrl":89,"openSource":58,"categories":90,"repo":92,"company":-1,"articles":101,"createdAt":102},"entity_01kzpjy8r7fvn8q0rcmcw8xg56","BenchLocal","benchlocal","Test LLMs on real tasks. Compare models side-by-side.","https:\u002F\u002Fbenchlocal.com\u002F","https:\u002F\u002Fgithub.com\u002Fstevibe\u002FBenchLocal","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fstevibe-cd02357b.png",[91],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":93,"url":88,"stars":94,"forks":95,"openIssues":96,"language":97,"license":98,"topics":99,"pushedAt":100,"archived":12},"stevibe\u002FBenchLocal",402,45,12,"TypeScript","MIT",[],"2026-08-10T15:10:36Z",[],1786390717,{"id":104,"name":105,"slug":106,"description":107,"url":108,"githubUrl":-1,"logoUrl":109,"openSource":12,"categories":110,"repo":-1,"company":-1,"articles":112,"createdAt":127},"entity_01kyb2mdndeshatkngff89rdmc","Braintrust","braintrust","Ship quality agents at scale. Braintrust is the AI observability platform for tracing production, running evals, and catching regressions before they reach users.","https:\u002F\u002Fwww.braintrust.dev\u002F","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ficon180-ef918139.png",[111],{"id":-1,"name":15,"slug":16,"count":17},[113],{"id":114,"name":115,"slug":116,"summary":117,"url":118,"kind":119,"platform":120,"author":121,"authorHandle":122,"publisher":123,"publishedAt":124,"about":125,"writtenBy":126,"createdAt":-1},"entity_01kyb1ndrmf6esz3naek99dvk8","The context gold rush: Why everyone is building the same thing.","the-context-gold-rush-why-everyone-is-building-the-same-thing","You either die building product or live long enough to do context management.","https:\u002F\u002Fx.com\u002Fsamzliu\u002Fstatus\u002F2080210797465379147","blog","x","Sam Z Liu","samzliu","X","2026-07-23",[],[],1784930776,{"id":129,"name":130,"slug":131,"description":132,"url":133,"githubUrl":134,"logoUrl":135,"openSource":58,"categories":136,"repo":138,"company":-1,"articles":151,"createdAt":164},"entity_01kzsa6ykgextv2krhnxvx212d","Dynobox","dynobox","Cross-harness testing for multi-step agent flows","https:\u002F\u002Fdynobox.xyz","https:\u002F\u002Fgithub.com\u002Fdynobox\u002Fdynobox","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon-6008b356.svg",[137],{"id":-1,"name":15,"slug":16,"count":17},{"fullName":139,"url":134,"stars":65,"forks":17,"openIssues":17,"language":97,"license":66,"topics":140,"pushedAt":150,"archived":12},"dynobox\u002Fdynobox",[141,142,143,144,145,146,16,147,148,149],"agent","agent-skills","claude-code","cli","codex","devtools","llm","opencode","testing","2026-08-12T16:11:17Z",[152],{"id":153,"name":154,"slug":155,"summary":156,"url":157,"kind":158,"platform":120,"author":159,"authorHandle":160,"publisher":123,"publishedAt":161,"about":162,"writtenBy":163,"createdAt":-1},"entity_01kzsa7tgbextv2kt75xknj5mg","Over engineering a pipeline to email me about strangers’ skills for dynobox","over-engineering-a-pipeline-to-email-me-about-strangers-skills-for-dynobox","For the past few weeks I've been building Dynobox - a local test runner that records what an agent does inside a harness (Claude Code, Codex, OpenCode etc.) and lets you write deterministic assertions against its non-deterministic behavior: tool calls, shell commands, files created or changed and so on.","https:\u002F\u002Fx.com\u002Fbhkdotdev\u002Fstatus\u002F2087281766440546814","article","bhk","bhkdotdev","2026-08-11",[],[],1786482227]