[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"$fwJIGrtFX1XNQozazZydT8A1xJN6Pk7mZaPl9TdCTvwM":3},{"tool":4,"categoryPool":30},{"id":5,"name":6,"slug":7,"description":8,"url":9,"githubUrl":10,"logoUrl":11,"openSource":12,"categories":13,"repo":18,"articles":28,"createdAt":29},"entity_01m09j68rhexyvmvdfcfd6fzwr","LLM-as-a-Verifier","llm-as-a-verifier","LLM-as-a-Verifier is a general-purpose framework that provides fine-grained feedback for any agent without requiring additional training. It achieves SOTA performance across coding, robotics, and medical agentic benchmarks.","https:\u002F\u002Fllm-as-a-verifier.com\u002F","https:\u002F\u002Fgithub.com\u002Fllm-as-a-verifier\u002Fllm-as-a-verifier","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fhq-e861a2b4.png",true,[14],{"name":15,"slug":16,"count":17},"Evals","evals",0,{"fullName":19,"url":10,"stars":20,"forks":21,"openIssues":22,"language":23,"license":24,"topics":25,"pushedAt":26,"archived":27},"llm-as-a-verifier\u002Fllm-as-a-verifier",852,68,5,"Python","MIT",[],"2026-08-14T20:32:51Z",false,[],1787027464,[31,61,81,113],{"id":32,"name":33,"slug":34,"description":35,"url":36,"githubUrl":36,"logoUrl":37,"openSource":12,"categories":38,"repo":40,"company":53,"articles":59,"createdAt":60},"entity_01kzhe2n9gfagsad1ydt1e7gkn","AACR-Bench","aacr-bench","An Alibaba open-source multi-language benchmark for evaluating LLMs in repository-level automatic code review, featuring an AI-assisted and expert-verified dataset.","https:\u002F\u002Fgithub.com\u002Falibaba\u002Faacr-bench","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Falibaba-0f6c077a.png",[39],{"name":15,"slug":16,"count":17},{"fullName":41,"url":36,"stars":42,"forks":43,"openIssues":44,"language":23,"license":45,"topics":46,"pushedAt":52,"archived":27},"alibaba\u002Faacr-bench",210,19,6,"Apache-2.0",[47,48,49,50,51],"benchmark","code-review","multi-language","repository-level-context","software-engineering","2026-08-04T05:36:38Z",{"id":54,"name":55,"slug":56,"url":57,"bio":58,"githubOrg":56},"entity_01kzmeveq2f22bbn6jcepyspwk","Alibaba","alibaba","https:\u002F\u002Fwww.alibabagroup.com","Alibaba Open Source",[],1786217846,{"id":62,"name":63,"slug":64,"description":65,"url":66,"githubUrl":67,"logoUrl":68,"openSource":12,"categories":69,"repo":71,"articles":79,"createdAt":80},"entity_01kzpjy8r7fvn8q0rcmcw8xg56","BenchLocal","benchlocal","Test LLMs on real tasks. Compare models side-by-side.","https:\u002F\u002Fbenchlocal.com\u002F","https:\u002F\u002Fgithub.com\u002Fstevibe\u002FBenchLocal","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fstevibe-cd02357b.png",[70],{"name":15,"slug":16,"count":17},{"fullName":72,"url":67,"stars":73,"forks":74,"openIssues":75,"language":76,"license":24,"topics":77,"pushedAt":78,"archived":27},"stevibe\u002FBenchLocal",407,46,14,"TypeScript",[],"2026-08-10T15:10:36Z",[],1786390717,{"id":82,"name":83,"slug":84,"description":85,"url":86,"logoUrl":87,"openSource":27,"categories":88,"company":93,"articles":97,"createdAt":112},"entity_01kyb2mdndeshatkngff89rdmc","Braintrust","braintrust","Ship quality agents at scale. Braintrust is the AI observability platform for tracing production, running evals, and catching regressions before they reach users.","https:\u002F\u002Fwww.braintrust.dev\u002F","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ficon180-ef918139.png",[89,90],{"name":15,"slug":16,"count":17},{"name":91,"slug":92,"count":17},"Observability","observability",{"id":94,"name":83,"slug":84,"url":95,"bio":85,"githubOrg":96},"entity_01kzmevhdaf22bbn7meq79a9a8","https:\u002F\u002Fbraintrust.dev\u002F","braintrustdata",[98],{"id":99,"name":100,"slug":101,"summary":102,"url":103,"kind":104,"platform":105,"author":106,"authorHandle":107,"publisher":108,"publishedAt":109,"about":110,"writtenBy":111},"entity_01kyb1ndrmf6esz3naek99dvk8","The context gold rush: Why everyone is building the same thing.","the-context-gold-rush-why-everyone-is-building-the-same-thing","You either die building product or live long enough to do context management.","https:\u002F\u002Fx.com\u002Fsamzliu\u002Fstatus\u002F2080210797465379147","blog","x","Sam Z Liu","samzliu","X","2026-07-23",[],[],1784930776,{"id":114,"name":115,"slug":116,"description":117,"url":118,"logoUrl":119,"openSource":27,"categories":120,"company":122,"articles":129,"createdAt":141},"entity_01kzyfk0nweh5b7faygzq2rgzz","DataBench","databench","DataBench scores frontier AI on the analytics work that matters — reasoning, reporting, and investigation on realistic, messy warehouse data.","https:\u002F\u002Fhex.tech\u002Fdatabench\u002F","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon-32ee2db0.svg",[121],{"name":15,"slug":16,"count":17},{"id":123,"name":124,"slug":125,"url":126,"bio":127,"githubOrg":128},"entity_01kzyfhc52eh5b7fac540s653f","Hex","hex","https:\u002F\u002Fhex.tech","Finally — anyone can get data insights grounded in the facts of their business. Hex has a flexible approach to context that earns trust without slowing you down.","hex-inc",[130],{"id":131,"name":132,"slug":133,"summary":134,"url":135,"kind":136,"platform":105,"author":124,"authorHandle":137,"publisher":108,"publishedAt":138,"about":139,"writtenBy":140},"entity_01kzye6311esg9ae93nr6enft3","Introducing DataBench","introducing-databench","DataBench v1: 100 realistic analytical tasks across Q&A and open-ended prompts, run in a synthetic Hex workspace — built because existing analytics benchmarks test \"overspecified pub trivia\" rather than the vague, directional questions people actually ask.","https:\u002F\u002Fx.com\u002F_hex_tech\u002Fstatus\u002F2087946398390206512","article","_hex_tech","2026-08-13",[],[],1786655638]