[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"tool-databench":3},{"tool":4,"categoryPool":39},{"id":5,"name":6,"slug":7,"description":8,"url":9,"githubUrl":-1,"logoUrl":10,"openSource":11,"categories":12,"repo":-1,"company":17,"articles":24,"createdAt":38},"entity_01kzyfk0nweh5b7faygzq2rgzz","DataBench","databench","DataBench scores frontier AI on the analytics work that matters — reasoning, reporting, and investigation on realistic, messy warehouse data.","https:\u002F\u002Fhex.tech\u002Fdatabench\u002F","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon-32ee2db0.svg",false,[13],{"id":-1,"name":14,"slug":15,"count":16},"Evals","evals",0,{"id":18,"name":19,"slug":20,"url":21,"bio":22,"githubOrg":23},"entity_01kzyfhc52eh5b7fac540s653f","Hex","hex","https:\u002F\u002Fhex.tech","Finally — anyone can get data insights grounded in the facts of their business. Hex has a flexible approach to context that earns trust without slowing you down.","hex-inc",[25],{"id":26,"name":27,"slug":28,"summary":29,"url":30,"kind":31,"platform":32,"author":19,"authorHandle":33,"publisher":34,"publishedAt":35,"about":36,"writtenBy":37,"createdAt":-1},"entity_01kzye6311esg9ae93nr6enft3","Introducing DataBench","introducing-databench","DataBench v1: 100 realistic analytical tasks across Q&A and open-ended prompts, run in a synthetic Hex workspace — built because existing analytics benchmarks test \"overspecified pub trivia\" rather than the vague, directional questions people actually ask.","https:\u002F\u002Fx.com\u002F_hex_tech\u002Fstatus\u002F2087946398390206512","article","x","_hex_tech","X","2026-08-13",[],[],1786655638,[40,72,93,123],{"id":41,"name":42,"slug":43,"description":44,"url":45,"githubUrl":45,"logoUrl":46,"openSource":47,"categories":48,"repo":50,"company":64,"articles":70,"createdAt":71},"entity_01kzhe2n9gfagsad1ydt1e7gkn","AACR-Bench","aacr-bench","An Alibaba open-source multi-language benchmark for evaluating LLMs in repository-level automatic code review, featuring an AI-assisted and expert-verified dataset.","https:\u002F\u002Fgithub.com\u002Falibaba\u002Faacr-bench","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Falibaba-0f6c077a.png",true,[49],{"id":-1,"name":14,"slug":15,"count":16},{"fullName":51,"url":45,"stars":52,"forks":53,"openIssues":54,"language":55,"license":56,"topics":57,"pushedAt":63,"archived":11},"alibaba\u002Faacr-bench",206,17,6,"Python","Apache-2.0",[58,59,60,61,62],"benchmark","code-review","multi-language","repository-level-context","software-engineering","2026-08-04T05:36:38Z",{"id":65,"name":66,"slug":67,"url":68,"bio":69,"githubOrg":67},"entity_01kzmeveq2f22bbn6jcepyspwk","Alibaba","alibaba","https:\u002F\u002Fwww.alibabagroup.com","Alibaba Open Source",[],1786217846,{"id":73,"name":74,"slug":75,"description":76,"url":77,"githubUrl":78,"logoUrl":79,"openSource":47,"categories":80,"repo":82,"company":-1,"articles":91,"createdAt":92},"entity_01kzpjy8r7fvn8q0rcmcw8xg56","BenchLocal","benchlocal","Test LLMs on real tasks. Compare models side-by-side.","https:\u002F\u002Fbenchlocal.com\u002F","https:\u002F\u002Fgithub.com\u002Fstevibe\u002FBenchLocal","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fstevibe-cd02357b.png",[81],{"id":-1,"name":14,"slug":15,"count":16},{"fullName":83,"url":78,"stars":84,"forks":85,"openIssues":86,"language":87,"license":88,"topics":89,"pushedAt":90,"archived":11},"stevibe\u002FBenchLocal",404,45,12,"TypeScript","MIT",[],"2026-08-10T15:10:36Z",[],1786390717,{"id":94,"name":95,"slug":96,"description":97,"url":98,"githubUrl":-1,"logoUrl":99,"openSource":11,"categories":100,"repo":-1,"company":105,"articles":109,"createdAt":122},"entity_01kyb2mdndeshatkngff89rdmc","Braintrust","braintrust","Ship quality agents at scale. Braintrust is the AI observability platform for tracing production, running evals, and catching regressions before they reach users.","https:\u002F\u002Fwww.braintrust.dev\u002F","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ficon180-ef918139.png",[101,102],{"id":-1,"name":14,"slug":15,"count":16},{"id":-1,"name":103,"slug":104,"count":16},"Observability","observability",{"id":106,"name":95,"slug":96,"url":107,"bio":97,"githubOrg":108},"entity_01kzmevhdaf22bbn7meq79a9a8","https:\u002F\u002Fbraintrust.dev\u002F","braintrustdata",[110],{"id":111,"name":112,"slug":113,"summary":114,"url":115,"kind":116,"platform":32,"author":117,"authorHandle":118,"publisher":34,"publishedAt":119,"about":120,"writtenBy":121,"createdAt":-1},"entity_01kyb1ndrmf6esz3naek99dvk8","The context gold rush: Why everyone is building the same thing.","the-context-gold-rush-why-everyone-is-building-the-same-thing","You either die building product or live long enough to do context management.","https:\u002F\u002Fx.com\u002Fsamzliu\u002Fstatus\u002F2080210797465379147","blog","Sam Z Liu","samzliu","2026-07-23",[],[],1784930776,{"id":5,"name":6,"slug":7,"description":8,"url":9,"githubUrl":-1,"logoUrl":10,"openSource":11,"categories":124,"repo":-1,"company":126,"articles":127,"createdAt":38},[125],{"id":-1,"name":14,"slug":15,"count":16},{"id":18,"name":19,"slug":20,"url":21,"bio":22,"githubOrg":23},[128],{"id":26,"name":27,"slug":28,"summary":29,"url":30,"kind":31,"platform":32,"author":19,"authorHandle":33,"publisher":34,"publishedAt":35,"about":129,"writtenBy":130,"createdAt":-1},[],[]]