[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"$f9lgrRPXimank-uGAYW__gU8aM2YlaH-NK4M5oSpEiq0":3},{"tool":4,"categoryPool":59},{"id":5,"name":6,"slug":7,"description":8,"url":9,"githubUrl":10,"logoUrl":11,"openSource":12,"categories":13,"repo":18,"articles":44,"createdAt":58},"entity_01m0eay3fefk4bkrky9ryetwx6","SkillEvaluator","skillevaluator","Multi-tier framework for evaluating AI agent skills with quality gates, semantic overlap detection, synthetic evaluation dataset generation, and live agent evaluation that measures how skills affect agent behavior.","https:\u002F\u002Fdocs.nvidia.com\u002Fskills\u002Fskillevaluator\u002F","https:\u002F\u002Fgithub.com\u002FNVIDIA\u002FSkillEvaluator","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon-4e56a97d.ico",true,[14],{"name":15,"slug":16,"count":17},"Agent Testing","agent-testing",0,{"fullName":19,"url":10,"stars":20,"forks":21,"openIssues":22,"language":23,"license":24,"topics":25,"pushedAt":41,"archived":42,"licenseSpdx":24,"licenseOsi":12,"licenseVerifiedAt":43},"NVIDIA\u002FSkillEvaluator",100,8,12,"Python","Apache-2.0",[26,27,28,29,30,31,32,33,34,35,36,37,38,39,40],"agent-evaluation","agent-security","agent-skills","agentic-ai","benchmark","claude-code","codex","evaluate","evaluation","security-scanner","skill-eval","skill-evals","skill-evaluation","skill-evaluator","skills","2026-08-20T02:36:48Z",false,"2026-08-20T01:00:34.282Z",[45],{"id":46,"name":47,"slug":48,"summary":49,"url":50,"kind":51,"platform":52,"author":53,"publisher":54,"publishedAt":55,"about":56,"writtenBy":57},"entity_01m0eazbszesjvfq76a49mp85h","Evaluating AI Agent Skill Performance with NVIDIA SkillEvaluator","evaluating-ai-agent-skill-performance-with-nvidia-skillevaluator","AI agents are only as effective as the context they receive. Even with capable models and well-documented NVIDIA libraries, agents can spend extra steps finding the right tools, burn tokens on dead ends, or struggle with specialized tasks. NVIDIA SkillEvaluator measures how skills affect agent performance through static checks and real-world task runs with and without each skill.","https:\u002F\u002Fdeveloper.nvidia.com\u002Fblog\u002Fevaluating-ai-agent-skill-performance-with-nvidia-skillevaluator","article","web","Michelle Horton","NVIDIA Technical Blog","2026-08-19",[],[],1787187629,[60,98,107],{"id":61,"name":62,"slug":63,"description":64,"url":65,"githubUrl":66,"logoUrl":67,"openSource":12,"categories":68,"repo":70,"articles":84,"createdAt":97},"entity_01kzsa6ykgextv2krhnxvx212d","Dynobox","dynobox","Cross-harness testing for multi-step agent flows","https:\u002F\u002Fdynobox.xyz","https:\u002F\u002Fgithub.com\u002Fdynobox\u002Fdynobox","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon-6008b356.svg",[69],{"name":15,"slug":16,"count":17},{"fullName":71,"url":66,"stars":72,"forks":17,"openIssues":17,"language":73,"license":24,"topics":74,"pushedAt":82,"archived":42,"licenseSpdx":24,"licenseOsi":12,"licenseVerifiedAt":83},"dynobox\u002Fdynobox",10,"TypeScript",[75,28,31,76,32,77,78,79,80,81],"agent","cli","devtools","evals","llm","opencode","testing","2026-08-18T22:49:47Z","2026-08-11",[85],{"id":86,"name":87,"slug":88,"summary":89,"url":90,"kind":51,"platform":91,"author":92,"authorHandle":93,"publisher":94,"publishedAt":83,"about":95,"writtenBy":96},"entity_01kzsa7tgbextv2kt75xknj5mg","Over engineering a pipeline to email me about strangers’ skills for dynobox","over-engineering-a-pipeline-to-email-me-about-strangers-skills-for-dynobox","For the past few weeks I've been building Dynobox - a local test runner that records what an agent does inside a harness (Claude Code, Codex, OpenCode etc.) and lets you write deterministic assertions against its non-deterministic behavior: tool calls, shell commands, files created or changed and so on.","https:\u002F\u002Fx.com\u002Fbhkdotdev\u002Fstatus\u002F2087281766440546814","x","bhk","bhkdotdev","X",[],[],1786482227,{"id":5,"name":6,"slug":7,"description":8,"url":9,"githubUrl":10,"logoUrl":11,"openSource":12,"categories":99,"repo":101,"articles":103,"createdAt":58},[100],{"name":15,"slug":16,"count":17},{"fullName":19,"url":10,"stars":20,"forks":21,"openIssues":22,"language":23,"license":24,"topics":102,"pushedAt":41,"archived":42,"licenseSpdx":24,"licenseOsi":12,"licenseVerifiedAt":43},[26,27,28,29,30,31,32,33,34,35,36,37,38,39,40],[104],{"id":46,"name":47,"slug":48,"summary":49,"url":50,"kind":51,"platform":52,"author":53,"publisher":54,"publishedAt":55,"about":105,"writtenBy":106},[],[],{"id":108,"name":109,"slug":110,"description":111,"url":112,"githubUrl":112,"logoUrl":113,"openSource":12,"categories":114,"repo":116,"articles":126,"createdAt":127},"entity_01m0bp7xa7e4fbqpswv3j249vj","SkillForge","skillforge","A skill creator that proves its skills work. Evidence-driven skill creation for Claude Code and Codex: baseline-tested generation, per-skill regression evals, ecosystem doctor, cross-runtime compile, and an opt-in proactive advisor.","https:\u002F\u002Fgithub.com\u002Ftripleyak\u002FSkillForge","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ftripleyak-8696ff1c.png",[115],{"name":15,"slug":16,"count":17},{"fullName":117,"url":112,"stars":118,"forks":119,"openIssues":17,"language":23,"license":120,"topics":121,"pushedAt":124,"archived":42,"licenseSpdx":120,"licenseOsi":12,"licenseVerifiedAt":125},"tripleyak\u002FSkillForge",868,89,"MIT",[28,122,31,123,32,78],"claude-ai","claude-skills","2026-07-29T16:31:19Z","2026-08-19T00:20:26.791Z",[],1787098821]