[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"tool-anydoc":3},{"tool":4,"categoryPool":29},{"id":5,"name":6,"slug":6,"description":7,"url":8,"githubUrl":9,"logoUrl":10,"openSource":11,"categories":12,"repo":17,"company":-1,"articles":27,"createdAt":28},"entity_01kzpezrpjfvn8q04kx1e37pqp","anydoc","Convert Word, PowerPoint, Excel, OpenDocument, RTF, EPUB, CSV, and PDF to clean Markdown. Built in Rust, with Node.js and Python bindings.","https:\u002F\u002Ffirecrawl.github.io\u002Fanydoc\u002F","https:\u002F\u002Fgithub.com\u002Ffirecrawl\u002Fanydoc","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Flogo-c40f9840.svg",true,[13],{"id":-1,"name":14,"slug":15,"count":16},"Extraction","extraction",0,{"fullName":18,"url":9,"stars":19,"forks":20,"openIssues":21,"language":22,"license":23,"topics":24,"pushedAt":25,"archived":26},"firecrawl\u002Fanydoc",15407,824,57,"Rust","MIT",[],"2026-08-10T23:33:04Z",false,[],1786386571,[30,63,105,122],{"id":31,"name":32,"slug":33,"description":34,"url":35,"githubUrl":36,"logoUrl":37,"openSource":11,"categories":38,"repo":40,"company":-1,"articles":61,"createdAt":62},"entity_01ky0s7wgrf239sywgdkw7e8gz","DocETL","docetl","Open-source toolkit, built by the EPIC Data Lab at UC Berkeley, for creating LLM-powered pipelines that extract, transform, and link knowledge from unstructured documents.","https:\u002F\u002Fwww.docetl.org\u002F","https:\u002F\u002Fgithub.com\u002Fucbepic\u002Fdocetl","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fdocetl-favicon-color-9d034c86.png",[39],{"id":-1,"name":14,"slug":15,"count":16},{"fullName":41,"url":36,"stars":42,"forks":43,"openIssues":44,"language":45,"license":23,"topics":46,"pushedAt":60,"archived":26},"ucbepic\u002Fdocetl",3964,423,43,"Python",[47,48,49,50,51,52,53,54,55,56,57,58,59],"agents","data","data-pipelines","document-analysis","document-processing","elt","etl","llm","python","semantic-data","unstructured-data","unstructured-data-analysis","workflow","2026-08-10T15:00:22Z",[],1784585384,{"id":64,"name":65,"slug":66,"description":67,"url":68,"githubUrl":69,"logoUrl":70,"openSource":11,"categories":71,"repo":76,"company":-1,"articles":91,"createdAt":104},"entity_01kzs3gy7rfantmwgtrzpzt8yn","ExtractBench","extractbench","A benchmark for schema-guided extraction from real enterprise documents. 370 documents, 4,869 pages, 67 document types, each with its own JSON Schema. Scored on value accuracy, completeness, and evidence.","https:\u002F\u002Fwww.extractbench.ai\u002F","https:\u002F\u002Fgithub.com\u002Frun-llama\u002FExtractBench","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fapple-touch-icon-da922544.png",[72,75],{"id":-1,"name":73,"slug":74,"count":16},"Evals","evals",{"id":-1,"name":14,"slug":15,"count":16},{"fullName":77,"url":69,"stars":78,"forks":79,"openIssues":16,"language":45,"license":80,"topics":81,"pushedAt":90,"archived":26},"run-llama\u002FExtractBench",39,4,"Apache-2.0",[82,83,84,85,86,87,88,54,89],"benchmark","coding-agent","document-ai","evaluation","extract","extract-data","llamaindex","vision-language-model","2026-08-08T18:34:23Z",[92],{"id":93,"name":94,"slug":95,"summary":96,"url":97,"kind":98,"platform":99,"author":-1,"authorHandle":-1,"publisher":100,"publishedAt":101,"about":102,"writtenBy":103,"createdAt":-1},"entity_01kzs3krzxfantmwj9zrcz3hbt","ExtractBench: The Most Comprehensive Extraction Benchmark","introducing-extractbench","The most comprehensive document extraction benchmark: 14 systems scored on accuracy, completeness, grounding, and cost across 370 enterprise documents.","https:\u002F\u002Fwww.llamaindex.ai\u002Fblog\u002Fintroducing-extractbench","announcement","web","LlamaIndex","2026-08-11",[],[],1786475215,{"id":106,"name":107,"slug":108,"description":109,"url":110,"githubUrl":110,"logoUrl":-1,"openSource":11,"categories":111,"repo":113,"company":-1,"articles":120,"createdAt":121},"entity_01kzhfcs4xeh2vtra99mntepyy","Knowledge Graph Builder","knowledge-graph-builder","Repository for building knowledge graphs from specific datasets using generative language model through ollama","https:\u002F\u002Fgithub.com\u002FFabioYanezRomero\u002FKnowledge-Graph-Builder",[112],{"id":-1,"name":14,"slug":15,"count":16},{"fullName":114,"url":110,"stars":115,"forks":116,"openIssues":117,"language":45,"license":23,"topics":118,"pushedAt":119,"archived":26},"FabioYanezRomero\u002FKnowledge-Graph-Builder",98,12,2,[],"2026-07-04T18:27:20Z",[],1786219226,{"id":123,"name":124,"slug":125,"description":126,"url":127,"githubUrl":128,"logoUrl":129,"openSource":11,"categories":130,"repo":132,"company":-1,"articles":148,"createdAt":149},"entity_01ky09k25gfanafx44rj2f9dpf","LangExtract","langextract","A Python library for extracting structured information from unstructured text using LLMs with precise source grounding and interactive visualization.","https:\u002F\u002Fpypi.org\u002Fproject\u002Flangextract\u002F","https:\u002F\u002Fgithub.com\u002Fgoogle\u002Flangextract","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon.35549fe8-6dd120da.ico",[131],{"id":-1,"name":14,"slug":15,"count":16},{"fullName":133,"url":128,"stars":134,"forks":135,"openIssues":136,"language":45,"license":80,"topics":137,"pushedAt":147,"archived":26},"google\u002Flangextract",38325,2685,120,[138,139,140,141,142,143,144,54,145,55,146],"gemini","gemini-ai","gemini-api","gemini-flash","gemini-pro","information-extration","large-language-models","nlp","structured-data","2026-08-11T15:31:39Z",[],1784568973]