[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"tool-pdf-inspector":3},{"tool":4,"categoryPool":45},{"id":5,"name":6,"slug":6,"description":7,"url":8,"githubUrl":9,"logoUrl":10,"openSource":11,"categories":12,"repo":17,"company":37,"articles":43,"createdAt":44},"entity_01kz1wb1t0f238b0zfjefcg0p2","pdf-inspector","Fast Rust library for PDF inspection, classification, and text extraction. Intelligently detects scanned vs text-based PDFs to enable smart routing decisions.","https:\u002F\u002Ffirecrawl.github.io\u002Fpdf-inspector\u002F","https:\u002F\u002Fgithub.com\u002Ffirecrawl\u002Fpdf-inspector","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffirecrawl-mark-ab2c3766.svg",true,[13],{"id":-1,"name":14,"slug":15,"count":16},"Extraction","extraction",0,{"fullName":18,"url":9,"stars":19,"forks":20,"openIssues":21,"language":22,"license":23,"topics":24,"pushedAt":35,"archived":36},"firecrawl\u002Fpdf-inspector",15172,1045,150,"Rust","MIT",[25,26,27,28,29,30,31,32,33,34],"markdown","nodejs","ocr-routing","pdf","pdf-classification","pdf-extraction","pdf-parser","python","rust","text-extraction","2026-08-13T00:07:37Z",false,{"id":38,"name":39,"slug":40,"url":41,"bio":42,"githubOrg":40},"entity_01kz1wdt5tf238b10s8exgwwj5","Firecrawl","firecrawl","https:\u002F\u002Ffirecrawl.dev","Firecrawl is the context API to search, scrape, and interact with the web at scale. Turn any source into clean Markdown or structured data your agents can ship with.",[],1785695930,[46,78,120,137],{"id":47,"name":48,"slug":49,"description":50,"url":51,"githubUrl":52,"logoUrl":53,"openSource":11,"categories":54,"repo":56,"company":-1,"articles":76,"createdAt":77},"entity_01ky0s7wgrf239sywgdkw7e8gz","DocETL","docetl","Open-source toolkit, built by the EPIC Data Lab at UC Berkeley, for creating LLM-powered pipelines that extract, transform, and link knowledge from unstructured documents.","https:\u002F\u002Fwww.docetl.org\u002F","https:\u002F\u002Fgithub.com\u002Fucbepic\u002Fdocetl","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fdocetl-favicon-color-9d034c86.png",[55],{"id":-1,"name":14,"slug":15,"count":16},{"fullName":57,"url":52,"stars":58,"forks":59,"openIssues":60,"language":61,"license":23,"topics":62,"pushedAt":75,"archived":36},"ucbepic\u002Fdocetl",3964,423,43,"Python",[63,64,65,66,67,68,69,70,32,71,72,73,74],"agents","data","data-pipelines","document-analysis","document-processing","elt","etl","llm","semantic-data","unstructured-data","unstructured-data-analysis","workflow","2026-08-10T15:00:22Z",[],1784585384,{"id":79,"name":80,"slug":81,"description":82,"url":83,"githubUrl":84,"logoUrl":85,"openSource":11,"categories":86,"repo":91,"company":-1,"articles":106,"createdAt":119},"entity_01kzs3gy7rfantmwgtrzpzt8yn","ExtractBench","extractbench","A benchmark for schema-guided extraction from real enterprise documents. 370 documents, 4,869 pages, 67 document types, each with its own JSON Schema. Scored on value accuracy, completeness, and evidence.","https:\u002F\u002Fwww.extractbench.ai\u002F","https:\u002F\u002Fgithub.com\u002Frun-llama\u002FExtractBench","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Fapple-touch-icon-da922544.png",[87,90],{"id":-1,"name":88,"slug":89,"count":16},"Evals","evals",{"id":-1,"name":14,"slug":15,"count":16},{"fullName":92,"url":84,"stars":93,"forks":94,"openIssues":16,"language":61,"license":95,"topics":96,"pushedAt":105,"archived":36},"run-llama\u002FExtractBench",39,4,"Apache-2.0",[97,98,99,100,101,102,103,70,104],"benchmark","coding-agent","document-ai","evaluation","extract","extract-data","llamaindex","vision-language-model","2026-08-08T18:34:23Z",[107],{"id":108,"name":109,"slug":110,"summary":111,"url":112,"kind":113,"platform":114,"author":-1,"authorHandle":-1,"publisher":115,"publishedAt":116,"about":117,"writtenBy":118,"createdAt":-1},"entity_01kzs3krzxfantmwj9zrcz3hbt","ExtractBench: The Most Comprehensive Extraction Benchmark","introducing-extractbench","The most comprehensive document extraction benchmark: 14 systems scored on accuracy, completeness, grounding, and cost across 370 enterprise documents.","https:\u002F\u002Fwww.llamaindex.ai\u002Fblog\u002Fintroducing-extractbench","announcement","web","LlamaIndex","2026-08-11",[],[],1786475215,{"id":121,"name":122,"slug":123,"description":124,"url":125,"githubUrl":125,"logoUrl":-1,"openSource":11,"categories":126,"repo":128,"company":-1,"articles":135,"createdAt":136},"entity_01kzhfcs4xeh2vtra99mntepyy","Knowledge Graph Builder","knowledge-graph-builder","Repository for building knowledge graphs from specific datasets using generative language model through ollama","https:\u002F\u002Fgithub.com\u002FFabioYanezRomero\u002FKnowledge-Graph-Builder",[127],{"id":-1,"name":14,"slug":15,"count":16},{"fullName":129,"url":125,"stars":130,"forks":131,"openIssues":132,"language":61,"license":23,"topics":133,"pushedAt":134,"archived":36},"FabioYanezRomero\u002FKnowledge-Graph-Builder",98,12,2,[],"2026-07-04T18:27:20Z",[],1786219226,{"id":138,"name":139,"slug":140,"description":141,"url":142,"githubUrl":143,"logoUrl":144,"openSource":11,"categories":145,"repo":147,"company":-1,"articles":163,"createdAt":164},"entity_01ky09k25gfanafx44rj2f9dpf","LangExtract","langextract","A Python library for extracting structured information from unstructured text using LLMs with precise source grounding and interactive visualization.","https:\u002F\u002Fpypi.org\u002Fproject\u002Flangextract\u002F","https:\u002F\u002Fgithub.com\u002Fgoogle\u002Flangextract","https:\u002F\u002Fr2.minima.ltd\u002Forg_01jakhtww5fk6b0jmxj0qfp9rf\u002Ffavicon.35549fe8-6dd120da.ico",[146],{"id":-1,"name":14,"slug":15,"count":16},{"fullName":148,"url":143,"stars":149,"forks":150,"openIssues":151,"language":61,"license":95,"topics":152,"pushedAt":162,"archived":36},"google\u002Flangextract",38325,2685,120,[153,154,155,156,157,158,159,70,160,32,161],"gemini","gemini-ai","gemini-api","gemini-flash","gemini-pro","information-extration","large-language-models","nlp","structured-data","2026-08-11T15:31:39Z",[],1784568973]