{"adoption": {"forks": 155, "observed_at": "2026-08-28T04:06:11.908367+00:00", "stars": 2085}, "canonical_url": "https://ross.abutalabs.com/products/docext", "card": {"archived": false, "artifact_type": "library", "description": "An on-premises, OCR-free unstructured data extraction, markdown conversion and benchmarking toolkit. (https://idp-leaderboard.org/)", "domain": ["machine-learning", "pdf", "developer-tools", "artificial-intelligence"], "enriched": true, "function": ["ocr", "nlp", "machine-learning", "pdf", "benchmarking", "rag"], "health_score": 77, "homepage": "https://nanonets.com/document-parsing-and-extraction", "language": "Python", "license": "Apache-2.0", "license_family": "permissive", "maturity": "active", "member_repos": ["NanoNets/docext"], "name": "NanoNets/docext", "platform": ["python", "self-hosted", "cross-platform"], "pushed_at": "2026-03-17T09:41:55+00:00", "repo": "NanoNets/docext", "stars": 2085, "tags": ["vision-language-models", "document-intelligence", "information-extraction", "markdown-conversion", "on-premises", "key-information-extraction", "table-extraction", "idp-leaderboard", "natural-language-processing"], "topics": ["document", "document-analysis", "extraction", "llms", "machine-learning", "nlp", "ocr", "rag", "unstructured-data", "vlms", "onprem", "document-data-extraction", "ocr-onpremise", "llm-ocr", "onprem-ocr", "onprem-vision", "onpremise", "table-extraction", "document-information-extraction", "ocr-benchmark"], "urls": [], "use_cases": ["extract structured fields from invoices and passports without OCR", "convert PDFs and scanned images to markdown with tables and LaTeX equations", "detect signatures and watermarks in documents", "benchmark vision-language models on document extraction tasks", "run document data extraction fully on-premises for privacy", "extract tables from unstructured documents with confidence scores"], "what_it_is": "docext is an on-premises document intelligence toolkit powered by vision-language models, offering OCR-free structured data extraction, PDF/image-to-markdown conversion, and a benchmarking leaderboard for document processing tasks. It is a Python library (pip-installable) from NanoNets that also ships the Nanonets-OCR-s model for image-to-markdown conversion.", "when_to_avoid": ["you need a lightweight traditional OCR engine like Tesseract", "you lack GPU resources for running vision-language models", "you only need simple text extraction from digital PDFs", "you need a managed cloud service with SLAs"], "when_to_choose": ["you need on-premises/private document extraction without sending data to cloud APIs", "you want OCR-free structured extraction using vision-language models", "you need PDF/image to markdown conversion with semantic tagging", "you want to benchmark VLMs on IDP tasks like KIE and table extraction"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/docext", "repo": "NanoNets/docext", "role": "main", "score": 50}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:06:11.908367+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T02:55:49.783846+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "91b56618a3404ad5e40ee2aaec25c645d1a61c8a953dbdc36319bff1a322709e", "fetched_at": "2026-08-28T04:06:11.908367+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NanoNets/docext"}, {"content_hash": "4a900fd45d4870d3cc76005da69cd0296dab7ee123576c0a4edf9c5c85a9da1c", "fetched_at": "2026-08-29T10:35:46.539658+00:00", "kind": "homepage", "missing": false, "url": "https://nanonets.com/document-parsing-and-extraction"}, {"content_hash": "3068d093752b51986335c9c152ec13718f06d52aeb87ba96b818379fa9fae0a5", "fetched_at": "2026-08-29T10:35:46.548441+00:00", "kind": "registry_pypi", "missing": false, "url": "https://pypi.org/pypi/docext/json"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:06:11.908367+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T02:55:49.783846+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "91b56618a3404ad5e40ee2aaec25c645d1a61c8a953dbdc36319bff1a322709e", "fetched_at": "2026-08-28T04:06:11.908367+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NanoNets/docext"}, {"content_hash": "4a900fd45d4870d3cc76005da69cd0296dab7ee123576c0a4edf9c5c85a9da1c", "fetched_at": "2026-08-29T10:35:46.539658+00:00", "kind": "homepage", "missing": false, "url": "https://nanonets.com/document-parsing-and-extraction"}, {"content_hash": "3068d093752b51986335c9c152ec13718f06d52aeb87ba96b818379fa9fae0a5", "fetched_at": "2026-08-29T10:35:46.548441+00:00", "kind": "registry_pypi", "missing": false, "url": "https://pypi.org/pypi/docext/json"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T02:55:49.783846+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "91b56618a3404ad5e40ee2aaec25c645d1a61c8a953dbdc36319bff1a322709e", "fetched_at": "2026-08-28T04:06:11.908367+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NanoNets/docext"}, {"content_hash": "4a900fd45d4870d3cc76005da69cd0296dab7ee123576c0a4edf9c5c85a9da1c", "fetched_at": "2026-08-29T10:35:46.539658+00:00", "kind": "homepage", "missing": false, "url": "https://nanonets.com/document-parsing-and-extraction"}, {"content_hash": "3068d093752b51986335c9c152ec13718f06d52aeb87ba96b818379fa9fae0a5", "fetched_at": "2026-08-29T10:35:46.548441+00:00", "kind": "registry_pypi", "missing": false, "url": "https://pypi.org/pypi/docext/json"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:06:11.908367+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:06:11.908367+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:06:11.908367+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T02:55:49.783846+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "91b56618a3404ad5e40ee2aaec25c645d1a61c8a953dbdc36319bff1a322709e", "fetched_at": "2026-08-28T04:06:11.908367+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NanoNets/docext"}, {"content_hash": "4a900fd45d4870d3cc76005da69cd0296dab7ee123576c0a4edf9c5c85a9da1c", "fetched_at": "2026-08-29T10:35:46.539658+00:00", "kind": "homepage", "missing": false, "url": "https://nanonets.com/document-parsing-and-extraction"}, {"content_hash": "3068d093752b51986335c9c152ec13718f06d52aeb87ba96b818379fa9fae0a5", "fetched_at": "2026-08-29T10:35:46.548441+00:00", "kind": "registry_pypi", "missing": false, "url": "https://pypi.org/pypi/docext/json"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:06:11.908367+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:06:11.908367+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T02:55:49.783846+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "91b56618a3404ad5e40ee2aaec25c645d1a61c8a953dbdc36319bff1a322709e", "fetched_at": "2026-08-28T04:06:11.908367+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NanoNets/docext"}, {"content_hash": "4a900fd45d4870d3cc76005da69cd0296dab7ee123576c0a4edf9c5c85a9da1c", "fetched_at": "2026-08-29T10:35:46.539658+00:00", "kind": "homepage", "missing": false, "url": "https://nanonets.com/document-parsing-and-extraction"}, {"content_hash": "3068d093752b51986335c9c152ec13718f06d52aeb87ba96b818379fa9fae0a5", "fetched_at": "2026-08-29T10:35:46.548441+00:00", "kind": "registry_pypi", "missing": false, "url": "https://pypi.org/pypi/docext/json"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:06:11.908367+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:06:11.908367+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:06:11.908367+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T02:55:49.783846+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "91b56618a3404ad5e40ee2aaec25c645d1a61c8a953dbdc36319bff1a322709e", "fetched_at": "2026-08-28T04:06:11.908367+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NanoNets/docext"}, {"content_hash": "4a900fd45d4870d3cc76005da69cd0296dab7ee123576c0a4edf9c5c85a9da1c", "fetched_at": "2026-08-29T10:35:46.539658+00:00", "kind": "homepage", "missing": false, "url": "https://nanonets.com/document-parsing-and-extraction"}, {"content_hash": "3068d093752b51986335c9c152ec13718f06d52aeb87ba96b818379fa9fae0a5", "fetched_at": "2026-08-29T10:35:46.548441+00:00", "kind": "registry_pypi", "missing": false, "url": "https://pypi.org/pypi/docext/json"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:06:11.908367+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:06:11.908367+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T02:55:49.783846+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "91b56618a3404ad5e40ee2aaec25c645d1a61c8a953dbdc36319bff1a322709e", "fetched_at": "2026-08-28T04:06:11.908367+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NanoNets/docext"}, {"content_hash": "4a900fd45d4870d3cc76005da69cd0296dab7ee123576c0a4edf9c5c85a9da1c", "fetched_at": "2026-08-29T10:35:46.539658+00:00", "kind": "homepage", "missing": false, "url": "https://nanonets.com/document-parsing-and-extraction"}, {"content_hash": "3068d093752b51986335c9c152ec13718f06d52aeb87ba96b818379fa9fae0a5", "fetched_at": "2026-08-29T10:35:46.548441+00:00", "kind": "registry_pypi", "missing": false, "url": "https://pypi.org/pypi/docext/json"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T02:55:49.783846+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "91b56618a3404ad5e40ee2aaec25c645d1a61c8a953dbdc36319bff1a322709e", "fetched_at": "2026-08-28T04:06:11.908367+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NanoNets/docext"}, {"content_hash": "4a900fd45d4870d3cc76005da69cd0296dab7ee123576c0a4edf9c5c85a9da1c", "fetched_at": "2026-08-29T10:35:46.539658+00:00", "kind": "homepage", "missing": false, "url": "https://nanonets.com/document-parsing-and-extraction"}, {"content_hash": "3068d093752b51986335c9c152ec13718f06d52aeb87ba96b818379fa9fae0a5", "fetched_at": "2026-08-29T10:35:46.548441+00:00", "kind": "registry_pypi", "missing": false, "url": "https://pypi.org/pypi/docext/json"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T02:55:49.783846+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "91b56618a3404ad5e40ee2aaec25c645d1a61c8a953dbdc36319bff1a322709e", "fetched_at": "2026-08-28T04:06:11.908367+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NanoNets/docext"}, {"content_hash": "4a900fd45d4870d3cc76005da69cd0296dab7ee123576c0a4edf9c5c85a9da1c", "fetched_at": "2026-08-29T10:35:46.539658+00:00", "kind": "homepage", "missing": false, "url": "https://nanonets.com/document-parsing-and-extraction"}, {"content_hash": "3068d093752b51986335c9c152ec13718f06d52aeb87ba96b818379fa9fae0a5", "fetched_at": "2026-08-29T10:35:46.548441+00:00", "kind": "registry_pypi", "missing": false, "url": "https://pypi.org/pypi/docext/json"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T02:55:49.783846+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "91b56618a3404ad5e40ee2aaec25c645d1a61c8a953dbdc36319bff1a322709e", "fetched_at": "2026-08-28T04:06:11.908367+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NanoNets/docext"}, {"content_hash": "4a900fd45d4870d3cc76005da69cd0296dab7ee123576c0a4edf9c5c85a9da1c", "fetched_at": "2026-08-29T10:35:46.539658+00:00", "kind": "homepage", "missing": false, "url": "https://nanonets.com/document-parsing-and-extraction"}, {"content_hash": "3068d093752b51986335c9c152ec13718f06d52aeb87ba96b818379fa9fae0a5", "fetched_at": "2026-08-29T10:35:46.548441+00:00", "kind": "registry_pypi", "missing": false, "url": "https://pypi.org/pypi/docext/json"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 72, "longevity": 37, "rhythm": 28}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 526, "days_push": 169, "days_rel": 429, "gap_med": 42.5, "n_releases_24m": 3}, "score": 50, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}