{"adoption": {"forks": 750, "observed_at": "2026-08-28T04:10:39.185358+00:00", "stars": 9993}, "canonical_url": "https://ross.abutalabs.com/products/pdf-extract-kit", "card": {"archived": false, "artifact_type": "library", "description": "A Comprehensive Toolkit for High-Quality PDF Content Extraction", "domain": ["pdf", "machine-learning", "developer-tools"], "enriched": true, "function": ["ocr", "pdf", "machine-learning", "parser", "image-processing"], "health_score": 29, "homepage": "https://pdf-extract-kit.readthedocs.io/zh-cn/latest/index.html", "language": "Python", "license": "AGPL-3.0", "license_family": "copyleft", "maturity": "active", "member_repos": ["opendatalab/PDF-Extract-Kit"], "name": "opendatalab/PDF-Extract-Kit", "platform": ["python", "windows"], "pushed_at": "2025-01-03T02:00:20+00:00", "repo": "opendatalab/PDF-Extract-Kit", "stars": 9993, "tags": ["pdf-extraction", "layout-detection", "formula-recognition", "table-recognition", "document-parsing", "model-toolbox", "natural-language-processing", "linux", "macos", "gpu"], "topics": [], "urls": [], "use_cases": ["extract text and structure from complex pdf documents", "convert scanned pdfs to markdown with ocr", "detect and recognize math formulas in pdfs", "recognize tables in pdf documents", "detect document layout regions in pdfs", "build document qa or translation apps on top of pdf parsing models", "benchmark pdf parsing models on evaluation datasets"], "what_it_is": "PDF-Extract-Kit is a Python model toolbox for high-quality PDF content extraction, integrating state-of-the-art models for layout detection, formula detection and recognition, OCR, table recognition, and reading order. Its modular design lets developers combine components to build applications like document translation or Q&A, while end users seeking PDF-to-Markdown conversion are pointed to the companion MinerU tool.", "when_to_avoid": ["you just want a turnkey pdf-to-markdown converter - use MinerU instead", "you need a simple text extraction utility without deep learning models", "you cannot run GPU or heavyweight ML models in your environment", "your project requires a permissive license - this is AGPL-3.0"], "when_to_choose": ["you need modular, model-level building blocks for document parsing tasks", "you want state-of-the-art layout, formula, OCR, and table recognition models with benchmarks", "you are a developer building custom document processing applications", "you need fine-grained control over individual parsing stages"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/pdf-extract-kit", "repo": "opendatalab/PDF-Extract-Kit", "role": "main", "score": 25}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:10:39.185358+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T17:20:01.476065+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cf05af0ba81f365b301fc92e60d5c97e0d358a9122a5f000652de377cf783947", "fetched_at": "2026-08-28T04:10:39.185358+00:00", "kind": "readme", "missing": false, "url": "https://github.com/opendatalab/PDF-Extract-Kit"}, {"content_hash": "5f2cd7992d2c753bb1f3e651938bcf8243679d7fc81037851d7e5d706dcac330", "fetched_at": "2026-08-29T08:19:46.868449+00:00", "kind": "homepage", "missing": false, "url": "https://pdf-extract-kit.readthedocs.io/zh-cn/latest/index.html"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:10:39.185358+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T17:20:01.476065+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cf05af0ba81f365b301fc92e60d5c97e0d358a9122a5f000652de377cf783947", "fetched_at": "2026-08-28T04:10:39.185358+00:00", "kind": "readme", "missing": false, "url": "https://github.com/opendatalab/PDF-Extract-Kit"}, {"content_hash": "5f2cd7992d2c753bb1f3e651938bcf8243679d7fc81037851d7e5d706dcac330", "fetched_at": "2026-08-29T08:19:46.868449+00:00", "kind": "homepage", "missing": false, "url": "https://pdf-extract-kit.readthedocs.io/zh-cn/latest/index.html"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T17:20:01.476065+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cf05af0ba81f365b301fc92e60d5c97e0d358a9122a5f000652de377cf783947", "fetched_at": "2026-08-28T04:10:39.185358+00:00", "kind": "readme", "missing": false, "url": "https://github.com/opendatalab/PDF-Extract-Kit"}, {"content_hash": "5f2cd7992d2c753bb1f3e651938bcf8243679d7fc81037851d7e5d706dcac330", "fetched_at": "2026-08-29T08:19:46.868449+00:00", "kind": "homepage", "missing": false, "url": "https://pdf-extract-kit.readthedocs.io/zh-cn/latest/index.html"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:10:39.185358+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:10:39.185358+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:10:39.185358+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T17:20:01.476065+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cf05af0ba81f365b301fc92e60d5c97e0d358a9122a5f000652de377cf783947", "fetched_at": "2026-08-28T04:10:39.185358+00:00", "kind": "readme", "missing": false, "url": "https://github.com/opendatalab/PDF-Extract-Kit"}, {"content_hash": "5f2cd7992d2c753bb1f3e651938bcf8243679d7fc81037851d7e5d706dcac330", "fetched_at": "2026-08-29T08:19:46.868449+00:00", "kind": "homepage", "missing": false, "url": "https://pdf-extract-kit.readthedocs.io/zh-cn/latest/index.html"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:10:39.185358+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:10:39.185358+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T17:20:01.476065+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cf05af0ba81f365b301fc92e60d5c97e0d358a9122a5f000652de377cf783947", "fetched_at": "2026-08-28T04:10:39.185358+00:00", "kind": "readme", "missing": false, "url": "https://github.com/opendatalab/PDF-Extract-Kit"}, {"content_hash": "5f2cd7992d2c753bb1f3e651938bcf8243679d7fc81037851d7e5d706dcac330", "fetched_at": "2026-08-29T08:19:46.868449+00:00", "kind": "homepage", "missing": false, "url": "https://pdf-extract-kit.readthedocs.io/zh-cn/latest/index.html"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:10:39.185358+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:10:39.185358+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:10:39.185358+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T17:20:01.476065+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cf05af0ba81f365b301fc92e60d5c97e0d358a9122a5f000652de377cf783947", "fetched_at": "2026-08-28T04:10:39.185358+00:00", "kind": "readme", "missing": false, "url": "https://github.com/opendatalab/PDF-Extract-Kit"}, {"content_hash": "5f2cd7992d2c753bb1f3e651938bcf8243679d7fc81037851d7e5d706dcac330", "fetched_at": "2026-08-29T08:19:46.868449+00:00", "kind": "homepage", "missing": false, "url": "https://pdf-extract-kit.readthedocs.io/zh-cn/latest/index.html"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:10:39.185358+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:10:39.185358+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T17:20:01.476065+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cf05af0ba81f365b301fc92e60d5c97e0d358a9122a5f000652de377cf783947", "fetched_at": "2026-08-28T04:10:39.185358+00:00", "kind": "readme", "missing": false, "url": "https://github.com/opendatalab/PDF-Extract-Kit"}, {"content_hash": "5f2cd7992d2c753bb1f3e651938bcf8243679d7fc81037851d7e5d706dcac330", "fetched_at": "2026-08-29T08:19:46.868449+00:00", "kind": "homepage", "missing": false, "url": "https://pdf-extract-kit.readthedocs.io/zh-cn/latest/index.html"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T17:20:01.476065+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cf05af0ba81f365b301fc92e60d5c97e0d358a9122a5f000652de377cf783947", "fetched_at": "2026-08-28T04:10:39.185358+00:00", "kind": "readme", "missing": false, "url": "https://github.com/opendatalab/PDF-Extract-Kit"}, {"content_hash": "5f2cd7992d2c753bb1f3e651938bcf8243679d7fc81037851d7e5d706dcac330", "fetched_at": "2026-08-29T08:19:46.868449+00:00", "kind": "homepage", "missing": false, "url": "https://pdf-extract-kit.readthedocs.io/zh-cn/latest/index.html"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T17:20:01.476065+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cf05af0ba81f365b301fc92e60d5c97e0d358a9122a5f000652de377cf783947", "fetched_at": "2026-08-28T04:10:39.185358+00:00", "kind": "readme", "missing": false, "url": "https://github.com/opendatalab/PDF-Extract-Kit"}, {"content_hash": "5f2cd7992d2c753bb1f3e651938bcf8243679d7fc81037851d7e5d706dcac330", "fetched_at": "2026-08-29T08:19:46.868449+00:00", "kind": "homepage", "missing": false, "url": "https://pdf-extract-kit.readthedocs.io/zh-cn/latest/index.html"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T17:20:01.476065+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "cf05af0ba81f365b301fc92e60d5c97e0d358a9122a5f000652de377cf783947", "fetched_at": "2026-08-28T04:10:39.185358+00:00", "kind": "readme", "missing": false, "url": "https://github.com/opendatalab/PDF-Extract-Kit"}, {"content_hash": "5f2cd7992d2c753bb1f3e651938bcf8243679d7fc81037851d7e5d706dcac330", "fetched_at": "2026-08-29T08:19:46.868449+00:00", "kind": "homepage", "missing": false, "url": "https://pdf-extract-kit.readthedocs.io/zh-cn/latest/index.html"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 0, "longevity": 56, "rhythm": 40}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 797, "days_push": 608, "days_rel": 691, "gap_med": 14.5, "n_releases_24m": 3}, "score": 25, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}