{"adoption": {"forks": 142, "observed_at": "2026-08-28T04:04:43.745313+00:00", "stars": 1437}, "canonical_url": "https://ross.abutalabs.com/products/openocr", "card": {"archived": false, "artifact_type": "library", "description": "OpenOCR: An Open-Source Toolkit for General-OCR Research and Applications, integrates a unified training and evaluation benchmark, commercial-grade OCR and Document Parsing systems, and faithful reproductions of the core implementations from a wide range of academic papers.", "domain": ["computer-vision", "artificial-intelligence", "pdf"], "enriched": true, "function": ["ocr", "machine-learning", "deep-learning", "image-processing", "pdf"], "health_score": 88, "homepage": null, "language": "Python", "license": "Apache-2.0", "license_family": "permissive", "maturity": "active", "member_repos": ["Topdu/OpenOCR"], "name": "Topdu/OpenOCR", "platform": ["python", "windows"], "pushed_at": "2026-08-04T04:54:39+00:00", "repo": "Topdu/OpenOCR", "stars": 1437, "tags": ["text-detection", "text-recognition", "document-parsing", "table-recognition", "formula-recognition", "scene-text", "pytorch", "research-benchmark", "natural-language-processing", "linux", "macos", "gpu"], "topics": ["chineseocr", "ocr", "ocr-pytorch", "scene-text-detection", "scene-text-recognition", "document-analysis", "document-parsing", "document-processing"], "urls": [], "use_cases": ["extract text from images with ocr", "parse documents into structured text", "recognize tables in scanned pdfs", "detect and recognize scene text in photos", "convert formulas in images to latex", "benchmark ocr models on standard datasets", "train custom text recognition models in pytorch", "chinese and english ocr"], "what_it_is": "OpenOCR is an open-source Python toolkit for general OCR research and applications, covering text detection and recognition, formula and table recognition, and document parsing. It provides commercial-grade OCR systems, a unified training/evaluation benchmark, and reproductions of academic paper implementations.", "when_to_avoid": ["you only need speech-to-text rather than visual text recognition", "you need handwriting-heavy or low-resource language OCR beyond the toolkit's supported models", "you want a fully managed cloud OCR API without running models yourself"], "when_to_choose": ["you need a full OCR pipeline including detection, recognition, and document parsing", "you want reproducible implementations of OCR research papers for benchmarking", "you need commercial-grade OCR with lightweight deployable models", "you work with Chinese and English text recognition"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/openocr", "repo": "Topdu/OpenOCR", "role": "main", "score": 58}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:04:43.745313+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T04:36:42.794759+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2d7a60f71be546f76231e7341bcdf817c89286d085bd9a6679728f75f1e8af88", "fetched_at": "2026-08-28T04:04:43.745313+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Topdu/OpenOCR"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:04:43.745313+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T04:36:42.794759+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2d7a60f71be546f76231e7341bcdf817c89286d085bd9a6679728f75f1e8af88", "fetched_at": "2026-08-28T04:04:43.745313+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Topdu/OpenOCR"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T04:36:42.794759+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2d7a60f71be546f76231e7341bcdf817c89286d085bd9a6679728f75f1e8af88", "fetched_at": "2026-08-28T04:04:43.745313+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Topdu/OpenOCR"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:04:43.745313+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:04:43.745313+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:04:43.745313+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T04:36:42.794759+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2d7a60f71be546f76231e7341bcdf817c89286d085bd9a6679728f75f1e8af88", "fetched_at": "2026-08-28T04:04:43.745313+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Topdu/OpenOCR"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:04:43.745313+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:04:43.745313+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T04:36:42.794759+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2d7a60f71be546f76231e7341bcdf817c89286d085bd9a6679728f75f1e8af88", "fetched_at": "2026-08-28T04:04:43.745313+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Topdu/OpenOCR"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:04:43.745313+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:04:43.745313+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:04:43.745313+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T04:36:42.794759+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2d7a60f71be546f76231e7341bcdf817c89286d085bd9a6679728f75f1e8af88", "fetched_at": "2026-08-28T04:04:43.745313+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Topdu/OpenOCR"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:04:43.745313+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:04:43.745313+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T04:36:42.794759+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2d7a60f71be546f76231e7341bcdf817c89286d085bd9a6679728f75f1e8af88", "fetched_at": "2026-08-28T04:04:43.745313+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Topdu/OpenOCR"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T04:36:42.794759+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2d7a60f71be546f76231e7341bcdf817c89286d085bd9a6679728f75f1e8af88", "fetched_at": "2026-08-28T04:04:43.745313+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Topdu/OpenOCR"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T04:36:42.794759+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2d7a60f71be546f76231e7341bcdf817c89286d085bd9a6679728f75f1e8af88", "fetched_at": "2026-08-28T04:04:43.745313+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Topdu/OpenOCR"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T04:36:42.794759+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "2d7a60f71be546f76231e7341bcdf817c89286d085bd9a6679728f75f1e8af88", "fetched_at": "2026-08-28T04:04:43.745313+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Topdu/OpenOCR"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 96, "longevity": 58, "rhythm": 8}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 824, "days_push": 29, "days_rel": 648, "gap_med": null, "n_releases_24m": 1}, "score": 58, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}