{"adoption": {"forks": 197, "observed_at": "2026-08-28T04:05:42.528594+00:00", "stars": 1834}, "canonical_url": "https://ross.abutalabs.com/products/advancedliteratemachinery", "card": {"archived": false, "artifact_type": "library", "description": "A collection of original, innovative ideas and algorithms towards Advanced Literate Machinery. This project is maintained by the OCR Team in the Language Technology Lab, Tongyi Lab, Alibaba Group.", "domain": ["computer-vision", "artificial-intelligence", "image-processing", "pdf"], "enriched": true, "function": ["ocr", "computer-vision", "machine-learning", "nlp", "data-science"], "health_score": 70, "homepage": null, "language": "C++", "license": "Apache-2.0", "license_family": "permissive", "maturity": "active", "member_repos": ["AlibabaResearch/AdvancedLiterateMachinery"], "name": "AlibabaResearch/AdvancedLiterateMachinery", "platform": ["python", "cpp"], "pushed_at": "2026-03-17T02:55:55+00:00", "repo": "AlibabaResearch/AdvancedLiterateMachinery", "stars": 1834, "tags": ["document-ai", "scene-text-recognition", "vision-language-models", "benchmark", "research-code", "multimodal", "natural-language-processing", "linux", "gpu"], "topics": ["artificial-intelligence", "documentai", "multimodal", "multimodal-deep-learning", "ocr", "computer-vision", "vision-language-transformer", "end-to-end-ocr", "scene-text-detection", "scene-text-detection-recognition", "scene-text-recognition", "text-detection", "text-recognition", "vision-language", "document", "document-analysis", "document-recognition", "document-understanding", "document-intelligence", "vision-language-model"], "urls": [], "use_cases": ["extract text from images with ocr", "recognize scene text in photos", "parse and understand documents", "evaluate multimodal llms on ocr benchmarks", "extract key information from scanned documents", "generate synthetic images containing text", "multilingual text recognition"], "what_it_is": "A collection of original OCR and document understanding models, algorithms, and benchmarks from Alibaba's Tongyi Lab, including models like Platypus and the CC-OCR benchmark for evaluating large multimodal models. It provides research code and pretrained models for reading text from images and documents.", "when_to_avoid": ["you need a simple production OCR API with commercial support", "you want a lightweight plug-and-play text extraction tool without GPU resources", "your project is unrelated to text reading or document intelligence"], "when_to_choose": ["you need state-of-the-art OCR or document parsing models", "you want to benchmark large multimodal models on OCR tasks", "you are researching scene text detection, recognition, or visual text generation", "you need key information extraction from documents"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/advancedliteratemachinery", "repo": "AlibabaResearch/AdvancedLiterateMachinery", "role": "main", "score": 55}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.528594+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T03:18:33.149574+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "be8640ffe746b1df2211f5bce0f8100a2bd01a4051ca93e534415ac39aaf0622", "fetched_at": "2026-08-28T04:05:42.528594+00:00", "kind": "readme", "missing": false, "url": "https://github.com/AlibabaResearch/AdvancedLiterateMachinery"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.528594+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T03:18:33.149574+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "be8640ffe746b1df2211f5bce0f8100a2bd01a4051ca93e534415ac39aaf0622", "fetched_at": "2026-08-28T04:05:42.528594+00:00", "kind": "readme", "missing": false, "url": "https://github.com/AlibabaResearch/AdvancedLiterateMachinery"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T03:18:33.149574+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "be8640ffe746b1df2211f5bce0f8100a2bd01a4051ca93e534415ac39aaf0622", "fetched_at": "2026-08-28T04:05:42.528594+00:00", "kind": "readme", "missing": false, "url": "https://github.com/AlibabaResearch/AdvancedLiterateMachinery"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.528594+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.528594+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.528594+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T03:18:33.149574+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "be8640ffe746b1df2211f5bce0f8100a2bd01a4051ca93e534415ac39aaf0622", "fetched_at": "2026-08-28T04:05:42.528594+00:00", "kind": "readme", "missing": false, "url": "https://github.com/AlibabaResearch/AdvancedLiterateMachinery"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.528594+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.528594+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T03:18:33.149574+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "be8640ffe746b1df2211f5bce0f8100a2bd01a4051ca93e534415ac39aaf0622", "fetched_at": "2026-08-28T04:05:42.528594+00:00", "kind": "readme", "missing": false, "url": "https://github.com/AlibabaResearch/AdvancedLiterateMachinery"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.528594+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.528594+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.528594+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T03:18:33.149574+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "be8640ffe746b1df2211f5bce0f8100a2bd01a4051ca93e534415ac39aaf0622", "fetched_at": "2026-08-28T04:05:42.528594+00:00", "kind": "readme", "missing": false, "url": "https://github.com/AlibabaResearch/AdvancedLiterateMachinery"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.528594+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:05:42.528594+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T03:18:33.149574+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "be8640ffe746b1df2211f5bce0f8100a2bd01a4051ca93e534415ac39aaf0622", "fetched_at": "2026-08-28T04:05:42.528594+00:00", "kind": "readme", "missing": false, "url": "https://github.com/AlibabaResearch/AdvancedLiterateMachinery"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T03:18:33.149574+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "be8640ffe746b1df2211f5bce0f8100a2bd01a4051ca93e534415ac39aaf0622", "fetched_at": "2026-08-28T04:05:42.528594+00:00", "kind": "readme", "missing": false, "url": "https://github.com/AlibabaResearch/AdvancedLiterateMachinery"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T03:18:33.149574+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "be8640ffe746b1df2211f5bce0f8100a2bd01a4051ca93e534415ac39aaf0622", "fetched_at": "2026-08-28T04:05:42.528594+00:00", "kind": "readme", "missing": false, "url": "https://github.com/AlibabaResearch/AdvancedLiterateMachinery"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T03:18:33.149574+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "be8640ffe746b1df2211f5bce0f8100a2bd01a4051ca93e534415ac39aaf0622", "fetched_at": "2026-08-28T04:05:42.528594+00:00", "kind": "readme", "missing": false, "url": "https://github.com/AlibabaResearch/AdvancedLiterateMachinery"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 72, "longevity": 100, "rhythm": 8}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 1435, "days_push": 169, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 55, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}