{"adoption": {"forks": 167, "observed_at": "2026-08-28T04:03:25.834551+00:00", "stars": 1060}, "canonical_url": "https://ross.abutalabs.com/products/publaynet", "card": {"archived": false, "artifact_type": "dataset", "description": null, "domain": ["computer-vision", "machine-learning"], "enriched": true, "function": ["computer-vision", "ocr", "machine-learning", "data-science"], "health_score": 38, "homepage": null, "language": "Jupyter Notebook", "license": "NOASSERTION", "license_family": "other", "maturity": "maintenance", "member_repos": ["ibm-aur-nlp/PubLayNet"], "name": "ibm-aur-nlp/PubLayNet", "platform": ["python", "cross-platform"], "pushed_at": "2025-07-09T01:13:59+00:00", "repo": "ibm-aur-nlp/PubLayNet", "stars": 1060, "tags": ["document-layout-analysis", "object-detection", "image-segmentation", "scientific-documents", "pubmed", "faster-rcnn", "mask-rcnn", "icdar", "natural-language-processing", "datasets"], "topics": [], "urls": [], "use_cases": ["train a document layout detection model", "segment text, tables, and figures in scientific paper images", "benchmark object detection models on document images", "extract structure from PDF pages of research articles", "download pre-trained Mask-RCNN for document analysis", "participate in scientific literature parsing competitions"], "what_it_is": "PubLayNet is a large annotated dataset of over 360k document images from PubMed Central, with bounding boxes and polygonal segmentations for layout elements like text, titles, lists, tables, and figures. The repository also hosts pre-trained Faster-RCNN and Mask-RCNN models and ICDAR 2021 Scientific Literature Parsing competition materials.", "when_to_avoid": ["you need table structure recognition rather than layout detection (see PubTabNet)", "you need annotated documents outside the scientific/medical domain", "you need a ready-made application rather than a dataset and models"], "when_to_choose": ["you need large-scale labeled data for document layout analysis", "you are training or evaluating object detection models on scientific documents", "you want pre-trained models for parsing PubMed-style paper layouts"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/publaynet", "repo": "ibm-aur-nlp/PubLayNet", "role": "main", "score": 46}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:03:25.834551+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T06:56:56.697436+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "58969915b22847a6a12940213e0a9fea5b5933a24fc820211939587e0318cd74", "fetched_at": "2026-08-28T04:03:25.834551+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ibm-aur-nlp/PubLayNet"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:03:25.834551+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T06:56:56.697436+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "58969915b22847a6a12940213e0a9fea5b5933a24fc820211939587e0318cd74", "fetched_at": "2026-08-28T04:03:25.834551+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ibm-aur-nlp/PubLayNet"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T06:56:56.697436+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "58969915b22847a6a12940213e0a9fea5b5933a24fc820211939587e0318cd74", "fetched_at": "2026-08-28T04:03:25.834551+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ibm-aur-nlp/PubLayNet"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:03:25.834551+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:03:25.834551+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:03:25.834551+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T06:56:56.697436+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "58969915b22847a6a12940213e0a9fea5b5933a24fc820211939587e0318cd74", "fetched_at": "2026-08-28T04:03:25.834551+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ibm-aur-nlp/PubLayNet"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:03:25.834551+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:03:25.834551+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T06:56:56.697436+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "58969915b22847a6a12940213e0a9fea5b5933a24fc820211939587e0318cd74", "fetched_at": "2026-08-28T04:03:25.834551+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ibm-aur-nlp/PubLayNet"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:03:25.834551+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:03:25.834551+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:03:25.834551+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T06:56:56.697436+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "58969915b22847a6a12940213e0a9fea5b5933a24fc820211939587e0318cd74", "fetched_at": "2026-08-28T04:03:25.834551+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ibm-aur-nlp/PubLayNet"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:03:25.834551+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:03:25.834551+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T06:56:56.697436+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "58969915b22847a6a12940213e0a9fea5b5933a24fc820211939587e0318cd74", "fetched_at": "2026-08-28T04:03:25.834551+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ibm-aur-nlp/PubLayNet"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T06:56:56.697436+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "58969915b22847a6a12940213e0a9fea5b5933a24fc820211939587e0318cd74", "fetched_at": "2026-08-28T04:03:25.834551+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ibm-aur-nlp/PubLayNet"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T06:56:56.697436+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "58969915b22847a6a12940213e0a9fea5b5933a24fc820211939587e0318cd74", "fetched_at": "2026-08-28T04:03:25.834551+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ibm-aur-nlp/PubLayNet"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T06:56:56.697436+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "58969915b22847a6a12940213e0a9fea5b5933a24fc820211939587e0318cd74", "fetched_at": "2026-08-28T04:03:25.834551+00:00", "kind": "readme", "missing": false, "url": "https://github.com/ibm-aur-nlp/PubLayNet"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 30, "longevity": 100, "rhythm": 35}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": ["no_releases", "no_license"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 2680, "days_push": 421, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 46, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}