{"adoption": {"forks": 567, "observed_at": "2026-08-28T04:09:09.796336+00:00", "stars": 5105}, "canonical_url": "https://ross.abutalabs.com/products/grobid", "card": {"archived": false, "artifact_type": "library", "description": "A machine learning software for extracting information from scholarly documents", "domain": ["machine-learning", "pdf", "developer-tools"], "enriched": true, "function": ["machine-learning", "nlp", "pdf", "parser", "ocr"], "health_score": 100, "homepage": "https://grobid.readthedocs.io", "language": "Java", "license": "Apache-2.0", "license_family": "permissive", "maturity": "active", "member_repos": ["grobidOrg/grobid"], "name": "grobidOrg/grobid", "platform": ["jvm", "cross-platform"], "pushed_at": "2026-08-25T10:04:52+00:00", "repo": "grobidOrg/grobid", "stars": 5105, "tags": ["pdf-parsing", "tei-xml", "scholarly-documents", "bibliographic-references", "metadata-extraction", "deep-learning", "crf", "transformers", "natural-language-processing", "docker", "web-server"], "topics": ["machine-learning", "scientific-articles", "pdf", "metadata", "fulltext", "bibliographical-references", "hamburger-to-cow", "deep-learning", "rnn", "transformers", "crf"], "urls": [], "use_cases": ["extract metadata (title, authors, abstract) from scientific PDFs", "parse bibliographic references from scholarly articles", "convert PDF papers into structured TEI XML", "resolve citation contexts to full references", "segment and structure full text of academic PDFs", "batch process large corpora of scientific publications"], "what_it_is": "GROBID is a Java machine learning library that extracts, parses, and re-structures raw documents such as PDFs into structured XML/TEI, focused on technical and scientific publications. It provides header extraction, reference parsing, citation context resolution, and full-text document segmentation, with a REST API and Docker deployment.", "when_to_avoid": ["you only need simple text extraction without structure (pdftotext suffices)", "your documents are not technical/scientific publications", "you cannot run a JVM-based service or ML models", "you need extraction from scanned PDFs without OCR preprocessing"], "when_to_choose": ["you need high-accuracy extraction of bibliographic data from scholarly PDFs", "you want structured XML/TEI output from unstructured scientific documents", "you need a self-hosted, Docker-deployable PDF parsing service", "you are building scholarly search, citation analysis, or literature review tooling"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/grobid", "repo": "grobidOrg/grobid", "role": "main", "score": 87}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:09:09.796336+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T18:02:11.512700+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b6e85af9a6d2d9fed5af927ecc133c80effc2e38fcdeefade02257566ffeec26", "fetched_at": "2026-08-28T04:09:09.796336+00:00", "kind": "readme", "missing": false, "url": "https://github.com/grobidOrg/grobid"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:09:09.796336+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T18:02:11.512700+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b6e85af9a6d2d9fed5af927ecc133c80effc2e38fcdeefade02257566ffeec26", "fetched_at": "2026-08-28T04:09:09.796336+00:00", "kind": "readme", "missing": false, "url": "https://github.com/grobidOrg/grobid"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T18:02:11.512700+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b6e85af9a6d2d9fed5af927ecc133c80effc2e38fcdeefade02257566ffeec26", "fetched_at": "2026-08-28T04:09:09.796336+00:00", "kind": "readme", "missing": false, "url": "https://github.com/grobidOrg/grobid"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:09:09.796336+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:09:09.796336+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:09:09.796336+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T18:02:11.512700+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b6e85af9a6d2d9fed5af927ecc133c80effc2e38fcdeefade02257566ffeec26", "fetched_at": "2026-08-28T04:09:09.796336+00:00", "kind": "readme", "missing": false, "url": "https://github.com/grobidOrg/grobid"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:09:09.796336+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:09:09.796336+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T18:02:11.512700+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b6e85af9a6d2d9fed5af927ecc133c80effc2e38fcdeefade02257566ffeec26", "fetched_at": "2026-08-28T04:09:09.796336+00:00", "kind": "readme", "missing": false, "url": "https://github.com/grobidOrg/grobid"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:09:09.796336+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:09:09.796336+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:09:09.796336+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T18:02:11.512700+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b6e85af9a6d2d9fed5af927ecc133c80effc2e38fcdeefade02257566ffeec26", "fetched_at": "2026-08-28T04:09:09.796336+00:00", "kind": "readme", "missing": false, "url": "https://github.com/grobidOrg/grobid"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:09:09.796336+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:09:09.796336+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T18:02:11.512700+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b6e85af9a6d2d9fed5af927ecc133c80effc2e38fcdeefade02257566ffeec26", "fetched_at": "2026-08-28T04:09:09.796336+00:00", "kind": "readme", "missing": false, "url": "https://github.com/grobidOrg/grobid"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T18:02:11.512700+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b6e85af9a6d2d9fed5af927ecc133c80effc2e38fcdeefade02257566ffeec26", "fetched_at": "2026-08-28T04:09:09.796336+00:00", "kind": "readme", "missing": false, "url": "https://github.com/grobidOrg/grobid"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T18:02:11.512700+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b6e85af9a6d2d9fed5af927ecc133c80effc2e38fcdeefade02257566ffeec26", "fetched_at": "2026-08-28T04:09:09.796336+00:00", "kind": "readme", "missing": false, "url": "https://github.com/grobidOrg/grobid"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T18:02:11.512700+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b6e85af9a6d2d9fed5af927ecc133c80effc2e38fcdeefade02257566ffeec26", "fetched_at": "2026-08-28T04:09:09.796336+00:00", "kind": "readme", "missing": false, "url": "https://github.com/grobidOrg/grobid"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 99, "longevity": 100, "rhythm": 64}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 5102, "days_push": 8, "days_rel": 29, "gap_med": 239, "n_releases_24m": 4}, "score": 87, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}