{"adoption": {"forks": 279, "observed_at": "2026-08-28T04:07:47.565676+00:00", "stars": 3175}, "canonical_url": "https://ross.abutalabs.com/products/text-extract-api", "card": {"archived": false, "artifact_type": "service", "description": "Document (PDF, Word, PPTX ...) extraction and parse API using state of the art modern OCRs + Ollama supported models. Anonymize documents. Remove PII. Convert any document or picture to structured JSON or Markdown", "domain": ["pdf", "privacy", "developer-tools", "large-language-models", "self-hosted"], "enriched": true, "function": ["ocr", "pdf", "llm-inference", "api-framework", "caching", "privacy", "nlp"], "health_score": 66, "homepage": "https://demo.doctractor.com", "language": "Python", "license": "MIT", "license_family": "permissive", "maturity": "active", "member_repos": ["CatchTheTornado/text-extract-api"], "name": "CatchTheTornado/text-extract-api", "platform": ["self-hosted", "python", "cli"], "pushed_at": "2025-12-08T21:27:14+00:00", "repo": "CatchTheTornado/text-extract-api", "stars": 3175, "tags": ["document-parsing", "pii-removal", "anonymization", "markdown-conversion", "easyocr", "ollama", "fastapi", "celery", "structured-extraction", "ocr", "docker", "web-server"], "topics": ["api", "extract", "json", "llm", "pdf", "anonymization", "ocr", "ocr-python", "pii"], "urls": [], "use_cases": ["convert pdf to markdown", "extract structured json from invoices", "remove pii from documents", "ocr scanned documents locally", "parse word and pptx files to text", "anonymize documents before sharing", "improve ocr results with an llm"], "what_it_is": "A self-hosted FastAPI-based API that converts PDFs, Office documents, and images into Markdown or structured JSON using OCR engines (EasyOCR, marker-pdf) and Ollama LLM models. It also supports PII removal/anonymization, Redis caching of OCR results, and asynchronous processing via Celery.", "when_to_avoid": ["you only need simple text extraction without OCR or LLM overhead", "you cannot run Ollama models or lack suitable hardware", "you need a fully managed cloud document API", "you need Apple GPU support via Docker, which is not supported"], "when_to_choose": ["you need self-hosted document extraction with no cloud dependencies", "you want LLM-enhanced OCR with structured JSON output", "you need PII removal or document anonymization", "you process many documents and want async queue processing with caching"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/text-extract-api", "repo": "CatchTheTornado/text-extract-api", "role": "main", "score": 45}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:07:47.565676+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T18:45:25.237035+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7eb042cbae2ef0b70043fd7ae130449aab52f3ef046a9ecc0edafc50a197a74d", "fetched_at": "2026-08-28T04:07:47.565676+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CatchTheTornado/text-extract-api"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:07:47.565676+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T18:45:25.237035+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7eb042cbae2ef0b70043fd7ae130449aab52f3ef046a9ecc0edafc50a197a74d", "fetched_at": "2026-08-28T04:07:47.565676+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CatchTheTornado/text-extract-api"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T18:45:25.237035+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7eb042cbae2ef0b70043fd7ae130449aab52f3ef046a9ecc0edafc50a197a74d", "fetched_at": "2026-08-28T04:07:47.565676+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CatchTheTornado/text-extract-api"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:07:47.565676+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:07:47.565676+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:07:47.565676+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T18:45:25.237035+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7eb042cbae2ef0b70043fd7ae130449aab52f3ef046a9ecc0edafc50a197a74d", "fetched_at": "2026-08-28T04:07:47.565676+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CatchTheTornado/text-extract-api"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:07:47.565676+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:07:47.565676+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T18:45:25.237035+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7eb042cbae2ef0b70043fd7ae130449aab52f3ef046a9ecc0edafc50a197a74d", "fetched_at": "2026-08-28T04:07:47.565676+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CatchTheTornado/text-extract-api"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:07:47.565676+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:07:47.565676+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:07:47.565676+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T18:45:25.237035+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7eb042cbae2ef0b70043fd7ae130449aab52f3ef046a9ecc0edafc50a197a74d", "fetched_at": "2026-08-28T04:07:47.565676+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CatchTheTornado/text-extract-api"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:07:47.565676+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:07:47.565676+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T18:45:25.237035+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7eb042cbae2ef0b70043fd7ae130449aab52f3ef046a9ecc0edafc50a197a74d", "fetched_at": "2026-08-28T04:07:47.565676+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CatchTheTornado/text-extract-api"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T18:45:25.237035+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7eb042cbae2ef0b70043fd7ae130449aab52f3ef046a9ecc0edafc50a197a74d", "fetched_at": "2026-08-28T04:07:47.565676+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CatchTheTornado/text-extract-api"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T18:45:25.237035+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7eb042cbae2ef0b70043fd7ae130449aab52f3ef046a9ecc0edafc50a197a74d", "fetched_at": "2026-08-28T04:07:47.565676+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CatchTheTornado/text-extract-api"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T18:45:25.237035+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7eb042cbae2ef0b70043fd7ae130449aab52f3ef046a9ecc0edafc50a197a74d", "fetched_at": "2026-08-28T04:07:47.565676+00:00", "kind": "readme", "missing": false, "url": "https://github.com/CatchTheTornado/text-extract-api"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 56, "longevity": 48, "rhythm": 28}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 679, "days_push": 268, "days_rel": 492, "gap_med": 51.0, "n_releases_24m": 3}, "score": 45, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}