{"adoption": {"forks": 214, "observed_at": "2026-08-28T04:07:36.098840+00:00", "stars": 2993}, "canonical_url": "https://ross.abutalabs.com/products/llm_aided_ocr", "card": {"archived": false, "artifact_type": "cli-tool", "description": "Enhances Tesseract OCR output using LLMs (local or API) for error correction, smart chunking, and markdown formatting of scanned PDFs", "domain": ["pdf", "developer-tools", "large-language-models"], "enriched": true, "function": ["ocr", "nlp", "llm-inference", "pdf", "markdown", "image-processing"], "health_score": 77, "homepage": null, "language": "Python", "license": "NOASSERTION", "license_family": "other", "maturity": "active", "member_repos": ["Dicklesworthstone/llm_aided_ocr"], "name": "Dicklesworthstone/llm_aided_ocr", "platform": ["python", "cli"], "pushed_at": "2026-08-03T04:19:28+00:00", "repo": "Dicklesworthstone/llm_aided_ocr", "stars": 2993, "tags": ["tesseract", "ocr-correction", "pdf-to-markdown", "llm-post-processing", "document-digitization", "natural-language-processing", "linux", "macos", "gpu"], "topics": ["ai-assist", "llama2", "llm", "ocr", "tesseract", "ocr-correction"], "urls": [], "use_cases": ["convert scanned pdfs to clean markdown", "fix OCR errors in digitized documents with an LLM", "correct tesseract output using local llama models", "turn old scanned letters or books into readable text", "remove headers and page numbers from OCR output", "batch process scanned pdfs into accurate text files"], "what_it_is": "A Python tool that converts scanned PDFs to text via Tesseract OCR, then uses LLMs (local or API-based like OpenAI/Anthropic) to correct OCR errors, remove duplicates, and format output as clean Markdown. It includes smart chunking, image preprocessing, quality assessment, and optional GPU-accelerated local inference.", "when_to_avoid": ["you need high-throughput OCR without LLM inference costs or latency", "you only need basic text extraction from clean digital PDFs", "you require a license-guaranteed dependency (license is non-standard)"], "when_to_choose": ["you have scanned PDFs whose raw OCR output is error-ridden and needs cleanup", "you want Markdown-formatted output from scanned documents", "you prefer running correction with a local LLM for privacy or cost reasons"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/llm_aided_ocr", "repo": "Dicklesworthstone/llm_aided_ocr", "role": "main", "score": 71}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:07:36.098840+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T18:47:26.433947+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "4cffaf8fdd6c164ad72a4d386c697489c49f821403aa1bf2a858a07dd819d94e", "fetched_at": "2026-08-28T04:07:36.098840+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dicklesworthstone/llm_aided_ocr"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:07:36.098840+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T18:47:26.433947+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "4cffaf8fdd6c164ad72a4d386c697489c49f821403aa1bf2a858a07dd819d94e", "fetched_at": "2026-08-28T04:07:36.098840+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dicklesworthstone/llm_aided_ocr"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T18:47:26.433947+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "4cffaf8fdd6c164ad72a4d386c697489c49f821403aa1bf2a858a07dd819d94e", "fetched_at": "2026-08-28T04:07:36.098840+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dicklesworthstone/llm_aided_ocr"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:07:36.098840+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:07:36.098840+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:07:36.098840+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T18:47:26.433947+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "4cffaf8fdd6c164ad72a4d386c697489c49f821403aa1bf2a858a07dd819d94e", "fetched_at": "2026-08-28T04:07:36.098840+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dicklesworthstone/llm_aided_ocr"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:07:36.098840+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:07:36.098840+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T18:47:26.433947+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "4cffaf8fdd6c164ad72a4d386c697489c49f821403aa1bf2a858a07dd819d94e", "fetched_at": "2026-08-28T04:07:36.098840+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dicklesworthstone/llm_aided_ocr"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:07:36.098840+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:07:36.098840+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:07:36.098840+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T18:47:26.433947+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "4cffaf8fdd6c164ad72a4d386c697489c49f821403aa1bf2a858a07dd819d94e", "fetched_at": "2026-08-28T04:07:36.098840+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dicklesworthstone/llm_aided_ocr"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:07:36.098840+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:07:36.098840+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T18:47:26.433947+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "4cffaf8fdd6c164ad72a4d386c697489c49f821403aa1bf2a858a07dd819d94e", "fetched_at": "2026-08-28T04:07:36.098840+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dicklesworthstone/llm_aided_ocr"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T18:47:26.433947+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "4cffaf8fdd6c164ad72a4d386c697489c49f821403aa1bf2a858a07dd819d94e", "fetched_at": "2026-08-28T04:07:36.098840+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dicklesworthstone/llm_aided_ocr"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T18:47:26.433947+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "4cffaf8fdd6c164ad72a4d386c697489c49f821403aa1bf2a858a07dd819d94e", "fetched_at": "2026-08-28T04:07:36.098840+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dicklesworthstone/llm_aided_ocr"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T18:47:26.433947+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "4cffaf8fdd6c164ad72a4d386c697489c49f821403aa1bf2a858a07dd819d94e", "fetched_at": "2026-08-28T04:07:36.098840+00:00", "kind": "readme", "missing": false, "url": "https://github.com/Dicklesworthstone/llm_aided_ocr"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 95, "longevity": 81, "rhythm": 35}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": ["no_releases", "no_license"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 1134, "days_push": 30, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 71, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}