{"adoption": {"forks": 68, "observed_at": "2026-08-28T04:03:36.634502+00:00", "stars": 1107}, "canonical_url": "https://ross.abutalabs.com/products/prometheus-eval", "card": {"archived": false, "artifact_type": "library", "description": "Evaluate your LLM's response with Prometheus and GPT4 💯", "domain": ["large-language-models", "machine-learning", "developer-tools"], "enriched": true, "function": ["machine-learning", "llm-inference", "benchmarking", "data-science"], "health_score": 39, "homepage": null, "language": "Python", "license": "Apache-2.0", "license_family": "permissive", "maturity": "active", "member_repos": ["prometheus-eval/prometheus-eval"], "name": "prometheus-eval/prometheus-eval", "platform": ["python"], "pushed_at": "2025-04-25T03:58:37+00:00", "repo": "prometheus-eval/prometheus-eval", "stars": 1107, "tags": ["llm-as-a-judge", "evaluation", "litellm", "vllm", "llmops", "prometheus", "gpt4", "meta-evaluation", "natural-language-processing"], "topics": ["evaluation", "litellm", "llm", "llmops", "python", "vllm", "gpt4", "llm-as-a-judge", "llm-as-evaluator"], "urls": [], "use_cases": ["evaluate llm responses with an open-source judge model", "score model outputs with llm-as-a-judge", "run pairwise comparisons between two llm answers", "benchmark my fine-tuned model against gpt-4 evaluation", "grade llm outputs against custom evaluation criteria", "build an automated evaluation pipeline for my chatbot", "use prometheus 2 with vllm for fast evaluation"], "what_it_is": "Prometheus-Eval is a Python library for evaluating LLM generation outputs using the Prometheus family of open evaluator models and GPT-4 as LLM judges. It provides evaluation pipelines, datasets like BiGGen-Bench, and trained judge models for absolute grading and pairwise comparison tasks.", "when_to_avoid": ["you only need simple unit tests for code, not LLM output quality assessment", "you cannot host GPU inference and only want a hosted eval service", "you need evaluation of non-text modalities like images or audio"], "when_to_choose": ["you need reproducible, open-source LLM evaluation instead of closed APIs", "you want to run an LLM judge locally with vLLM", "you need absolute grading or pairwise ranking of model outputs", "you are doing LLM research on meta-evaluation or reward models"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/prometheus-eval", "repo": "prometheus-eval/prometheus-eval", "role": "main", "score": 23}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:03:36.634502+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T06:44:03.039000+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7a2c3b33eeb8dac4b5e9bec3d4c9280655d33b1fc2793b8819529a6c73198d70", "fetched_at": "2026-08-28T04:03:36.634502+00:00", "kind": "readme", "missing": false, "url": "https://github.com/prometheus-eval/prometheus-eval"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:03:36.634502+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T06:44:03.039000+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7a2c3b33eeb8dac4b5e9bec3d4c9280655d33b1fc2793b8819529a6c73198d70", "fetched_at": "2026-08-28T04:03:36.634502+00:00", "kind": "readme", "missing": false, "url": "https://github.com/prometheus-eval/prometheus-eval"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T06:44:03.039000+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7a2c3b33eeb8dac4b5e9bec3d4c9280655d33b1fc2793b8819529a6c73198d70", "fetched_at": "2026-08-28T04:03:36.634502+00:00", "kind": "readme", "missing": false, "url": "https://github.com/prometheus-eval/prometheus-eval"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:03:36.634502+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:03:36.634502+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:03:36.634502+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T06:44:03.039000+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7a2c3b33eeb8dac4b5e9bec3d4c9280655d33b1fc2793b8819529a6c73198d70", "fetched_at": "2026-08-28T04:03:36.634502+00:00", "kind": "readme", "missing": false, "url": "https://github.com/prometheus-eval/prometheus-eval"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:03:36.634502+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:03:36.634502+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T06:44:03.039000+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7a2c3b33eeb8dac4b5e9bec3d4c9280655d33b1fc2793b8819529a6c73198d70", "fetched_at": "2026-08-28T04:03:36.634502+00:00", "kind": "readme", "missing": false, "url": "https://github.com/prometheus-eval/prometheus-eval"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:03:36.634502+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:03:36.634502+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:03:36.634502+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T06:44:03.039000+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7a2c3b33eeb8dac4b5e9bec3d4c9280655d33b1fc2793b8819529a6c73198d70", "fetched_at": "2026-08-28T04:03:36.634502+00:00", "kind": "readme", "missing": false, "url": "https://github.com/prometheus-eval/prometheus-eval"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:03:36.634502+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:03:36.634502+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T06:44:03.039000+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7a2c3b33eeb8dac4b5e9bec3d4c9280655d33b1fc2793b8819529a6c73198d70", "fetched_at": "2026-08-28T04:03:36.634502+00:00", "kind": "readme", "missing": false, "url": "https://github.com/prometheus-eval/prometheus-eval"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T06:44:03.039000+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7a2c3b33eeb8dac4b5e9bec3d4c9280655d33b1fc2793b8819529a6c73198d70", "fetched_at": "2026-08-28T04:03:36.634502+00:00", "kind": "readme", "missing": false, "url": "https://github.com/prometheus-eval/prometheus-eval"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T06:44:03.039000+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7a2c3b33eeb8dac4b5e9bec3d4c9280655d33b1fc2793b8819529a6c73198d70", "fetched_at": "2026-08-28T04:03:36.634502+00:00", "kind": "readme", "missing": false, "url": "https://github.com/prometheus-eval/prometheus-eval"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T06:44:03.039000+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "7a2c3b33eeb8dac4b5e9bec3d4c9280655d33b1fc2793b8819529a6c73198d70", "fetched_at": "2026-08-28T04:03:36.634502+00:00", "kind": "readme", "missing": false, "url": "https://github.com/prometheus-eval/prometheus-eval"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 18, "longevity": 61, "rhythm": 8}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 867, "days_push": 495, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 23, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}