{"adoption": {"forks": 271, "observed_at": "2026-08-28T04:06:53.423928+00:00", "stars": 2460}, "canonical_url": "https://ross.abutalabs.com/products/api-for-open-llm", "card": {"archived": false, "artifact_type": "service", "description": "Openai style api for open large language models, using LLMs just as chatgpt! Support for LLaMA, LLaMA-2, BLOOM, Falcon, Baichuan, Qwen, Xverse, SqlCoder, CodeLLaMA, ChatGLM, ChatGLM2, ChatGLM3 etc. 开源大模型的统一后端接口", "domain": ["large-language-models", "artificial-intelligence", "self-hosted", "apis"], "enriched": true, "function": ["llm-inference", "api-framework", "http-server", "rag"], "health_score": 27, "homepage": null, "language": "Python", "license": "Apache-2.0", "license_family": "permissive", "maturity": "active", "member_repos": ["xusenlinzy/api-for-open-llm"], "name": "xusenlinzy/api-for-open-llm", "platform": ["python", "self-hosted"], "pushed_at": "2024-09-26T07:47:57+00:00", "repo": "xusenlinzy/api-for-open-llm", "stars": 2460, "tags": ["openai-compatible-api", "chatglm", "llama", "qwen", "vllm", "langchain", "streaming", "embeddings", "docker", "gpu", "linux"], "topics": ["docker", "langchain", "llms", "nlp", "openai", "chatglm", "llama", "baichuan", "internlm", "llama2", "qwen", "xverse", "sqlcoder", "code-llama"], "urls": [], "use_cases": ["serve open-source llm with openai compatible api", "self-host chatgpt alternative backend", "use langchain with local open llm", "run chatglm or qwen behind openai api", "deploy llm server with streaming responses", "expose embeddings endpoint for rag knowledge base"], "what_it_is": "A Python server that exposes open-source large language models (LLaMA, ChatGLM, Qwen, Baichuan, Falcon, etc.) behind an OpenAI-compatible API. It supports streaming responses, embeddings, reranking, LoRA adapters, and vLLM-accelerated inference so open models can drop in as ChatGPT replacements.", "when_to_avoid": ["you only need OpenAI's hosted models", "you need a production-grade inference server with broad model coverage like vLLM or TGI directly", "you need non-OpenAI API styles or heavy enterprise features"], "when_to_choose": ["you want existing OpenAI SDK clients to work with open models", "you need LangChain compatibility and embeddings for RAG", "you want one unified backend for many open LLM families", "you want vLLM-accelerated serving with concurrency"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/api-for-open-llm", "repo": "xusenlinzy/api-for-open-llm", "role": "main", "score": 20}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:06:53.423928+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T02:29:20.761140+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9e7199ff6b3a99e37ffc2e24d5306866626d2a4318370ce6fc95d417fb568964", "fetched_at": "2026-08-28T04:06:53.423928+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xusenlinzy/api-for-open-llm"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:06:53.423928+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T02:29:20.761140+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9e7199ff6b3a99e37ffc2e24d5306866626d2a4318370ce6fc95d417fb568964", "fetched_at": "2026-08-28T04:06:53.423928+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xusenlinzy/api-for-open-llm"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T02:29:20.761140+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9e7199ff6b3a99e37ffc2e24d5306866626d2a4318370ce6fc95d417fb568964", "fetched_at": "2026-08-28T04:06:53.423928+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xusenlinzy/api-for-open-llm"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:06:53.423928+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:06:53.423928+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:06:53.423928+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T02:29:20.761140+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9e7199ff6b3a99e37ffc2e24d5306866626d2a4318370ce6fc95d417fb568964", "fetched_at": "2026-08-28T04:06:53.423928+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xusenlinzy/api-for-open-llm"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:06:53.423928+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:06:53.423928+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T02:29:20.761140+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9e7199ff6b3a99e37ffc2e24d5306866626d2a4318370ce6fc95d417fb568964", "fetched_at": "2026-08-28T04:06:53.423928+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xusenlinzy/api-for-open-llm"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:06:53.423928+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:06:53.423928+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:06:53.423928+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T02:29:20.761140+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9e7199ff6b3a99e37ffc2e24d5306866626d2a4318370ce6fc95d417fb568964", "fetched_at": "2026-08-28T04:06:53.423928+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xusenlinzy/api-for-open-llm"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:06:53.423928+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:06:53.423928+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T02:29:20.761140+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9e7199ff6b3a99e37ffc2e24d5306866626d2a4318370ce6fc95d417fb568964", "fetched_at": "2026-08-28T04:06:53.423928+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xusenlinzy/api-for-open-llm"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T02:29:20.761140+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9e7199ff6b3a99e37ffc2e24d5306866626d2a4318370ce6fc95d417fb568964", "fetched_at": "2026-08-28T04:06:53.423928+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xusenlinzy/api-for-open-llm"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T02:29:20.761140+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9e7199ff6b3a99e37ffc2e24d5306866626d2a4318370ce6fc95d417fb568964", "fetched_at": "2026-08-28T04:06:53.423928+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xusenlinzy/api-for-open-llm"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T02:29:20.761140+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "9e7199ff6b3a99e37ffc2e24d5306866626d2a4318370ce6fc95d417fb568964", "fetched_at": "2026-08-28T04:06:53.423928+00:00", "kind": "readme", "missing": false, "url": "https://github.com/xusenlinzy/api-for-open-llm"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 0, "longevity": 85, "rhythm": 8}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 1198, "days_push": 706, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 20, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}