{"adoption": {"forks": 980, "observed_at": "2026-08-28T04:09:47.966233+00:00", "stars": 6710}, "canonical_url": "https://ross.abutalabs.com/products/pkuseg-python", "card": {"archived": false, "artifact_type": "library", "description": "pkuseg多领域中文分词工具; The pkuseg toolkit for multi-domain Chinese word segmentation", "domain": ["machine-learning", "developer-tools"], "enriched": true, "function": ["nlp", "parser", "machine-learning"], "health_score": 20, "homepage": null, "language": "Python", "license": "MIT", "license_family": "permissive", "maturity": "maintenance", "member_repos": ["lancopku/pkuseg-python"], "name": "lancopku/pkuseg-python", "platform": ["python", "windows"], "pushed_at": "2022-11-05T13:37:41+00:00", "repo": "lancopku/pkuseg-python", "stars": 6710, "tags": ["chinese-word-segmentation", "tokenization", "part-of-speech-tagging", "multi-domain", "pretrained-models", "natural-language-processing", "linux", "macos"], "topics": ["chinese-word-segmentation"], "urls": [], "use_cases": ["segment Chinese text into words", "tokenize Chinese text for a specific domain like medicine or news", "perform part-of-speech tagging on Chinese text", "train a custom Chinese word segmentation model on my own labeled data", "get higher Chinese segmentation accuracy than jieba or THULAC", "preprocess Chinese text for downstream NLP tasks"], "what_it_is": "pkuseg is a Python toolkit for multi-domain Chinese word segmentation based on the paper by Luo et al. (2019). It provides pre-trained models for specific domains (news, web, medicine, tourism, mixed), supports part-of-speech tagging, and allows users to train custom models on their own annotated data.", "when_to_avoid": ["you need a lightweight, fast general-purpose tokenizer without model downloads", "you require active development or recent updates, as the project has not seen releases since 2022", "you need segmentation for languages other than Chinese", "you are on a platform other than 64-bit Linux, macOS, or Windows and cannot compile locally"], "when_to_choose": ["you need accurate Chinese word segmentation, especially in a known domain like news, web, medicine, or tourism", "you want to train a segmentation model on your own annotated corpus", "you need part-of-speech tagging alongside segmentation", "benchmark results show it outperforms jieba or THULAC on your data"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/pkuseg-python", "repo": "lancopku/pkuseg-python", "role": "main", "score": 23}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:09:47.966233+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T17:42:56.598166+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "114b4f7f7e389d8f616be6f71ac708e6f6532a318d8e1469608ad4165e568784", "fetched_at": "2026-08-28T04:09:47.966233+00:00", "kind": "readme", "missing": false, "url": "https://github.com/lancopku/pkuseg-python"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:09:47.966233+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T17:42:56.598166+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "114b4f7f7e389d8f616be6f71ac708e6f6532a318d8e1469608ad4165e568784", "fetched_at": "2026-08-28T04:09:47.966233+00:00", "kind": "readme", "missing": false, "url": "https://github.com/lancopku/pkuseg-python"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T17:42:56.598166+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "114b4f7f7e389d8f616be6f71ac708e6f6532a318d8e1469608ad4165e568784", "fetched_at": "2026-08-28T04:09:47.966233+00:00", "kind": "readme", "missing": false, "url": "https://github.com/lancopku/pkuseg-python"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:09:47.966233+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:09:47.966233+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:09:47.966233+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T17:42:56.598166+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "114b4f7f7e389d8f616be6f71ac708e6f6532a318d8e1469608ad4165e568784", "fetched_at": "2026-08-28T04:09:47.966233+00:00", "kind": "readme", "missing": false, "url": "https://github.com/lancopku/pkuseg-python"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:09:47.966233+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:09:47.966233+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T17:42:56.598166+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "114b4f7f7e389d8f616be6f71ac708e6f6532a318d8e1469608ad4165e568784", "fetched_at": "2026-08-28T04:09:47.966233+00:00", "kind": "readme", "missing": false, "url": "https://github.com/lancopku/pkuseg-python"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:09:47.966233+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:09:47.966233+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:09:47.966233+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T17:42:56.598166+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "114b4f7f7e389d8f616be6f71ac708e6f6532a318d8e1469608ad4165e568784", "fetched_at": "2026-08-28T04:09:47.966233+00:00", "kind": "readme", "missing": false, "url": "https://github.com/lancopku/pkuseg-python"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:09:47.966233+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:09:47.966233+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T17:42:56.598166+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "114b4f7f7e389d8f616be6f71ac708e6f6532a318d8e1469608ad4165e568784", "fetched_at": "2026-08-28T04:09:47.966233+00:00", "kind": "readme", "missing": false, "url": "https://github.com/lancopku/pkuseg-python"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T17:42:56.598166+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "114b4f7f7e389d8f616be6f71ac708e6f6532a318d8e1469608ad4165e568784", "fetched_at": "2026-08-28T04:09:47.966233+00:00", "kind": "readme", "missing": false, "url": "https://github.com/lancopku/pkuseg-python"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T17:42:56.598166+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "114b4f7f7e389d8f616be6f71ac708e6f6532a318d8e1469608ad4165e568784", "fetched_at": "2026-08-28T04:09:47.966233+00:00", "kind": "readme", "missing": false, "url": "https://github.com/lancopku/pkuseg-python"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T17:42:56.598166+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "114b4f7f7e389d8f616be6f71ac708e6f6532a318d8e1469608ad4165e568784", "fetched_at": "2026-08-28T04:09:47.966233+00:00", "kind": "readme", "missing": false, "url": "https://github.com/lancopku/pkuseg-python"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 0, "longevity": 100, "rhythm": 8}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 2950, "days_push": 1397, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 23, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}