{"adoption": {"forks": 492, "observed_at": "2026-08-28T04:04:55.046597+00:00", "stars": 1507}, "canonical_url": "https://ross.abutalabs.com/products/nlp-lang", "card": {"archived": false, "artifact_type": "library", "description": "这个项目是一个基本包.封装了大多数nlp项目中常用工具", "domain": ["developer-tools"], "enriched": true, "function": ["nlp", "parser", "search-engine", "data-science"], "health_score": 78, "homepage": null, "language": "Java", "license": "Apache-2.0", "license_family": "permissive", "maturity": "maintenance", "member_repos": ["NLPchina/nlp-lang"], "name": "NLPchina/nlp-lang", "platform": ["jvm"], "pushed_at": "2026-08-13T18:47:52+00:00", "repo": "NLPchina/nlp-lang", "stars": 1507, "tags": ["chinese-nlp", "trie-tree", "double-array-trie", "pinyin", "simplified-traditional-chinese", "simhash", "bloom-filter", "viterbi", "text-processing", "natural-language-processing", "algorithms"], "topics": ["java", "nlp", "nlp-lang", "tire"], "urls": [], "use_cases": ["build a trie or double-array trie in java", "convert chinese characters to pinyin", "convert simplified chinese to traditional chinese", "compute document similarity with simhash", "deduplicate text by fingerprint", "split chinese text into sentences", "word frequency and idf statistics", "in-memory search autocomplete suggestions"], "what_it_is": "A Java base library from NLPchina that packages common utilities used across NLP projects, such as trie/double-array trie structures, word normalization, sentence splitting, and Viterbi. It also includes components like Chinese-to-pinyin conversion, simplified/traditional Chinese conversion, SimHash similarity, fingerprint deduplication, and in-memory search suggestions.", "when_to_avoid": ["you need modern deep-learning-based NLP pipelines", "you work outside the JVM ecosystem", "you need actively developed features beyond the current utility set"], "when_to_choose": ["you need common Chinese NLP utilities in a JVM project", "you want fast trie-based string lookup structures", "you need pinyin conversion or simplified/traditional Chinese mapping", "you need lightweight text deduplication or similarity without heavy dependencies"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/nlp-lang", "repo": "NLPchina/nlp-lang", "role": "main", "score": 66}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:04:55.046597+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T04:32:35.690370+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "75fc31a62e9608354559b6562d8c69925cfe0e9ba2ffd2461a056a4add48ca77", "fetched_at": "2026-08-28T04:04:55.046597+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NLPchina/nlp-lang"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:04:55.046597+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T04:32:35.690370+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "75fc31a62e9608354559b6562d8c69925cfe0e9ba2ffd2461a056a4add48ca77", "fetched_at": "2026-08-28T04:04:55.046597+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NLPchina/nlp-lang"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T04:32:35.690370+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "75fc31a62e9608354559b6562d8c69925cfe0e9ba2ffd2461a056a4add48ca77", "fetched_at": "2026-08-28T04:04:55.046597+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NLPchina/nlp-lang"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:04:55.046597+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:04:55.046597+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:04:55.046597+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T04:32:35.690370+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "75fc31a62e9608354559b6562d8c69925cfe0e9ba2ffd2461a056a4add48ca77", "fetched_at": "2026-08-28T04:04:55.046597+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NLPchina/nlp-lang"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:04:55.046597+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:04:55.046597+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T04:32:35.690370+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "75fc31a62e9608354559b6562d8c69925cfe0e9ba2ffd2461a056a4add48ca77", "fetched_at": "2026-08-28T04:04:55.046597+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NLPchina/nlp-lang"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:04:55.046597+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:04:55.046597+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:04:55.046597+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T04:32:35.690370+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "75fc31a62e9608354559b6562d8c69925cfe0e9ba2ffd2461a056a4add48ca77", "fetched_at": "2026-08-28T04:04:55.046597+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NLPchina/nlp-lang"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:04:55.046597+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:04:55.046597+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T04:32:35.690370+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "75fc31a62e9608354559b6562d8c69925cfe0e9ba2ffd2461a056a4add48ca77", "fetched_at": "2026-08-28T04:04:55.046597+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NLPchina/nlp-lang"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T04:32:35.690370+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "75fc31a62e9608354559b6562d8c69925cfe0e9ba2ffd2461a056a4add48ca77", "fetched_at": "2026-08-28T04:04:55.046597+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NLPchina/nlp-lang"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T04:32:35.690370+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "75fc31a62e9608354559b6562d8c69925cfe0e9ba2ffd2461a056a4add48ca77", "fetched_at": "2026-08-28T04:04:55.046597+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NLPchina/nlp-lang"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T04:32:35.690370+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "75fc31a62e9608354559b6562d8c69925cfe0e9ba2ffd2461a056a4add48ca77", "fetched_at": "2026-08-28T04:04:55.046597+00:00", "kind": "readme", "missing": false, "url": "https://github.com/NLPchina/nlp-lang"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 97, "longevity": 100, "rhythm": 8}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 4539, "days_push": 20, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 66, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}