{"adoption": {"forks": 89, "observed_at": "2026-08-28T04:04:01.103801+00:00", "stars": 1215}, "canonical_url": "https://ross.abutalabs.com/products/chinese_speech_pretrain", "card": {"archived": false, "artifact_type": "dataset", "description": "chinese speech pretrained models", "domain": ["speech-processing", "machine-learning", "artificial-intelligence"], "enriched": true, "function": ["speech-recognition", "machine-learning", "transformers", "audio-processing"], "health_score": 20, "homepage": null, "language": "Shell", "license": null, "license_family": "other", "maturity": "maintenance", "member_repos": ["TencentGameMate/chinese_speech_pretrain"], "name": "TencentGameMate/chinese_speech_pretrain", "platform": ["python"], "pushed_at": "2024-08-23T03:14:03+00:00", "repo": "TencentGameMate/chinese_speech_pretrain", "stars": 1215, "tags": ["wav2vec2", "hubert", "self-supervised-pretraining", "fairseq", "chinese-asr", "wenetspeech", "pretrained-models", "huggingface", "natural-language-processing", "gpu", "linux"], "topics": [], "urls": [], "use_cases": ["pretrain wav2vec2 or hubert models on chinese speech", "use chinese speech representations as features for asr", "improve mandarin speech recognition with low-resource fine-tuning", "download chinese wav2vec2 checkpoints for huggingface transformers", "benchmark chinese speech recognition on aishell and wenetspeech", "extract self-supervised speech embeddings for downstream audio tasks"], "what_it_is": "A collection of Chinese speech pretrained models (wav2vec 2.0 and HuBERT, BASE and LARGE) trained by Tencent on 10,000 hours of WenetSpeech data using Fairseq. Checkpoints are hosted on Hugging Face and Baidu Pan, with ESPnet-based ASR recipes demonstrating their use as feature extractors for Conformer speech recognition.", "when_to_avoid": ["you need pretrained models for non-chinese languages", "you want a ready-to-use end-to-end speech recognition product rather than pretrained features", "you cannot work with fairseq or espnet tooling"], "when_to_choose": ["you need chinese-language speech pretrained models for asr or audio feature extraction", "you want wav2vec2/hubert checkpoints trained on large-scale mandarin data", "you are fine-tuning speech recognition on limited labeled chinese audio"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/chinese_speech_pretrain", "repo": "TencentGameMate/chinese_speech_pretrain", "role": "main", "score": 32}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:04:01.103801+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T06:17:14.107152+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "18750fd41ff308e82298d19fff9f89d66de00320bc07b0844a15da708576f513", "fetched_at": "2026-08-28T04:04:01.103801+00:00", "kind": "readme", "missing": false, "url": "https://github.com/TencentGameMate/chinese_speech_pretrain"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:04:01.103801+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T06:17:14.107152+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "18750fd41ff308e82298d19fff9f89d66de00320bc07b0844a15da708576f513", "fetched_at": "2026-08-28T04:04:01.103801+00:00", "kind": "readme", "missing": false, "url": "https://github.com/TencentGameMate/chinese_speech_pretrain"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T06:17:14.107152+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "18750fd41ff308e82298d19fff9f89d66de00320bc07b0844a15da708576f513", "fetched_at": "2026-08-28T04:04:01.103801+00:00", "kind": "readme", "missing": false, "url": "https://github.com/TencentGameMate/chinese_speech_pretrain"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:04:01.103801+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:04:01.103801+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:04:01.103801+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T06:17:14.107152+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "18750fd41ff308e82298d19fff9f89d66de00320bc07b0844a15da708576f513", "fetched_at": "2026-08-28T04:04:01.103801+00:00", "kind": "readme", "missing": false, "url": "https://github.com/TencentGameMate/chinese_speech_pretrain"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:04:01.103801+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:04:01.103801+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T06:17:14.107152+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "18750fd41ff308e82298d19fff9f89d66de00320bc07b0844a15da708576f513", "fetched_at": "2026-08-28T04:04:01.103801+00:00", "kind": "readme", "missing": false, "url": "https://github.com/TencentGameMate/chinese_speech_pretrain"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:04:01.103801+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:04:01.103801+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:04:01.103801+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T06:17:14.107152+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "18750fd41ff308e82298d19fff9f89d66de00320bc07b0844a15da708576f513", "fetched_at": "2026-08-28T04:04:01.103801+00:00", "kind": "readme", "missing": false, "url": "https://github.com/TencentGameMate/chinese_speech_pretrain"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:04:01.103801+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:04:01.103801+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T06:17:14.107152+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "18750fd41ff308e82298d19fff9f89d66de00320bc07b0844a15da708576f513", "fetched_at": "2026-08-28T04:04:01.103801+00:00", "kind": "readme", "missing": false, "url": "https://github.com/TencentGameMate/chinese_speech_pretrain"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T06:17:14.107152+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "18750fd41ff308e82298d19fff9f89d66de00320bc07b0844a15da708576f513", "fetched_at": "2026-08-28T04:04:01.103801+00:00", "kind": "readme", "missing": false, "url": "https://github.com/TencentGameMate/chinese_speech_pretrain"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T06:17:14.107152+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "18750fd41ff308e82298d19fff9f89d66de00320bc07b0844a15da708576f513", "fetched_at": "2026-08-28T04:04:01.103801+00:00", "kind": "readme", "missing": false, "url": "https://github.com/TencentGameMate/chinese_speech_pretrain"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T06:17:14.107152+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "18750fd41ff308e82298d19fff9f89d66de00320bc07b0844a15da708576f513", "fetched_at": "2026-08-28T04:04:01.103801+00:00", "kind": "readme", "missing": false, "url": "https://github.com/TencentGameMate/chinese_speech_pretrain"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 0, "longevity": 100, "rhythm": 35}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": ["no_releases", "no_license"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 1561, "days_push": 740, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 32, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}