{"adoption": {"forks": 188, "observed_at": "2026-08-28T04:04:20.610556+00:00", "stars": 1315}, "canonical_url": "https://ross.abutalabs.com/products/svoice", "card": {"archived": true, "artifact_type": "library", "description": "We provide a PyTorch implementation of the paper Voice Separation with an Unknown Number of Multiple Speakers In which, we present a new method for separating a mixed audio sequence, in which multiple voices speak simultaneously. The new method employs gated neural networks that are trained to separate the voices at multiple processing steps, while maintaining the speaker in each output channel fixed. A different model is trained for every number of possible speakers, and the model with the largest number of speakers is employed to select the actual number of speakers in a given sample. Our method greatly outperforms the current state of the art, which, as we show, is not competitive for more than two speakers. ", "domain": ["speech-processing", "machine-learning"], "enriched": true, "function": ["audio-processing", "machine-learning", "deep-learning"], "health_score": 10, "homepage": null, "language": "Python", "license": "NOASSERTION", "license_family": "other", "maturity": "maintenance", "member_repos": ["facebookresearch/svoice"], "name": "facebookresearch/svoice", "platform": ["python"], "pushed_at": "2023-11-16T13:46:20+00:00", "repo": "facebookresearch/svoice", "stars": 1315, "tags": ["speech-separation", "pytorch", "research-code", "speaker-diarization", "icml-paper", "audio", "linux", "macos"], "topics": [], "urls": [], "use_cases": ["separate overlapping voices in a mixed audio recording", "estimate how many speakers are talking in an audio clip", "split a multi-speaker conversation into individual speaker tracks", "research speech separation models in PyTorch", "preprocess noisy multi-speaker audio for downstream speech recognition", "reproduce results from the SVoice ICML paper"], "what_it_is": "SVoice is a PyTorch implementation of the ICML paper 'Voice Separation with an Unknown Number of Multiple Speakers' from Facebook AI Research. It separates mixed audio containing multiple simultaneous speakers using gated neural networks, with models trained per speaker count to also estimate the number of speakers in a sample.", "when_to_avoid": ["you need a production-ready, actively maintained speech separation service", "you need speaker identity loss (IDloss) as described in the paper, which is not included", "you need a simple pretrained CLI tool with no training setup", "you work outside Python/PyTorch environments"], "when_to_choose": ["you need to separate audio with more than two simultaneous speakers", "you want a research-grade PyTorch baseline for speaker separation", "the number of speakers in the mixture is unknown", "you want to reproduce or extend the SVoice paper"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/svoice", "repo": "facebookresearch/svoice", "role": "main", "score": 10}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:04:20.610556+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T04:48:40.816952+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1145e557f6e5611b83e7177c1bac36f02f44a1733fe7ee02fa9da7490bbf996c", "fetched_at": "2026-08-28T04:04:20.610556+00:00", "kind": "readme", "missing": false, "url": "https://github.com/facebookresearch/svoice"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:04:20.610556+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T04:48:40.816952+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1145e557f6e5611b83e7177c1bac36f02f44a1733fe7ee02fa9da7490bbf996c", "fetched_at": "2026-08-28T04:04:20.610556+00:00", "kind": "readme", "missing": false, "url": "https://github.com/facebookresearch/svoice"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T04:48:40.816952+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1145e557f6e5611b83e7177c1bac36f02f44a1733fe7ee02fa9da7490bbf996c", "fetched_at": "2026-08-28T04:04:20.610556+00:00", "kind": "readme", "missing": false, "url": "https://github.com/facebookresearch/svoice"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:04:20.610556+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:04:20.610556+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:04:20.610556+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T04:48:40.816952+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1145e557f6e5611b83e7177c1bac36f02f44a1733fe7ee02fa9da7490bbf996c", "fetched_at": "2026-08-28T04:04:20.610556+00:00", "kind": "readme", "missing": false, "url": "https://github.com/facebookresearch/svoice"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:04:20.610556+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:04:20.610556+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T04:48:40.816952+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1145e557f6e5611b83e7177c1bac36f02f44a1733fe7ee02fa9da7490bbf996c", "fetched_at": "2026-08-28T04:04:20.610556+00:00", "kind": "readme", "missing": false, "url": "https://github.com/facebookresearch/svoice"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:04:20.610556+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:04:20.610556+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:04:20.610556+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T04:48:40.816952+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1145e557f6e5611b83e7177c1bac36f02f44a1733fe7ee02fa9da7490bbf996c", "fetched_at": "2026-08-28T04:04:20.610556+00:00", "kind": "readme", "missing": false, "url": "https://github.com/facebookresearch/svoice"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:04:20.610556+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:04:20.610556+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T04:48:40.816952+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1145e557f6e5611b83e7177c1bac36f02f44a1733fe7ee02fa9da7490bbf996c", "fetched_at": "2026-08-28T04:04:20.610556+00:00", "kind": "readme", "missing": false, "url": "https://github.com/facebookresearch/svoice"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T04:48:40.816952+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1145e557f6e5611b83e7177c1bac36f02f44a1733fe7ee02fa9da7490bbf996c", "fetched_at": "2026-08-28T04:04:20.610556+00:00", "kind": "readme", "missing": false, "url": "https://github.com/facebookresearch/svoice"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T04:48:40.816952+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1145e557f6e5611b83e7177c1bac36f02f44a1733fe7ee02fa9da7490bbf996c", "fetched_at": "2026-08-28T04:04:20.610556+00:00", "kind": "readme", "missing": false, "url": "https://github.com/facebookresearch/svoice"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T04:48:40.816952+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1145e557f6e5611b83e7177c1bac36f02f44a1733fe7ee02fa9da7490bbf996c", "fetched_at": "2026-08-28T04:04:20.610556+00:00", "kind": "readme", "missing": false, "url": "https://github.com/facebookresearch/svoice"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 0, "longevity": 100, "rhythm": 35}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": ["no_releases", "archived", "no_license"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 2113, "days_push": 1021, "days_rel": null, "gap_med": null, "n_releases_24m": 0}, "score": 10, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}