{"adoption": {"forks": 131, "observed_at": "2026-08-28T04:05:19.111941+00:00", "stars": 1666}, "canonical_url": "https://ross.abutalabs.com/products/sa2va", "card": {"archived": false, "artifact_type": "library", "description": "Official Repo For Pixel-LLM Codebase: Sa2VA (T-PAMI-26), SAMTok (CVPR-26), VRT (Arxiv-25), SaSaSa2VA (1-st solution for LSVOS)", "domain": ["computer-vision", "large-language-models", "machine-learning", "artificial-intelligence"], "enriched": true, "function": ["machine-learning", "computer-vision", "deep-learning", "nlp", "image-processing", "video-processing"], "health_score": 93, "homepage": null, "language": "Python", "license": "Apache-2.0", "license_family": "permissive", "maturity": "active", "member_repos": ["bytedance/Sa2VA"], "name": "bytedance/Sa2VA", "platform": ["python"], "pushed_at": "2026-08-04T23:56:43+00:00", "repo": "bytedance/Sa2VA", "stars": 1666, "tags": ["multimodal-llm", "segmentation", "sam2", "grounded-understanding", "referring-segmentation", "visual-prompting", "research-code", "image-chat", "video-chat", "mask-tokens", "gpu", "linux"], "topics": ["computer-vision", "mllm", "large-language-models"], "urls": [], "use_cases": ["referring image and video segmentation with a multimodal LLM", "grounded conversation about images and videos", "chat with an AI about specific pixels or regions in an image", "generate segmentation masks from natural language instructions", "visual prompting and region-level question answering", "evaluate grounded visual reasoning with VRT-Bench", "train a multimodal LLM that outputs segmentation masks", "video object segmentation benchmark submission"], "what_it_is": "Sa2VA is a family of research models and codebases from ByteDance that combine SAM-2 with multimodal LLMs for pixel-level grounded understanding of images and videos. It includes the core Sa2VA model plus extensions like SAMTok, VRT, Pixel-SAIL, and SaSaSa2VA, with pretrained checkpoints on Hugging Face.", "when_to_avoid": ["you need a production-ready, latency-optimized vision API", "you lack a GPU or cannot run large multimodal models", "you only need simple image classification or object detection", "you want a no-code GUI tool rather than a Python research codebase"], "when_to_choose": ["you need unified referring segmentation and visual chat in one model", "you want pixel-level grounding on top of InternVL or Qwen-VL backbones", "you are doing research on grounded multimodal understanding", "you need pretrained checkpoints for image and video segmentation tasks"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/sa2va", "repo": "bytedance/Sa2VA", "role": "main", "score": 70}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:05:19.111941+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-30T03:42:57.832601+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b80c36c49b3078850eebb70c0a79d1037ef75a2bfacc78b97b5375c7bfc828d1", "fetched_at": "2026-08-28T04:05:19.111941+00:00", "kind": "readme", "missing": false, "url": "https://github.com/bytedance/Sa2VA"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:05:19.111941+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-30T03:42:57.832601+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b80c36c49b3078850eebb70c0a79d1037ef75a2bfacc78b97b5375c7bfc828d1", "fetched_at": "2026-08-28T04:05:19.111941+00:00", "kind": "readme", "missing": false, "url": "https://github.com/bytedance/Sa2VA"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-30T03:42:57.832601+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b80c36c49b3078850eebb70c0a79d1037ef75a2bfacc78b97b5375c7bfc828d1", "fetched_at": "2026-08-28T04:05:19.111941+00:00", "kind": "readme", "missing": false, "url": "https://github.com/bytedance/Sa2VA"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:05:19.111941+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:05:19.111941+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:05:19.111941+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-30T03:42:57.832601+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b80c36c49b3078850eebb70c0a79d1037ef75a2bfacc78b97b5375c7bfc828d1", "fetched_at": "2026-08-28T04:05:19.111941+00:00", "kind": "readme", "missing": false, "url": "https://github.com/bytedance/Sa2VA"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:05:19.111941+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:05:19.111941+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-30T03:42:57.832601+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b80c36c49b3078850eebb70c0a79d1037ef75a2bfacc78b97b5375c7bfc828d1", "fetched_at": "2026-08-28T04:05:19.111941+00:00", "kind": "readme", "missing": false, "url": "https://github.com/bytedance/Sa2VA"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:05:19.111941+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:05:19.111941+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:05:19.111941+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-30T03:42:57.832601+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b80c36c49b3078850eebb70c0a79d1037ef75a2bfacc78b97b5375c7bfc828d1", "fetched_at": "2026-08-28T04:05:19.111941+00:00", "kind": "readme", "missing": false, "url": "https://github.com/bytedance/Sa2VA"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:05:19.111941+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:05:19.111941+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-30T03:42:57.832601+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b80c36c49b3078850eebb70c0a79d1037ef75a2bfacc78b97b5375c7bfc828d1", "fetched_at": "2026-08-28T04:05:19.111941+00:00", "kind": "readme", "missing": false, "url": "https://github.com/bytedance/Sa2VA"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-30T03:42:57.832601+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b80c36c49b3078850eebb70c0a79d1037ef75a2bfacc78b97b5375c7bfc828d1", "fetched_at": "2026-08-28T04:05:19.111941+00:00", "kind": "readme", "missing": false, "url": "https://github.com/bytedance/Sa2VA"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-30T03:42:57.832601+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b80c36c49b3078850eebb70c0a79d1037ef75a2bfacc78b97b5375c7bfc828d1", "fetched_at": "2026-08-28T04:05:19.111941+00:00", "kind": "readme", "missing": false, "url": "https://github.com/bytedance/Sa2VA"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-30T03:42:57.832601+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "b80c36c49b3078850eebb70c0a79d1037ef75a2bfacc78b97b5375c7bfc828d1", "fetched_at": "2026-08-28T04:05:19.111941+00:00", "kind": "readme", "missing": false, "url": "https://github.com/bytedance/Sa2VA"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 96, "longevity": 43, "rhythm": 52}, "computed_at": "2026-09-03T02:20:16.233290+00:00", "flags": [], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 604, "days_push": 29, "days_rel": 320, "gap_med": 1, "n_releases_24m": 2}, "score": 70, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}