Implementing NVLabs C-RADIOv3 Embeddings Model as Remotely Sourced Zoo Model for FiftyOne
Python
27
32 commits
updated Feb 3, 2026

This repository provides FiftyOne integration for C-RADIO models from NVIDIA Labs. RADIO models are state-of-the-art models that produce rich spatial features and global summaries, making them excellent for image embeddings, similarity search, attention visualization, and downstream computer vision tasks.
# Install FiftyOne
pip install fiftyone
# Register the RADIO model source
import fiftyone.zoo as foz
foz.register_zoo_model_source(
"https://github.com/harpreetsahota204/NVLabs_CRADIOV3",
)
import fiftyone as fo
import fiftyone.zoo as foz
# Load a dataset
dataset = foz.load_zoo_dataset("quickstart", shuffle=True)
# Load RADIO model for embeddings
model = foz.load_zoo_model("nv_labs/c-radio_v3-h")
# Compute embeddings
dataset.compute_embeddings(
model=model,
embeddings_field="radio_embeddings",
)
# Launch FiftyOne App
session = fo.launch_app(dataset)
| Model Name | Description | Architecture | Patch Size | Best For |
|---|---|---|---|---|
nv_labs/c-radio_v3-b | C-RADIOv3-B | ViT-B/16 | 16Γ16 | Fast inference, moderate accuracy |
nv_labs/c-radio_v3-l | C-RADIOv3-L | ViT-L/16 | 16Γ16 | Balanced performance |
nv_labs/c-radio_v3-h | C-RADIOv3-H | ViT-H/16 | 16Γ16 | High accuracy, recommended |
nv_labs/c-radio_v3-g | C-RADIOv3-g | ViT-H/14 | 14Γ14 | Maximum performance |
# Global image embeddings (default)
model = foz.load_zoo_model(
"nv_labs/c-radio_v3-h",
output_type="summary" # Global semantic features
)
# Spatial attention features
spatial_model = foz.load_zoo_model(
"nv_labs/c-radio_v3-h",
output_type="spatial" # Patch-level spatial features
)
# returns a 1D embedding vector, dimensions 3048
model = foz.load_zoo_model(
"nv_labs/c-radio_v3-h",
output_type="summary",
feature_format="NCHW" # "NCHW": [Batch, Channels, Height, Width] , or you can use "NLC":[Batch, Num_patches, Channels]
)
# returns spatial features which are parsed as a FiftyOne Heatmap
model = foz.load_zoo_model(
"nv_labs/c-radio_v3-h",
output_type="spatial",
feature_format="NCHW" # can only use this format for spatial features
)
model = foz.load_zoo_model(
"nv_labs/c-radio_v3-h",
# Core settings
output_type="spatial", # "summary" or "spatial"
feature_format="NCHW", # "NCHW" or "NLC" (NCHW only for spatial)
# Performance options
use_mixed_precision=True, # Auto-detected, bfloat16 on Ampere+
use_external_preprocessor=False, # Advanced preprocessing
# Spatial heatmap options (when output_type="spatial")
apply_smoothing=True, # Smooth attention heatmaps
smoothing_sigma=1.51, # Gaussian smoothing strength
)
Extract high-level semantic representations for similarity search and clustering:
# Compute embeddings
dataset.compute_embeddings(
model=model,
embeddings_field="radio_embeddings"
)
Visualize what regions the model pays attention to:
# Load spatial model with smoothing
spatial_model = foz.load_zoo_model(
"nv_labs/c-radio_v3-h",
output_type="spatial",
apply_smoothing=True, # or False
smoothing_sigma=0.51, # used only when apply_smoothing=True
feature_format="NCHW"
)
# Generate attention heatmaps
dataset.apply_model(spatial_model, "radio_heatmap")
# View heatmaps in FiftyOne App
session = fo.launch_app(dataset)
Create 2D visualizations of your image embeddings:
import fiftyone.brain as fob
# First compute embeddings
dataset.compute_embeddings(
model=model,
embeddings_field="radio_embeddings"
)
# Create UMAP visualization
results = fob.compute_visualization(
dataset,
method="umap", # Also supports "tsne", "pca"
brain_key="radio_viz",
embeddings="radio_embeddings"
)
# Explore in the App
session = fo.launch_app(dataset)
Build powerful similarity search with RADIO embeddings:
import fiftyone.brain as fob
# Build similarity index
results = fob.compute_similarity(
dataset,
backend="sklearn", # Fast sklearn backend
brain_key="radio_sim",
embeddings="radio_embeddings"
)
# Find similar images
sample_id = dataset.first().id
similar_samples = dataset.sort_by_similarity(
sample_id,
brain_key="radio_sim",
k=10 # Top 10 most similar
)
# View results
session = fo.launch_app(similar_samples)
Score how representative each sample is of your dataset:
import fiftyone.brain as fob
# Compute representativeness scores
fob.compute_representativeness(
dataset,
representativeness_field="radio_represent",
method="cluster-center",
embeddings="radio_embeddings"
)
# Find most representative samples
representative_view = dataset.sort_by("radio_represent", reverse=True)
Find and remove near-duplicate images:
import fiftyone.brain as fob
# Detect duplicates using embeddings
results = fob.compute_uniqueness(
dataset,
embeddings="radio_embeddings"
)
# Filter to most unique samples
unique_view = dataset.sort_by("uniqueness", reverse=True)
Combine multiple RADIO outputs for comprehensive analysis:
# Step 1: Global embeddings for similarity
embedding_model = foz.load_zoo_model("nv_labs/c-radio_v3-h")
dataset.compute_embeddings(embedding_model, "radio_embeddings")
# Step 2: Spatial heatmaps for attention analysis
spatial_model = foz.load_zoo_model(
"nv_labs/c-radio_v3-h",
output_type="spatial",
apply_smoothing=True,
smoothing_sigma=0.8
)
dataset.apply_model(spatial_model, "radio_heatmap")
# Step 3: Build similarity index
import fiftyone.brain as fob
fob.compute_similarity(dataset, embeddings="radio_embeddings", brain_key="radio_sim")
# Step 4: Comprehensive analysis
session = fo.launch_app(dataset)
GPU Memory Errors
# Use smaller models or disable mixed precision
model = foz.load_zoo_model(
"nv_labs/c-radio_v3-b", # Use smaller model
use_mixed_precision=False
)
Mixed Precision Issues
# Disable mixed precision on older GPUs
model = foz.load_zoo_model(
"nv_labs/c-radio_v3-h",
use_mixed_precision=False
)
@misc{heinrich2025radiov25improvedbaselinesagglomerative,
title={RADIOv2.5: Improved Baselines for Agglomerative Vision Foundation Models},
author={Greg Heinrich and Mike Ranzinger and Hongxu and Yin and Yao Lu and Jan Kautz and Andrew Tao and Bryan Catanzaro and Pavlo Molchanov},
year={2024},
eprint={2412.07679},
archivePrefix={arXiv},
primaryClass={cs.CV},
url={https://arxiv.org/abs/2412.07679},
}
This implementation follows the original RADIO model license. Please refer to the NVIDIA RADIO repository for complete license details.
Contributions are welcome! Please feel free to submit:
26 followers Β· starred Jul 2025
Python
76.5%
Jupyter Notebook
23.5%
Implementing NVLabs C-RADIOv3 Embeddings Model as Remotely Sourced Zoo Model for FiftyOne
Python
27
32 commits
updated Feb 3, 2026

This repository provides FiftyOne integration for C-RADIO models from NVIDIA Labs. RADIO models are state-of-the-art models that produce rich spatial features and global summaries, making them excellent for image embeddings, similarity search, attention visualization, and downstream computer vision tasks.
# Install FiftyOne
pip install fiftyone
# Register the RADIO model source
import fiftyone.zoo as foz
foz.register_zoo_model_source(
"https://github.com/harpreetsahota204/NVLabs_CRADIOV3",
)
import fiftyone as fo
import fiftyone.zoo as foz
# Load a dataset
dataset = foz.load_zoo_dataset("quickstart", shuffle=True)
# Load RADIO model for embeddings
model = foz.load_zoo_model("nv_labs/c-radio_v3-h")
# Compute embeddings
dataset.compute_embeddings(
model=model,
embeddings_field="radio_embeddings",
)
# Launch FiftyOne App
session = fo.launch_app(dataset)
| Model Name | Description | Architecture | Patch Size | Best For |
|---|---|---|---|---|
nv_labs/c-radio_v3-b | C-RADIOv3-B | ViT-B/16 | 16Γ16 | Fast inference, moderate accuracy |
nv_labs/c-radio_v3-l | C-RADIOv3-L | ViT-L/16 | 16Γ16 | Balanced performance |
nv_labs/c-radio_v3-h | C-RADIOv3-H | ViT-H/16 | 16Γ16 | High accuracy, recommended |
nv_labs/c-radio_v3-g | C-RADIOv3-g | ViT-H/14 | 14Γ14 | Maximum performance |
# Global image embeddings (default)
model = foz.load_zoo_model(
"nv_labs/c-radio_v3-h",
output_type="summary" # Global semantic features
)
# Spatial attention features
spatial_model = foz.load_zoo_model(
"nv_labs/c-radio_v3-h",
output_type="spatial" # Patch-level spatial features
)
# returns a 1D embedding vector, dimensions 3048
model = foz.load_zoo_model(
"nv_labs/c-radio_v3-h",
output_type="summary",
feature_format="NCHW" # "NCHW": [Batch, Channels, Height, Width] , or you can use "NLC":[Batch, Num_patches, Channels]
)
# returns spatial features which are parsed as a FiftyOne Heatmap
model = foz.load_zoo_model(
"nv_labs/c-radio_v3-h",
output_type="spatial",
feature_format="NCHW" # can only use this format for spatial features
)
model = foz.load_zoo_model(
"nv_labs/c-radio_v3-h",
# Core settings
output_type="spatial", # "summary" or "spatial"
feature_format="NCHW", # "NCHW" or "NLC" (NCHW only for spatial)
# Performance options
use_mixed_precision=True, # Auto-detected, bfloat16 on Ampere+
use_external_preprocessor=False, # Advanced preprocessing
# Spatial heatmap options (when output_type="spatial")
apply_smoothing=True, # Smooth attention heatmaps
smoothing_sigma=1.51, # Gaussian smoothing strength
)
Extract high-level semantic representations for similarity search and clustering:
# Compute embeddings
dataset.compute_embeddings(
model=model,
embeddings_field="radio_embeddings"
)
Visualize what regions the model pays attention to:
# Load spatial model with smoothing
spatial_model = foz.load_zoo_model(
"nv_labs/c-radio_v3-h",
output_type="spatial",
apply_smoothing=True, # or False
smoothing_sigma=0.51, # used only when apply_smoothing=True
feature_format="NCHW"
)
# Generate attention heatmaps
dataset.apply_model(spatial_model, "radio_heatmap")
# View heatmaps in FiftyOne App
session = fo.launch_app(dataset)
Create 2D visualizations of your image embeddings:
import fiftyone.brain as fob
# First compute embeddings
dataset.compute_embeddings(
model=model,
embeddings_field="radio_embeddings"
)
# Create UMAP visualization
results = fob.compute_visualization(
dataset,
method="umap", # Also supports "tsne", "pca"
brain_key="radio_viz",
embeddings="radio_embeddings"
)
# Explore in the App
session = fo.launch_app(dataset)
Build powerful similarity search with RADIO embeddings:
import fiftyone.brain as fob
# Build similarity index
results = fob.compute_similarity(
dataset,
backend="sklearn", # Fast sklearn backend
brain_key="radio_sim",
embeddings="radio_embeddings"
)
# Find similar images
sample_id = dataset.first().id
similar_samples = dataset.sort_by_similarity(
sample_id,
brain_key="radio_sim",
k=10 # Top 10 most similar
)
# View results
session = fo.launch_app(similar_samples)
Score how representative each sample is of your dataset:
import fiftyone.brain as fob
# Compute representativeness scores
fob.compute_representativeness(
dataset,
representativeness_field="radio_represent",
method="cluster-center",
embeddings="radio_embeddings"
)
# Find most representative samples
representative_view = dataset.sort_by("radio_represent", reverse=True)
Find and remove near-duplicate images:
import fiftyone.brain as fob
# Detect duplicates using embeddings
results = fob.compute_uniqueness(
dataset,
embeddings="radio_embeddings"
)
# Filter to most unique samples
unique_view = dataset.sort_by("uniqueness", reverse=True)
Combine multiple RADIO outputs for comprehensive analysis:
# Step 1: Global embeddings for similarity
embedding_model = foz.load_zoo_model("nv_labs/c-radio_v3-h")
dataset.compute_embeddings(embedding_model, "radio_embeddings")
# Step 2: Spatial heatmaps for attention analysis
spatial_model = foz.load_zoo_model(
"nv_labs/c-radio_v3-h",
output_type="spatial",
apply_smoothing=True,
smoothing_sigma=0.8
)
dataset.apply_model(spatial_model, "radio_heatmap")
# Step 3: Build similarity index
import fiftyone.brain as fob
fob.compute_similarity(dataset, embeddings="radio_embeddings", brain_key="radio_sim")
# Step 4: Comprehensive analysis
session = fo.launch_app(dataset)
GPU Memory Errors
# Use smaller models or disable mixed precision
model = foz.load_zoo_model(
"nv_labs/c-radio_v3-b", # Use smaller model
use_mixed_precision=False
)
Mixed Precision Issues
# Disable mixed precision on older GPUs
model = foz.load_zoo_model(
"nv_labs/c-radio_v3-h",
use_mixed_precision=False
)
@misc{heinrich2025radiov25improvedbaselinesagglomerative,
title={RADIOv2.5: Improved Baselines for Agglomerative Vision Foundation Models},
author={Greg Heinrich and Mike Ranzinger and Hongxu and Yin and Yao Lu and Jan Kautz and Andrew Tao and Bryan Catanzaro and Pavlo Molchanov},
year={2024},
eprint={2412.07679},
archivePrefix={arXiv},
primaryClass={cs.CV},
url={https://arxiv.org/abs/2412.07679},
}
This implementation follows the original RADIO model license. Please refer to the NVIDIA RADIO repository for complete license details.
Contributions are welcome! Please feel free to submit:
26 followers Β· starred Jul 2025
Python
76.5%
Jupyter Notebook
23.5%