Sentence Similarity
sentence-transformers
Safetensors
English
nvomniembed
text
image
video
audio
vidore
multimodal-embedding
Text-to-Video retrieval
Text-to-Audio retrieval
Visual Document Retrieval
feature-extraction
custom_code
Instructions to use nvidia/omni-embed-nemotron-3b with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- sentence-transformers
How to use nvidia/omni-embed-nemotron-3b with sentence-transformers:
from sentence_transformers import SentenceTransformer model = SentenceTransformer("nvidia/omni-embed-nemotron-3b", trust_remote_code=True) sentences = [ "That is a happy person", "That is a happy dog", "That is a very happy person", "Today is a sunny day" ] embeddings = model.encode(sentences) similarities = model.similarity(embeddings, embeddings) print(similarities.shape) # [4, 4] - Notebooks
- Google Colab
- Kaggle
Download modeling_nv_omni_embed.py from nvidia/omni-embed-nemotron-3b: direct link, hf CLI and curl.
- Browser
- Download file 1.87 kB
-
https://huggingface.co/nvidia/omni-embed-nemotron-3b/resolve/main/modeling_nv_omni_embed.py
- Command line
-
hf download hf://nvidia/omni-embed-nemotron-3b/modeling_nv_omni_embed.py
-
curl -L -o modeling_nv_omni_embed.py https://huggingface.co/nvidia/omni-embed-nemotron-3b/resolve/main/modeling_nv_omni_embed.py
1.87 kB
| import torch | |
| from transformers import Qwen2_5OmniThinkerTextModel, Qwen2_5OmniThinkerForConditionalGeneration | |
| from transformers.cache_utils import Cache | |
| from transformers.modeling_attn_mask_utils import _prepare_4d_attention_mask | |
| from transformers.models.qwen2_5_omni.configuration_qwen2_5_omni import Qwen2_5OmniThinkerConfig | |
| class BidirectQwen2_5OmniThinkerTextModel(Qwen2_5OmniThinkerTextModel): | |
| def __init__(self, config): | |
| super().__init__(config) | |
| for layer in self.layers: | |
| layer.self_attn.is_causal = False | |
| # override the _update_causal_mask method to generate bi-directional attention | |
| def _update_causal_mask( | |
| self, | |
| attention_mask: torch.Tensor, | |
| input_tensor: torch.Tensor, | |
| cache_position: torch.Tensor, | |
| past_key_values: Cache, | |
| output_attentions: bool = False, | |
| ): | |
| calculated_attention_mask = super()._update_causal_mask( | |
| attention_mask, | |
| input_tensor, | |
| cache_position, | |
| past_key_values, | |
| output_attentions) | |
| if calculated_attention_mask is None: | |
| return None | |
| if self.config._attn_implementation == "flash_attention_2": | |
| if attention_mask is not None and 0.0 in attention_mask: | |
| return attention_mask | |
| causal_mask = _prepare_4d_attention_mask( | |
| attention_mask, | |
| dtype=input_tensor.dtype, | |
| ) | |
| return causal_mask | |
| class NVOmniEmbedConfig(Qwen2_5OmniThinkerConfig): | |
| model_type = "nvomniembed" | |
| class NVOmniEmbedModel(Qwen2_5OmniThinkerForConditionalGeneration): | |
| config_class = NVOmniEmbedConfig | |
| def __init__(self, config): | |
| super().__init__(config) | |
| self.model = BidirectQwen2_5OmniThinkerTextModel._from_config( | |
| config.text_config, attn_implementation=config._attn_implementation | |
| ) | |