{
 "id": "r7d093d4eb4c",
 "found": true,
 "parent_id": "rbe28438d028",
 "created_at": "2026-08-31 22:01:47.457577+00:00",
 "rows": [
  {
   "url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
   "title": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16 \u00b7 Hugging Face",
   "published_at": "2026-08-14T21:27:48",
   "release": {
    "organization": "NVIDIA",
    "model": "NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
    "release_date": "2026-06-04",
    "parameters": "550B (55B active)",
    "context_window": "1M tokens",
    "license": "OpenMDW-1.1",
    "summary": "Frontier-scale general purpose reasoning and chat model optimized for complex agentic workflows, long-context reasoning, and high-stakes analytical workloads.",
    "organization_evidence": [
     "# NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
     "## Model Overview",
     "**Model Developer:** NVIDIA Corporation"
    ],
    "model_evidence": [
     "# NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16"
    ],
    "release_date_evidence": [
     "# NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
     "## Model Summary",
     "| **Release Date** | June 4, 2026 |"
    ],
    "parameters_evidence": [
     "# NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
     "## Model Summary",
     "| **Total Parameters** | 550B (55B active) |"
    ],
    "context_window_evidence": [
     "# NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
     "## Model Summary",
     "| **Context Length** | Up to 1M tokens |"
    ],
    "license_evidence": [
     "# NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
     "## Model Summary",
     "| **License** | [OpenMDW License Agreement, version 1.1](https://raw.githubusercontent.com/OpenMDW/OpenMDW/refs/heads/main/1.1/LICENSE.OpenMDW-1.1) |"
    ],
    "summary_evidence": [
     "# NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16",
     "### Use Case",
     "NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16 is a frontier-scale general purpose reasoning and chat model intended to be used in English, Code, and supported multilingual contexts."
    ]
   }
  },
  {
   "url": "https://huggingface.co/google/gemma-4-31B-it-qat-q4_0-unquantized",
   "title": "google/gemma-4-31B-it-qat-q4_0-unquantized \u00b7 Hugging Face",
   "published_at": "2026-07-21T21:12:15",
   "release": {
    "organization": "Google DeepMind",
    "model": "Gemma 4",
    "release_date": "2026-07-02",
    "parameters": null,
    "context_window": "256K tokens",
    "license": "Apache 2.0",
    "summary": "multimodal, handling text and image input (with audio supported on E2B, E4B, and 12B) and generating text output",
    "organization_evidence": [
     "Gemma is a family of open models built by Google DeepMind."
    ],
    "model_evidence": [
     "Gemma is a family of open models built by Google DeepMind."
    ],
    "release_date_evidence": [
     "Gemma is a family of open models built by Google DeepMind.",
     "[Technical Report](https://arxiv.org/abs/2607.02770)"
    ],
    "parameters_evidence": null,
    "context_window_evidence": [
     "Gemma is a family of open models built by Google DeepMind.",
     "Gemma 4 features a context window of up to 256K tokens and maintains multilingual support in over 140 languages."
    ],
    "license_evidence": [
     "Gemma is a family of open models built by Google DeepMind.",
     "**License**: [Apache 2.0](https://ai.google.dev/gemma/docs/gemma_4_license) | **Authors**: [Google DeepMind](https://deepmind.google/models/gemma/)"
    ],
    "summary_evidence": [
     "Gemma is a family of open models built by Google DeepMind.",
     "Gemma 4 models are multimodal, handling text and image input (with audio supported on E2B, E4B, and 12B) and generating text output."
    ]
   }
  },
  {
   "url": "https://developer.nvidia.com/blog/post-train-nvidia-cosmos-3-edge-for-on-device-robot-control",
   "title": "Post-Train NVIDIA Cosmos 3 Edge for On-Device Robot Control | NVIDIA Technical Blog",
   "published_at": "2026-08-19T16:00:00",
   "release": {
    "organization": "NVIDIA",
    "model": "Cosmos 3 Edge",
    "release_date": "2026-08-19",
    "parameters": "4B",
    "context_window": null,
    "license": "OpenMDW1.1",
    "summary": "on-device robot manipulation policies and physical AI control",
    "organization_evidence": [
     "Post-Train NVIDIA Cosmos 3 Edge for On-Device Robot Control"
    ],
    "model_evidence": [
     "Post-Train NVIDIA Cosmos 3 Edge for On-Device Robot Control"
    ],
    "release_date_evidence": [
     "Post-Train NVIDIA Cosmos 3 Edge for On-Device Robot Control",
     "Aug 19, 2026"
    ],
    "parameters_evidence": [
     "Post-Train NVIDIA Cosmos 3 Edge for On-Device Robot Control"
    ],
    "context_window_evidence": null,
    "license_evidence": [
     "Post-Train NVIDIA Cosmos 3 Edge for On-Device Robot Control",
     "More broadly, Cosmos\u2019s open weights and framework, with the OpenMDW1.1 license, make post-training a powerful, flexible, and easy way to create specialized, highly performant, and accurate custom models for physical AI."
    ],
    "summary_evidence": [
     "Post-Train NVIDIA Cosmos 3 Edge for On-Device Robot Control",
     "- [NVIDIA Jetson Thor](https://www.nvidia.com/en-us/autonomous-machines/embedded-systems/jetson-thor/) can run the 4B Cosmos 3 Edge omni-model natively for on-device robot manipulation policies."
    ]
   }
  },
  {
   "url": "https://huggingface.co/nvidia/DeepSeek-V4-Flash-0731-NVFP4",
   "title": "nvidia/DeepSeek-V4-Flash-0731-NVFP4 \u00b7 Hugging Face",
   "published_at": "2026-08-31T16:53:50",
   "release": {
    "organization": "NVIDIA",
    "model": "DeepSeek-V4-Flash-0731-NVFP4",
    "release_date": "2026-08-31",
    "parameters": "304B in total and 13B activated",
    "context_window": "1 million tokens",
    "license": null,
    "summary": "well-suited for advanced reasoning, agentic AI applications, tool use scenarios, and complex problem-solving in domains such as mathematics, software engineering, and enterprise AI assistants",
    "organization_evidence": [
     "# Model Overview",
     "## Description:",
     "The NVIDIA DeepSeek-V4-Flash-0731-NVFP4 model is the quantized version of DeepSeek AI's DeepSeek-V4-Flash-0731 model, an autoregressive Mixture-of-Experts language model that uses an optimized Transformer architecture with hybrid attention (Compressed Sparse Attention and Heavily Compressed Attention) and Manifold-Constrained Hyper-Connections."
    ],
    "model_evidence": [
     "# Model Overview",
     "## Description:",
     "The NVIDIA DeepSeek-V4-Flash-0731-NVFP4 model is the quantized version of DeepSeek AI's DeepSeek-V4-Flash-0731 model, an autoregressive Mixture-of-Experts language model that uses an optimized Transformer architecture with hybrid attention (Compressed Sparse Attention and Heavily Compressed Attention) and Manifold-Constrained Hyper-Connections."
    ],
    "release_date_evidence": [
     "# Model Overview",
     "## Release Date:",
     "Hugging Face 08/31/2026 via [https://huggingface.co/nvidia/DeepSeek-V4-Flash-0731-NVFP4](https://huggingface.co/nvidia/DeepSeek-V4-Flash-0731-NVFP4)"
    ],
    "parameters_evidence": [
     "# Model Overview",
     "## Model Architecture:",
     "**Number of Model Parameters:** 304B in total and 13B activated"
    ],
    "context_window_evidence": [
     "# Model Overview",
     "## Input:",
     "Maximum context length of 1 million tokens."
    ],
    "license_evidence": null,
    "summary_evidence": [
     "# Model Overview",
     "### Use Case:",
     "DeepSeek V4 is well-suited for advanced reasoning, agentic AI applications, tool use scenarios, and complex problem-solving in domains such as mathematics, software engineering, and enterprise AI assistants."
    ]
   }
  },
  {
   "url": "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4",
   "title": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4 \u00b7 Hugging Face",
   "published_at": "2026-08-17T15:06:50",
   "release": {
    "organization": "NVIDIA",
    "model": "Nemotron-3-Super-120B-A12B-NVFP4",
    "release_date": "2026-03-11",
    "parameters": "120B Total / 12B Active",
    "context_window": "1M tokens",
    "license": "NVIDIA Nemotron Open Model License",
    "summary": "general purpose reasoning and chat model optimized for collaborative agents and high-volume workloads",
    "organization_evidence": [
     "# NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4",
     "## Description",
     "**Nemotron-3-Super-120B-A12B-NVFP4** is a large language model (LLM) trained by NVIDIA, designed to deliver strong agentic, reasoning, and conversational capabilities."
    ],
    "model_evidence": [
     "# NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4",
     "## Description",
     "**Nemotron-3-Super-120B-A12B-NVFP4** is a large language model (LLM) trained by NVIDIA, designed to deliver strong agentic, reasoning, and conversational capabilities."
    ],
    "release_date_evidence": [
     "# NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4",
     "## Model Summary",
     "| **Release Date** | March 11, 2026 |"
    ],
    "parameters_evidence": [
     "# NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4",
     "## Model Architecture",
     "- **Number of model parameters:** 120B Total / 12B Active"
    ],
    "context_window_evidence": [
     "# NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4",
     "## Model Summary",
     "| **Context Length** | Up to 1M tokens |"
    ],
    "license_evidence": [
     "# NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4",
     "## Model Summary",
     "| **License** | [NVIDIA Nemotron Open Model License](https://www.nvidia.com/en-us/agreements/enterprise-software/nvidia-nemotron-open-model-license/) |"
    ],
    "summary_evidence": [
     "# NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4",
     "### Use Case",
     "NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4 is a general purpose reasoning and chat model intended to be used in English, Code, and supported multilingual contexts."
    ]
   }
  },
  {
   "url": "https://build.nvidia.com/nvidia/nemotron-3-super-120b-a12b/modelcard?:~:text=Description_and%20120B%20parameters%20in%20total.&text=This%20model%20is%20ready%20for%20commercial%20use.",
   "title": "nemotron-3-super-120b-a12b Model by NVIDIA | NVIDIA NIM",
   "published_at": null,
   "release": {
    "organization": "NVIDIA",
    "model": "Nemotron-3-Super-120B-A12B",
    "release_date": "2026-03-11",
    "parameters": "120B (12B active)",
    "context_window": "1M tokens",
    "license": "NVIDIA Nemotron Open Model License",
    "summary": "general purpose reasoning and chat model optimized for collaborative agents and high-volume workloads",
    "organization_evidence": [
     "nemotron-3-super-120b-a12b Model by NVIDIA | NVIDIA NIM NVIDIA-Nemotron-3-Super-120B-A12B Model Summary"
    ],
    "model_evidence": [
     "nemotron-3-super-120b-a12b Model by NVIDIA | NVIDIA NIM NVIDIA-Nemotron-3-Super-120B-A12B Model Summary",
     "NVIDIA-Nemotron-3-Super-120B-A12B Model Summary   Total Parameters 120B (12B active)  Architecture LatentMoE - Mamba-2 + MoE + Attention hybrid with Multi-Token Prediction (MTP)  Context Length Up to 1M tokens  Minimum GPU Requirement 8\u00d7 H100-80GB  Supported Languages English, French, German, Italian, Japanese, Spanish, Chinese  Best For Agentic workflows, long-context reasoning, high-volume workloads (e.g. IT ticket automation), tool use, RAG  Reasoning Mode Configurable on/off via chat template (enable_thinking=True/False)  License NVIDIA Nemotron Open Model License  Release Date March 11, 2026"
    ],
    "release_date_evidence": [
     "nemotron-3-super-120b-a12b Model by NVIDIA | NVIDIA NIM NVIDIA-Nemotron-3-Super-120B-A12B Model Summary",
     "Release Date March 11, 2026"
    ],
    "parameters_evidence": [
     "nemotron-3-super-120b-a12b Model by NVIDIA | NVIDIA NIM NVIDIA-Nemotron-3-Super-120B-A12B Model Summary",
     "Total Parameters 120B (12B active)"
    ],
    "context_window_evidence": [
     "nemotron-3-super-120b-a12b Model by NVIDIA | NVIDIA NIM NVIDIA-Nemotron-3-Super-120B-A12B Model Summary",
     "Context Length Up to 1M tokens"
    ],
    "license_evidence": [
     "nemotron-3-super-120b-a12b Model by NVIDIA | NVIDIA NIM NVIDIA-Nemotron-3-Super-120B-A12B Model Summary",
     "License NVIDIA Nemotron Open Model License"
    ],
    "summary_evidence": [
     "nemotron-3-super-120b-a12b Model by NVIDIA | NVIDIA NIM NVIDIA-Nemotron-3-Super-120B-A12B Model Summary",
     "NVIDIA-Nemotron-3-Super-120B-A12B-BF16 is a general purpose reasoning and chat model intended to be used in English, Code, and supported multilingual contexts."
    ]
   }
  }
 ],
 "notes": [
  "Merged 8 queries (1528 total) \u2192 1524 unique results",
  "Search result set: rbe28438d028 (1524 rows), reference it as FROM rbe28438d028 in follow-up queries"
 ]
}