<!-- mobian-agent-page publisher="dailydev" canonical="https://daily.dev/tags/tensorrt" -->

---
title: TensorRT News &amp; Updates | daily.dev
description: TensorRT news and updates for the NVIDIA SDK that compiles and optimizes trained neural networks for inference on NVIDIA GPUs. Readers can learn about TensorRT-LLM, engine building, quantization, Triton deployment, RTX desktop support and reported latency results.
canonical: https://daily.dev/tags/tensorrt
twitter:card: summary_large_image
twitter:site: @dailydotdev
og:url: https://daily.dev/tags/tensorrt
og:type: website
og:site_name: daily.dev
og:title: TensorRT News &amp; Updates | daily.dev
og:description: TensorRT news and updates for the NVIDIA SDK that compiles and optimizes trained neural networks for inference on NVIDIA GPUs. Readers can learn about TensorRT-LLM, engine building, quantization, Triton deployment, RTX desktop support and reported latency results.
og:image: https://api.daily.dev/og/tags/tensorrt.png
og:image:width: 1200
og:image:height: 630
---

## Recommended TensorRT stories

## Who to follow for TensorRT

## Top sources covering TensorRT

## Most upvoted TensorRT posts

## Best discussed TensorRT posts

## All posts about TensorRT

```json
{"@context":"https://schema.org","@graph":[{"@type":"Organization","@id":"https://daily.dev/#organization","name":"daily.dev","url":"https://daily.dev","logo":{"@type":"ImageObject","url":"https://daily.dev/apple-touch-icon.png","width":180,"height":180},"sameAs":["https://twitter.com/dailydotdev","https://github.com/dailydotdev","https://www.linkedin.com/company/daily-dev-ltd"]},{"@type":"WebSite","@id":"https://daily.dev/#website","url":"https://daily.dev","name":"daily.dev","publisher":{"@id":"https://daily.dev/#organization"},"potentialAction":{"@type":"SearchAction","target":{"@type":"EntryPoint","urlTemplate":"https://daily.dev/search?q={search_term_string}"},"query-input":"required name=search_term_string"}}]}
{"@context":"https://schema.org","@graph":[{"@type":"CollectionPage","@id":"https://daily.dev/tags/tensorrt#page","url":"https://daily.dev/tags/tensorrt","name":"TensorRT News & Updates","description":"TensorRT news and updates for the NVIDIA SDK that compiles and optimizes trained neural networks for inference on NVIDIA GPUs. Readers can learn about TensorRT-LLM, engine building, quantization, Triton deployment, RTX desktop support and reported latency results.","isPartOf":{"@type":"WebSite","url":"https://daily.dev"}},{"@type":"ItemList","@id":"https://daily.dev/tags/tensorrt#items","numberOfItems":10,"itemListElement":[{"@type":"ListItem","position":1,"url":"https://daily.dev/posts/accelerate-generative-ai-inference-performance-with-nvidia-tensorrt-model-optimizer-now-publicly-av-v5wsawqic","name":"Accelerate Generative AI Inference Performance with NVIDIA TensorRT Model Optimizer, Now Publicly Available"},{"@type":"ListItem","position":2,"url":"https://daily.dev/posts/how-tensorrt-accelerates-ai-on-rtx-pcs-dps8jvl6k","name":"How TensorRT Accelerates AI on RTX PCs"},{"@type":"ListItem","position":3,"url":"https://daily.dev/posts/city-of-raleigh-taps-nvidia-metropolis-to-improve-traffic-zd7mhib4t","name":"City of Raleigh Taps NVIDIA Metropolis to Improve Traffic"},{"@type":"ListItem","position":4,"url":"https://daily.dev/posts/nvidia-tensorrt-accelerates-stable-diffusion-nearly-2x-faster-with-8-bit-post-training-quantization-hylvx9m1t","name":"NVIDIA TensorRT Accelerates Stable Diffusion Nearly 2x Faster with 8-bit Post-Training Quantization"},{"@type":"ListItem","position":5,"url":"https://daily.dev/posts/detecting-real-time-waste-contamination-using-edge-computing-and-video-analytics-k9j5utzcq","name":"Detecting Real-Time Waste Contamination Using Edge Computing and Video Analytics"},{"@type":"ListItem","position":6,"url":"https://daily.dev/posts/deploying-llms-into-production-using-tensorrt-llm-yjxi5o90x","name":"Deploying LLMs Into Production Using TensorRT LLM"},{"@type":"ListItem","position":7,"url":"https://daily.dev/posts/emulating-the-attention-mechanism-in-transformer-models-with-a-fully-convolutional-network-u6wgehqit","name":"Emulating the Attention Mechanism in Transformer Models with a Fully Convolutional Network"},{"@type":"ListItem","position":8,"url":"https://daily.dev/posts/collabora-whisperfusion-whisperfusion-builds-upon-the-capabilities-of-whisperlive-and-whisperspeech-rsjmtlly0","name":"collabora/WhisperFusion: WhisperFusion builds upon the capabilities of WhisperLive and WhisperSpeech to provide a seamless conversations with an AI."},{"@type":"ListItem","position":9,"url":"https://daily.dev/posts/how-amazon-and-nvidia-help-sellers-create-better-product-listings-with-ai-yehmxyoz7","name":"How Amazon and NVIDIA Help Sellers Create Better Product Listings With AI"},{"@type":"ListItem","position":10,"url":"https://daily.dev/posts/colossal-ai-team-open-sources-swiftinfer-a-tensorrt-based-implementation-of-the-streamingllm-algori-l32crlqpu","name":"Colossal-AI Team Open-Sources SwiftInfer: A TensorRT-Based Implementation of the StreamingLLM Algorithm"}]},{"@type":"BreadcrumbList","itemListElement":[{"@type":"ListItem","position":1,"name":"Home","item":"https://daily.dev"},{"@type":"ListItem","position":2,"name":"Tags","item":"https://daily.dev/tags"},{"@type":"ListItem","position":3,"name":"TensorRT"}]}]}
```

