{"catalog":{"title":"AI chips compared: H100, B200, Rubin, MI355X, TPU and Trainium accelerator specs","path":"/ai-chips","updatedAt":"2026-10-05T05:12:32.825Z","fields":[{"id":"vendor","label":"Vendor","type":"select","options":["Nvidia","AMD","Google","AWS","Intel","Cerebras","Huawei","Microsoft","Meta"]},{"id":"architecture","label":"Architecture","type":"text"},{"id":"memory","label":"Memory","type":"text"},{"id":"memoryGb","label":"Memory (GB)","type":"number","format":{"suffix":" GB","maximumFractionDigits":1}},{"id":"memoryBandwidthTbs","label":"Memory bandwidth (TB/s)","type":"number","format":{"suffix":" TB/s","maximumFractionDigits":3}},{"id":"fp8Tflops","label":"FP8 (TFLOPS)","type":"number","format":{"suffix":" TFLOPS","maximumFractionDigits":0}},{"id":"fp4Pflops","label":"FP4 (PFLOPS)","type":"number","format":{"suffix":" PFLOPS","maximumFractionDigits":3}},{"id":"precisionNote","label":"Precision note (sparse or dense)","type":"textarea"},{"id":"tdpW","label":"Power (W)","type":"number","format":{"suffix":" W","maximumFractionDigits":0}},{"id":"formFactor","label":"Form factor","type":"text"},{"id":"launch","label":"Launch","type":"text"},{"id":"status","label":"Status","type":"select","options":["Shipping","Announced","In production"]},{"id":"datasheet","label":"Vendor source","type":"url"},{"id":"checked","label":"Checked","type":"date"}]},"entries":[{"slug":"amd-instinct-mi300x","name":"AMD Instinct MI300X","path":"/ai-chips/amd-instinct-mi300x","category":null,"updatedAt":"2026-10-05T05:05:26.026Z","fields":{"vendor":"AMD","architecture":"CDNA 3","memory":"192 GB HBM3","memoryGb":192,"memoryBandwidthTbs":5.3,"fp8Tflops":2610,"fp4Pflops":null,"precisionNote":"2.61 PFLOPS FP8 (E5M2, E4M3) is AMD's figure without sparsity; 5.22 PFLOPS with structured sparsity.","tdpW":750,"formFactor":"OAM module","launch":"Launch date December 6, 2023","status":"Shipping","datasheet":"https://www.amd.com/en/products/accelerators/instinct/mi300/mi300x.html","checked":"2026-10-05"}},{"slug":"amd-instinct-mi325x","name":"AMD Instinct MI325X","path":"/ai-chips/amd-instinct-mi325x","category":null,"updatedAt":"2026-10-05T05:05:28.830Z","fields":{"vendor":"AMD","architecture":"CDNA 3","memory":"256 GB HBM3E","memoryGb":256,"memoryBandwidthTbs":6,"fp8Tflops":2610,"fp4Pflops":null,"precisionNote":"2.61 PFLOPS FP8 (E5M2, E4M3) is AMD's figure without sparsity; 5.22 PFLOPS with structured sparsity.","tdpW":1000,"formFactor":"OAM module","launch":"Launch date October 10, 2024","status":"Shipping","datasheet":"https://www.amd.com/en/products/accelerators/instinct/mi300/mi325x.html","checked":"2026-10-05"}},{"slug":"amd-instinct-mi355x","name":"AMD Instinct MI355X","path":"/ai-chips/amd-instinct-mi355x","category":null,"updatedAt":"2026-10-05T05:05:32.766Z","fields":{"vendor":"AMD","architecture":"CDNA 4","memory":"288 GB HBM3E","memoryGb":288,"memoryBandwidthTbs":8,"fp8Tflops":5000,"fp4Pflops":10.1,"precisionNote":"OCP-FP8 5 PFLOPS is without sparsity; 10.1 PFLOPS with structured sparsity. MXFP4 10.1 PFLOPS (no sparsity variant listed).","tdpW":1400,"formFactor":"OAM module","launch":"Launch date June 12, 2025","status":"Shipping","datasheet":"https://www.amd.com/en/products/accelerators/instinct/mi350/mi355x.html","checked":"2026-10-05"}},{"slug":"amd-instinct-mi430x","name":"AMD Instinct MI430X","path":"/ai-chips/amd-instinct-mi430x","category":null,"updatedAt":"2026-10-05T05:05:36.311Z","fields":{"vendor":"AMD","architecture":"Instinct MI400 series","memory":"432 GB HBM4","memoryGb":432,"memoryBandwidthTbs":23.3,"fp8Tflops":null,"fp4Pflops":9.2,"precisionNote":"AMD gives up to 9.2 PFLOPS peak theoretical FP4/MXFP4 and 288 TFLOPS hardware FP64, with no sparsity label.","tdpW":null,"formFactor":"GPU for HPC and sovereign AI systems","launch":"Launched July 23, 2026; expected availability 2027","status":"Announced","datasheet":"https://www.amd.com/en/products/accelerators/instinct/mi400/mi430x.html","checked":"2026-10-05"}},{"slug":"amd-instinct-mi455x","name":"AMD Instinct MI455X","path":"/ai-chips/amd-instinct-mi455x","category":null,"updatedAt":"2026-10-05T05:05:40.174Z","fields":{"vendor":"AMD","architecture":"CDNA 5","memory":"432 GB HBM4","memoryGb":432,"memoryBandwidthTbs":23.3,"fp8Tflops":20100,"fp4Pflops":40.3,"precisionNote":"AMD lists peak OCP FP8 20.1 PFLOPS and OCP MXFP4 40.3 PFLOPS with no sparsity label (AMD gives separate structured-sparsity figures only for FP16, BF16 and INT8).","tdpW":null,"formFactor":"Enhanced Accelerator Module (EAM), direct liquid cooling, Helios rack","launch":"Launched July 23, 2026 (Advancing AI 2026)","status":"Announced","datasheet":"https://www.amd.com/en/products/accelerators/instinct/mi400/mi455x.html","checked":"2026-10-05"}},{"slug":"aws-trainium2","name":"AWS Trainium2","path":"/ai-chips/aws-trainium2","category":null,"updatedAt":"2026-10-05T05:05:43.466Z","fields":{"vendor":"AWS","architecture":"NeuronCore-v3 (8 cores)","memory":"96 GiB HBM","memoryGb":96,"memoryBandwidthTbs":2.9,"fp8Tflops":1299,"fp4Pflops":null,"precisionNote":"1,299 TFLOPS FP8 is AWS's dense figure; AWS lists 2,563 TFLOPS for FP8/FP16/BF16/TF32 with sparsity.","tdpW":null,"formFactor":"Cloud instances (EC2 Trn2, Trn2 UltraServers)","launch":null,"status":"Shipping","datasheet":"https://awsdocs-neuron.readthedocs-hosted.com/en/latest/about-neuron/arch/neuron-hardware/trainium2.html","checked":"2026-10-05"}},{"slug":"aws-trainium3","name":"AWS Trainium3","path":"/ai-chips/aws-trainium3","category":null,"updatedAt":"2026-10-05T05:05:48.396Z","fields":{"vendor":"AWS","architecture":"NeuronCore-v4 (8 cores)","memory":"144 GB HBM3e","memoryGb":144,"memoryBandwidthTbs":4.9,"fp8Tflops":2517,"fp4Pflops":2.517,"precisionNote":"AWS lists 2,517 TFLOPS for MXFP8/MXFP4 with no sparsity label; its separate sparse figure (2,517 TFLOPS) covers FP16/BF16/TF32.","tdpW":null,"formFactor":"Cloud instances (Trainium3 UltraServers, up to 144 chips)","launch":null,"status":"Shipping","datasheet":"https://awsdocs-neuron.readthedocs-hosted.com/en/latest/about-neuron/arch/neuron-hardware/trainium3.html","checked":"2026-10-05"}},{"slug":"cerebras-wse-3t-cs-4","name":"Cerebras WSE-3T (CS-4)","path":"/ai-chips/cerebras-wse-3t-cs-4","category":null,"updatedAt":"2026-10-05T05:05:51.066Z","fields":{"vendor":"Cerebras","architecture":"Wafer Scale Engine 3 Turbo","memory":"44 GB on-wafer SRAM","memoryGb":44,"memoryBandwidthTbs":43200,"fp8Tflops":null,"fp4Pflops":null,"precisionNote":"Cerebras rates AI compute at 250 PFLOPS per wafer (750 PFLOPS per CS-4), shown as sparse FP16. Bandwidth is on-wafer SRAM, 43.2 PB/s.","tdpW":null,"formFactor":"Wafer-scale processor; CS-4 rack holds three","launch":"Unveiled August 18, 2026; first CS-4 shipments planned for Q3 2026","status":"Announced","datasheet":"https://www.cerebras.ai/cs4-datasheet","checked":"2026-10-05"}},{"slug":"google-tpu-8i","name":"Google TPU 8i","path":"/ai-chips/google-tpu-8i","category":null,"updatedAt":"2026-10-05T05:05:54.896Z","fields":{"vendor":"Google","architecture":"8th-gen TPU (inference)","memory":"288 GB HBM","memoryGb":288,"memoryBandwidthTbs":8.601,"fp8Tflops":null,"fp4Pflops":10.1,"precisionNote":"Google lists peak FP4 per chip as 10.1 PFLOPS with no sparse/dense label. Bandwidth given as 8,601 GB/s.","tdpW":null,"formFactor":"Cloud TPU","launch":"Announced April 22, 2026 (Google Cloud Next); general availability planned later in 2026","status":"Announced","datasheet":"https://cloud.google.com/blog/products/compute/tpu-8t-and-tpu-8i-technical-deep-dive","checked":"2026-10-05"}},{"slug":"google-tpu-8t","name":"Google TPU 8t","path":"/ai-chips/google-tpu-8t","category":null,"updatedAt":"2026-10-05T05:05:57.816Z","fields":{"vendor":"Google","architecture":"8th-gen TPU (training)","memory":"216 GB HBM","memoryGb":216,"memoryBandwidthTbs":6.528,"fp8Tflops":null,"fp4Pflops":12.6,"precisionNote":"Google lists peak FP4 per chip as 12.6 PFLOPS with no sparse/dense label. Bandwidth given as 6,528 GB/s.","tdpW":null,"formFactor":"Cloud TPU (superpods up to 9,600 chips)","launch":"Announced April 22, 2026 (Google Cloud Next); general availability planned later in 2026","status":"Announced","datasheet":"https://cloud.google.com/blog/products/compute/tpu-8t-and-tpu-8i-technical-deep-dive","checked":"2026-10-05"}},{"slug":"google-tpu7x-ironwood","name":"Google TPU7x (Ironwood)","path":"/ai-chips/google-tpu7x-ironwood","category":null,"updatedAt":"2026-10-05T05:06:01.526Z","fields":{"vendor":"Google","architecture":"Ironwood (7th-gen TPU), dual chiplet","memory":"192 GB HBM","memoryGb":192,"memoryBandwidthTbs":7.37,"fp8Tflops":4614,"fp4Pflops":null,"precisionNote":"Google lists peak FP8 compute per chip as 4,614 TFLOPS (BF16 2,307) with no sparse/dense label. Bandwidth stated as approximately 7.37 TB/s (7,380 GBps in Google's table).","tdpW":null,"formFactor":"Cloud TPU (4-chip VMs, pods up to 9,216 chips)","launch":null,"status":"Shipping","datasheet":"https://cloud.google.com/tpu/docs/tpu7x","checked":"2026-10-05"}},{"slug":"huawei-ascend-950dt","name":"Huawei Ascend 950DT","path":"/ai-chips/huawei-ascend-950dt","category":null,"updatedAt":"2026-10-05T05:12:32.825Z","fields":{"vendor":"Huawei","architecture":"Ascend 950 (SIMD + SIMT)","memory":"144 GB HiZQ 2.0 (Huawei HBM)","memoryGb":144,"memoryBandwidthTbs":4,"fp8Tflops":1000,"fp4Pflops":2,"precisionNote":"Huawei states 1 PFLOPS in FP8, MXFP8 and HiF8 and 2 PFLOPS in MXFP4 for the Ascend 950 series, with no sparse/dense label.","tdpW":null,"formFactor":"NPU in Atlas 950 SuperPoD","launch":"Specs announced September 18, 2025; availability Q4 2026 per Huawei","status":"Announced","datasheet":"https://www.huawei.com/en/news/2025/9/hc-xu-keynote-speech","checked":"2026-10-05"}},{"slug":"huawei-ascend-960","name":"Huawei Ascend 960 (Atlas 960E SuperPoD)","path":"/ai-chips/huawei-ascend-960","category":null,"updatedAt":"2026-10-05T05:06:05.715Z","fields":{"vendor":"Huawei","architecture":"Ascend 960","memory":"Up to 1 PB HBM per Atlas 960E SuperPoD (4,096 NPUs)","memoryGb":null,"memoryBandwidthTbs":null,"fp8Tflops":2000,"fp4Pflops":4,"precisionNote":"Per-chip 2 PFLOPS FP8 and 4 PFLOPS FP4 from Huawei's 2025 roadmap, no sparse/dense label. Atlas 960E SuperPoD: 8 EFLOPS FP8.","tdpW":null,"formFactor":"NPU in Atlas 960E SuperPoD","launch":"Atlas 960E unveiled September 17, 2026; Ascend 960DT Q1 2027, 960PR Q3 2027 per Huawei","status":"Announced","datasheet":"https://www.huawei.com/en/news/2026/9/hc-wang-keynote","checked":"2026-10-05"}},{"slug":"intel-gaudi-3","name":"Intel Gaudi 3","path":"/ai-chips/intel-gaudi-3","category":null,"updatedAt":"2026-10-05T05:06:09.593Z","fields":{"vendor":"Intel","architecture":"Gaudi (5th-gen TPC), 5nm","memory":"128 GB HBM2e","memoryGb":128,"memoryBandwidthTbs":3.7,"fp8Tflops":1678,"fp4Pflops":null,"precisionNote":"1,678 TFLOPS FP8 MME (matrix engine) from Intel's white paper, no sparsity label; Intel's Gaudi 3 e-book lists 1,835 TFLOPS FP8 MME.","tdpW":900,"formFactor":"OAM mezzanine card (HL-325L); PCIe card (HL-338)","launch":null,"status":"Shipping","datasheet":"https://cdrdv2-public.intel.com/817486/gaudi-3-ai-accelerator-white-paper.pdf","checked":"2026-10-05"}},{"slug":"meta-mtia-200","name":"Meta MTIA 200","path":"/ai-chips/meta-mtia-200","category":null,"updatedAt":"2026-10-02T12:09:52.625Z","fields":{"vendor":"Meta","architecture":"MTIA, 8x8 PE grid, TSMC 5nm","memory":"128 GB LPDDR5 (off-chip) + 256 MB SRAM","memoryGb":128,"memoryBandwidthTbs":0.2048,"fp8Tflops":null,"fp4Pflops":null,"precisionNote":"Meta lists INT8 GEMM 354 TFLOPS dense / 708 with sparsity and FP16/BF16 177 dense / 354 with sparsity. Bandwidth is the LPDDR5 figure (204.8 GB/s); on-chip SRAM runs at 2.7 TB/s.","tdpW":90,"formFactor":"Accelerator board (two per board, up to 72 per rack)","launch":"Published April 10, 2024","status":"Shipping","datasheet":"https://ai.meta.com/blog/next-generation-meta-training-inference-accelerator-AI-MTIA/","checked":"2026-10-02"}},{"slug":"microsoft-maia-200","name":"Microsoft Maia 200","path":"/ai-chips/microsoft-maia-200","category":null,"updatedAt":"2026-10-05T05:06:13.240Z","fields":{"vendor":"Microsoft","architecture":"Maia 200, TSMC 3nm","memory":"216 GB HBM3e","memoryGb":216,"memoryBandwidthTbs":7,"fp8Tflops":5000,"fp4Pflops":10,"precisionNote":"Microsoft states \"over 10 petaFLOPS\" FP4 and \"over 5 petaFLOPS\" FP8 per chip, with no sparse/dense label.","tdpW":750,"formFactor":"Azure servers (four accelerators per tray)","launch":"Introduced January 26, 2026","status":"Shipping","datasheet":"https://blogs.microsoft.com/blog/2026/01/26/maia-200-the-ai-accelerator-built-for-inference/","checked":"2026-10-05"}},{"slug":"nvidia-b200","name":"Nvidia B200 (Blackwell)","path":"/ai-chips/nvidia-b200","category":null,"updatedAt":"2026-10-05T05:06:17.916Z","fields":{"vendor":"Nvidia","architecture":"Blackwell","memory":"Up to 192 GB HBM3E (180 GB per GPU in HGX B200)","memoryGb":192,"memoryBandwidthTbs":8,"fp8Tflops":5000,"fp4Pflops":10,"precisionNote":"Dense figures from Nvidia's Blackwell GPU table: FP8 5 PFLOPS dense / 10 sparse; NVFP4 10 PFLOPS dense / 20 sparse. HGX B200 8-GPU board: FP8 72 PFLOPS sparse, FP4 144 sparse / 72 dense.","tdpW":1200,"formFactor":"SXM module (HGX B200 / DGX B200)","launch":null,"status":"Shipping","datasheet":"https://developer.nvidia.com/blog/inside-nvidia-blackwell-ultra-the-chip-powering-the-ai-factory-era/","checked":"2026-10-05"}},{"slug":"nvidia-b300","name":"Nvidia B300 (Blackwell Ultra)","path":"/ai-chips/nvidia-b300","category":null,"updatedAt":"2026-10-05T05:06:21.763Z","fields":{"vendor":"Nvidia","architecture":"Blackwell Ultra","memory":"288 GB HBM3E","memoryGb":288,"memoryBandwidthTbs":8,"fp8Tflops":5000,"fp4Pflops":15,"precisionNote":"Dense figures from Nvidia's chip table: FP8 5 PFLOPS dense / 10 sparse; NVFP4 15 PFLOPS dense / 20 sparse. HGX B300 board: FP4 144 sparse / 108 dense PFLOPS, FP8 72 PFLOPS sparse.","tdpW":1400,"formFactor":"SXM module (HGX B300 / DGX B300)","launch":null,"status":"Shipping","datasheet":"https://developer.nvidia.com/blog/inside-nvidia-blackwell-ultra-the-chip-powering-the-ai-factory-era/","checked":"2026-10-05"}},{"slug":"nvidia-gb200-nvl72","name":"Nvidia GB200 NVL72","path":"/ai-chips/nvidia-gb200-nvl72","category":null,"updatedAt":"2026-10-05T05:06:25.012Z","fields":{"vendor":"Nvidia","architecture":"Grace Blackwell","memory":"13.4 TB HBM3E (72 GPUs)","memoryGb":13400,"memoryBandwidthTbs":576,"fp8Tflops":720000,"fp4Pflops":720,"precisionNote":"Rack totals. FP8/FP6 720 PFLOPS is sparse (Nvidia: dense is one-half). NVFP4 1,440 PFLOPS sparse / 720 PFLOPS dense.","tdpW":null,"formFactor":"Rack-scale system: 72 GPUs, 36 Grace CPUs","launch":null,"status":"Shipping","datasheet":"https://www.nvidia.com/en-us/data-center/gb200-nvl72/","checked":"2026-10-05"}},{"slug":"nvidia-gb300-nvl72","name":"Nvidia GB300 NVL72","path":"/ai-chips/nvidia-gb300-nvl72","category":null,"updatedAt":"2026-10-05T05:06:28.905Z","fields":{"vendor":"Nvidia","architecture":"Grace Blackwell Ultra","memory":"20 TB HBM3E (72 GPUs)","memoryGb":20000,"memoryBandwidthTbs":576,"fp8Tflops":720000,"fp4Pflops":1080,"precisionNote":"Rack totals. Nvidia: all Tensor Core specs are with sparsity unless noted. FP8/FP6 720 PFLOPS sparse; FP4 1,440 PFLOPS sparse / 1,080 PFLOPS without sparsity. Bandwidth is stated as up to 576 TB/s.","tdpW":null,"formFactor":"Rack-scale system: 72 GPUs, 36 Grace CPUs","launch":null,"status":"Shipping","datasheet":"https://www.nvidia.com/en-us/data-center/gb300-nvl72/","checked":"2026-10-05"}},{"slug":"nvidia-groq-3-lpx","name":"Nvidia Groq 3 LPX","path":"/ai-chips/nvidia-groq-3-lpx","category":null,"updatedAt":"2026-10-02T12:11:11.157Z","fields":{"vendor":"Nvidia","architecture":"Groq 3 LPU","memory":"128 GB SRAM per rack (256 LPUs)","memoryGb":128,"memoryBandwidthTbs":40000,"fp8Tflops":null,"fp4Pflops":null,"precisionNote":"Rack totals of on-chip SRAM: 128 GB and 40 PB/s memory bandwidth, plus 640 TB/s scale-up bandwidth per rack.","tdpW":null,"formFactor":"Rack-scale system: 256 LPUs","launch":"Full production announced August 24, 2026","status":"In production","datasheet":"https://nvidianews.nvidia.com/news/nvidia-groq-3-lpx-now-in-full-production-with-world-class-speed-for-agentic-ai","checked":"2026-10-02"}},{"slug":"nvidia-h100","name":"Nvidia H100 SXM","path":"/ai-chips/nvidia-h100","category":null,"updatedAt":"2026-10-05T05:06:32.596Z","fields":{"vendor":"Nvidia","architecture":"Hopper","memory":"80 GB HBM3","memoryGb":80,"memoryBandwidthTbs":3.35,"fp8Tflops":3958,"fp4Pflops":null,"precisionNote":"3,958 TFLOPS FP8 Tensor Core is stated with sparsity on Nvidia's H100 page; Nvidia's Blackwell Ultra chip table gives Hopper FP8 as 2 PFLOPS dense / 4 sparse. H100 NVL: 3,341 TFLOPS with sparsity.","tdpW":700,"formFactor":"SXM module (HGX H100 / DGX H100)","launch":null,"status":"Shipping","datasheet":"https://www.nvidia.com/en-us/data-center/h100/","checked":"2026-10-05"}},{"slug":"nvidia-h200","name":"Nvidia H200 SXM","path":"/ai-chips/nvidia-h200","category":null,"updatedAt":"2026-10-05T05:06:35.946Z","fields":{"vendor":"Nvidia","architecture":"Hopper","memory":"141 GB HBM3e","memoryGb":141,"memoryBandwidthTbs":4.8,"fp8Tflops":3958,"fp4Pflops":null,"precisionNote":"3,958 TFLOPS FP8 Tensor Core is stated with sparsity on Nvidia's H200 page (H200 NVL: 3,341 TFLOPS with sparsity).","tdpW":700,"formFactor":"SXM module (HGX H200)","launch":null,"status":"Shipping","datasheet":"https://www.nvidia.com/en-us/data-center/h200/","checked":"2026-10-05"}},{"slug":"nvidia-rubin-gpu","name":"Nvidia Rubin GPU","path":"/ai-chips/nvidia-rubin-gpu","category":null,"updatedAt":"2026-10-05T05:06:39.855Z","fields":{"vendor":"Nvidia","architecture":"Rubin","memory":"288 GB HBM4","memoryGb":288,"memoryBandwidthTbs":22,"fp8Tflops":17500,"fp4Pflops":35,"precisionNote":"FP8/FP6 17.5 PFLOPS and NVFP4 35 PFLOPS are Nvidia's training figures, marked dense. NVFP4 inference is listed as 50 PFLOPS without a sparse/dense label. Preliminary per Nvidia.","tdpW":null,"formFactor":"GPU in Vera Rubin Superchip / NVL72 rack","launch":"Launched January 5, 2026 (CES); full production announced May 31, 2026; production shipments from fall 2026","status":"In production","datasheet":"https://www.nvidia.com/en-us/data-center/vera-rubin-nvl72/","checked":"2026-10-05"}},{"slug":"nvidia-vera-rubin-nvl72","name":"Nvidia Vera Rubin NVL72","path":"/ai-chips/nvidia-vera-rubin-nvl72","category":null,"updatedAt":"2026-10-05T05:05:08.969Z","fields":{"vendor":"Nvidia","architecture":"Vera Rubin","memory":"20.7 TB HBM4 (72 GPUs)","memoryGb":20700,"memoryBandwidthTbs":1580,"fp8Tflops":1260000,"fp4Pflops":2520,"precisionNote":"Rack totals. FP8/FP6 1,260 PFLOPS and NVFP4 2,520 PFLOPS are training figures marked dense; NVFP4 inference is listed as 3,600 PFLOPS without a sparse/dense label.","tdpW":null,"formFactor":"Rack-scale system: 72 GPUs, 36 Vera CPUs","launch":"Launched January 5, 2026 (CES); full production announced May 31, 2026","status":"In production","datasheet":"https://www.nvidia.com/en-us/data-center/vera-rubin-nvl72/","checked":"2026-10-05"}}]}