{"version":"2026-09-06.2","models":[{"id":"qwen3-5-0-8b","name":"Qwen3.5 0.8B","maker":"Qwen","repo":"Qwen/Qwen3.5-0.8B","parametersB":0.873438784,"checkpointParameters":873438784,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":24,"kvHeads":2,"headDim":256,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Publisher configuration default.","category":"General","architecture":"Dense","summary":"Small text and image model for basic tasks on limited memory.","license":"Apache 2.0","source":"https://huggingface.co/Qwen/Qwen3.5-0.8B","configSource":"https://huggingface.co/Qwen/Qwen3.5-0.8B/blob/2fc06364715b967f1860aea9cf38778875588b17/config.json","metadataSource":"https://huggingface.co/api/models/Qwen/Qwen3.5-0.8B","revision":"2fc06364715b967f1860aea9cf38778875588b17","reviewedAt":"2026-09-06","modalities":["text","image","video"],"cache":{"kind":"hybrid","attention":[{"layers":6,"kvHeads":2,"keyDim":256,"valueDim":256}],"notes":"Gated DeltaNet states use FP32; convolution states use the model dtype. Full-attention layers alone grow with text length. MTP speculation is excluded from cache.","source":"https://huggingface.co/Qwen/Qwen3.5-0.8B/blob/2fc06364715b967f1860aea9cf38778875588b17/config.json","recurrent":{"layers":18,"keyHeads":16,"valueHeads":16,"keyDim":128,"valueDim":128,"convKernel":4,"stateBytes":4,"convBytes":2}},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra."},{"id":"qwen3-5-2b","name":"Qwen3.5 2B","maker":"Qwen","repo":"Qwen/Qwen3.5-2B","parametersB":2.274069824,"checkpointParameters":2274069824,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":24,"kvHeads":2,"headDim":256,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Publisher configuration default.","category":"General","architecture":"Dense","summary":"Compact text and image model for local chat and simple tools.","license":"Apache 2.0","source":"https://huggingface.co/Qwen/Qwen3.5-2B","configSource":"https://huggingface.co/Qwen/Qwen3.5-2B/blob/15852e8c16360a2fea060d615a32b45270f8a8fc/config.json","metadataSource":"https://huggingface.co/api/models/Qwen/Qwen3.5-2B","revision":"15852e8c16360a2fea060d615a32b45270f8a8fc","reviewedAt":"2026-09-06","modalities":["text","image","video"],"cache":{"kind":"hybrid","attention":[{"layers":6,"kvHeads":2,"keyDim":256,"valueDim":256}],"notes":"Gated DeltaNet states use FP32; convolution states use the model dtype. Full-attention layers alone grow with text length. MTP speculation is excluded from cache.","source":"https://huggingface.co/Qwen/Qwen3.5-2B/blob/15852e8c16360a2fea060d615a32b45270f8a8fc/config.json","recurrent":{"layers":18,"keyHeads":16,"valueHeads":16,"keyDim":128,"valueDim":128,"convKernel":4,"stateBytes":4,"convBytes":2}},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra."},{"id":"smollm3-3b","name":"SmolLM3 3B","maker":"Hugging Face","repo":"HuggingFaceTB/SmolLM3-3B","parametersB":3.075098624,"checkpointParameters":3075098624,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":36,"kvHeads":4,"headDim":128,"contextLimit":65536,"configuredContextLimit":65536,"contextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","category":"General","architecture":"Dense","summary":"Small text model with optional reasoning and a fully published training recipe.","license":"Apache 2.0","source":"https://huggingface.co/HuggingFaceTB/SmolLM3-3B","configSource":"https://huggingface.co/HuggingFaceTB/SmolLM3-3B/blob/a07cc9a04f16550a088caea529712d1d335b0ac1/config.json","metadataSource":"https://huggingface.co/api/models/HuggingFaceTB/SmolLM3-3B","revision":"a07cc9a04f16550a088caea529712d1d335b0ac1","reviewedAt":"2026-09-06","modalities":["text"],"extendedContextLimit":131072,"extendedContextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","cache":{"kind":"standard","attention":[{"layers":36,"kvHeads":4,"keyDim":128,"valueDim":128}],"notes":"Text decoding cache; image/audio processing buffers and runtime workspaces are additional.","source":"https://huggingface.co/HuggingFaceTB/SmolLM3-3B/blob/a07cc9a04f16550a088caea529712d1d335b0ac1/config.json"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support.","multimodal":false,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support."},{"id":"granite-4-2-3b","name":"Granite 4.2 3B","maker":"IBM","repo":"ibm-granite/granite-4.2-3b","parametersB":3.6597376,"checkpointParameters":3659737600,"parameterBasis":"Hugging Face normalized whole-checkpoint tensor count.","layers":40,"kvHeads":8,"headDim":64,"contextLimit":131072,"configuredContextLimit":131072,"category":"Reasoning","architecture":"Dense","summary":"Small current Granite with optional reasoning and tool use.","license":"Apache 2.0","source":"https://huggingface.co/ibm-granite/granite-4.2-3b","configSource":"https://huggingface.co/ibm-granite/granite-4.2-3b/blob/e459acceac81e5fe67c07d9cfc72329a332e7eb1/config.json","metadataSource":"https://huggingface.co/api/models/ibm-granite/granite-4.2-3b","revision":"e459acceac81e5fe67c07d9cfc72329a332e7eb1","reviewedAt":"2026-09-06","modalities":["text"],"cache":{"kind":"standard","attention":[{"layers":40,"kvHeads":8,"keyDim":64,"valueDim":64}],"source":"https://huggingface.co/ibm-granite/granite-4.2-3b/blob/e459acceac81e5fe67c07d9cfc72329a332e7eb1/config.json","notes":"Text decoding cache; runtime workspaces and optional speculative caches are additional."},"quantizationCaveat":"Precision estimates need quantization scales and some higher-precision tensors. Backend support and actual file sizes vary. Whole-checkpoint counts can include auxiliary weights.","extendedContextLimit":524288,"extendedContextNote":"Publisher advertises 512K extension from a 128K native context; requires validated long-context runtime settings.","multimodal":false,"weightNote":"Precision estimates need quantization scales and some higher-precision tensors. Backend support and actual file sizes vary. Whole-checkpoint counts can include auxiliary weights.","contextNote":"Publisher advertises 512K extension from a 128K native context; requires validated long-context runtime settings."},{"id":"ministral-3-3b","name":"Ministral 3 3B","maker":"Mistral AI","repo":"mistralai/Ministral-3-3B-Instruct-2512","parametersB":3.849090048,"checkpointParameters":3849090048,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":26,"kvHeads":8,"headDim":128,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Publisher configuration default.","category":"General","architecture":"Dense","summary":"Small text and image model with instruction following and tool use.","license":"Apache 2.0","source":"https://huggingface.co/mistralai/Ministral-3-3B-Instruct-2512","configSource":"https://huggingface.co/mistralai/Ministral-3-3B-Instruct-2512/blob/b35d4dfe56c142746f54dbd64f579faab2744308/config.json","metadataSource":"https://huggingface.co/api/models/mistralai/Ministral-3-3B-Instruct-2512","revision":"b35d4dfe56c142746f54dbd64f579faab2744308","reviewedAt":"2026-09-06","modalities":["text","image"],"cache":{"kind":"standard","attention":[{"layers":26,"kvHeads":8,"keyDim":128,"valueDim":128}],"notes":"Text decoding cache; image/audio processing buffers and runtime workspaces are additional.","source":"https://huggingface.co/mistralai/Ministral-3-3B-Instruct-2512/blob/b35d4dfe56c142746f54dbd64f579faab2744308/config.json"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included."},{"id":"qwen3-5-4b","name":"Qwen3.5 4B","maker":"Qwen","repo":"Qwen/Qwen3.5-4B","parametersB":4.659865088,"checkpointParameters":4659865088,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":32,"kvHeads":4,"headDim":256,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","category":"General","architecture":"Dense","summary":"Small text and image model with reasoning and tool use.","license":"Apache 2.0","source":"https://huggingface.co/Qwen/Qwen3.5-4B","configSource":"https://huggingface.co/Qwen/Qwen3.5-4B/blob/851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a/config.json","metadataSource":"https://huggingface.co/api/models/Qwen/Qwen3.5-4B","revision":"851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a","reviewedAt":"2026-09-06","modalities":["text","image","video"],"extendedContextLimit":1010000,"extendedContextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","cache":{"kind":"hybrid","attention":[{"layers":8,"kvHeads":4,"keyDim":256,"valueDim":256}],"notes":"Gated DeltaNet states use FP32; convolution states use the model dtype. Full-attention layers alone grow with text length. MTP speculation is excluded from cache.","source":"https://huggingface.co/Qwen/Qwen3.5-4B/blob/851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a/config.json","recurrent":{"layers":24,"keyHeads":16,"valueHeads":32,"keyDim":128,"valueDim":128,"convKernel":4,"stateBytes":4,"convBytes":2}},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra."},{"id":"gemma-4-e2b","name":"Gemma 4 E2B","maker":"Google","repo":"google/gemma-4-E2B-it","parametersB":5.123178051,"checkpointParameters":5123178051,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":35,"kvHeads":1,"headDim":256,"contextLimit":131072,"configuredContextLimit":131072,"contextNote":"Publisher configuration default.","category":"General","architecture":"Dense","summary":"Compact text, image and audio model; its full weights exceed 5B parameters.","license":"Apache 2.0","source":"https://huggingface.co/google/gemma-4-E2B-it","configSource":"https://huggingface.co/google/gemma-4-E2B-it/blob/3e22461f65e89153144f8adb70e3b8c2cc9845a7/config.json","metadataSource":"https://huggingface.co/api/models/google/gemma-4-E2B-it","revision":"3e22461f65e89153144f8adb70e3b8c2cc9845a7","reviewedAt":"2026-09-06","modalities":["text","image","video","audio"],"cache":{"kind":"sliding","attention":[{"layers":3,"kvHeads":1,"keyDim":512,"valueDim":512},{"layers":12,"kvHeads":1,"keyDim":256,"valueDim":256,"window":512}],"notes":"Only first 15 layers allocate unique caches; later shared layers reuse earlier caches of the same type. Global/sliding dimensions differ. attention_k_eq_v shares a projection, NOT the stored K and V caches. Assumes runtime sliding-cache eviction.","source":"https://huggingface.co/google/gemma-4-E2B-it/blob/3e22461f65e89153144f8adb70e3b8c2cc9845a7/config.json","sharedLayers":20},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included."},{"id":"olmo-3-7b-instruct","name":"OLMo 3 7B Instruct","maker":"Ai2","repo":"allenai/Olmo-3-7B-Instruct","parametersB":7.298011136,"checkpointParameters":7298011136,"parameterBasis":"Hugging Face normalized whole-checkpoint tensor count.","layers":32,"kvHeads":32,"headDim":128,"contextLimit":65536,"configuredContextLimit":65536,"category":"General","architecture":"Dense","summary":"Fully open small chat model with published training data and recipes.","license":"Apache 2.0","source":"https://huggingface.co/allenai/Olmo-3-7B-Instruct","configSource":"https://huggingface.co/allenai/Olmo-3-7B-Instruct/blob/6e5971d9eba42665f5bd5a0fcf047f299ce1dccc/config.json","metadataSource":"https://huggingface.co/api/models/allenai/Olmo-3-7B-Instruct","revision":"6e5971d9eba42665f5bd5a0fcf047f299ce1dccc","reviewedAt":"2026-09-06","modalities":["text"],"cache":{"kind":"sliding","attention":[{"layers":8,"kvHeads":32,"keyDim":128,"valueDim":128},{"layers":24,"kvHeads":32,"keyDim":128,"valueDim":128,"window":4096}],"source":"https://huggingface.co/allenai/Olmo-3-7B-Instruct/blob/6e5971d9eba42665f5bd5a0fcf047f299ce1dccc/config.json","notes":"Per-layer sliding/full attention schedule from publisher config. Cache savings require runtime eviction of sliding-window entries."},"quantizationCaveat":"Precision estimates need quantization scales and some higher-precision tensors. Backend support and actual file sizes vary. Whole-checkpoint counts can include auxiliary weights.","multimodal":false,"weightNote":"Precision estimates need quantization scales and some higher-precision tensors. Backend support and actual file sizes vary. Whole-checkpoint counts can include auxiliary weights."},{"id":"gemma-4-e4b","name":"Gemma 4 E4B","maker":"Google","repo":"google/gemma-4-E4B-it","parametersB":7.99615649,"checkpointParameters":7996156490,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":42,"kvHeads":2,"headDim":256,"contextLimit":131072,"configuredContextLimit":131072,"contextNote":"Publisher configuration default.","category":"General","architecture":"Dense","summary":"Text, image and audio model with roughly 8B total stored parameters.","license":"Apache 2.0","source":"https://huggingface.co/google/gemma-4-E4B-it","configSource":"https://huggingface.co/google/gemma-4-E4B-it/blob/ee0ef6023621cff504d758262d4e04895a5af4a2/config.json","metadataSource":"https://huggingface.co/api/models/google/gemma-4-E4B-it","revision":"ee0ef6023621cff504d758262d4e04895a5af4a2","reviewedAt":"2026-09-06","modalities":["text","image","video","audio"],"cache":{"kind":"sliding","attention":[{"layers":4,"kvHeads":2,"keyDim":512,"valueDim":512},{"layers":20,"kvHeads":2,"keyDim":256,"valueDim":256,"window":512}],"notes":"Only first 24 layers allocate unique caches; later shared layers reuse earlier caches of the same type. Global/sliding dimensions differ. attention_k_eq_v shares a projection, NOT the stored K and V caches. Assumes runtime sliding-cache eviction.","source":"https://huggingface.co/google/gemma-4-E4B-it/blob/ee0ef6023621cff504d758262d4e04895a5af4a2/config.json","sharedLayers":18},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included."},{"id":"granite-4-2-8b","name":"Granite 4.2 8B","maker":"IBM","repo":"ibm-granite/granite-4.2-8b","parametersB":8.79159296,"checkpointParameters":8791592960,"parameterBasis":"Hugging Face normalized whole-checkpoint tensor count.","layers":40,"kvHeads":8,"headDim":128,"contextLimit":131072,"configuredContextLimit":131072,"category":"Reasoning","architecture":"Dense","summary":"Current dense Granite for reasoning, code and multilingual tools.","license":"Apache 2.0","source":"https://huggingface.co/ibm-granite/granite-4.2-8b","configSource":"https://huggingface.co/ibm-granite/granite-4.2-8b/blob/f8de16cdcdbc6c779ca517604e050d82cc119e44/config.json","metadataSource":"https://huggingface.co/api/models/ibm-granite/granite-4.2-8b","revision":"f8de16cdcdbc6c779ca517604e050d82cc119e44","reviewedAt":"2026-09-06","modalities":["text"],"cache":{"kind":"standard","attention":[{"layers":40,"kvHeads":8,"keyDim":128,"valueDim":128}],"source":"https://huggingface.co/ibm-granite/granite-4.2-8b/blob/f8de16cdcdbc6c779ca517604e050d82cc119e44/config.json","notes":"Text decoding cache; runtime workspaces and optional speculative caches are additional."},"quantizationCaveat":"Precision estimates need quantization scales and some higher-precision tensors. Backend support and actual file sizes vary. Whole-checkpoint counts can include auxiliary weights.","extendedContextLimit":524288,"extendedContextNote":"Publisher advertises 512K extension from a 128K native context; requires validated long-context runtime settings.","multimodal":false,"weightNote":"Precision estimates need quantization scales and some higher-precision tensors. Backend support and actual file sizes vary. Whole-checkpoint counts can include auxiliary weights.","contextNote":"Publisher advertises 512K extension from a 128K native context; requires validated long-context runtime settings."},{"id":"nemotron-nano-9b-v2","name":"Nemotron Nano 9B v2","maker":"NVIDIA","repo":"nvidia/NVIDIA-Nemotron-Nano-9B-v2","parametersB":8.888227328,"checkpointParameters":8888227328,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":56,"kvHeads":8,"headDim":128,"contextLimit":131072,"configuredContextLimit":131072,"contextNote":"Publisher configuration default.","category":"Reasoning","architecture":"Dense","summary":"Small dense hybrid model for reasoning and non-reasoning tasks.","license":"NVIDIA Open Model License","source":"https://huggingface.co/nvidia/NVIDIA-Nemotron-Nano-9B-v2","configSource":"https://huggingface.co/nvidia/NVIDIA-Nemotron-Nano-9B-v2/blob/6533e8de2c68e4536bf7c411d7a3ce5734111476/config.json","metadataSource":"https://huggingface.co/api/models/nvidia/NVIDIA-Nemotron-Nano-9B-v2","revision":"6533e8de2c68e4536bf7c411d7a3ce5734111476","reviewedAt":"2026-09-06","modalities":["text"],"cache":{"kind":"hybrid","attention":[{"layers":4,"kvHeads":8,"keyDim":128,"valueDim":128}],"recurrent":{"layers":27,"keyHeads":8,"valueHeads":128,"keyDim":128,"valueDim":80,"convKernel":4,"stateBytes":4,"convBytes":2},"notes":"Mamba-2 recurrent state is heads × head dimension × state size in FP32. Convolution channels = heads × head dimension + 2 × groups × state size in BF16. MTP speculative caches excluded.","source":"https://huggingface.co/nvidia/NVIDIA-Nemotron-Nano-9B-v2/blob/6533e8de2c68e4536bf7c411d7a3ce5734111476/config.json","recurrentType":"mamba2","blockCounts":{"M":27,"-":25,"*":4}},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support.","multimodal":false,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support."},{"id":"ministral-3-8b","name":"Ministral 3 8B","maker":"Mistral AI","repo":"mistralai/Ministral-3-8B-Instruct-2512","parametersB":8.918026716,"checkpointParameters":8918026716,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":34,"kvHeads":8,"headDim":128,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Publisher configuration default.","category":"General","architecture":"Dense","summary":"Text and image model with a 256K supported context.","license":"Apache 2.0","source":"https://huggingface.co/mistralai/Ministral-3-8B-Instruct-2512","configSource":"https://huggingface.co/mistralai/Ministral-3-8B-Instruct-2512/blob/5b26027e7b19eeb4b7352e1fed3926375dd2cb4d/config.json","metadataSource":"https://huggingface.co/api/models/mistralai/Ministral-3-8B-Instruct-2512","revision":"5b26027e7b19eeb4b7352e1fed3926375dd2cb4d","reviewedAt":"2026-09-06","modalities":["text","image"],"cache":{"kind":"standard","attention":[{"layers":34,"kvHeads":8,"keyDim":128,"valueDim":128}],"notes":"Text decoding cache; image/audio processing buffers and runtime workspaces are additional.","source":"https://huggingface.co/mistralai/Ministral-3-8B-Instruct-2512/blob/5b26027e7b19eeb4b7352e1fed3926375dd2cb4d/config.json"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included."},{"id":"qwen3-5-9b","name":"Qwen3.5 9B","maker":"Qwen","repo":"Qwen/Qwen3.5-9B","parametersB":9.653104368,"checkpointParameters":9653104368,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":32,"kvHeads":4,"headDim":256,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","category":"General","architecture":"Dense","summary":"Text, image and coding model with optional reasoning.","license":"Apache 2.0","source":"https://huggingface.co/Qwen/Qwen3.5-9B","configSource":"https://huggingface.co/Qwen/Qwen3.5-9B/blob/c202236235762e1c871ad0ccb60c8ee5ba337b9a/config.json","metadataSource":"https://huggingface.co/api/models/Qwen/Qwen3.5-9B","revision":"c202236235762e1c871ad0ccb60c8ee5ba337b9a","reviewedAt":"2026-09-06","modalities":["text","image","video"],"extendedContextLimit":1010000,"extendedContextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","cache":{"kind":"hybrid","attention":[{"layers":8,"kvHeads":4,"keyDim":256,"valueDim":256}],"notes":"Gated DeltaNet states use FP32; convolution states use the model dtype. Full-attention layers alone grow with text length. MTP speculation is excluded from cache.","source":"https://huggingface.co/Qwen/Qwen3.5-9B/blob/c202236235762e1c871ad0ccb60c8ee5ba337b9a/config.json","recurrent":{"layers":24,"keyHeads":16,"valueHeads":32,"keyDim":128,"valueDim":128,"convKernel":4,"stateBytes":4,"convBytes":2}},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra."},{"id":"glm-4-6v-flash-9b","name":"GLM-4.6V Flash 9B","maker":"Z.ai","repo":"zai-org/GLM-4.6V-Flash","parametersB":10.292777472,"checkpointParameters":10292777472,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":40,"kvHeads":2,"headDim":128,"contextLimit":131072,"configuredContextLimit":131072,"contextNote":"Publisher configuration default.","category":"General","architecture":"Dense","summary":"Small vision-language model with native tool use.","license":"MIT","source":"https://huggingface.co/zai-org/GLM-4.6V-Flash","configSource":"https://huggingface.co/zai-org/GLM-4.6V-Flash/blob/411bb4d77144a3f03accbf4b780f5acb8b7cde4e/config.json","metadataSource":"https://huggingface.co/api/models/zai-org/GLM-4.6V-Flash","revision":"411bb4d77144a3f03accbf4b780f5acb8b7cde4e","reviewedAt":"2026-09-06","modalities":["text","image"],"cache":{"kind":"standard","attention":[{"layers":40,"kvHeads":2,"keyDim":128,"valueDim":128}],"notes":"Text decoding cache; image/audio processing buffers and runtime workspaces are additional.","source":"https://huggingface.co/zai-org/GLM-4.6V-Flash/blob/411bb4d77144a3f03accbf4b780f5acb8b7cde4e/config.json"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included."},{"id":"gemma-4-12b","name":"Gemma 4 12B","maker":"Google","repo":"google/gemma-4-12B-it","parametersB":11.959730224,"checkpointParameters":11959730224,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":48,"kvHeads":8,"headDim":256,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Publisher configuration default.","category":"General","architecture":"Dense","summary":"Unified text, image and audio model with a 256K context.","license":"Apache 2.0","source":"https://huggingface.co/google/gemma-4-12B-it","configSource":"https://huggingface.co/google/gemma-4-12B-it/blob/707f0a3b8a3c7ad586ed01e27eafbad8a27dd0f7/config.json","metadataSource":"https://huggingface.co/api/models/google/gemma-4-12B-it","revision":"707f0a3b8a3c7ad586ed01e27eafbad8a27dd0f7","reviewedAt":"2026-09-06","modalities":["text","image","video","audio"],"cache":{"kind":"sliding","attention":[{"layers":8,"kvHeads":1,"keyDim":512,"valueDim":512},{"layers":40,"kvHeads":8,"keyDim":256,"valueDim":256,"window":1024}],"notes":"Only first 48 layers allocate unique caches; later shared layers reuse earlier caches of the same type. Global/sliding dimensions differ. attention_k_eq_v shares a projection, NOT the stored K and V caches. Assumes runtime sliding-cache eviction.","source":"https://huggingface.co/google/gemma-4-12B-it/blob/707f0a3b8a3c7ad586ed01e27eafbad8a27dd0f7/config.json","sharedLayers":0},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included."},{"id":"ministral-3-14b","name":"Ministral 3 14B","maker":"Mistral AI","repo":"mistralai/Ministral-3-14B-Instruct-2512","parametersB":13.94503224,"checkpointParameters":13945032240,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":40,"kvHeads":8,"headDim":128,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Publisher configuration default.","category":"General","architecture":"Dense","summary":"Largest small dense Ministral for text and image tasks.","license":"Apache 2.0","source":"https://huggingface.co/mistralai/Ministral-3-14B-Instruct-2512","configSource":"https://huggingface.co/mistralai/Ministral-3-14B-Instruct-2512/blob/29439f81c2be264d8d393273f99e7db9c0961120/config.json","metadataSource":"https://huggingface.co/api/models/mistralai/Ministral-3-14B-Instruct-2512","revision":"29439f81c2be264d8d393273f99e7db9c0961120","reviewedAt":"2026-09-06","modalities":["text","image"],"cache":{"kind":"standard","attention":[{"layers":40,"kvHeads":8,"keyDim":128,"valueDim":128}],"notes":"Text decoding cache; image/audio processing buffers and runtime workspaces are additional.","source":"https://huggingface.co/mistralai/Ministral-3-14B-Instruct-2512/blob/29439f81c2be264d8d393273f99e7db9c0961120/config.json"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included."},{"id":"phi-4-reasoning-plus-14b","name":"Phi-4 Reasoning Plus 14B","maker":"Microsoft","repo":"microsoft/Phi-4-reasoning-plus","parametersB":14.6595072,"checkpointParameters":14659507200,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":40,"kvHeads":10,"headDim":128,"contextLimit":32768,"configuredContextLimit":32768,"contextNote":"Publisher configuration default.","category":"Reasoning","architecture":"Dense","summary":"Dense reasoning specialist for mathematics, science and coding.","license":"MIT","source":"https://huggingface.co/microsoft/Phi-4-reasoning-plus","configSource":"https://huggingface.co/microsoft/Phi-4-reasoning-plus/blob/69baf8528e1bcf05f475034d9e5dd32875ed125f/config.json","metadataSource":"https://huggingface.co/api/models/microsoft/Phi-4-reasoning-plus","revision":"69baf8528e1bcf05f475034d9e5dd32875ed125f","reviewedAt":"2026-09-06","modalities":["text"],"cache":{"kind":"standard","attention":[{"layers":40,"kvHeads":10,"keyDim":128,"valueDim":128}],"notes":"Text decoding cache; image/audio processing buffers and runtime workspaces are additional.","source":"https://huggingface.co/microsoft/Phi-4-reasoning-plus/blob/69baf8528e1bcf05f475034d9e5dd32875ed125f/config.json"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support.","multimodal":false,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support."},{"id":"gpt-oss-20b","name":"gpt-oss 20B","maker":"OpenAI","repo":"openai/gpt-oss-20b","parametersB":20.914757184,"checkpointParameters":20914757184,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":24,"kvHeads":8,"headDim":64,"contextLimit":131072,"configuredContextLimit":131072,"contextNote":"Publisher configuration default.","category":"Reasoning","architecture":"Mixture of experts","summary":"Small OpenAI reasoning model supplied with native MXFP4 expert weights.","license":"Apache 2.0","source":"https://huggingface.co/openai/gpt-oss-20b","configSource":"https://huggingface.co/openai/gpt-oss-20b/blob/6cee5e81ee83917806bbde320786a8fb61efebee/config.json","metadataSource":"https://huggingface.co/api/models/openai/gpt-oss-20b","revision":"6cee5e81ee83917806bbde320786a8fb61efebee","reviewedAt":"2026-09-06","modalities":["text"],"activeParametersB":3.6,"cache":{"kind":"sliding","attention":[{"layers":12,"kvHeads":8,"keyDim":64,"valueDim":64},{"layers":12,"kvHeads":8,"keyDim":64,"valueDim":64,"window":128}],"notes":"Alternating full attention and 128-token sliding attention; savings require runtime sliding-cache eviction.","source":"https://huggingface.co/openai/gpt-oss-20b/blob/6cee5e81ee83917806bbde320786a8fb61efebee/config.json"},"quantizationCaveat":"Native MXFP4 quantizes expert matrices, not every tensor. Use the supplied native checkpoint bytes for native mode; generic 4-bit is a different estimate. Unpacking MXFP4 to BF16 changes residency.","nativeWeights":{"format":"MXFP4 experts + BF16 other weights","bytes":13761264768,"source":"https://huggingface.co/openai/gpt-oss-20b/blob/6cee5e81ee83917806bbde320786a8fb61efebee/model.safetensors.index.json"},"multimodal":false,"weightNote":"Native MXFP4 quantizes expert matrices, not every tensor. Use the supplied native checkpoint bytes for native mode; generic 4-bit is a different estimate. Unpacking MXFP4 to BF16 changes residency.","nativeWeightsGiB":12.816176533699036,"nativeWeightLabel":"Native MXFP4"},{"id":"ernie-4-5-21b-a3b-thinking","name":"ERNIE 4.5 21B-A3B Thinking","maker":"Baidu","repo":"baidu/ERNIE-4.5-21B-A3B-Thinking","parametersB":21.825437888,"checkpointParameters":21825437888,"parameterBasis":"Hugging Face normalized whole-checkpoint tensor count.","layers":28,"kvHeads":4,"headDim":128,"contextLimit":131072,"configuredContextLimit":131072,"category":"Reasoning","architecture":"Mixture of experts","summary":"Compact sparse ERNIE with reasoning and a 128K context.","license":"Apache 2.0","source":"https://huggingface.co/baidu/ERNIE-4.5-21B-A3B-Thinking","configSource":"https://huggingface.co/baidu/ERNIE-4.5-21B-A3B-Thinking/blob/4341bb42644d5422859509fa25d41544c57181f8/config.json","metadataSource":"https://huggingface.co/api/models/baidu/ERNIE-4.5-21B-A3B-Thinking","revision":"4341bb42644d5422859509fa25d41544c57181f8","reviewedAt":"2026-09-06","modalities":["text"],"cache":{"kind":"standard","attention":[{"layers":28,"kvHeads":4,"keyDim":128,"valueDim":128}],"source":"https://huggingface.co/baidu/ERNIE-4.5-21B-A3B-Thinking/blob/4341bb42644d5422859509fa25d41544c57181f8/config.json","notes":"Text decoding cache; runtime workspaces and optional speculative caches are additional."},"quantizationCaveat":"Precision estimates need quantization scales and some higher-precision tensors. Backend support and actual file sizes vary. Whole-checkpoint counts can include auxiliary weights.","activeParametersB":3,"multimodal":false,"weightNote":"Precision estimates need quantization scales and some higher-precision tensors. Backend support and actual file sizes vary. Whole-checkpoint counts can include auxiliary weights."},{"id":"devstral-small-2-24b","name":"Devstral Small 2 24B","maker":"Mistral AI","repo":"mistralai/Devstral-Small-2-24B-Instruct-2512","parametersB":24.01136184,"checkpointParameters":24011361840,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":40,"kvHeads":8,"headDim":128,"contextLimit":262144,"configuredContextLimit":393216,"contextNote":"Use the model-card supported 256K limit; configuration has a larger positional allocation.","category":"Coding","architecture":"Dense","summary":"Coding agent model with image input for software development.","license":"Apache 2.0","source":"https://huggingface.co/mistralai/Devstral-Small-2-24B-Instruct-2512","configSource":"https://huggingface.co/mistralai/Devstral-Small-2-24B-Instruct-2512/blob/55c5b41e98c2dbd21b0c8afffc540dcfc9eb5128/config.json","metadataSource":"https://huggingface.co/api/models/mistralai/Devstral-Small-2-24B-Instruct-2512","revision":"55c5b41e98c2dbd21b0c8afffc540dcfc9eb5128","reviewedAt":"2026-09-06","modalities":["text","image"],"cache":{"kind":"standard","attention":[{"layers":40,"kvHeads":8,"keyDim":128,"valueDim":128}],"notes":"Text decoding cache; image/audio processing buffers and runtime workspaces are additional.","source":"https://huggingface.co/mistralai/Devstral-Small-2-24B-Instruct-2512/blob/55c5b41e98c2dbd21b0c8afffc540dcfc9eb5128/config.json"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included."},{"id":"gemma-4-26b-a4b","name":"Gemma 4 26B-A4B","maker":"Google","repo":"google/gemma-4-26B-A4B-it","parametersB":25.805936206,"checkpointParameters":25805936206,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":30,"kvHeads":8,"headDim":256,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Publisher configuration default.","category":"General","architecture":"Mixture of experts","summary":"Sparse text and image model with about 3.8B active parameters.","license":"Apache 2.0","source":"https://huggingface.co/google/gemma-4-26B-A4B-it","configSource":"https://huggingface.co/google/gemma-4-26B-A4B-it/blob/4d7ae4984b7db7de8f8457170b3f1a419ee76d52/config.json","metadataSource":"https://huggingface.co/api/models/google/gemma-4-26B-A4B-it","revision":"4d7ae4984b7db7de8f8457170b3f1a419ee76d52","reviewedAt":"2026-09-06","modalities":["text","image","video"],"activeParametersB":3.8,"cache":{"kind":"sliding","attention":[{"layers":5,"kvHeads":2,"keyDim":512,"valueDim":512},{"layers":25,"kvHeads":8,"keyDim":256,"valueDim":256,"window":1024}],"notes":"Only first 30 layers allocate unique caches; later shared layers reuse earlier caches of the same type. Global/sliding dimensions differ. attention_k_eq_v shares a projection, NOT the stored K and V caches. Assumes runtime sliding-cache eviction.","source":"https://huggingface.co/google/gemma-4-26B-A4B-it/blob/4d7ae4984b7db7de8f8457170b3f1a419ee76d52/config.json","sharedLayers":0},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included."},{"id":"qwen3-8-27b","name":"Qwen3.8 27B","maker":"Qwen","repo":"Qwen/Qwen3.8-27B","parametersB":27.781427952,"checkpointParameters":27781427952,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":64,"kvHeads":4,"headDim":256,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","category":"General","architecture":"Dense","summary":"Current dense Qwen for coding, reasoning and image understanding.","license":"Apache 2.0","source":"https://huggingface.co/Qwen/Qwen3.8-27B","configSource":"https://huggingface.co/Qwen/Qwen3.8-27B/blob/1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0/config.json","metadataSource":"https://huggingface.co/api/models/Qwen/Qwen3.8-27B","revision":"1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0","reviewedAt":"2026-09-06","modalities":["text","image","video"],"extendedContextLimit":1000000,"extendedContextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","cache":{"kind":"hybrid","attention":[{"layers":16,"kvHeads":4,"keyDim":256,"valueDim":256}],"notes":"Gated DeltaNet states use FP32; convolution states use the model dtype. Full-attention layers alone grow with text length. MTP speculation is excluded from cache.","source":"https://huggingface.co/Qwen/Qwen3.8-27B/blob/1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0/config.json","recurrent":{"layers":48,"keyHeads":16,"valueHeads":48,"keyDim":128,"valueDim":128,"convKernel":4,"stateBytes":4,"convBytes":2}},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra."},{"id":"granite-4-2-30b","name":"Granite 4.2 30B","maker":"IBM","repo":"ibm-granite/granite-4.2-30b","parametersB":29.276770304,"checkpointParameters":29276770304,"parameterBasis":"Hugging Face normalized whole-checkpoint tensor count.","layers":64,"kvHeads":8,"headDim":128,"contextLimit":131072,"configuredContextLimit":131072,"category":"Reasoning","architecture":"Dense","summary":"Largest current dense Granite for reasoning and agent tasks.","license":"Apache 2.0","source":"https://huggingface.co/ibm-granite/granite-4.2-30b","configSource":"https://huggingface.co/ibm-granite/granite-4.2-30b/blob/9e668ce1c538387ef24d3644e9b0606647762636/config.json","metadataSource":"https://huggingface.co/api/models/ibm-granite/granite-4.2-30b","revision":"9e668ce1c538387ef24d3644e9b0606647762636","reviewedAt":"2026-09-06","modalities":["text"],"cache":{"kind":"standard","attention":[{"layers":64,"kvHeads":8,"keyDim":128,"valueDim":128}],"source":"https://huggingface.co/ibm-granite/granite-4.2-30b/blob/9e668ce1c538387ef24d3644e9b0606647762636/config.json","notes":"Text decoding cache; runtime workspaces and optional speculative caches are additional."},"quantizationCaveat":"Precision estimates need quantization scales and some higher-precision tensors. Backend support and actual file sizes vary. Whole-checkpoint counts can include auxiliary weights.","extendedContextLimit":524288,"extendedContextNote":"Publisher advertises 512K extension from a 128K native context; requires validated long-context runtime settings.","multimodal":false,"weightNote":"Precision estimates need quantization scales and some higher-precision tensors. Backend support and actual file sizes vary. Whole-checkpoint counts can include auxiliary weights.","contextNote":"Publisher advertises 512K extension from a 128K native context; requires validated long-context runtime settings."},{"id":"qwen3-coder-30b-a3b","name":"Qwen3 Coder 30B-A3B","maker":"Qwen","repo":"Qwen/Qwen3-Coder-30B-A3B-Instruct","parametersB":30.532122624,"checkpointParameters":30532122624,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":48,"kvHeads":4,"headDim":128,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","category":"Coding","architecture":"Mixture of experts","summary":"Dedicated coding model with conventional attention and a 256K native context.","license":"Apache 2.0","source":"https://huggingface.co/Qwen/Qwen3-Coder-30B-A3B-Instruct","configSource":"https://huggingface.co/Qwen/Qwen3-Coder-30B-A3B-Instruct/blob/b2cff646eb4bb1d68355c01b18ae02e7cf42d120/config.json","metadataSource":"https://huggingface.co/api/models/Qwen/Qwen3-Coder-30B-A3B-Instruct","revision":"b2cff646eb4bb1d68355c01b18ae02e7cf42d120","reviewedAt":"2026-09-06","modalities":["text"],"activeParametersB":3.3,"extendedContextLimit":1048576,"extendedContextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","cache":{"kind":"standard","attention":[{"layers":48,"kvHeads":4,"keyDim":128,"valueDim":128}],"notes":"Text decoding cache; image/audio processing buffers and runtime workspaces are additional.","source":"https://huggingface.co/Qwen/Qwen3-Coder-30B-A3B-Instruct/blob/b2cff646eb4bb1d68355c01b18ae02e7cf42d120/config.json"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support.","multimodal":false,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support."},{"id":"glm-4-7-flash-30b-a3b","name":"GLM-4.7 Flash 30B-A3B","maker":"Z.ai","repo":"zai-org/GLM-4.7-Flash","parametersB":31.221488576,"checkpointParameters":31221488576,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":47,"kvHeads":20,"headDim":256,"contextLimit":202752,"configuredContextLimit":202752,"contextNote":"Publisher configuration default.","category":"Reasoning","architecture":"Mixture of experts","summary":"Compact sparse GLM for reasoning, coding and tool use.","license":"MIT","source":"https://huggingface.co/zai-org/GLM-4.7-Flash","configSource":"https://huggingface.co/zai-org/GLM-4.7-Flash/blob/7dd20894a642a0aa287e9827cb1a1f7f91386b67/config.json","metadataSource":"https://huggingface.co/api/models/zai-org/GLM-4.7-Flash","revision":"7dd20894a642a0aa287e9827cb1a1f7f91386b67","reviewedAt":"2026-09-06","modalities":["text"],"activeParametersB":3,"cache":{"kind":"mla","attention":[],"latent":{"layers":47,"rank":512,"ropeDim":64},"notes":"Compressed MLA cache assumes a runtime that stores the shared latent and rotary key once per token. Expanded attention backends use more memory.","source":"https://huggingface.co/zai-org/GLM-4.7-Flash/blob/7dd20894a642a0aa287e9827cb1a1f7f91386b67/config.json"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":false,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra."},{"id":"gemma-4-31b","name":"Gemma 4 31B","maker":"Google","repo":"google/gemma-4-31B-it","parametersB":31.273088876,"checkpointParameters":31273088876,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":60,"kvHeads":16,"headDim":256,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Publisher configuration default.","category":"General","architecture":"Dense","summary":"Largest dense Gemma 4 for text and image tasks.","license":"Apache 2.0","source":"https://huggingface.co/google/gemma-4-31B-it","configSource":"https://huggingface.co/google/gemma-4-31B-it/blob/842da3794eaa0b77d5f08bae87a17459d91ff475/config.json","metadataSource":"https://huggingface.co/api/models/google/gemma-4-31B-it","revision":"842da3794eaa0b77d5f08bae87a17459d91ff475","reviewedAt":"2026-09-06","modalities":["text","image","video"],"cache":{"kind":"sliding","attention":[{"layers":10,"kvHeads":4,"keyDim":512,"valueDim":512},{"layers":50,"kvHeads":16,"keyDim":256,"valueDim":256,"window":1024}],"notes":"Only first 60 layers allocate unique caches; later shared layers reuse earlier caches of the same type. Global/sliding dimensions differ. attention_k_eq_v shares a projection, NOT the stored K and V caches. Assumes runtime sliding-cache eviction.","source":"https://huggingface.co/google/gemma-4-31B-it/blob/842da3794eaa0b77d5f08bae87a17459d91ff475/config.json","sharedLayers":0},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included."},{"id":"nemotron-3-nano-30b-a3b","name":"Nemotron 3 Nano 30B-A3B","maker":"NVIDIA","repo":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16","parametersB":31.577937344,"checkpointParameters":31577937344,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":52,"kvHeads":2,"headDim":128,"contextLimit":1048576,"configuredContextLimit":262144,"contextNote":"Publisher supports 1M. Default config is 256K; override the runtime maximum to use 1M.","category":"Reasoning","architecture":"Mixture of experts","summary":"Hybrid reasoning model with 1M context support and few full-attention layers.","license":"NVIDIA Nemotron Open Model License","source":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16","configSource":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16/blob/bf77c3174f68ad409e1c2aa60daeb46e32d1c606/config.json","metadataSource":"https://huggingface.co/api/models/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16","revision":"bf77c3174f68ad409e1c2aa60daeb46e32d1c606","reviewedAt":"2026-09-06","modalities":["text"],"activeParametersB":3.5,"cache":{"kind":"hybrid","attention":[{"layers":6,"kvHeads":2,"keyDim":128,"valueDim":128}],"recurrent":{"layers":23,"keyHeads":8,"valueHeads":64,"keyDim":128,"valueDim":64,"convKernel":4,"stateBytes":4,"convBytes":2},"notes":"Mamba-2 recurrent state is heads × head dimension × state size in FP32. Convolution channels = heads × head dimension + 2 × groups × state size in BF16. MTP speculative caches excluded.","source":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16/blob/bf77c3174f68ad409e1c2aa60daeb46e32d1c606/config.json","recurrentType":"mamba2","blockCounts":{"M":23,"E":23,"*":6}},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support.","multimodal":false,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support."},{"id":"olmo-3-1-32b-think","name":"OLMo 3.1 32B Think","maker":"Ai2","repo":"allenai/Olmo-3.1-32B-Think","parametersB":32.233522176,"checkpointParameters":32233522176,"parameterBasis":"Hugging Face normalized whole-checkpoint tensor count.","layers":64,"kvHeads":8,"headDim":128,"contextLimit":65536,"configuredContextLimit":65536,"category":"Reasoning","architecture":"Dense","summary":"Fully open dense reasoning model with published training data and recipes.","license":"Apache 2.0","source":"https://huggingface.co/allenai/Olmo-3.1-32B-Think","configSource":"https://huggingface.co/allenai/Olmo-3.1-32B-Think/blob/832c3f543499af8fe68b88359501de9cb7840544/config.json","metadataSource":"https://huggingface.co/api/models/allenai/Olmo-3.1-32B-Think","revision":"832c3f543499af8fe68b88359501de9cb7840544","reviewedAt":"2026-09-06","modalities":["text"],"cache":{"kind":"sliding","attention":[{"layers":16,"kvHeads":8,"keyDim":128,"valueDim":128},{"layers":48,"kvHeads":8,"keyDim":128,"valueDim":128,"window":4096}],"source":"https://huggingface.co/allenai/Olmo-3.1-32B-Think/blob/832c3f543499af8fe68b88359501de9cb7840544/config.json","notes":"Per-layer sliding/full attention schedule from publisher config. Cache savings require runtime eviction of sliding-window entries."},"quantizationCaveat":"Precision estimates need quantization scales and some higher-precision tensors. Backend support and actual file sizes vary. Whole-checkpoint counts can include auxiliary weights.","multimodal":false,"weightNote":"Precision estimates need quantization scales and some higher-precision tensors. Backend support and actual file sizes vary. Whole-checkpoint counts can include auxiliary weights."},{"id":"qwen3-6-35b-a3b","name":"Qwen3.6 35B-A3B","maker":"Qwen","repo":"Qwen/Qwen3.6-35B-A3B","parametersB":35.951822704,"checkpointParameters":35951822704,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":40,"kvHeads":2,"headDim":256,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","category":"General","architecture":"Mixture of experts","summary":"Sparse text and image model with coding and tool use.","license":"Apache 2.0","source":"https://huggingface.co/Qwen/Qwen3.6-35B-A3B","configSource":"https://huggingface.co/Qwen/Qwen3.6-35B-A3B/blob/995ad96eacd98c81ed38be0c5b274b04031597b0/config.json","metadataSource":"https://huggingface.co/api/models/Qwen/Qwen3.6-35B-A3B","revision":"995ad96eacd98c81ed38be0c5b274b04031597b0","reviewedAt":"2026-09-06","modalities":["text","image","video"],"activeParametersB":3,"extendedContextLimit":1010000,"extendedContextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","cache":{"kind":"hybrid","attention":[{"layers":10,"kvHeads":2,"keyDim":256,"valueDim":256}],"notes":"Gated DeltaNet states use FP32; convolution states use the model dtype. Full-attention layers alone grow with text length. MTP speculation is excluded from cache.","source":"https://huggingface.co/Qwen/Qwen3.6-35B-A3B/blob/995ad96eacd98c81ed38be0c5b274b04031597b0/config.json","recurrent":{"layers":30,"keyHeads":16,"valueHeads":32,"keyDim":128,"valueDim":128,"convKernel":4,"stateBytes":4,"convBytes":2}},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra."},{"id":"kimi-linear-48b-a3b","name":"Kimi Linear 48B-A3B","maker":"Moonshot AI","repo":"moonshotai/Kimi-Linear-48B-A3B-Instruct","parametersB":49.122681728,"checkpointParameters":49122681728,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":27,"kvHeads":32,"headDim":192,"contextLimit":1048576,"configuredContextLimit":1048576,"contextNote":"Publisher configuration default.","category":"General","architecture":"Mixture of experts","summary":"Smaller Kimi hybrid with recurrent attention and 1M context support.","license":"MIT","source":"https://huggingface.co/moonshotai/Kimi-Linear-48B-A3B-Instruct","configSource":"https://huggingface.co/moonshotai/Kimi-Linear-48B-A3B-Instruct/blob/e1df551a447157d4658b573f9a695d57658590e9/config.json","metadataSource":"https://huggingface.co/api/models/moonshotai/Kimi-Linear-48B-A3B-Instruct","revision":"e1df551a447157d4658b573f9a695d57658590e9","reviewedAt":"2026-09-06","modalities":["text"],"activeParametersB":3,"cache":{"kind":"hybrid","attention":[{"layers":7,"kvHeads":32,"keyDim":192,"valueDim":128}],"recurrent":{"layers":20,"keyHeads":32,"valueHeads":32,"keyDim":128,"valueDim":128,"convKernel":4},"runtime":"Moonshot Hugging Face reference with FLA, BF16 expanded attention cache","precisionLocked16Bit":true,"notes":"The reference retains expanded K/V in 7 attention layers, plus one FP32 32×128×128 recurrent state and three BF16 4096×4 convolution states in each of 20 KDA layers. NoPE still retains its 64-element shared key branch. Compressed-MLA engines use a different, smaller layout; speculative decoding and saved prefix states are excluded.","source":"https://huggingface.co/moonshotai/Kimi-Linear-48B-A3B-Instruct/blob/main/modeling_kimi.py"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support.","multimodal":false,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support.","cacheSources":["https://huggingface.co/moonshotai/Kimi-Linear-48B-A3B-Instruct/blob/e1df551a447157d4658b573f9a695d57658590e9/config.json","https://huggingface.co/moonshotai/Kimi-Linear-48B-A3B-Instruct/blob/main/modeling_kimi.py","https://github.com/fla-org/flash-linear-attention/blob/main/fla/modules/conv/short_conv.py","https://github.com/fla-org/flash-linear-attention/blob/main/fla/ops/kda/fused_recurrent.py","https://github.com/fla-org/flash-linear-attention/blob/main/fla/ops/kda/chunk.py"]},{"id":"llama-3-3-70b","name":"Llama 3.3 70B","maker":"Meta","repo":"meta-llama/Llama-3.3-70B-Instruct","parametersB":70.553706496,"checkpointParameters":70553706496,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":80,"kvHeads":8,"headDim":128,"contextLimit":131072,"configuredContextLimit":131072,"contextNote":"Publisher configuration default.","category":"General","architecture":"Dense","summary":"Current dense 70B Llama, retained for the established Llama ecosystem.","license":"Llama 3.3 Community","source":"https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct","configSource":"https://github.com/meta-llama/llama-models/blob/main/models/sku_list.py","metadataSource":"https://huggingface.co/api/models/meta-llama/Llama-3.3-70B-Instruct","revision":"6f6073b423013f6a7d4d9f39144961bfbfbc386b","reviewedAt":"2026-09-06","modalities":["text"],"cache":{"kind":"standard","attention":[{"layers":80,"kvHeads":8,"keyDim":128,"valueDim":128}],"notes":"Text decoding cache; image/audio processing buffers and runtime workspaces are additional.","source":"https://github.com/meta-llama/llama-models/blob/main/models/sku_list.py"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support.","multimodal":false,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support."},{"id":"qwen3-coder-next-80b-a3b","name":"Qwen3 Coder Next 80B-A3B","maker":"Qwen","repo":"Qwen/Qwen3-Coder-Next","parametersB":79.674391296,"checkpointParameters":79674391296,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":48,"kvHeads":2,"headDim":256,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Publisher configuration default.","category":"Coding","architecture":"Mixture of experts","summary":"Sparse coding model for repository work and coding agents.","license":"Apache 2.0","source":"https://huggingface.co/Qwen/Qwen3-Coder-Next","configSource":"https://huggingface.co/Qwen/Qwen3-Coder-Next/blob/a7fbcb5c0e12d62a448eaa0e260346bf5dcc0feb/config.json","metadataSource":"https://huggingface.co/api/models/Qwen/Qwen3-Coder-Next","revision":"a7fbcb5c0e12d62a448eaa0e260346bf5dcc0feb","reviewedAt":"2026-09-06","modalities":["text"],"activeParametersB":3,"cache":{"kind":"hybrid","attention":[{"layers":12,"kvHeads":2,"keyDim":256,"valueDim":256}],"notes":"Gated DeltaNet states use FP32; convolution states use the model dtype. Full-attention layers alone grow with text length. MTP speculation is excluded from cache.","source":"https://huggingface.co/Qwen/Qwen3-Coder-Next/blob/a7fbcb5c0e12d62a448eaa0e260346bf5dcc0feb/config.json","recurrent":{"layers":36,"keyHeads":16,"valueHeads":32,"keyDim":128,"valueDim":128,"convKernel":4,"stateBytes":4,"convBytes":2}},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support.","multimodal":false,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support."},{"id":"llama-4-scout-109b-a17b","name":"Llama 4 Scout 109B-A17B","maker":"Meta","repo":"meta-llama/Llama-4-Scout-17B-16E-Instruct","parametersB":108.641793536,"checkpointParameters":108641793536,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":48,"kvHeads":8,"headDim":128,"contextLimit":10485760,"configuredContextLimit":10485760,"contextNote":"Publisher configuration default.","category":"General","architecture":"Mixture of experts","summary":"Sparse text and image model with a distinctive 10M supported context.","license":"Llama 4 Community","source":"https://huggingface.co/meta-llama/Llama-4-Scout-17B-16E-Instruct","configSource":"https://github.com/meta-pytorch/torchtune/blob/main/torchtune/models/llama4/_model_builders.py","metadataSource":"https://huggingface.co/api/models/meta-llama/Llama-4-Scout-17B-16E-Instruct","revision":"92f3b1597a195b523d8d9e5700e57e4fbb8f20d3","reviewedAt":"2026-09-06","modalities":["text","image"],"activeParametersB":17,"verificationNote":"Meta-owned torchtune llama4_scout_17b_16e builder independently verifies 48 layers, 8 KV heads, 5120 hidden width / 40 query heads = 128 head dimension, 8192-token chunks, every fourth layer global, and 10485760 context limit.","additionalSources":["https://github.com/meta-llama/llama-models/blob/main/models/llama4/MODEL_CARD.md"],"cache":{"kind":"standard","attention":[{"layers":48,"kvHeads":8,"keyDim":128,"valueDim":128}],"runtime":"Meta reference, BF16 full-history allocation in all 48 layers","precisionLocked16Bit":true,"notes":"Meta reference allocates full configured history for K and V in every layer. Scout has 36 chunked 8192-token attention layers and 12 global layers, but that attention mask does not make the reference evict cache. This conservative profile does not assume an optimized chunk-cache engine.","source":"https://github.com/meta-llama/llama-models/blob/main/models/llama4/model.py"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included.","cacheSources":["https://github.com/meta-pytorch/torchtune/blob/main/torchtune/models/llama4/_model_builders.py","https://github.com/meta-llama/llama-models/blob/main/models/llama4/model.py"]},{"id":"gpt-oss-120b","name":"gpt-oss 120B","maker":"OpenAI","repo":"openai/gpt-oss-120b","parametersB":116.829156672,"checkpointParameters":116829156672,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":36,"kvHeads":8,"headDim":64,"contextLimit":131072,"configuredContextLimit":131072,"contextNote":"Publisher configuration default.","category":"Reasoning","architecture":"Mixture of experts","summary":"Larger OpenAI reasoning model supplied with native MXFP4 expert weights.","license":"Apache 2.0","source":"https://huggingface.co/openai/gpt-oss-120b","configSource":"https://huggingface.co/openai/gpt-oss-120b/blob/b5c939de8f754692c1647ca79fbf85e8c1e70f8a/config.json","metadataSource":"https://huggingface.co/api/models/openai/gpt-oss-120b","revision":"b5c939de8f754692c1647ca79fbf85e8c1e70f8a","reviewedAt":"2026-09-06","modalities":["text"],"activeParametersB":5.1,"cache":{"kind":"sliding","attention":[{"layers":18,"kvHeads":8,"keyDim":64,"valueDim":64},{"layers":18,"kvHeads":8,"keyDim":64,"valueDim":64,"window":128}],"notes":"Alternating full attention and 128-token sliding attention; savings require runtime sliding-cache eviction.","source":"https://huggingface.co/openai/gpt-oss-120b/blob/b5c939de8f754692c1647ca79fbf85e8c1e70f8a/config.json"},"quantizationCaveat":"Native MXFP4 quantizes expert matrices, not every tensor. Use the supplied native checkpoint bytes for native mode; generic 4-bit is a different estimate. Unpacking MXFP4 to BF16 changes residency.","nativeWeights":{"format":"MXFP4 experts + BF16 other weights","bytes":65248815744,"source":"https://huggingface.co/openai/gpt-oss-120b/blob/b5c939de8f754692c1647ca79fbf85e8c1e70f8a/model.safetensors.index.json"},"multimodal":false,"weightNote":"Native MXFP4 quantizes expert matrices, not every tensor. Use the supplied native checkpoint bytes for native mode; generic 4-bit is a different estimate. Unpacking MXFP4 to BF16 changes residency.","nativeWeightsGiB":60.76769506931305,"nativeWeightLabel":"Native MXFP4"},{"id":"mistral-small-4-119b-a6-5b","name":"Mistral Small 4 119B-A6.5B","maker":"Mistral AI","repo":"mistralai/Mistral-Small-4-119B-2603","parametersB":119.401317952,"checkpointParameters":119401317952,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":36,"kvHeads":32,"headDim":128,"contextLimit":262144,"configuredContextLimit":1048576,"contextNote":"Use the model-card supported 256K limit; configuration has a larger positional allocation.","category":"General","architecture":"Mixture of experts","summary":"Sparse model combining chat, reasoning, coding and image input.","license":"Apache 2.0","source":"https://huggingface.co/mistralai/Mistral-Small-4-119B-2603","configSource":"https://huggingface.co/mistralai/Mistral-Small-4-119B-2603/blob/a11f36bebf709121056b1dbcc943d1c6afbe494d/config.json","metadataSource":"https://huggingface.co/api/models/mistralai/Mistral-Small-4-119B-2603","revision":"a11f36bebf709121056b1dbcc943d1c6afbe494d","reviewedAt":"2026-09-06","modalities":["text","image"],"activeParametersB":6.5,"cache":{"kind":"mla","attention":[],"latent":{"layers":36,"rank":256,"ropeDim":64},"notes":"Compressed MLA cache assumes a runtime that stores the shared latent and rotary key once per token. Expanded attention backends use more memory.","source":"https://huggingface.co/mistralai/Mistral-Small-4-119B-2603/blob/a11f36bebf709121056b1dbcc943d1c6afbe494d/config.json"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included."},{"id":"nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B-A12B","maker":"NVIDIA","repo":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16","parametersB":123.611012096,"checkpointParameters":123611012096,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":88,"kvHeads":2,"headDim":128,"contextLimit":1048576,"configuredContextLimit":262144,"contextNote":"Publisher supports 1M. Default config is 256K; override the runtime maximum to use 1M.","category":"Reasoning","architecture":"Mixture of experts","summary":"Hybrid sparse model for reasoning and agents with 1M context support.","license":"NVIDIA Nemotron Open Model License","source":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16","configSource":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16/blob/2dc98e2afe4face0e4ce40972a915c45368bd34a/config.json","metadataSource":"https://huggingface.co/api/models/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16","revision":"2dc98e2afe4face0e4ce40972a915c45368bd34a","reviewedAt":"2026-09-06","modalities":["text"],"activeParametersB":12,"cache":{"kind":"hybrid","attention":[{"layers":8,"kvHeads":2,"keyDim":128,"valueDim":128}],"recurrent":{"layers":40,"keyHeads":8,"valueHeads":128,"keyDim":128,"valueDim":64,"convKernel":4,"stateBytes":4,"convBytes":2},"notes":"Mamba-2 recurrent state is heads × head dimension × state size in FP32. Convolution channels = heads × head dimension + 2 × groups × state size in BF16. MTP speculative caches excluded.","source":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-BF16/blob/2dc98e2afe4face0e4ce40972a915c45368bd34a/config.json","recurrentType":"mamba2","blockCounts":{"M":40,"E":40,"*":8}},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":false,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra."},{"id":"devstral-2-123b","name":"Devstral 2 123B","maker":"Mistral AI","repo":"mistralai/Devstral-2-123B-Instruct-2512","parametersB":125.02598984,"checkpointParameters":125025989840,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":88,"kvHeads":8,"headDim":128,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Publisher configuration default.","category":"Coding","architecture":"Dense","summary":"Large dense model for coding agents and repository changes.","license":"Modified MIT","source":"https://huggingface.co/mistralai/Devstral-2-123B-Instruct-2512","configSource":"https://huggingface.co/mistralai/Devstral-2-123B-Instruct-2512/blob/1613bf01adb5e1c6fdc196b46e6b173eae75eb4a/config.json","metadataSource":"https://huggingface.co/api/models/mistralai/Devstral-2-123B-Instruct-2512","revision":"1613bf01adb5e1c6fdc196b46e6b173eae75eb4a","reviewedAt":"2026-09-06","modalities":["text"],"cache":{"kind":"standard","attention":[{"layers":88,"kvHeads":8,"keyDim":128,"valueDim":128}],"notes":"Text decoding cache; image/audio processing buffers and runtime workspaces are additional.","source":"https://huggingface.co/mistralai/Devstral-2-123B-Instruct-2512/blob/1613bf01adb5e1c6fdc196b46e6b173eae75eb4a/config.json"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support.","multimodal":false,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support."},{"id":"qwen3-5-122b-a10b","name":"Qwen3.5 122B-A10B","maker":"Qwen","repo":"Qwen/Qwen3.5-122B-A10B","parametersB":125.086497008,"checkpointParameters":125086497008,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":48,"kvHeads":2,"headDim":256,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","category":"General","architecture":"Mixture of experts","summary":"Large sparse model for text, images, reasoning and coding.","license":"Apache 2.0","source":"https://huggingface.co/Qwen/Qwen3.5-122B-A10B","configSource":"https://huggingface.co/Qwen/Qwen3.5-122B-A10B/blob/dc4d348443bc740c68e2d77492492c11606384d5/config.json","metadataSource":"https://huggingface.co/api/models/Qwen/Qwen3.5-122B-A10B","revision":"dc4d348443bc740c68e2d77492492c11606384d5","reviewedAt":"2026-09-06","modalities":["text","image","video"],"activeParametersB":10,"extendedContextLimit":1010000,"extendedContextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","cache":{"kind":"hybrid","attention":[{"layers":12,"kvHeads":2,"keyDim":256,"valueDim":256}],"notes":"Gated DeltaNet states use FP32; convolution states use the model dtype. Full-attention layers alone grow with text length. MTP speculation is excluded from cache.","source":"https://huggingface.co/Qwen/Qwen3.5-122B-A10B/blob/dc4d348443bc740c68e2d77492492c11606384d5/config.json","recurrent":{"layers":36,"keyHeads":16,"valueHeads":64,"keyDim":128,"valueDim":128,"convKernel":4,"stateBytes":4,"convBytes":2}},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra."},{"id":"mistral-medium-3-5-128b","name":"Mistral Medium 3.5 128B","maker":"Mistral AI","repo":"mistralai/Mistral-Medium-3.5-128B","parametersB":127.704210176,"checkpointParameters":127704210176,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":88,"kvHeads":8,"headDim":128,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Publisher configuration default.","category":"General","architecture":"Dense","summary":"Dense text and image model combining chat, reasoning and coding.","license":"Modified MIT","source":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B","configSource":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B/blob/22b2b868a15677cfa6061277ed2f653d1349a9ab/config.json","metadataSource":"https://huggingface.co/api/models/mistralai/Mistral-Medium-3.5-128B","revision":"22b2b868a15677cfa6061277ed2f653d1349a9ab","reviewedAt":"2026-09-06","modalities":["text","image"],"cache":{"kind":"standard","attention":[{"layers":88,"kvHeads":8,"keyDim":128,"valueDim":128}],"notes":"Text decoding cache; image/audio processing buffers and runtime workspaces are additional.","source":"https://huggingface.co/mistralai/Mistral-Medium-3.5-128B/blob/22b2b868a15677cfa6061277ed2f653d1349a9ab/config.json"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Use conversions made after the publisher config correction c4be198050fb5789774a55b92ed697becfbf20ae.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Use conversions made after the publisher config correction c4be198050fb5789774a55b92ed697becfbf20ae."},{"id":"qwen3-8-flash-next","name":"Qwen3.8 Flash Next","maker":"Qwen","repo":"Qwen/Qwen3.8-Flash-Next","parametersB":179.999981459,"checkpointParameters":179999981459,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":48,"kvHeads":2,"headDim":256,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","category":"General","architecture":"Mixture of experts","summary":"Experimental sparse text and image model with large offloadable embedding tables.","license":"Qwen Community 1.0","source":"https://huggingface.co/Qwen/Qwen3.8-Flash-Next","configSource":"https://huggingface.co/Qwen/Qwen3.8-Flash-Next/blob/de4b8e4d43b917e7706784d8bb445c9af86a3540/config.json","metadataSource":"https://huggingface.co/api/models/Qwen/Qwen3.8-Flash-Next","revision":"de4b8e4d43b917e7706784d8bb445c9af86a3540","reviewedAt":"2026-09-06","modalities":["text","image","video"],"activeParametersB":6,"extendedContextLimit":1000000,"extendedContextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","parameterGroups":{"mainModelB":125,"ngramEmbeddingB":51,"mtpB":4,"placementNote":"All weights count for fully resident memory. Host offload changes host and device budgets separately."},"cache":{"kind":"hybrid","attention":[{"layers":12,"kvHeads":2,"keyDim":256,"valueDim":256}],"buffers":[{"name":"Pooled QSA index keys","layers":12,"elementsPerToken":128,"bytesPerElement":2,"divisor":4,"rounding":"ceil"},{"name":"QSA raw key and MRoPE tail","layers":12,"elementsPerToken":140,"bytesPerElement":2,"fixedTokens":4},{"name":"GDN recurrent matrices","layers":36,"elementsPerToken":786432,"bytesPerElement":4,"fixedTokens":1},{"name":"GDN convolution states","layers":36,"elementsPerToken":30720,"bytesPerElement":2,"fixedTokens":1},{"name":"PLE dilated convolution state","layers":1,"elementsPerToken":92160,"bytesPerElement":2,"fixedTokens":1},{"name":"PLE previous token IDs","layers":1,"elementsPerToken":2,"bytesPerElement":4,"fixedTokens":1}],"runtime":"vLLM GQA + QSA + GDN, BF16 cache","precisionLocked16Bit":true,"notes":"12 full-context GQA caches plus QSA index keys pooled over four tokens; four raw key/MRoPE tail slots per QSA layer. 36 GDN FP32 recurrent states plus BF16 convolutions and one PLE convolution state. Counts the complete model, including the resident n-gram table; host offload requires a separate memory allocation. Cache tensor payload for vLLM with BF16 main cache, FP32 recurrent state, no speculative decoding, and no prefix-state checkpoint copies. Page padding, attention workspaces and allocator reserves are additional. Sparse top-k selection reduces computation, not retention of the main full-context cache.","source":"https://github.com/vllm-project/vllm/blob/294fbb4f595f61743124fc2e56a99e128cbe10da/vllm/models/qwen4_exp/common/qsa_cache.py#L801-L864"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":true,"weightNote":"The PLE n-gram table remains at BF16 in these estimates; only the remaining weights use the selected precision. All weights are counted in the selected memory pool. Optional host placement requires a separate host-memory budget. Quantized model files and kernels must support this architecture.","cacheSources":["https://github.com/vllm-project/vllm/blob/294fbb4f595f61743124fc2e56a99e128cbe10da/vllm/models/qwen4_exp/nvidia/indexer_qsa.py#L148-L166","https://github.com/vllm-project/vllm/blob/294fbb4f595f61743124fc2e56a99e128cbe10da/vllm/model_executor/layers/mamba/mamba_utils.py#L264-L285","https://github.com/vllm-project/vllm/blob/294fbb4f595f61743124fc2e56a99e128cbe10da/vllm/models/qwen4_exp/nvidia/ple_layer.py#L533-L655","https://github.com/vllm-project/vllm/blob/294fbb4f595f61743124fc2e56a99e128cbe10da/vllm/models/qwen4_exp/nvidia/model_state.py#L25-L62","https://github.com/huggingface/transformers/blob/c93057d4835cd31752bb56f59989dd27696eb45b/src/transformers/models/qwen4_exp/modeling_qwen4_exp.py#L1078-L1180","https://huggingface.co/Qwen/Qwen3.8-Flash-Next/blob/de4b8e4d43b917e7706784d8bb445c9af86a3540/config.json"],"cacheBackendNote":"Explicit FP32 recurrent state in this profile. QSA raw cache includes all three int64 MRoPE positions (24bytes) per slot. ceil(T/4) conservatively reserves the open pool. PLE conv is TP-replicated; GDN states are TP-sharded. Audited vLLM PLE input path requires pipeline_parallel_size=1. No generic4-bit PLE lookup implementation was found; serializedFP8 PLE with a globalFP32 scale is separately supported but checkpoint availability was not verified.","fixedWeightGroups":[{"name":"PLE n-gram embedding table","parameters":51200245760,"bits":16,"source":"https://github.com/vllm-project/vllm/blob/294fbb4f595f61743124fc2e56a99e128cbe10da/vllm/models/qwen4_exp/nvidia/ple_layer.py#L85-L174"}]},{"id":"minimax-m2-7","name":"MiniMax M2.7","maker":"MiniMax","repo":"MiniMaxAI/MiniMax-M2.7","parametersB":228.689764864,"checkpointParameters":228689764864,"parameterBasis":"Hugging Face normalized whole-checkpoint tensor count.","layers":62,"kvHeads":8,"headDim":128,"contextLimit":204800,"configuredContextLimit":204800,"category":"Coding","architecture":"Mixture of experts","summary":"Current smaller MiniMax for coding, reasoning and office tasks.","license":"MiniMax M2.7 License","source":"https://huggingface.co/MiniMaxAI/MiniMax-M2.7","configSource":"https://huggingface.co/MiniMaxAI/MiniMax-M2.7/blob/d494266a4affc0d2995ba1fa35c8481cbd84294b/config.json","metadataSource":"https://huggingface.co/api/models/MiniMaxAI/MiniMax-M2.7","revision":"d494266a4affc0d2995ba1fa35c8481cbd84294b","reviewedAt":"2026-09-06","modalities":["text"],"cache":{"kind":"standard","attention":[{"layers":62,"kvHeads":8,"keyDim":128,"valueDim":128}],"source":"https://huggingface.co/MiniMaxAI/MiniMax-M2.7/blob/d494266a4affc0d2995ba1fa35c8481cbd84294b/config.json","notes":"Text decoding cache; runtime workspaces and optional speculative caches are additional."},"quantizationCaveat":"Precision estimates need quantization scales and some higher-precision tensors. Backend support and actual file sizes vary. Whole-checkpoint counts can include auxiliary weights.","multimodal":false,"weightNote":"Precision estimates need quantization scales and some higher-precision tensors. Backend support and actual file sizes vary. Whole-checkpoint counts can include auxiliary weights."},{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","maker":"DeepSeek","repo":"deepseek-ai/DeepSeek-V4-Flash-0731","parametersB":304.180418494,"checkpointParameters":304180418494,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":43,"kvHeads":1,"headDim":512,"contextLimit":1048576,"configuredContextLimit":1048576,"contextNote":"Publisher configuration default.","category":"Reasoning","architecture":"Mixture of experts","summary":"Updated DeepSeek Flash for reasoning, coding and tool use.","license":"MIT","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731","configSource":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731/blob/7872f01b1d1fe23eabc4c98b48bffcef5a386062/config.json","metadataSource":"https://huggingface.co/api/models/deepseek-ai/DeepSeek-V4-Flash-0731","revision":"7872f01b1d1fe23eabc4c98b48bffcef5a386062","reviewedAt":"2026-09-06","modalities":["text"],"cache":{"kind":"hybrid","attention":[],"buffers":[{"name":"Shared K/V sliding ring (BF16)","layers":43,"elementsPerToken":512,"bytesPerElement":2,"fixedTokens":128},{"name":"CSA compressed shared K/V (BF16)","layers":21,"elementsPerToken":512,"bytesPerElement":2,"divisor":4,"rounding":"floor"},{"name":"HCA compressed shared K/V (BF16)","layers":20,"elementsPerToken":512,"bytesPerElement":2,"divisor":128,"rounding":"floor"},{"name":"CSA compressed index vectors (BF16)","layers":21,"elementsPerToken":128,"bytesPerElement":2,"divisor":4,"rounding":"floor"},{"name":"CSA pooling values and scores (FP32)","layers":21,"elementsPerToken":2048,"bytesPerElement":4,"fixedTokens":8},{"name":"HCA pooling values and scores (FP32)","layers":20,"elementsPerToken":1024,"bytesPerElement":4,"fixedTokens":128},{"name":"CSA index pooling values and scores (FP32)","layers":21,"elementsPerToken":512,"bytesPerElement":4,"fixedTokens":8},{"name":"Shared rotary frequency tables (complex64)","layers":2,"elementsPerToken":32,"bytesPerElement":8,"perRequest":false}],"runtime":"DeepSeek reference BF16 cache (text, DSpark disabled)","precisionLocked16Bit":true,"source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731/blob/7872f01b1d1fe23eabc4c98b48bffcef5a386062/inference/model.py","notes":"Persistent cache uses BF16 vectors and FP32 pooling state; Q8 does not shrink it. Assumes text generation with speculative decoding disabled. Long prompts require chunked prefill; the reference demo’s full-prefill peak is not included."},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":false,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra."},{"id":"deepseek-v4-flash-vision","name":"DeepSeek V4 Flash Vision","maker":"DeepSeek","repo":"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","parametersB":304.646824126,"checkpointParameters":304646824126,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":43,"kvHeads":1,"headDim":512,"contextLimit":1048576,"configuredContextLimit":1048576,"contextNote":"Publisher configuration default.","category":"General","architecture":"Mixture of experts","summary":"Experimental vision version of the updated DeepSeek Flash.","license":"MIT","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","configSource":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp/blob/6821d6ad3681a4b137b066b76094fa82ebd0a380/config.json","metadataSource":"https://huggingface.co/api/models/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","revision":"6821d6ad3681a4b137b066b76094fa82ebd0a380","reviewedAt":"2026-09-06","modalities":["text","image"],"cache":{"kind":"hybrid","attention":[],"buffers":[{"name":"Shared K/V sliding ring (BF16)","layers":43,"elementsPerToken":512,"bytesPerElement":2,"fixedTokens":128},{"name":"CSA compressed shared K/V (BF16)","layers":21,"elementsPerToken":512,"bytesPerElement":2,"divisor":4,"rounding":"floor"},{"name":"HCA compressed shared K/V (BF16)","layers":20,"elementsPerToken":512,"bytesPerElement":2,"divisor":128,"rounding":"floor"},{"name":"CSA compressed index vectors (BF16)","layers":21,"elementsPerToken":128,"bytesPerElement":2,"divisor":4,"rounding":"floor"},{"name":"CSA pooling values and scores (FP32)","layers":21,"elementsPerToken":2048,"bytesPerElement":4,"fixedTokens":8},{"name":"HCA pooling values and scores (FP32)","layers":20,"elementsPerToken":1024,"bytesPerElement":4,"fixedTokens":128},{"name":"CSA index pooling values and scores (FP32)","layers":21,"elementsPerToken":512,"bytesPerElement":4,"fixedTokens":8},{"name":"Shared rotary frequency tables (complex64)","layers":2,"elementsPerToken":32,"bytesPerElement":8,"perRequest":false}],"runtime":"DeepSeek reference BF16 cache (text, DSpark disabled)","precisionLocked16Bit":true,"source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp/blob/6821d6ad3681a4b137b066b76094fa82ebd0a380/inference/model.py","notes":"Persistent cache uses BF16 vectors and FP32 pooling state; Q8 does not shrink it. Assumes text generation with speculative decoding disabled. Long prompts require chunked prefill; the reference demo’s full-prefill peak is not included."},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra."},{"id":"glm-5-3-flash","name":"GLM-5.3 Flash","maker":"Z.ai","repo":"zai-org/GLM-5.3-Flash","parametersB":321.32303139,"checkpointParameters":321323031390,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":45,"kvHeads":64,"headDim":256,"contextLimit":1048576,"configuredContextLimit":1048576,"contextNote":"Publisher configuration default.","category":"General","architecture":"Mixture of experts","summary":"Current multimodal GLM with hybrid recurrent and sparse attention.","license":"MIT","source":"https://huggingface.co/zai-org/GLM-5.3-Flash","configSource":"https://huggingface.co/zai-org/GLM-5.3-Flash/blob/690b705278a3a58e538fcb37c2ca8b5f9511213c/config.json","metadataSource":"https://huggingface.co/api/models/zai-org/GLM-5.3-Flash","revision":"690b705278a3a58e538fcb37c2ca8b5f9511213c","reviewedAt":"2026-09-06","modalities":["text","image"],"activeParametersB":18,"cache":{"kind":"hybrid","attention":[],"latent":{"layers":11,"rank":512,"ropeDim":0},"buffers":[{"name":"Pooled DSA index keys and scales","layers":11,"elementsPerToken":132,"bytesPerElement":1,"divisor":4,"rounding":"ceil"},{"name":"DSA raw key and gate tail","layers":11,"elementsPerToken":256,"bytesPerElement":2,"fixedTokens":4},{"name":"KDA recurrent matrices","layers":34,"elementsPerToken":1048576,"bytesPerElement":4,"fixedTokens":1},{"name":"KDA convolution states","layers":34,"elementsPerToken":73728,"bytesPerElement":2,"fixedTokens":1}],"runtime":"vLLM compressed MLA + KDA, BF16 cache","precisionLocked16Bit":true,"notes":"11 MLA layers retain 512-element latents; pooled DSA indices retain one 132-byte entry per four tokens plus four raw key/gate tail slots. 34 KDA layers keep FP32 64×128×128 recurrent matrices and BF16 convolution states. Cache tensor payload for vLLM with BF16 main cache, FP32 recurrent state, no speculative decoding, and no prefix-state checkpoint copies. Page padding, attention workspaces and allocator reserves are additional. Sparse top-k selection reduces computation, not retention of the main full-context cache.","source":"https://github.com/vllm-project/vllm/blob/294fbb4f595f61743124fc2e56a99e128cbe10da/vllm/models/glm5next/nvidia/attention.py#L78-L204"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","cacheSources":["https://github.com/vllm-project/vllm/blob/294fbb4f595f61743124fc2e56a99e128cbe10da/vllm/models/glm5next/nvidia/attention.py#L274-L307","https://github.com/vllm-project/vllm/blob/294fbb4f595f61743124fc2e56a99e128cbe10da/vllm/model_executor/layers/mamba/mamba_utils.py#L133-L139","https://github.com/vllm-project/vllm/blob/294fbb4f595f61743124fc2e56a99e128cbe10da/vllm/model_executor/layers/mamba/mamba_utils.py#L288-L311","https://github.com/vllm-project/vllm/blob/294fbb4f595f61743124fc2e56a99e128cbe10da/vllm/v1/attention/backends/mla/flashinfer_mla_sparse_sm90.py#L64-L151","https://huggingface.co/zai-org/GLM-5.3-Flash/blob/690b705278a3a58e538fcb37c2ca8b5f9511213c/config.json"],"cacheBackendNote":"FLASHINFER_MLA_SPARSE_SM90 supports BF16 and 512-element no-RoPE geometry; also supported by Blackwell FlashInfer sparse backend. Pooled index storage requires token block size multiple of128. ceil(T/4) reserves the open pool as a conservative extra index entry."},{"id":"qwen3-5-397b-a17b","name":"Qwen3.5 397B-A17B","maker":"Qwen","repo":"Qwen/Qwen3.5-397B-A17B","parametersB":403.397928944,"checkpointParameters":403397928944,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":60,"kvHeads":2,"headDim":256,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","category":"General","architecture":"Mixture of experts","summary":"Large sparse text and image model for high-memory systems.","license":"Apache 2.0","source":"https://huggingface.co/Qwen/Qwen3.5-397B-A17B","configSource":"https://huggingface.co/Qwen/Qwen3.5-397B-A17B/blob/8472618112abcbd45acbcdc58436aff4233c23f7/config.json","metadataSource":"https://huggingface.co/api/models/Qwen/Qwen3.5-397B-A17B","revision":"8472618112abcbd45acbcdc58436aff4233c23f7","reviewedAt":"2026-09-06","modalities":["text","image","video"],"activeParametersB":17,"extendedContextLimit":1010000,"extendedContextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","cache":{"kind":"hybrid","attention":[{"layers":15,"kvHeads":2,"keyDim":256,"valueDim":256}],"notes":"Gated DeltaNet states use FP32; convolution states use the model dtype. Full-attention layers alone grow with text length. MTP speculation is excluded from cache.","source":"https://huggingface.co/Qwen/Qwen3.5-397B-A17B/blob/8472618112abcbd45acbcdc58436aff4233c23f7/config.json","recurrent":{"layers":45,"keyHeads":16,"valueHeads":64,"keyDim":128,"valueDim":128,"convKernel":4,"stateBytes":4,"convBytes":2}},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra."},{"id":"minimax-m3","name":"MiniMax M3","maker":"MiniMax","repo":"MiniMaxAI/MiniMax-M3","parametersB":427.04014016,"checkpointParameters":427040140160,"parameterBasis":"Hugging Face normalized whole-checkpoint tensor count.","layers":60,"kvHeads":4,"headDim":128,"contextLimit":1048576,"configuredContextLimit":1048576,"category":"General","architecture":"Mixture of experts","summary":"Current multimodal MiniMax with sparse attention and 1M context.","license":"MiniMax Community License","source":"https://huggingface.co/MiniMaxAI/MiniMax-M3","configSource":"https://huggingface.co/MiniMaxAI/MiniMax-M3/blob/f0e1c1e04d40177e4673a22097036854f536e9c0/config.json","metadataSource":"https://huggingface.co/api/models/MiniMaxAI/MiniMax-M3","revision":"f0e1c1e04d40177e4673a22097036854f536e9c0","reviewedAt":"2026-09-06","modalities":["text","image","video"],"cache":{"kind":"standard","attention":[{"layers":60,"kvHeads":4,"keyDim":128,"valueDim":128}],"buffers":[{"name":"MSA shared index-key cache, 128-token blocks","layers":57,"elementsPerToken":16384,"bytesPerElement":2,"divisor":128,"rounding":"ceil"}],"runtime":"vLLM NVIDIA, BF16 attention and BF16 MSA index cache, MTP disabled","precisionLocked16Bit":true,"notes":"All 60 layers retain full-history K/V. The 57 sparse layers also retain one 128-element index key per token; the four index heads are query heads, not four cached key heads. Sparse block selection reduces computation without evicting the full cache. Assumes 128-token pages, no speculative/MTP cache, and separately budgeted scheduler and prefill workspace.","source":"https://github.com/vllm-project/vllm/blob/main/vllm/models/minimax_m3/nvidia/model.py"},"quantizationCaveat":"Precision estimates need quantization scales and some higher-precision tensors. Backend support and actual file sizes vary. Whole-checkpoint counts can include auxiliary weights.","activeParametersB":23,"multimodal":true,"weightNote":"Precision estimates need quantization scales and some higher-precision tensors. Backend support and actual file sizes vary. Whole-checkpoint counts can include auxiliary weights.","cacheSources":["https://huggingface.co/MiniMaxAI/MiniMax-M3/blob/main/config.json","https://github.com/vllm-project/vllm/blob/main/vllm/models/minimax_m3/nvidia/model.py","https://github.com/vllm-project/vllm/blob/main/vllm/models/minimax_m3/common/indexer.py","https://github.com/vllm-project/vllm/blob/main/vllm/v1/kv_cache_interface.py"]},{"id":"nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B-A55B","maker":"NVIDIA","repo":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","parametersB":560.524578816,"checkpointParameters":560524578816,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":108,"kvHeads":2,"headDim":128,"contextLimit":1048576,"configuredContextLimit":262144,"contextNote":"Publisher supports 1M. Default config is 256K; override the runtime maximum to use 1M.","category":"Reasoning","architecture":"Mixture of experts","summary":"Largest Nemotron 3 for reasoning and long-running agents.","license":"OpenMDW 1.1","source":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","configSource":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16/blob/77df655d5e9f8362164ed14dd8b48f8bce657498/config.json","metadataSource":"https://huggingface.co/api/models/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","revision":"77df655d5e9f8362164ed14dd8b48f8bce657498","reviewedAt":"2026-09-06","modalities":["text"],"activeParametersB":55,"cache":{"kind":"hybrid","attention":[{"layers":12,"kvHeads":2,"keyDim":128,"valueDim":128}],"recurrent":{"layers":48,"keyHeads":8,"valueHeads":256,"keyDim":128,"valueDim":64,"convKernel":4,"stateBytes":4,"convBytes":2},"notes":"Mamba-2 recurrent state is heads × head dimension × state size in FP32. Convolution channels = heads × head dimension + 2 × groups × state size in BF16. MTP speculative caches excluded.","source":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16/blob/77df655d5e9f8362164ed14dd8b48f8bce657498/config.json","recurrentType":"mamba2","blockCounts":{"mamba":48,"moe":48,"attention":12}},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":false,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra."},{"id":"mistral-large-3-675b-a41b","name":"Mistral Large 3 675B-A41B","maker":"Mistral AI","repo":"mistralai/Mistral-Large-3-675B-Instruct-2512","parametersB":675,"checkpointParameters":675000000000,"parameterBasis":"Publisher approximate total from model card.","layers":61,"kvHeads":128,"headDim":192,"contextLimit":262144,"configuredContextLimit":294912,"contextNote":"Use the model-card supported 256K limit; configuration has a larger positional allocation.","category":"General","architecture":"Mixture of experts","summary":"Large sparse text and image model for server deployments.","license":"Apache 2.0","source":"https://huggingface.co/mistralai/Mistral-Large-3-675B-Instruct-2512","configSource":"https://huggingface.co/mistralai/Mistral-Large-3-675B-Instruct-2512/blob/383ffea2c7d60dfd44ca960e8e691709d4fdb9cd/params.json","metadataSource":"https://huggingface.co/api/models/mistralai/Mistral-Large-3-675B-Instruct-2512","revision":"383ffea2c7d60dfd44ca960e8e691709d4fdb9cd","reviewedAt":"2026-09-06","modalities":["text","image"],"activeParametersB":41,"cache":{"kind":"mla","attention":[],"latent":{"layers":61,"rank":512,"ropeDim":64},"notes":"Compressed MLA cache assumes a runtime that stores the shared latent and rotary key once per token. Expanded attention backends use more memory.","source":"https://huggingface.co/mistralai/Mistral-Large-3-675B-Instruct-2512/blob/383ffea2c7d60dfd44ca960e8e691709d4fdb9cd/params.json"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included."},{"id":"glm-5-3","name":"GLM-5.3","maker":"Z.ai","repo":"zai-org/GLM-5.3","parametersB":753.32994048,"checkpointParameters":753329940480,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":78,"kvHeads":64,"headDim":192,"contextLimit":1048576,"configuredContextLimit":1048576,"contextNote":"Publisher configuration default.","category":"Reasoning","architecture":"Mixture of experts","summary":"Current large GLM for reasoning, coding and extended agent tasks.","license":"GLM-5.3 License","source":"https://huggingface.co/zai-org/GLM-5.3","configSource":"https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/config.json","metadataSource":"https://huggingface.co/api/models/zai-org/GLM-5.3","revision":"aca966e4e02791568aa6a4ced368624b3d897f42","reviewedAt":"2026-09-06","modalities":["text"],"cache":{"kind":"mla","attention":[],"latent":{"layers":78,"rank":512,"ropeDim":64},"buffers":[{"name":"DSA index keys and scales","layers":21,"elementsPerToken":132,"bytesPerElement":1}],"runtime":"vLLM compressed MLA, BF16 cache","precisionLocked16Bit":true,"notes":"78 independent MLA caches; only 21 layers own a 128-value FP8 index plus one FP32 scale per token. Shared index selections do not share main MLA caches. Cache tensor payload for vLLM with BF16 main cache, FP32 recurrent state, no speculative decoding, and no prefix-state checkpoint copies. Page padding, attention workspaces and allocator reserves are additional. Sparse top-k selection reduces computation, not retention of the main full-context cache.","source":"https://github.com/vllm-project/vllm/blob/294fbb4f595f61743124fc2e56a99e128cbe10da/vllm/model_executor/models/deepseek_v2.py#L633-L722"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":false,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","cacheSources":["https://github.com/vllm-project/vllm/blob/294fbb4f595f61743124fc2e56a99e128cbe10da/vllm/model_executor/models/deepseek_v2.py#L1130-L1179","https://github.com/vllm-project/vllm/blob/294fbb4f595f61743124fc2e56a99e128cbe10da/vllm/v1/attention/backends/mla/flashmla_sparse.py#L110-L151","https://huggingface.co/zai-org/GLM-5.3/blob/aca966e4e02791568aa6a4ced368624b3d897f42/config.json"],"cacheBackendNote":"FLASHMLA_SPARSE supports BF16 and this 576-element geometry on Hopper/Blackwell. Generic Transformers currently stores expanded per-head K/V instead and is not represented by this profile."},{"id":"kimi-k2-7-code","name":"Kimi K2.7 Code","maker":"Moonshot AI","repo":"moonshotai/Kimi-K2.7-Code","parametersB":1026.879376368,"checkpointParameters":1026879376368,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":61,"kvHeads":64,"headDim":192,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Publisher configuration default.","category":"Coding","architecture":"Mixture of experts","summary":"Dedicated Kimi coding model with vision and a 256K context.","license":"Modified MIT","source":"https://huggingface.co/moonshotai/Kimi-K2.7-Code","configSource":"https://huggingface.co/moonshotai/Kimi-K2.7-Code/blob/74797c9c62378b951a1f6fcf5c4631024e9b8bef/config.json","metadataSource":"https://huggingface.co/api/models/moonshotai/Kimi-K2.7-Code","revision":"74797c9c62378b951a1f6fcf5c4631024e9b8bef","reviewedAt":"2026-09-06","modalities":["text","image"],"activeParametersB":32,"cache":{"kind":"mla","attention":[],"latent":{"layers":61,"rank":512,"ropeDim":64},"notes":"Compressed MLA cache assumes a runtime that stores the shared latent and rotary key once per token. Expanded attention backends use more memory.","source":"https://huggingface.co/moonshotai/Kimi-K2.7-Code/blob/74797c9c62378b951a1f6fcf5c4631024e9b8bef/config.json"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included."},{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","maker":"DeepSeek","repo":"deepseek-ai/DeepSeek-V4-Pro-0813","parametersB":1650.497936906,"checkpointParameters":1650497936906,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":61,"kvHeads":1,"headDim":512,"contextLimit":1048576,"configuredContextLimit":1048576,"contextNote":"Publisher configuration default.","category":"Reasoning","architecture":"Mixture of experts","summary":"Latest large DeepSeek Pro checkpoint for multi-GPU servers.","license":"MIT","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","configSource":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/config.json","metadataSource":"https://huggingface.co/api/models/deepseek-ai/DeepSeek-V4-Pro-0813","revision":"72e1d3230f6c080a530b0a1d46f8eb4602340597","reviewedAt":"2026-09-06","modalities":["text"],"cache":{"kind":"hybrid","attention":[],"buffers":[{"name":"Shared K/V sliding ring (BF16)","layers":61,"elementsPerToken":512,"bytesPerElement":2,"fixedTokens":128},{"name":"CSA compressed shared K/V (BF16)","layers":30,"elementsPerToken":512,"bytesPerElement":2,"divisor":4,"rounding":"floor"},{"name":"HCA compressed shared K/V (BF16)","layers":31,"elementsPerToken":512,"bytesPerElement":2,"divisor":128,"rounding":"floor"},{"name":"CSA compressed index vectors (BF16)","layers":30,"elementsPerToken":128,"bytesPerElement":2,"divisor":4,"rounding":"floor"},{"name":"CSA pooling values and scores (FP32)","layers":30,"elementsPerToken":2048,"bytesPerElement":4,"fixedTokens":8},{"name":"HCA pooling values and scores (FP32)","layers":31,"elementsPerToken":1024,"bytesPerElement":4,"fixedTokens":128},{"name":"CSA index pooling values and scores (FP32)","layers":30,"elementsPerToken":512,"bytesPerElement":4,"fixedTokens":8},{"name":"Shared rotary frequency tables (complex64)","layers":1,"elementsPerToken":32,"bytesPerElement":8,"perRequest":false}],"runtime":"DeepSeek reference BF16 cache (text, DSpark disabled)","precisionLocked16Bit":true,"source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813/blob/72e1d3230f6c080a530b0a1d46f8eb4602340597/inference/model.py","notes":"Persistent cache uses BF16 vectors and FP32 pooling state; Q8 does not shrink it. Assumes text generation with speculative decoding disabled. Long prompts require chunked prefill; the reference demo’s full-prefill peak is not included."},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":false,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra."},{"id":"qwen3-8-2-4t-a95b","name":"Qwen3.8 2.4T-A95B","maker":"Qwen","repo":"Qwen/Qwen3.8-2.4T-A95B","parametersB":2446.182725504,"checkpointParameters":2446182725504,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":92,"kvHeads":4,"headDim":256,"contextLimit":262144,"configuredContextLimit":262144,"contextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","category":"Reasoning","architecture":"Mixture of experts","summary":"Frontier Qwen reasoning model for large multi-GPU servers.","license":"Qwen3.8 Max","source":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","configSource":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/config.json","metadataSource":"https://huggingface.co/api/models/Qwen/Qwen3.8-2.4T-A95B","revision":"207bd685a7e3696cfaff12ded7c6a7ea0f88c996","reviewedAt":"2026-09-06","modalities":["text"],"activeParametersB":95,"extendedContextLimit":1010000,"extendedContextNote":"Requires the publisher-documented YaRN/RoPE scaling configuration; selecting a longer context alone does not enable it.","cache":{"kind":"hybrid","attention":[{"layers":23,"kvHeads":4,"keyDim":256,"valueDim":256}],"notes":"Gated DeltaNet states use FP32; convolution states use the model dtype. Full-attention layers alone grow with text length. MTP speculation is excluded from cache.","source":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B/blob/207bd685a7e3696cfaff12ded7c6a7ea0f88c996/config.json","recurrent":{"layers":69,"keyHeads":16,"valueHeads":128,"keyDim":128,"valueDim":128,"convKernel":4,"stateBytes":4,"convBytes":2}},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra.","multimodal":false,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count may include optional MTP or draft weights; cached speculative tokens are extra."},{"id":"kimi-k3","name":"Kimi K3","maker":"Moonshot AI","repo":"moonshotai/Kimi-K3","parametersB":2779.931837184,"checkpointParameters":2779931837184,"parameterBasis":"Hugging Face normalized checkpoint tensor count, including auxiliary tensors when present.","layers":93,"kvHeads":96,"headDim":192,"contextLimit":1048576,"configuredContextLimit":1048576,"contextNote":"Publisher configuration default.","category":"General","architecture":"Mixture of experts","summary":"Current frontier Kimi with text, image and video input and 1M context.","license":"Kimi K3 License","source":"https://huggingface.co/moonshotai/Kimi-K3","configSource":"https://huggingface.co/moonshotai/Kimi-K3/blob/f831ab66814297da540d832a5235f8e904f29d06/config.json","metadataSource":"https://huggingface.co/api/models/moonshotai/Kimi-K3","revision":"f831ab66814297da540d832a5235f8e904f29d06","reviewedAt":"2026-09-06","modalities":["text","image","video"],"activeParametersB":104,"cache":{"kind":"hybrid","attention":[],"latent":{"layers":24,"rank":512,"ropeDim":64},"buffers":[{"name":"KDA FP32 recurrent state","layers":69,"elementsPerToken":1572864,"bytesPerElement":4,"fixedTokens":1},{"name":"KDA BF16 convolution state, three history slots","layers":69,"elementsPerToken":110592,"bytesPerElement":2,"fixedTokens":1}],"runtime":"vLLM NVIDIA, BF16 compressed MLA and one active KDA state per request","precisionLocked16Bit":true,"notes":"24 attention layers cache one 576-element latent per token, including the 64-element shared key branch even without RoPE. Each of 69 KDA layers keeps FP32 recurrent state and three BF16 convolution history slots. Assumes no prefix-state checkpoints, DSpark, speculative decoding or RecoverSSM records. Engine pool allocation and prefill workspaces are separate.","source":"https://github.com/vllm-project/vllm/blob/main/vllm/models/kimi_k3/nvidia/mla.py"},"quantizationCaveat":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included.","multimodal":true,"weightNote":"Precision is a storage estimate. Quantized files include scales and some higher-precision tensors; use the actual file size when known. A format also needs backend and hardware support. Full checkpoint count includes media/auxiliary weights when present; a text-only conversion can be smaller. Image/audio encoder working memory is not included.","cacheSources":["https://huggingface.co/moonshotai/Kimi-K3/blob/main/config.json","https://github.com/vllm-project/vllm/blob/main/vllm/models/kimi_k3/nvidia/model.py","https://github.com/vllm-project/vllm/blob/main/vllm/models/kimi_k3/nvidia/mla.py","https://github.com/vllm-project/vllm/blob/main/vllm/models/kimi_k3/nvidia/kda.py","https://github.com/vllm-project/vllm/blob/main/vllm/model_executor/layers/mamba/mamba_utils.py","https://github.com/vllm-project/vllm/blob/main/vllm/v1/kv_cache_interface.py"]}],"hardware":[{"id":"rtx-5050","name":"GeForce RTX 5050","maker":"NVIDIA","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":320,"count":1,"summary":"8 GB GDDR6 dedicated memory.","caveat":"Desktop reference card. Partner cooling and power limits vary. Use a CUDA/runtime build with Blackwell support; no NVLink.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR6","configuration":"GeForce RTX 5050, desktop/reference specification, 8 GB GDDR6","interconnect":"PCIe 5.0; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":130,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 5050","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-5060","name":"GeForce RTX 5060","maker":"NVIDIA","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":448,"count":1,"summary":"8 GB GDDR7 dedicated memory.","caveat":"Desktop reference card. Partner cooling and power limits vary. Use a CUDA/runtime build with Blackwell support; no NVLink.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR7","configuration":"GeForce RTX 5060, desktop/reference specification, 8 GB GDDR7","interconnect":"PCIe 5.0; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":145,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 5060","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-5060-ti-8gb","name":"GeForce RTX 5060 Ti 8 GB","maker":"NVIDIA","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":448,"count":1,"summary":"8 GB GDDR7 dedicated memory.","caveat":"Desktop reference card. Partner cooling and power limits vary. Use a CUDA/runtime build with Blackwell support; no NVLink.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR7","configuration":"GeForce RTX 5060 Ti 8 GB, desktop/reference specification, 8 GB GDDR7","interconnect":"PCIe 5.0; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":180,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 5060 Ti 8 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-5060-ti-16gb","name":"GeForce RTX 5060 Ti 16 GB","maker":"NVIDIA","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":448,"count":1,"summary":"16 GB GDDR7 dedicated memory.","caveat":"Desktop reference card. Partner cooling and power limits vary. Use a CUDA/runtime build with Blackwell support; no NVLink.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR7","configuration":"GeForce RTX 5060 Ti 16 GB, desktop/reference specification, 16 GB GDDR7","interconnect":"PCIe 5.0; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":180,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 5060 Ti 16 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-5070","name":"GeForce RTX 5070","maker":"NVIDIA","memoryGiB":12,"reserveGiB":1,"type":"GPU","bandwidthGBs":672,"count":1,"summary":"12 GB GDDR7 dedicated memory.","caveat":"Desktop reference card. Partner cooling and power limits vary. Use a CUDA/runtime build with Blackwell support; no NVLink.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":12,"memoryType":"GDDR7","configuration":"GeForce RTX 5070, desktop/reference specification, 12 GB GDDR7","interconnect":"PCIe 5.0; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":250,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 5070","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-5070-ti","name":"GeForce RTX 5070 Ti","maker":"NVIDIA","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":896,"count":1,"summary":"16 GB GDDR7 dedicated memory.","caveat":"Desktop reference card. Partner cooling and power limits vary. Use a CUDA/runtime build with Blackwell support; no NVLink.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR7","configuration":"GeForce RTX 5070 Ti, desktop/reference specification, 16 GB GDDR7","interconnect":"PCIe 5.0; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":300,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 5070 Ti","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-5080","name":"GeForce RTX 5080","maker":"NVIDIA","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":960,"count":1,"summary":"16 GB GDDR7 dedicated memory.","caveat":"Desktop reference card. Partner cooling and power limits vary. Use a CUDA/runtime build with Blackwell support; no NVLink.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR7","configuration":"GeForce RTX 5080, desktop/reference specification, 16 GB GDDR7","interconnect":"PCIe 5.0; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":360,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 5080","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-5090","name":"GeForce RTX 5090","maker":"NVIDIA","memoryGiB":32,"reserveGiB":1,"type":"GPU","bandwidthGBs":1792,"count":1,"summary":"32 GB GDDR7 dedicated memory.","caveat":"Desktop reference card. Partner cooling and power limits vary. Use a CUDA/runtime build with Blackwell support; no NVLink.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":32,"memoryType":"GDDR7","configuration":"GeForce RTX 5090, desktop/reference specification, 32 GB GDDR7","interconnect":"PCIe 5.0; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":575,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 5090","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-3060-12gb","name":"GeForce RTX 3060 12 GB","maker":"NVIDIA","memoryGiB":12,"reserveGiB":1,"type":"GPU","bandwidthGBs":360,"count":1,"summary":"12 GB GDDR6 dedicated memory.","caveat":"12 GB desktop variant. The 8 GB card has a narrower memory bus. Partner power/cooling can vary.","source":"https://cs.pny.com.tw/en/upload/download_files/en_download_list_22h16_93qzs8s5n6.pdf","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":12,"memoryType":"GDDR6","configuration":"GeForce RTX 3060 12 GB, desktop/reference specification, 12 GB GDDR6","interconnect":"PCIe 4.0 x16; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"15 Gbit/s per pin × 192 bits ÷ 8 = 360 GB/s. PNY labels the bandwidth row Gbps in error; the data rate and bus width establish the GB/s result.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://cs.pny.com.tw/en/upload/download_files/en_download_list_22h16_93qzs8s5n6.pdf"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":170,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 3060 12 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"15 Gbit/s per pin × 192 bits ÷ 8 = 360 GB/s. PNY labels the bandwidth row Gbps in error; the data rate and bus width establish the GB/s result."},{"id":"rtx-4060-ti-16gb","name":"GeForce RTX 4060 Ti 16 GB","maker":"NVIDIA","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":288,"count":1,"summary":"16 GB GDDR6 dedicated memory.","caveat":"CUDA backend; use a runtime build that supports this GPU generation.","source":"https://cs.pny.com.tw/en/upload/download_files/en_download_list_23h01_h9yegm9tak.pdf","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR6","configuration":"GeForce RTX 4060 Ti 16 GB, desktop/reference specification, 16 GB GDDR6","interconnect":"PCIe 4.0 x8; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"18 Gbit/s per pin × 128 bits ÷ 8 = 288 GB/s; do not use cache-adjusted effective bandwidth.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://cs.pny.com.tw/en/upload/download_files/en_download_list_23h01_h9yegm9tak.pdf"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":165,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 4060 Ti 16 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"18 Gbit/s per pin × 128 bits ÷ 8 = 288 GB/s; do not use cache-adjusted effective bandwidth."},{"id":"rtx-4070-ti-super","name":"GeForce RTX 4070 Ti SUPER","maker":"NVIDIA","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":672,"count":1,"summary":"16 GB GDDR6X dedicated memory.","caveat":"CUDA backend; use a runtime build that supports this GPU generation.","source":"https://www.pny.com/pny-geforce-rtx-4070-ti-super-16gb-verto-triple-fan-oc-edition","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR6X","configuration":"GeForce RTX 4070 Ti SUPER, desktop/reference specification, 16 GB GDDR6X","interconnect":"PCIe 4.0 x16; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.pny.com/pny-geforce-rtx-4070-ti-super-16gb-verto-triple-fan-oc-edition"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":285,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 4070 Ti SUPER","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-3090","name":"GeForce RTX 3090","maker":"NVIDIA","memoryGiB":24,"reserveGiB":1,"type":"GPU","bandwidthGBs":936,"count":1,"summary":"24 GB GDDR6X dedicated memory.","caveat":"Desktop 24 GB card. NVLink requires a compatible bridge and runtime; without it, communication uses PCIe. Used-card power and cooling need to match the host.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3090-3090ti/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":24,"memoryType":"GDDR6X","configuration":"GeForce RTX 3090, desktop/reference specification, 24 GB GDDR6X","interconnect":"PCIe 4.0 x16; optional two-card NVLink bridge","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Reference GDDR6X data rate 19.5 Gbit/s × 384 bits ÷ 8 = 936 GB/s.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3090-3090ti/"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/content/PDF/nvidia-ampere-ga-102-gpu-architecture-whitepaper-v2.pdf"}],"powerWatts":350,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 3090","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Reference GDDR6X data rate 19.5 Gbit/s × 384 bits ÷ 8 = 936 GB/s."},{"id":"rtx-3090-ti","name":"GeForce RTX 3090 Ti","maker":"NVIDIA","memoryGiB":24,"reserveGiB":1,"type":"GPU","bandwidthGBs":1008,"count":1,"summary":"24 GB GDDR6X dedicated memory.","caveat":"CUDA backend; use a runtime build that supports this GPU generation.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3090-3090ti/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":24,"memoryType":"GDDR6X","configuration":"GeForce RTX 3090 Ti, desktop/reference specification, 24 GB GDDR6X","interconnect":"PCIe 4.0 x16; optional two-card NVLink bridge","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3090-3090ti/"},{"title":"Manufacturer source 2","url":"https://images.nvidia.cn/aem-dam/Solutions/geforce/ada/nvidia-ada-gpu-architecture.pdf"}],"powerWatts":450,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 3090 Ti","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-4090","name":"GeForce RTX 4090","maker":"NVIDIA","memoryGiB":24,"reserveGiB":1,"type":"GPU","bandwidthGBs":1008,"count":1,"summary":"24 GB GDDR6X dedicated memory.","caveat":"CUDA backend; use a runtime build that supports this GPU generation.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/40-series/rtx-4090/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":24,"memoryType":"GDDR6X","configuration":"GeForce RTX 4090, desktop/reference specification, 24 GB GDDR6X","interconnect":"PCIe 4.0 x16; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/40-series/rtx-4090/"},{"title":"Manufacturer source 2","url":"https://images.nvidia.cn/aem-dam/Solutions/geforce/ada/nvidia-ada-gpu-architecture.pdf"}],"powerWatts":450,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 4090","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-4060","name":"GeForce RTX 4060","maker":"NVIDIA","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":272,"count":1,"summary":"8 GB GDDR6 dedicated memory.","caveat":"CUDA backend; use a runtime build that supports this GPU generation.","source":"https://www.pny.com/pny-geforce-rtx-4060-8gb-verto-dual-fan","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR6","configuration":"GeForce RTX 4060, desktop/reference specification, 8 GB GDDR6","interconnect":"PCIe 4.0 x8; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.pny.com/pny-geforce-rtx-4060-8gb-verto-dual-fan"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":115,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 4060","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-4060-ti-8gb","name":"GeForce RTX 4060 Ti 8 GB","maker":"NVIDIA","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":288,"count":1,"summary":"8 GB GDDR6 dedicated memory.","caveat":"CUDA backend; use a runtime build that supports this GPU generation.","source":"https://www.pny.com/geforce-rtx-4060-ti-8gb-verto-dual-fan-oc","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR6","configuration":"GeForce RTX 4060 Ti 8 GB, desktop/reference specification, 8 GB GDDR6","interconnect":"PCIe 4.0 x8; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.pny.com/geforce-rtx-4060-ti-8gb-verto-dual-fan-oc"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":160,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 4060 Ti 8 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-4070-gddr6x","name":"GeForce RTX 4070 GDDR6X","maker":"NVIDIA","memoryGiB":12,"reserveGiB":1,"type":"GPU","bandwidthGBs":504,"count":1,"summary":"12 GB GDDR6X dedicated memory.","caveat":"CUDA backend; use a runtime build that supports this GPU generation.","source":"https://www.pny.com/geforce-rtx-4070-12gb-verto-dual-fan","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":12,"memoryType":"GDDR6X","configuration":"GeForce RTX 4070 GDDR6X, desktop/reference specification, 12 GB GDDR6X","interconnect":"PCIe 4.0 x16; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.pny.com/geforce-rtx-4070-12gb-verto-dual-fan"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":200,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 4070 GDDR6X","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-4070-super","name":"GeForce RTX 4070 SUPER","maker":"NVIDIA","memoryGiB":12,"reserveGiB":1,"type":"GPU","bandwidthGBs":504,"count":1,"summary":"12 GB GDDR6X dedicated memory.","caveat":"CUDA backend; use a runtime build that supports this GPU generation.","source":"https://www.pny.com/PNY-GeForce-RTX-4070-Super-12-GB-VERTO-OC-Edition-Dual-Fan","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":12,"memoryType":"GDDR6X","configuration":"GeForce RTX 4070 SUPER, desktop/reference specification, 12 GB GDDR6X","interconnect":"PCIe 4.0 x16; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.pny.com/PNY-GeForce-RTX-4070-Super-12-GB-VERTO-OC-Edition-Dual-Fan"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":220,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 4070 SUPER","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-4070-ti","name":"GeForce RTX 4070 Ti","maker":"NVIDIA","memoryGiB":12,"reserveGiB":1,"type":"GPU","bandwidthGBs":504,"count":1,"summary":"12 GB GDDR6X dedicated memory.","caveat":"CUDA backend; use a runtime build that supports this GPU generation.","source":"https://www.pny.com/pny-geforce-rtx-4070-ti-12gb-verto-triple-fan","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":12,"memoryType":"GDDR6X","configuration":"GeForce RTX 4070 Ti, desktop/reference specification, 12 GB GDDR6X","interconnect":"PCIe 4.0 x16; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.pny.com/pny-geforce-rtx-4070-ti-12gb-verto-triple-fan"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":285,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 4070 Ti","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-4080","name":"GeForce RTX 4080","maker":"NVIDIA","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":716.8,"count":1,"summary":"16 GB GDDR6X dedicated memory.","caveat":"CUDA backend; use a runtime build that supports this GPU generation.","source":"https://www.pny.com/pny-geforce-rtx-4080-16gb-verto-triple-fan-led","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR6X","configuration":"GeForce RTX 4080, desktop/reference specification, 16 GB GDDR6X","interconnect":"PCIe 4.0 x16; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.pny.com/pny-geforce-rtx-4080-16gb-verto-triple-fan-led"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":320,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 4080","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-4080-super","name":"GeForce RTX 4080 SUPER","maker":"NVIDIA","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":736,"count":1,"summary":"16 GB GDDR6X dedicated memory.","caveat":"CUDA backend; use a runtime build that supports this GPU generation.","source":"https://www.pny.com/pny-geforce-rtx-4080-super-16-gb-xlr8-gaming-verto-epic-x-rgb-triple-fan-oc","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR6X","configuration":"GeForce RTX 4080 SUPER, desktop/reference specification, 16 GB GDDR6X","interconnect":"PCIe 4.0 x16; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.pny.com/pny-geforce-rtx-4080-super-16-gb-xlr8-gaming-verto-epic-x-rgb-triple-fan-oc"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":320,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"GeForce RTX 4080 SUPER","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-5090-laptop-24gb","name":"GeForce RTX 5090 Laptop GPU 24 GB","maker":"NVIDIA","memoryGiB":24,"reserveGiB":1,"type":"GPU","bandwidthGBs":896,"count":1,"summary":"24 GB GDDR7 dedicated memory.","caveat":"Laptop GPU, with different specifications from the desktop card of the same number. GPU subsystem power is configurable by the laptop maker; Dynamic Boost, thermals, firmware and AC/battery mode change sustained performance. Check the exact laptop specification.","source":"https://www.nvidia.com/en-us/geforce/laptops/compare/","reviewedAt":"2026-09-06","category":"Laptop GPU","memoryPerDeviceGiB":24,"memoryType":"GDDR7","configuration":"GeForce RTX 5090 Laptop GPU 24 GB, 256-bit memory interface; OEM-configured laptop cooling and GPU power limit","interconnect":"Laptop-internal PCIe 5.0; lane wiring varies by OEM; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"NVIDIA-published reference peak bandwidth; memory power states and OEM configuration affect sustained bandwidth.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/laptops/compare/"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/laptops/50-series/"}],"powerRangeWatts":[95,150],"powerBasis":"NVIDIA GPU Subsystem Power range; OEM Dynamic Boost may add power beyond this base range. Not a measured inference draw or whole-laptop maximum.","family":"GeForce RTX 5090 Laptop GPU 24 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"NVIDIA-published reference peak bandwidth; memory power states and OEM configuration affect sustained bandwidth."},{"id":"rtx-5080-laptop-16gb","name":"GeForce RTX 5080 Laptop GPU 16 GB","maker":"NVIDIA","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":896,"count":1,"summary":"16 GB GDDR7 dedicated memory.","caveat":"Laptop GPU, with different specifications from the desktop card of the same number. GPU subsystem power is configurable by the laptop maker; Dynamic Boost, thermals, firmware and AC/battery mode change sustained performance. Check the exact laptop specification.","source":"https://www.nvidia.com/en-us/geforce/laptops/compare/","reviewedAt":"2026-09-06","category":"Laptop GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR7","configuration":"GeForce RTX 5080 Laptop GPU 16 GB, 256-bit memory interface; OEM-configured laptop cooling and GPU power limit","interconnect":"Laptop-internal PCIe 5.0; lane wiring varies by OEM; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"NVIDIA-published reference peak bandwidth; memory power states and OEM configuration affect sustained bandwidth.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/laptops/compare/"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/laptops/50-series/"}],"powerRangeWatts":[80,150],"powerBasis":"NVIDIA GPU Subsystem Power range; OEM Dynamic Boost may add power beyond this base range. Not a measured inference draw or whole-laptop maximum.","family":"GeForce RTX 5080 Laptop GPU 16 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"NVIDIA-published reference peak bandwidth; memory power states and OEM configuration affect sustained bandwidth."},{"id":"rtx-5070-ti-laptop-12gb","name":"GeForce RTX 5070 Ti Laptop GPU 12 GB","maker":"NVIDIA","memoryGiB":12,"reserveGiB":1,"type":"GPU","bandwidthGBs":672,"count":1,"summary":"12 GB GDDR7 dedicated memory.","caveat":"Laptop GPU, with different specifications from the desktop card of the same number. GPU subsystem power is configurable by the laptop maker; Dynamic Boost, thermals, firmware and AC/battery mode change sustained performance. Check the exact laptop specification.","source":"https://www.nvidia.com/en-us/geforce/laptops/compare/","reviewedAt":"2026-09-06","category":"Laptop GPU","memoryPerDeviceGiB":12,"memoryType":"GDDR7","configuration":"GeForce RTX 5070 Ti Laptop GPU 12 GB, 192-bit memory interface; OEM-configured laptop cooling and GPU power limit","interconnect":"Laptop-internal PCIe 5.0; lane wiring varies by OEM; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"NVIDIA-published reference peak bandwidth; memory power states and OEM configuration affect sustained bandwidth.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/laptops/compare/"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/laptops/50-series/"}],"powerRangeWatts":[60,115],"powerBasis":"NVIDIA GPU Subsystem Power range; OEM Dynamic Boost may add power beyond this base range. Not a measured inference draw or whole-laptop maximum.","family":"GeForce RTX 5070 Ti Laptop GPU 12 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"NVIDIA-published reference peak bandwidth; memory power states and OEM configuration affect sustained bandwidth."},{"id":"rtx-5070-laptop-12gb","name":"GeForce RTX 5070 Laptop GPU 12 GB","maker":"NVIDIA","memoryGiB":12,"reserveGiB":1,"type":"GPU","bandwidthGBs":384,"count":1,"summary":"12 GB GDDR7 dedicated memory.","caveat":"Laptop GPU, with different specifications from the desktop card of the same number. GPU subsystem power is configurable by the laptop maker; Dynamic Boost, thermals, firmware and AC/battery mode change sustained performance. Check the exact laptop specification.","source":"https://www.nvidia.com/en-us/geforce/laptops/compare/","reviewedAt":"2026-09-06","category":"Laptop GPU","memoryPerDeviceGiB":12,"memoryType":"GDDR7","configuration":"GeForce RTX 5070 Laptop GPU 12 GB, 128-bit memory interface; OEM-configured laptop cooling and GPU power limit","interconnect":"Laptop-internal PCIe 5.0; lane wiring varies by OEM; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"NVIDIA-published reference peak bandwidth; memory power states and OEM configuration affect sustained bandwidth.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/laptops/compare/"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/laptops/50-series/"},{"title":"Manufacturer source 3","url":"https://www.nvidia.com/en-us/geforce/news/conan-exiles-enhanced-geforce-game-ready-driver/"}],"powerRangeWatts":[50,100],"powerBasis":"NVIDIA GPU Subsystem Power range; OEM Dynamic Boost may add power beyond this base range. Not a measured inference draw or whole-laptop maximum.","family":"GeForce RTX 5070 Laptop GPU 12 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"NVIDIA-published reference peak bandwidth; memory power states and OEM configuration affect sustained bandwidth."},{"id":"rtx-5070-laptop-8gb","name":"GeForce RTX 5070 Laptop GPU 8 GB","maker":"NVIDIA","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":384,"count":1,"summary":"8 GB GDDR7 dedicated memory.","caveat":"Laptop GPU, with different specifications from the desktop card of the same number. GPU subsystem power is configurable by the laptop maker; Dynamic Boost, thermals, firmware and AC/battery mode change sustained performance. Check the exact laptop specification.","source":"https://www.nvidia.com/en-us/geforce/laptops/compare/","reviewedAt":"2026-09-06","category":"Laptop GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR7","configuration":"GeForce RTX 5070 Laptop GPU 8 GB, 128-bit memory interface; OEM-configured laptop cooling and GPU power limit","interconnect":"Laptop-internal PCIe 5.0; lane wiring varies by OEM; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"NVIDIA-published reference peak bandwidth; memory power states and OEM configuration affect sustained bandwidth.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/laptops/compare/"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/laptops/50-series/"}],"powerRangeWatts":[50,100],"powerBasis":"NVIDIA GPU Subsystem Power range; OEM Dynamic Boost may add power beyond this base range. Not a measured inference draw or whole-laptop maximum.","family":"GeForce RTX 5070 Laptop GPU 8 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"NVIDIA-published reference peak bandwidth; memory power states and OEM configuration affect sustained bandwidth."},{"id":"rtx-5060-laptop-8gb","name":"GeForce RTX 5060 Laptop GPU 8 GB","maker":"NVIDIA","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":384,"count":1,"summary":"8 GB GDDR7 dedicated memory.","caveat":"Laptop GPU, with different specifications from the desktop card of the same number. GPU subsystem power is configurable by the laptop maker; Dynamic Boost, thermals, firmware and AC/battery mode change sustained performance. Check the exact laptop specification.","source":"https://www.nvidia.com/en-us/geforce/laptops/compare/","reviewedAt":"2026-09-06","category":"Laptop GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR7","configuration":"GeForce RTX 5060 Laptop GPU 8 GB, 128-bit memory interface; OEM-configured laptop cooling and GPU power limit","interconnect":"Laptop-internal PCIe 5.0; lane wiring varies by OEM; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"NVIDIA-published reference peak bandwidth; memory power states and OEM configuration affect sustained bandwidth.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/laptops/compare/"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/laptops/50-series/"}],"powerRangeWatts":[45,100],"powerBasis":"NVIDIA GPU Subsystem Power range; OEM Dynamic Boost may add power beyond this base range. Not a measured inference draw or whole-laptop maximum.","family":"GeForce RTX 5060 Laptop GPU 8 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"NVIDIA-published reference peak bandwidth; memory power states and OEM configuration affect sustained bandwidth."},{"id":"rtx-5050-laptop-8gb","name":"GeForce RTX 5050 Laptop GPU 8 GB","maker":"NVIDIA","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":384,"count":1,"summary":"8 GB GDDR7 dedicated memory.","caveat":"Laptop GPU, with different specifications from the desktop card of the same number. GPU subsystem power is configurable by the laptop maker; Dynamic Boost, thermals, firmware and AC/battery mode change sustained performance. Check the exact laptop specification.","source":"https://www.nvidia.com/en-us/geforce/laptops/compare/","reviewedAt":"2026-09-06","category":"Laptop GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR7","configuration":"GeForce RTX 5050 Laptop GPU 8 GB, 128-bit memory interface; OEM-configured laptop cooling and GPU power limit","interconnect":"Laptop-internal PCIe 5.0; lane wiring varies by OEM; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"NVIDIA-published reference peak bandwidth; memory power states and OEM configuration affect sustained bandwidth.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/laptops/compare/"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/laptops/50-series/"}],"powerRangeWatts":[35,100],"powerBasis":"NVIDIA GPU Subsystem Power range; OEM Dynamic Boost may add power beyond this base range. Not a measured inference draw or whole-laptop maximum.","family":"GeForce RTX 5050 Laptop GPU 8 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"NVIDIA-published reference peak bandwidth; memory power states and OEM configuration affect sustained bandwidth."},{"id":"rtx-4090-laptop-16gb","name":"GeForce RTX 4090 Laptop GPU 16 GB","maker":"NVIDIA","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":null,"count":1,"summary":"16 GB GDDR6 dedicated memory.","caveat":"Laptop GPU, with different specifications from the desktop card of the same number. GPU subsystem power is configurable by the laptop maker; Dynamic Boost, thermals, firmware and AC/battery mode change sustained performance. Check the exact laptop specification.","source":"https://www.nvidia.com/en-us/geforce/laptops/compare/","reviewedAt":"2026-09-06","category":"Laptop GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR6","configuration":"GeForce RTX 4090 Laptop GPU 16 GB, 256-bit memory interface; OEM-configured laptop cooling and GPU power limit","interconnect":"Laptop-internal PCIe 4.0; lane wiring varies by OEM; no NVLink","availability":"Available","bandwidthKind":"Unknown for the exact OEM configuration","bandwidthNote":"The cited manufacturer table does not publish memory data rate or bandwidth. No bandwidth-derived speed estimate is supplied; obtain the exact laptop memory clock and bus width.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/laptops/compare/"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/laptops/40-series/"}],"powerRangeWatts":[80,150],"powerBasis":"NVIDIA GPU Subsystem Power range; OEM Dynamic Boost may add power beyond this base range. Not a measured inference draw or whole-laptop maximum.","family":"GeForce RTX 4090 Laptop GPU 16 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"The cited manufacturer table does not publish memory data rate or bandwidth. No bandwidth-derived speed estimate is supplied; obtain the exact laptop memory clock and bus width."},{"id":"rtx-4080-laptop-12gb","name":"GeForce RTX 4080 Laptop GPU 12 GB","maker":"NVIDIA","memoryGiB":12,"reserveGiB":1,"type":"GPU","bandwidthGBs":null,"count":1,"summary":"12 GB GDDR6 dedicated memory.","caveat":"Laptop GPU, with different specifications from the desktop card of the same number. GPU subsystem power is configurable by the laptop maker; Dynamic Boost, thermals, firmware and AC/battery mode change sustained performance. Check the exact laptop specification.","source":"https://www.nvidia.com/en-us/geforce/laptops/compare/","reviewedAt":"2026-09-06","category":"Laptop GPU","memoryPerDeviceGiB":12,"memoryType":"GDDR6","configuration":"GeForce RTX 4080 Laptop GPU 12 GB, 192-bit memory interface; OEM-configured laptop cooling and GPU power limit","interconnect":"Laptop-internal PCIe 4.0; lane wiring varies by OEM; no NVLink","availability":"Available","bandwidthKind":"Unknown for the exact OEM configuration","bandwidthNote":"The cited manufacturer table does not publish memory data rate or bandwidth. No bandwidth-derived speed estimate is supplied; obtain the exact laptop memory clock and bus width.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/laptops/compare/"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/laptops/40-series/"}],"powerRangeWatts":[60,150],"powerBasis":"NVIDIA GPU Subsystem Power range; OEM Dynamic Boost may add power beyond this base range. Not a measured inference draw or whole-laptop maximum.","family":"GeForce RTX 4080 Laptop GPU 12 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"The cited manufacturer table does not publish memory data rate or bandwidth. No bandwidth-derived speed estimate is supplied; obtain the exact laptop memory clock and bus width."},{"id":"rtx-4070-laptop-8gb","name":"GeForce RTX 4070 Laptop GPU 8 GB","maker":"NVIDIA","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":null,"count":1,"summary":"8 GB GDDR6 dedicated memory.","caveat":"Laptop GPU, with different specifications from the desktop card of the same number. GPU subsystem power is configurable by the laptop maker; Dynamic Boost, thermals, firmware and AC/battery mode change sustained performance. Check the exact laptop specification.","source":"https://www.nvidia.com/en-us/geforce/laptops/compare/","reviewedAt":"2026-09-06","category":"Laptop GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR6","configuration":"GeForce RTX 4070 Laptop GPU 8 GB, 128-bit memory interface; OEM-configured laptop cooling and GPU power limit","interconnect":"Laptop-internal PCIe 4.0; lane wiring varies by OEM; no NVLink","availability":"Available","bandwidthKind":"Unknown for the exact OEM configuration","bandwidthNote":"The cited manufacturer table does not publish memory data rate or bandwidth. No bandwidth-derived speed estimate is supplied; obtain the exact laptop memory clock and bus width.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/laptops/compare/"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/laptops/40-series/"}],"powerRangeWatts":[35,115],"powerBasis":"NVIDIA GPU Subsystem Power range; OEM Dynamic Boost may add power beyond this base range. Not a measured inference draw or whole-laptop maximum.","family":"GeForce RTX 4070 Laptop GPU 8 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"The cited manufacturer table does not publish memory data rate or bandwidth. No bandwidth-derived speed estimate is supplied; obtain the exact laptop memory clock and bus width."},{"id":"rtx-4060-laptop-8gb","name":"GeForce RTX 4060 Laptop GPU 8 GB","maker":"NVIDIA","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":null,"count":1,"summary":"8 GB GDDR6 dedicated memory.","caveat":"Laptop GPU, with different specifications from the desktop card of the same number. GPU subsystem power is configurable by the laptop maker; Dynamic Boost, thermals, firmware and AC/battery mode change sustained performance. Check the exact laptop specification.","source":"https://www.nvidia.com/en-us/geforce/laptops/compare/","reviewedAt":"2026-09-06","category":"Laptop GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR6","configuration":"GeForce RTX 4060 Laptop GPU 8 GB, 128-bit memory interface; OEM-configured laptop cooling and GPU power limit","interconnect":"Laptop-internal PCIe 4.0; lane wiring varies by OEM; no NVLink","availability":"Available","bandwidthKind":"Unknown for the exact OEM configuration","bandwidthNote":"The cited manufacturer table does not publish memory data rate or bandwidth. No bandwidth-derived speed estimate is supplied; obtain the exact laptop memory clock and bus width.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/laptops/compare/"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/laptops/40-series/"}],"powerRangeWatts":[35,115],"powerBasis":"NVIDIA GPU Subsystem Power range; OEM Dynamic Boost may add power beyond this base range. Not a measured inference draw or whole-laptop maximum.","family":"GeForce RTX 4060 Laptop GPU 8 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"The cited manufacturer table does not publish memory data rate or bandwidth. No bandwidth-derived speed estimate is supplied; obtain the exact laptop memory clock and bus width."},{"id":"rtx-4050-laptop-6gb","name":"GeForce RTX 4050 Laptop GPU 6 GB","maker":"NVIDIA","memoryGiB":6,"reserveGiB":1,"type":"GPU","bandwidthGBs":null,"count":1,"summary":"6 GB GDDR6 dedicated memory.","caveat":"Laptop GPU, with different specifications from the desktop card of the same number. GPU subsystem power is configurable by the laptop maker; Dynamic Boost, thermals, firmware and AC/battery mode change sustained performance. Check the exact laptop specification.","source":"https://www.nvidia.com/en-us/geforce/laptops/compare/","reviewedAt":"2026-09-06","category":"Laptop GPU","memoryPerDeviceGiB":6,"memoryType":"GDDR6","configuration":"GeForce RTX 4050 Laptop GPU 6 GB, 96-bit memory interface; OEM-configured laptop cooling and GPU power limit","interconnect":"Laptop-internal PCIe 4.0; lane wiring varies by OEM; no NVLink","availability":"Available","bandwidthKind":"Unknown for the exact OEM configuration","bandwidthNote":"The cited manufacturer table does not publish memory data rate or bandwidth. No bandwidth-derived speed estimate is supplied; obtain the exact laptop memory clock and bus width.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/laptops/compare/"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/en-us/geforce/laptops/40-series/"}],"powerRangeWatts":[35,115],"powerBasis":"NVIDIA GPU Subsystem Power range; OEM Dynamic Boost may add power beyond this base range. Not a measured inference draw or whole-laptop maximum.","family":"GeForce RTX 4050 Laptop GPU 6 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"The cited manufacturer table does not publish memory data rate or bandwidth. No bandwidth-derived speed estimate is supplied; obtain the exact laptop memory clock and bus width."},{"id":"rtx-pro-2000-blackwell","name":"RTX PRO 2000 Blackwell","maker":"NVIDIA","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":288,"count":1,"summary":"16 GB GDDR7 ECC dedicated memory.","caveat":"Exact desktop edition shown. Blackwell needs a compatible CUDA runtime. ECC and driver allocations can reduce free capacity; each card remains a separate memory pool.","source":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-2000/","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR7 ECC","configuration":"RTX PRO 2000 Blackwell, desktop/reference specification, 16 GB GDDR7 ECC","interconnect":"PCIe 5.0; low profile, dual slot","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-2000/"}],"powerWatts":70,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"RTX PRO 2000 Blackwell","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-pro-4000-sff-blackwell","name":"RTX PRO 4000 Blackwell SFF","maker":"NVIDIA","memoryGiB":24,"reserveGiB":1,"type":"GPU","bandwidthGBs":432,"count":1,"summary":"24 GB GDDR7 ECC dedicated memory.","caveat":"Exact desktop edition shown. Blackwell needs a compatible CUDA runtime. ECC and driver allocations can reduce free capacity; each card remains a separate memory pool.","source":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-4000-sff/","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":24,"memoryType":"GDDR7 ECC","configuration":"RTX PRO 4000 Blackwell SFF, desktop/reference specification, 24 GB GDDR7 ECC","interconnect":"PCIe 5.0 x8; low profile, dual slot","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-4000-sff/"}],"powerWatts":70,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"RTX PRO 4000 Blackwell SFF","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-pro-4000-blackwell","name":"RTX PRO 4000 Blackwell","maker":"NVIDIA","memoryGiB":24,"reserveGiB":1,"type":"GPU","bandwidthGBs":672,"count":1,"summary":"24 GB GDDR7 ECC dedicated memory.","caveat":"Exact desktop edition shown. Blackwell needs a compatible CUDA runtime. ECC and driver allocations can reduce free capacity; each card remains a separate memory pool.","source":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-4000/","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":24,"memoryType":"GDDR7 ECC","configuration":"RTX PRO 4000 Blackwell, desktop/reference specification, 24 GB GDDR7 ECC","interconnect":"PCIe 5.0 x16; full height, single slot","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-4000/"}],"powerWatts":145,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"RTX PRO 4000 Blackwell","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-pro-4500-blackwell","name":"RTX PRO 4500 Blackwell Workstation","maker":"NVIDIA","memoryGiB":32,"reserveGiB":1,"type":"GPU","bandwidthGBs":896,"count":1,"summary":"32 GB GDDR7 ECC dedicated memory.","caveat":"Exact desktop edition shown. Blackwell needs a compatible CUDA runtime. ECC and driver allocations can reduce free capacity; each card remains a separate memory pool.","source":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-4500/","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":32,"memoryType":"GDDR7 ECC","configuration":"RTX PRO 4500 Blackwell Workstation, desktop/reference specification, 32 GB GDDR7 ECC","interconnect":"PCIe 5.0 x16; full height, dual slot","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-4500/"}],"powerWatts":200,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"RTX PRO 4500 Blackwell Workstation","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-pro-5000-blackwell-48gb","name":"RTX PRO 5000 Blackwell 48 GB","maker":"NVIDIA","memoryGiB":48,"reserveGiB":1,"type":"GPU","bandwidthGBs":1344,"count":1,"summary":"48 GB GDDR7 ECC dedicated memory.","caveat":"Exact desktop edition shown. Blackwell needs a compatible CUDA runtime. ECC and driver allocations can reduce free capacity; each card remains a separate memory pool.","source":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-5000/","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":48,"memoryType":"GDDR7 ECC","configuration":"RTX PRO 5000 Blackwell 48 GB, desktop/reference specification, 48 GB GDDR7 ECC","interconnect":"PCIe 5.0 x16; full height, dual slot","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-5000/"}],"powerWatts":300,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"RTX PRO 5000 Blackwell 48 GB","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-pro-5000-blackwell-72gb","name":"RTX PRO 5000 Blackwell 72 GB","maker":"NVIDIA","memoryGiB":72,"reserveGiB":1,"type":"GPU","bandwidthGBs":1344,"count":1,"summary":"72 GB GDDR7 ECC dedicated memory.","caveat":"Exact desktop edition shown. Blackwell needs a compatible CUDA runtime. ECC and driver allocations can reduce free capacity; each card remains a separate memory pool.","source":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-5000/","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":72,"memoryType":"GDDR7 ECC","configuration":"RTX PRO 5000 Blackwell 72 GB, desktop/reference specification, 72 GB GDDR7 ECC","interconnect":"PCIe 5.0 x16; full height, dual slot","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-5000/"}],"powerWatts":300,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"RTX PRO 5000 Blackwell 72 GB","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-pro-6000-blackwell","name":"RTX PRO 6000 Blackwell Workstation","maker":"NVIDIA","memoryGiB":96,"reserveGiB":1,"type":"GPU","bandwidthGBs":1792,"count":1,"summary":"96 GB GDDR7 ECC dedicated memory.","caveat":"Exact desktop edition shown. Blackwell needs a compatible CUDA runtime. ECC and driver allocations can reduce free capacity; each card remains a separate memory pool.","source":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-6000/","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":96,"memoryType":"GDDR7 ECC","configuration":"RTX PRO 6000 Blackwell Workstation, desktop/reference specification, 96 GB GDDR7 ECC","interconnect":"PCIe 5.0 x16","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-6000/"}],"powerWatts":600,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"RTX PRO 6000 Blackwell Workstation","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-pro-6000-blackwell-max-q","name":"RTX PRO 6000 Blackwell Max-Q","maker":"NVIDIA","memoryGiB":96,"reserveGiB":1,"type":"GPU","bandwidthGBs":1792,"count":1,"summary":"96 GB GDDR7 ECC dedicated memory.","caveat":"Exact desktop edition shown. Blackwell needs a compatible CUDA runtime. ECC and driver allocations can reduce free capacity; each card remains a separate memory pool.","source":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-6000-max-q/","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":96,"memoryType":"GDDR7 ECC","configuration":"RTX PRO 6000 Blackwell Max-Q, desktop/reference specification, 96 GB GDDR7 ECC","interconnect":"PCIe 5.0 x16; full height, dual slot","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-6000-max-q/"}],"powerWatts":300,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"RTX PRO 6000 Blackwell Max-Q","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-a6000","name":"RTX A6000","maker":"NVIDIA","memoryGiB":48,"reserveGiB":1,"type":"GPU","bandwidthGBs":768,"count":1,"summary":"48 GB GDDR6 ECC dedicated memory.","caveat":"CUDA backend; use a runtime build that supports this GPU generation.","source":"https://www.nvidia.com/content/dam/en-zz/Solutions/design-visualization/quadro-product-literature/proviz-print-nvidia-rtx-a6000-datasheet-us-nvidia-1454980-r9-web%20%281%29.pdf","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":48,"memoryType":"GDDR6 ECC","configuration":"RTX A6000, desktop/reference specification, 48 GB GDDR6 ECC","interconnect":"PCIe 4.0 x16; optional two-card NVLink, 112.5 GB/s bidirectional","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/content/dam/en-zz/Solutions/design-visualization/quadro-product-literature/proviz-print-nvidia-rtx-a6000-datasheet-us-nvidia-1454980-r9-web%20%281%29.pdf"}],"powerWatts":300,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"RTX A6000","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-6000-ada","name":"RTX 6000 Ada Generation","maker":"NVIDIA","memoryGiB":48,"reserveGiB":1,"type":"GPU","bandwidthGBs":960,"count":1,"summary":"48 GB GDDR6 ECC dedicated memory.","caveat":"CUDA backend; use a runtime build that supports this GPU generation.","source":"https://www.nvidia.com/content/dam/en-zz/Solutions/design-visualization/rtx-6000/proviz-print-rtx6000-datasheet-web-2504660.pdf","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":48,"memoryType":"GDDR6 ECC","configuration":"RTX 6000 Ada Generation, desktop/reference specification, 48 GB GDDR6 ECC","interconnect":"PCIe 4.0 x16; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/content/dam/en-zz/Solutions/design-visualization/rtx-6000/proviz-print-rtx6000-datasheet-web-2504660.pdf"}],"powerWatts":300,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"RTX 6000 Ada Generation","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rtx-5000-ada","name":"RTX 5000 Ada Generation","maker":"NVIDIA","memoryGiB":32,"reserveGiB":1,"type":"GPU","bandwidthGBs":576,"count":1,"summary":"32 GB GDDR6 ECC dedicated memory.","caveat":"CUDA backend; use a runtime build that supports this GPU generation.","source":"https://www.pny.com/nvidia-rtx-5000-ada","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":32,"memoryType":"GDDR6 ECC","configuration":"RTX 5000 Ada Generation, desktop/reference specification, 32 GB GDDR6 ECC","interconnect":"PCIe 4.0 x16; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.pny.com/nvidia-rtx-5000-ada"}],"family":"RTX 5000 Ada Generation","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"a100-80gb-pcie","name":"A100 80 GB PCIe","maker":"NVIDIA","memoryGiB":80,"reserveGiB":2,"type":"GPU","bandwidthGBs":1935,"count":1,"summary":"80 GB HBM2e ECC dedicated memory.","caveat":"Requires a compatible server, cooling and power delivery. Capacity is for one full GPU; a MIG/vGPU partition exposes less. Runtime support and actual free memory must be checked.","source":"https://www.nvidia.com/en-us/data-center/a100/","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":80,"memoryType":"HBM2e ECC","configuration":"A100 80 GB PCIe, server card/module, 80 GB HBM2e ECC","interconnect":"PCIe 4.0 x16; NVLink topology depends on server","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/data-center/a100/"}],"powerWatts":300,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"A100 80 GB PCIe","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"a100-80gb-sxm","name":"A100 80 GB SXM","maker":"NVIDIA","memoryGiB":80,"reserveGiB":2,"type":"GPU","bandwidthGBs":2039,"count":1,"summary":"80 GB HBM2e ECC dedicated memory.","caveat":"Requires a compatible server, cooling and power delivery. Capacity is for one full GPU; a MIG/vGPU partition exposes less. Runtime support and actual free memory must be checked.","source":"https://www.nvidia.com/en-us/data-center/a100/","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":80,"memoryType":"HBM2e ECC","configuration":"A100 80 GB SXM, server card/module, 80 GB HBM2e ECC","interconnect":"SXM module; NVLink/NVSwitch server baseboard","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/data-center/a100/"}],"powerWatts":400,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"A100 80 GB SXM","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"h100-sxm","name":"H100 80 GB SXM","maker":"NVIDIA","memoryGiB":80,"reserveGiB":2,"type":"GPU","bandwidthGBs":3350,"count":1,"summary":"80 GB HBM3 ECC dedicated memory.","caveat":"Requires a compatible server, cooling and power delivery. Capacity is for one full GPU; a MIG/vGPU partition exposes less. Runtime support and actual free memory must be checked.","source":"https://www.nvidia.com/en-us/data-center/h100/","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":80,"memoryType":"HBM3 ECC","configuration":"H100 80 GB SXM, server card/module, 80 GB HBM3 ECC","interconnect":"SXM module; NVLink up to 900 GB/s bidirectional","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/data-center/h100/"}],"powerWatts":700,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"H100 80 GB SXM","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"h100-nvl-94gb","name":"H100 NVL 94 GB PCIe","maker":"NVIDIA","memoryGiB":94,"reserveGiB":2,"type":"GPU","bandwidthGBs":3900,"count":1,"summary":"94 GB HBM3 ECC dedicated memory.","caveat":"Requires a compatible server, cooling and power delivery. Capacity is for one full GPU; a MIG/vGPU partition exposes less. Runtime support and actual free memory must be checked.","source":"https://www.nvidia.com/en-us/data-center/h100/","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":94,"memoryType":"HBM3 ECC","configuration":"H100 NVL 94 GB PCIe, server card/module, 94 GB HBM3 ECC","interconnect":"PCIe 5.0 x16; NVLink bridge up to 600 GB/s bidirectional","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/data-center/h100/"}],"powerWatts":400,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"H100 NVL 94 GB PCIe","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"h200-sxm","name":"H200 141 GB SXM","maker":"NVIDIA","memoryGiB":141,"reserveGiB":2,"type":"GPU","bandwidthGBs":4800,"count":1,"summary":"141 GB HBM3e ECC dedicated memory.","caveat":"Requires a compatible server, cooling and power delivery. Capacity is for one full GPU; a MIG/vGPU partition exposes less. Runtime support and actual free memory must be checked.","source":"https://www.nvidia.com/en-us/data-center/h200/","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":141,"memoryType":"HBM3e ECC","configuration":"H200 141 GB SXM, server card/module, 141 GB HBM3e ECC","interconnect":"SXM module; NVLink up to 900 GB/s bidirectional","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/data-center/h200/"}],"powerWatts":700,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"H200 141 GB SXM","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"h200-nvl","name":"H200 NVL 141 GB PCIe","maker":"NVIDIA","memoryGiB":141,"reserveGiB":2,"type":"GPU","bandwidthGBs":4800,"count":1,"summary":"141 GB HBM3e ECC dedicated memory.","caveat":"Requires a compatible server, cooling and power delivery. Capacity is for one full GPU; a MIG/vGPU partition exposes less. Runtime support and actual free memory must be checked.","source":"https://www.nvidia.com/en-us/data-center/h200/","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":141,"memoryType":"HBM3e ECC","configuration":"H200 NVL 141 GB PCIe, server card/module, 141 GB HBM3e ECC","interconnect":"PCIe 5.0 x16; 2/4-way NVLink bridge up to 900 GB/s per GPU","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/data-center/h200/"}],"powerWatts":600,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"H200 NVL 141 GB PCIe","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"l40s","name":"L40S","maker":"NVIDIA","memoryGiB":48,"reserveGiB":2,"type":"GPU","bandwidthGBs":864,"count":1,"summary":"48 GB GDDR6 ECC dedicated memory.","caveat":"Requires a compatible server, cooling and power delivery. Capacity is for one full GPU; a MIG/vGPU partition exposes less. Runtime support and actual free memory must be checked.","source":"https://www.nvidia.com/en-us/data-center/l40s/","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":48,"memoryType":"GDDR6 ECC","configuration":"L40S, server card/module, 48 GB GDDR6 ECC","interconnect":"PCIe 4.0 x16; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/data-center/l40s/"}],"powerWatts":350,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"L40S","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"b200-sxm","name":"B200 180 GB SXM","maker":"NVIDIA","memoryGiB":180,"reserveGiB":2,"type":"GPU","bandwidthGBs":8000,"count":1,"summary":"180 GB HBM3e ECC dedicated memory.","caveat":"Requires a compatible server, cooling and power delivery. Capacity is for one full GPU; a MIG/vGPU partition exposes less. Runtime support and actual free memory must be checked.","source":"https://docs.nvidia.com/enterprise-reference-architectures/hgx-ai-factory/latest/components.html","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":180,"memoryType":"HBM3e ECC","configuration":"One full B200 180 GB SXM GPU in a compatible HGX server; not a PCIe add-in card","interconnect":"SXM module in HGX; NVLink/NVSwitch up to 1800 GB/s per GPU","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://docs.nvidia.com/enterprise-reference-architectures/hgx-ai-factory/latest/components.html"}],"family":"B200 180 GB SXM","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"b300-sxm","name":"B300 288 GB SXM","maker":"NVIDIA","memoryGiB":288,"reserveGiB":2,"type":"GPU","bandwidthGBs":8000,"count":1,"summary":"288 GB HBM3e ECC dedicated memory.","caveat":"Requires a compatible server, cooling and power delivery. Capacity is for one full GPU; a MIG/vGPU partition exposes less. Runtime support and actual free memory must be checked.","source":"https://docs.nvidia.com/enterprise-reference-architectures/hgx-ai-factory/latest/components.html","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":288,"memoryType":"HBM3e ECC","configuration":"One full B300 288 GB SXM GPU in a compatible HGX server; not a PCIe add-in card","interconnect":"SXM module in HGX; NVLink/NVSwitch up to 1800 GB/s per GPU","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://docs.nvidia.com/enterprise-reference-architectures/hgx-ai-factory/latest/components.html"}],"family":"B300 288 GB SXM","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"a40","name":"A40","maker":"NVIDIA","memoryGiB":48,"reserveGiB":2,"type":"GPU","bandwidthGBs":696,"count":1,"summary":"48 GB GDDR6 ECC dedicated memory.","caveat":"Requires a compatible server, cooling and power delivery. Capacity is for one full GPU; a MIG/vGPU partition exposes less. Runtime support and actual free memory must be checked.","source":"https://images.nvidia.com/content/Solutions/data-center/a40/nvidia-a40-datasheet.pdf","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":48,"memoryType":"GDDR6 ECC","configuration":"A40, server card/module, 48 GB GDDR6 ECC","interconnect":"PCIe 4.0 x16; optional two-card NVLink 112.5 GB/s bidirectional","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://images.nvidia.com/content/Solutions/data-center/a40/nvidia-a40-datasheet.pdf"}],"powerWatts":300,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"A40","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"rx-6800","name":"Radeon RX 6800","maker":"AMD","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":512,"count":1,"summary":"16 GB GDDR6 dedicated memory.","caveat":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/6000-series/amd-radeon-rx-6800.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR6","configuration":"Radeon RX 6800, desktop/reference specification, 16 GB GDDR6","interconnect":"PCIe; no dedicated inter-GPU memory fabric","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/6000-series/amd-radeon-rx-6800.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":250,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Radeon RX 6800","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates."},{"id":"rx-6800-xt","name":"Radeon RX 6800 XT","maker":"AMD","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":512,"count":1,"summary":"16 GB GDDR6 dedicated memory.","caveat":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/6000-series/amd-radeon-rx-6800-xt.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR6","configuration":"Radeon RX 6800 XT, desktop/reference specification, 16 GB GDDR6","interconnect":"PCIe; no dedicated inter-GPU memory fabric","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/6000-series/amd-radeon-rx-6800-xt.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":300,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Radeon RX 6800 XT","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates."},{"id":"rx-6950-xt","name":"Radeon RX 6950 XT","maker":"AMD","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":576,"count":1,"summary":"16 GB GDDR6 dedicated memory.","caveat":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/6000-series/amd-radeon-rx-6950-xt.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR6","configuration":"Radeon RX 6950 XT, desktop/reference specification, 16 GB GDDR6","interconnect":"PCIe; no dedicated inter-GPU memory fabric","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/6000-series/amd-radeon-rx-6950-xt.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":335,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Radeon RX 6950 XT","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates."},{"id":"rx-7600-xt","name":"Radeon RX 7600 XT","maker":"AMD","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":288,"count":1,"summary":"16 GB GDDR6 dedicated memory.","caveat":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/7000-series/amd-radeon-rx-7600-xt.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR6","configuration":"Radeon RX 7600 XT, desktop/reference specification, 16 GB GDDR6","interconnect":"PCIe; no dedicated inter-GPU memory fabric","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/7000-series/amd-radeon-rx-7600-xt.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":190,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Radeon RX 7600 XT","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates."},{"id":"rx-7700-xt","name":"Radeon RX 7700 XT","maker":"AMD","memoryGiB":12,"reserveGiB":1,"type":"GPU","bandwidthGBs":432,"count":1,"summary":"12 GB GDDR6 dedicated memory.","caveat":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/7000-series/amd-radeon-rx-7700-xt.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":12,"memoryType":"GDDR6","configuration":"Radeon RX 7700 XT, desktop/reference specification, 12 GB GDDR6","interconnect":"PCIe; no dedicated inter-GPU memory fabric","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/7000-series/amd-radeon-rx-7700-xt.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":245,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Radeon RX 7700 XT","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates."},{"id":"rx-7800-xt","name":"Radeon RX 7800 XT","maker":"AMD","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":624,"count":1,"summary":"16 GB GDDR6 dedicated memory.","caveat":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/7000-series/amd-radeon-rx-7800-xt.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR6","configuration":"Radeon RX 7800 XT, desktop/reference specification, 16 GB GDDR6","interconnect":"PCIe; no dedicated inter-GPU memory fabric","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/7000-series/amd-radeon-rx-7800-xt.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":263,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Radeon RX 7800 XT","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates."},{"id":"rx-7900-gre","name":"Radeon RX 7900 GRE","maker":"AMD","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":576,"count":1,"summary":"16 GB GDDR6 dedicated memory.","caveat":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/7000-series/amd-radeon-rx-7900-gre.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR6","configuration":"Radeon RX 7900 GRE, desktop/reference specification, 16 GB GDDR6","interconnect":"PCIe; no dedicated inter-GPU memory fabric","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/7000-series/amd-radeon-rx-7900-gre.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":260,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Radeon RX 7900 GRE","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates."},{"id":"rx-7900-xt","name":"Radeon RX 7900 XT","maker":"AMD","memoryGiB":20,"reserveGiB":1,"type":"GPU","bandwidthGBs":800,"count":1,"summary":"20 GB GDDR6 dedicated memory.","caveat":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/7000-series/amd-radeon-rx-7900xt.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":20,"memoryType":"GDDR6","configuration":"Radeon RX 7900 XT, desktop/reference specification, 20 GB GDDR6","interconnect":"PCIe; no dedicated inter-GPU memory fabric","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/7000-series/amd-radeon-rx-7900xt.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":315,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Radeon RX 7900 XT","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates."},{"id":"rx-7900-xtx","name":"Radeon RX 7900 XTX","maker":"AMD","memoryGiB":24,"reserveGiB":1,"type":"GPU","bandwidthGBs":960,"count":1,"summary":"24 GB GDDR6 dedicated memory.","caveat":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/7000-series/amd-radeon-rx-7900xtx.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":24,"memoryType":"GDDR6","configuration":"Radeon RX 7900 XTX, desktop/reference specification, 24 GB GDDR6","interconnect":"PCIe; no dedicated inter-GPU memory fabric","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/7000-series/amd-radeon-rx-7900xtx.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":355,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Radeon RX 7900 XTX","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates."},{"id":"rx-9060-xt-16gb","name":"Radeon RX 9060 XT 16 GB","maker":"AMD","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":320,"count":1,"summary":"16 GB GDDR6 dedicated memory.","caveat":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/9000-series/amd-radeon-rx-9060xt.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR6","configuration":"Radeon RX 9060 XT 16 GB, desktop/reference specification, 16 GB GDDR6","interconnect":"PCIe; no dedicated inter-GPU memory fabric","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/9000-series/amd-radeon-rx-9060xt.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":160,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Radeon RX 9060 XT 16 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates."},{"id":"rx-9070-gre","name":"Radeon RX 9070 GRE","maker":"AMD","memoryGiB":12,"reserveGiB":1,"type":"GPU","bandwidthGBs":432,"count":1,"summary":"12 GB GDDR6 dedicated memory.","caveat":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/9000-series/amd-radeon-rx-9070-gre.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":12,"memoryType":"GDDR6","configuration":"Radeon RX 9070 GRE, desktop/reference specification, 12 GB GDDR6","interconnect":"PCIe; no dedicated inter-GPU memory fabric","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/9000-series/amd-radeon-rx-9070-gre.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":220,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Radeon RX 9070 GRE","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates."},{"id":"rx-9070","name":"Radeon RX 9070","maker":"AMD","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":640,"count":1,"summary":"16 GB GDDR6 dedicated memory.","caveat":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/9000-series/amd-radeon-rx-9070.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR6","configuration":"Radeon RX 9070, desktop/reference specification, 16 GB GDDR6","interconnect":"PCIe; no dedicated inter-GPU memory fabric","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/9000-series/amd-radeon-rx-9070.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":220,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Radeon RX 9070","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates."},{"id":"rx-9070-xt","name":"Radeon RX 9070 XT","maker":"AMD","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":640,"count":1,"summary":"16 GB GDDR6 dedicated memory.","caveat":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/9000-series/amd-radeon-rx-9070xt.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR6","configuration":"Radeon RX 9070 XT, desktop/reference specification, 16 GB GDDR6","interconnect":"PCIe; no dedicated inter-GPU memory fabric","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/9000-series/amd-radeon-rx-9070xt.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":304,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Radeon RX 9070 XT","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Raw GDDR bandwidth, excluding AMD Infinity Cache effective-bandwidth marketing estimates."},{"id":"radeon-ai-pro-r9700","name":"Radeon AI PRO R9700","maker":"AMD","memoryGiB":32,"reserveGiB":1,"type":"GPU","bandwidthGBs":640,"count":1,"summary":"32 GB GDDR6 dedicated memory.","caveat":"32 GB desktop card. ECC is supported on Linux. Verify the exact ROCm/PyTorch build and supported quantization; multi-card placement depends on the host.","source":"https://www.amd.com/en/products/graphics/workstations/radeon-ai-pro/ai-9000-series/amd-radeon-ai-pro-r9700.html","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":32,"memoryType":"GDDR6","configuration":"Radeon AI PRO R9700, desktop/reference specification, 32 GB GDDR6","interconnect":"PCIe 5.0 x16","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/workstations/radeon-ai-pro/ai-9000-series/amd-radeon-ai-pro-r9700.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":300,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Radeon AI PRO R9700","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"radeon-pro-w7800-32gb","name":"Radeon PRO W7800 32 GB","maker":"AMD","memoryGiB":32,"reserveGiB":1,"type":"GPU","bandwidthGBs":576,"count":1,"summary":"32 GB GDDR6 ECC dedicated memory.","caveat":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","source":"https://www.amd.com/en/newsroom/press-releases/2023-4-13-amd-unveils-the-most-powerful-amd-radeon-pro-graph.html","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":32,"memoryType":"GDDR6 ECC","configuration":"Radeon PRO W7800 32 GB, desktop/reference specification, 32 GB GDDR6 ECC","interconnect":"PCIe 4.0 x16","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/newsroom/press-releases/2023-4-13-amd-unveils-the-most-powerful-amd-radeon-pro-graph.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":260,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Radeon PRO W7800 32 GB","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"radeon-pro-w7800-48gb","name":"Radeon PRO W7800 48 GB","maker":"AMD","memoryGiB":48,"reserveGiB":1,"type":"GPU","bandwidthGBs":864,"count":1,"summary":"48 GB GDDR6 ECC dedicated memory.","caveat":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","source":"https://www.amd.com/en/products/graphics/workstations/radeon-pro/w7800-48gb.html","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":48,"memoryType":"GDDR6 ECC","configuration":"Radeon PRO W7800 48 GB, desktop/reference specification, 48 GB GDDR6 ECC","interconnect":"PCIe 4.0 x16","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/workstations/radeon-pro/w7800-48gb.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":260,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Radeon PRO W7800 48 GB","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"radeon-pro-w7900","name":"Radeon PRO W7900","maker":"AMD","memoryGiB":48,"reserveGiB":1,"type":"GPU","bandwidthGBs":864,"count":1,"summary":"48 GB GDDR6 ECC dedicated memory.","caveat":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","source":"https://www.amd.com/en/newsroom/press-releases/2023-4-13-amd-unveils-the-most-powerful-amd-radeon-pro-graph.html","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":48,"memoryType":"GDDR6 ECC","configuration":"Radeon PRO W7900, desktop/reference specification, 48 GB GDDR6 ECC","interconnect":"PCIe 4.0 x16","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/newsroom/press-releases/2023-4-13-amd-unveils-the-most-powerful-amd-radeon-pro-graph.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":295,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Radeon PRO W7900","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"mi210-pcie","name":"Instinct MI210 PCIe","maker":"AMD","memoryGiB":64,"reserveGiB":2,"type":"GPU","bandwidthGBs":1600,"count":1,"summary":"64 GB HBM2e ECC dedicated memory.","caveat":"Requires a compatible server, cooling and power delivery. Capacity is for one full GPU; a MIG/vGPU partition exposes less. Runtime support and actual free memory must be checked.","source":"https://www.amd.com/en/products/accelerators/instinct/mi200/mi210.html","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":64,"memoryType":"HBM2e ECC","configuration":"Instinct MI210 PCIe, server card/module, 64 GB HBM2e ECC","interconnect":"PCIe 4.0 x16; up to three Infinity Fabric links with bridges","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/accelerators/instinct/mi200/mi210.html"},{"title":"Manufacturer source 2","url":"https://instinct.docs.amd.com/projects/system-acceptance/en/latest/gpus/mi210.html"}],"powerWatts":300,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Instinct MI210 PCIe","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"mi300x","name":"Instinct MI300X OAM","maker":"AMD","memoryGiB":192,"reserveGiB":2,"type":"GPU","bandwidthGBs":5300,"count":1,"summary":"192 GB HBM3 ECC dedicated memory.","caveat":"Requires a compatible server, cooling and power delivery. Capacity is for one full GPU; a MIG/vGPU partition exposes less. Runtime support and actual free memory must be checked.","source":"https://www.amd.com/en/products/accelerators/instinct/mi300/mi300x.html","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":192,"memoryType":"HBM3 ECC","configuration":"Instinct MI300X OAM, server card/module, 192 GB HBM3 ECC","interconnect":"OAM module; Infinity Fabric on compatible server baseboard","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/accelerators/instinct/mi300/mi300x.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/en/docs-7.2.4/how-to/rocm-for-ai/inference-optimization/workload.html"}],"powerWatts":750,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Instinct MI300X OAM","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"mi325x","name":"Instinct MI325X OAM","maker":"AMD","memoryGiB":256,"reserveGiB":2,"type":"GPU","bandwidthGBs":6000,"count":1,"summary":"256 GB HBM3e ECC dedicated memory.","caveat":"Requires a compatible server, cooling and power delivery. Capacity is for one full GPU; a MIG/vGPU partition exposes less. Runtime support and actual free memory must be checked.","source":"https://www.amd.com/en/products/accelerators/instinct/mi300/mi325x.html","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":256,"memoryType":"HBM3e ECC","configuration":"Instinct MI325X OAM, server card/module, 256 GB HBM3e ECC","interconnect":"OAM module; Infinity Fabric on compatible server baseboard","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/accelerators/instinct/mi300/mi325x.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/en/docs-7.2.4/how-to/rocm-for-ai/inference-optimization/workload.html"}],"powerWatts":1000,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Instinct MI325X OAM","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"mi350x","name":"Instinct MI350X OAM","maker":"AMD","memoryGiB":288,"reserveGiB":2,"type":"GPU","bandwidthGBs":8000,"count":1,"summary":"288 GB HBM3e ECC dedicated memory.","caveat":"Requires a compatible server, cooling and power delivery. Capacity is for one full GPU; a MIG/vGPU partition exposes less. Runtime support and actual free memory must be checked.","source":"https://www.amd.com/en/products/accelerators/instinct/mi350/mi350x.html","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":288,"memoryType":"HBM3e ECC","configuration":"Instinct MI350X OAM, server card/module, 288 GB HBM3e ECC","interconnect":"OAM module; Infinity Fabric on compatible server baseboard","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/accelerators/instinct/mi350/mi350x.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/en/docs-7.2.4/how-to/rocm-for-ai/inference-optimization/workload.html"}],"powerWatts":1000,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Instinct MI350X OAM","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"mi355x","name":"Instinct MI355X OAM","maker":"AMD","memoryGiB":288,"reserveGiB":2,"type":"GPU","bandwidthGBs":8000,"count":1,"summary":"288 GB HBM3e ECC dedicated memory.","caveat":"Requires a compatible server, cooling and power delivery. Capacity is for one full GPU; a MIG/vGPU partition exposes less. Runtime support and actual free memory must be checked. MI355X is designed for direct liquid cooling.","source":"https://www.amd.com/en/products/accelerators/instinct/mi350/mi355x.html","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":288,"memoryType":"HBM3e ECC","configuration":"Instinct MI355X OAM, server card/module, 288 GB HBM3e ECC","interconnect":"OAM module; Infinity Fabric on compatible server baseboard","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/accelerators/instinct/mi350/mi355x.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/en/docs-7.2.4/how-to/rocm-for-ai/inference-optimization/workload.html"}],"powerWatts":1400,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Instinct MI355X OAM","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"mi350p","name":"Instinct MI350P PCIe","maker":"AMD","memoryGiB":144,"reserveGiB":2,"type":"GPU","bandwidthGBs":4000,"count":1,"summary":"144 GB HBM3e ECC dedicated memory.","caveat":"Server PCIe card with passive cooling. Board power is configurable to 450 W, 600 W maximum; verify qualified server and ROCm support.","source":"https://www.amd.com/en/products/accelerators/instinct/mi350/mi350p.html","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":144,"memoryType":"HBM3e ECC","configuration":"Instinct MI350P PCIe, server card/module, 144 GB HBM3e ECC","interconnect":"PCIe 5.0 x16; passive double-slot card","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/accelerators/instinct/mi350/mi350p.html"}],"powerWatts":600,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Instinct MI350P PCIe","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"arc-pro-b50","name":"Arc Pro B50","maker":"Intel","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":224,"count":1,"summary":"16 GB GDDR6 dedicated memory.","caveat":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","source":"https://www.intel.com/content/dam/www/central-libraries/us/en/documents/2026-03/intel-arc-pro-b-series-graphics-quick-reference-guide-v1-0.pdf","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR6","configuration":"Arc Pro B50, desktop/reference specification, 16 GB GDDR6","interconnect":"PCIe; topology and lane width depend on host","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","sources":[{"title":"Manufacturer source 1","url":"https://www.intel.com/content/dam/www/central-libraries/us/en/documents/2026-03/intel-arc-pro-b-series-graphics-quick-reference-guide-v1-0.pdf"},{"title":"Manufacturer source 2","url":"https://www.intel.com/content/www/us/en/ark/products/series/242616/intel-arc-pro-b-series-graphics.html"}],"powerWatts":70,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Arc Pro B50","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"arc-pro-b60","name":"Arc Pro B60","maker":"Intel","memoryGiB":24,"reserveGiB":1,"type":"GPU","bandwidthGBs":456,"count":1,"summary":"24 GB GDDR6 dedicated memory.","caveat":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","source":"https://www.intel.com/content/dam/www/central-libraries/us/en/documents/2026-03/intel-arc-pro-b-series-graphics-quick-reference-guide-v1-0.pdf","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":24,"memoryType":"GDDR6","configuration":"Arc Pro B60, desktop/reference specification, 24 GB GDDR6","interconnect":"PCIe; topology and lane width depend on host","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","sources":[{"title":"Manufacturer source 1","url":"https://www.intel.com/content/dam/www/central-libraries/us/en/documents/2026-03/intel-arc-pro-b-series-graphics-quick-reference-guide-v1-0.pdf"},{"title":"Manufacturer source 2","url":"https://www.intel.com/content/www/us/en/ark/products/series/242616/intel-arc-pro-b-series-graphics.html"}],"powerRangeWatts":[120,200],"family":"Arc Pro B60","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"arc-pro-b65","name":"Arc Pro B65","maker":"Intel","memoryGiB":32,"reserveGiB":1,"type":"GPU","bandwidthGBs":608,"count":1,"summary":"32 GB GDDR6 dedicated memory.","caveat":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","source":"https://www.intel.com/content/dam/www/central-libraries/us/en/documents/2026-03/intel-arc-pro-b-series-graphics-quick-reference-guide-v1-0.pdf","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":32,"memoryType":"GDDR6","configuration":"Arc Pro B65, desktop/reference specification, 32 GB GDDR6","interconnect":"PCIe; topology and lane width depend on host","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","sources":[{"title":"Manufacturer source 1","url":"https://www.intel.com/content/dam/www/central-libraries/us/en/documents/2026-03/intel-arc-pro-b-series-graphics-quick-reference-guide-v1-0.pdf"},{"title":"Manufacturer source 2","url":"https://www.intel.com/content/www/us/en/ark/products/series/242616/intel-arc-pro-b-series-graphics.html"}],"powerWatts":200,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Arc Pro B65","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"arc-pro-b70","name":"Arc Pro B70","maker":"Intel","memoryGiB":32,"reserveGiB":1,"type":"GPU","bandwidthGBs":608,"count":1,"summary":"32 GB GDDR6 dedicated memory.","caveat":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","source":"https://www.intel.com/content/dam/www/central-libraries/us/en/documents/2026-03/intel-arc-pro-b-series-graphics-quick-reference-guide-v1-0.pdf","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":32,"memoryType":"GDDR6","configuration":"Arc Pro B70, desktop/reference specification, 32 GB GDDR6","interconnect":"PCIe; topology and lane width depend on host","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","sources":[{"title":"Manufacturer source 1","url":"https://www.intel.com/content/dam/www/central-libraries/us/en/documents/2026-03/intel-arc-pro-b-series-graphics-quick-reference-guide-v1-0.pdf"},{"title":"Manufacturer source 2","url":"https://www.intel.com/content/www/us/en/ark/products/series/242616/intel-arc-pro-b-series-graphics.html"}],"powerRangeWatts":[160,290],"family":"Arc Pro B70","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"arc-a770-16gb","name":"Arc A770 16 GB","maker":"Intel","memoryGiB":16,"reserveGiB":1,"type":"GPU","bandwidthGBs":560,"count":1,"summary":"16 GB GDDR6 dedicated memory.","caveat":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","source":"https://www.intel.com/content/www/us/en/products/sku/229151/intel-arc-a770-graphics-16gb/specifications.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR6","configuration":"Intel Arc A770 16 GB reference memory configuration; 17.5 Gbit/s GDDR6","interconnect":"PCIe 4.0 x16","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","sources":[{"title":"Manufacturer source 1","url":"https://www.intel.com/content/www/us/en/products/sku/229151/intel-arc-a770-graphics-16gb/specifications.html"}],"powerWatts":225,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Arc A770 16 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"arc-b570","name":"Arc B570","maker":"Intel","memoryGiB":10,"reserveGiB":1,"type":"GPU","bandwidthGBs":380,"count":1,"summary":"10 GB GDDR6 dedicated memory.","caveat":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","source":"https://www.intel.com/content/www/us/en/products/details/discrete-gpus/arc/desktop/b-series.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":10,"memoryType":"GDDR6","configuration":"Arc B570, desktop/reference specification, 10 GB GDDR6","interconnect":"PCIe 4.0 x8","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","sources":[{"title":"Manufacturer source 1","url":"https://www.intel.com/content/www/us/en/products/details/discrete-gpus/arc/desktop/b-series.html"},{"title":"Manufacturer source 2","url":"https://cdrdv2-public.intel.com/839907/Intel%20Arc%20B-Series%20Graphics%20Quick%20Reference%20Guide%20V1.1.pdf"}],"powerWatts":150,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Arc B570","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"arc-b580","name":"Arc B580","maker":"Intel","memoryGiB":12,"reserveGiB":1,"type":"GPU","bandwidthGBs":456,"count":1,"summary":"12 GB GDDR6 dedicated memory.","caveat":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","source":"https://www.intel.com/content/www/us/en/products/details/discrete-gpus/arc/desktop/b-series.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":12,"memoryType":"GDDR6","configuration":"Arc B580, desktop/reference specification, 12 GB GDDR6","interconnect":"PCIe 4.0 x8","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","sources":[{"title":"Manufacturer source 1","url":"https://www.intel.com/content/www/us/en/products/details/discrete-gpus/arc/desktop/b-series.html"},{"title":"Manufacturer source 2","url":"https://cdrdv2-public.intel.com/839907/Intel%20Arc%20B-Series%20Graphics%20Quick%20Reference%20Guide%20V1.1.pdf"}],"powerWatts":190,"powerBasis":"Maximum/reference board power; excludes the host system and is not measured inference power","family":"Arc B580","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","memoryBandwidthNote":"Manufacturer peak memory bandwidth; not a measured sustained rate or a token/s estimate."},{"id":"mac-mini-m1-8cpu-8gpu-8gb","name":"Mac mini M1 · 8 GB (8-core GPU)","maker":"Apple","memoryGiB":8,"reserveGiB":2,"type":"Unified memory","bandwidthGBs":null,"count":1,"summary":"8-core CPU, 8-core GPU; 8 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111894","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":8,"memoryType":"Unified memory","configuration":"Mac mini, Apple M1, 8-core CPU / 8-core GPU / 8 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Not published by Apple in the cited specification","bandwidthNote":"Memory capacity and core bin are verified. Apple does not publish a bandwidth figure in this specification; no bandwidth-derived speed estimate is supplied.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111894"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M1","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Memory capacity and core bin are verified. Apple does not publish a bandwidth figure in this specification; no bandwidth-derived speed estimate is supplied."},{"id":"mac-mini-m1-8cpu-8gpu-16gb","name":"Mac mini M1 · 16 GB (8-core GPU)","maker":"Apple","memoryGiB":16,"reserveGiB":4,"type":"Unified memory","bandwidthGBs":null,"count":1,"summary":"8-core CPU, 8-core GPU; 16 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111894","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":16,"memoryType":"Unified memory","configuration":"Mac mini, Apple M1, 8-core CPU / 8-core GPU / 16 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Not published by Apple in the cited specification","bandwidthNote":"Memory capacity and core bin are verified. Apple does not publish a bandwidth figure in this specification; no bandwidth-derived speed estimate is supplied.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111894"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M1","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Memory capacity and core bin are verified. Apple does not publish a bandwidth figure in this specification; no bandwidth-derived speed estimate is supplied."},{"id":"mac-mini-m2-8cpu-10gpu-8gb","name":"Mac mini M2 · 8 GB (10-core GPU)","maker":"Apple","memoryGiB":8,"reserveGiB":2,"type":"Unified memory","bandwidthGBs":100,"count":1,"summary":"8-core CPU, 10-core GPU; 8 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111837","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":8,"memoryType":"Unified memory","configuration":"Mac mini, Apple M2, 8-core CPU / 10-core GPU / 8 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111837"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M2","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m2-8cpu-10gpu-16gb","name":"Mac mini M2 · 16 GB (10-core GPU)","maker":"Apple","memoryGiB":16,"reserveGiB":4,"type":"Unified memory","bandwidthGBs":100,"count":1,"summary":"8-core CPU, 10-core GPU; 16 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111837","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":16,"memoryType":"Unified memory","configuration":"Mac mini, Apple M2, 8-core CPU / 10-core GPU / 16 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111837"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M2","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m2-8cpu-10gpu-24gb","name":"Mac mini M2 · 24 GB (10-core GPU)","maker":"Apple","memoryGiB":24,"reserveGiB":6,"type":"Unified memory","bandwidthGBs":100,"count":1,"summary":"8-core CPU, 10-core GPU; 24 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111837","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":24,"memoryType":"Unified memory","configuration":"Mac mini, Apple M2, 8-core CPU / 10-core GPU / 24 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111837"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M2","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m2-pro-10cpu-16gpu-16gb","name":"Mac mini M2 Pro · 16 GB (16-core GPU)","maker":"Apple","memoryGiB":16,"reserveGiB":4,"type":"Unified memory","bandwidthGBs":200,"count":1,"summary":"10-core CPU, 16-core GPU; 16 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111837","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":16,"memoryType":"Unified memory","configuration":"Mac mini, Apple M2 Pro, 10-core CPU / 16-core GPU / 16 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111837"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M2 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m2-pro-10cpu-16gpu-32gb","name":"Mac mini M2 Pro · 32 GB (16-core GPU)","maker":"Apple","memoryGiB":32,"reserveGiB":8,"type":"Unified memory","bandwidthGBs":200,"count":1,"summary":"10-core CPU, 16-core GPU; 32 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111837","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":32,"memoryType":"Unified memory","configuration":"Mac mini, Apple M2 Pro, 10-core CPU / 16-core GPU / 32 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111837"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M2 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m2-pro-12cpu-19gpu-16gb","name":"Mac mini M2 Pro · 16 GB (19-core GPU)","maker":"Apple","memoryGiB":16,"reserveGiB":4,"type":"Unified memory","bandwidthGBs":200,"count":1,"summary":"12-core CPU, 19-core GPU; 16 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111837","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":16,"memoryType":"Unified memory","configuration":"Mac mini, Apple M2 Pro, 12-core CPU / 19-core GPU / 16 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111837"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M2 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m2-pro-12cpu-19gpu-32gb","name":"Mac mini M2 Pro · 32 GB (19-core GPU)","maker":"Apple","memoryGiB":32,"reserveGiB":8,"type":"Unified memory","bandwidthGBs":200,"count":1,"summary":"12-core CPU, 19-core GPU; 32 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111837","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":32,"memoryType":"Unified memory","configuration":"Mac mini, Apple M2 Pro, 12-core CPU / 19-core GPU / 32 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111837"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M2 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-air-m3-8cpu-8gpu-8gb","name":"MacBook Air M3 · 8 GB (8-core GPU)","maker":"Apple","memoryGiB":8,"reserveGiB":2,"type":"Unified memory","bandwidthGBs":100,"count":1,"summary":"8-core CPU, 8-core GPU; 8 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. Fanless MacBook Air; sustained workloads can be limited by thermals.","source":"https://support.apple.com/en-us/118551","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":8,"memoryType":"Unified memory","configuration":"MacBook Air, Apple M3, 8-core CPU / 8-core GPU / 8 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/118551"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Air M3","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-air-m3-8cpu-8gpu-16gb","name":"MacBook Air M3 · 16 GB (8-core GPU)","maker":"Apple","memoryGiB":16,"reserveGiB":4,"type":"Unified memory","bandwidthGBs":100,"count":1,"summary":"8-core CPU, 8-core GPU; 16 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. Fanless MacBook Air; sustained workloads can be limited by thermals.","source":"https://support.apple.com/en-us/118551","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":16,"memoryType":"Unified memory","configuration":"MacBook Air, Apple M3, 8-core CPU / 8-core GPU / 16 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/118551"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Air M3","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-air-m3-8cpu-8gpu-24gb","name":"MacBook Air M3 · 24 GB (8-core GPU)","maker":"Apple","memoryGiB":24,"reserveGiB":6,"type":"Unified memory","bandwidthGBs":100,"count":1,"summary":"8-core CPU, 8-core GPU; 24 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. Fanless MacBook Air; sustained workloads can be limited by thermals.","source":"https://support.apple.com/en-us/118551","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":24,"memoryType":"Unified memory","configuration":"MacBook Air, Apple M3, 8-core CPU / 8-core GPU / 24 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/118551"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Air M3","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-air-m3-8cpu-10gpu-8gb","name":"MacBook Air M3 · 8 GB (10-core GPU)","maker":"Apple","memoryGiB":8,"reserveGiB":2,"type":"Unified memory","bandwidthGBs":100,"count":1,"summary":"8-core CPU, 10-core GPU; 8 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. Fanless MacBook Air; sustained workloads can be limited by thermals.","source":"https://support.apple.com/en-us/118551","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":8,"memoryType":"Unified memory","configuration":"MacBook Air, Apple M3, 8-core CPU / 10-core GPU / 8 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/118551"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Air M3","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-air-m3-8cpu-10gpu-16gb","name":"MacBook Air M3 · 16 GB (10-core GPU)","maker":"Apple","memoryGiB":16,"reserveGiB":4,"type":"Unified memory","bandwidthGBs":100,"count":1,"summary":"8-core CPU, 10-core GPU; 16 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. Fanless MacBook Air; sustained workloads can be limited by thermals.","source":"https://support.apple.com/en-us/118551","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":16,"memoryType":"Unified memory","configuration":"MacBook Air, Apple M3, 8-core CPU / 10-core GPU / 16 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/118551"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Air M3","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-air-m3-8cpu-10gpu-24gb","name":"MacBook Air M3 · 24 GB (10-core GPU)","maker":"Apple","memoryGiB":24,"reserveGiB":6,"type":"Unified memory","bandwidthGBs":100,"count":1,"summary":"8-core CPU, 10-core GPU; 24 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. Fanless MacBook Air; sustained workloads can be limited by thermals.","source":"https://support.apple.com/en-us/118551","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":24,"memoryType":"Unified memory","configuration":"MacBook Air, Apple M3, 8-core CPU / 10-core GPU / 24 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/118551"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Air M3","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m1-pro-10cpu-16gpu-16gb","name":"MacBook Pro M1 Pro · 16 GB (16-core GPU)","maker":"Apple","memoryGiB":16,"reserveGiB":4,"type":"Unified memory","bandwidthGBs":200,"count":1,"summary":"10-core CPU, 16-core GPU; 16 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111901","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":16,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M1 Pro, 10-core CPU / 16-core GPU / 16 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111901"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M1 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m1-pro-10cpu-16gpu-32gb","name":"MacBook Pro M1 Pro · 32 GB (16-core GPU)","maker":"Apple","memoryGiB":32,"reserveGiB":8,"type":"Unified memory","bandwidthGBs":200,"count":1,"summary":"10-core CPU, 16-core GPU; 32 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111901","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":32,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M1 Pro, 10-core CPU / 16-core GPU / 32 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111901"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M1 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m1-max-10cpu-24gpu-32gb","name":"Mac Studio M1 Max · 32 GB (24-core GPU)","maker":"Apple","memoryGiB":32,"reserveGiB":8,"type":"Unified memory","bandwidthGBs":400,"count":1,"summary":"10-core CPU, 24-core GPU; 32 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111900","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":32,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M1 Max, 10-core CPU / 24-core GPU / 32 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111900"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M1 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m1-max-10cpu-24gpu-64gb","name":"Mac Studio M1 Max · 64 GB (24-core GPU)","maker":"Apple","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":400,"count":1,"summary":"10-core CPU, 24-core GPU; 64 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111900","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":64,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M1 Max, 10-core CPU / 24-core GPU / 64 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111900"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M1 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m1-max-10cpu-32gpu-32gb","name":"Mac Studio M1 Max · 32 GB (32-core GPU)","maker":"Apple","memoryGiB":32,"reserveGiB":8,"type":"Unified memory","bandwidthGBs":400,"count":1,"summary":"10-core CPU, 32-core GPU; 32 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111900","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":32,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M1 Max, 10-core CPU / 32-core GPU / 32 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111900"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M1 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m1-max-10cpu-32gpu-64gb","name":"Mac Studio M1 Max · 64 GB (32-core GPU)","maker":"Apple","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":400,"count":1,"summary":"10-core CPU, 32-core GPU; 64 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111900","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":64,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M1 Max, 10-core CPU / 32-core GPU / 64 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111900"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M1 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m1-ultra-20cpu-48gpu-64gb","name":"Mac Studio M1 Ultra · 64 GB (48-core GPU)","maker":"Apple","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":800,"count":1,"summary":"20-core CPU, 48-core GPU; 64 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111900","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":64,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M1 Ultra, 20-core CPU / 48-core GPU / 64 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111900"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M1 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m1-ultra-20cpu-48gpu-128gb","name":"Mac Studio M1 Ultra · 128 GB (48-core GPU)","maker":"Apple","memoryGiB":128,"reserveGiB":32,"type":"Unified memory","bandwidthGBs":800,"count":1,"summary":"20-core CPU, 48-core GPU; 128 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111900","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":128,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M1 Ultra, 20-core CPU / 48-core GPU / 128 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111900"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M1 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m1-ultra-20cpu-64gpu-64gb","name":"Mac Studio M1 Ultra · 64 GB (64-core GPU)","maker":"Apple","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":800,"count":1,"summary":"20-core CPU, 64-core GPU; 64 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111900","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":64,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M1 Ultra, 20-core CPU / 64-core GPU / 64 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111900"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M1 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m1-ultra-20cpu-64gpu-128gb","name":"Mac Studio M1 Ultra · 128 GB (64-core GPU)","maker":"Apple","memoryGiB":128,"reserveGiB":32,"type":"Unified memory","bandwidthGBs":800,"count":1,"summary":"20-core CPU, 64-core GPU; 128 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111900","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":128,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M1 Ultra, 20-core CPU / 64-core GPU / 128 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111900"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M1 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m2-max-12cpu-30gpu-32gb","name":"Mac Studio M2 Max · 32 GB (30-core GPU)","maker":"Apple","memoryGiB":32,"reserveGiB":8,"type":"Unified memory","bandwidthGBs":400,"count":1,"summary":"12-core CPU, 30-core GPU; 32 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111835","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":32,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M2 Max, 12-core CPU / 30-core GPU / 32 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111835"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M2 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m2-max-12cpu-30gpu-64gb","name":"Mac Studio M2 Max · 64 GB (30-core GPU)","maker":"Apple","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":400,"count":1,"summary":"12-core CPU, 30-core GPU; 64 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111835","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":64,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M2 Max, 12-core CPU / 30-core GPU / 64 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111835"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M2 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m2-max-12cpu-38gpu-32gb","name":"Mac Studio M2 Max · 32 GB (38-core GPU)","maker":"Apple","memoryGiB":32,"reserveGiB":8,"type":"Unified memory","bandwidthGBs":400,"count":1,"summary":"12-core CPU, 38-core GPU; 32 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111835","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":32,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M2 Max, 12-core CPU / 38-core GPU / 32 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111835"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M2 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m2-max-12cpu-38gpu-64gb","name":"Mac Studio M2 Max · 64 GB (38-core GPU)","maker":"Apple","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":400,"count":1,"summary":"12-core CPU, 38-core GPU; 64 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111835","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":64,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M2 Max, 12-core CPU / 38-core GPU / 64 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111835"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M2 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m2-max-12cpu-38gpu-96gb","name":"Mac Studio M2 Max · 96 GB (38-core GPU)","maker":"Apple","memoryGiB":96,"reserveGiB":24,"type":"Unified memory","bandwidthGBs":400,"count":1,"summary":"12-core CPU, 38-core GPU; 96 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111835","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":96,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M2 Max, 12-core CPU / 38-core GPU / 96 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111835"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M2 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m2-ultra-24cpu-60gpu-64gb","name":"Mac Studio M2 Ultra · 64 GB (60-core GPU)","maker":"Apple","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":800,"count":1,"summary":"24-core CPU, 60-core GPU; 64 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111835","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":64,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M2 Ultra, 24-core CPU / 60-core GPU / 64 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111835"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M2 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m2-ultra-24cpu-60gpu-128gb","name":"Mac Studio M2 Ultra · 128 GB (60-core GPU)","maker":"Apple","memoryGiB":128,"reserveGiB":32,"type":"Unified memory","bandwidthGBs":800,"count":1,"summary":"24-core CPU, 60-core GPU; 128 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111835","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":128,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M2 Ultra, 24-core CPU / 60-core GPU / 128 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111835"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M2 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m2-ultra-24cpu-60gpu-192gb","name":"Mac Studio M2 Ultra · 192 GB (60-core GPU)","maker":"Apple","memoryGiB":192,"reserveGiB":48,"type":"Unified memory","bandwidthGBs":800,"count":1,"summary":"24-core CPU, 60-core GPU; 192 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111835","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":192,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M2 Ultra, 24-core CPU / 60-core GPU / 192 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111835"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M2 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m2-ultra-24cpu-76gpu-64gb","name":"Mac Studio M2 Ultra · 64 GB (76-core GPU)","maker":"Apple","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":800,"count":1,"summary":"24-core CPU, 76-core GPU; 64 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111835","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":64,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M2 Ultra, 24-core CPU / 76-core GPU / 64 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111835"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M2 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m2-ultra-24cpu-76gpu-128gb","name":"Mac Studio M2 Ultra · 128 GB (76-core GPU)","maker":"Apple","memoryGiB":128,"reserveGiB":32,"type":"Unified memory","bandwidthGBs":800,"count":1,"summary":"24-core CPU, 76-core GPU; 128 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111835","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":128,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M2 Ultra, 24-core CPU / 76-core GPU / 128 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111835"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M2 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m2-ultra-24cpu-76gpu-192gb","name":"Mac Studio M2 Ultra · 192 GB (76-core GPU)","maker":"Apple","memoryGiB":192,"reserveGiB":48,"type":"Unified memory","bandwidthGBs":800,"count":1,"summary":"24-core CPU, 76-core GPU; 192 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/111835","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":192,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M2 Ultra, 24-core CPU / 76-core GPU / 192 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/111835"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M2 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m3-pro-12cpu-18gpu-18gb","name":"MacBook Pro M3 Pro · 18 GB (18-core GPU)","maker":"Apple","memoryGiB":18,"reserveGiB":4.5,"type":"Unified memory","bandwidthGBs":150,"count":1,"summary":"12-core CPU, 18-core GPU; 18 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/117737","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":18,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M3 Pro, 12-core CPU / 18-core GPU / 18 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/117737"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M3 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m3-pro-12cpu-18gpu-36gb","name":"MacBook Pro M3 Pro · 36 GB (18-core GPU)","maker":"Apple","memoryGiB":36,"reserveGiB":9,"type":"Unified memory","bandwidthGBs":150,"count":1,"summary":"12-core CPU, 18-core GPU; 36 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/117737","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":36,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M3 Pro, 12-core CPU / 18-core GPU / 36 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/117737"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M3 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m3-max-14cpu-30gpu-36gb","name":"MacBook Pro M3 Max · 36 GB (30-core GPU)","maker":"Apple","memoryGiB":36,"reserveGiB":9,"type":"Unified memory","bandwidthGBs":300,"count":1,"summary":"14-core CPU, 30-core GPU; 36 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/117737","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":36,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M3 Max, 14-core CPU / 30-core GPU / 36 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/117737"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M3 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m3-max-14cpu-30gpu-96gb","name":"MacBook Pro M3 Max · 96 GB (30-core GPU)","maker":"Apple","memoryGiB":96,"reserveGiB":24,"type":"Unified memory","bandwidthGBs":300,"count":1,"summary":"14-core CPU, 30-core GPU; 96 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/117737","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":96,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M3 Max, 14-core CPU / 30-core GPU / 96 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/117737"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M3 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m3-max-16cpu-40gpu-48gb","name":"MacBook Pro M3 Max · 48 GB (40-core GPU)","maker":"Apple","memoryGiB":48,"reserveGiB":12,"type":"Unified memory","bandwidthGBs":400,"count":1,"summary":"16-core CPU, 40-core GPU; 48 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/117737","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":48,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M3 Max, 16-core CPU / 40-core GPU / 48 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/117737"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M3 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m3-max-16cpu-40gpu-64gb","name":"MacBook Pro M3 Max · 64 GB (40-core GPU)","maker":"Apple","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":400,"count":1,"summary":"16-core CPU, 40-core GPU; 64 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/117737","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":64,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M3 Max, 16-core CPU / 40-core GPU / 64 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/117737"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M3 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m3-max-16cpu-40gpu-128gb","name":"MacBook Pro M3 Max · 128 GB (40-core GPU)","maker":"Apple","memoryGiB":128,"reserveGiB":32,"type":"Unified memory","bandwidthGBs":400,"count":1,"summary":"16-core CPU, 40-core GPU; 128 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/117737","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":128,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M3 Max, 16-core CPU / 40-core GPU / 128 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/117737"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M3 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m4-10cpu-10gpu-16gb","name":"Mac mini M4 · 16 GB (10-core GPU)","maker":"Apple","memoryGiB":16,"reserveGiB":4,"type":"Unified memory","bandwidthGBs":120,"count":1,"summary":"10-core CPU, 10-core GPU; 16 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/121555","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":16,"memoryType":"Unified memory","configuration":"Mac mini, Apple M4, 10-core CPU / 10-core GPU / 16 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/121555"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M4","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m4-10cpu-10gpu-24gb","name":"Mac mini M4 · 24 GB (10-core GPU)","maker":"Apple","memoryGiB":24,"reserveGiB":6,"type":"Unified memory","bandwidthGBs":120,"count":1,"summary":"10-core CPU, 10-core GPU; 24 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/121555","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":24,"memoryType":"Unified memory","configuration":"Mac mini, Apple M4, 10-core CPU / 10-core GPU / 24 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/121555"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M4","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m4-10cpu-10gpu-32gb","name":"Mac mini M4 · 32 GB (10-core GPU)","maker":"Apple","memoryGiB":32,"reserveGiB":8,"type":"Unified memory","bandwidthGBs":120,"count":1,"summary":"10-core CPU, 10-core GPU; 32 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/121555","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":32,"memoryType":"Unified memory","configuration":"Mac mini, Apple M4, 10-core CPU / 10-core GPU / 32 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/121555"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M4","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m4-pro-12cpu-16gpu-24gb","name":"Mac mini M4 Pro · 24 GB (16-core GPU)","maker":"Apple","memoryGiB":24,"reserveGiB":6,"type":"Unified memory","bandwidthGBs":273,"count":1,"summary":"12-core CPU, 16-core GPU; 24 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/121555","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":24,"memoryType":"Unified memory","configuration":"Mac mini, Apple M4 Pro, 12-core CPU / 16-core GPU / 24 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/121555"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M4 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m4-pro-12cpu-16gpu-48gb","name":"Mac mini M4 Pro · 48 GB (16-core GPU)","maker":"Apple","memoryGiB":48,"reserveGiB":12,"type":"Unified memory","bandwidthGBs":273,"count":1,"summary":"12-core CPU, 16-core GPU; 48 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/121555","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":48,"memoryType":"Unified memory","configuration":"Mac mini, Apple M4 Pro, 12-core CPU / 16-core GPU / 48 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/121555"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M4 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m4-pro-12cpu-16gpu-64gb","name":"Mac mini M4 Pro · 64 GB (16-core GPU)","maker":"Apple","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":273,"count":1,"summary":"12-core CPU, 16-core GPU; 64 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/121555","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":64,"memoryType":"Unified memory","configuration":"Mac mini, Apple M4 Pro, 12-core CPU / 16-core GPU / 64 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/121555"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M4 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m4-pro-14cpu-20gpu-24gb","name":"Mac mini M4 Pro · 24 GB (20-core GPU)","maker":"Apple","memoryGiB":24,"reserveGiB":6,"type":"Unified memory","bandwidthGBs":273,"count":1,"summary":"14-core CPU, 20-core GPU; 24 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/121555","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":24,"memoryType":"Unified memory","configuration":"Mac mini, Apple M4 Pro, 14-core CPU / 20-core GPU / 24 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/121555"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M4 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m4-pro-14cpu-20gpu-48gb","name":"Mac mini M4 Pro · 48 GB (20-core GPU)","maker":"Apple","memoryGiB":48,"reserveGiB":12,"type":"Unified memory","bandwidthGBs":273,"count":1,"summary":"14-core CPU, 20-core GPU; 48 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/121555","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":48,"memoryType":"Unified memory","configuration":"Mac mini, Apple M4 Pro, 14-core CPU / 20-core GPU / 48 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/121555"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M4 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m4-pro-14cpu-20gpu-64gb","name":"Mac mini M4 Pro · 64 GB (20-core GPU)","maker":"Apple","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":273,"count":1,"summary":"14-core CPU, 20-core GPU; 64 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/121555","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":64,"memoryType":"Unified memory","configuration":"Mac mini, Apple M4 Pro, 14-core CPU / 20-core GPU / 64 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/121555"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac mini M4 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m4-max-14cpu-32gpu-36gb","name":"Mac Studio M4 Max · 36 GB (32-core GPU)","maker":"Apple","memoryGiB":36,"reserveGiB":9,"type":"Unified memory","bandwidthGBs":410,"count":1,"summary":"14-core CPU, 32-core GPU; 36 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/122211","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":36,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M4 Max, 14-core CPU / 32-core GPU / 36 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/122211"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M4 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m4-max-16cpu-40gpu-48gb","name":"Mac Studio M4 Max · 48 GB (40-core GPU)","maker":"Apple","memoryGiB":48,"reserveGiB":12,"type":"Unified memory","bandwidthGBs":546,"count":1,"summary":"16-core CPU, 40-core GPU; 48 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/122211","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":48,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M4 Max, 16-core CPU / 40-core GPU / 48 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/122211"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M4 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m4-max-64gb","name":"Mac Studio M4 Max · 64 GB (40-core GPU)","maker":"Apple","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":546,"count":1,"summary":"16-core CPU, 40-core GPU; 64 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/122211","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":64,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M4 Max, 16-core CPU / 40-core GPU / 64 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/122211"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M4 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m4-max-16cpu-40gpu-128gb","name":"Mac Studio M4 Max · 128 GB (40-core GPU)","maker":"Apple","memoryGiB":128,"reserveGiB":32,"type":"Unified memory","bandwidthGBs":546,"count":1,"summary":"16-core CPU, 40-core GPU; 128 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/122211","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":128,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M4 Max, 16-core CPU / 40-core GPU / 128 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/122211"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M4 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m3-ultra-28cpu-60gpu-96gb","name":"Mac Studio M3 Ultra · 96 GB (60-core GPU)","maker":"Apple","memoryGiB":96,"reserveGiB":24,"type":"Unified memory","bandwidthGBs":819,"count":1,"summary":"28-core CPU, 60-core GPU; 96 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/122211","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":96,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M3 Ultra, 28-core CPU / 60-core GPU / 96 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/122211"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M3 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m3-ultra-28cpu-60gpu-256gb","name":"Mac Studio M3 Ultra · 256 GB (60-core GPU)","maker":"Apple","memoryGiB":256,"reserveGiB":64,"type":"Unified memory","bandwidthGBs":819,"count":1,"summary":"28-core CPU, 60-core GPU; 256 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/122211","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":256,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M3 Ultra, 28-core CPU / 60-core GPU / 256 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/122211"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M3 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m3-ultra-32cpu-80gpu-96gb","name":"Mac Studio M3 Ultra · 96 GB (80-core GPU)","maker":"Apple","memoryGiB":96,"reserveGiB":24,"type":"Unified memory","bandwidthGBs":819,"count":1,"summary":"32-core CPU, 80-core GPU; 96 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/122211","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":96,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M3 Ultra, 32-core CPU / 80-core GPU / 96 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/122211"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M3 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m3-ultra-256gb","name":"Mac Studio M3 Ultra · 256 GB (80-core GPU)","maker":"Apple","memoryGiB":256,"reserveGiB":64,"type":"Unified memory","bandwidthGBs":819,"count":1,"summary":"32-core CPU, 80-core GPU; 256 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/122211","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":256,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M3 Ultra, 32-core CPU / 80-core GPU / 256 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/122211"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M3 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m3-ultra-32cpu-80gpu-512gb","name":"Mac Studio M3 Ultra · 512 GB (80-core GPU)","maker":"Apple","memoryGiB":512,"reserveGiB":128,"type":"Unified memory","bandwidthGBs":819,"count":1,"summary":"32-core CPU, 80-core GPU; 512 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. 512 GB was sold at launch; Apple later removed it from current configuration options. This entry is for existing or resale machines.","source":"https://www.apple.com/newsroom/2025/03/apple-unveils-new-mac-studio-the-most-powerful-mac-ever/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":512,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M3 Ultra, 32-core CPU / 80-core GPU / 512 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Used / previous generation","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/newsroom/2025/03/apple-unveils-new-mac-studio-the-most-powerful-mac-ever/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"Mac Studio M3 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m5-10cpu-10gpu-16gb","name":"MacBook Pro M5 · 16 GB (10-core GPU)","maker":"Apple","memoryGiB":16,"reserveGiB":4,"type":"Unified memory","bandwidthGBs":153,"count":1,"summary":"10-core CPU, 10-core GPU; 16 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/125405","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":16,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M5, 10-core CPU / 10-core GPU / 16 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/125405"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M5","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m5-10cpu-10gpu-24gb","name":"MacBook Pro M5 · 24 GB (10-core GPU)","maker":"Apple","memoryGiB":24,"reserveGiB":6,"type":"Unified memory","bandwidthGBs":153,"count":1,"summary":"10-core CPU, 10-core GPU; 24 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/125405","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":24,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M5, 10-core CPU / 10-core GPU / 24 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/125405"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M5","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m5-10cpu-10gpu-32gb","name":"MacBook Pro M5 · 32 GB (10-core GPU)","maker":"Apple","memoryGiB":32,"reserveGiB":8,"type":"Unified memory","bandwidthGBs":153,"count":1,"summary":"10-core CPU, 10-core GPU; 32 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/125405","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":32,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M5, 10-core CPU / 10-core GPU / 32 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/125405"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M5","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m5-pro-15cpu-16gpu-24gb","name":"MacBook Pro M5 Pro · 24 GB (16-core GPU)","maker":"Apple","memoryGiB":24,"reserveGiB":6,"type":"Unified memory","bandwidthGBs":307,"count":1,"summary":"15-core CPU, 16-core GPU; 24 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/126318","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":24,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M5 Pro, 15-core CPU / 16-core GPU / 24 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/126318"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M5 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m5-pro-15cpu-16gpu-48gb","name":"MacBook Pro M5 Pro · 48 GB (16-core GPU)","maker":"Apple","memoryGiB":48,"reserveGiB":12,"type":"Unified memory","bandwidthGBs":307,"count":1,"summary":"15-core CPU, 16-core GPU; 48 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/126318","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":48,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M5 Pro, 15-core CPU / 16-core GPU / 48 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/126318"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M5 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m5-pro-18cpu-20gpu-24gb","name":"MacBook Pro M5 Pro · 24 GB (20-core GPU)","maker":"Apple","memoryGiB":24,"reserveGiB":6,"type":"Unified memory","bandwidthGBs":307,"count":1,"summary":"18-core CPU, 20-core GPU; 24 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/126318","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":24,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M5 Pro, 18-core CPU / 20-core GPU / 24 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/126318"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M5 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m5-pro-18cpu-20gpu-48gb","name":"MacBook Pro M5 Pro · 48 GB (20-core GPU)","maker":"Apple","memoryGiB":48,"reserveGiB":12,"type":"Unified memory","bandwidthGBs":307,"count":1,"summary":"18-core CPU, 20-core GPU; 48 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/126318","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":48,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M5 Pro, 18-core CPU / 20-core GPU / 48 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/126318"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M5 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m5-pro-18cpu-20gpu-64gb","name":"MacBook Pro M5 Pro · 64 GB (20-core GPU)","maker":"Apple","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":307,"count":1,"summary":"18-core CPU, 20-core GPU; 64 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/126318","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":64,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M5 Pro, 18-core CPU / 20-core GPU / 64 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/126318"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M5 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m5-max-18cpu-32gpu-36gb","name":"MacBook Pro M5 Max · 36 GB (32-core GPU)","maker":"Apple","memoryGiB":36,"reserveGiB":9,"type":"Unified memory","bandwidthGBs":460,"count":1,"summary":"18-core CPU, 32-core GPU; 36 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/126319","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":36,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M5 Max, 18-core CPU / 32-core GPU / 36 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/126319"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M5 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m5-max-18cpu-40gpu-48gb","name":"MacBook Pro M5 Max · 48 GB (40-core GPU)","maker":"Apple","memoryGiB":48,"reserveGiB":12,"type":"Unified memory","bandwidthGBs":614,"count":1,"summary":"18-core CPU, 40-core GPU; 48 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/126319","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":48,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M5 Max, 18-core CPU / 40-core GPU / 48 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/126319"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M5 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m5-max-18cpu-40gpu-64gb","name":"MacBook Pro M5 Max · 64 GB (40-core GPU)","maker":"Apple","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":614,"count":1,"summary":"18-core CPU, 40-core GPU; 64 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/126319","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":64,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M5 Max, 18-core CPU / 40-core GPU / 64 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/126319"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M5 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"macbook-pro-m5-max-18cpu-40gpu-128gb","name":"MacBook Pro M5 Max · 128 GB (40-core GPU)","maker":"Apple","memoryGiB":128,"reserveGiB":32,"type":"Unified memory","bandwidthGBs":614,"count":1,"summary":"18-core CPU, 40-core GPU; 128 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://support.apple.com/en-us/126319","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":128,"memoryType":"Unified memory","configuration":"MacBook Pro, Apple M5 Max, 18-core CPU / 40-core GPU / 128 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://support.apple.com/en-us/126319"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"}],"family":"MacBook Pro M5 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m5-max-18cpu-32gpu-36gb","name":"Mac Studio M5 Max · 36 GB (32-core GPU)","maker":"Apple","memoryGiB":36,"reserveGiB":9,"type":"Unified memory","bandwidthGBs":460,"count":1,"summary":"18-core CPU, 32-core GPU; 36 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://www.apple.com/mac-studio/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":36,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M5 Max, 18-core CPU / 32-core GPU / 36 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-studio/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/newsroom/2026/08/apple-introduces-new-mac-studio-with-m5-max-and-m5-ultra/"}],"availableFrom":"2026-09-22","family":"Mac Studio M5 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m5-max-18cpu-40gpu-48gb","name":"Mac Studio M5 Max · 48 GB (40-core GPU)","maker":"Apple","memoryGiB":48,"reserveGiB":12,"type":"Unified memory","bandwidthGBs":614,"count":1,"summary":"18-core CPU, 40-core GPU; 48 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://www.apple.com/mac-studio/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":48,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M5 Max, 18-core CPU / 40-core GPU / 48 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-studio/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/newsroom/2026/08/apple-introduces-new-mac-studio-with-m5-max-and-m5-ultra/"}],"availableFrom":"2026-09-22","family":"Mac Studio M5 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m5-max-18cpu-40gpu-64gb","name":"Mac Studio M5 Max · 64 GB (40-core GPU)","maker":"Apple","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":614,"count":1,"summary":"18-core CPU, 40-core GPU; 64 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://www.apple.com/mac-studio/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":64,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M5 Max, 18-core CPU / 40-core GPU / 64 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-studio/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/newsroom/2026/08/apple-introduces-new-mac-studio-with-m5-max-and-m5-ultra/"}],"availableFrom":"2026-09-22","family":"Mac Studio M5 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m5-max-18cpu-40gpu-128gb","name":"Mac Studio M5 Max · 128 GB (40-core GPU)","maker":"Apple","memoryGiB":128,"reserveGiB":32,"type":"Unified memory","bandwidthGBs":614,"count":1,"summary":"18-core CPU, 40-core GPU; 128 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://www.apple.com/mac-studio/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":128,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M5 Max, 18-core CPU / 40-core GPU / 128 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-studio/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/newsroom/2026/08/apple-introduces-new-mac-studio-with-m5-max-and-m5-ultra/"}],"availableFrom":"2026-09-22","family":"Mac Studio M5 Max","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m5-ultra-30cpu-64gpu-96gb","name":"Mac Studio M5 Ultra · 96 GB (64-core GPU)","maker":"Apple","memoryGiB":96,"reserveGiB":24,"type":"Unified memory","bandwidthGBs":1200,"count":1,"summary":"30-core CPU, 64-core GPU; 96 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://www.apple.com/mac-studio/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":96,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M5 Ultra, 30-core CPU / 64-core GPU / 96 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-studio/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/newsroom/2026/08/apple-introduces-new-mac-studio-with-m5-max-and-m5-ultra/"}],"availableFrom":"2026-09-22","family":"Mac Studio M5 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m5-ultra-30cpu-64gpu-256gb","name":"Mac Studio M5 Ultra · 256 GB (64-core GPU)","maker":"Apple","memoryGiB":256,"reserveGiB":64,"type":"Unified memory","bandwidthGBs":1200,"count":1,"summary":"30-core CPU, 64-core GPU; 256 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://www.apple.com/mac-studio/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":256,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M5 Ultra, 30-core CPU / 64-core GPU / 256 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-studio/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/newsroom/2026/08/apple-introduces-new-mac-studio-with-m5-max-and-m5-ultra/"}],"availableFrom":"2026-09-22","family":"Mac Studio M5 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m5-ultra-36cpu-80gpu-96gb","name":"Mac Studio M5 Ultra · 96 GB (80-core GPU)","maker":"Apple","memoryGiB":96,"reserveGiB":24,"type":"Unified memory","bandwidthGBs":1200,"count":1,"summary":"36-core CPU, 80-core GPU; 96 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://www.apple.com/mac-studio/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":96,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M5 Ultra, 36-core CPU / 80-core GPU / 96 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-studio/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/newsroom/2026/08/apple-introduces-new-mac-studio-with-m5-max-and-m5-ultra/"}],"availableFrom":"2026-09-22","family":"Mac Studio M5 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m5-ultra-36cpu-80gpu-256gb","name":"Mac Studio M5 Ultra · 256 GB (80-core GPU)","maker":"Apple","memoryGiB":256,"reserveGiB":64,"type":"Unified memory","bandwidthGBs":1200,"count":1,"summary":"36-core CPU, 80-core GPU; 256 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://www.apple.com/mac-studio/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":256,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M5 Ultra, 36-core CPU / 80-core GPU / 256 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-studio/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/newsroom/2026/08/apple-introduces-new-mac-studio-with-m5-max-and-m5-ultra/"}],"availableFrom":"2026-09-22","family":"Mac Studio M5 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-studio-m5-ultra-36cpu-80gpu-512gb","name":"Mac Studio M5 Ultra · 512 GB (80-core GPU)","maker":"Apple","memoryGiB":512,"reserveGiB":128,"type":"Unified memory","bandwidthGBs":1200,"count":1,"summary":"36-core CPU, 80-core GPU; 512 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. Apple says the 512 GB configuration is coming in late October.","source":"https://www.apple.com/mac-studio/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":512,"memoryType":"Unified memory","configuration":"Mac Studio, Apple M5 Ultra, 36-core CPU / 80-core GPU / 512 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-studio/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/newsroom/2026/08/apple-introduces-new-mac-studio-with-m5-max-and-m5-ultra/"}],"availableFrom":"2026-10","family":"Mac Studio M5 Ultra","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m6-12cpu-12gpu-16gb","name":"Mac mini M6 · 16 GB (12-core GPU)","maker":"Apple","memoryGiB":16,"reserveGiB":4,"type":"Unified memory","bandwidthGBs":153,"count":1,"summary":"12-core CPU, 12-core GPU; 16 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://www.apple.com/mac-mini/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":16,"memoryType":"Unified memory","configuration":"Mac mini, Apple M6, 12-core CPU / 12-core GPU / 16 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-mini/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/uk/newsroom/2026/08/apple-introduces-m6-and-m5-ultra-for-a-big-leap-in-performance-and-ai-compute/"}],"availableFrom":"2026-09-22","family":"Mac mini M6","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m6-12cpu-12gpu-24gb","name":"Mac mini M6 · 24 GB (12-core GPU)","maker":"Apple","memoryGiB":24,"reserveGiB":6,"type":"Unified memory","bandwidthGBs":170,"count":1,"summary":"12-core CPU, 12-core GPU; 24 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://www.apple.com/mac-mini/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":24,"memoryType":"Unified memory","configuration":"Mac mini, Apple M6, 12-core CPU / 12-core GPU / 24 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-mini/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/uk/newsroom/2026/08/apple-introduces-m6-and-m5-ultra-for-a-big-leap-in-performance-and-ai-compute/"}],"availableFrom":"2026-09-22","family":"Mac mini M6","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m6-12cpu-12gpu-32gb","name":"Mac mini M6 · 32 GB (12-core GPU)","maker":"Apple","memoryGiB":32,"reserveGiB":8,"type":"Unified memory","bandwidthGBs":170,"count":1,"summary":"12-core CPU, 12-core GPU; 32 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://www.apple.com/mac-mini/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":32,"memoryType":"Unified memory","configuration":"Mac mini, Apple M6, 12-core CPU / 12-core GPU / 32 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-mini/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/uk/newsroom/2026/08/apple-introduces-m6-and-m5-ultra-for-a-big-leap-in-performance-and-ai-compute/"}],"availableFrom":"2026-09-22","family":"Mac mini M6","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m5-pro-15cpu-16gpu-24gb","name":"Mac mini M5 Pro · 24 GB (16-core GPU)","maker":"Apple","memoryGiB":24,"reserveGiB":6,"type":"Unified memory","bandwidthGBs":307,"count":1,"summary":"15-core CPU, 16-core GPU; 24 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://www.apple.com/mac-mini/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":24,"memoryType":"Unified memory","configuration":"Mac mini, Apple M5 Pro, 15-core CPU / 16-core GPU / 24 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-mini/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/uk/newsroom/2026/08/apple-introduces-m6-and-m5-ultra-for-a-big-leap-in-performance-and-ai-compute/"}],"availableFrom":"2026-09-22","family":"Mac mini M5 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m5-pro-15cpu-16gpu-48gb","name":"Mac mini M5 Pro · 48 GB (16-core GPU)","maker":"Apple","memoryGiB":48,"reserveGiB":12,"type":"Unified memory","bandwidthGBs":307,"count":1,"summary":"15-core CPU, 16-core GPU; 48 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://www.apple.com/mac-mini/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":48,"memoryType":"Unified memory","configuration":"Mac mini, Apple M5 Pro, 15-core CPU / 16-core GPU / 48 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-mini/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/uk/newsroom/2026/08/apple-introduces-m6-and-m5-ultra-for-a-big-leap-in-performance-and-ai-compute/"}],"availableFrom":"2026-09-22","family":"Mac mini M5 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m5-pro-15cpu-16gpu-64gb","name":"Mac mini M5 Pro · 64 GB (16-core GPU)","maker":"Apple","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":307,"count":1,"summary":"15-core CPU, 16-core GPU; 64 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://www.apple.com/mac-mini/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":64,"memoryType":"Unified memory","configuration":"Mac mini, Apple M5 Pro, 15-core CPU / 16-core GPU / 64 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-mini/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/uk/newsroom/2026/08/apple-introduces-m6-and-m5-ultra-for-a-big-leap-in-performance-and-ai-compute/"}],"availableFrom":"2026-09-22","family":"Mac mini M5 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m5-pro-18cpu-20gpu-24gb","name":"Mac mini M5 Pro · 24 GB (20-core GPU)","maker":"Apple","memoryGiB":24,"reserveGiB":6,"type":"Unified memory","bandwidthGBs":307,"count":1,"summary":"18-core CPU, 20-core GPU; 24 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://www.apple.com/mac-mini/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":24,"memoryType":"Unified memory","configuration":"Mac mini, Apple M5 Pro, 18-core CPU / 20-core GPU / 24 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-mini/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/uk/newsroom/2026/08/apple-introduces-m6-and-m5-ultra-for-a-big-leap-in-performance-and-ai-compute/"}],"availableFrom":"2026-09-22","family":"Mac mini M5 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m5-pro-18cpu-20gpu-48gb","name":"Mac mini M5 Pro · 48 GB (20-core GPU)","maker":"Apple","memoryGiB":48,"reserveGiB":12,"type":"Unified memory","bandwidthGBs":307,"count":1,"summary":"18-core CPU, 20-core GPU; 48 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://www.apple.com/mac-mini/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":48,"memoryType":"Unified memory","configuration":"Mac mini, Apple M5 Pro, 18-core CPU / 20-core GPU / 48 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-mini/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/uk/newsroom/2026/08/apple-introduces-m6-and-m5-ultra-for-a-big-leap-in-performance-and-ai-compute/"}],"availableFrom":"2026-09-22","family":"Mac mini M5 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"mac-mini-m5-pro-18cpu-20gpu-64gb","name":"Mac mini M5 Pro · 64 GB (20-core GPU)","maker":"Apple","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":307,"count":1,"summary":"18-core CPU, 20-core GPU; 64 GB shared by macOS, CPU and GPU.","caveat":"Use Metal or MLX with support for the chosen model. The OS and other apps share memory; Metal working-set limits vary. ","source":"https://www.apple.com/mac-mini/specs/","reviewedAt":"2026-09-06","category":"Apple silicon","memoryPerDeviceGiB":64,"memoryType":"Unified memory","configuration":"Mac mini, Apple M5 Pro, 18-core CPU / 20-core GPU / 64 GB unified memory","interconnect":"On-chip unified memory; no discrete VRAM pool","availability":"Announced","bandwidthKind":"Peak theoretical, decimal GB/s for the whole chip","bandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference.","reserveAssumption":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtimeSupport":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","sources":[{"title":"Manufacturer source 1","url":"https://www.apple.com/mac-mini/specs/"},{"title":"Manufacturer source 2","url":"https://developer.apple.com/documentation/metal/mtldevice/recommendedmaxworkingsetsize"},{"title":"Manufacturer source 3","url":"https://www.apple.com/uk/newsroom/2026/08/apple-introduces-m6-and-m5-ultra-for-a-big-leap-in-performance-and-ai-compute/"}],"availableFrom":"2026-09-22","family":"Mac mini M5 Pro","segment":"Personal computer","reserveNote":"Planner assumption: 25% of installed memory reserved for macOS, other apps and allocation headroom. This is adjustable, not an Apple GPU-allocation guarantee.","runtime":"Metal / MLX. Actual usable memory is runtime- and OS-dependent; query recommendedMaxWorkingSetSize rather than treating all RAM as VRAM.","memoryBandwidthNote":"Apple-published chip bandwidth; shared CPU/GPU traffic reduces bandwidth available to inference."},{"id":"ryzen-ai-max-plus-395-32gb","name":"Ryzen AI Max+ 395 · 32 GB","maker":"AMD","memoryGiB":32,"reserveGiB":8,"type":"Unified memory","bandwidthGBs":256,"count":1,"summary":"32 GB shared system memory.","caveat":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","source":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-395.html","reviewedAt":"2026-09-06","category":"Unified-memory PC","memoryPerDeviceGiB":32,"memoryType":"LPDDR5X","configuration":"Ryzen AI Max+ 395 / Radeon 8060S, 16 CPU cores / 40 GPU CUs, 32 GB LPDDR5X-8000, 256-bit memory interface","interconnect":"CPU/GPU shared memory","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole memory interface","bandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate.","reserveAssumption":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtimeSupport":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-395.html"},{"title":"Manufacturer source 2","url":"https://www.amd.com/en/blogs/2025/amd-ryzen-ai-max-395-processor-breakthrough-ai-.html"}],"powerRangeWatts":[45,120],"powerBasis":"AMD configurable SoC TDP range; not whole-system or wall power","family":"Ryzen AI Max+ 395","segment":"Personal computer","reserveNote":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtime":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","memoryBandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate."},{"id":"ryzen-ai-max-plus-395-64gb","name":"Ryzen AI Max+ 395 · 64 GB","maker":"AMD","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":256,"count":1,"summary":"64 GB shared system memory.","caveat":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","source":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-395.html","reviewedAt":"2026-09-06","category":"Unified-memory PC","memoryPerDeviceGiB":64,"memoryType":"LPDDR5X","configuration":"Ryzen AI Max+ 395 / Radeon 8060S, 16 CPU cores / 40 GPU CUs, 64 GB LPDDR5X-8000, 256-bit memory interface","interconnect":"CPU/GPU shared memory","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole memory interface","bandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate.","reserveAssumption":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtimeSupport":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-395.html"},{"title":"Manufacturer source 2","url":"https://www.amd.com/en/blogs/2025/amd-ryzen-ai-max-395-processor-breakthrough-ai-.html"}],"powerRangeWatts":[45,120],"powerBasis":"AMD configurable SoC TDP range; not whole-system or wall power","family":"Ryzen AI Max+ 395","segment":"Personal computer","reserveNote":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtime":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","memoryBandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate."},{"id":"ryzen-ai-max-plus-395-128gb","name":"Ryzen AI Max+ 395 · 128 GB","maker":"AMD","memoryGiB":128,"reserveGiB":32,"type":"Unified memory","bandwidthGBs":256,"count":1,"summary":"128 GB shared system memory.","caveat":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","source":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-395.html","reviewedAt":"2026-09-06","category":"Unified-memory PC","memoryPerDeviceGiB":128,"memoryType":"LPDDR5X","configuration":"Ryzen AI Max+ 395 / Radeon 8060S, 16 CPU cores / 40 GPU CUs, 128 GB LPDDR5X-8000, 256-bit memory interface","interconnect":"CPU/GPU shared memory","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole memory interface","bandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate.","reserveAssumption":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtimeSupport":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-395.html"},{"title":"Manufacturer source 2","url":"https://www.amd.com/en/blogs/2025/amd-ryzen-ai-max-395-processor-breakthrough-ai-.html"}],"powerRangeWatts":[45,120],"powerBasis":"AMD configurable SoC TDP range; not whole-system or wall power","family":"Ryzen AI Max+ 395","segment":"Personal computer","reserveNote":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtime":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","memoryBandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate."},{"id":"ryzen-ai-max-plus-392-32gb","name":"Ryzen AI Max+ 392 · 32 GB","maker":"AMD","memoryGiB":32,"reserveGiB":8,"type":"Unified memory","bandwidthGBs":256,"count":1,"summary":"32 GB shared system memory.","caveat":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","source":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-392.html","reviewedAt":"2026-09-06","category":"Unified-memory PC","memoryPerDeviceGiB":32,"memoryType":"LPDDR5X","configuration":"Ryzen AI Max+ 392 / Radeon 8060S, 12 CPU cores / 40 GPU CUs, 32 GB LPDDR5X-8000, 256-bit memory interface","interconnect":"CPU/GPU shared memory","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole memory interface","bandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate.","reserveAssumption":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtimeSupport":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-392.html"},{"title":"Manufacturer source 2","url":"https://www.amd.com/en/blogs/2025/amd-ryzen-ai-max-395-processor-breakthrough-ai-.html"}],"powerRangeWatts":[45,120],"powerBasis":"AMD configurable SoC TDP range; not whole-system or wall power","family":"Ryzen AI Max+ 392","segment":"Personal computer","reserveNote":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtime":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","memoryBandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate."},{"id":"ryzen-ai-max-plus-392-64gb","name":"Ryzen AI Max+ 392 · 64 GB","maker":"AMD","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":256,"count":1,"summary":"64 GB shared system memory.","caveat":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","source":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-392.html","reviewedAt":"2026-09-06","category":"Unified-memory PC","memoryPerDeviceGiB":64,"memoryType":"LPDDR5X","configuration":"Ryzen AI Max+ 392 / Radeon 8060S, 12 CPU cores / 40 GPU CUs, 64 GB LPDDR5X-8000, 256-bit memory interface","interconnect":"CPU/GPU shared memory","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole memory interface","bandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate.","reserveAssumption":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtimeSupport":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-392.html"},{"title":"Manufacturer source 2","url":"https://www.amd.com/en/blogs/2025/amd-ryzen-ai-max-395-processor-breakthrough-ai-.html"}],"powerRangeWatts":[45,120],"powerBasis":"AMD configurable SoC TDP range; not whole-system or wall power","family":"Ryzen AI Max+ 392","segment":"Personal computer","reserveNote":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtime":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","memoryBandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate."},{"id":"ryzen-ai-max-plus-392-128gb","name":"Ryzen AI Max+ 392 · 128 GB","maker":"AMD","memoryGiB":128,"reserveGiB":32,"type":"Unified memory","bandwidthGBs":256,"count":1,"summary":"128 GB shared system memory.","caveat":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","source":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-392.html","reviewedAt":"2026-09-06","category":"Unified-memory PC","memoryPerDeviceGiB":128,"memoryType":"LPDDR5X","configuration":"Ryzen AI Max+ 392 / Radeon 8060S, 12 CPU cores / 40 GPU CUs, 128 GB LPDDR5X-8000, 256-bit memory interface","interconnect":"CPU/GPU shared memory","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole memory interface","bandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate.","reserveAssumption":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtimeSupport":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-392.html"},{"title":"Manufacturer source 2","url":"https://www.amd.com/en/blogs/2025/amd-ryzen-ai-max-395-processor-breakthrough-ai-.html"}],"powerRangeWatts":[45,120],"powerBasis":"AMD configurable SoC TDP range; not whole-system or wall power","family":"Ryzen AI Max+ 392","segment":"Personal computer","reserveNote":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtime":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","memoryBandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate."},{"id":"ryzen-ai-max-plus-388-32gb","name":"Ryzen AI Max+ 388 · 32 GB","maker":"AMD","memoryGiB":32,"reserveGiB":8,"type":"Unified memory","bandwidthGBs":256,"count":1,"summary":"32 GB shared system memory.","caveat":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","source":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-388.html","reviewedAt":"2026-09-06","category":"Unified-memory PC","memoryPerDeviceGiB":32,"memoryType":"LPDDR5X","configuration":"Ryzen AI Max+ 388 / Radeon 8060S, 8 CPU cores / 40 GPU CUs, 32 GB LPDDR5X-8000, 256-bit memory interface","interconnect":"CPU/GPU shared memory","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole memory interface","bandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate.","reserveAssumption":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtimeSupport":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-388.html"},{"title":"Manufacturer source 2","url":"https://www.amd.com/en/blogs/2025/amd-ryzen-ai-max-395-processor-breakthrough-ai-.html"}],"powerRangeWatts":[45,120],"powerBasis":"AMD configurable SoC TDP range; not whole-system or wall power","family":"Ryzen AI Max+ 388","segment":"Personal computer","reserveNote":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtime":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","memoryBandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate."},{"id":"ryzen-ai-max-plus-388-64gb","name":"Ryzen AI Max+ 388 · 64 GB","maker":"AMD","memoryGiB":64,"reserveGiB":16,"type":"Unified memory","bandwidthGBs":256,"count":1,"summary":"64 GB shared system memory.","caveat":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","source":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-388.html","reviewedAt":"2026-09-06","category":"Unified-memory PC","memoryPerDeviceGiB":64,"memoryType":"LPDDR5X","configuration":"Ryzen AI Max+ 388 / Radeon 8060S, 8 CPU cores / 40 GPU CUs, 64 GB LPDDR5X-8000, 256-bit memory interface","interconnect":"CPU/GPU shared memory","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole memory interface","bandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate.","reserveAssumption":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtimeSupport":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-388.html"},{"title":"Manufacturer source 2","url":"https://www.amd.com/en/blogs/2025/amd-ryzen-ai-max-395-processor-breakthrough-ai-.html"}],"powerRangeWatts":[45,120],"powerBasis":"AMD configurable SoC TDP range; not whole-system or wall power","family":"Ryzen AI Max+ 388","segment":"Personal computer","reserveNote":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtime":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","memoryBandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate."},{"id":"ryzen-ai-max-plus-388-128gb","name":"Ryzen AI Max+ 388 · 128 GB","maker":"AMD","memoryGiB":128,"reserveGiB":32,"type":"Unified memory","bandwidthGBs":256,"count":1,"summary":"128 GB shared system memory.","caveat":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","source":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-388.html","reviewedAt":"2026-09-06","category":"Unified-memory PC","memoryPerDeviceGiB":128,"memoryType":"LPDDR5X","configuration":"Ryzen AI Max+ 388 / Radeon 8060S, 8 CPU cores / 40 GPU CUs, 128 GB LPDDR5X-8000, 256-bit memory interface","interconnect":"CPU/GPU shared memory","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole memory interface","bandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate.","reserveAssumption":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtimeSupport":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/processors/laptop/ryzen/ai-300-series/amd-ryzen-ai-max-plus-388.html"},{"title":"Manufacturer source 2","url":"https://www.amd.com/en/blogs/2025/amd-ryzen-ai-max-395-processor-breakthrough-ai-.html"}],"powerRangeWatts":[45,120],"powerBasis":"AMD configurable SoC TDP range; not whole-system or wall power","family":"Ryzen AI Max+ 388","segment":"Personal computer","reserveNote":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtime":"Memory configuration planning profile: assumes LPDDR5X-8000 across the full 256-bit interface: 8000 MT/s × 256 bits ÷ 8 = 256 GB/s. Check OEM availability of the selected installed capacity; memory speed and power limits vary. AMD Variable Graphics Memory exposes up to 96 GB of a 128 GB configuration; verify OS and runtime support.","memoryBandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate."},{"id":"dgx-spark-128gb","name":"NVIDIA DGX Spark · 128 GB","maker":"NVIDIA","memoryGiB":128,"reserveGiB":32,"type":"Unified memory","bandwidthGBs":273,"count":1,"summary":"128 GB shared system memory.","caveat":"Arm64 Linux and GB10 require compatible CUDA packages. The CPU, OS and GPU share the 128 GB pool. 140 W is chip TDP; the external supply is rated at 240 W.","source":"https://docs.nvidia.com/dgx/dgx-spark/system-overview.html","reviewedAt":"2026-09-06","category":"Unified-memory PC","memoryPerDeviceGiB":128,"memoryType":"LPDDR5X","configuration":"DGX Spark GB10, 20-core Arm CPU / Blackwell GPU / 128 GB coherent LPDDR5X, 256-bit interface","interconnect":"CPU/GPU shared memory","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s for the whole memory interface","bandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate.","reserveAssumption":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtimeSupport":"Arm64 Linux and GB10 require compatible CUDA packages. The CPU, OS and GPU share the 128 GB pool. 140 W is chip TDP; the external supply is rated at 240 W.","sources":[{"title":"Manufacturer source 1","url":"https://docs.nvidia.com/dgx/dgx-spark/system-overview.html"}],"powerWatts":140,"powerBasis":"SoC TDP; not whole-system or wall power","family":"NVIDIA DGX Spark","segment":"Personal computer","reserveNote":"Planner assumption: reserve 25% for the operating system, CPU workloads and allocation headroom; runtime-specific GPU memory limits also apply.","runtime":"Arm64 Linux and GB10 require compatible CUDA packages. The CPU, OS and GPU share the 128 GB pool. 140 W is chip TDP; the external supply is rated at 240 W.","memoryBandwidthNote":"Shared memory-controller bandwidth; not a measured sustained inference rate."},{"id":"dual-rtx-3090","name":"2 × GeForce RTX 3090","maker":"NVIDIA","memoryGiB":48,"reserveGiB":2,"type":"Multi-GPU","bandwidthGBs":936,"count":2,"summary":"2 separate 24 GB memory pools; 48 GB total installed.","caveat":"Combined capacity is a planning upper bound. The runtime must split weights and KV cache; each device must fit its own layers and buffers. Layer splitting does not multiply single-request memory bandwidth. Host PCIe lanes, P2P support, cooling and power must be validated.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3090-3090ti/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":24,"memoryType":"GDDR6X","configuration":"2 identical GeForce RTX 3090 devices, 24 GiB nominal memory each; runtime-managed model splitting","interconnect":"PCIe 4.0; optional two-card NVLink bridge, not assumed","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency.","reserveAssumption":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3090-3090ti/"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/content/PDF/nvidia-ampere-ga-102-gpu-architecture-whitepaper-v2.pdf"}],"powerWatts":700,"powerBasis":"Sum of device maximum/reference board power; host, cooling and power-supply losses excluded","family":"2 × GeForce RTX 3090","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency."},{"id":"4x-rtx-3090","name":"4 × GeForce RTX 3090","maker":"NVIDIA","memoryGiB":96,"reserveGiB":4,"type":"Multi-GPU","bandwidthGBs":936,"count":4,"summary":"4 separate 24 GB memory pools; 96 GB total installed.","caveat":"Combined capacity is a planning upper bound. The runtime must split weights and KV cache; each device must fit its own layers and buffers. Layer splitting does not multiply single-request memory bandwidth. Host PCIe lanes, P2P support, cooling and power must be validated.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3090-3090ti/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":24,"memoryType":"GDDR6X","configuration":"4 identical GeForce RTX 3090 devices, 24 GiB nominal memory each; runtime-managed model splitting","interconnect":"Host PCIe; NVLink can bridge pairs, not create a four-card fully connected fabric","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency.","reserveAssumption":"Planner assumption: 1 GiB reserve on each of 4 devices; placement and duplicated buffers can require additional memory.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3090-3090ti/"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/content/PDF/nvidia-ampere-ga-102-gpu-architecture-whitepaper-v2.pdf"}],"powerWatts":1400,"powerBasis":"Sum of device maximum/reference board power; host, cooling and power-supply losses excluded","family":"4 × GeForce RTX 3090","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB reserve on each of 4 devices; placement and duplicated buffers can require additional memory.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency."},{"id":"2x-rtx-4090","name":"2 × GeForce RTX 4090","maker":"NVIDIA","memoryGiB":48,"reserveGiB":2,"type":"Multi-GPU","bandwidthGBs":1008,"count":2,"summary":"2 separate 24 GB memory pools; 48 GB total installed.","caveat":"Combined capacity is a planning upper bound. The runtime must split weights and KV cache; each device must fit its own layers and buffers. Layer splitting does not multiply single-request memory bandwidth. Host PCIe lanes, P2P support, cooling and power must be validated.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/40-series/rtx-4090/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":24,"memoryType":"GDDR6X","configuration":"2 identical GeForce RTX 4090 devices, 24 GiB nominal memory each; runtime-managed model splitting","interconnect":"Host PCIe 4.0; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency.","reserveAssumption":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/40-series/rtx-4090/"},{"title":"Manufacturer source 2","url":"https://images.nvidia.cn/aem-dam/Solutions/geforce/ada/nvidia-ada-gpu-architecture.pdf"}],"powerWatts":900,"powerBasis":"Sum of device maximum/reference board power; host, cooling and power-supply losses excluded","family":"2 × GeForce RTX 4090","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency."},{"id":"2x-rtx-5090","name":"2 × GeForce RTX 5090","maker":"NVIDIA","memoryGiB":64,"reserveGiB":2,"type":"Multi-GPU","bandwidthGBs":1792,"count":2,"summary":"2 separate 32 GB memory pools; 64 GB total installed.","caveat":"Combined capacity is a planning upper bound. The runtime must split weights and KV cache; each device must fit its own layers and buffers. Layer splitting does not multiply single-request memory bandwidth. Host PCIe lanes, P2P support, cooling and power must be validated.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":32,"memoryType":"GDDR7","configuration":"2 identical GeForce RTX 5090 devices, 32 GiB nominal memory each; runtime-managed model splitting","interconnect":"Host PCIe 5.0; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency.","reserveAssumption":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":1150,"powerBasis":"Sum of device maximum/reference board power; host, cooling and power-supply losses excluded","family":"2 × GeForce RTX 5090","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency."},{"id":"4x-rtx-5090","name":"4 × GeForce RTX 5090","maker":"NVIDIA","memoryGiB":128,"reserveGiB":4,"type":"Multi-GPU","bandwidthGBs":1792,"count":4,"summary":"4 separate 32 GB memory pools; 128 GB total installed.","caveat":"Combined capacity is a planning upper bound. The runtime must split weights and KV cache; each device must fit its own layers and buffers. Layer splitting does not multiply single-request memory bandwidth. Host PCIe lanes, P2P support, cooling and power must be validated.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":32,"memoryType":"GDDR7","configuration":"4 identical GeForce RTX 5090 devices, 32 GiB nominal memory each; runtime-managed model splitting","interconnect":"Host PCIe 5.0; no NVLink; workstation/server with enough lanes and power required","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency.","reserveAssumption":"Planner assumption: 1 GiB reserve on each of 4 devices; placement and duplicated buffers can require additional memory.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":2300,"powerBasis":"Sum of device maximum/reference board power; host, cooling and power-supply losses excluded","family":"4 × GeForce RTX 5090","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB reserve on each of 4 devices; placement and duplicated buffers can require additional memory.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency."},{"id":"2x-rtx-5060-ti-16gb","name":"2 × GeForce RTX 5060 Ti 16 GB","maker":"NVIDIA","memoryGiB":32,"reserveGiB":2,"type":"Multi-GPU","bandwidthGBs":448,"count":2,"summary":"2 separate 16 GB memory pools; 32 GB total installed.","caveat":"Combined capacity is a planning upper bound. The runtime must split weights and KV cache; each device must fit its own layers and buffers. Layer splitting does not multiply single-request memory bandwidth. Host PCIe lanes, P2P support, cooling and power must be validated.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":16,"memoryType":"GDDR7","configuration":"2 identical GeForce RTX 5060 Ti 16 GB devices, 16 GiB nominal memory each; runtime-managed model splitting","interconnect":"Host PCIe; each GPU has a PCIe 5.0 interface","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency.","reserveAssumption":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/compare/"}],"powerWatts":360,"powerBasis":"Sum of device maximum/reference board power; host, cooling and power-supply losses excluded","family":"2 × GeForce RTX 5060 Ti 16 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency."},{"id":"2x-rtx-a6000","name":"2 × RTX A6000","maker":"NVIDIA","memoryGiB":96,"reserveGiB":2,"type":"Multi-GPU","bandwidthGBs":768,"count":2,"summary":"2 separate 48 GB memory pools; 96 GB total installed.","caveat":"Combined capacity is a planning upper bound. The runtime must split weights and KV cache; each device must fit its own layers and buffers. Layer splitting does not multiply single-request memory bandwidth. Host PCIe lanes, P2P support, cooling and power must be validated.","source":"https://www.nvidia.com/content/dam/en-zz/Solutions/design-visualization/quadro-product-literature/proviz-print-nvidia-rtx-a6000-datasheet-us-nvidia-1454980-r9-web%20%281%29.pdf","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":48,"memoryType":"GDDR6 ECC","configuration":"2 identical RTX A6000 devices, 48 GiB nominal memory each; runtime-managed model splitting","interconnect":"Optional two-card NVLink bridge 112.5 GB/s bidirectional; otherwise host PCIe","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency.","reserveAssumption":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/content/dam/en-zz/Solutions/design-visualization/quadro-product-literature/proviz-print-nvidia-rtx-a6000-datasheet-us-nvidia-1454980-r9-web%20%281%29.pdf"}],"powerWatts":600,"powerBasis":"Sum of device maximum/reference board power; host, cooling and power-supply losses excluded","family":"2 × RTX A6000","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency."},{"id":"2x-rtx-pro-5000-blackwell-72gb","name":"2 × RTX PRO 5000 Blackwell 72 GB","maker":"NVIDIA","memoryGiB":144,"reserveGiB":2,"type":"Multi-GPU","bandwidthGBs":1344,"count":2,"summary":"2 separate 72 GB memory pools; 144 GB total installed.","caveat":"Combined capacity is a planning upper bound. The runtime must split weights and KV cache; each device must fit its own layers and buffers. Layer splitting does not multiply single-request memory bandwidth. Host PCIe lanes, P2P support, cooling and power must be validated.","source":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-5000/","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":72,"memoryType":"GDDR7 ECC","configuration":"2 identical RTX PRO 5000 Blackwell 72 GB devices, 72 GiB nominal memory each; runtime-managed model splitting","interconnect":"PCIe 5.0 x16; full height, dual slot","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency.","reserveAssumption":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-5000/"}],"powerWatts":600,"powerBasis":"Sum of device maximum/reference board power; host, cooling and power-supply losses excluded","family":"2 × RTX PRO 5000 Blackwell 72 GB","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency."},{"id":"2x-rtx-pro-6000-blackwell-max-q","name":"2 × RTX PRO 6000 Blackwell Max-Q","maker":"NVIDIA","memoryGiB":192,"reserveGiB":2,"type":"Multi-GPU","bandwidthGBs":1792,"count":2,"summary":"2 separate 96 GB memory pools; 192 GB total installed.","caveat":"Combined capacity is a planning upper bound. The runtime must split weights and KV cache; each device must fit its own layers and buffers. Layer splitting does not multiply single-request memory bandwidth. Host PCIe lanes, P2P support, cooling and power must be validated.","source":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-6000-max-q/","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":96,"memoryType":"GDDR7 ECC","configuration":"2 identical RTX PRO 6000 Blackwell Max-Q devices, 96 GiB nominal memory each; runtime-managed model splitting","interconnect":"PCIe 5.0 x16; full height, dual slot","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency.","reserveAssumption":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-6000-max-q/"}],"powerWatts":600,"powerBasis":"Sum of device maximum/reference board power; host, cooling and power-supply losses excluded","family":"2 × RTX PRO 6000 Blackwell Max-Q","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency."},{"id":"4x-rtx-pro-6000-blackwell-max-q","name":"4 × RTX PRO 6000 Blackwell Max-Q","maker":"NVIDIA","memoryGiB":384,"reserveGiB":4,"type":"Multi-GPU","bandwidthGBs":1792,"count":4,"summary":"4 separate 96 GB memory pools; 384 GB total installed.","caveat":"Combined capacity is a planning upper bound. The runtime must split weights and KV cache; each device must fit its own layers and buffers. Layer splitting does not multiply single-request memory bandwidth. Host PCIe lanes, P2P support, cooling and power must be validated.","source":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-6000-max-q/","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":96,"memoryType":"GDDR7 ECC","configuration":"4 identical RTX PRO 6000 Blackwell Max-Q devices, 96 GiB nominal memory each; runtime-managed model splitting","interconnect":"PCIe 5.0 x16; full height, dual slot","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency.","reserveAssumption":"Planner assumption: 1 GiB reserve on each of 4 devices; placement and duplicated buffers can require additional memory.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-6000-max-q/"}],"powerWatts":1200,"powerBasis":"Sum of device maximum/reference board power; host, cooling and power-supply losses excluded","family":"4 × RTX PRO 6000 Blackwell Max-Q","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB reserve on each of 4 devices; placement and duplicated buffers can require additional memory.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency."},{"id":"2x-radeon-ai-pro-r9700","name":"2 × Radeon AI PRO R9700","maker":"AMD","memoryGiB":64,"reserveGiB":2,"type":"Multi-GPU","bandwidthGBs":640,"count":2,"summary":"2 separate 32 GB memory pools; 64 GB total installed.","caveat":"Combined capacity is a planning upper bound. The runtime must split weights and KV cache; each device must fit its own layers and buffers. Layer splitting does not multiply single-request memory bandwidth. Host PCIe lanes, P2P support, cooling and power must be validated.","source":"https://www.amd.com/en/products/graphics/workstations/radeon-ai-pro/ai-9000-series/amd-radeon-ai-pro-r9700.html","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":32,"memoryType":"GDDR6","configuration":"2 identical Radeon AI PRO R9700 devices, 32 GiB nominal memory each; runtime-managed model splitting","interconnect":"PCIe 5.0 x16","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency.","reserveAssumption":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/workstations/radeon-ai-pro/ai-9000-series/amd-radeon-ai-pro-r9700.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":600,"powerBasis":"Sum of device maximum/reference board power; host, cooling and power-supply losses excluded","family":"2 × Radeon AI PRO R9700","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency."},{"id":"4x-radeon-ai-pro-r9700","name":"4 × Radeon AI PRO R9700","maker":"AMD","memoryGiB":128,"reserveGiB":4,"type":"Multi-GPU","bandwidthGBs":640,"count":4,"summary":"4 separate 32 GB memory pools; 128 GB total installed.","caveat":"Combined capacity is a planning upper bound. The runtime must split weights and KV cache; each device must fit its own layers and buffers. Layer splitting does not multiply single-request memory bandwidth. Host PCIe lanes, P2P support, cooling and power must be validated.","source":"https://www.amd.com/en/products/graphics/workstations/radeon-ai-pro/ai-9000-series/amd-radeon-ai-pro-r9700.html","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":32,"memoryType":"GDDR6","configuration":"4 identical Radeon AI PRO R9700 devices, 32 GiB nominal memory each; runtime-managed model splitting","interconnect":"PCIe 5.0 x16","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency.","reserveAssumption":"Planner assumption: 1 GiB reserve on each of 4 devices; placement and duplicated buffers can require additional memory.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/workstations/radeon-ai-pro/ai-9000-series/amd-radeon-ai-pro-r9700.html"},{"title":"Manufacturer source 2","url":"https://rocm.docs.amd.com/projects/radeon/en/latest/docs/compatibility.html"}],"powerWatts":1200,"powerBasis":"Sum of device maximum/reference board power; host, cooling and power-supply losses excluded","family":"4 × Radeon AI PRO R9700","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB reserve on each of 4 devices; placement and duplicated buffers can require additional memory.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency."},{"id":"2x-arc-pro-b70","name":"2 × Arc Pro B70","maker":"Intel","memoryGiB":64,"reserveGiB":2,"type":"Multi-GPU","bandwidthGBs":608,"count":2,"summary":"2 separate 32 GB memory pools; 64 GB total installed.","caveat":"Combined capacity is a planning upper bound. The runtime must split weights and KV cache; each device must fit its own layers and buffers. Layer splitting does not multiply single-request memory bandwidth. Host PCIe lanes, P2P support, cooling and power must be validated.","source":"https://www.intel.com/content/dam/www/central-libraries/us/en/documents/2026-03/intel-arc-pro-b-series-graphics-quick-reference-guide-v1-0.pdf","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":32,"memoryType":"GDDR6","configuration":"2 identical Arc Pro B70 devices, 32 GiB nominal memory each; runtime-managed model splitting","interconnect":"PCIe; topology and lane width depend on host","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency.","reserveAssumption":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtimeSupport":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","sources":[{"title":"Manufacturer source 1","url":"https://www.intel.com/content/dam/www/central-libraries/us/en/documents/2026-03/intel-arc-pro-b-series-graphics-quick-reference-guide-v1-0.pdf"},{"title":"Manufacturer source 2","url":"https://www.intel.com/content/www/us/en/ark/products/series/242616/intel-arc-pro-b-series-graphics.html"}],"powerRangeWatts":[160,290],"family":"2 × Arc Pro B70","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtime":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","memoryBandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency."},{"id":"4x-arc-pro-b70","name":"4 × Arc Pro B70","maker":"Intel","memoryGiB":128,"reserveGiB":4,"type":"Multi-GPU","bandwidthGBs":608,"count":4,"summary":"4 separate 32 GB memory pools; 128 GB total installed.","caveat":"Combined capacity is a planning upper bound. The runtime must split weights and KV cache; each device must fit its own layers and buffers. Layer splitting does not multiply single-request memory bandwidth. Host PCIe lanes, P2P support, cooling and power must be validated.","source":"https://www.intel.com/content/dam/www/central-libraries/us/en/documents/2026-03/intel-arc-pro-b-series-graphics-quick-reference-guide-v1-0.pdf","reviewedAt":"2026-09-06","category":"Workstation GPU","memoryPerDeviceGiB":32,"memoryType":"GDDR6","configuration":"4 identical Arc Pro B70 devices, 32 GiB nominal memory each; runtime-managed model splitting","interconnect":"PCIe; topology and lane width depend on host","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency.","reserveAssumption":"Planner assumption: 1 GiB reserve on each of 4 devices; placement and duplicated buffers can require additional memory.","runtimeSupport":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","sources":[{"title":"Manufacturer source 1","url":"https://www.intel.com/content/dam/www/central-libraries/us/en/documents/2026-03/intel-arc-pro-b-series-graphics-quick-reference-guide-v1-0.pdf"},{"title":"Manufacturer source 2","url":"https://www.intel.com/content/www/us/en/ark/products/series/242616/intel-arc-pro-b-series-graphics.html"}],"powerRangeWatts":[160,290],"family":"4 × Arc Pro B70","segment":"Workstation","reserveNote":"Planner assumption: 1 GiB reserve on each of 4 devices; placement and duplicated buffers can require additional memory.","runtime":"Use an Intel-compatible backend such as SYCL, OpenVINO or Vulkan; model and quantization support varies by backend.","memoryBandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency."},{"id":"2x-mi210-pcie","name":"2 × Instinct MI210 PCIe","maker":"AMD","memoryGiB":128,"reserveGiB":4,"type":"Multi-GPU","bandwidthGBs":1600,"count":2,"summary":"2 separate 64 GB memory pools; 128 GB total installed.","caveat":"Combined capacity is a planning upper bound. The runtime must split weights and KV cache; each device must fit its own layers and buffers. Layer splitting does not multiply single-request memory bandwidth. Host PCIe lanes, P2P support, cooling and power must be validated.","source":"https://www.amd.com/en/products/accelerators/instinct/mi200/mi210.html","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":64,"memoryType":"HBM2e ECC","configuration":"2 identical Instinct MI210 PCIe devices, 64 GiB nominal memory each; runtime-managed model splitting","interconnect":"PCIe 4.0; optional Infinity Fabric bridge in supported server","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency.","reserveAssumption":"Planner assumption: 2 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtimeSupport":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/accelerators/instinct/mi200/mi210.html"},{"title":"Manufacturer source 2","url":"https://instinct.docs.amd.com/projects/system-acceptance/en/latest/gpus/mi210.html"}],"powerWatts":600,"powerBasis":"Sum of device maximum/reference board power; host, cooling and power-supply losses excluded","family":"2 × Instinct MI210 PCIe","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtime":"ROCm/HIP support depends on the exact GPU, OS and runtime release; Vulkan is another backend where supported.","memoryBandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency."},{"id":"2x-h200-nvl","name":"2 × H200 NVL 141 GB PCIe","maker":"NVIDIA","memoryGiB":282,"reserveGiB":4,"type":"Multi-GPU","bandwidthGBs":4800,"count":2,"summary":"2 separate 141 GB memory pools; 282 GB total installed.","caveat":"Combined capacity is a planning upper bound. The runtime must split weights and KV cache; each device must fit its own layers and buffers. Layer splitting does not multiply single-request memory bandwidth. Host PCIe lanes, P2P support, cooling and power must be validated.","source":"https://www.nvidia.com/en-us/data-center/h200/","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":141,"memoryType":"HBM3e ECC","configuration":"2 identical H200 NVL 141 GB PCIe devices, 141 GiB nominal memory each; runtime-managed model splitting","interconnect":"Two-way NVLink bridge, up to 900 GB/s bidirectional per GPU; PCIe 5.0 host","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency.","reserveAssumption":"Planner assumption: 2 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/data-center/h200/"}],"powerWatts":1200,"powerBasis":"Sum of device maximum/reference board power; host, cooling and power-supply losses excluded","family":"2 × H200 NVL 141 GB PCIe","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB reserve on each of 2 devices; placement and duplicated buffers can require additional memory.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency."},{"id":"4x-h200-nvl","name":"4 × H200 NVL 141 GB PCIe","maker":"NVIDIA","memoryGiB":564,"reserveGiB":8,"type":"Multi-GPU","bandwidthGBs":4800,"count":4,"summary":"4 separate 141 GB memory pools; 564 GB total installed.","caveat":"Combined capacity is a planning upper bound. The runtime must split weights and KV cache; each device must fit its own layers and buffers. Layer splitting does not multiply single-request memory bandwidth. Host PCIe lanes, P2P support, cooling and power must be validated.","source":"https://www.nvidia.com/en-us/data-center/h200/","reviewedAt":"2026-09-06","category":"Data center GPU","memoryPerDeviceGiB":141,"memoryType":"HBM3e ECC","configuration":"4 identical H200 NVL 141 GB PCIe devices, 141 GiB nominal memory each; runtime-managed model splitting","interconnect":"Four-way NVLink bridge, up to 900 GB/s bidirectional per GPU; PCIe 5.0 host","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency.","reserveAssumption":"Planner assumption: 2 GiB reserve on each of 4 devices; placement and duplicated buffers can require additional memory.","runtimeSupport":"CUDA backend; use a runtime build that supports this GPU generation.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/data-center/h200/"}],"powerWatts":2400,"powerBasis":"Sum of device maximum/reference board power; host, cooling and power-supply losses excluded","family":"4 × H200 NVL 141 GB PCIe","segment":"Datacenter","reserveNote":"Planner assumption: 2 GiB reserve on each of 4 devices; placement and duplicated buffers can require additional memory.","runtime":"CUDA backend; use a runtime build that supports this GPU generation.","memoryBandwidthNote":"This is peak bandwidth of each device, not an aggregate rate. Do not multiply it by device count to predict sequential layer-split token latency."},{"id":"rtx-3050-6gb","name":"GeForce RTX 3050 6 GB","maker":"NVIDIA","memoryGiB":6,"reserveGiB":1,"type":"GPU","bandwidthGBs":168,"count":1,"summary":"6 GB GDDR6 dedicated memory.","caveat":"Desktop reference memory configuration. Partner cooling, power limits and sustained performance vary. 6 GB version has a narrower bus and lower power limit than the 8 GB model.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3050/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":6,"memoryType":"GDDR6","configuration":"GeForce RTX 3050 6 GB, desktop, 6 GB GDDR6, 96-bit bus at 14 Gbps","interconnect":"PCIe 4.0 x8; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Derived from manufacturer memory data rate: 14 Gbps × 96 bits / 8 = 168 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend for Ampere; model and quantization kernels still require runtime support.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3050/"},{"title":"Manufacturer source 2","url":"https://www.msi.com/Graphics-Card/GeForce-RTX-3050-GAMING-6G/Specification"}],"powerWatts":70,"powerBasis":"Manufacturer reference board power; excludes host system and is not measured inference power.","family":"GeForce RTX 3050 6 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend for Ampere; model and quantization kernels still require runtime support.","memoryBandwidthNote":"Derived from manufacturer memory data rate: 14 Gbps × 96 bits / 8 = 168 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used."},{"id":"rtx-3050-8gb","name":"GeForce RTX 3050 8 GB","maker":"NVIDIA","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":224,"count":1,"summary":"8 GB GDDR6 dedicated memory.","caveat":"Desktop reference memory configuration. Partner cooling, power limits and sustained performance vary. OEM CUDA core counts can differ; this is the 8 GB, 128-bit memory configuration.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3050/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR6","configuration":"GeForce RTX 3050 8 GB, desktop, 8 GB GDDR6, 128-bit bus at 14 Gbps","interconnect":"PCIe 4.0 x8; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Derived from manufacturer memory data rate: 14 Gbps × 128 bits / 8 = 224 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend for Ampere; model and quantization kernels still require runtime support.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3050/"},{"title":"Manufacturer source 2","url":"https://www.msi.com/Graphics-Card/GeForce-RTX-3050-GAMING-8G/Specification"}],"powerWatts":130,"powerBasis":"Manufacturer reference board power; excludes host system and is not measured inference power.","family":"GeForce RTX 3050 8 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend for Ampere; model and quantization kernels still require runtime support.","memoryBandwidthNote":"Derived from manufacturer memory data rate: 14 Gbps × 128 bits / 8 = 224 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used."},{"id":"rtx-3060-ti-gddr6","name":"GeForce RTX 3060 Ti 8 GB GDDR6","maker":"NVIDIA","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":448,"count":1,"summary":"8 GB GDDR6 dedicated memory.","caveat":"Desktop reference memory configuration. Partner cooling, power limits and sustained performance vary. This entry is the original GDDR6 variant. Later GDDR6X variants have different memory bandwidth.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3060-3060ti/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR6","configuration":"GeForce RTX 3060 Ti 8 GB GDDR6, desktop, 8 GB GDDR6, 256-bit bus at 14 Gbps","interconnect":"PCIe 4.0; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Derived from manufacturer memory data rate: 14 Gbps × 256 bits / 8 = 448 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend for Ampere; model and quantization kernels still require runtime support.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3060-3060ti/"},{"title":"Manufacturer source 2","url":"https://storage-asset.msi.com/datasheet/vga/latam/GeForce-RTX-3060-Ti-VENTUS-3X-OC.pdf"}],"powerWatts":200,"powerBasis":"Manufacturer reference board power; excludes host system and is not measured inference power.","family":"GeForce RTX 3060 Ti 8 GB GDDR6","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend for Ampere; model and quantization kernels still require runtime support.","memoryBandwidthNote":"Derived from manufacturer memory data rate: 14 Gbps × 256 bits / 8 = 448 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used."},{"id":"rtx-3070","name":"GeForce RTX 3070","maker":"NVIDIA","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":448,"count":1,"summary":"8 GB GDDR6 dedicated memory.","caveat":"Desktop reference memory configuration. Partner cooling, power limits and sustained performance vary.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3070-3070ti/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR6","configuration":"GeForce RTX 3070, desktop, 8 GB GDDR6, 256-bit bus at 14 Gbps","interconnect":"PCIe 4.0; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Derived from manufacturer memory data rate: 14 Gbps × 256 bits / 8 = 448 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend for Ampere; model and quantization kernels still require runtime support.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3070-3070ti/"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/content/dam/en-zz/Solutions/geforce/ampere/pdf/NVIDIA-ampere-GA102-GPU-Architecture-Whitepaper-V1.pdf"}],"powerWatts":220,"powerBasis":"Manufacturer reference board power; excludes host system and is not measured inference power.","family":"GeForce RTX 3070","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend for Ampere; model and quantization kernels still require runtime support.","memoryBandwidthNote":"Derived from manufacturer memory data rate: 14 Gbps × 256 bits / 8 = 448 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used."},{"id":"rtx-3070-ti","name":"GeForce RTX 3070 Ti","maker":"NVIDIA","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":608,"count":1,"summary":"8 GB GDDR6X dedicated memory.","caveat":"Desktop reference memory configuration. Partner cooling, power limits and sustained performance vary.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3070-3070ti/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR6X","configuration":"GeForce RTX 3070 Ti, desktop, 8 GB GDDR6X, 256-bit bus at 19 Gbps","interconnect":"PCIe 4.0; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Derived from manufacturer memory data rate: 19 Gbps × 256 bits / 8 = 608 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend for Ampere; model and quantization kernels still require runtime support.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3070-3070ti/"},{"title":"Manufacturer source 2","url":"https://storage-asset.msi.com/datasheet/vga/uk/GeForce-RTX-3070-Ti-VENTUS-3X-8G-OC.pdf"}],"powerWatts":290,"powerBasis":"Manufacturer reference board power; excludes host system and is not measured inference power.","family":"GeForce RTX 3070 Ti","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend for Ampere; model and quantization kernels still require runtime support.","memoryBandwidthNote":"Derived from manufacturer memory data rate: 19 Gbps × 256 bits / 8 = 608 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used."},{"id":"rtx-3080-10gb","name":"GeForce RTX 3080 10 GB","maker":"NVIDIA","memoryGiB":10,"reserveGiB":1,"type":"GPU","bandwidthGBs":760,"count":1,"summary":"10 GB GDDR6X dedicated memory.","caveat":"Desktop reference memory configuration. Partner cooling, power limits and sustained performance vary. The 12 GB model also has a wider memory bus; capacity is not its only difference.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3080-3080ti/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":10,"memoryType":"GDDR6X","configuration":"GeForce RTX 3080 10 GB, desktop, 10 GB GDDR6X, 320-bit bus at 19 Gbps","interconnect":"PCIe 4.0; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Derived from manufacturer memory data rate: 19 Gbps × 320 bits / 8 = 760 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend for Ampere; model and quantization kernels still require runtime support.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3080-3080ti/"},{"title":"Manufacturer source 2","url":"https://www.nvidia.com/content/dam/en-zz/Solutions/geforce/ampere/pdf/NVIDIA-ampere-GA102-GPU-Architecture-Whitepaper-V1.pdf"}],"powerWatts":320,"powerBasis":"Manufacturer reference board power; excludes host system and is not measured inference power.","family":"GeForce RTX 3080 10 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend for Ampere; model and quantization kernels still require runtime support.","memoryBandwidthNote":"Derived from manufacturer memory data rate: 19 Gbps × 320 bits / 8 = 760 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used."},{"id":"rtx-3080-12gb","name":"GeForce RTX 3080 12 GB","maker":"NVIDIA","memoryGiB":12,"reserveGiB":1,"type":"GPU","bandwidthGBs":912,"count":1,"summary":"12 GB GDDR6X dedicated memory.","caveat":"Desktop reference memory configuration. Partner cooling, power limits and sustained performance vary.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3080-3080ti/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":12,"memoryType":"GDDR6X","configuration":"GeForce RTX 3080 12 GB, desktop, 12 GB GDDR6X, 384-bit bus at 19 Gbps","interconnect":"PCIe 4.0; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Derived from manufacturer memory data rate: 19 Gbps × 384 bits / 8 = 912 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend for Ampere; model and quantization kernels still require runtime support.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3080-3080ti/"},{"title":"Manufacturer source 2","url":"https://www.msi.com/Graphics-Card/GeForce-RTX-3080-VENTUS-3X-PLUS-12G-OC-LHR/Specification"}],"powerWatts":350,"powerBasis":"Manufacturer reference board power; excludes host system and is not measured inference power.","family":"GeForce RTX 3080 12 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend for Ampere; model and quantization kernels still require runtime support.","memoryBandwidthNote":"Derived from manufacturer memory data rate: 19 Gbps × 384 bits / 8 = 912 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used."},{"id":"rtx-3080-ti","name":"GeForce RTX 3080 Ti","maker":"NVIDIA","memoryGiB":12,"reserveGiB":1,"type":"GPU","bandwidthGBs":912,"count":1,"summary":"12 GB GDDR6X dedicated memory.","caveat":"Desktop reference memory configuration. Partner cooling, power limits and sustained performance vary.","source":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3080-3080ti/","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":12,"memoryType":"GDDR6X","configuration":"GeForce RTX 3080 Ti, desktop, 12 GB GDDR6X, 384-bit bus at 19 Gbps","interconnect":"PCIe 4.0; no NVLink","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Derived from manufacturer memory data rate: 19 Gbps × 384 bits / 8 = 912 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"CUDA backend for Ampere; model and quantization kernels still require runtime support.","sources":[{"title":"Manufacturer source 1","url":"https://www.nvidia.com/en-us/geforce/graphics-cards/30-series/rtx-3080-3080ti/"},{"title":"Manufacturer source 2","url":"https://storage-asset.msi.com/datasheet/vga/global/GeForce-RTX-3080-Ti-VENTUS-3X-12G-OC.pdf"}],"powerWatts":350,"powerBasis":"Manufacturer reference board power; excludes host system and is not measured inference power.","family":"GeForce RTX 3080 Ti","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"CUDA backend for Ampere; model and quantization kernels still require runtime support.","memoryBandwidthNote":"Derived from manufacturer memory data rate: 19 Gbps × 384 bits / 8 = 912 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used."},{"id":"rx-6600","name":"Radeon RX 6600","maker":"AMD","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":224,"count":1,"summary":"8 GB GDDR6 dedicated memory.","caveat":"Desktop reference memory configuration. Partner cooling, power limits and sustained performance vary. Infinity Cache effective bandwidth is not substituted for physical GDDR6 bandwidth.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/6000-series/amd-radeon-rx-6600.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR6","configuration":"Radeon RX 6600, desktop, 8 GB GDDR6, 128-bit bus at 14 Gbps","interconnect":"PCIe; dedicated card memory","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Derived from manufacturer memory data rate: 14 Gbps × 128 bits / 8 = 224 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"RDNA 2: check exact ROCm hardware/OS support; a Vulkan-capable inference backend is an alternative when supported by the model.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/6000-series/amd-radeon-rx-6600.html"}],"powerWatts":132,"powerBasis":"Manufacturer reference board power; excludes host system and is not measured inference power.","family":"Radeon RX 6600","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"RDNA 2: check exact ROCm hardware/OS support; a Vulkan-capable inference backend is an alternative when supported by the model.","memoryBandwidthNote":"Derived from manufacturer memory data rate: 14 Gbps × 128 bits / 8 = 224 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used."},{"id":"rx-6600-xt","name":"Radeon RX 6600 XT","maker":"AMD","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":256,"count":1,"summary":"8 GB GDDR6 dedicated memory.","caveat":"Desktop reference memory configuration. Partner cooling, power limits and sustained performance vary. Infinity Cache effective bandwidth is not substituted for physical GDDR6 bandwidth.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/6000-series/amd-radeon-rx-6600-xt.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR6","configuration":"Radeon RX 6600 XT, desktop, 8 GB GDDR6, 128-bit bus at 16 Gbps","interconnect":"PCIe; dedicated card memory","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Derived from manufacturer memory data rate: 16 Gbps × 128 bits / 8 = 256 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"RDNA 2: check exact ROCm hardware/OS support; a Vulkan-capable inference backend is an alternative when supported by the model.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/6000-series/amd-radeon-rx-6600-xt.html"}],"powerWatts":160,"powerBasis":"Manufacturer reference board power; excludes host system and is not measured inference power.","family":"Radeon RX 6600 XT","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"RDNA 2: check exact ROCm hardware/OS support; a Vulkan-capable inference backend is an alternative when supported by the model.","memoryBandwidthNote":"Derived from manufacturer memory data rate: 16 Gbps × 128 bits / 8 = 256 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used."},{"id":"rx-6650-xt","name":"Radeon RX 6650 XT","maker":"AMD","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":280,"count":1,"summary":"8 GB GDDR6 dedicated memory.","caveat":"Desktop reference memory configuration. Partner cooling, power limits and sustained performance vary. Infinity Cache effective bandwidth is not substituted for physical GDDR6 bandwidth.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/6000-series/amd-radeon-rx-6650-xt.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR6","configuration":"Radeon RX 6650 XT, desktop, 8 GB GDDR6, 128-bit bus at 17.5 Gbps","interconnect":"PCIe; dedicated card memory","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Derived from manufacturer memory data rate: 17.5 Gbps × 128 bits / 8 = 280 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"RDNA 2: check exact ROCm hardware/OS support; a Vulkan-capable inference backend is an alternative when supported by the model.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/6000-series/amd-radeon-rx-6650-xt.html"}],"powerWatts":180,"powerBasis":"Manufacturer reference board power; excludes host system and is not measured inference power.","family":"Radeon RX 6650 XT","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"RDNA 2: check exact ROCm hardware/OS support; a Vulkan-capable inference backend is an alternative when supported by the model.","memoryBandwidthNote":"Derived from manufacturer memory data rate: 17.5 Gbps × 128 bits / 8 = 280 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used."},{"id":"rx-6700-xt","name":"Radeon RX 6700 XT","maker":"AMD","memoryGiB":12,"reserveGiB":1,"type":"GPU","bandwidthGBs":384,"count":1,"summary":"12 GB GDDR6 dedicated memory.","caveat":"Desktop reference memory configuration. Partner cooling, power limits and sustained performance vary. Infinity Cache effective bandwidth is not substituted for physical GDDR6 bandwidth.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/6000-series/amd-radeon-rx-6700-xt.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":12,"memoryType":"GDDR6","configuration":"Radeon RX 6700 XT, desktop, 12 GB GDDR6, 192-bit bus at 16 Gbps","interconnect":"PCIe; dedicated card memory","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Derived from manufacturer memory data rate: 16 Gbps × 192 bits / 8 = 384 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"RDNA 2: check exact ROCm hardware/OS support; a Vulkan-capable inference backend is an alternative when supported by the model.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/6000-series/amd-radeon-rx-6700-xt.html"}],"powerWatts":230,"powerBasis":"Manufacturer reference board power; excludes host system and is not measured inference power.","family":"Radeon RX 6700 XT","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"RDNA 2: check exact ROCm hardware/OS support; a Vulkan-capable inference backend is an alternative when supported by the model.","memoryBandwidthNote":"Derived from manufacturer memory data rate: 16 Gbps × 192 bits / 8 = 384 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used."},{"id":"rx-6750-xt","name":"Radeon RX 6750 XT","maker":"AMD","memoryGiB":12,"reserveGiB":1,"type":"GPU","bandwidthGBs":432,"count":1,"summary":"12 GB GDDR6 dedicated memory.","caveat":"Desktop reference memory configuration. Partner cooling, power limits and sustained performance vary. Infinity Cache effective bandwidth is not substituted for physical GDDR6 bandwidth.","source":"https://www.amd.com/en/products/graphics/desktops/radeon/6000-series/amd-radeon-rx-6750-xt.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":12,"memoryType":"GDDR6","configuration":"Radeon RX 6750 XT, desktop, 12 GB GDDR6, 192-bit bus at 18 Gbps","interconnect":"PCIe; dedicated card memory","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Derived from manufacturer memory data rate: 18 Gbps × 192 bits / 8 = 432 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"RDNA 2: check exact ROCm hardware/OS support; a Vulkan-capable inference backend is an alternative when supported by the model.","sources":[{"title":"Manufacturer source 1","url":"https://www.amd.com/en/products/graphics/desktops/radeon/6000-series/amd-radeon-rx-6750-xt.html"}],"powerWatts":250,"powerBasis":"Manufacturer reference board power; excludes host system and is not measured inference power.","family":"Radeon RX 6750 XT","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"RDNA 2: check exact ROCm hardware/OS support; a Vulkan-capable inference backend is an alternative when supported by the model.","memoryBandwidthNote":"Derived from manufacturer memory data rate: 18 Gbps × 192 bits / 8 = 432 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used."},{"id":"arc-a750-8gb","name":"Arc A750 8 GB","maker":"Intel","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":512,"count":1,"summary":"8 GB GDDR6 dedicated memory.","caveat":"Desktop reference memory configuration. Partner cooling, power limits and sustained performance vary. Intel reference TBP; partner settings can differ.","source":"https://www.intel.com/content/www/us/en/products/docs/discrete-gpus/arc/desktop/a-series/3.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR6","configuration":"Arc A750 8 GB, desktop, 8 GB GDDR6","interconnect":"PCIe; dedicated card memory","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Manufacturer peak memory bandwidth. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"Use a compatible SYCL, Vulkan or OpenVINO backend; CUDA builds do not run directly on Arc.","sources":[{"title":"Manufacturer source 1","url":"https://www.intel.com/content/www/us/en/products/docs/discrete-gpus/arc/desktop/a-series/3.html"},{"title":"Manufacturer source 2","url":"https://www.intel.com/content/www/us/en/support/articles/000092523/graphics.html"}],"powerWatts":225,"powerBasis":"Manufacturer reference board power; excludes host system and is not measured inference power.","family":"Arc A750 8 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"Use a compatible SYCL, Vulkan or OpenVINO backend; CUDA builds do not run directly on Arc.","memoryBandwidthNote":"Manufacturer peak memory bandwidth. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used."},{"id":"arc-a770-8gb","name":"Arc A770 8 GB","maker":"Intel","memoryGiB":8,"reserveGiB":1,"type":"GPU","bandwidthGBs":512,"count":1,"summary":"8 GB GDDR6 dedicated memory.","caveat":"Desktop reference memory configuration. Partner cooling, power limits and sustained performance vary. 8 GB version has 512 GB/s; the existing 16 GB version has 560 GB/s.","source":"https://www.intel.com/content/www/us/en/products/sku/227955/intel-arc-a770-graphics-8gb/specifications.html","reviewedAt":"2026-09-06","category":"Consumer GPU","memoryPerDeviceGiB":8,"memoryType":"GDDR6","configuration":"Arc A770 8 GB, desktop, 8 GB GDDR6, 256-bit bus at 16 Gbps","interconnect":"PCIe 4.0 x16; dedicated card memory","availability":"Available","bandwidthKind":"Peak theoretical, decimal GB/s per device","bandwidthNote":"Derived from manufacturer memory data rate: 16 Gbps × 256 bits / 8 = 512 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used.","reserveAssumption":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtimeSupport":"Use a compatible SYCL, Vulkan or OpenVINO backend; CUDA builds do not run directly on Arc.","sources":[{"title":"Manufacturer source 1","url":"https://www.intel.com/content/www/us/en/products/sku/227955/intel-arc-a770-graphics-8gb/specifications.html"},{"title":"Manufacturer source 2","url":"https://www.intel.com/content/www/us/en/support/articles/000092523/graphics.html"}],"powerWatts":225,"powerBasis":"Manufacturer reference board power; excludes host system and is not measured inference power.","family":"Arc A770 8 GB","segment":"Consumer","reserveNote":"Planner assumption: 1 GiB per device for driver/display and allocation headroom. Runtime buffers are modeled separately; actual free memory must be checked.","runtime":"Use a compatible SYCL, Vulkan or OpenVINO backend; CUDA builds do not run directly on Arc.","memoryBandwidthNote":"Derived from manufacturer memory data rate: 16 Gbps × 256 bits / 8 = 512 GB/s. Not measured sustained bandwidth or token throughput; cache-enhanced effective bandwidth is not used."}]}
