[ { "id": "mac-mini-m6-16", "family": "Mac mini", "chip": "M6", "chip_variant": "12-core CPU / 12-core GPU", "unified_memory_gb": 16, "memory_bandwidth_gbs": 153, "usable_memory_gb": 10.5, "price_usd": 899, "idle_watts": null, "load_watts": 65, "load_watts_status": "stand_in", "load_watts_note": "Apple has not published power figures for the 2026 machines yet (support articles 103253 / 102027 stop at the 2024/2025 models). Stand-in: Apple's published maximum for the previous chip (M4 Mac mini, 65 W), which is a CPU+GPU stress figure, so it likely overstates inference draw.", "generation": "current", "status": "Pre-order; ships 2026-09-22.", "notes": "macOS lets the GPU wire roughly two-thirds of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://www.apple.com/mac-mini/specs/", "https://www.apple.com/newsroom/2026/08/apple-unveils-a-more-powerful-mac-mini-featuring-the-all-new-m6-and-m5-pro/", "https://daringfireball.net/2026/08/configurations_and_pricing_for_new_mac_minis_and_mac_studios" ] }, { "id": "mac-mini-m6-24", "family": "Mac mini", "chip": "M6", "chip_variant": "12-core CPU / 12-core GPU", "unified_memory_gb": 24, "memory_bandwidth_gbs": 170, "usable_memory_gb": 16.0, "price_usd": 1099, "idle_watts": null, "load_watts": 65, "load_watts_status": "stand_in", "load_watts_note": "Apple has not published power figures for the 2026 machines yet (support articles 103253 / 102027 stop at the 2024/2025 models). Stand-in: Apple's published maximum for the previous chip (M4 Mac mini, 65 W), which is a CPU+GPU stress figure, so it likely overstates inference draw.", "generation": "current", "status": "Pre-order; ships 2026-09-22.", "notes": "macOS lets the GPU wire roughly two-thirds of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://www.apple.com/mac-mini/specs/", "https://www.apple.com/newsroom/2026/08/apple-unveils-a-more-powerful-mac-mini-featuring-the-all-new-m6-and-m5-pro/", "https://daringfireball.net/2026/08/configurations_and_pricing_for_new_mac_minis_and_mac_studios" ] }, { "id": "mac-mini-m6-32", "family": "Mac mini", "chip": "M6", "chip_variant": "12-core CPU / 12-core GPU", "unified_memory_gb": 32, "memory_bandwidth_gbs": 170, "usable_memory_gb": 21.0, "price_usd": 1299, "idle_watts": null, "load_watts": 65, "load_watts_status": "stand_in", "load_watts_note": "Apple has not published power figures for the 2026 machines yet (support articles 103253 / 102027 stop at the 2024/2025 models). Stand-in: Apple's published maximum for the previous chip (M4 Mac mini, 65 W), which is a CPU+GPU stress figure, so it likely overstates inference draw.", "generation": "current", "status": "Pre-order; ships 2026-09-22.", "notes": "macOS lets the GPU wire roughly two-thirds of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://www.apple.com/mac-mini/specs/", "https://www.apple.com/newsroom/2026/08/apple-unveils-a-more-powerful-mac-mini-featuring-the-all-new-m6-and-m5-pro/", "https://daringfireball.net/2026/08/configurations_and_pricing_for_new_mac_minis_and_mac_studios" ] }, { "id": "mac-mini-m5-pro-24", "family": "Mac mini", "chip": "M5 Pro", "chip_variant": "15-core CPU / 16-core GPU", "unified_memory_gb": 24, "memory_bandwidth_gbs": 307, "usable_memory_gb": 16.0, "price_usd": 1699, "idle_watts": null, "load_watts": 140, "load_watts_status": "stand_in", "load_watts_note": "Apple has not published power figures for the 2026 machines yet (support articles 103253 / 102027 stop at the 2024/2025 models). Stand-in: Apple's published maximum for the previous chip (M4 Pro Mac mini, 140 W), which is a CPU+GPU stress figure, so it likely overstates inference draw.", "generation": "current", "status": "Pre-order; ships 2026-09-22.", "notes": "macOS lets the GPU wire roughly two-thirds of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://www.apple.com/mac-mini/specs/", "https://www.apple.com/newsroom/2026/08/apple-unveils-a-more-powerful-mac-mini-featuring-the-all-new-m6-and-m5-pro/", "https://daringfireball.net/2026/08/configurations_and_pricing_for_new_mac_minis_and_mac_studios" ] }, { "id": "mac-mini-m5-pro-48", "family": "Mac mini", "chip": "M5 Pro", "chip_variant": "15-core CPU / 16-core GPU", "unified_memory_gb": 48, "memory_bandwidth_gbs": 307, "usable_memory_gb": 36.0, "price_usd": 2299, "idle_watts": null, "load_watts": 140, "load_watts_status": "stand_in", "load_watts_note": "Apple has not published power figures for the 2026 machines yet (support articles 103253 / 102027 stop at the 2024/2025 models). Stand-in: Apple's published maximum for the previous chip (M4 Pro Mac mini, 140 W), which is a CPU+GPU stress figure, so it likely overstates inference draw.", "generation": "current", "status": "Pre-order; ships 2026-09-22.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://www.apple.com/mac-mini/specs/", "https://www.apple.com/newsroom/2026/08/apple-unveils-a-more-powerful-mac-mini-featuring-the-all-new-m6-and-m5-pro/", "https://daringfireball.net/2026/08/configurations_and_pricing_for_new_mac_minis_and_mac_studios" ] }, { "id": "mac-mini-m5-pro-64", "family": "Mac mini", "chip": "M5 Pro", "chip_variant": "15-core CPU / 16-core GPU", "unified_memory_gb": 64, "memory_bandwidth_gbs": 307, "usable_memory_gb": 48.0, "price_usd": 2699, "idle_watts": null, "load_watts": 140, "load_watts_status": "stand_in", "load_watts_note": "Apple has not published power figures for the 2026 machines yet (support articles 103253 / 102027 stop at the 2024/2025 models). Stand-in: Apple's published maximum for the previous chip (M4 Pro Mac mini, 140 W), which is a CPU+GPU stress figure, so it likely overstates inference draw.", "generation": "current", "status": "Pre-order; ships 2026-09-22.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://www.apple.com/mac-mini/specs/", "https://www.apple.com/newsroom/2026/08/apple-unveils-a-more-powerful-mac-mini-featuring-the-all-new-m6-and-m5-pro/", "https://daringfireball.net/2026/08/configurations_and_pricing_for_new_mac_minis_and_mac_studios" ] }, { "id": "mac-studio-m5-max-36", "family": "Mac Studio", "chip": "M5 Max", "chip_variant": "18-core CPU / 32-core GPU", "unified_memory_gb": 36, "memory_bandwidth_gbs": 460, "usable_memory_gb": 27.0, "price_usd": 2499, "idle_watts": null, "load_watts": 145, "load_watts_status": "stand_in", "load_watts_note": "Apple has not published power figures for the 2026 machines yet (support articles 103253 / 102027 stop at the 2024/2025 models). Stand-in: Apple's published maximum for the previous chip (M4 Max Mac Studio, 145 W), which is a CPU+GPU stress figure, so it likely overstates inference draw.", "generation": "current", "status": "Pre-order; ships 2026-09-22.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://www.apple.com/mac-studio/specs/", "https://www.apple.com/newsroom/2026/08/apple-introduces-new-mac-studio-with-m5-max-and-m5-ultra/", "https://daringfireball.net/2026/08/configurations_and_pricing_for_new_mac_minis_and_mac_studios" ] }, { "id": "mac-studio-m5-max-48", "family": "Mac Studio", "chip": "M5 Max", "chip_variant": "18-core CPU / 40-core GPU", "unified_memory_gb": 48, "memory_bandwidth_gbs": 614, "usable_memory_gb": 36.0, "price_usd": 3099, "idle_watts": null, "load_watts": 145, "load_watts_status": "stand_in", "load_watts_note": "Apple has not published power figures for the 2026 machines yet (support articles 103253 / 102027 stop at the 2024/2025 models). Stand-in: Apple's published maximum for the previous chip (M4 Max Mac Studio, 145 W), which is a CPU+GPU stress figure, so it likely overstates inference draw.", "generation": "current", "status": "Pre-order; ships 2026-09-22.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://www.apple.com/mac-studio/specs/", "https://daringfireball.net/2026/08/configurations_and_pricing_for_new_mac_minis_and_mac_studios" ] }, { "id": "mac-studio-m5-max-64", "family": "Mac Studio", "chip": "M5 Max", "chip_variant": "18-core CPU / 40-core GPU", "unified_memory_gb": 64, "memory_bandwidth_gbs": 614, "usable_memory_gb": 48.0, "price_usd": 3499, "idle_watts": null, "load_watts": 145, "load_watts_status": "stand_in", "load_watts_note": "Apple has not published power figures for the 2026 machines yet (support articles 103253 / 102027 stop at the 2024/2025 models). Stand-in: Apple's published maximum for the previous chip (M4 Max Mac Studio, 145 W), which is a CPU+GPU stress figure, so it likely overstates inference draw.", "generation": "current", "status": "Pre-order; ships 2026-09-22.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://www.apple.com/mac-studio/specs/", "https://daringfireball.net/2026/08/configurations_and_pricing_for_new_mac_minis_and_mac_studios" ] }, { "id": "mac-studio-m5-max-128", "family": "Mac Studio", "chip": "M5 Max", "chip_variant": "18-core CPU / 40-core GPU", "unified_memory_gb": 128, "memory_bandwidth_gbs": 614, "usable_memory_gb": 96.0, "price_usd": 5099, "idle_watts": null, "load_watts": 145, "load_watts_status": "stand_in", "load_watts_note": "Apple has not published power figures for the 2026 machines yet (support articles 103253 / 102027 stop at the 2024/2025 models). Stand-in: Apple's published maximum for the previous chip (M4 Max Mac Studio, 145 W), which is a CPU+GPU stress figure, so it likely overstates inference draw.", "generation": "current", "status": "Pre-order; ships 2026-09-22.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://www.apple.com/mac-studio/specs/", "https://daringfireball.net/2026/08/configurations_and_pricing_for_new_mac_minis_and_mac_studios" ] }, { "id": "mac-studio-m5-ultra-96", "family": "Mac Studio", "chip": "M5 Ultra", "chip_variant": "30-core CPU / 64-core GPU", "unified_memory_gb": 96, "memory_bandwidth_gbs": 1200, "usable_memory_gb": 72.0, "price_usd": 5499, "idle_watts": null, "load_watts": 270, "load_watts_status": "stand_in", "load_watts_note": "Apple has not published power figures for the 2026 machines yet (support articles 103253 / 102027 stop at the 2024/2025 models). Stand-in: Apple's published maximum for the previous chip (M3 Ultra Mac Studio, 270 W), which is a CPU+GPU stress figure, so it likely overstates inference draw.", "generation": "current", "status": "Pre-order; ships 2026-09-22.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://www.apple.com/mac-studio/specs/", "https://www.apple.com/newsroom/2026/08/apple-introduces-new-mac-studio-with-m5-max-and-m5-ultra/", "https://daringfireball.net/2026/08/configurations_and_pricing_for_new_mac_minis_and_mac_studios" ] }, { "id": "mac-studio-m5-ultra-256", "family": "Mac Studio", "chip": "M5 Ultra", "chip_variant": "36-core CPU / 80-core GPU", "unified_memory_gb": 256, "memory_bandwidth_gbs": 1200, "usable_memory_gb": 192.0, "price_usd": 10799, "idle_watts": null, "load_watts": 270, "load_watts_status": "stand_in", "load_watts_note": "Apple has not published power figures for the 2026 machines yet (support articles 103253 / 102027 stop at the 2024/2025 models). Stand-in: Apple's published maximum for the previous chip (M3 Ultra Mac Studio, 270 W), which is a CPU+GPU stress figure, so it likely overstates inference draw.", "generation": "current", "status": "Pre-order; ships 2026-09-22.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://www.apple.com/mac-studio/specs/", "https://daringfireball.net/2026/08/configurations_and_pricing_for_new_mac_minis_and_mac_studios", "https://appleinsider.com/articles/26/08/25/you-can-spend-18299-on-a-mac-studio-today-or-more-in-october" ] }, { "id": "mac-studio-m5-ultra-512", "family": "Mac Studio", "chip": "M5 Ultra", "chip_variant": "36-core CPU / 80-core GPU", "unified_memory_gb": 512, "memory_bandwidth_gbs": 1200, "usable_memory_gb": 384.0, "price_usd": null, "idle_watts": null, "load_watts": 270, "load_watts_status": "stand_in", "load_watts_note": "Apple has not published power figures for the 2026 machines yet (support articles 103253 / 102027 stop at the 2024/2025 models). Stand-in: Apple's published maximum for the previous chip (M3 Ultra Mac Studio, 270 W), which is a CPU+GPU stress figure, so it likely overstates inference draw.", "generation": "current", "status": "Announced for late October 2026; Apple has not published a price.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://www.apple.com/mac-studio/specs/", "https://www.apple.com/mac-studio/" ], "TODO": "price_usd: Apple has not announced the 512GB price." }, { "id": "mac-mini-m4-16", "family": "Mac mini", "chip": "M4", "chip_variant": "10-core CPU / 10-core GPU", "unified_memory_gb": 16, "memory_bandwidth_gbs": 120, "usable_memory_gb": 10.5, "price_usd": 599, "idle_watts": 4, "load_watts": 65, "load_watts_status": "published", "load_watts_note": "Apple's published 'max' figure for this chip (support article 103253), a CPU+GPU stress number; inference typically draws less.", "generation": "previous", "status": "Discontinued 2026-08-25. Launch price; Apple no longer sells it new, so treat as a refurbished/used reference.", "notes": "macOS lets the GPU wire roughly two-thirds of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://support.apple.com/en-us/121555", "https://support.apple.com/en-us/103253", "https://lowendmac.com/2024/mac-mini-2024/" ] }, { "id": "mac-mini-m4-24", "family": "Mac mini", "chip": "M4", "chip_variant": "10-core CPU / 10-core GPU", "unified_memory_gb": 24, "memory_bandwidth_gbs": 120, "usable_memory_gb": 16.0, "price_usd": 799, "idle_watts": 4, "load_watts": 65, "load_watts_status": "published", "load_watts_note": "Apple's published 'max' figure for this chip (support article 103253), a CPU+GPU stress number; inference typically draws less.", "generation": "previous", "status": "Discontinued 2026-08-25. Launch price; Apple no longer sells it new, so treat as a refurbished/used reference.", "notes": "macOS lets the GPU wire roughly two-thirds of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://support.apple.com/en-us/121555", "https://support.apple.com/en-us/103253", "https://lowendmac.com/2024/mac-mini-2024/" ] }, { "id": "mac-mini-m4-32", "family": "Mac mini", "chip": "M4", "chip_variant": "10-core CPU / 10-core GPU", "unified_memory_gb": 32, "memory_bandwidth_gbs": 120, "usable_memory_gb": 21.0, "price_usd": 999, "idle_watts": 4, "load_watts": 65, "load_watts_status": "published", "load_watts_note": "Apple's published 'max' figure for this chip (support article 103253), a CPU+GPU stress number; inference typically draws less.", "generation": "previous", "status": "Discontinued 2026-08-25. Launch price; Apple no longer sells it new, so treat as a refurbished/used reference.", "notes": "macOS lets the GPU wire roughly two-thirds of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://support.apple.com/en-us/121555", "https://support.apple.com/en-us/103253", "https://lowendmac.com/2024/mac-mini-2024/" ] }, { "id": "mac-mini-m4-pro-24", "family": "Mac mini", "chip": "M4 Pro", "chip_variant": "12-core CPU / 16-core GPU", "unified_memory_gb": 24, "memory_bandwidth_gbs": 273, "usable_memory_gb": 16.0, "price_usd": 1399, "idle_watts": 5, "load_watts": 140, "load_watts_status": "published", "load_watts_note": "Apple's published 'max' figure for this chip (support article 103253), a CPU+GPU stress number; inference typically draws less.", "generation": "previous", "status": "Discontinued 2026-08-25. Launch price; Apple no longer sells it new, so treat as a refurbished/used reference.", "notes": "macOS lets the GPU wire roughly two-thirds of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://support.apple.com/en-us/121555", "https://support.apple.com/en-us/103253", "https://lowendmac.com/2024/mac-mini-2024/" ] }, { "id": "mac-mini-m4-pro-48", "family": "Mac mini", "chip": "M4 Pro", "chip_variant": "12-core CPU / 16-core GPU", "unified_memory_gb": 48, "memory_bandwidth_gbs": 273, "usable_memory_gb": 36.0, "price_usd": 1799, "idle_watts": 5, "load_watts": 140, "load_watts_status": "published", "load_watts_note": "Apple's published 'max' figure for this chip (support article 103253), a CPU+GPU stress number; inference typically draws less.", "generation": "previous", "status": "Discontinued 2026-08-25. Launch price; Apple no longer sells it new, so treat as a refurbished/used reference.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://support.apple.com/en-us/121555", "https://support.apple.com/en-us/103253", "https://lowendmac.com/2024/mac-mini-2024/" ] }, { "id": "mac-mini-m4-pro-64", "family": "Mac mini", "chip": "M4 Pro", "chip_variant": "12-core CPU / 16-core GPU", "unified_memory_gb": 64, "memory_bandwidth_gbs": 273, "usable_memory_gb": 48.0, "price_usd": 1999, "idle_watts": 5, "load_watts": 140, "load_watts_status": "published", "load_watts_note": "Apple's published 'max' figure for this chip (support article 103253), a CPU+GPU stress number; inference typically draws less.", "generation": "previous", "status": "Discontinued 2026-08-25. Launch price; Apple no longer sells it new, so treat as a refurbished/used reference.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://support.apple.com/en-us/121555", "https://support.apple.com/en-us/103253", "https://lowendmac.com/2024/mac-mini-2024/" ] }, { "id": "mac-studio-m4-max-36", "family": "Mac Studio", "chip": "M4 Max", "chip_variant": "14-core CPU / 32-core GPU", "unified_memory_gb": 36, "memory_bandwidth_gbs": 410, "usable_memory_gb": 27.0, "price_usd": 1999, "idle_watts": 6, "load_watts": 145, "load_watts_status": "published", "load_watts_note": "Apple's published 'max' figure for this chip (support article 102027), a CPU+GPU stress number; inference typically draws less. Apple measured the 14-core/32-GPU config.", "generation": "previous", "status": "Discontinued 2026-08-25. Launch price; Apple no longer sells it new, so treat as a refurbished/used reference.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://support.apple.com/en-us/122211", "https://support.apple.com/en-us/102027", "https://lowendmac.com/2025/mac-studio-early-2025/" ] }, { "id": "mac-studio-m4-max-48", "family": "Mac Studio", "chip": "M4 Max", "chip_variant": "16-core CPU / 40-core GPU", "unified_memory_gb": 48, "memory_bandwidth_gbs": 546, "usable_memory_gb": 36.0, "price_usd": 2499, "idle_watts": 6, "load_watts": 145, "load_watts_status": "published", "load_watts_note": "Apple's published 'max' figure for this chip (support article 102027), a CPU+GPU stress number; inference typically draws less. Apple measured the 14-core/32-GPU config only.", "generation": "previous", "status": "Discontinued 2026-08-25. Launch price; Apple no longer sells it new, so treat as a refurbished/used reference.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://support.apple.com/en-us/122211", "https://support.apple.com/en-us/102027", "https://lowendmac.com/2025/mac-studio-early-2025/" ] }, { "id": "mac-studio-m4-max-64", "family": "Mac Studio", "chip": "M4 Max", "chip_variant": "16-core CPU / 40-core GPU", "unified_memory_gb": 64, "memory_bandwidth_gbs": 546, "usable_memory_gb": 48.0, "price_usd": 2699, "idle_watts": 6, "load_watts": 145, "load_watts_status": "published", "load_watts_note": "Apple's published 'max' figure for this chip (support article 102027), a CPU+GPU stress number; inference typically draws less. Apple measured the 14-core/32-GPU config only.", "generation": "previous", "status": "Discontinued 2026-08-25. Launch price; Apple no longer sells it new, so treat as a refurbished/used reference.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://support.apple.com/en-us/122211", "https://support.apple.com/en-us/102027", "https://lowendmac.com/2025/mac-studio-early-2025/" ] }, { "id": "mac-studio-m4-max-128", "family": "Mac Studio", "chip": "M4 Max", "chip_variant": "16-core CPU / 40-core GPU", "unified_memory_gb": 128, "memory_bandwidth_gbs": 546, "usable_memory_gb": 96.0, "price_usd": 3499, "idle_watts": 6, "load_watts": 145, "load_watts_status": "published", "load_watts_note": "Apple's published 'max' figure for this chip (support article 102027), a CPU+GPU stress number; inference typically draws less. Apple measured the 14-core/32-GPU config only.", "generation": "previous", "status": "Discontinued 2026-08-25. Launch price; Apple no longer sells it new, so treat as a refurbished/used reference.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://support.apple.com/en-us/122211", "https://support.apple.com/en-us/102027", "https://lowendmac.com/2025/mac-studio-early-2025/" ] }, { "id": "mac-studio-m3-ultra-96", "family": "Mac Studio", "chip": "M3 Ultra", "chip_variant": "28-core CPU / 60-core GPU", "unified_memory_gb": 96, "memory_bandwidth_gbs": 819, "usable_memory_gb": 72.0, "price_usd": 3999, "idle_watts": 9, "load_watts": 270, "load_watts_status": "published", "load_watts_note": "Apple's published 'max' figure for this chip (support article 102027), a CPU+GPU stress number; inference typically draws less. Apple measured the 32-core/80-GPU, 512GB config.", "generation": "previous", "status": "Discontinued 2026-08-25. Launch price; Apple no longer sells it new, so treat as a refurbished/used reference.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://support.apple.com/en-us/122211", "https://support.apple.com/en-us/102027", "https://lowendmac.com/2025/mac-studio-early-2025/" ] }, { "id": "mac-studio-m3-ultra-256", "family": "Mac Studio", "chip": "M3 Ultra", "chip_variant": "32-core CPU / 80-core GPU", "unified_memory_gb": 256, "memory_bandwidth_gbs": 819, "usable_memory_gb": 192.0, "price_usd": 7099, "idle_watts": 9, "load_watts": 270, "load_watts_status": "published", "load_watts_note": "Apple's published 'max' figure for this chip (support article 102027), a CPU+GPU stress number; inference typically draws less.", "generation": "previous", "status": "Discontinued 2026-08-25. Launch price; Apple no longer sells it new, so treat as a refurbished/used reference. 256GB required the 32-core chip.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://support.apple.com/en-us/122211", "https://support.apple.com/en-us/102027", "https://lowendmac.com/2025/mac-studio-early-2025/" ] }, { "id": "mac-studio-m3-ultra-512", "family": "Mac Studio", "chip": "M3 Ultra", "chip_variant": "32-core CPU / 80-core GPU", "unified_memory_gb": 512, "memory_bandwidth_gbs": 819, "usable_memory_gb": 384.0, "price_usd": 9499, "idle_watts": 9, "load_watts": 270, "load_watts_status": "published", "load_watts_note": "Apple's published 'max' figure for this chip (support article 102027), a CPU+GPU stress number; inference typically draws less.", "generation": "previous", "status": "Discontinued 2026-08-25. Launch price; Apple no longer sells it new, so treat as a refurbished/used reference. 512GB option withdrawn March 2026.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. Memory is treated as GB throughout, which is slightly conservative.", "sources": [ "https://support.apple.com/en-us/122211", "https://support.apple.com/en-us/102027", "https://lowendmac.com/2025/mac-studio-early-2025/" ] }, { "id": "nvidia-dgx-spark-128", "family": "DGX Spark", "chip": "GB10 Grace Blackwell", "chip_variant": "Founders Edition, 4TB", "unified_memory_gb": 128, "memory_bandwidth_gbs": 273, "usable_memory_gb": 119.5, "price_usd": 4699, "idle_watts": 5, "load_watts": 150, "load_watts_status": "third_party_measured", "load_watts_note": "NVIDIA publishes no system draw (240 W supply, GB10 TDP 140 W). 150 W is a single third-party unit's peak under inference (kubesimplify); Jeff Geerling measured 122.8 W running gpt-oss-20b.", "generation": "current", "status": "Shipping. MSRP raised from $3,999 to $4,699 in February 2026.", "notes": "CUDA deviceQuery reports 122,570 MiB (about 119.7 GiB) visible to the GPU under DGX OS; treated as 119.5 GB usable. Partner boxes (ASUS, Dell, Lenovo, Acer) exist but none was verifiably cheaper and in stock when checked.", "sources": [ "https://www.nvidia.com/en-us/products/workstations/dgx-spark/", "https://forums.developer.nvidia.com/t/2-23-2026-price-change-announcement/361713", "https://blog.kubesimplify.com/day-3-the-dgx-spark-unpacked-gb10-unified-memory-sm-121-and-the-one-reason-this-hardware-exists", "https://github.com/geerlingguy/ai-benchmarks" ], "estimate_efficiency_moe": 0.65, "estimate_note": "MoE decode on CUDA reaches ~65% of the bandwidth ceiling here (Qwen3 30B-A3B: 89 tok/s measured vs 136 ceiling), against ~30% on Apple Silicon." }, { "id": "framework-desktop-395-128", "family": "Strix Halo", "chip": "Framework Desktop", "chip_variant": "Ryzen AI Max+ 395 · Radeon 8060S, 40 CU", "unified_memory_gb": 128, "memory_bandwidth_gbs": 256, "usable_memory_gb": 96, "price_usd": 3449, "idle_watts": null, "load_watts": 133, "load_watts_status": "third_party_measured", "load_watts_note": "Jeff Geerling's ai-benchmarks: Framework Desktop mainboard at 133 W running Llama 3.1 70B in Ollama (98 W on gpt-oss-120b via Vulkan). Meter method not stated.", "generation": "current", "status": "System $3,449 (out of stock when checked); mainboard alone $3,149 on pre-order. Launched at $1,999 in 2025; LPDDR5x prices rose since.", "notes": "Windows lets you dedicate up to 75% of RAM to the GPU (Variable Graphics Memory in AMD Adrenalin; Framework documents 96 GB on a 128 GB board). On Linux the amdgpu GTT pool can be raised to roughly 108-120 GB with ttm.pages_limit / amdgpu.gttsize boot parameters. The Windows figure is used here as the conservative one. Measured GPU bandwidth is ~212-215 GB/s of the 256 GB/s theoretical, so estimates here run a little high.", "sources": [ "https://frame.work/products/desktop-diy-amd-aimax300/configuration/new", "https://frame.work/blog/framework-desktop-deep-dive-ryzen-ai-max", "https://frame.work/desktop?tab=machine-learning", "https://community.frame.work/t/updated-commands-to-increase-max-unified-memory-usage-on-framework-desktop-under-fedora-43/78460", "https://github.com/geerlingguy/ai-benchmarks" ], "estimate_efficiency_moe": 0.55, "estimate_note": "MoE decode on Strix Halo reaches ~50-65% of the bandwidth ceiling (Qwen3 30B-A3B: 66-86 tok/s measured vs 128 ceiling), against ~30% on Apple Silicon." }, { "id": "framework-desktop-395-64", "family": "Strix Halo", "chip": "Framework Desktop", "chip_variant": "Ryzen AI Max+ 395 · Radeon 8060S, 40 CU", "unified_memory_gb": 64, "memory_bandwidth_gbs": 256, "usable_memory_gb": 48, "price_usd": 1959, "idle_watts": null, "load_watts": 133, "load_watts_status": "third_party_measured", "load_watts_note": "Jeff Geerling's ai-benchmarks: Framework Desktop mainboard at 133 W running Llama 3.1 70B in Ollama (98 W on gpt-oss-120b via Vulkan). Meter method not stated. Measured on the 128 GB board; same chip.", "generation": "current", "status": "System $1,959 (out of stock when checked); mainboard alone $1,659. Framework lists 48 GB dedicated to the GPU.", "notes": "Windows lets you dedicate up to 75% of RAM to the GPU (Variable Graphics Memory in AMD Adrenalin; Framework documents 96 GB on a 128 GB board). On Linux the amdgpu GTT pool can be raised to roughly 108-120 GB with ttm.pages_limit / amdgpu.gttsize boot parameters. The Windows figure is used here as the conservative one. Measured GPU bandwidth is ~212-215 GB/s of the 256 GB/s theoretical, so estimates here run a little high.", "sources": [ "https://frame.work/products/desktop-diy-amd-aimax300/configuration/new", "https://frame.work/desktop?tab=machine-learning", "https://github.com/geerlingguy/ai-benchmarks" ], "estimate_efficiency_moe": 0.55, "estimate_note": "MoE decode on Strix Halo reaches ~50-65% of the bandwidth ceiling (Qwen3 30B-A3B: 66-86 tok/s measured vs 128 ceiling), against ~30% on Apple Silicon." }, { "id": "framework-desktop-385-32", "family": "Strix Halo", "chip": "Framework Desktop", "chip_variant": "Ryzen AI Max 385 · Radeon 8050S, 32 CU", "unified_memory_gb": 32, "memory_bandwidth_gbs": 256, "usable_memory_gb": 24, "price_usd": 1269, "idle_watts": null, "load_watts": 133, "load_watts_status": "stand_in", "load_watts_note": "No measurement for the Max 385 board; stand-in is the Max+ 395 board's measured 133 W, which likely overstates this smaller GPU.", "generation": "current", "status": "System $1,269 (out of stock when checked); mainboard alone $969. Note: the 32 GB tier is the smaller Ryzen AI Max 385 with a 32-CU GPU.", "notes": "Windows lets you dedicate up to 75% of RAM to the GPU (Variable Graphics Memory in AMD Adrenalin; Framework documents 96 GB on a 128 GB board). On Linux the amdgpu GTT pool can be raised to roughly 108-120 GB with ttm.pages_limit / amdgpu.gttsize boot parameters. The Windows figure is used here as the conservative one. Measured GPU bandwidth is ~212-215 GB/s of the 256 GB/s theoretical, so estimates here run a little high.", "sources": [ "https://frame.work/products/desktop-diy-amd-aimax300/configuration/new", "https://frame.work/desktop?tab=machine-learning" ], "estimate_efficiency_moe": 0.55, "estimate_note": "MoE decode on Strix Halo reaches ~50-65% of the bandwidth ceiling (Qwen3 30B-A3B: 66-86 tok/s measured vs 128 ceiling), against ~30% on Apple Silicon." }, { "id": "gmktec-evo-x2-128", "family": "Strix Halo", "chip": "GMKtec EVO-X2", "chip_variant": "Ryzen AI Max+ 395 · Radeon 8060S, 40 CU", "unified_memory_gb": 128, "memory_bandwidth_gbs": 256, "usable_memory_gb": 96, "price_usd": 3500, "idle_watts": null, "load_watts": 150, "load_watts_status": "third_party_measured", "load_watts_note": "ServeTheHome measured 147-160 W at the wall running Llama 3.3 70B Q6_K; 8-14 W idle.", "generation": "current", "status": "128 GB / 1 TB $3,499.99 in stock when checked (2 TB $3,649.99). Launched at about $2,000 in May 2025.", "notes": "Windows lets you dedicate up to 75% of RAM to the GPU (Variable Graphics Memory in AMD Adrenalin; Framework documents 96 GB on a 128 GB board). On Linux the amdgpu GTT pool can be raised to roughly 108-120 GB with ttm.pages_limit / amdgpu.gttsize boot parameters. The Windows figure is used here as the conservative one. Measured GPU bandwidth is ~212-215 GB/s of the 256 GB/s theoretical, so estimates here run a little high.", "sources": [ "https://www.gmktec.com/products/amd-ryzen%e2%84%a2-ai-max-395-evo-x2-ai-mini-pc", "https://www.servethehome.com/gmktec-evo-x2-review-an-amd-ryzen-ai-max-395-powerhouse/4/" ], "estimate_efficiency_moe": 0.55, "estimate_note": "MoE decode on Strix Halo reaches ~50-65% of the bandwidth ceiling (Qwen3 30B-A3B: 66-86 tok/s measured vs 128 ceiling), against ~30% on Apple Silicon." }, { "id": "gmktec-evo-x2-64", "family": "Strix Halo", "chip": "GMKtec EVO-X2", "chip_variant": "Ryzen AI Max+ 395 · Radeon 8060S, 40 CU", "unified_memory_gb": 64, "memory_bandwidth_gbs": 256, "usable_memory_gb": 48, "price_usd": 2200, "idle_watts": null, "load_watts": 150, "load_watts_status": "third_party_measured", "load_watts_note": "ServeTheHome's 147-160 W figure was on the 128 GB unit; same chip.", "generation": "current", "status": "64 GB / 1 TB $2,199.99 in stock when checked. 48 GB usable follows the same 75% Windows split Framework documents for a 64 GB board.", "notes": "Windows lets you dedicate up to 75% of RAM to the GPU (Variable Graphics Memory in AMD Adrenalin; Framework documents 96 GB on a 128 GB board). On Linux the amdgpu GTT pool can be raised to roughly 108-120 GB with ttm.pages_limit / amdgpu.gttsize boot parameters. The Windows figure is used here as the conservative one. Measured GPU bandwidth is ~212-215 GB/s of the 256 GB/s theoretical, so estimates here run a little high.", "sources": [ "https://www.gmktec.com/products/amd-ryzen%e2%84%a2-ai-max-395-evo-x2-ai-mini-pc", "https://www.servethehome.com/gmktec-evo-x2-review-an-amd-ryzen-ai-max-395-powerhouse/4/", "https://frame.work/desktop?tab=machine-learning" ], "estimate_efficiency_moe": 0.55, "estimate_note": "MoE decode on Strix Halo reaches ~50-65% of the bandwidth ceiling (Qwen3 30B-A3B: 66-86 tok/s measured vs 128 ceiling), against ~30% on Apple Silicon." }, { "id": "beelink-gtr9-pro-128", "family": "Strix Halo", "chip": "Beelink GTR9 Pro", "chip_variant": "Ryzen AI Max+ 395 · Radeon 8060S, 40 CU", "unified_memory_gb": 128, "memory_bandwidth_gbs": 256, "usable_memory_gb": 96, "price_usd": 4349, "idle_watts": null, "load_watts": 133, "load_watts_status": "stand_in", "load_watts_note": "No wall measurement found for this unit; stand-in is the Framework Desktop's measured 133 W on the same chip at the same 120 W sustained TDP.", "generation": "current", "status": "Only sold as 128 GB / 2 TB, $4,349 when checked (launched at $1,985).", "notes": "Windows lets you dedicate up to 75% of RAM to the GPU (Variable Graphics Memory in AMD Adrenalin; Framework documents 96 GB on a 128 GB board). On Linux the amdgpu GTT pool can be raised to roughly 108-120 GB with ttm.pages_limit / amdgpu.gttsize boot parameters. The Windows figure is used here as the conservative one. Measured GPU bandwidth is ~212-215 GB/s of the 256 GB/s theoretical, so estimates here run a little high.", "sources": [ "https://www.bee-link.com/products/beelink-gtr9-pro-amd-ryzen-ai-max-395", "https://github.com/hogeheer499-commits/strix-halo-guide" ], "estimate_efficiency_moe": 0.55, "estimate_note": "MoE decode on Strix Halo reaches ~50-65% of the bandwidth ceiling (Qwen3 30B-A3B: 66-86 tok/s measured vs 128 ceiling), against ~30% on Apple Silicon." }, { "id": "minisforum-ms-s1-max-128", "family": "Strix Halo", "chip": "Minisforum MS-S1 Max", "chip_variant": "Ryzen AI Max+ 395 · Radeon 8060S, 40 CU", "unified_memory_gb": 128, "memory_bandwidth_gbs": 256, "usable_memory_gb": 96, "price_usd": 3799, "idle_watts": null, "load_watts": 133, "load_watts_status": "stand_in", "load_watts_note": "No wall measurement found for this unit; stand-in is the Framework Desktop's measured 133 W on the same chip. Minisforum's performance mode allows 130 W.", "generation": "current", "status": "128 GB / 2 TB $3,799 on sale when checked (regular $4,749), shipping mid-September.", "notes": "Windows lets you dedicate up to 75% of RAM to the GPU (Variable Graphics Memory in AMD Adrenalin; Framework documents 96 GB on a 128 GB board). On Linux the amdgpu GTT pool can be raised to roughly 108-120 GB with ttm.pages_limit / amdgpu.gttsize boot parameters. The Windows figure is used here as the conservative one. Measured GPU bandwidth is ~212-215 GB/s of the 256 GB/s theoretical, so estimates here run a little high.", "sources": [ "https://store.minisforum.com/products/minisforum-ms-s1-max-mini-pc" ], "estimate_efficiency_moe": 0.55, "estimate_note": "MoE decode on Strix Halo reaches ~50-65% of the bandwidth ceiling (Qwen3 30B-A3B: 66-86 tok/s measured vs 128 ceiling), against ~30% on Apple Silicon." }, { "id": "hp-z2-mini-g1a-128", "family": "Strix Halo", "chip": "HP Z2 Mini G1a", "chip_variant": "Ryzen AI Max+ PRO 395 · Radeon 8060S, 40 CU", "unified_memory_gb": 128, "memory_bandwidth_gbs": 256, "usable_memory_gb": 96, "price_usd": 5544, "idle_watts": null, "load_watts": 133, "load_watts_status": "stand_in", "load_watts_note": "StorageReview saw the GPU alone drawing just under 100 W on gpt-oss-120b (GPU telemetry, not the wall); stand-in is the Framework Desktop's measured 133 W on the same silicon.", "generation": "current", "status": "128 GB / 1 TB $5,543.92 at Staples when checked; StorageReview cited a $4,781 list price in 2025. Enterprise pricing varies a lot by reseller.", "notes": "Windows lets you dedicate up to 75% of RAM to the GPU (Variable Graphics Memory in AMD Adrenalin; Framework documents 96 GB on a 128 GB board). On Linux the amdgpu GTT pool can be raised to roughly 108-120 GB with ttm.pages_limit / amdgpu.gttsize boot parameters. The Windows figure is used here as the conservative one. Measured GPU bandwidth is ~212-215 GB/s of the 256 GB/s theoretical, so estimates here run a little high.", "sources": [ "https://www.staples.com/hp-z2-mini-g1a-desktop-computer-ryzen-ai-max-pro-395-128gb-ram-1tb-ssd-windows-11-pro-mouse-keyboard-included/product_IM1YS5467", "https://www.storagereview.com/review/hp-z2-mini-g1a-review-running-gpt-oss-120b-without-a-discrete-gpu" ], "estimate_efficiency_moe": 0.55, "estimate_note": "MoE decode on Strix Halo reaches ~50-65% of the bandwidth ceiling (Qwen3 30B-A3B: 66-86 tok/s measured vs 128 ceiling), against ~30% on Apple Silicon." }, { "id": "macbook-air-m5-16", "family": "MacBook Air", "chip": "M5 (13-inch)", "chip_variant": "10-core CPU / 10-core GPU", "unified_memory_gb": 16, "memory_bandwidth_gbs": 153, "usable_memory_gb": 10.5, "price_usd": 1299, "idle_watts": null, "load_watts": 65, "load_watts_status": "stand_in", "load_watts_note": "Apple has published no power figures for the 2026 Macs, and none at all for sustained GPU load on a laptop. Stand-in: the desktop maximum for the equivalent chip (M4 Mac mini, 65 W). A laptop draws less than that, so the electricity line here is an over-estimate.", "generation": "current", "status": "Shipping. Fanless, so it throttles hardest of any Mac here under a long generation.", "notes": "macOS lets the GPU wire roughly two-thirds of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. On a laptop, sustained speed drops once the chassis warms up and the fans cap out, so a desktop with the same chip will hold a higher tokens/sec over a long generation than the figures here suggest.", "sources": [ "https://www.apple.com/macbook-air/specs/" ] }, { "id": "macbook-pro-14-m5-16", "family": "MacBook Pro", "chip": "M5 (14-inch)", "chip_variant": "10-core CPU / 10-core GPU", "unified_memory_gb": 16, "memory_bandwidth_gbs": 153, "usable_memory_gb": 10.5, "price_usd": 1999, "idle_watts": null, "load_watts": 65, "load_watts_status": "stand_in", "load_watts_note": "Apple has published no power figures for the 2026 Macs, and none at all for sustained GPU load on a laptop. Stand-in: the desktop maximum for the equivalent chip (M4 Mac mini, 65 W). A laptop draws less than that, so the electricity line here is an over-estimate.", "generation": "current", "status": "Shipping.", "notes": "macOS lets the GPU wire roughly two-thirds of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. On a laptop, sustained speed drops once the chassis warms up and the fans cap out, so a desktop with the same chip will hold a higher tokens/sec over a long generation than the figures here suggest.", "sources": [ "https://www.apple.com/macbook-pro/specs/", "https://www.apple.com/shop/buy-mac/macbook-pro", "https://support.apple.com/en-us/102059" ] }, { "id": "macbook-pro-14-m5-32", "family": "MacBook Pro", "chip": "M5 (14-inch)", "chip_variant": "10-core CPU / 10-core GPU", "unified_memory_gb": 32, "memory_bandwidth_gbs": 153, "usable_memory_gb": 21.0, "price_usd": 2399, "idle_watts": null, "load_watts": 65, "load_watts_status": "stand_in", "load_watts_note": "Apple has published no power figures for the 2026 Macs, and none at all for sustained GPU load on a laptop. Stand-in: the desktop maximum for the equivalent chip (M4 Mac mini, 65 W). A laptop draws less than that, so the electricity line here is an over-estimate.", "generation": "current", "status": "Shipping.", "notes": "macOS lets the GPU wire roughly two-thirds of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. On a laptop, sustained speed drops once the chassis warms up and the fans cap out, so a desktop with the same chip will hold a higher tokens/sec over a long generation than the figures here suggest.", "sources": [ "https://www.apple.com/macbook-pro/specs/", "https://www.apple.com/shop/buy-mac/macbook-pro", "https://support.apple.com/en-us/102059" ] }, { "id": "macbook-pro-16-m5-pro-24", "family": "MacBook Pro", "chip": "M5 Pro (16-inch)", "chip_variant": "15-core CPU / 16-core GPU", "unified_memory_gb": 24, "memory_bandwidth_gbs": 307, "usable_memory_gb": 16.0, "price_usd": 2999, "idle_watts": null, "load_watts": 140, "load_watts_status": "stand_in", "load_watts_note": "Apple has published no power figures for the 2026 Macs, and none at all for sustained GPU load on a laptop. Stand-in: the desktop maximum for the equivalent chip (M4 Pro Mac mini, 140 W). A laptop draws less than that, so the electricity line here is an over-estimate.", "generation": "current", "status": "Shipping.", "notes": "macOS lets the GPU wire roughly two-thirds of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. On a laptop, sustained speed drops once the chassis warms up and the fans cap out, so a desktop with the same chip will hold a higher tokens/sec over a long generation than the figures here suggest.", "sources": [ "https://www.apple.com/macbook-pro/specs/", "https://www.apple.com/shop/buy-mac/macbook-pro", "https://support.apple.com/en-us/102059" ] }, { "id": "macbook-pro-16-m5-pro-48", "family": "MacBook Pro", "chip": "M5 Pro (16-inch)", "chip_variant": "15-core CPU / 16-core GPU", "unified_memory_gb": 48, "memory_bandwidth_gbs": 307, "usable_memory_gb": 36.0, "price_usd": 3599, "idle_watts": null, "load_watts": 140, "load_watts_status": "stand_in", "load_watts_note": "Apple has published no power figures for the 2026 Macs, and none at all for sustained GPU load on a laptop. Stand-in: the desktop maximum for the equivalent chip (M4 Pro Mac mini, 140 W). A laptop draws less than that, so the electricity line here is an over-estimate.", "generation": "current", "status": "Shipping.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. On a laptop, sustained speed drops once the chassis warms up and the fans cap out, so a desktop with the same chip will hold a higher tokens/sec over a long generation than the figures here suggest.", "sources": [ "https://www.apple.com/macbook-pro/specs/", "https://www.apple.com/shop/buy-mac/macbook-pro", "https://support.apple.com/en-us/102059" ] }, { "id": "macbook-pro-16-m5-pro-64", "family": "MacBook Pro", "chip": "M5 Pro (16-inch)", "chip_variant": "15-core CPU / 16-core GPU", "unified_memory_gb": 64, "memory_bandwidth_gbs": 307, "usable_memory_gb": 48.0, "price_usd": 3999, "idle_watts": null, "load_watts": 140, "load_watts_status": "stand_in", "load_watts_note": "Apple has published no power figures for the 2026 Macs, and none at all for sustained GPU load on a laptop. Stand-in: the desktop maximum for the equivalent chip (M4 Pro Mac mini, 140 W). A laptop draws less than that, so the electricity line here is an over-estimate.", "generation": "current", "status": "Shipping.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. On a laptop, sustained speed drops once the chassis warms up and the fans cap out, so a desktop with the same chip will hold a higher tokens/sec over a long generation than the figures here suggest.", "sources": [ "https://www.apple.com/macbook-pro/specs/", "https://www.apple.com/shop/buy-mac/macbook-pro", "https://support.apple.com/en-us/102059" ] }, { "id": "macbook-pro-16-m5-max-48", "family": "MacBook Pro", "chip": "M5 Max (16-inch)", "chip_variant": "18-core CPU / 40-core GPU", "unified_memory_gb": 48, "memory_bandwidth_gbs": 614, "usable_memory_gb": 36.0, "price_usd": 4999, "idle_watts": null, "load_watts": 145, "load_watts_status": "stand_in", "load_watts_note": "Apple has published no power figures for the 2026 Macs, and none at all for sustained GPU load on a laptop. Stand-in: the desktop maximum for the equivalent chip (M4 Max Mac Studio, 145 W). A laptop draws less than that, so the electricity line here is an over-estimate.", "generation": "current", "status": "Shipping.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. On a laptop, sustained speed drops once the chassis warms up and the fans cap out, so a desktop with the same chip will hold a higher tokens/sec over a long generation than the figures here suggest.", "sources": [ "https://www.apple.com/macbook-pro/specs/", "https://www.apple.com/shop/buy-mac/macbook-pro", "https://support.apple.com/en-us/102059" ] }, { "id": "macbook-pro-16-m5-max-64", "family": "MacBook Pro", "chip": "M5 Max (16-inch)", "chip_variant": "18-core CPU / 40-core GPU", "unified_memory_gb": 64, "memory_bandwidth_gbs": 614, "usable_memory_gb": 48.0, "price_usd": 5399, "idle_watts": null, "load_watts": 145, "load_watts_status": "stand_in", "load_watts_note": "Apple has published no power figures for the 2026 Macs, and none at all for sustained GPU load on a laptop. Stand-in: the desktop maximum for the equivalent chip (M4 Max Mac Studio, 145 W). A laptop draws less than that, so the electricity line here is an over-estimate.", "generation": "current", "status": "Shipping.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. On a laptop, sustained speed drops once the chassis warms up and the fans cap out, so a desktop with the same chip will hold a higher tokens/sec over a long generation than the figures here suggest.", "sources": [ "https://www.apple.com/macbook-pro/specs/", "https://www.apple.com/shop/buy-mac/macbook-pro", "https://support.apple.com/en-us/102059" ] }, { "id": "macbook-pro-16-m5-max-128", "family": "MacBook Pro", "chip": "M5 Max (16-inch)", "chip_variant": "18-core CPU / 40-core GPU", "unified_memory_gb": 128, "memory_bandwidth_gbs": 614, "usable_memory_gb": 96.0, "price_usd": 6999, "idle_watts": null, "load_watts": 145, "load_watts_status": "stand_in", "load_watts_note": "Apple has published no power figures for the 2026 Macs, and none at all for sustained GPU load on a laptop. Stand-in: the desktop maximum for the equivalent chip (M4 Max Mac Studio, 145 W). A laptop draws less than that, so the electricity line here is an over-estimate.", "generation": "current", "status": "Shipping.", "notes": "macOS lets the GPU wire roughly 75% of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. On a laptop, sustained speed drops once the chassis warms up and the fans cap out, so a desktop with the same chip will hold a higher tokens/sec over a long generation than the figures here suggest.", "sources": [ "https://www.apple.com/macbook-pro/specs/", "https://www.apple.com/shop/buy-mac/macbook-pro", "https://support.apple.com/en-us/102059" ] }, { "id": "gmktec-evo-x3-128", "family": "Strix Halo", "chip": "GMKtec EVO-X3", "chip_variant": "Ryzen AI Max+ 395 · Radeon 8060S, 40 CU", "unified_memory_gb": 128, "memory_bandwidth_gbs": 256, "usable_memory_gb": 96, "price_usd": 3600, "idle_watts": null, "load_watts": 150, "load_watts_status": "third_party_measured", "load_watts_note": "No separate measurement for the X3; ServeTheHome measured 147-160 W at the wall on the EVO-X2, same chip and TDP.", "generation": "current", "status": "Shipping. Replaced the EVO-X2 in July 2026.", "notes": "Windows lets you dedicate up to 75% of RAM to the GPU (Variable Graphics Memory in AMD Adrenalin). On Linux the amdgpu GTT pool can be raised to roughly 108-120 GB with ttm.pages_limit / amdgpu.gttsize boot parameters. The Windows figure is used here as the conservative one. Measured GPU bandwidth is ~212-215 GB/s of the 256 GB/s theoretical, so estimates here run a little high.", "sources": [ "https://www.gmktec.com/products/amd-ryzen-ai-max-395-evo-x3-ai-mini-pc" ], "estimate_efficiency_moe": 0.55, "estimate_note": "MoE decode on Strix Halo reaches ~50-65% of the bandwidth ceiling (Qwen3 30B-A3B: 66-86 tok/s measured vs 128 ceiling), against ~30% on Apple Silicon." }, { "id": "framework-desktop-495-192", "family": "Strix Halo", "chip": "Framework Desktop", "chip_variant": "Ryzen AI Max+ PRO 495 'Gorgon Halo' · 192 GB", "unified_memory_gb": 192, "memory_bandwidth_gbs": 273, "usable_memory_gb": 160, "price_usd": null, "idle_watts": null, "load_watts": 133, "load_watts_status": "stand_in", "load_watts_note": "No measurement for Gorgon Halo; stand-in is the Framework Desktop's measured 133 W on the previous chip.", "generation": "current", "status": "Announced, not yet priced or orderable.", "notes": "192 GB unified, of which AMD states 160 GB can be allocated to the GPU — the largest GPU-addressable pool of anything on this list under $10,000, if it ships at a sane price. Windows lets you dedicate up to 75% of RAM to the GPU (Variable Graphics Memory in AMD Adrenalin). On Linux the amdgpu GTT pool can be raised to roughly 108-120 GB with ttm.pages_limit / amdgpu.gttsize boot parameters. The Windows figure is used here as the conservative one. Measured GPU bandwidth is ~212-215 GB/s of the 256 GB/s theoretical, so estimates here run a little high.", "sources": [ "https://frame.work/desktop" ], "estimate_efficiency_moe": 0.55, "estimate_note": "MoE decode on Strix Halo reaches ~50-65% of the bandwidth ceiling (Qwen3 30B-A3B: 66-86 tok/s measured vs 128 ceiling), against ~30% on Apple Silicon.", "TODO": "price_usd: Framework has not announced the 192 GB price." }, { "id": "macbook-air-m5-15-16", "family": "MacBook Air", "chip": "M5 (15-inch)", "chip_variant": "10-core CPU / 10-core GPU", "unified_memory_gb": 16, "memory_bandwidth_gbs": 153, "usable_memory_gb": 10.5, "price_usd": 1499, "idle_watts": null, "load_watts": 65, "load_watts_status": "stand_in", "load_watts_note": "Apple has published no power figures for the 2026 Macs, and none at all for sustained GPU load on a laptop. Stand-in: the desktop maximum for the equivalent chip (M4 Mac mini, 65 W). A laptop draws less than that, so the electricity line here is an over-estimate.", "generation": "current", "status": "Shipping. Fanless, so it throttles hardest of any Mac here under a long generation.", "notes": "macOS lets the GPU wire roughly two-thirds of unified memory by default (llama.cpp discussion #2182: 66.7% at 32 GiB or less, 75% above). Raise it with `sudo sysctl iogpu.wired_limit_mb`. On a laptop, sustained speed drops once the chassis warms up and the fans cap out, so a desktop with the same chip will hold a higher tokens/sec over a long generation than the figures here suggest.", "sources": [ "https://www.apple.com/macbook-air/specs/" ] } ]