[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-minimax-m25":3},{"id":4,"slug":5,"name":6,"hf_id":7,"organization":8,"architecture":9,"architecture_config":10,"parameters":11,"parameters_label":12,"context_length":13,"license":14,"downloads":15,"likes":16,"status":17,"description":10,"huggingface_url":18,"created_at":19,"seo":20,"category":23,"family":26,"score":29,"variants":30,"quantizations":31,"requirements":58,"gpu_recommendations":143,"fitting_gpus":271,"community":279},236,"minimax-m25","MiniMax M2.5","MiniMaxAI\u002FMiniMax-M2.5","MiniMaxAI","moe_transformer",null,230000000000,"230B",200000,"MIT",699232,1502,"published","https:\u002F\u002Fhuggingface.co\u002FMiniMaxAI\u002FMiniMax-M2.5","2026-08-12T03:11:18+00:00",{"title":6,"description":21,"image":10,"robots":22,"canonical_url":10},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":24,"slug":25},"LLM","llm",{"name":27,"slug":28},"MiniMax","minimax",0.2677,[],[32,37,42,47,51,55],{"format":33,"bits":34,"size_factor":35,"quality_loss":36},"AWQ",4,0.263,0.03,{"format":38,"bits":39,"size_factor":40,"quality_loss":41},"FP16",16,1,0,{"format":43,"bits":44,"size_factor":45,"quality_loss":46},"FP8",8,0.5,0.01,{"format":48,"bits":34,"size_factor":49,"quality_loss":50},"GGUF",0.25,0.05,{"format":52,"bits":34,"size_factor":53,"quality_loss":54},"GPTQ",0.275,0.04,{"format":56,"bits":44,"size_factor":45,"quality_loss":57},"INT8",0.02,{"FP16":59,"FP8":83,"INT8":97,"AWQ":101,"GPTQ":115,"GGUF":129},{"minimum":60,"production":69,"recommended":76},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":63,"kv_cache_gb":45,"vram_gb":64,"ram_gb":65,"disk_gb":66,"requirement_source":67,"confidence":68},"minimum",4096,428.41,546.89,820.33,707.32,"estimated",0.95,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":63,"kv_cache_gb":72,"vram_gb":73,"ram_gb":74,"disk_gb":75,"requirement_source":67,"confidence":68},"production",16384,64,708.8,1063.21,772.32,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":63,"kv_cache_gb":44,"vram_gb":80,"ram_gb":81,"disk_gb":82,"requirement_source":67,"confidence":68},"recommended",8192,2,587.27,880.91,733.32,{"minimum":84,"production":89,"recommended":93},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":85,"kv_cache_gb":45,"vram_gb":86,"ram_gb":87,"disk_gb":88,"requirement_source":67,"confidence":68},214.2,275.92,413.88,373.16,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":85,"kv_cache_gb":72,"vram_gb":90,"ram_gb":91,"disk_gb":92,"requirement_source":67,"confidence":68},413.2,619.8,438.16,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":85,"kv_cache_gb":44,"vram_gb":94,"ram_gb":95,"disk_gb":96,"requirement_source":67,"confidence":68},303.99,455.98,399.16,{"minimum":98,"production":99,"recommended":100},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":85,"kv_cache_gb":45,"vram_gb":86,"ram_gb":87,"disk_gb":88,"requirement_source":67,"confidence":68},{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":85,"kv_cache_gb":72,"vram_gb":90,"ram_gb":91,"disk_gb":92,"requirement_source":67,"confidence":68},{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":85,"kv_cache_gb":44,"vram_gb":94,"ram_gb":95,"disk_gb":96,"requirement_source":67,"confidence":68},{"minimum":102,"production":107,"recommended":111},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":103,"kv_cache_gb":45,"vram_gb":104,"ram_gb":105,"disk_gb":106,"requirement_source":67,"confidence":68},111.12,145.51,218.27,212.34,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":103,"kv_cache_gb":72,"vram_gb":108,"ram_gb":109,"disk_gb":110,"requirement_source":67,"confidence":68},270.94,406.42,277.34,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":103,"kv_cache_gb":44,"vram_gb":112,"ram_gb":113,"disk_gb":114,"requirement_source":67,"confidence":68},167.65,251.48,238.34,{"minimum":116,"production":121,"recommended":125},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":117,"kv_cache_gb":45,"vram_gb":118,"ram_gb":119,"disk_gb":120,"requirement_source":67,"confidence":68},112.46,147.21,220.81,214.43,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":117,"kv_cache_gb":72,"vram_gb":122,"ram_gb":123,"disk_gb":124,"requirement_source":67,"confidence":68},272.79,409.19,279.43,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":117,"kv_cache_gb":44,"vram_gb":126,"ram_gb":127,"disk_gb":128,"requirement_source":67,"confidence":68},169.42,254.14,240.43,{"minimum":130,"production":135,"recommended":139},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":131,"kv_cache_gb":45,"vram_gb":132,"ram_gb":133,"disk_gb":134,"requirement_source":67,"confidence":68},115.13,150.6,225.89,218.61,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":131,"kv_cache_gb":72,"vram_gb":136,"ram_gb":137,"disk_gb":138,"requirement_source":67,"confidence":68},276.49,414.73,283.61,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":131,"kv_cache_gb":44,"vram_gb":140,"ram_gb":141,"disk_gb":142,"requirement_source":67,"confidence":68},172.97,259.45,244.61,{"recommended":144,"cheapest":164,"performance":181,"min_complexity":190,"cluster_options":199,"no_match":268,"required_vram_gb":112,"candidate_count":269,"usage":270},{"gpu_model":145,"gpu_model_id":146,"gpu_count":34,"total_vram_gb":147,"hour_price":148,"monthly_cost":149,"providers":150,"bandwidth_gbps":152,"fp16_tflops":153,"interconnect_score":154,"parallel_efficiency":155,"score":156,"tier":157,"perf_metric":158,"perf_label":159,"perf_value":160,"workload":10,"cost_metric":161},"H200 SXM",11,564,3.5,10220,[151],"lambda",4800,989,40,0.6,0.3486,"balanced","tok\u002Fs","103.7 tok\u002Fs",103.7,{"effective_value":162,"effective_unit":158,"capacity_factor":163},72.6,0.7,{"gpu_model":165,"gpu_model_id":166,"gpu_count":44,"total_vram_gb":167,"hour_price":168,"monthly_cost":169,"providers":170,"bandwidth_gbps":173,"fp16_tflops":174,"interconnect_score":154,"parallel_efficiency":175,"score":176,"tier":157,"perf_metric":158,"perf_label":177,"perf_value":178,"workload":10,"cost_metric":179},"RTX 3090",3,192,0.0678,395.95,[171,172],"vast","tensordock",936,71,0.4,0.5403,"27 tok\u002Fs",27,{"effective_value":180,"effective_unit":158,"capacity_factor":163},18.9,{"gpu_model":145,"gpu_model_id":146,"gpu_count":44,"total_vram_gb":182,"hour_price":148,"monthly_cost":183,"providers":184,"bandwidth_gbps":152,"fp16_tflops":153,"interconnect_score":154,"parallel_efficiency":175,"score":185,"tier":157,"perf_metric":158,"perf_label":186,"perf_value":187,"workload":10,"cost_metric":188},1128,20440,[151],0.2197,"138.2 tok\u002Fs",138.2,{"effective_value":189,"effective_unit":158,"capacity_factor":163},96.7,{"gpu_model":191,"gpu_model_id":178,"gpu_count":40,"total_vram_gb":192,"hour_price":193,"monthly_cost":194,"providers":195,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":154,"parallel_efficiency":40,"score":196,"tier":157,"perf_metric":158,"perf_label":197,"perf_value":41,"workload":10,"cost_metric":198},"B200",180,4.1271,3012.78,[171],0.5789,"0 tok\u002Fs",{"effective_value":41,"effective_unit":158,"capacity_factor":163},[200,211,216,225,232,241,248,259],{"gpu_model":201,"gpu_model_id":202,"gpu_count":44,"total_vram_gb":203,"hour_price":204,"monthly_cost":205,"providers":206,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":154,"parallel_efficiency":175,"insufficient":207,"vram_shortfall_gb":208,"score":209,"tier":210},"Tesla V100",17,128,0.0272,158.85,[171],true,39.65,0.5034,"cheapest",{"gpu_model":212,"gpu_model_id":213,"gpu_count":44,"total_vram_gb":203,"hour_price":168,"monthly_cost":169,"providers":214,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":154,"parallel_efficiency":175,"insufficient":207,"vram_shortfall_gb":208,"score":215,"tier":157},"RTX 4070S Ti",22,[171],0.4984,{"gpu_model":217,"gpu_model_id":218,"gpu_count":44,"total_vram_gb":203,"hour_price":219,"monthly_cost":220,"providers":221,"bandwidth_gbps":222,"fp16_tflops":223,"interconnect_score":154,"parallel_efficiency":175,"insufficient":207,"vram_shortfall_gb":208,"score":224,"tier":157},"RTX 5070 Ti",18,0.0849,495.82,[171],896,88,0.5168,{"gpu_model":226,"gpu_model_id":227,"gpu_count":44,"total_vram_gb":203,"hour_price":228,"monthly_cost":229,"providers":230,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":154,"parallel_efficiency":175,"insufficient":207,"vram_shortfall_gb":208,"score":231,"tier":157},"RTX 4080S",21,0.1192,696.13,[171],0.4922,{"gpu_model":233,"gpu_model_id":234,"gpu_count":44,"total_vram_gb":203,"hour_price":235,"monthly_cost":236,"providers":237,"bandwidth_gbps":238,"fp16_tflops":239,"interconnect_score":154,"parallel_efficiency":175,"insufficient":207,"vram_shortfall_gb":208,"score":240,"tier":157},"RTX 5080",23,0.1344,784.9,[171],960,113,0.5123,{"gpu_model":242,"gpu_model_id":243,"gpu_count":44,"total_vram_gb":203,"hour_price":244,"monthly_cost":245,"providers":246,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":154,"parallel_efficiency":175,"insufficient":207,"vram_shortfall_gb":208,"score":247,"tier":157},"RTX PRO 4500",25,0.3222,1881.65,[171],0.4675,{"gpu_model":249,"gpu_model_id":250,"gpu_count":44,"total_vram_gb":251,"hour_price":252,"monthly_cost":253,"providers":254,"bandwidth_gbps":255,"fp16_tflops":256,"interconnect_score":154,"parallel_efficiency":175,"insufficient":207,"vram_shortfall_gb":257,"score":258,"tier":210},"RTX 5070",28,96,0.0101,58.98,[171],672,61,71.65,0.5208,{"gpu_model":260,"gpu_model_id":261,"gpu_count":44,"total_vram_gb":72,"hour_price":262,"monthly_cost":263,"providers":264,"bandwidth_gbps":265,"fp16_tflops":261,"interconnect_score":154,"parallel_efficiency":175,"insufficient":207,"vram_shortfall_gb":266,"score":267,"tier":157},"RTX 5060 Ti",24,0.0544,317.7,[171],448,103.65,0.5103,false,12,"value",[272,276],{"name":191,"vram_gb":192,"cheapest_hour":193,"cost":273,"provider":171,"estimated_tps":41},{"hourly":193,"daily":274,"monthly":194,"yearly":275},99.05,36153.4,{"name":277,"vram_gb":192,"cheapest_hour":10,"cost":10,"provider":10,"estimated_tps":278},"B200 SXM",70.2,{"posts_count":41,"benchmarks_count":41}]