[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-qwen35-397b-a17b":3},{"id":4,"slug":5,"name":6,"hf_id":7,"organization":8,"architecture":9,"architecture_config":10,"parameters":11,"parameters_label":12,"context_length":13,"license":14,"downloads":15,"likes":16,"status":17,"description":10,"huggingface_url":18,"created_at":19,"seo":20,"category":23,"family":26,"score":28,"variants":29,"quantizations":30,"requirements":57,"gpu_recommendations":142,"fitting_gpus":329,"community":330},334,"qwen35-397b-a17b","Qwen3.5 397B-A17B","Qwen\u002FQwen3.5-397B-A17B","Qwen","moe_transformer",null,397000000000,"397B",262144,"MIT",315709,1550,"published","https:\u002F\u002Fhuggingface.co\u002FQwen\u002FQwen3.5-397B-A17B","2026-08-12T03:11:23+00:00",{"title":6,"description":21,"image":10,"robots":22,"canonical_url":10},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":24,"slug":25},"LLM","llm",{"name":8,"slug":27},"qwen",0.2676,[],[31,36,41,46,50,54],{"format":32,"bits":33,"size_factor":34,"quality_loss":35},"AWQ",4,0.263,0.03,{"format":37,"bits":38,"size_factor":39,"quality_loss":40},"FP16",16,1,0,{"format":42,"bits":43,"size_factor":44,"quality_loss":45},"FP8",8,0.5,0.01,{"format":47,"bits":33,"size_factor":48,"quality_loss":49},"GGUF",0.25,0.05,{"format":51,"bits":33,"size_factor":52,"quality_loss":53},"GPTQ",0.275,0.04,{"format":55,"bits":43,"size_factor":44,"quality_loss":56},"INT8",0.02,{"FP16":58,"FP8":82,"INT8":96,"AWQ":100,"GPTQ":114,"GGUF":128},{"minimum":59,"production":68,"recommended":75},{"scenario":60,"context_length":61,"batch_size":39,"weight_gb":62,"kv_cache_gb":44,"vram_gb":63,"ram_gb":64,"disk_gb":65,"requirement_source":66,"confidence":67},"minimum",4096,739.47,940.38,1410.57,1192.57,"estimated",0.95,{"scenario":69,"context_length":70,"batch_size":33,"weight_gb":62,"kv_cache_gb":71,"vram_gb":72,"ram_gb":73,"disk_gb":74,"requirement_source":66,"confidence":67},"production",16384,64,1138.07,1707.1,1257.57,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":62,"kv_cache_gb":43,"vram_gb":79,"ram_gb":80,"disk_gb":81,"requirement_source":66,"confidence":67},"recommended",8192,2,998.65,1497.97,1218.57,{"minimum":83,"production":88,"recommended":92},{"scenario":60,"context_length":61,"batch_size":39,"weight_gb":84,"kv_cache_gb":44,"vram_gb":85,"ram_gb":86,"disk_gb":87,"requirement_source":66,"confidence":67},369.74,472.66,709,615.79,{"scenario":69,"context_length":70,"batch_size":33,"weight_gb":84,"kv_cache_gb":71,"vram_gb":89,"ram_gb":90,"disk_gb":91,"requirement_source":66,"confidence":67},627.83,941.75,680.79,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":84,"kv_cache_gb":43,"vram_gb":93,"ram_gb":94,"disk_gb":95,"requirement_source":66,"confidence":67},509.67,764.51,641.79,{"minimum":97,"production":98,"recommended":99},{"scenario":60,"context_length":61,"batch_size":39,"weight_gb":84,"kv_cache_gb":44,"vram_gb":85,"ram_gb":86,"disk_gb":87,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":33,"weight_gb":84,"kv_cache_gb":71,"vram_gb":89,"ram_gb":90,"disk_gb":91,"requirement_source":66,"confidence":67},{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":84,"kv_cache_gb":43,"vram_gb":93,"ram_gb":94,"disk_gb":95,"requirement_source":66,"confidence":67},{"minimum":101,"production":106,"recommended":110},{"scenario":60,"context_length":61,"batch_size":39,"weight_gb":102,"kv_cache_gb":44,"vram_gb":103,"ram_gb":104,"disk_gb":105,"requirement_source":66,"confidence":67},191.8,247.58,371.37,338.21,{"scenario":69,"context_length":70,"batch_size":33,"weight_gb":102,"kv_cache_gb":71,"vram_gb":107,"ram_gb":108,"disk_gb":109,"requirement_source":66,"confidence":67},382.28,573.43,403.21,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":102,"kv_cache_gb":43,"vram_gb":111,"ram_gb":112,"disk_gb":113,"requirement_source":66,"confidence":67},274.36,411.53,364.21,{"minimum":115,"production":120,"recommended":124},{"scenario":60,"context_length":61,"batch_size":39,"weight_gb":116,"kv_cache_gb":44,"vram_gb":117,"ram_gb":118,"disk_gb":119,"requirement_source":66,"confidence":67},194.11,250.5,375.75,341.81,{"scenario":69,"context_length":70,"batch_size":33,"weight_gb":116,"kv_cache_gb":71,"vram_gb":121,"ram_gb":122,"disk_gb":123,"requirement_source":66,"confidence":67},385.47,578.21,406.81,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":116,"kv_cache_gb":43,"vram_gb":125,"ram_gb":126,"disk_gb":127,"requirement_source":66,"confidence":67},277.41,416.12,367.81,{"minimum":129,"production":134,"recommended":138},{"scenario":60,"context_length":61,"batch_size":39,"weight_gb":130,"kv_cache_gb":44,"vram_gb":131,"ram_gb":132,"disk_gb":133,"requirement_source":66,"confidence":67},198.73,256.35,384.52,349.02,{"scenario":69,"context_length":70,"batch_size":33,"weight_gb":130,"kv_cache_gb":71,"vram_gb":135,"ram_gb":136,"disk_gb":137,"requirement_source":66,"confidence":67},391.85,587.78,414.02,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":130,"kv_cache_gb":43,"vram_gb":139,"ram_gb":140,"disk_gb":141,"requirement_source":66,"confidence":67},283.52,425.29,375.02,{"recommended":143,"cheapest":163,"performance":178,"min_complexity":188,"cluster_options":197,"no_match":327,"required_vram_gb":111,"candidate_count":43,"usage":328},{"gpu_model":144,"gpu_model_id":145,"gpu_count":43,"total_vram_gb":146,"hour_price":147,"monthly_cost":148,"providers":149,"bandwidth_gbps":151,"fp16_tflops":152,"interconnect_score":153,"parallel_efficiency":154,"score":155,"tier":156,"perf_metric":157,"perf_label":158,"perf_value":159,"workload":10,"cost_metric":160},"H200 SXM",11,1128,3.5,20440,[150],"lambda",4800,989,40,0.4,0.2846,"balanced","tok\u002Fs","80.1 tok\u002Fs",80.1,{"effective_value":161,"effective_unit":157,"capacity_factor":162},56.1,0.7,{"gpu_model":164,"gpu_model_id":165,"gpu_count":43,"total_vram_gb":166,"hour_price":167,"monthly_cost":168,"providers":169,"bandwidth_gbps":171,"fp16_tflops":172,"interconnect_score":153,"parallel_efficiency":154,"score":173,"tier":156,"perf_metric":157,"perf_label":174,"perf_value":175,"workload":10,"cost_metric":176},"RTX A6000",6,384,0.2874,1678.42,[170,150],"vast",768,77,0.5331,"12.8 tok\u002Fs",12.8,{"effective_value":177,"effective_unit":157,"capacity_factor":162},9,{"gpu_model":144,"gpu_model_id":145,"gpu_count":33,"total_vram_gb":179,"hour_price":147,"monthly_cost":180,"providers":181,"bandwidth_gbps":151,"fp16_tflops":152,"interconnect_score":153,"parallel_efficiency":182,"score":183,"tier":156,"perf_metric":157,"perf_label":184,"perf_value":185,"workload":10,"cost_metric":186},564,10220,[150],0.6,0.3991,"60.1 tok\u002Fs",60.1,{"effective_value":187,"effective_unit":157,"capacity_factor":162},42.1,{"gpu_model":144,"gpu_model_id":145,"gpu_count":78,"total_vram_gb":189,"hour_price":147,"monthly_cost":190,"providers":191,"bandwidth_gbps":151,"fp16_tflops":152,"interconnect_score":153,"parallel_efficiency":192,"score":193,"tier":156,"perf_metric":157,"perf_label":194,"perf_value":153,"workload":10,"cost_metric":195},282,5110,[150],0.8,0.5426,"40 tok\u002Fs",{"effective_value":196,"effective_unit":157,"capacity_factor":162},28,[198,210,222,230,238,247,256,263,273,278,287,294,301,308,318],{"gpu_model":199,"gpu_model_id":78,"gpu_count":43,"total_vram_gb":200,"hour_price":201,"monthly_cost":202,"providers":203,"bandwidth_gbps":205,"fp16_tflops":206,"interconnect_score":153,"parallel_efficiency":154,"insufficient":207,"vram_shortfall_gb":208,"score":209,"tier":156},"RTX 5090",256,0.337,1968.08,[170,204],"runpod",1792,209,true,18.36,0.5066,{"gpu_model":211,"gpu_model_id":212,"gpu_count":43,"total_vram_gb":213,"hour_price":214,"monthly_cost":215,"providers":216,"bandwidth_gbps":218,"fp16_tflops":219,"interconnect_score":153,"parallel_efficiency":154,"insufficient":207,"vram_shortfall_gb":220,"score":221,"tier":156},"RTX 3090",3,192,0.0678,395.95,[170,217],"tensordock",936,71,82.36,0.5198,{"gpu_model":223,"gpu_model_id":39,"gpu_count":43,"total_vram_gb":213,"hour_price":224,"monthly_cost":225,"providers":226,"bandwidth_gbps":227,"fp16_tflops":228,"interconnect_score":153,"parallel_efficiency":154,"insufficient":207,"vram_shortfall_gb":220,"score":229,"tier":156},"RTX 4090",0.1344,784.9,[170,217,204,150],1008,165,0.5134,{"gpu_model":231,"gpu_model_id":232,"gpu_count":43,"total_vram_gb":213,"hour_price":233,"monthly_cost":234,"providers":235,"bandwidth_gbps":227,"fp16_tflops":236,"interconnect_score":153,"parallel_efficiency":154,"insufficient":207,"vram_shortfall_gb":220,"score":237,"tier":156},"RTX 4090D",5,0.1602,935.57,[170],82.6,0.5102,{"gpu_model":239,"gpu_model_id":240,"gpu_count":43,"total_vram_gb":213,"hour_price":241,"monthly_cost":242,"providers":243,"bandwidth_gbps":244,"fp16_tflops":245,"interconnect_score":153,"parallel_efficiency":154,"insufficient":207,"vram_shortfall_gb":220,"score":246,"tier":156},"A10",7,0.2015,1176.76,[170,204],600,125,0.4959,{"gpu_model":248,"gpu_model_id":249,"gpu_count":43,"total_vram_gb":213,"hour_price":250,"monthly_cost":251,"providers":252,"bandwidth_gbps":253,"fp16_tflops":254,"interconnect_score":153,"parallel_efficiency":154,"insufficient":207,"vram_shortfall_gb":220,"score":255,"tier":156},"L4",13,0.2022,1180.85,[170],300,121,0.4889,{"gpu_model":257,"gpu_model_id":258,"gpu_count":43,"total_vram_gb":213,"hour_price":259,"monthly_cost":260,"providers":261,"bandwidth_gbps":40,"fp16_tflops":40,"interconnect_score":153,"parallel_efficiency":154,"insufficient":207,"vram_shortfall_gb":220,"score":262,"tier":156},"RTX PRO 5000",26,0.483,2820.72,[170],0.4479,{"gpu_model":264,"gpu_model_id":265,"gpu_count":43,"total_vram_gb":266,"hour_price":267,"monthly_cost":268,"providers":269,"bandwidth_gbps":40,"fp16_tflops":40,"interconnect_score":153,"parallel_efficiency":154,"insufficient":207,"vram_shortfall_gb":270,"score":271,"tier":272},"Tesla V100",17,128,0.0272,158.85,[170],146.36,0.5034,"cheapest",{"gpu_model":274,"gpu_model_id":275,"gpu_count":43,"total_vram_gb":266,"hour_price":214,"monthly_cost":215,"providers":276,"bandwidth_gbps":40,"fp16_tflops":40,"interconnect_score":153,"parallel_efficiency":154,"insufficient":207,"vram_shortfall_gb":270,"score":277,"tier":156},"RTX 4070S Ti",22,[170],0.4984,{"gpu_model":279,"gpu_model_id":280,"gpu_count":43,"total_vram_gb":266,"hour_price":281,"monthly_cost":282,"providers":283,"bandwidth_gbps":284,"fp16_tflops":285,"interconnect_score":153,"parallel_efficiency":154,"insufficient":207,"vram_shortfall_gb":270,"score":286,"tier":156},"RTX 5070 Ti",18,0.0849,495.82,[170],896,88,0.5168,{"gpu_model":288,"gpu_model_id":289,"gpu_count":43,"total_vram_gb":266,"hour_price":290,"monthly_cost":291,"providers":292,"bandwidth_gbps":40,"fp16_tflops":40,"interconnect_score":153,"parallel_efficiency":154,"insufficient":207,"vram_shortfall_gb":270,"score":293,"tier":156},"RTX 4080S",21,0.1192,696.13,[170],0.4922,{"gpu_model":295,"gpu_model_id":296,"gpu_count":43,"total_vram_gb":266,"hour_price":224,"monthly_cost":225,"providers":297,"bandwidth_gbps":298,"fp16_tflops":299,"interconnect_score":153,"parallel_efficiency":154,"insufficient":207,"vram_shortfall_gb":270,"score":300,"tier":156},"RTX 5080",23,[170],960,113,0.5123,{"gpu_model":302,"gpu_model_id":303,"gpu_count":43,"total_vram_gb":266,"hour_price":304,"monthly_cost":305,"providers":306,"bandwidth_gbps":40,"fp16_tflops":40,"interconnect_score":153,"parallel_efficiency":154,"insufficient":207,"vram_shortfall_gb":270,"score":307,"tier":156},"RTX PRO 4500",25,0.3222,1881.65,[170],0.4675,{"gpu_model":309,"gpu_model_id":196,"gpu_count":43,"total_vram_gb":310,"hour_price":311,"monthly_cost":312,"providers":313,"bandwidth_gbps":314,"fp16_tflops":315,"interconnect_score":153,"parallel_efficiency":154,"insufficient":207,"vram_shortfall_gb":316,"score":317,"tier":272},"RTX 5070",96,0.0101,58.98,[170],672,61,178.36,0.5208,{"gpu_model":319,"gpu_model_id":320,"gpu_count":43,"total_vram_gb":71,"hour_price":321,"monthly_cost":322,"providers":323,"bandwidth_gbps":324,"fp16_tflops":320,"interconnect_score":153,"parallel_efficiency":154,"insufficient":207,"vram_shortfall_gb":325,"score":326,"tier":156},"RTX 5060 Ti",24,0.0544,317.7,[170],448,210.36,0.5103,false,"value",[],{"posts_count":40,"benchmarks_count":40}]