[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-deepseek-v31":3},{"id":4,"slug":5,"name":6,"hf_id":7,"organization":8,"architecture":9,"architecture_config":10,"parameters":11,"parameters_label":12,"context_length":13,"license":14,"downloads":15,"likes":16,"status":17,"description":10,"huggingface_url":18,"created_at":19,"seo":20,"category":23,"family":26,"score":29,"variants":30,"quantizations":31,"requirements":58,"gpu_recommendations":143,"fitting_gpus":358,"community":359},111,"deepseek-v31","DeepSeek V3.1","deepseek-ai\u002FDeepSeek-V3.1","deepseek-ai","moe_transformer",null,671000000000,"671B",131072,"MIT",224432,825,"published","https:\u002F\u002Fhuggingface.co\u002Fdeepseek-ai\u002FDeepSeek-V3.1","2026-08-12T03:11:12+00:00",{"title":6,"description":21,"image":10,"robots":22,"canonical_url":10},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":24,"slug":25},"LLM","llm",{"name":27,"slug":28},"DeepSeek","deepseek",0.0095,[],[32,37,42,47,51,55],{"format":33,"bits":34,"size_factor":35,"quality_loss":36},"AWQ",4,0.263,0.03,{"format":38,"bits":39,"size_factor":40,"quality_loss":41},"FP16",16,1,0,{"format":43,"bits":44,"size_factor":45,"quality_loss":46},"FP8",8,0.5,0.01,{"format":48,"bits":34,"size_factor":49,"quality_loss":50},"GGUF",0.25,0.05,{"format":52,"bits":34,"size_factor":53,"quality_loss":54},"GPTQ",0.275,0.04,{"format":56,"bits":44,"size_factor":45,"quality_loss":57},"INT8",0.02,{"FP16":59,"FP8":83,"INT8":97,"AWQ":101,"GPTQ":115,"GGUF":129},{"minimum":60,"production":69,"recommended":76},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":63,"kv_cache_gb":45,"vram_gb":64,"ram_gb":65,"disk_gb":66,"requirement_source":67,"confidence":68},"minimum",4096,1249.83,1585.99,2378.99,1988.74,"estimated",0.95,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":63,"kv_cache_gb":72,"vram_gb":73,"ram_gb":74,"disk_gb":75,"requirement_source":67,"confidence":68},"production",16384,64,1842.37,2763.56,2053.74,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":63,"kv_cache_gb":44,"vram_gb":80,"ram_gb":81,"disk_gb":82,"requirement_source":67,"confidence":68},"recommended",8192,2,1673.61,2510.41,2014.74,{"minimum":84,"production":89,"recommended":93},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":85,"kv_cache_gb":45,"vram_gb":86,"ram_gb":87,"disk_gb":88,"requirement_source":67,"confidence":68},624.92,795.47,1193.21,1013.87,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":85,"kv_cache_gb":72,"vram_gb":90,"ram_gb":91,"disk_gb":92,"requirement_source":67,"confidence":68},979.99,1469.98,1078.87,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":85,"kv_cache_gb":44,"vram_gb":94,"ram_gb":95,"disk_gb":96,"requirement_source":67,"confidence":68},847.15,1270.73,1039.87,{"minimum":98,"production":99,"recommended":100},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":85,"kv_cache_gb":45,"vram_gb":86,"ram_gb":87,"disk_gb":88,"requirement_source":67,"confidence":68},{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":85,"kv_cache_gb":72,"vram_gb":90,"ram_gb":91,"disk_gb":92,"requirement_source":67,"confidence":68},{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":85,"kv_cache_gb":44,"vram_gb":94,"ram_gb":95,"disk_gb":96,"requirement_source":67,"confidence":68},{"minimum":102,"production":107,"recommended":111},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":103,"kv_cache_gb":45,"vram_gb":104,"ram_gb":105,"disk_gb":106,"requirement_source":67,"confidence":68},324.18,415.03,622.55,544.71,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":103,"kv_cache_gb":72,"vram_gb":108,"ram_gb":109,"disk_gb":110,"requirement_source":67,"confidence":68},564.96,847.44,609.71,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":103,"kv_cache_gb":44,"vram_gb":112,"ram_gb":113,"disk_gb":114,"requirement_source":67,"confidence":68},449.42,674.13,570.71,{"minimum":116,"production":121,"recommended":125},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":117,"kv_cache_gb":45,"vram_gb":118,"ram_gb":119,"disk_gb":120,"requirement_source":67,"confidence":68},328.08,419.97,629.96,550.81,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":117,"kv_cache_gb":72,"vram_gb":122,"ram_gb":123,"disk_gb":124,"requirement_source":67,"confidence":68},570.35,855.53,615.81,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":117,"kv_cache_gb":44,"vram_gb":126,"ram_gb":127,"disk_gb":128,"requirement_source":67,"confidence":68},454.59,681.88,576.81,{"minimum":130,"production":135,"recommended":139},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":131,"kv_cache_gb":45,"vram_gb":132,"ram_gb":133,"disk_gb":134,"requirement_source":67,"confidence":68},335.89,429.85,644.78,562.99,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":131,"kv_cache_gb":72,"vram_gb":136,"ram_gb":137,"disk_gb":138,"requirement_source":67,"confidence":68},581.13,871.7,627.99,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":131,"kv_cache_gb":44,"vram_gb":140,"ram_gb":141,"disk_gb":142,"requirement_source":67,"confidence":68},464.92,697.38,588.99,{"recommended":144,"cheapest":164,"performance":181,"min_complexity":191,"cluster_options":194,"no_match":356,"required_vram_gb":112,"candidate_count":34,"usage":357},{"gpu_model":145,"gpu_model_id":146,"gpu_count":44,"total_vram_gb":147,"hour_price":148,"monthly_cost":149,"providers":150,"bandwidth_gbps":152,"fp16_tflops":153,"interconnect_score":154,"parallel_efficiency":155,"score":156,"tier":157,"perf_metric":158,"perf_label":159,"perf_value":160,"workload":10,"cost_metric":161},"H200 SXM",11,1128,3.5,20440,[151],"lambda",4800,989,40,0.4,0.326,"balanced","tok\u002Fs","47.4 tok\u002Fs",47.4,{"effective_value":162,"effective_unit":158,"capacity_factor":163},33.2,0.7,{"gpu_model":165,"gpu_model_id":166,"gpu_count":44,"total_vram_gb":167,"hour_price":168,"monthly_cost":169,"providers":170,"bandwidth_gbps":174,"fp16_tflops":175,"interconnect_score":154,"parallel_efficiency":155,"score":176,"tier":157,"perf_metric":158,"perf_label":177,"perf_value":178,"workload":10,"cost_metric":179},"A100 80GB",9,640,1.2,7008,[171,172,173,151],"vast","tensordock","runpod",2039,77.97,0.4479,"20.1 tok\u002Fs",20.1,{"effective_value":180,"effective_unit":158,"capacity_factor":163},14.1,{"gpu_model":145,"gpu_model_id":146,"gpu_count":34,"total_vram_gb":182,"hour_price":148,"monthly_cost":183,"providers":184,"bandwidth_gbps":152,"fp16_tflops":153,"interconnect_score":154,"parallel_efficiency":185,"score":186,"tier":157,"perf_metric":158,"perf_label":187,"perf_value":188,"workload":10,"cost_metric":189},564,10220,[151],0.6,0.4569,"35.5 tok\u002Fs",35.5,{"effective_value":190,"effective_unit":158,"capacity_factor":163},24.8,{"gpu_model":145,"gpu_model_id":146,"gpu_count":34,"total_vram_gb":182,"hour_price":148,"monthly_cost":183,"providers":192,"bandwidth_gbps":152,"fp16_tflops":153,"interconnect_score":154,"parallel_efficiency":185,"score":186,"tier":157,"perf_metric":158,"perf_label":187,"perf_value":188,"workload":10,"cost_metric":193},[151],{"effective_value":190,"effective_unit":158,"capacity_factor":163},[195,207,216,223,230,240,251,259,267,276,285,291,301,306,315,322,329,336,347],{"gpu_model":196,"gpu_model_id":197,"gpu_count":44,"total_vram_gb":198,"hour_price":199,"monthly_cost":200,"providers":201,"bandwidth_gbps":202,"fp16_tflops":203,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":205,"score":206,"tier":157},"RTX A6000",6,384,0.2874,1678.42,[171,151],768,77,true,65.42,0.4893,{"gpu_model":208,"gpu_model_id":209,"gpu_count":44,"total_vram_gb":198,"hour_price":210,"monthly_cost":211,"providers":212,"bandwidth_gbps":213,"fp16_tflops":214,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":205,"score":215,"tier":157},"L40S",12,0.4004,2338.34,[171,172,173],864,362,0.4777,{"gpu_model":217,"gpu_model_id":218,"gpu_count":44,"total_vram_gb":198,"hour_price":219,"monthly_cost":220,"providers":221,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":205,"score":222,"tier":157},"RTX PRO 6000 WS",19,0.6689,3906.38,[171],0.4253,{"gpu_model":224,"gpu_model_id":225,"gpu_count":44,"total_vram_gb":198,"hour_price":226,"monthly_cost":227,"providers":228,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":205,"score":229,"tier":157},"RTX PRO 6000 S",20,0.7339,4285.98,[171],0.4174,{"gpu_model":231,"gpu_model_id":79,"gpu_count":44,"total_vram_gb":232,"hour_price":233,"monthly_cost":234,"providers":235,"bandwidth_gbps":236,"fp16_tflops":237,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":238,"score":239,"tier":157},"RTX 5090",256,0.337,1968.08,[171,173],1792,209,193.42,0.5066,{"gpu_model":241,"gpu_model_id":242,"gpu_count":44,"total_vram_gb":243,"hour_price":244,"monthly_cost":245,"providers":246,"bandwidth_gbps":247,"fp16_tflops":248,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":249,"score":250,"tier":157},"RTX 3090",3,192,0.0678,395.95,[171,172],936,71,257.42,0.5198,{"gpu_model":252,"gpu_model_id":40,"gpu_count":44,"total_vram_gb":243,"hour_price":253,"monthly_cost":254,"providers":255,"bandwidth_gbps":256,"fp16_tflops":257,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":249,"score":258,"tier":157},"RTX 4090",0.1344,784.9,[171,172,173,151],1008,165,0.5134,{"gpu_model":260,"gpu_model_id":261,"gpu_count":44,"total_vram_gb":243,"hour_price":262,"monthly_cost":263,"providers":264,"bandwidth_gbps":256,"fp16_tflops":265,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":249,"score":266,"tier":157},"RTX 4090D",5,0.1602,935.57,[171],82.6,0.5102,{"gpu_model":268,"gpu_model_id":269,"gpu_count":44,"total_vram_gb":243,"hour_price":270,"monthly_cost":271,"providers":272,"bandwidth_gbps":273,"fp16_tflops":274,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":249,"score":275,"tier":157},"A10",7,0.2015,1176.76,[171,173],600,125,0.4959,{"gpu_model":277,"gpu_model_id":278,"gpu_count":44,"total_vram_gb":243,"hour_price":279,"monthly_cost":280,"providers":281,"bandwidth_gbps":282,"fp16_tflops":283,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":249,"score":284,"tier":157},"L4",13,0.2022,1180.85,[171],300,121,0.4889,{"gpu_model":286,"gpu_model_id":287,"gpu_count":44,"total_vram_gb":243,"hour_price":288,"monthly_cost":289,"providers":290,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":249,"score":176,"tier":157},"RTX PRO 5000",26,0.483,2820.72,[171],{"gpu_model":292,"gpu_model_id":293,"gpu_count":44,"total_vram_gb":294,"hour_price":295,"monthly_cost":296,"providers":297,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":298,"score":299,"tier":300},"Tesla V100",17,128,0.0272,158.85,[171],321.42,0.5034,"cheapest",{"gpu_model":302,"gpu_model_id":303,"gpu_count":44,"total_vram_gb":294,"hour_price":244,"monthly_cost":245,"providers":304,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":298,"score":305,"tier":157},"RTX 4070S Ti",22,[171],0.4984,{"gpu_model":307,"gpu_model_id":308,"gpu_count":44,"total_vram_gb":294,"hour_price":309,"monthly_cost":310,"providers":311,"bandwidth_gbps":312,"fp16_tflops":313,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":298,"score":314,"tier":157},"RTX 5070 Ti",18,0.0849,495.82,[171],896,88,0.5168,{"gpu_model":316,"gpu_model_id":317,"gpu_count":44,"total_vram_gb":294,"hour_price":318,"monthly_cost":319,"providers":320,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":298,"score":321,"tier":157},"RTX 4080S",21,0.1192,696.13,[171],0.4922,{"gpu_model":323,"gpu_model_id":324,"gpu_count":44,"total_vram_gb":294,"hour_price":253,"monthly_cost":254,"providers":325,"bandwidth_gbps":326,"fp16_tflops":327,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":298,"score":328,"tier":157},"RTX 5080",23,[171],960,113,0.5123,{"gpu_model":330,"gpu_model_id":331,"gpu_count":44,"total_vram_gb":294,"hour_price":332,"monthly_cost":333,"providers":334,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":298,"score":335,"tier":157},"RTX PRO 4500",25,0.3222,1881.65,[171],0.4675,{"gpu_model":337,"gpu_model_id":338,"gpu_count":44,"total_vram_gb":339,"hour_price":340,"monthly_cost":341,"providers":342,"bandwidth_gbps":343,"fp16_tflops":344,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":345,"score":346,"tier":300},"RTX 5070",28,96,0.0101,58.98,[171],672,61,353.42,0.5208,{"gpu_model":348,"gpu_model_id":349,"gpu_count":44,"total_vram_gb":72,"hour_price":350,"monthly_cost":351,"providers":352,"bandwidth_gbps":353,"fp16_tflops":349,"interconnect_score":154,"parallel_efficiency":155,"insufficient":204,"vram_shortfall_gb":354,"score":355,"tier":157},"RTX 5060 Ti",24,0.0544,317.7,[171],448,385.42,0.5103,false,"value",[],{"posts_count":41,"benchmarks_count":41}]