[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-deepseek-v4-flash":3},{"id":4,"slug":5,"name":6,"hf_id":7,"organization":8,"architecture":9,"architecture_config":10,"parameters":11,"parameters_label":12,"context_length":13,"license":14,"downloads":15,"likes":16,"status":17,"description":10,"huggingface_url":18,"created_at":19,"seo":20,"category":23,"family":26,"score":29,"variants":30,"quantizations":31,"requirements":58,"gpu_recommendations":143,"fitting_gpus":321,"community":322},108,"deepseek-v4-flash","DeepSeek V4 Flash","deepseek-ai\u002FDeepSeek-V4-Flash","deepseek-ai","moe_transformer",null,284000000000,"284B",1048576,"MIT",2283943,2082,"published","https:\u002F\u002Fhuggingface.co\u002Fdeepseek-ai\u002FDeepSeek-V4-Flash","2026-08-12T03:11:11+00:00",{"title":6,"description":21,"image":10,"robots":22,"canonical_url":10},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":24,"slug":25},"LLM","llm",{"name":27,"slug":28},"DeepSeek","deepseek",0.0266,[],[32,37,42,47,51,55],{"format":33,"bits":34,"size_factor":35,"quality_loss":36},"AWQ",4,0.263,0.03,{"format":38,"bits":39,"size_factor":40,"quality_loss":41},"FP16",16,1,0,{"format":43,"bits":44,"size_factor":45,"quality_loss":46},"FP8",8,0.5,0.01,{"format":48,"bits":34,"size_factor":49,"quality_loss":50},"GGUF",0.25,0.05,{"format":52,"bits":34,"size_factor":53,"quality_loss":54},"GPTQ",0.275,0.04,{"format":56,"bits":44,"size_factor":45,"quality_loss":57},"INT8",0.02,{"FP16":59,"FP8":83,"INT8":97,"AWQ":101,"GPTQ":115,"GGUF":129},{"minimum":60,"production":69,"recommended":76},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":63,"kv_cache_gb":45,"vram_gb":64,"ram_gb":65,"disk_gb":66,"requirement_source":67,"confidence":68},"minimum",4096,528.99,674.12,1011.19,864.23,"estimated",0.95,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":63,"kv_cache_gb":72,"vram_gb":73,"ram_gb":74,"disk_gb":75,"requirement_source":67,"confidence":68},"production",16384,64,847.61,1271.41,929.23,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":63,"kv_cache_gb":44,"vram_gb":80,"ram_gb":81,"disk_gb":82,"requirement_source":67,"confidence":68},"recommended",8192,2,720.29,1080.44,890.23,{"minimum":84,"production":89,"recommended":93},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":85,"kv_cache_gb":45,"vram_gb":86,"ram_gb":87,"disk_gb":88,"requirement_source":67,"confidence":68},264.5,339.54,509.31,451.61,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":85,"kv_cache_gb":72,"vram_gb":90,"ram_gb":91,"disk_gb":92,"requirement_source":67,"confidence":68},482.6,723.91,516.61,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":85,"kv_cache_gb":44,"vram_gb":94,"ram_gb":95,"disk_gb":96,"requirement_source":67,"confidence":68},370.5,555.74,477.61,{"minimum":98,"production":99,"recommended":100},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":85,"kv_cache_gb":45,"vram_gb":86,"ram_gb":87,"disk_gb":88,"requirement_source":67,"confidence":68},{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":85,"kv_cache_gb":72,"vram_gb":90,"ram_gb":91,"disk_gb":92,"requirement_source":67,"confidence":68},{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":85,"kv_cache_gb":44,"vram_gb":94,"ram_gb":95,"disk_gb":96,"requirement_source":67,"confidence":68},{"minimum":102,"production":107,"recommended":111},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":103,"kv_cache_gb":45,"vram_gb":104,"ram_gb":105,"disk_gb":106,"requirement_source":67,"confidence":68},137.21,178.52,267.78,253.04,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":103,"kv_cache_gb":72,"vram_gb":108,"ram_gb":109,"disk_gb":110,"requirement_source":67,"confidence":68},306.95,460.42,318.04,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":103,"kv_cache_gb":44,"vram_gb":112,"ram_gb":113,"disk_gb":114,"requirement_source":67,"confidence":68},202.16,303.23,279.04,{"minimum":116,"production":121,"recommended":125},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":117,"kv_cache_gb":45,"vram_gb":118,"ram_gb":119,"disk_gb":120,"requirement_source":67,"confidence":68},138.86,180.61,270.91,255.62,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":117,"kv_cache_gb":72,"vram_gb":122,"ram_gb":123,"disk_gb":124,"requirement_source":67,"confidence":68},309.23,463.84,320.62,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":117,"kv_cache_gb":44,"vram_gb":126,"ram_gb":127,"disk_gb":128,"requirement_source":67,"confidence":68},204.34,306.51,281.62,{"minimum":130,"production":135,"recommended":139},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":131,"kv_cache_gb":45,"vram_gb":132,"ram_gb":133,"disk_gb":134,"requirement_source":67,"confidence":68},142.17,184.79,277.19,260.78,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":131,"kv_cache_gb":72,"vram_gb":136,"ram_gb":137,"disk_gb":138,"requirement_source":67,"confidence":68},313.79,470.68,325.78,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":131,"kv_cache_gb":44,"vram_gb":140,"ram_gb":141,"disk_gb":142,"requirement_source":67,"confidence":68},208.72,313.07,286.78,{"recommended":144,"cheapest":164,"performance":179,"min_complexity":189,"cluster_options":199,"no_match":319,"required_vram_gb":112,"candidate_count":44,"usage":320},{"gpu_model":145,"gpu_model_id":146,"gpu_count":44,"total_vram_gb":147,"hour_price":148,"monthly_cost":149,"providers":150,"bandwidth_gbps":152,"fp16_tflops":153,"interconnect_score":154,"parallel_efficiency":155,"score":156,"tier":157,"perf_metric":158,"perf_label":159,"perf_value":160,"workload":10,"cost_metric":161},"H200 SXM",11,1128,3.5,20440,[151],"lambda",4800,989,40,0.4,0.2675,"balanced","tok\u002Fs","111.9 tok\u002Fs",111.9,{"effective_value":162,"effective_unit":158,"capacity_factor":163},78.3,0.7,{"gpu_model":165,"gpu_model_id":166,"gpu_count":44,"total_vram_gb":167,"hour_price":168,"monthly_cost":169,"providers":170,"bandwidth_gbps":172,"fp16_tflops":173,"interconnect_score":154,"parallel_efficiency":155,"score":174,"tier":157,"perf_metric":158,"perf_label":175,"perf_value":176,"workload":10,"cost_metric":177},"RTX A6000",6,384,0.2874,1678.42,[171,151],"vast",768,77,0.483,"17.9 tok\u002Fs",17.9,{"effective_value":178,"effective_unit":158,"capacity_factor":163},12.5,{"gpu_model":145,"gpu_model_id":146,"gpu_count":34,"total_vram_gb":180,"hour_price":148,"monthly_cost":181,"providers":182,"bandwidth_gbps":152,"fp16_tflops":153,"interconnect_score":154,"parallel_efficiency":183,"score":184,"tier":157,"perf_metric":158,"perf_label":185,"perf_value":186,"workload":10,"cost_metric":187},564,10220,[151],0.6,0.365,"84 tok\u002Fs",84,{"effective_value":188,"effective_unit":158,"capacity_factor":163},58.8,{"gpu_model":145,"gpu_model_id":146,"gpu_count":79,"total_vram_gb":190,"hour_price":148,"monthly_cost":191,"providers":192,"bandwidth_gbps":152,"fp16_tflops":153,"interconnect_score":154,"parallel_efficiency":193,"score":194,"tier":157,"perf_metric":158,"perf_label":195,"perf_value":196,"workload":10,"cost_metric":197},282,5110,[151],0.8,0.5871,"56 tok\u002Fs",56,{"effective_value":198,"effective_unit":158,"capacity_factor":163},39.2,[200,213,222,230,239,248,254,264,269,278,285,292,299,310],{"gpu_model":201,"gpu_model_id":202,"gpu_count":44,"total_vram_gb":203,"hour_price":204,"monthly_cost":205,"providers":206,"bandwidth_gbps":208,"fp16_tflops":209,"interconnect_score":154,"parallel_efficiency":155,"insufficient":210,"vram_shortfall_gb":211,"score":212,"tier":157},"RTX 3090",3,192,0.0678,395.95,[171,207],"tensordock",936,71,true,10.16,0.5198,{"gpu_model":214,"gpu_model_id":40,"gpu_count":44,"total_vram_gb":203,"hour_price":215,"monthly_cost":216,"providers":217,"bandwidth_gbps":219,"fp16_tflops":220,"interconnect_score":154,"parallel_efficiency":155,"insufficient":210,"vram_shortfall_gb":211,"score":221,"tier":157},"RTX 4090",0.1344,784.9,[171,207,218,151],"runpod",1008,165,0.5134,{"gpu_model":223,"gpu_model_id":224,"gpu_count":44,"total_vram_gb":203,"hour_price":225,"monthly_cost":226,"providers":227,"bandwidth_gbps":219,"fp16_tflops":228,"interconnect_score":154,"parallel_efficiency":155,"insufficient":210,"vram_shortfall_gb":211,"score":229,"tier":157},"RTX 4090D",5,0.1602,935.57,[171],82.6,0.5102,{"gpu_model":231,"gpu_model_id":232,"gpu_count":44,"total_vram_gb":203,"hour_price":233,"monthly_cost":234,"providers":235,"bandwidth_gbps":236,"fp16_tflops":237,"interconnect_score":154,"parallel_efficiency":155,"insufficient":210,"vram_shortfall_gb":211,"score":238,"tier":157},"A10",7,0.2015,1176.76,[171,218],600,125,0.4959,{"gpu_model":240,"gpu_model_id":241,"gpu_count":44,"total_vram_gb":203,"hour_price":242,"monthly_cost":243,"providers":244,"bandwidth_gbps":245,"fp16_tflops":246,"interconnect_score":154,"parallel_efficiency":155,"insufficient":210,"vram_shortfall_gb":211,"score":247,"tier":157},"L4",13,0.2022,1180.85,[171],300,121,0.4889,{"gpu_model":249,"gpu_model_id":250,"gpu_count":44,"total_vram_gb":203,"hour_price":174,"monthly_cost":251,"providers":252,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":154,"parallel_efficiency":155,"insufficient":210,"vram_shortfall_gb":211,"score":253,"tier":157},"RTX PRO 5000",26,2820.72,[171],0.4479,{"gpu_model":255,"gpu_model_id":256,"gpu_count":44,"total_vram_gb":257,"hour_price":258,"monthly_cost":259,"providers":260,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":154,"parallel_efficiency":155,"insufficient":210,"vram_shortfall_gb":261,"score":262,"tier":263},"Tesla V100",17,128,0.0272,158.85,[171],74.16,0.5034,"cheapest",{"gpu_model":265,"gpu_model_id":266,"gpu_count":44,"total_vram_gb":257,"hour_price":204,"monthly_cost":205,"providers":267,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":154,"parallel_efficiency":155,"insufficient":210,"vram_shortfall_gb":261,"score":268,"tier":157},"RTX 4070S Ti",22,[171],0.4984,{"gpu_model":270,"gpu_model_id":271,"gpu_count":44,"total_vram_gb":257,"hour_price":272,"monthly_cost":273,"providers":274,"bandwidth_gbps":275,"fp16_tflops":276,"interconnect_score":154,"parallel_efficiency":155,"insufficient":210,"vram_shortfall_gb":261,"score":277,"tier":157},"RTX 5070 Ti",18,0.0849,495.82,[171],896,88,0.5168,{"gpu_model":279,"gpu_model_id":280,"gpu_count":44,"total_vram_gb":257,"hour_price":281,"monthly_cost":282,"providers":283,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":154,"parallel_efficiency":155,"insufficient":210,"vram_shortfall_gb":261,"score":284,"tier":157},"RTX 4080S",21,0.1192,696.13,[171],0.4922,{"gpu_model":286,"gpu_model_id":287,"gpu_count":44,"total_vram_gb":257,"hour_price":215,"monthly_cost":216,"providers":288,"bandwidth_gbps":289,"fp16_tflops":290,"interconnect_score":154,"parallel_efficiency":155,"insufficient":210,"vram_shortfall_gb":261,"score":291,"tier":157},"RTX 5080",23,[171],960,113,0.5123,{"gpu_model":293,"gpu_model_id":294,"gpu_count":44,"total_vram_gb":257,"hour_price":295,"monthly_cost":296,"providers":297,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":154,"parallel_efficiency":155,"insufficient":210,"vram_shortfall_gb":261,"score":298,"tier":157},"RTX PRO 4500",25,0.3222,1881.65,[171],0.4675,{"gpu_model":300,"gpu_model_id":301,"gpu_count":44,"total_vram_gb":302,"hour_price":303,"monthly_cost":304,"providers":305,"bandwidth_gbps":306,"fp16_tflops":307,"interconnect_score":154,"parallel_efficiency":155,"insufficient":210,"vram_shortfall_gb":308,"score":309,"tier":263},"RTX 5070",28,96,0.0101,58.98,[171],672,61,106.16,0.5208,{"gpu_model":311,"gpu_model_id":312,"gpu_count":44,"total_vram_gb":72,"hour_price":313,"monthly_cost":314,"providers":315,"bandwidth_gbps":316,"fp16_tflops":312,"interconnect_score":154,"parallel_efficiency":155,"insufficient":210,"vram_shortfall_gb":317,"score":318,"tier":157},"RTX 5060 Ti",24,0.0544,317.7,[171],448,138.16,0.5103,false,"value",[],{"posts_count":41,"benchmarks_count":41}]