[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-qwen3-vl-8b-instruct":3},{"id":4,"slug":5,"name":6,"hf_id":6,"organization":7,"architecture":8,"architecture_config":9,"parameters":14,"parameters_label":15,"context_length":16,"license":17,"downloads":18,"likes":19,"status":20,"description":17,"huggingface_url":21,"created_at":22,"seo":23,"category":17,"family":26,"score":28,"variants":29,"quantizations":30,"requirements":56,"gpu_recommendations":135,"fitting_gpus":200,"community":231},492,"qwen3-vl-8b-instruct","Qwen\u002FQwen3-VL-8B-Instruct","Qwen","qwen3_vl",{"layers":10,"head_dim":11,"kv_heads":12,"hidden_size":13},36,128,8,4096,8252293120,"8.3B",262144,null,4936347,1044,"published","https:\u002F\u002Fhuggingface.co\u002FQwen\u002FQwen3-VL-8B-Instruct","2026-08-13T18:10:01+00:00",{"title":6,"description":24,"image":17,"robots":25,"canonical_url":17},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":7,"slug":27},"qwen",0.0193,[],[31,36,41,45,49,53],{"format":32,"bits":33,"size_factor":34,"quality_loss":35},"AWQ",4,0.263,0.03,{"format":37,"bits":38,"size_factor":39,"quality_loss":40},"FP16",16,1,0,{"format":42,"bits":12,"size_factor":43,"quality_loss":44},"FP8",0.5,0.01,{"format":46,"bits":33,"size_factor":47,"quality_loss":48},"GGUF",0.25,0.05,{"format":50,"bits":33,"size_factor":51,"quality_loss":52},"GPTQ",0.275,0.04,{"format":54,"bits":12,"size_factor":43,"quality_loss":55},"INT8",0.02,{"FP16":57,"FP8":82,"INT8":95,"AWQ":99,"GPTQ":111,"GGUF":123},{"minimum":58,"production":68,"recommended":76},{"scenario":59,"context_length":60,"batch_size":39,"weight_gb":61,"kv_cache_gb":62,"vram_gb":63,"ram_gb":64,"disk_gb":65,"requirement_source":66,"confidence":67},"minimum",2048,15.37,0.28,23.19,34.79,62.98,"inferred",0.95,{"scenario":69,"context_length":70,"batch_size":71,"weight_gb":61,"kv_cache_gb":72,"vram_gb":73,"ram_gb":74,"disk_gb":75,"requirement_source":66,"confidence":67},"production",8192,2,9,45.21,67.82,127.98,{"scenario":77,"context_length":13,"batch_size":39,"weight_gb":61,"kv_cache_gb":78,"vram_gb":79,"ram_gb":80,"disk_gb":81,"requirement_source":66,"confidence":67},"recommended",1.13,26.51,39.76,88.98,{"minimum":83,"production":88,"recommended":92},{"scenario":59,"context_length":60,"batch_size":39,"weight_gb":84,"kv_cache_gb":62,"vram_gb":85,"ram_gb":86,"disk_gb":87,"requirement_source":66,"confidence":67},7.69,13.47,32,50.99,{"scenario":69,"context_length":70,"batch_size":71,"weight_gb":84,"kv_cache_gb":72,"vram_gb":89,"ram_gb":90,"disk_gb":91,"requirement_source":66,"confidence":67},34.61,51.91,115.99,{"scenario":77,"context_length":13,"batch_size":39,"weight_gb":84,"kv_cache_gb":78,"vram_gb":93,"ram_gb":86,"disk_gb":94,"requirement_source":66,"confidence":67},16.35,76.99,{"minimum":96,"production":97,"recommended":98},{"scenario":59,"context_length":60,"batch_size":39,"weight_gb":84,"kv_cache_gb":62,"vram_gb":85,"ram_gb":86,"disk_gb":87,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":71,"weight_gb":84,"kv_cache_gb":72,"vram_gb":89,"ram_gb":90,"disk_gb":91,"requirement_source":66,"confidence":67},{"scenario":77,"context_length":13,"batch_size":39,"weight_gb":84,"kv_cache_gb":78,"vram_gb":93,"ram_gb":86,"disk_gb":94,"requirement_source":66,"confidence":67},{"minimum":100,"production":104,"recommended":108},{"scenario":59,"context_length":60,"batch_size":39,"weight_gb":101,"kv_cache_gb":62,"vram_gb":102,"ram_gb":86,"disk_gb":103,"requirement_source":66,"confidence":67},3.99,8.79,45.22,{"scenario":69,"context_length":70,"batch_size":71,"weight_gb":101,"kv_cache_gb":72,"vram_gb":105,"ram_gb":106,"disk_gb":107,"requirement_source":66,"confidence":67},29.5,44.25,110.22,{"scenario":77,"context_length":13,"batch_size":39,"weight_gb":101,"kv_cache_gb":78,"vram_gb":109,"ram_gb":86,"disk_gb":110,"requirement_source":66,"confidence":67},11.45,71.22,{"minimum":112,"production":116,"recommended":120},{"scenario":59,"context_length":60,"batch_size":39,"weight_gb":113,"kv_cache_gb":62,"vram_gb":114,"ram_gb":86,"disk_gb":115,"requirement_source":66,"confidence":67},4.03,8.85,45.29,{"scenario":69,"context_length":70,"batch_size":71,"weight_gb":113,"kv_cache_gb":72,"vram_gb":117,"ram_gb":118,"disk_gb":119,"requirement_source":66,"confidence":67},29.57,44.35,110.29,{"scenario":77,"context_length":13,"batch_size":39,"weight_gb":113,"kv_cache_gb":78,"vram_gb":121,"ram_gb":86,"disk_gb":122,"requirement_source":66,"confidence":67},11.52,71.29,{"minimum":124,"production":128,"recommended":132},{"scenario":59,"context_length":60,"batch_size":39,"weight_gb":125,"kv_cache_gb":62,"vram_gb":126,"ram_gb":86,"disk_gb":127,"requirement_source":66,"confidence":67},4.13,8.97,45.44,{"scenario":69,"context_length":70,"batch_size":71,"weight_gb":125,"kv_cache_gb":72,"vram_gb":129,"ram_gb":130,"disk_gb":131,"requirement_source":66,"confidence":67},29.7,44.55,110.44,{"scenario":77,"context_length":13,"batch_size":39,"weight_gb":125,"kv_cache_gb":78,"vram_gb":133,"ram_gb":86,"disk_gb":134,"requirement_source":66,"confidence":67},11.64,71.44,{"recommended":136,"cheapest":154,"performance":163,"min_complexity":182,"cluster_options":196,"no_match":197,"required_vram_gb":109,"candidate_count":198,"usage":199},{"gpu_model":137,"gpu_model_id":138,"gpu_count":39,"total_vram_gb":38,"hour_price":139,"monthly_cost":140,"providers":141,"bandwidth_gbps":143,"fp16_tflops":144,"interconnect_score":145,"parallel_efficiency":39,"score":146,"tier":147,"perf_metric":148,"perf_label":149,"perf_value":150,"workload":17,"cost_metric":151},"RTX 5080",23,0.1339,97.75,[142],"vast",960,113,40,0.6857,"cheapest","tok\u002Fs","240.6 tok\u002Fs",240.6,{"effective_value":152,"effective_unit":148,"capacity_factor":153},168.4,0.7,{"gpu_model":155,"gpu_model_id":156,"gpu_count":39,"total_vram_gb":38,"hour_price":157,"monthly_cost":158,"providers":159,"bandwidth_gbps":40,"fp16_tflops":40,"interconnect_score":145,"parallel_efficiency":39,"score":160,"tier":147,"perf_metric":148,"perf_label":161,"perf_value":40,"workload":17,"cost_metric":162},"Tesla V100",17,0.0289,21.1,[142],0.6804,"0 tok\u002Fs",{"effective_value":40,"effective_unit":148,"capacity_factor":153},{"gpu_model":164,"gpu_model_id":165,"gpu_count":12,"total_vram_gb":166,"hour_price":167,"monthly_cost":168,"providers":169,"bandwidth_gbps":173,"fp16_tflops":174,"interconnect_score":145,"parallel_efficiency":175,"score":176,"tier":177,"perf_metric":148,"perf_label":178,"perf_value":179,"workload":17,"cost_metric":180},"H100 SXM",10,640,1.6139,9425.18,[142,170,171,172],"tensordock","lambda","runpod",3350,66.9,0.4,0.2402,"balanced","2686.7 tok\u002Fs",2686.7,{"effective_value":181,"effective_unit":148,"capacity_factor":153},1880.7,{"gpu_model":183,"gpu_model_id":184,"gpu_count":39,"total_vram_gb":185,"hour_price":186,"monthly_cost":187,"providers":188,"bandwidth_gbps":189,"fp16_tflops":190,"interconnect_score":145,"parallel_efficiency":39,"score":191,"tier":147,"perf_metric":148,"perf_label":192,"perf_value":193,"workload":17,"cost_metric":194},"RTX 3090",3,24,0.0622,45.41,[142,170],936,71,0.623,"234.6 tok\u002Fs",234.6,{"effective_value":195,"effective_unit":148,"capacity_factor":153},164.2,[],false,22,"value",[201,205,209,216,223],{"name":155,"vram_gb":38,"cheapest_hour":157,"cost":202,"provider":142,"estimated_tps":40},{"hourly":157,"daily":203,"monthly":158,"yearly":204},0.69,253.16,{"name":183,"vram_gb":185,"cheapest_hour":186,"cost":206,"provider":142,"estimated_tps":193},{"hourly":186,"daily":207,"monthly":187,"yearly":208},1.49,544.87,{"name":210,"vram_gb":38,"cheapest_hour":211,"cost":212,"provider":142,"estimated_tps":40},"RTX 4070S Ti",0.0678,{"hourly":211,"daily":213,"monthly":214,"yearly":215},1.63,49.49,593.93,{"name":217,"vram_gb":38,"cheapest_hour":218,"cost":219,"provider":142,"estimated_tps":40},"RTX 4080S",0.0685,{"hourly":218,"daily":220,"monthly":221,"yearly":222},1.64,50.01,600.06,{"name":224,"vram_gb":225,"cheapest_hour":226,"cost":227,"provider":142,"estimated_tps":152},"RTX 5070",12,0.0804,{"hourly":226,"daily":228,"monthly":229,"yearly":230},1.93,58.69,704.3,{"posts_count":40,"benchmarks_count":40}]