[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-tiny-qwen2forcausallm-2-5":3},{"id":4,"slug":5,"name":6,"hf_id":6,"organization":7,"architecture":8,"architecture_config":9,"parameters":12,"parameters_label":13,"context_length":14,"license":15,"downloads":16,"likes":17,"status":18,"description":15,"huggingface_url":19,"created_at":20,"seo":21,"category":24,"family":27,"score":30,"variants":31,"quantizations":32,"requirements":58,"gpu_recommendations":100,"fitting_gpus":155,"community":185},447,"tiny-qwen2forcausallm-2-5","trl-internal-testing\u002Ftiny-Qwen2ForCausalLM-2.5","trl-internal-testing","qwen2",{"layers":10,"head_dim":10,"kv_heads":10,"hidden_size":11},2,8,1214856,"1.2M",32768,null,15222284,20,"published","https:\u002F\u002Fhuggingface.co\u002Ftrl-internal-testing\u002Ftiny-Qwen2ForCausalLM-2.5","2026-08-13T18:02:10+00:00",{"title":6,"description":22,"image":15,"robots":23,"canonical_url":15},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":25,"slug":26},"LLM","llm",{"name":28,"slug":29},"Qwen","qwen",0.0327,[],[33,38,43,47,51,55],{"format":34,"bits":35,"size_factor":36,"quality_loss":37},"AWQ",4,0.263,0.03,{"format":39,"bits":40,"size_factor":41,"quality_loss":42},"FP16",16,1,0,{"format":44,"bits":11,"size_factor":45,"quality_loss":46},"FP8",0.5,0.01,{"format":48,"bits":35,"size_factor":49,"quality_loss":50},"GGUF",0.25,0.05,{"format":52,"bits":35,"size_factor":53,"quality_loss":54},"GPTQ",0.275,0.04,{"format":56,"bits":11,"size_factor":45,"quality_loss":57},"INT8",0.02,{"FP16":59,"FP8":78,"INT8":84,"AWQ":88,"GPTQ":92,"GGUF":96},{"minimum":60,"production":68,"recommended":73},{"scenario":61,"context_length":62,"batch_size":41,"weight_gb":42,"kv_cache_gb":42,"vram_gb":63,"ram_gb":64,"disk_gb":65,"requirement_source":66,"confidence":67},"minimum",4096,2.2,32,39,"inferred",0.95,{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":42,"kv_cache_gb":57,"vram_gb":71,"ram_gb":64,"disk_gb":72,"requirement_source":66,"confidence":67},"production",16384,2.43,104,{"scenario":74,"context_length":75,"batch_size":10,"weight_gb":42,"kv_cache_gb":42,"vram_gb":76,"ram_gb":64,"disk_gb":77,"requirement_source":66,"confidence":67},"recommended",8192,2.31,65,{"minimum":79,"production":80,"recommended":82},{"scenario":61,"context_length":62,"batch_size":41,"weight_gb":42,"kv_cache_gb":42,"vram_gb":63,"ram_gb":64,"disk_gb":65,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":42,"kv_cache_gb":57,"vram_gb":81,"ram_gb":64,"disk_gb":72,"requirement_source":66,"confidence":67},2.42,{"scenario":74,"context_length":75,"batch_size":10,"weight_gb":42,"kv_cache_gb":42,"vram_gb":83,"ram_gb":64,"disk_gb":77,"requirement_source":66,"confidence":67},2.3,{"minimum":85,"production":86,"recommended":87},{"scenario":61,"context_length":62,"batch_size":41,"weight_gb":42,"kv_cache_gb":42,"vram_gb":63,"ram_gb":64,"disk_gb":65,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":42,"kv_cache_gb":57,"vram_gb":81,"ram_gb":64,"disk_gb":72,"requirement_source":66,"confidence":67},{"scenario":74,"context_length":75,"batch_size":10,"weight_gb":42,"kv_cache_gb":42,"vram_gb":83,"ram_gb":64,"disk_gb":77,"requirement_source":66,"confidence":67},{"minimum":89,"production":90,"recommended":91},{"scenario":61,"context_length":62,"batch_size":41,"weight_gb":42,"kv_cache_gb":42,"vram_gb":63,"ram_gb":64,"disk_gb":65,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":42,"kv_cache_gb":57,"vram_gb":81,"ram_gb":64,"disk_gb":72,"requirement_source":66,"confidence":67},{"scenario":74,"context_length":75,"batch_size":10,"weight_gb":42,"kv_cache_gb":42,"vram_gb":83,"ram_gb":64,"disk_gb":77,"requirement_source":66,"confidence":67},{"minimum":93,"production":94,"recommended":95},{"scenario":61,"context_length":62,"batch_size":41,"weight_gb":42,"kv_cache_gb":42,"vram_gb":63,"ram_gb":64,"disk_gb":65,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":42,"kv_cache_gb":57,"vram_gb":81,"ram_gb":64,"disk_gb":72,"requirement_source":66,"confidence":67},{"scenario":74,"context_length":75,"batch_size":10,"weight_gb":42,"kv_cache_gb":42,"vram_gb":83,"ram_gb":64,"disk_gb":77,"requirement_source":66,"confidence":67},{"minimum":97,"production":98,"recommended":99},{"scenario":61,"context_length":62,"batch_size":41,"weight_gb":42,"kv_cache_gb":42,"vram_gb":63,"ram_gb":64,"disk_gb":65,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":42,"kv_cache_gb":57,"vram_gb":81,"ram_gb":64,"disk_gb":72,"requirement_source":66,"confidence":67},{"scenario":74,"context_length":75,"batch_size":10,"weight_gb":42,"kv_cache_gb":42,"vram_gb":83,"ram_gb":64,"disk_gb":77,"requirement_source":66,"confidence":67},{"recommended":101,"cheapest":119,"performance":127,"min_complexity":141,"cluster_options":151,"no_match":152,"required_vram_gb":83,"candidate_count":153,"usage":154},{"gpu_model":102,"gpu_model_id":103,"gpu_count":41,"total_vram_gb":104,"hour_price":105,"monthly_cost":106,"providers":107,"bandwidth_gbps":110,"fp16_tflops":111,"interconnect_score":112,"parallel_efficiency":41,"score":113,"tier":114,"perf_metric":115,"perf_label":116,"perf_value":42,"workload":15,"cost_metric":117},"RTX 3090",3,24,0.0622,45.41,[108,109],"vast","tensordock",936,71,40,0.4957,"cheapest","tok\u002Fs","0 tok\u002Fs",{"effective_value":42,"effective_unit":115,"capacity_factor":118},0.7,{"gpu_model":120,"gpu_model_id":121,"gpu_count":41,"total_vram_gb":40,"hour_price":122,"monthly_cost":123,"providers":124,"bandwidth_gbps":42,"fp16_tflops":42,"interconnect_score":112,"parallel_efficiency":41,"score":125,"tier":114,"perf_metric":115,"perf_label":116,"perf_value":42,"workload":15,"cost_metric":126},"Tesla V100",17,0.0289,21.1,[108],0.4896,{"effective_value":42,"effective_unit":115,"capacity_factor":118},{"gpu_model":128,"gpu_model_id":129,"gpu_count":11,"total_vram_gb":130,"hour_price":131,"monthly_cost":132,"providers":133,"bandwidth_gbps":135,"fp16_tflops":136,"interconnect_score":112,"parallel_efficiency":137,"score":138,"tier":139,"perf_metric":115,"perf_label":116,"perf_value":42,"workload":15,"cost_metric":140},"H200 SXM",11,1128,3.5,20440,[134],"lambda",4800,989,0.4,0.2197,"balanced",{"effective_value":42,"effective_unit":115,"capacity_factor":118},{"gpu_model":142,"gpu_model_id":143,"gpu_count":41,"total_vram_gb":40,"hour_price":144,"monthly_cost":145,"providers":146,"bandwidth_gbps":147,"fp16_tflops":148,"interconnect_score":112,"parallel_efficiency":41,"score":149,"tier":114,"perf_metric":115,"perf_label":116,"perf_value":42,"workload":15,"cost_metric":150},"RTX 5080",23,0.1339,97.75,[108],960,113,0.4948,{"effective_value":42,"effective_unit":115,"capacity_factor":118},[],false,26,"value",[156,160,164,171,178],{"name":120,"vram_gb":40,"cheapest_hour":122,"cost":157,"provider":108,"estimated_tps":42},{"hourly":122,"daily":158,"monthly":123,"yearly":159},0.69,253.16,{"name":102,"vram_gb":104,"cheapest_hour":105,"cost":161,"provider":108,"estimated_tps":42},{"hourly":105,"daily":162,"monthly":106,"yearly":163},1.49,544.87,{"name":165,"vram_gb":40,"cheapest_hour":166,"cost":167,"provider":108,"estimated_tps":42},"RTX 4070S Ti",0.0678,{"hourly":166,"daily":168,"monthly":169,"yearly":170},1.63,49.49,593.93,{"name":172,"vram_gb":40,"cheapest_hour":173,"cost":174,"provider":108,"estimated_tps":42},"RTX 4080S",0.0685,{"hourly":173,"daily":175,"monthly":176,"yearly":177},1.64,50.01,600.06,{"name":179,"vram_gb":11,"cheapest_hour":180,"cost":181,"provider":108,"estimated_tps":42},"RTX 5060 Ti",0.0719,{"hourly":180,"daily":182,"monthly":183,"yearly":184},1.73,52.49,629.84,{"posts_count":42,"benchmarks_count":42}]