[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-tiny-qwen2forcausallm-2-5":3},{"id":4,"slug":5,"name":6,"hf_id":6,"organization":7,"architecture":8,"architecture_config":9,"parameters":12,"parameters_label":13,"context_length":14,"license":15,"downloads":16,"likes":17,"status":18,"description":15,"huggingface_url":19,"created_at":20,"seo":21,"category":24,"family":27,"score":30,"variants":31,"quantizations":32,"requirements":58,"gpu_recommendations":100,"fitting_gpus":153,"community":177},447,"tiny-qwen2forcausallm-2-5","trl-internal-testing\u002Ftiny-Qwen2ForCausalLM-2.5","trl-internal-testing","qwen2",{"layers":10,"head_dim":10,"kv_heads":10,"hidden_size":11},2,8,1214856,"1.2M",32768,null,15222284,20,"published","https:\u002F\u002Fhuggingface.co\u002Ftrl-internal-testing\u002Ftiny-Qwen2ForCausalLM-2.5","2026-08-13T18:02:10+00:00",{"title":6,"description":22,"image":15,"robots":23,"canonical_url":15},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":25,"slug":26},"LLM","llm",{"name":28,"slug":29},"Qwen","qwen",0.0327,[],[33,38,43,47,51,55],{"format":34,"bits":35,"size_factor":36,"quality_loss":37},"AWQ",4,0.263,0.03,{"format":39,"bits":40,"size_factor":41,"quality_loss":42},"FP16",16,1,0,{"format":44,"bits":11,"size_factor":45,"quality_loss":46},"FP8",0.5,0.01,{"format":48,"bits":35,"size_factor":49,"quality_loss":50},"GGUF",0.25,0.05,{"format":52,"bits":35,"size_factor":53,"quality_loss":54},"GPTQ",0.275,0.04,{"format":56,"bits":11,"size_factor":45,"quality_loss":57},"INT8",0.02,{"FP16":59,"FP8":78,"INT8":84,"AWQ":88,"GPTQ":92,"GGUF":96},{"minimum":60,"production":68,"recommended":73},{"scenario":61,"context_length":62,"batch_size":41,"weight_gb":42,"kv_cache_gb":42,"vram_gb":63,"ram_gb":64,"disk_gb":65,"requirement_source":66,"confidence":67},"minimum",4096,2.2,32,39,"inferred",0.95,{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":42,"kv_cache_gb":57,"vram_gb":71,"ram_gb":64,"disk_gb":72,"requirement_source":66,"confidence":67},"production",16384,2.43,104,{"scenario":74,"context_length":75,"batch_size":10,"weight_gb":42,"kv_cache_gb":42,"vram_gb":76,"ram_gb":64,"disk_gb":77,"requirement_source":66,"confidence":67},"recommended",8192,2.31,65,{"minimum":79,"production":80,"recommended":82},{"scenario":61,"context_length":62,"batch_size":41,"weight_gb":42,"kv_cache_gb":42,"vram_gb":63,"ram_gb":64,"disk_gb":65,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":42,"kv_cache_gb":57,"vram_gb":81,"ram_gb":64,"disk_gb":72,"requirement_source":66,"confidence":67},2.42,{"scenario":74,"context_length":75,"batch_size":10,"weight_gb":42,"kv_cache_gb":42,"vram_gb":83,"ram_gb":64,"disk_gb":77,"requirement_source":66,"confidence":67},2.3,{"minimum":85,"production":86,"recommended":87},{"scenario":61,"context_length":62,"batch_size":41,"weight_gb":42,"kv_cache_gb":42,"vram_gb":63,"ram_gb":64,"disk_gb":65,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":42,"kv_cache_gb":57,"vram_gb":81,"ram_gb":64,"disk_gb":72,"requirement_source":66,"confidence":67},{"scenario":74,"context_length":75,"batch_size":10,"weight_gb":42,"kv_cache_gb":42,"vram_gb":83,"ram_gb":64,"disk_gb":77,"requirement_source":66,"confidence":67},{"minimum":89,"production":90,"recommended":91},{"scenario":61,"context_length":62,"batch_size":41,"weight_gb":42,"kv_cache_gb":42,"vram_gb":63,"ram_gb":64,"disk_gb":65,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":42,"kv_cache_gb":57,"vram_gb":81,"ram_gb":64,"disk_gb":72,"requirement_source":66,"confidence":67},{"scenario":74,"context_length":75,"batch_size":10,"weight_gb":42,"kv_cache_gb":42,"vram_gb":83,"ram_gb":64,"disk_gb":77,"requirement_source":66,"confidence":67},{"minimum":93,"production":94,"recommended":95},{"scenario":61,"context_length":62,"batch_size":41,"weight_gb":42,"kv_cache_gb":42,"vram_gb":63,"ram_gb":64,"disk_gb":65,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":42,"kv_cache_gb":57,"vram_gb":81,"ram_gb":64,"disk_gb":72,"requirement_source":66,"confidence":67},{"scenario":74,"context_length":75,"batch_size":10,"weight_gb":42,"kv_cache_gb":42,"vram_gb":83,"ram_gb":64,"disk_gb":77,"requirement_source":66,"confidence":67},{"minimum":97,"production":98,"recommended":99},{"scenario":61,"context_length":62,"batch_size":41,"weight_gb":42,"kv_cache_gb":42,"vram_gb":63,"ram_gb":64,"disk_gb":65,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":42,"kv_cache_gb":57,"vram_gb":81,"ram_gb":64,"disk_gb":72,"requirement_source":66,"confidence":67},{"scenario":74,"context_length":75,"batch_size":10,"weight_gb":42,"kv_cache_gb":42,"vram_gb":83,"ram_gb":64,"disk_gb":77,"requirement_source":66,"confidence":67},{"recommended":101,"cheapest":116,"performance":124,"min_complexity":138,"cluster_options":149,"no_match":150,"required_vram_gb":83,"candidate_count":151,"usage":152},{"gpu_model":102,"gpu_model_id":103,"gpu_count":41,"total_vram_gb":11,"hour_price":104,"monthly_cost":105,"providers":106,"bandwidth_gbps":108,"fp16_tflops":103,"interconnect_score":109,"parallel_efficiency":41,"score":110,"tier":111,"perf_metric":112,"perf_label":113,"perf_value":42,"workload":15,"cost_metric":114},"RTX 5060 Ti",24,0.0554,40.44,[107],"vast",448,40,0.569,"cheapest","tok\u002Fs","0 tok\u002Fs",{"effective_value":42,"effective_unit":112,"capacity_factor":115},0.7,{"gpu_model":117,"gpu_model_id":118,"gpu_count":41,"total_vram_gb":40,"hour_price":119,"monthly_cost":120,"providers":121,"bandwidth_gbps":42,"fp16_tflops":42,"interconnect_score":109,"parallel_efficiency":41,"score":122,"tier":111,"perf_metric":112,"perf_label":113,"perf_value":42,"workload":15,"cost_metric":123},"Tesla V100",17,0.0272,19.86,[107],0.4896,{"effective_value":42,"effective_unit":112,"capacity_factor":115},{"gpu_model":125,"gpu_model_id":126,"gpu_count":11,"total_vram_gb":127,"hour_price":128,"monthly_cost":129,"providers":130,"bandwidth_gbps":132,"fp16_tflops":133,"interconnect_score":109,"parallel_efficiency":134,"score":135,"tier":136,"perf_metric":112,"perf_label":113,"perf_value":42,"workload":15,"cost_metric":137},"H200 SXM",11,1128,3.5,20440,[131],"lambda",4800,989,0.4,0.2197,"balanced",{"effective_value":42,"effective_unit":112,"capacity_factor":115},{"gpu_model":139,"gpu_model_id":140,"gpu_count":41,"total_vram_gb":103,"hour_price":141,"monthly_cost":142,"providers":143,"bandwidth_gbps":145,"fp16_tflops":146,"interconnect_score":109,"parallel_efficiency":41,"score":147,"tier":111,"perf_metric":112,"perf_label":113,"perf_value":42,"workload":15,"cost_metric":148},"RTX 3090",3,0.0678,49.49,[107,144],"tensordock",936,71,0.4957,{"effective_value":42,"effective_unit":112,"capacity_factor":115},[],false,27,"value",[154,158,162,166,169],{"name":117,"vram_gb":40,"cheapest_hour":119,"cost":155,"provider":107,"estimated_tps":42},{"hourly":119,"daily":156,"monthly":120,"yearly":157},0.65,238.27,{"name":102,"vram_gb":11,"cheapest_hour":104,"cost":159,"provider":107,"estimated_tps":42},{"hourly":104,"daily":160,"monthly":105,"yearly":161},1.33,485.3,{"name":139,"vram_gb":103,"cheapest_hour":141,"cost":163,"provider":107,"estimated_tps":42},{"hourly":141,"daily":164,"monthly":142,"yearly":165},1.63,593.93,{"name":167,"vram_gb":40,"cheapest_hour":141,"cost":168,"provider":107,"estimated_tps":42},"RTX 4070S Ti",{"hourly":141,"daily":164,"monthly":142,"yearly":165},{"name":170,"vram_gb":171,"cheapest_hour":172,"cost":173,"provider":107,"estimated_tps":42},"RTX 5070",12,0.0804,{"hourly":172,"daily":174,"monthly":175,"yearly":176},1.93,58.69,704.3,{"posts_count":42,"benchmarks_count":42}]