[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-tiny-random-llamaforcausallm":3},{"id":4,"slug":5,"name":6,"hf_id":6,"organization":7,"architecture":8,"architecture_config":9,"parameters":14,"parameters_label":15,"context_length":16,"license":17,"downloads":18,"likes":19,"status":20,"description":17,"huggingface_url":21,"created_at":22,"seo":23,"category":26,"family":29,"score":31,"variants":32,"quantizations":33,"requirements":57,"gpu_recommendations":97,"fitting_gpus":152,"community":182},503,"tiny-random-llamaforcausallm","hmellor\u002Ftiny-random-LlamaForCausalLM","hmellor","llama",{"layers":10,"head_dim":11,"kv_heads":12,"hidden_size":13},2,64,4,16,1062992,"1.1M",8192,null,4171785,1,"published","https:\u002F\u002Fhuggingface.co\u002Fhmellor\u002Ftiny-random-LlamaForCausalLM","2026-08-13T18:12:32+00:00",{"title":6,"description":24,"image":17,"robots":25,"canonical_url":17},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":27,"slug":28},"LLM","llm",{"name":30,"slug":8},"Llama",0.0066,[],[34,38,41,46,50,54],{"format":35,"bits":12,"size_factor":36,"quality_loss":37},"AWQ",0.263,0.03,{"format":39,"bits":13,"size_factor":19,"quality_loss":40},"FP16",0,{"format":42,"bits":43,"size_factor":44,"quality_loss":45},"FP8",8,0.5,0.01,{"format":47,"bits":12,"size_factor":48,"quality_loss":49},"GGUF",0.25,0.05,{"format":51,"bits":12,"size_factor":52,"quality_loss":53},"GPTQ",0.275,0.04,{"format":55,"bits":43,"size_factor":44,"quality_loss":56},"INT8",0.02,{"FP16":58,"FP8":77,"INT8":81,"AWQ":85,"GPTQ":89,"GGUF":93},{"minimum":59,"production":67,"recommended":72},{"scenario":60,"context_length":61,"batch_size":19,"weight_gb":40,"kv_cache_gb":45,"vram_gb":62,"ram_gb":63,"disk_gb":64,"requirement_source":65,"confidence":66},"minimum",4096,2.21,32,39,"inferred",0.95,{"scenario":68,"context_length":69,"batch_size":12,"weight_gb":40,"kv_cache_gb":19,"vram_gb":70,"ram_gb":63,"disk_gb":71,"requirement_source":65,"confidence":66},"production",16384,3.61,104,{"scenario":73,"context_length":16,"batch_size":10,"weight_gb":40,"kv_cache_gb":74,"vram_gb":75,"ram_gb":63,"disk_gb":76,"requirement_source":65,"confidence":66},"recommended",0.13,2.45,65,{"minimum":78,"production":79,"recommended":80},{"scenario":60,"context_length":61,"batch_size":19,"weight_gb":40,"kv_cache_gb":45,"vram_gb":62,"ram_gb":63,"disk_gb":64,"requirement_source":65,"confidence":66},{"scenario":68,"context_length":69,"batch_size":12,"weight_gb":40,"kv_cache_gb":19,"vram_gb":70,"ram_gb":63,"disk_gb":71,"requirement_source":65,"confidence":66},{"scenario":73,"context_length":16,"batch_size":10,"weight_gb":40,"kv_cache_gb":74,"vram_gb":75,"ram_gb":63,"disk_gb":76,"requirement_source":65,"confidence":66},{"minimum":82,"production":83,"recommended":84},{"scenario":60,"context_length":61,"batch_size":19,"weight_gb":40,"kv_cache_gb":45,"vram_gb":62,"ram_gb":63,"disk_gb":64,"requirement_source":65,"confidence":66},{"scenario":68,"context_length":69,"batch_size":12,"weight_gb":40,"kv_cache_gb":19,"vram_gb":70,"ram_gb":63,"disk_gb":71,"requirement_source":65,"confidence":66},{"scenario":73,"context_length":16,"batch_size":10,"weight_gb":40,"kv_cache_gb":74,"vram_gb":75,"ram_gb":63,"disk_gb":76,"requirement_source":65,"confidence":66},{"minimum":86,"production":87,"recommended":88},{"scenario":60,"context_length":61,"batch_size":19,"weight_gb":40,"kv_cache_gb":45,"vram_gb":62,"ram_gb":63,"disk_gb":64,"requirement_source":65,"confidence":66},{"scenario":68,"context_length":69,"batch_size":12,"weight_gb":40,"kv_cache_gb":19,"vram_gb":70,"ram_gb":63,"disk_gb":71,"requirement_source":65,"confidence":66},{"scenario":73,"context_length":16,"batch_size":10,"weight_gb":40,"kv_cache_gb":74,"vram_gb":75,"ram_gb":63,"disk_gb":76,"requirement_source":65,"confidence":66},{"minimum":90,"production":91,"recommended":92},{"scenario":60,"context_length":61,"batch_size":19,"weight_gb":40,"kv_cache_gb":45,"vram_gb":62,"ram_gb":63,"disk_gb":64,"requirement_source":65,"confidence":66},{"scenario":68,"context_length":69,"batch_size":12,"weight_gb":40,"kv_cache_gb":19,"vram_gb":70,"ram_gb":63,"disk_gb":71,"requirement_source":65,"confidence":66},{"scenario":73,"context_length":16,"batch_size":10,"weight_gb":40,"kv_cache_gb":74,"vram_gb":75,"ram_gb":63,"disk_gb":76,"requirement_source":65,"confidence":66},{"minimum":94,"production":95,"recommended":96},{"scenario":60,"context_length":61,"batch_size":19,"weight_gb":40,"kv_cache_gb":45,"vram_gb":62,"ram_gb":63,"disk_gb":64,"requirement_source":65,"confidence":66},{"scenario":68,"context_length":69,"batch_size":12,"weight_gb":40,"kv_cache_gb":19,"vram_gb":70,"ram_gb":63,"disk_gb":71,"requirement_source":65,"confidence":66},{"scenario":73,"context_length":16,"batch_size":10,"weight_gb":40,"kv_cache_gb":74,"vram_gb":75,"ram_gb":63,"disk_gb":76,"requirement_source":65,"confidence":66},{"recommended":98,"cheapest":114,"performance":122,"min_complexity":136,"cluster_options":148,"no_match":149,"required_vram_gb":75,"candidate_count":150,"usage":151},{"gpu_model":99,"gpu_model_id":100,"gpu_count":19,"total_vram_gb":13,"hour_price":101,"monthly_cost":102,"providers":103,"bandwidth_gbps":105,"fp16_tflops":106,"interconnect_score":107,"parallel_efficiency":19,"score":108,"tier":109,"perf_metric":110,"perf_label":111,"perf_value":40,"workload":17,"cost_metric":112},"RTX 5080",23,0.1339,97.75,[104],"vast",960,113,40,0.5357,"cheapest","tok\u002Fs","0 tok\u002Fs",{"effective_value":40,"effective_unit":110,"capacity_factor":113},0.7,{"gpu_model":115,"gpu_model_id":116,"gpu_count":19,"total_vram_gb":13,"hour_price":117,"monthly_cost":118,"providers":119,"bandwidth_gbps":40,"fp16_tflops":40,"interconnect_score":107,"parallel_efficiency":19,"score":120,"tier":109,"perf_metric":110,"perf_label":111,"perf_value":40,"workload":17,"cost_metric":121},"Tesla V100",17,0.0289,21.1,[104],0.5304,{"effective_value":40,"effective_unit":110,"capacity_factor":113},{"gpu_model":123,"gpu_model_id":124,"gpu_count":43,"total_vram_gb":125,"hour_price":126,"monthly_cost":127,"providers":128,"bandwidth_gbps":130,"fp16_tflops":131,"interconnect_score":107,"parallel_efficiency":132,"score":133,"tier":134,"perf_metric":110,"perf_label":111,"perf_value":40,"workload":17,"cost_metric":135},"H200 SXM",11,1128,3.5,20440,[129],"lambda",4800,989,0.4,0.2197,"balanced",{"effective_value":40,"effective_unit":110,"capacity_factor":113},{"gpu_model":137,"gpu_model_id":138,"gpu_count":19,"total_vram_gb":139,"hour_price":140,"monthly_cost":141,"providers":142,"bandwidth_gbps":144,"fp16_tflops":145,"interconnect_score":107,"parallel_efficiency":19,"score":146,"tier":109,"perf_metric":110,"perf_label":111,"perf_value":40,"workload":17,"cost_metric":147},"RTX 3090",3,24,0.0622,45.41,[104,143],"tensordock",936,71,0.4957,{"effective_value":40,"effective_unit":110,"capacity_factor":113},[],false,26,"value",[153,157,161,168,175],{"name":115,"vram_gb":13,"cheapest_hour":117,"cost":154,"provider":104,"estimated_tps":40},{"hourly":117,"daily":155,"monthly":118,"yearly":156},0.69,253.16,{"name":137,"vram_gb":139,"cheapest_hour":140,"cost":158,"provider":104,"estimated_tps":40},{"hourly":140,"daily":159,"monthly":141,"yearly":160},1.49,544.87,{"name":162,"vram_gb":13,"cheapest_hour":163,"cost":164,"provider":104,"estimated_tps":40},"RTX 4070S Ti",0.0678,{"hourly":163,"daily":165,"monthly":166,"yearly":167},1.63,49.49,593.93,{"name":169,"vram_gb":13,"cheapest_hour":170,"cost":171,"provider":104,"estimated_tps":40},"RTX 4080S",0.0685,{"hourly":170,"daily":172,"monthly":173,"yearly":174},1.64,50.01,600.06,{"name":176,"vram_gb":43,"cheapest_hour":177,"cost":178,"provider":104,"estimated_tps":40},"RTX 5060 Ti",0.0719,{"hourly":177,"daily":179,"monthly":180,"yearly":181},1.73,52.49,629.84,{"posts_count":40,"benchmarks_count":40}]