[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-tiny-random-llamaforcausallm":3},{"id":4,"slug":5,"name":6,"hf_id":6,"organization":7,"architecture":8,"architecture_config":9,"parameters":14,"parameters_label":15,"context_length":16,"license":17,"downloads":18,"likes":19,"status":20,"description":17,"huggingface_url":21,"created_at":22,"seo":23,"category":26,"family":29,"score":31,"variants":32,"quantizations":33,"requirements":57,"gpu_recommendations":97,"fitting_gpus":149,"community":176},503,"tiny-random-llamaforcausallm","hmellor\u002Ftiny-random-LlamaForCausalLM","hmellor","llama",{"layers":10,"head_dim":11,"kv_heads":12,"hidden_size":13},2,64,4,16,1062992,"1.1M",8192,null,4171785,1,"published","https:\u002F\u002Fhuggingface.co\u002Fhmellor\u002Ftiny-random-LlamaForCausalLM","2026-08-13T18:12:32+00:00",{"title":6,"description":24,"image":17,"robots":25,"canonical_url":17},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":27,"slug":28},"LLM","llm",{"name":30,"slug":8},"Llama",0.0066,[],[34,38,41,46,50,54],{"format":35,"bits":12,"size_factor":36,"quality_loss":37},"AWQ",0.263,0.03,{"format":39,"bits":13,"size_factor":19,"quality_loss":40},"FP16",0,{"format":42,"bits":43,"size_factor":44,"quality_loss":45},"FP8",8,0.5,0.01,{"format":47,"bits":12,"size_factor":48,"quality_loss":49},"GGUF",0.25,0.05,{"format":51,"bits":12,"size_factor":52,"quality_loss":53},"GPTQ",0.275,0.04,{"format":55,"bits":43,"size_factor":44,"quality_loss":56},"INT8",0.02,{"FP16":58,"FP8":77,"INT8":81,"AWQ":85,"GPTQ":89,"GGUF":93},{"minimum":59,"production":67,"recommended":72},{"scenario":60,"context_length":61,"batch_size":19,"weight_gb":40,"kv_cache_gb":45,"vram_gb":62,"ram_gb":63,"disk_gb":64,"requirement_source":65,"confidence":66},"minimum",4096,2.21,32,39,"inferred",0.95,{"scenario":68,"context_length":69,"batch_size":12,"weight_gb":40,"kv_cache_gb":19,"vram_gb":70,"ram_gb":63,"disk_gb":71,"requirement_source":65,"confidence":66},"production",16384,3.61,104,{"scenario":73,"context_length":16,"batch_size":10,"weight_gb":40,"kv_cache_gb":74,"vram_gb":75,"ram_gb":63,"disk_gb":76,"requirement_source":65,"confidence":66},"recommended",0.13,2.45,65,{"minimum":78,"production":79,"recommended":80},{"scenario":60,"context_length":61,"batch_size":19,"weight_gb":40,"kv_cache_gb":45,"vram_gb":62,"ram_gb":63,"disk_gb":64,"requirement_source":65,"confidence":66},{"scenario":68,"context_length":69,"batch_size":12,"weight_gb":40,"kv_cache_gb":19,"vram_gb":70,"ram_gb":63,"disk_gb":71,"requirement_source":65,"confidence":66},{"scenario":73,"context_length":16,"batch_size":10,"weight_gb":40,"kv_cache_gb":74,"vram_gb":75,"ram_gb":63,"disk_gb":76,"requirement_source":65,"confidence":66},{"minimum":82,"production":83,"recommended":84},{"scenario":60,"context_length":61,"batch_size":19,"weight_gb":40,"kv_cache_gb":45,"vram_gb":62,"ram_gb":63,"disk_gb":64,"requirement_source":65,"confidence":66},{"scenario":68,"context_length":69,"batch_size":12,"weight_gb":40,"kv_cache_gb":19,"vram_gb":70,"ram_gb":63,"disk_gb":71,"requirement_source":65,"confidence":66},{"scenario":73,"context_length":16,"batch_size":10,"weight_gb":40,"kv_cache_gb":74,"vram_gb":75,"ram_gb":63,"disk_gb":76,"requirement_source":65,"confidence":66},{"minimum":86,"production":87,"recommended":88},{"scenario":60,"context_length":61,"batch_size":19,"weight_gb":40,"kv_cache_gb":45,"vram_gb":62,"ram_gb":63,"disk_gb":64,"requirement_source":65,"confidence":66},{"scenario":68,"context_length":69,"batch_size":12,"weight_gb":40,"kv_cache_gb":19,"vram_gb":70,"ram_gb":63,"disk_gb":71,"requirement_source":65,"confidence":66},{"scenario":73,"context_length":16,"batch_size":10,"weight_gb":40,"kv_cache_gb":74,"vram_gb":75,"ram_gb":63,"disk_gb":76,"requirement_source":65,"confidence":66},{"minimum":90,"production":91,"recommended":92},{"scenario":60,"context_length":61,"batch_size":19,"weight_gb":40,"kv_cache_gb":45,"vram_gb":62,"ram_gb":63,"disk_gb":64,"requirement_source":65,"confidence":66},{"scenario":68,"context_length":69,"batch_size":12,"weight_gb":40,"kv_cache_gb":19,"vram_gb":70,"ram_gb":63,"disk_gb":71,"requirement_source":65,"confidence":66},{"scenario":73,"context_length":16,"batch_size":10,"weight_gb":40,"kv_cache_gb":74,"vram_gb":75,"ram_gb":63,"disk_gb":76,"requirement_source":65,"confidence":66},{"minimum":94,"production":95,"recommended":96},{"scenario":60,"context_length":61,"batch_size":19,"weight_gb":40,"kv_cache_gb":45,"vram_gb":62,"ram_gb":63,"disk_gb":64,"requirement_source":65,"confidence":66},{"scenario":68,"context_length":69,"batch_size":12,"weight_gb":40,"kv_cache_gb":19,"vram_gb":70,"ram_gb":63,"disk_gb":71,"requirement_source":65,"confidence":66},{"scenario":73,"context_length":16,"batch_size":10,"weight_gb":40,"kv_cache_gb":74,"vram_gb":75,"ram_gb":63,"disk_gb":76,"requirement_source":65,"confidence":66},{"recommended":98,"cheapest":114,"performance":122,"min_complexity":136,"cluster_options":145,"no_match":146,"required_vram_gb":75,"candidate_count":147,"usage":148},{"gpu_model":99,"gpu_model_id":100,"gpu_count":19,"total_vram_gb":13,"hour_price":101,"monthly_cost":102,"providers":103,"bandwidth_gbps":105,"fp16_tflops":106,"interconnect_score":107,"parallel_efficiency":19,"score":108,"tier":109,"perf_metric":110,"perf_label":111,"perf_value":40,"workload":17,"cost_metric":112},"RTX 5080",23,0.1078,78.69,[104],"vast",960,113,40,0.5361,"cheapest","tok\u002Fs","0 tok\u002Fs",{"effective_value":40,"effective_unit":110,"capacity_factor":113},0.7,{"gpu_model":115,"gpu_model_id":116,"gpu_count":19,"total_vram_gb":13,"hour_price":117,"monthly_cost":118,"providers":119,"bandwidth_gbps":40,"fp16_tflops":40,"interconnect_score":107,"parallel_efficiency":19,"score":120,"tier":109,"perf_metric":110,"perf_label":111,"perf_value":40,"workload":17,"cost_metric":121},"Tesla V100",17,0.0272,19.86,[104],0.5304,{"effective_value":40,"effective_unit":110,"capacity_factor":113},{"gpu_model":123,"gpu_model_id":124,"gpu_count":43,"total_vram_gb":125,"hour_price":126,"monthly_cost":127,"providers":128,"bandwidth_gbps":130,"fp16_tflops":131,"interconnect_score":107,"parallel_efficiency":132,"score":133,"tier":134,"perf_metric":110,"perf_label":111,"perf_value":40,"workload":17,"cost_metric":135},"H200 SXM",11,1128,3.5,20440,[129],"lambda",4800,989,0.4,0.2197,"balanced",{"effective_value":40,"effective_unit":110,"capacity_factor":113},{"gpu_model":137,"gpu_model_id":138,"gpu_count":19,"total_vram_gb":43,"hour_price":139,"monthly_cost":140,"providers":141,"bandwidth_gbps":142,"fp16_tflops":138,"interconnect_score":107,"parallel_efficiency":19,"score":143,"tier":109,"perf_metric":110,"perf_label":111,"perf_value":40,"workload":17,"cost_metric":144},"RTX 5060 Ti",24,0.0554,40.44,[104],448,0.574,{"effective_value":40,"effective_unit":110,"capacity_factor":113},[],false,27,"value",[150,154,158,165,168],{"name":115,"vram_gb":13,"cheapest_hour":117,"cost":151,"provider":104,"estimated_tps":40},{"hourly":117,"daily":152,"monthly":118,"yearly":153},0.65,238.27,{"name":137,"vram_gb":43,"cheapest_hour":139,"cost":155,"provider":104,"estimated_tps":40},{"hourly":139,"daily":156,"monthly":140,"yearly":157},1.33,485.3,{"name":159,"vram_gb":138,"cheapest_hour":160,"cost":161,"provider":104,"estimated_tps":40},"RTX 3090",0.0678,{"hourly":160,"daily":162,"monthly":163,"yearly":164},1.63,49.49,593.93,{"name":166,"vram_gb":13,"cheapest_hour":160,"cost":167,"provider":104,"estimated_tps":40},"RTX 4070S Ti",{"hourly":160,"daily":162,"monthly":163,"yearly":164},{"name":169,"vram_gb":170,"cheapest_hour":171,"cost":172,"provider":104,"estimated_tps":40},"RTX 5070",12,0.0804,{"hourly":171,"daily":173,"monthly":174,"yearly":175},1.93,58.69,704.3,{"posts_count":40,"benchmarks_count":40}]