[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-codellama-7b":3},{"id":4,"slug":5,"name":6,"hf_id":7,"organization":8,"architecture":9,"architecture_config":10,"parameters":11,"parameters_label":12,"context_length":13,"license":14,"downloads":15,"likes":16,"status":17,"description":10,"huggingface_url":18,"created_at":19,"seo":20,"category":23,"family":26,"score":29,"variants":30,"quantizations":31,"requirements":58,"gpu_recommendations":139,"fitting_gpus":201,"community":232},77,"codellama-7b","CodeLlama 7B","codellama\u002FCodeLlama-7b-hf","codellama","transformer",null,7000000000,"7B",16384,"MIT",474809,377,"published","https:\u002F\u002Fhuggingface.co\u002Fcodellama\u002FCodeLlama-7b-hf","2026-08-12T03:11:10+00:00",{"title":6,"description":21,"image":10,"robots":22,"canonical_url":10},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":24,"slug":25},"Coding AI","coding",{"name":27,"slug":28},"Llama","llama",0.0049,[],[32,37,42,47,51,55],{"format":33,"bits":34,"size_factor":35,"quality_loss":36},"AWQ",4,0.263,0.03,{"format":38,"bits":39,"size_factor":40,"quality_loss":41},"FP16",16,1,0,{"format":43,"bits":44,"size_factor":45,"quality_loss":46},"FP8",8,0.5,0.01,{"format":48,"bits":34,"size_factor":49,"quality_loss":50},"GGUF",0.25,0.05,{"format":52,"bits":34,"size_factor":53,"quality_loss":54},"GPTQ",0.275,0.04,{"format":56,"bits":44,"size_factor":45,"quality_loss":57},"INT8",0.02,{"FP16":59,"FP8":82,"INT8":96,"AWQ":100,"GPTQ":113,"GGUF":126},{"minimum":60,"production":69,"recommended":75},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":63,"kv_cache_gb":45,"vram_gb":64,"ram_gb":65,"disk_gb":66,"requirement_source":67,"confidence":68},"minimum",4096,13.04,21.44,32.17,59.34,"estimated",0.95,{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":63,"kv_cache_gb":71,"vram_gb":72,"ram_gb":73,"disk_gb":74,"requirement_source":67,"confidence":68},"production",64,135.59,203.39,124.34,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":63,"kv_cache_gb":44,"vram_gb":79,"ram_gb":80,"disk_gb":81,"requirement_source":67,"confidence":68},"recommended",8192,2,37.94,56.92,85.34,{"minimum":83,"production":88,"recommended":92},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":84,"kv_cache_gb":45,"vram_gb":85,"ram_gb":86,"disk_gb":87,"requirement_source":67,"confidence":68},6.52,13.2,32,49.17,{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":84,"kv_cache_gb":71,"vram_gb":89,"ram_gb":90,"disk_gb":91,"requirement_source":67,"confidence":68},126.6,189.89,114.17,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":84,"kv_cache_gb":44,"vram_gb":93,"ram_gb":94,"disk_gb":95,"requirement_source":67,"confidence":68},29.32,43.98,75.17,{"minimum":97,"production":98,"recommended":99},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":84,"kv_cache_gb":45,"vram_gb":85,"ram_gb":86,"disk_gb":87,"requirement_source":67,"confidence":68},{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":84,"kv_cache_gb":71,"vram_gb":89,"ram_gb":90,"disk_gb":91,"requirement_source":67,"confidence":68},{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":84,"kv_cache_gb":44,"vram_gb":93,"ram_gb":94,"disk_gb":95,"requirement_source":67,"confidence":68},{"minimum":101,"production":105,"recommended":109},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":102,"kv_cache_gb":45,"vram_gb":103,"ram_gb":86,"disk_gb":104,"requirement_source":67,"confidence":68},3.38,9.23,44.28,{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":102,"kv_cache_gb":71,"vram_gb":106,"ram_gb":107,"disk_gb":108,"requirement_source":67,"confidence":68},122.27,183.4,109.28,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":102,"kv_cache_gb":44,"vram_gb":110,"ram_gb":111,"disk_gb":112,"requirement_source":67,"confidence":68},25.17,37.76,70.28,{"minimum":114,"production":118,"recommended":122},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":115,"kv_cache_gb":45,"vram_gb":116,"ram_gb":86,"disk_gb":117,"requirement_source":67,"confidence":68},3.42,9.28,44.34,{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":115,"kv_cache_gb":71,"vram_gb":119,"ram_gb":120,"disk_gb":121,"requirement_source":67,"confidence":68},122.32,183.48,109.34,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":115,"kv_cache_gb":44,"vram_gb":123,"ram_gb":124,"disk_gb":125,"requirement_source":67,"confidence":68},25.23,37.84,70.34,{"minimum":127,"production":131,"recommended":135},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":128,"kv_cache_gb":45,"vram_gb":129,"ram_gb":86,"disk_gb":130,"requirement_source":67,"confidence":68},3.5,9.38,44.47,{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":128,"kv_cache_gb":71,"vram_gb":132,"ram_gb":133,"disk_gb":134,"requirement_source":67,"confidence":68},122.44,183.65,109.47,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":128,"kv_cache_gb":44,"vram_gb":136,"ram_gb":137,"disk_gb":138,"requirement_source":67,"confidence":68},25.33,38,70.47,{"recommended":140,"cheapest":158,"performance":168,"min_complexity":184,"cluster_options":197,"no_match":198,"required_vram_gb":110,"candidate_count":199,"usage":200},{"gpu_model":141,"gpu_model_id":78,"gpu_count":40,"total_vram_gb":86,"hour_price":142,"monthly_cost":143,"providers":144,"bandwidth_gbps":147,"fp16_tflops":148,"interconnect_score":149,"parallel_efficiency":40,"score":150,"tier":151,"perf_metric":152,"perf_label":153,"perf_value":154,"workload":10,"cost_metric":155},"RTX 5090",0.3222,235.21,[145,146],"vast","runpod",1792,209,40,0.6881,"cheapest","tok\u002Fs","530.2 tok\u002Fs",530.2,{"effective_value":156,"effective_unit":152,"capacity_factor":157},371.1,0.7,{"gpu_model":159,"gpu_model_id":160,"gpu_count":78,"total_vram_gb":86,"hour_price":161,"monthly_cost":162,"providers":163,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":149,"parallel_efficiency":164,"score":165,"tier":151,"perf_metric":152,"perf_label":166,"perf_value":41,"workload":10,"cost_metric":167},"Tesla V100",17,0.0272,39.71,[145],0.8,0.6369,"0 tok\u002Fs",{"effective_value":41,"effective_unit":152,"capacity_factor":157},{"gpu_model":169,"gpu_model_id":170,"gpu_count":44,"total_vram_gb":171,"hour_price":128,"monthly_cost":172,"providers":173,"bandwidth_gbps":175,"fp16_tflops":176,"interconnect_score":149,"parallel_efficiency":177,"score":178,"tier":179,"perf_metric":152,"perf_label":180,"perf_value":181,"workload":10,"cost_metric":182},"H200 SXM",11,1128,20440,[174],"lambda",4800,989,0.4,0.2197,"balanced","4544.4 tok\u002Fs",4544.4,{"effective_value":183,"effective_unit":152,"capacity_factor":157},3181.1,{"gpu_model":185,"gpu_model_id":186,"gpu_count":40,"total_vram_gb":187,"hour_price":188,"monthly_cost":189,"providers":190,"bandwidth_gbps":191,"fp16_tflops":4,"interconnect_score":149,"parallel_efficiency":40,"score":192,"tier":151,"perf_metric":152,"perf_label":193,"perf_value":194,"workload":10,"cost_metric":195},"RTX A6000",6,48,0.3001,219.07,[145,174],768,0.6308,"227.2 tok\u002Fs",227.2,{"effective_value":196,"effective_unit":152,"capacity_factor":157},159,[],false,24,"value",[202,206,210,218,225],{"name":185,"vram_gb":187,"cheapest_hour":188,"cost":203,"provider":145,"estimated_tps":194},{"hourly":188,"daily":204,"monthly":189,"yearly":205},7.2,2628.88,{"name":141,"vram_gb":86,"cheapest_hour":142,"cost":207,"provider":145,"estimated_tps":154},{"hourly":142,"daily":208,"monthly":143,"yearly":209},7.73,2822.47,{"name":211,"vram_gb":187,"cheapest_hour":212,"cost":213,"provider":145,"estimated_tps":217},"L40S",0.4669,{"hourly":212,"daily":214,"monthly":215,"yearly":216},11.21,340.84,4090.04,255.6,{"name":219,"vram_gb":187,"cheapest_hour":220,"cost":221,"provider":145,"estimated_tps":41},"RTX PRO 6000 WS",0.8022,{"hourly":220,"daily":222,"monthly":223,"yearly":224},19.25,585.61,7027.27,{"name":226,"vram_gb":187,"cheapest_hour":227,"cost":228,"provider":145,"estimated_tps":41},"RTX PRO 6000 S",1.0131,{"hourly":227,"daily":229,"monthly":230,"yearly":231},24.31,739.56,8874.76,{"posts_count":41,"benchmarks_count":41}]