[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-codellama-7b":3},{"id":4,"slug":5,"name":6,"hf_id":7,"organization":8,"architecture":9,"architecture_config":10,"parameters":11,"parameters_label":12,"context_length":13,"license":14,"downloads":15,"likes":16,"status":17,"description":10,"huggingface_url":18,"created_at":19,"seo":20,"category":23,"family":26,"score":29,"variants":30,"quantizations":31,"requirements":58,"gpu_recommendations":139,"fitting_gpus":205,"community":236},77,"codellama-7b","CodeLlama 7B","codellama\u002FCodeLlama-7b-hf","codellama","transformer",null,7000000000,"7B",16384,"MIT",474809,377,"published","https:\u002F\u002Fhuggingface.co\u002Fcodellama\u002FCodeLlama-7b-hf","2026-08-12T03:11:10+00:00",{"title":6,"description":21,"image":10,"robots":22,"canonical_url":10},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":24,"slug":25},"Coding AI","coding",{"name":27,"slug":28},"Llama","llama",0.0049,[],[32,37,42,47,51,55],{"format":33,"bits":34,"size_factor":35,"quality_loss":36},"AWQ",4,0.263,0.03,{"format":38,"bits":39,"size_factor":40,"quality_loss":41},"FP16",16,1,0,{"format":43,"bits":44,"size_factor":45,"quality_loss":46},"FP8",8,0.5,0.01,{"format":48,"bits":34,"size_factor":49,"quality_loss":50},"GGUF",0.25,0.05,{"format":52,"bits":34,"size_factor":53,"quality_loss":54},"GPTQ",0.275,0.04,{"format":56,"bits":44,"size_factor":45,"quality_loss":57},"INT8",0.02,{"FP16":59,"FP8":82,"INT8":96,"AWQ":100,"GPTQ":113,"GGUF":126},{"minimum":60,"production":69,"recommended":75},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":63,"kv_cache_gb":45,"vram_gb":64,"ram_gb":65,"disk_gb":66,"requirement_source":67,"confidence":68},"minimum",4096,13.04,21.44,32.17,59.34,"estimated",0.95,{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":63,"kv_cache_gb":71,"vram_gb":72,"ram_gb":73,"disk_gb":74,"requirement_source":67,"confidence":68},"production",64,135.59,203.39,124.34,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":63,"kv_cache_gb":44,"vram_gb":79,"ram_gb":80,"disk_gb":81,"requirement_source":67,"confidence":68},"recommended",8192,2,37.94,56.92,85.34,{"minimum":83,"production":88,"recommended":92},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":84,"kv_cache_gb":45,"vram_gb":85,"ram_gb":86,"disk_gb":87,"requirement_source":67,"confidence":68},6.52,13.2,32,49.17,{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":84,"kv_cache_gb":71,"vram_gb":89,"ram_gb":90,"disk_gb":91,"requirement_source":67,"confidence":68},126.6,189.89,114.17,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":84,"kv_cache_gb":44,"vram_gb":93,"ram_gb":94,"disk_gb":95,"requirement_source":67,"confidence":68},29.32,43.98,75.17,{"minimum":97,"production":98,"recommended":99},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":84,"kv_cache_gb":45,"vram_gb":85,"ram_gb":86,"disk_gb":87,"requirement_source":67,"confidence":68},{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":84,"kv_cache_gb":71,"vram_gb":89,"ram_gb":90,"disk_gb":91,"requirement_source":67,"confidence":68},{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":84,"kv_cache_gb":44,"vram_gb":93,"ram_gb":94,"disk_gb":95,"requirement_source":67,"confidence":68},{"minimum":101,"production":105,"recommended":109},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":102,"kv_cache_gb":45,"vram_gb":103,"ram_gb":86,"disk_gb":104,"requirement_source":67,"confidence":68},3.38,9.23,44.28,{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":102,"kv_cache_gb":71,"vram_gb":106,"ram_gb":107,"disk_gb":108,"requirement_source":67,"confidence":68},122.27,183.4,109.28,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":102,"kv_cache_gb":44,"vram_gb":110,"ram_gb":111,"disk_gb":112,"requirement_source":67,"confidence":68},25.17,37.76,70.28,{"minimum":114,"production":118,"recommended":122},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":115,"kv_cache_gb":45,"vram_gb":116,"ram_gb":86,"disk_gb":117,"requirement_source":67,"confidence":68},3.42,9.28,44.34,{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":115,"kv_cache_gb":71,"vram_gb":119,"ram_gb":120,"disk_gb":121,"requirement_source":67,"confidence":68},122.32,183.48,109.34,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":115,"kv_cache_gb":44,"vram_gb":123,"ram_gb":124,"disk_gb":125,"requirement_source":67,"confidence":68},25.23,37.84,70.34,{"minimum":127,"production":131,"recommended":135},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":128,"kv_cache_gb":45,"vram_gb":129,"ram_gb":86,"disk_gb":130,"requirement_source":67,"confidence":68},3.5,9.38,44.47,{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":128,"kv_cache_gb":71,"vram_gb":132,"ram_gb":133,"disk_gb":134,"requirement_source":67,"confidence":68},122.44,183.65,109.47,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":128,"kv_cache_gb":44,"vram_gb":136,"ram_gb":137,"disk_gb":138,"requirement_source":67,"confidence":68},25.33,38,70.47,{"recommended":140,"cheapest":158,"performance":173,"min_complexity":189,"cluster_options":201,"no_match":202,"required_vram_gb":110,"candidate_count":203,"usage":204},{"gpu_model":141,"gpu_model_id":78,"gpu_count":40,"total_vram_gb":86,"hour_price":142,"monthly_cost":143,"providers":144,"bandwidth_gbps":147,"fp16_tflops":148,"interconnect_score":149,"parallel_efficiency":40,"score":150,"tier":151,"perf_metric":152,"perf_label":153,"perf_value":154,"workload":10,"cost_metric":155},"RTX 5090",0.337,246.01,[145,146],"vast","runpod",1792,209,40,0.6879,"cheapest","tok\u002Fs","530.2 tok\u002Fs",530.2,{"effective_value":156,"effective_unit":152,"capacity_factor":157},371.1,0.7,{"gpu_model":159,"gpu_model_id":160,"gpu_count":34,"total_vram_gb":161,"hour_price":162,"monthly_cost":163,"providers":164,"bandwidth_gbps":165,"fp16_tflops":166,"interconnect_score":149,"parallel_efficiency":167,"score":168,"tier":151,"perf_metric":152,"perf_label":169,"perf_value":170,"workload":10,"cost_metric":171},"RTX 5070",28,48,0.0101,29.49,[145],672,61,0.6,0.5507,"477.2 tok\u002Fs",477.2,{"effective_value":172,"effective_unit":152,"capacity_factor":157},334,{"gpu_model":174,"gpu_model_id":175,"gpu_count":44,"total_vram_gb":176,"hour_price":128,"monthly_cost":177,"providers":178,"bandwidth_gbps":180,"fp16_tflops":181,"interconnect_score":149,"parallel_efficiency":182,"score":183,"tier":184,"perf_metric":152,"perf_label":185,"perf_value":186,"workload":10,"cost_metric":187},"H200 SXM",11,1128,20440,[179],"lambda",4800,989,0.4,0.2197,"balanced","4544.4 tok\u002Fs",4544.4,{"effective_value":188,"effective_unit":152,"capacity_factor":157},3181.1,{"gpu_model":190,"gpu_model_id":191,"gpu_count":40,"total_vram_gb":161,"hour_price":192,"monthly_cost":193,"providers":194,"bandwidth_gbps":195,"fp16_tflops":4,"interconnect_score":149,"parallel_efficiency":40,"score":196,"tier":151,"perf_metric":152,"perf_label":197,"perf_value":198,"workload":10,"cost_metric":199},"RTX A6000",6,0.2874,209.8,[145,179],768,0.6309,"227.2 tok\u002Fs",227.2,{"effective_value":200,"effective_unit":152,"capacity_factor":157},159,[],false,23,"value",[206,210,214,222,229],{"name":190,"vram_gb":161,"cheapest_hour":192,"cost":207,"provider":145,"estimated_tps":198},{"hourly":192,"daily":208,"monthly":193,"yearly":209},6.9,2517.62,{"name":141,"vram_gb":86,"cheapest_hour":142,"cost":211,"provider":145,"estimated_tps":154},{"hourly":142,"daily":212,"monthly":143,"yearly":213},8.09,2952.12,{"name":215,"vram_gb":161,"cheapest_hour":216,"cost":217,"provider":145,"estimated_tps":221},"L40S",0.4004,{"hourly":216,"daily":218,"monthly":219,"yearly":220},9.61,292.29,3507.5,255.6,{"name":223,"vram_gb":161,"cheapest_hour":224,"cost":225,"provider":145,"estimated_tps":41},"RTX PRO 6000 WS",0.6689,{"hourly":224,"daily":226,"monthly":227,"yearly":228},16.05,488.3,5859.56,{"name":230,"vram_gb":161,"cheapest_hour":231,"cost":232,"provider":145,"estimated_tps":41},"RTX PRO 6000 S",0.7339,{"hourly":231,"daily":233,"monthly":234,"yearly":235},17.61,535.75,6428.96,{"posts_count":41,"benchmarks_count":41}]