[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-codellama-70b":3},{"id":4,"slug":5,"name":6,"hf_id":7,"organization":8,"architecture":9,"architecture_config":10,"parameters":11,"parameters_label":12,"context_length":13,"license":14,"downloads":15,"likes":16,"status":17,"description":10,"huggingface_url":18,"created_at":19,"seo":20,"category":23,"family":26,"score":29,"variants":30,"quantizations":31,"requirements":58,"gpu_recommendations":142,"fitting_gpus":214,"community":242},85,"codellama-70b","CodeLlama 70B","codellama\u002FCodeLlama-70b-hf","codellama","transformer",null,70000000000,"70B",16384,"MIT",553,324,"published","https:\u002F\u002Fhuggingface.co\u002Fcodellama\u002FCodeLlama-70b-hf","2026-08-12T03:11:10+00:00",{"title":6,"description":21,"image":10,"robots":22,"canonical_url":10},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":24,"slug":25},"Coding AI","coding",{"name":27,"slug":28},"Llama","llama",0.0036,[],[32,37,42,47,51,55],{"format":33,"bits":34,"size_factor":35,"quality_loss":36},"AWQ",4,0.263,0.03,{"format":38,"bits":39,"size_factor":40,"quality_loss":41},"FP16",16,1,0,{"format":43,"bits":44,"size_factor":45,"quality_loss":46},"FP8",8,0.5,0.01,{"format":48,"bits":34,"size_factor":49,"quality_loss":50},"GGUF",0.25,0.05,{"format":52,"bits":34,"size_factor":53,"quality_loss":54},"GPTQ",0.275,0.04,{"format":56,"bits":44,"size_factor":45,"quality_loss":57},"INT8",0.02,{"FP16":59,"FP8":82,"INT8":96,"AWQ":100,"GPTQ":114,"GGUF":128},{"minimum":60,"production":69,"recommended":75},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":63,"kv_cache_gb":45,"vram_gb":64,"ram_gb":65,"disk_gb":66,"requirement_source":67,"confidence":68},"minimum",4096,130.39,169.89,254.83,242.4,"estimated",0.95,{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":63,"kv_cache_gb":71,"vram_gb":72,"ram_gb":73,"disk_gb":74,"requirement_source":67,"confidence":68},"production",64,297.53,446.3,307.4,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":63,"kv_cache_gb":44,"vram_gb":79,"ram_gb":80,"disk_gb":81,"requirement_source":67,"confidence":68},"recommended",8192,2,193.13,289.7,268.4,{"minimum":83,"production":88,"recommended":92},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":84,"kv_cache_gb":45,"vram_gb":85,"ram_gb":86,"disk_gb":87,"requirement_source":67,"confidence":68},65.19,87.42,131.13,140.7,{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":84,"kv_cache_gb":71,"vram_gb":89,"ram_gb":90,"disk_gb":91,"requirement_source":67,"confidence":68},207.57,311.35,205.7,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":84,"kv_cache_gb":44,"vram_gb":93,"ram_gb":94,"disk_gb":95,"requirement_source":67,"confidence":68},106.92,160.38,166.7,{"minimum":97,"production":98,"recommended":99},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":84,"kv_cache_gb":45,"vram_gb":85,"ram_gb":86,"disk_gb":87,"requirement_source":67,"confidence":68},{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":84,"kv_cache_gb":71,"vram_gb":89,"ram_gb":90,"disk_gb":91,"requirement_source":67,"confidence":68},{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":84,"kv_cache_gb":44,"vram_gb":93,"ram_gb":94,"disk_gb":95,"requirement_source":67,"confidence":68},{"minimum":101,"production":106,"recommended":110},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":102,"kv_cache_gb":45,"vram_gb":103,"ram_gb":104,"disk_gb":105,"requirement_source":67,"confidence":68},33.82,47.73,71.6,91.76,{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":102,"kv_cache_gb":71,"vram_gb":107,"ram_gb":108,"disk_gb":109,"requirement_source":67,"confidence":68},164.27,246.4,156.76,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":102,"kv_cache_gb":44,"vram_gb":111,"ram_gb":112,"disk_gb":113,"requirement_source":67,"confidence":68},65.43,98.14,117.76,{"minimum":115,"production":120,"recommended":124},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":116,"kv_cache_gb":45,"vram_gb":117,"ram_gb":118,"disk_gb":119,"requirement_source":67,"confidence":68},34.23,48.25,72.37,92.39,{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":116,"kv_cache_gb":71,"vram_gb":121,"ram_gb":122,"disk_gb":123,"requirement_source":67,"confidence":68},164.83,247.25,157.39,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":116,"kv_cache_gb":44,"vram_gb":125,"ram_gb":126,"disk_gb":127,"requirement_source":67,"confidence":68},65.96,98.95,118.39,{"minimum":129,"production":134,"recommended":138},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":130,"kv_cache_gb":45,"vram_gb":131,"ram_gb":132,"disk_gb":133,"requirement_source":67,"confidence":68},35.04,49.28,73.92,93.66,{"scenario":70,"context_length":13,"batch_size":34,"weight_gb":130,"kv_cache_gb":71,"vram_gb":135,"ram_gb":136,"disk_gb":137,"requirement_source":67,"confidence":68},165.96,248.93,158.66,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":130,"kv_cache_gb":44,"vram_gb":139,"ram_gb":140,"disk_gb":141,"requirement_source":67,"confidence":68},67.04,100.56,119.66,{"recommended":143,"cheapest":162,"performance":175,"min_complexity":184,"cluster_options":200,"no_match":211,"required_vram_gb":111,"candidate_count":212,"usage":213},{"gpu_model":144,"gpu_model_id":145,"gpu_count":40,"total_vram_gb":146,"hour_price":147,"monthly_cost":148,"providers":149,"bandwidth_gbps":151,"fp16_tflops":152,"interconnect_score":153,"parallel_efficiency":40,"score":154,"tier":155,"perf_metric":156,"perf_label":157,"perf_value":158,"workload":10,"cost_metric":159},"H200 SXM",11,141,3.5,2555,[150],"lambda",4800,989,40,0.5948,"balanced","tok\u002Fs","141.9 tok\u002Fs",141.9,{"effective_value":160,"effective_unit":156,"capacity_factor":161},99.3,0.7,{"gpu_model":163,"gpu_model_id":164,"gpu_count":44,"total_vram_gb":165,"hour_price":166,"monthly_cost":167,"providers":168,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":153,"parallel_efficiency":170,"score":171,"tier":172,"perf_metric":156,"perf_label":173,"perf_value":41,"workload":10,"cost_metric":174},"Tesla V100",17,128,0.0272,158.85,[169],"vast",0.4,0.493,"cheapest","0 tok\u002Fs",{"effective_value":41,"effective_unit":156,"capacity_factor":161},{"gpu_model":144,"gpu_model_id":145,"gpu_count":44,"total_vram_gb":176,"hour_price":147,"monthly_cost":177,"providers":178,"bandwidth_gbps":151,"fp16_tflops":152,"interconnect_score":153,"parallel_efficiency":170,"score":179,"tier":155,"perf_metric":156,"perf_label":180,"perf_value":181,"workload":10,"cost_metric":182},1128,20440,[150],0.2197,"454.2 tok\u002Fs",454.2,{"effective_value":183,"effective_unit":156,"capacity_factor":161},317.9,{"gpu_model":185,"gpu_model_id":186,"gpu_count":40,"total_vram_gb":187,"hour_price":188,"monthly_cost":189,"providers":190,"bandwidth_gbps":193,"fp16_tflops":194,"interconnect_score":153,"parallel_efficiency":40,"score":195,"tier":155,"perf_metric":156,"perf_label":196,"perf_value":197,"workload":10,"cost_metric":198},"A100 80GB",9,80,1.2,876,[169,191,192,150],"tensordock","runpod",2039,77.97,0.6682,"60.3 tok\u002Fs",60.3,{"effective_value":199,"effective_unit":156,"capacity_factor":161},42.2,[201],{"gpu_model":202,"gpu_model_id":203,"gpu_count":44,"total_vram_gb":71,"hour_price":204,"monthly_cost":205,"providers":206,"bandwidth_gbps":207,"fp16_tflops":203,"interconnect_score":153,"parallel_efficiency":170,"insufficient":208,"vram_shortfall_gb":209,"score":210,"tier":155},"RTX 5060 Ti",24,0.0554,323.54,[169],448,true,1.43,0.5102,false,19,"value",[215,219,227,231,239],{"name":185,"vram_gb":187,"cheapest_hour":188,"cost":216,"provider":169,"estimated_tps":197},{"hourly":188,"daily":217,"monthly":189,"yearly":218},28.8,10512,{"name":220,"vram_gb":187,"cheapest_hour":221,"cost":222,"provider":169,"estimated_tps":226},"H100 SXM",1.6022,{"hourly":221,"daily":223,"monthly":224,"yearly":225},38.45,1169.61,14035.27,99.1,{"name":144,"vram_gb":146,"cheapest_hour":147,"cost":228,"provider":150,"estimated_tps":158},{"hourly":147,"daily":229,"monthly":148,"yearly":230},84,30660,{"name":232,"vram_gb":233,"cheapest_hour":234,"cost":235,"provider":169,"estimated_tps":41},"B200",180,4.1271,{"hourly":234,"daily":236,"monthly":237,"yearly":238},99.05,3012.78,36153.4,{"name":240,"vram_gb":233,"cheapest_hour":10,"cost":10,"provider":10,"estimated_tps":241},"B200 SXM",230.6,{"posts_count":41,"benchmarks_count":41}]