[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-gemma-4-e2b":3},{"id":4,"slug":5,"name":6,"hf_id":7,"organization":8,"architecture":9,"architecture_config":10,"parameters":11,"parameters_label":12,"context_length":13,"license":14,"downloads":15,"likes":16,"status":17,"description":10,"huggingface_url":18,"created_at":19,"seo":20,"category":23,"family":26,"score":29,"variants":30,"quantizations":31,"requirements":58,"gpu_recommendations":139,"fitting_gpus":206,"community":242},151,"gemma-4-e2b","Gemma 4 E2B","google\u002Fgemma-4-e2b-it","google","transformer",null,5100000000,"5.1B",131072,"MIT",3780950,883,"published","https:\u002F\u002Fhuggingface.co\u002Fgoogle\u002Fgemma-4-e2b-it","2026-08-12T03:11:14+00:00",{"title":6,"description":21,"image":10,"robots":22,"canonical_url":10},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":24,"slug":25},"LLM","llm",{"name":27,"slug":28},"Gemma","gemma",0.0723,[],[32,37,42,47,51,55],{"format":33,"bits":34,"size_factor":35,"quality_loss":36},"AWQ",4,0.263,0.03,{"format":38,"bits":39,"size_factor":40,"quality_loss":41},"FP16",16,1,0,{"format":43,"bits":44,"size_factor":45,"quality_loss":46},"FP8",8,0.5,0.01,{"format":48,"bits":34,"size_factor":49,"quality_loss":50},"GGUF",0.25,0.05,{"format":52,"bits":34,"size_factor":53,"quality_loss":54},"GPTQ",0.275,0.04,{"format":56,"bits":44,"size_factor":45,"quality_loss":57},"INT8",0.02,{"FP16":59,"FP8":83,"INT8":96,"AWQ":100,"GPTQ":113,"GGUF":126},{"minimum":60,"production":69,"recommended":76},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":63,"kv_cache_gb":45,"vram_gb":64,"ram_gb":65,"disk_gb":66,"requirement_source":67,"confidence":68},"minimum",4096,9.5,16.97,32,53.82,"estimated",0.95,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":63,"kv_cache_gb":72,"vram_gb":73,"ram_gb":74,"disk_gb":75,"requirement_source":67,"confidence":68},"production",16384,64,130.71,196.06,118.82,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":63,"kv_cache_gb":44,"vram_gb":80,"ram_gb":81,"disk_gb":82,"requirement_source":67,"confidence":68},"recommended",8192,2,33.26,49.89,79.82,{"minimum":84,"production":88,"recommended":92},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":85,"kv_cache_gb":45,"vram_gb":86,"ram_gb":65,"disk_gb":87,"requirement_source":67,"confidence":68},4.75,10.96,46.41,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":85,"kv_cache_gb":72,"vram_gb":89,"ram_gb":90,"disk_gb":91,"requirement_source":67,"confidence":68},124.15,186.23,111.41,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":85,"kv_cache_gb":44,"vram_gb":93,"ram_gb":94,"disk_gb":95,"requirement_source":67,"confidence":68},26.98,40.47,72.41,{"minimum":97,"production":98,"recommended":99},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":85,"kv_cache_gb":45,"vram_gb":86,"ram_gb":65,"disk_gb":87,"requirement_source":67,"confidence":68},{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":85,"kv_cache_gb":72,"vram_gb":89,"ram_gb":90,"disk_gb":91,"requirement_source":67,"confidence":68},{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":85,"kv_cache_gb":44,"vram_gb":93,"ram_gb":94,"disk_gb":95,"requirement_source":67,"confidence":68},{"minimum":101,"production":105,"recommended":109},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":102,"kv_cache_gb":45,"vram_gb":103,"ram_gb":65,"disk_gb":104,"requirement_source":67,"confidence":68},2.46,8.07,42.84,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":102,"kv_cache_gb":72,"vram_gb":106,"ram_gb":107,"disk_gb":108,"requirement_source":67,"confidence":68},121,181.5,107.84,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":102,"kv_cache_gb":44,"vram_gb":110,"ram_gb":111,"disk_gb":112,"requirement_source":67,"confidence":68},23.96,35.94,68.84,{"minimum":114,"production":118,"recommended":122},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":115,"kv_cache_gb":45,"vram_gb":116,"ram_gb":65,"disk_gb":117,"requirement_source":67,"confidence":68},2.49,8.1,42.89,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":115,"kv_cache_gb":72,"vram_gb":119,"ram_gb":120,"disk_gb":121,"requirement_source":67,"confidence":68},121.04,181.56,107.89,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":115,"kv_cache_gb":44,"vram_gb":123,"ram_gb":124,"disk_gb":125,"requirement_source":67,"confidence":68},24,36,68.89,{"minimum":127,"production":131,"recommended":135},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":128,"kv_cache_gb":45,"vram_gb":129,"ram_gb":65,"disk_gb":130,"requirement_source":67,"confidence":68},2.55,8.18,42.98,{"scenario":70,"context_length":71,"batch_size":34,"weight_gb":128,"kv_cache_gb":72,"vram_gb":132,"ram_gb":133,"disk_gb":134,"requirement_source":67,"confidence":68},121.12,181.68,107.98,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":128,"kv_cache_gb":44,"vram_gb":136,"ram_gb":137,"disk_gb":138,"requirement_source":67,"confidence":68},24.08,36.11,68.98,{"recommended":140,"cheapest":158,"performance":172,"min_complexity":189,"cluster_options":203,"no_match":204,"required_vram_gb":110,"candidate_count":123,"usage":205},{"gpu_model":141,"gpu_model_id":79,"gpu_count":40,"total_vram_gb":65,"hour_price":142,"monthly_cost":143,"providers":144,"bandwidth_gbps":147,"fp16_tflops":148,"interconnect_score":149,"parallel_efficiency":40,"score":150,"tier":151,"perf_metric":152,"perf_label":153,"perf_value":154,"workload":10,"cost_metric":155},"RTX 5090",0.337,246.01,[145,146],"vast","runpod",1792,209,40,0.6973,"cheapest","tok\u002Fs","728.5 tok\u002Fs",728.5,{"effective_value":156,"effective_unit":152,"capacity_factor":157},510,0.7,{"gpu_model":159,"gpu_model_id":160,"gpu_count":79,"total_vram_gb":123,"hour_price":161,"monthly_cost":162,"providers":163,"bandwidth_gbps":164,"fp16_tflops":165,"interconnect_score":149,"parallel_efficiency":166,"score":167,"tier":151,"perf_metric":152,"perf_label":168,"perf_value":169,"workload":10,"cost_metric":170},"RTX 5070",28,0.0101,14.75,[145],672,61,0.8,0.6015,"437.1 tok\u002Fs",437.1,{"effective_value":171,"effective_unit":152,"capacity_factor":157},306,{"gpu_model":173,"gpu_model_id":174,"gpu_count":44,"total_vram_gb":175,"hour_price":176,"monthly_cost":177,"providers":178,"bandwidth_gbps":180,"fp16_tflops":181,"interconnect_score":149,"parallel_efficiency":182,"score":183,"tier":184,"perf_metric":152,"perf_label":185,"perf_value":186,"workload":10,"cost_metric":187},"H200 SXM",11,1128,3.5,20440,[179],"lambda",4800,989,0.4,0.2197,"balanced","6243.9 tok\u002Fs",6243.9,{"effective_value":188,"effective_unit":152,"capacity_factor":157},4370.7,{"gpu_model":190,"gpu_model_id":191,"gpu_count":40,"total_vram_gb":123,"hour_price":192,"monthly_cost":193,"providers":194,"bandwidth_gbps":196,"fp16_tflops":197,"interconnect_score":149,"parallel_efficiency":40,"score":198,"tier":151,"perf_metric":152,"perf_label":199,"perf_value":200,"workload":10,"cost_metric":201},"RTX 3090",3,0.0678,49.49,[145,195],"tensordock",936,71,0.6423,"380.5 tok\u002Fs",380.5,{"effective_value":202,"effective_unit":152,"capacity_factor":157},266.3,[],false,"value",[207,211,219,226,234],{"name":190,"vram_gb":123,"cheapest_hour":192,"cost":208,"provider":145,"estimated_tps":200},{"hourly":192,"daily":209,"monthly":193,"yearly":210},1.63,593.93,{"name":212,"vram_gb":123,"cheapest_hour":213,"cost":214,"provider":145,"estimated_tps":218},"RTX 4090",0.1344,{"hourly":213,"daily":215,"monthly":216,"yearly":217},3.23,98.11,1177.34,409.8,{"name":220,"vram_gb":123,"cheapest_hour":221,"cost":222,"provider":145,"estimated_tps":218},"RTX 4090D",0.1602,{"hourly":221,"daily":223,"monthly":224,"yearly":225},3.84,116.95,1403.35,{"name":227,"vram_gb":123,"cheapest_hour":228,"cost":229,"provider":145,"estimated_tps":233},"A10",0.2015,{"hourly":228,"daily":230,"monthly":231,"yearly":232},4.84,147.1,1765.14,243.9,{"name":235,"vram_gb":123,"cheapest_hour":236,"cost":237,"provider":145,"estimated_tps":241},"L4",0.2022,{"hourly":236,"daily":238,"monthly":239,"yearly":240},4.85,147.61,1771.27,122,{"posts_count":41,"benchmarks_count":41}]