[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-gemma-4-e4b-it":3},{"id":4,"slug":5,"name":6,"hf_id":6,"organization":7,"architecture":8,"architecture_config":9,"parameters":14,"parameters_label":15,"context_length":16,"license":17,"downloads":18,"likes":19,"status":20,"description":17,"huggingface_url":21,"created_at":22,"seo":23,"category":17,"family":26,"score":29,"variants":30,"quantizations":31,"requirements":58,"gpu_recommendations":134,"fitting_gpus":198,"community":234},488,"gemma-4-e4b-it","google\u002Fgemma-4-E4B-it","google","gemma4",{"layers":10,"head_dim":11,"kv_heads":12,"hidden_size":13},42,256,2,2560,4087349248,"4.1B",131072,null,5195709,1474,"published","https:\u002F\u002Fhuggingface.co\u002Fgoogle\u002Fgemma-4-E4B-it","2026-08-13T18:09:26+00:00",{"title":6,"description":24,"image":17,"robots":25,"canonical_url":17},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":27,"slug":28},"Gemma","gemma",0.0244,[],[32,37,42,47,51,55],{"format":33,"bits":34,"size_factor":35,"quality_loss":36},"AWQ",4,0.263,0.03,{"format":38,"bits":39,"size_factor":40,"quality_loss":41},"FP16",16,1,0,{"format":43,"bits":44,"size_factor":45,"quality_loss":46},"FP8",8,0.5,0.01,{"format":48,"bits":34,"size_factor":49,"quality_loss":50},"GGUF",0.25,0.05,{"format":52,"bits":34,"size_factor":53,"quality_loss":54},"GPTQ",0.275,0.04,{"format":56,"bits":44,"size_factor":45,"quality_loss":57},"INT8",0.02,{"FP16":59,"FP8":83,"INT8":95,"AWQ":99,"GPTQ":111,"GGUF":122},{"minimum":60,"production":70,"recommended":76},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":63,"kv_cache_gb":64,"vram_gb":65,"ram_gb":66,"disk_gb":67,"requirement_source":68,"confidence":69},"minimum",4096,7.61,0.33,14,32,50.88,"inferred",0.95,{"scenario":71,"context_length":72,"batch_size":34,"weight_gb":63,"kv_cache_gb":10,"vram_gb":73,"ram_gb":74,"disk_gb":75,"requirement_source":68,"confidence":69},"production",16384,94.81,142.21,115.88,{"scenario":77,"context_length":78,"batch_size":12,"weight_gb":63,"kv_cache_gb":79,"vram_gb":80,"ram_gb":81,"disk_gb":82,"requirement_source":68,"confidence":69},"recommended",8192,5.25,25.95,38.93,76.88,{"minimum":84,"production":88,"recommended":92},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":85,"kv_cache_gb":64,"vram_gb":86,"ram_gb":66,"disk_gb":87,"requirement_source":68,"confidence":69},3.81,9.18,44.94,{"scenario":71,"context_length":72,"batch_size":34,"weight_gb":85,"kv_cache_gb":10,"vram_gb":89,"ram_gb":90,"disk_gb":91,"requirement_source":68,"confidence":69},89.55,134.33,109.94,{"scenario":77,"context_length":78,"batch_size":12,"weight_gb":85,"kv_cache_gb":79,"vram_gb":93,"ram_gb":66,"disk_gb":94,"requirement_source":68,"confidence":69},20.92,70.94,{"minimum":96,"production":97,"recommended":98},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":85,"kv_cache_gb":64,"vram_gb":86,"ram_gb":66,"disk_gb":87,"requirement_source":68,"confidence":69},{"scenario":71,"context_length":72,"batch_size":34,"weight_gb":85,"kv_cache_gb":10,"vram_gb":89,"ram_gb":90,"disk_gb":91,"requirement_source":68,"confidence":69},{"scenario":77,"context_length":78,"batch_size":12,"weight_gb":85,"kv_cache_gb":79,"vram_gb":93,"ram_gb":66,"disk_gb":94,"requirement_source":68,"confidence":69},{"minimum":100,"production":104,"recommended":108},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":101,"kv_cache_gb":64,"vram_gb":102,"ram_gb":66,"disk_gb":103,"requirement_source":68,"confidence":69},1.97,6.86,42.08,{"scenario":71,"context_length":72,"batch_size":34,"weight_gb":101,"kv_cache_gb":10,"vram_gb":105,"ram_gb":106,"disk_gb":107,"requirement_source":68,"confidence":69},87.03,130.54,107.08,{"scenario":77,"context_length":78,"batch_size":12,"weight_gb":101,"kv_cache_gb":79,"vram_gb":109,"ram_gb":66,"disk_gb":110,"requirement_source":68,"confidence":69},18.5,68.08,{"minimum":112,"production":115,"recommended":119},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":12,"kv_cache_gb":64,"vram_gb":113,"ram_gb":66,"disk_gb":114,"requirement_source":68,"confidence":69},6.89,42.12,{"scenario":71,"context_length":72,"batch_size":34,"weight_gb":12,"kv_cache_gb":10,"vram_gb":116,"ram_gb":117,"disk_gb":118,"requirement_source":68,"confidence":69},87.06,130.59,107.12,{"scenario":77,"context_length":78,"batch_size":12,"weight_gb":12,"kv_cache_gb":79,"vram_gb":120,"ram_gb":66,"disk_gb":121,"requirement_source":68,"confidence":69},18.53,68.12,{"minimum":123,"production":127,"recommended":131},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":124,"kv_cache_gb":64,"vram_gb":125,"ram_gb":66,"disk_gb":126,"requirement_source":68,"confidence":69},2.05,6.95,42.19,{"scenario":71,"context_length":72,"batch_size":34,"weight_gb":124,"kv_cache_gb":10,"vram_gb":128,"ram_gb":129,"disk_gb":130,"requirement_source":68,"confidence":69},87.12,130.69,107.19,{"scenario":77,"context_length":78,"batch_size":12,"weight_gb":124,"kv_cache_gb":79,"vram_gb":132,"ram_gb":66,"disk_gb":133,"requirement_source":68,"confidence":69},18.59,68.19,{"recommended":135,"cheapest":153,"performance":162,"min_complexity":179,"cluster_options":194,"no_match":195,"required_vram_gb":109,"candidate_count":196,"usage":197},{"gpu_model":136,"gpu_model_id":12,"gpu_count":40,"total_vram_gb":66,"hour_price":137,"monthly_cost":138,"providers":139,"bandwidth_gbps":142,"fp16_tflops":143,"interconnect_score":144,"parallel_efficiency":40,"score":145,"tier":146,"perf_metric":147,"perf_label":148,"perf_value":149,"workload":17,"cost_metric":150},"RTX 5090",0.3337,243.6,[140,141],"vast","runpod",1792,209,40,0.6519,"cheapest","tok\u002Fs","909.6 tok\u002Fs",909.6,{"effective_value":151,"effective_unit":147,"capacity_factor":152},636.7,0.7,{"gpu_model":154,"gpu_model_id":155,"gpu_count":12,"total_vram_gb":66,"hour_price":156,"monthly_cost":126,"providers":157,"bandwidth_gbps":41,"fp16_tflops":41,"interconnect_score":144,"parallel_efficiency":158,"score":159,"tier":146,"perf_metric":147,"perf_label":160,"perf_value":41,"workload":17,"cost_metric":161},"Tesla V100",17,0.0289,[140],0.8,0.6008,"0 tok\u002Fs",{"effective_value":41,"effective_unit":147,"capacity_factor":152},{"gpu_model":163,"gpu_model_id":164,"gpu_count":44,"total_vram_gb":165,"hour_price":166,"monthly_cost":167,"providers":168,"bandwidth_gbps":170,"fp16_tflops":171,"interconnect_score":144,"parallel_efficiency":172,"score":173,"tier":174,"perf_metric":147,"perf_label":175,"perf_value":176,"workload":17,"cost_metric":177},"H200 SXM",11,1128,3.5,20440,[169],"lambda",4800,989,0.4,0.2197,"balanced","7797 tok\u002Fs",7797,{"effective_value":178,"effective_unit":147,"capacity_factor":152},5457.9,{"gpu_model":180,"gpu_model_id":181,"gpu_count":40,"total_vram_gb":182,"hour_price":183,"monthly_cost":184,"providers":185,"bandwidth_gbps":187,"fp16_tflops":188,"interconnect_score":144,"parallel_efficiency":40,"score":189,"tier":146,"perf_metric":147,"perf_label":190,"perf_value":191,"workload":17,"cost_metric":192},"RTX 3090",3,24,0.0622,45.41,[140,186],"tensordock",936,71,0.6902,"475.1 tok\u002Fs",475.1,{"effective_value":193,"effective_unit":147,"capacity_factor":152},332.6,[],false,25,"value",[199,203,211,218,226],{"name":180,"vram_gb":182,"cheapest_hour":183,"cost":200,"provider":140,"estimated_tps":191},{"hourly":183,"daily":201,"monthly":184,"yearly":202},1.49,544.87,{"name":204,"vram_gb":182,"cheapest_hour":205,"cost":206,"provider":140,"estimated_tps":210},"RTX 4090",0.1356,{"hourly":205,"daily":207,"monthly":208,"yearly":209},3.25,98.99,1187.86,511.7,{"name":212,"vram_gb":182,"cheapest_hour":213,"cost":214,"provider":140,"estimated_tps":210},"RTX 4090D",0.1602,{"hourly":213,"daily":215,"monthly":216,"yearly":217},3.84,116.95,1403.35,{"name":219,"vram_gb":182,"cheapest_hour":220,"cost":221,"provider":140,"estimated_tps":225},"A10",0.2015,{"hourly":220,"daily":222,"monthly":223,"yearly":224},4.84,147.1,1765.14,304.6,{"name":227,"vram_gb":182,"cheapest_hour":228,"cost":229,"provider":140,"estimated_tps":233},"L4",0.2022,{"hourly":228,"daily":230,"monthly":231,"yearly":232},4.85,147.61,1771.27,152.3,{"posts_count":41,"benchmarks_count":41}]