[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-ms-marco-minilm-l4-v2":3},{"id":4,"slug":5,"name":6,"hf_id":6,"organization":7,"architecture":8,"architecture_config":9,"parameters":13,"parameters_label":14,"context_length":15,"license":16,"downloads":17,"likes":18,"status":19,"description":16,"huggingface_url":20,"created_at":21,"seo":22,"category":16,"family":25,"score":27,"variants":28,"quantizations":29,"requirements":55,"gpu_recommendations":107,"fitting_gpus":171,"community":202},467,"ms-marco-minilm-l4-v2","cross-encoder\u002Fms-marco-MiniLM-L4-v2","cross-encoder","bert",{"layers":10,"head_dim":11,"hidden_size":12},4,32,384,18798336,"18.8M",512,null,8796019,28,"published","https:\u002F\u002Fhuggingface.co\u002Fcross-encoder\u002Fms-marco-MiniLM-L4-v2","2026-08-13T18:05:36+00:00",{"title":6,"description":23,"image":16,"robots":24,"canonical_url":16},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":26,"slug":7},"Cross-encoder",0.0205,[],[30,34,39,44,48,52],{"format":31,"bits":10,"size_factor":32,"quality_loss":33},"AWQ",0.263,0.03,{"format":35,"bits":36,"size_factor":37,"quality_loss":38},"FP16",16,1,0,{"format":40,"bits":41,"size_factor":42,"quality_loss":43},"FP8",8,0.5,0.01,{"format":45,"bits":10,"size_factor":46,"quality_loss":47},"GGUF",0.25,0.05,{"format":49,"bits":10,"size_factor":50,"quality_loss":51},"GPTQ",0.275,0.04,{"format":53,"bits":41,"size_factor":42,"quality_loss":54},"INT8",0.02,{"FP16":56,"FP8":75,"INT8":85,"AWQ":89,"GPTQ":99,"GGUF":103},{"minimum":57,"production":64,"recommended":70},{"scenario":58,"context_length":59,"batch_size":37,"weight_gb":51,"kv_cache_gb":54,"vram_gb":60,"ram_gb":11,"disk_gb":61,"requirement_source":62,"confidence":63},"minimum",4096,2.29,39.05,"inferred",0.95,{"scenario":65,"context_length":66,"batch_size":10,"weight_gb":51,"kv_cache_gb":67,"vram_gb":68,"ram_gb":11,"disk_gb":69,"requirement_source":62,"confidence":63},"production",16384,2,5.3,104.05,{"scenario":71,"context_length":72,"batch_size":67,"weight_gb":51,"kv_cache_gb":46,"vram_gb":73,"ram_gb":11,"disk_gb":74,"requirement_source":62,"confidence":63},"recommended",8192,2.74,65.05,{"minimum":76,"production":79,"recommended":82},{"scenario":58,"context_length":59,"batch_size":37,"weight_gb":54,"kv_cache_gb":54,"vram_gb":77,"ram_gb":11,"disk_gb":78,"requirement_source":62,"confidence":63},2.27,39.03,{"scenario":65,"context_length":66,"batch_size":10,"weight_gb":54,"kv_cache_gb":67,"vram_gb":80,"ram_gb":11,"disk_gb":81,"requirement_source":62,"confidence":63},5.27,104.03,{"scenario":71,"context_length":72,"batch_size":67,"weight_gb":54,"kv_cache_gb":46,"vram_gb":83,"ram_gb":11,"disk_gb":84,"requirement_source":62,"confidence":63},2.72,65.03,{"minimum":86,"production":87,"recommended":88},{"scenario":58,"context_length":59,"batch_size":37,"weight_gb":54,"kv_cache_gb":54,"vram_gb":77,"ram_gb":11,"disk_gb":78,"requirement_source":62,"confidence":63},{"scenario":65,"context_length":66,"batch_size":10,"weight_gb":54,"kv_cache_gb":67,"vram_gb":80,"ram_gb":11,"disk_gb":81,"requirement_source":62,"confidence":63},{"scenario":71,"context_length":72,"batch_size":67,"weight_gb":54,"kv_cache_gb":46,"vram_gb":83,"ram_gb":11,"disk_gb":84,"requirement_source":62,"confidence":63},{"minimum":90,"production":93,"recommended":96},{"scenario":58,"context_length":59,"batch_size":37,"weight_gb":43,"kv_cache_gb":54,"vram_gb":91,"ram_gb":11,"disk_gb":92,"requirement_source":62,"confidence":63},2.25,39.01,{"scenario":65,"context_length":66,"batch_size":10,"weight_gb":43,"kv_cache_gb":67,"vram_gb":94,"ram_gb":11,"disk_gb":95,"requirement_source":62,"confidence":63},5.26,104.01,{"scenario":71,"context_length":72,"batch_size":67,"weight_gb":43,"kv_cache_gb":46,"vram_gb":97,"ram_gb":11,"disk_gb":98,"requirement_source":62,"confidence":63},2.71,65.01,{"minimum":100,"production":101,"recommended":102},{"scenario":58,"context_length":59,"batch_size":37,"weight_gb":43,"kv_cache_gb":54,"vram_gb":91,"ram_gb":11,"disk_gb":92,"requirement_source":62,"confidence":63},{"scenario":65,"context_length":66,"batch_size":10,"weight_gb":43,"kv_cache_gb":67,"vram_gb":94,"ram_gb":11,"disk_gb":95,"requirement_source":62,"confidence":63},{"scenario":71,"context_length":72,"batch_size":67,"weight_gb":43,"kv_cache_gb":46,"vram_gb":97,"ram_gb":11,"disk_gb":98,"requirement_source":62,"confidence":63},{"minimum":104,"production":105,"recommended":106},{"scenario":58,"context_length":59,"batch_size":37,"weight_gb":43,"kv_cache_gb":54,"vram_gb":91,"ram_gb":11,"disk_gb":92,"requirement_source":62,"confidence":63},{"scenario":65,"context_length":66,"batch_size":10,"weight_gb":43,"kv_cache_gb":67,"vram_gb":94,"ram_gb":11,"disk_gb":95,"requirement_source":62,"confidence":63},{"scenario":71,"context_length":72,"batch_size":67,"weight_gb":43,"kv_cache_gb":46,"vram_gb":97,"ram_gb":11,"disk_gb":98,"requirement_source":62,"confidence":63},{"recommended":108,"cheapest":126,"performance":135,"min_complexity":152,"cluster_options":167,"no_match":168,"required_vram_gb":97,"candidate_count":169,"usage":170},{"gpu_model":109,"gpu_model_id":110,"gpu_count":37,"total_vram_gb":36,"hour_price":111,"monthly_cost":112,"providers":113,"bandwidth_gbps":115,"fp16_tflops":116,"interconnect_score":117,"parallel_efficiency":37,"score":118,"tier":119,"perf_metric":120,"perf_label":121,"perf_value":122,"workload":16,"cost_metric":123},"RTX 5080",23,0.1339,97.75,[114],"vast",960,113,40,0.54,"cheapest","tok\u002Fs","96000 tok\u002Fs",96000,{"effective_value":124,"effective_unit":120,"capacity_factor":125},67200,0.7,{"gpu_model":127,"gpu_model_id":128,"gpu_count":37,"total_vram_gb":36,"hour_price":129,"monthly_cost":130,"providers":131,"bandwidth_gbps":38,"fp16_tflops":38,"interconnect_score":117,"parallel_efficiency":37,"score":132,"tier":119,"perf_metric":120,"perf_label":133,"perf_value":38,"workload":16,"cost_metric":134},"Tesla V100",17,0.0289,21.1,[114],0.5347,"0 tok\u002Fs",{"effective_value":38,"effective_unit":120,"capacity_factor":125},{"gpu_model":136,"gpu_model_id":137,"gpu_count":41,"total_vram_gb":138,"hour_price":139,"monthly_cost":140,"providers":141,"bandwidth_gbps":143,"fp16_tflops":144,"interconnect_score":117,"parallel_efficiency":145,"score":146,"tier":147,"perf_metric":120,"perf_label":148,"perf_value":149,"workload":16,"cost_metric":150},"H200 SXM",11,1128,3.5,20440,[142],"lambda",4800,989,0.4,0.2197,"balanced","1536000 tok\u002Fs",1536000,{"effective_value":151,"effective_unit":120,"capacity_factor":125},1075200,{"gpu_model":153,"gpu_model_id":154,"gpu_count":37,"total_vram_gb":155,"hour_price":156,"monthly_cost":157,"providers":158,"bandwidth_gbps":160,"fp16_tflops":161,"interconnect_score":117,"parallel_efficiency":37,"score":162,"tier":119,"perf_metric":120,"perf_label":163,"perf_value":164,"workload":16,"cost_metric":165},"RTX 3090",3,24,0.0622,45.41,[114,159],"tensordock",936,71,0.4957,"93600 tok\u002Fs",93600,{"effective_value":166,"effective_unit":120,"capacity_factor":125},65520,[],false,26,"value",[172,176,180,187,194],{"name":127,"vram_gb":36,"cheapest_hour":129,"cost":173,"provider":114,"estimated_tps":38},{"hourly":129,"daily":174,"monthly":130,"yearly":175},0.69,253.16,{"name":153,"vram_gb":155,"cheapest_hour":156,"cost":177,"provider":114,"estimated_tps":164},{"hourly":156,"daily":178,"monthly":157,"yearly":179},1.49,544.87,{"name":181,"vram_gb":36,"cheapest_hour":182,"cost":183,"provider":114,"estimated_tps":38},"RTX 4070S Ti",0.0678,{"hourly":182,"daily":184,"monthly":185,"yearly":186},1.63,49.49,593.93,{"name":188,"vram_gb":36,"cheapest_hour":189,"cost":190,"provider":114,"estimated_tps":38},"RTX 4080S",0.0685,{"hourly":189,"daily":191,"monthly":192,"yearly":193},1.64,50.01,600.06,{"name":195,"vram_gb":41,"cheapest_hour":196,"cost":197,"provider":114,"estimated_tps":201},"RTX 5060 Ti",0.0719,{"hourly":196,"daily":198,"monthly":199,"yearly":200},1.73,52.49,629.84,44800,{"posts_count":38,"benchmarks_count":38}]