[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-multilingual-e5-small":3},{"id":4,"slug":5,"name":6,"hf_id":6,"organization":7,"architecture":8,"architecture_config":9,"parameters":13,"parameters_label":14,"context_length":15,"license":16,"downloads":17,"likes":12,"status":18,"description":16,"huggingface_url":19,"created_at":20,"seo":21,"category":24,"family":27,"score":30,"variants":31,"quantizations":32,"requirements":59,"gpu_recommendations":115,"fitting_gpus":158,"community":190},453,"multilingual-e5-small","intfloat\u002Fmultilingual-e5-small","intfloat","bert",{"layers":10,"head_dim":11,"hidden_size":12},12,32,384,117247872,"117.2M",512,null,12214525,"published","https:\u002F\u002Fhuggingface.co\u002Fintfloat\u002Fmultilingual-e5-small","2026-08-13T18:03:05+00:00",{"title":6,"description":22,"image":16,"robots":23,"canonical_url":16},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":25,"slug":26},"Embedding \u002F RAG","embedding",{"name":28,"slug":29},"Yi","yi",0.034,[],[33,38,43,48,52,56],{"format":34,"bits":35,"size_factor":36,"quality_loss":37},"AWQ",4,0.263,0.03,{"format":39,"bits":40,"size_factor":41,"quality_loss":42},"FP16",16,1,0,{"format":44,"bits":45,"size_factor":46,"quality_loss":47},"FP8",8,0.5,0.01,{"format":49,"bits":35,"size_factor":50,"quality_loss":51},"GGUF",0.25,0.05,{"format":53,"bits":35,"size_factor":54,"quality_loss":55},"GPTQ",0.275,0.04,{"format":57,"bits":45,"size_factor":46,"quality_loss":58},"INT8",0.02,{"FP16":60,"FP8":80,"INT8":91,"AWQ":95,"GPTQ":106,"GGUF":111},{"minimum":61,"production":69,"recommended":74},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":64,"kv_cache_gb":42,"vram_gb":65,"ram_gb":11,"disk_gb":66,"requirement_source":67,"confidence":68},"minimum",4096,0.22,1.52,39.34,"inferred",0.95,{"scenario":70,"context_length":71,"batch_size":35,"weight_gb":64,"kv_cache_gb":42,"vram_gb":72,"ram_gb":11,"disk_gb":73,"requirement_source":67,"confidence":68},"production",16384,2.02,104.34,{"scenario":75,"context_length":76,"batch_size":77,"weight_gb":64,"kv_cache_gb":42,"vram_gb":78,"ram_gb":11,"disk_gb":79,"requirement_source":67,"confidence":68},"recommended",8192,2,1.71,65.34,{"minimum":81,"production":85,"recommended":88},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":82,"kv_cache_gb":42,"vram_gb":83,"ram_gb":11,"disk_gb":84,"requirement_source":67,"confidence":68},0.11,1.37,39.17,{"scenario":70,"context_length":71,"batch_size":35,"weight_gb":82,"kv_cache_gb":42,"vram_gb":86,"ram_gb":11,"disk_gb":87,"requirement_source":67,"confidence":68},1.85,104.17,{"scenario":75,"context_length":76,"batch_size":77,"weight_gb":82,"kv_cache_gb":42,"vram_gb":89,"ram_gb":11,"disk_gb":90,"requirement_source":67,"confidence":68},1.54,65.17,{"minimum":92,"production":93,"recommended":94},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":82,"kv_cache_gb":42,"vram_gb":83,"ram_gb":11,"disk_gb":84,"requirement_source":67,"confidence":68},{"scenario":70,"context_length":71,"batch_size":35,"weight_gb":82,"kv_cache_gb":42,"vram_gb":86,"ram_gb":11,"disk_gb":87,"requirement_source":67,"confidence":68},{"scenario":75,"context_length":76,"batch_size":77,"weight_gb":82,"kv_cache_gb":42,"vram_gb":89,"ram_gb":11,"disk_gb":90,"requirement_source":67,"confidence":68},{"minimum":96,"production":100,"recommended":103},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":97,"kv_cache_gb":42,"vram_gb":98,"ram_gb":11,"disk_gb":99,"requirement_source":67,"confidence":68},0.06,1.29,39.09,{"scenario":70,"context_length":71,"batch_size":35,"weight_gb":97,"kv_cache_gb":42,"vram_gb":101,"ram_gb":11,"disk_gb":102,"requirement_source":67,"confidence":68},1.77,104.09,{"scenario":75,"context_length":76,"batch_size":77,"weight_gb":97,"kv_cache_gb":42,"vram_gb":104,"ram_gb":11,"disk_gb":105,"requirement_source":67,"confidence":68},1.46,65.09,{"minimum":107,"production":108,"recommended":109},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":97,"kv_cache_gb":42,"vram_gb":98,"ram_gb":11,"disk_gb":99,"requirement_source":67,"confidence":68},{"scenario":70,"context_length":71,"batch_size":35,"weight_gb":97,"kv_cache_gb":42,"vram_gb":101,"ram_gb":11,"disk_gb":102,"requirement_source":67,"confidence":68},{"scenario":75,"context_length":76,"batch_size":77,"weight_gb":97,"kv_cache_gb":42,"vram_gb":110,"ram_gb":11,"disk_gb":105,"requirement_source":67,"confidence":68},1.47,{"minimum":112,"production":113,"recommended":114},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":97,"kv_cache_gb":42,"vram_gb":98,"ram_gb":11,"disk_gb":99,"requirement_source":67,"confidence":68},{"scenario":70,"context_length":71,"batch_size":35,"weight_gb":97,"kv_cache_gb":42,"vram_gb":101,"ram_gb":11,"disk_gb":102,"requirement_source":67,"confidence":68},{"scenario":75,"context_length":76,"batch_size":77,"weight_gb":97,"kv_cache_gb":42,"vram_gb":110,"ram_gb":11,"disk_gb":105,"requirement_source":67,"confidence":68},{"recommended":116,"cheapest":133,"performance":136,"min_complexity":147,"cluster_options":155,"no_match":156,"required_vram_gb":104,"candidate_count":35,"usage":157},{"gpu_model":117,"gpu_model_id":118,"gpu_count":41,"total_vram_gb":45,"hour_price":119,"monthly_cost":120,"providers":121,"bandwidth_gbps":123,"fp16_tflops":118,"interconnect_score":124,"parallel_efficiency":41,"score":125,"tier":126,"perf_metric":127,"perf_label":128,"perf_value":129,"workload":16,"cost_metric":130},"RTX 5060 Ti",24,0.0554,40.44,[122],"vast",448,40,0.541,"cheapest","vectors\u002Fs","14933 vec\u002Fs",14933,{"effective_value":131,"effective_unit":127,"capacity_factor":132},11946.4,0.8,{"gpu_model":117,"gpu_model_id":118,"gpu_count":41,"total_vram_gb":45,"hour_price":119,"monthly_cost":120,"providers":134,"bandwidth_gbps":123,"fp16_tflops":118,"interconnect_score":124,"parallel_efficiency":41,"score":125,"tier":126,"perf_metric":127,"perf_label":128,"perf_value":129,"workload":16,"cost_metric":135},[122],{"effective_value":131,"effective_unit":127,"capacity_factor":132},{"gpu_model":117,"gpu_model_id":118,"gpu_count":45,"total_vram_gb":137,"hour_price":119,"monthly_cost":138,"providers":139,"bandwidth_gbps":123,"fp16_tflops":118,"interconnect_score":124,"parallel_efficiency":140,"score":141,"tier":142,"perf_metric":127,"perf_label":143,"perf_value":144,"workload":16,"cost_metric":145},64,323.54,[122],0.4,0.3635,"balanced","119467 vec\u002Fs",119467,{"effective_value":146,"effective_unit":127,"capacity_factor":132},95573.6,{"gpu_model":117,"gpu_model_id":118,"gpu_count":77,"total_vram_gb":40,"hour_price":119,"monthly_cost":148,"providers":149,"bandwidth_gbps":123,"fp16_tflops":118,"interconnect_score":124,"parallel_efficiency":132,"score":150,"tier":126,"perf_metric":127,"perf_label":151,"perf_value":152,"workload":16,"cost_metric":153},80.88,[122],0.4509,"29867 vec\u002Fs",29867,{"effective_value":154,"effective_unit":127,"capacity_factor":132},23893.6,[],false,"value",[159,166,171,179,182],{"name":160,"vram_gb":40,"cheapest_hour":161,"cost":162,"provider":122,"estimated_tps":42},"Tesla V100",0.0272,{"hourly":161,"daily":163,"monthly":164,"yearly":165},0.65,19.86,238.27,{"name":117,"vram_gb":45,"cheapest_hour":119,"cost":167,"provider":122,"estimated_tps":170},{"hourly":119,"daily":168,"monthly":120,"yearly":169},1.33,485.3,7466.7,{"name":172,"vram_gb":118,"cheapest_hour":173,"cost":174,"provider":122,"estimated_tps":178},"RTX 3090",0.0678,{"hourly":173,"daily":175,"monthly":176,"yearly":177},1.63,49.49,593.93,15600,{"name":180,"vram_gb":40,"cheapest_hour":173,"cost":181,"provider":122,"estimated_tps":42},"RTX 4070S Ti",{"hourly":173,"daily":175,"monthly":176,"yearly":177},{"name":183,"vram_gb":10,"cheapest_hour":184,"cost":185,"provider":122,"estimated_tps":189},"RTX 5070",0.0804,{"hourly":184,"daily":186,"monthly":187,"yearly":188},1.93,58.69,704.3,11200,{"posts_count":42,"benchmarks_count":42}]