[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-bge-small-zh-v1-5":3},{"id":4,"slug":5,"name":6,"hf_id":6,"organization":7,"architecture":8,"architecture_config":9,"parameters":13,"parameters_label":14,"context_length":12,"license":15,"downloads":16,"likes":17,"status":18,"description":15,"huggingface_url":19,"created_at":20,"seo":21,"category":24,"family":27,"score":30,"variants":31,"quantizations":32,"requirements":58,"gpu_recommendations":111,"fitting_gpus":153,"community":188},494,"bge-small-zh-v1-5","BAAI\u002Fbge-small-zh-v1.5","BAAI","bert",{"layers":10,"head_dim":11,"hidden_size":12},4,64,512,23400448,"23.4M",null,4787964,136,"published","https:\u002F\u002Fhuggingface.co\u002FBAAI\u002Fbge-small-zh-v1.5","2026-08-13T18:11:23+00:00",{"title":6,"description":22,"image":15,"robots":23,"canonical_url":15},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":25,"slug":26},"Embedding \u002F RAG","embedding",{"name":28,"slug":29},"BGE","bge",0.009,[],[33,37,42,47,51,55],{"format":34,"bits":10,"size_factor":35,"quality_loss":36},"AWQ",0.263,0.03,{"format":38,"bits":39,"size_factor":40,"quality_loss":41},"FP16",16,1,0,{"format":43,"bits":44,"size_factor":45,"quality_loss":46},"FP8",8,0.5,0.01,{"format":48,"bits":10,"size_factor":49,"quality_loss":50},"GGUF",0.25,0.05,{"format":52,"bits":10,"size_factor":53,"quality_loss":54},"GPTQ",0.275,0.04,{"format":56,"bits":44,"size_factor":45,"quality_loss":57},"INT8",0.02,{"FP16":59,"FP8":79,"INT8":89,"AWQ":93,"GPTQ":103,"GGUF":107},{"minimum":60,"production":68,"recommended":73},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":54,"kv_cache_gb":41,"vram_gb":63,"ram_gb":64,"disk_gb":65,"requirement_source":66,"confidence":67},"minimum",4096,1.27,32,39.07,"inferred",0.95,{"scenario":69,"context_length":70,"batch_size":10,"weight_gb":54,"kv_cache_gb":41,"vram_gb":71,"ram_gb":64,"disk_gb":72,"requirement_source":66,"confidence":67},"production",16384,1.75,104.07,{"scenario":74,"context_length":75,"batch_size":76,"weight_gb":54,"kv_cache_gb":41,"vram_gb":77,"ram_gb":64,"disk_gb":78,"requirement_source":66,"confidence":67},"recommended",8192,2,1.45,65.07,{"minimum":80,"production":83,"recommended":86},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":57,"kv_cache_gb":41,"vram_gb":81,"ram_gb":64,"disk_gb":82,"requirement_source":66,"confidence":67},1.24,39.03,{"scenario":69,"context_length":70,"batch_size":10,"weight_gb":57,"kv_cache_gb":41,"vram_gb":84,"ram_gb":64,"disk_gb":85,"requirement_source":66,"confidence":67},1.71,104.03,{"scenario":74,"context_length":75,"batch_size":76,"weight_gb":57,"kv_cache_gb":41,"vram_gb":87,"ram_gb":64,"disk_gb":88,"requirement_source":66,"confidence":67},1.41,65.03,{"minimum":90,"production":91,"recommended":92},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":57,"kv_cache_gb":41,"vram_gb":81,"ram_gb":64,"disk_gb":82,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":10,"weight_gb":57,"kv_cache_gb":41,"vram_gb":84,"ram_gb":64,"disk_gb":85,"requirement_source":66,"confidence":67},{"scenario":74,"context_length":75,"batch_size":76,"weight_gb":57,"kv_cache_gb":41,"vram_gb":87,"ram_gb":64,"disk_gb":88,"requirement_source":66,"confidence":67},{"minimum":94,"production":97,"recommended":100},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":46,"kv_cache_gb":41,"vram_gb":95,"ram_gb":64,"disk_gb":96,"requirement_source":66,"confidence":67},1.23,39.02,{"scenario":69,"context_length":70,"batch_size":10,"weight_gb":46,"kv_cache_gb":41,"vram_gb":98,"ram_gb":64,"disk_gb":99,"requirement_source":66,"confidence":67},1.7,104.02,{"scenario":74,"context_length":75,"batch_size":76,"weight_gb":46,"kv_cache_gb":41,"vram_gb":101,"ram_gb":64,"disk_gb":102,"requirement_source":66,"confidence":67},1.4,65.02,{"minimum":104,"production":105,"recommended":106},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":46,"kv_cache_gb":41,"vram_gb":95,"ram_gb":64,"disk_gb":96,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":10,"weight_gb":46,"kv_cache_gb":41,"vram_gb":98,"ram_gb":64,"disk_gb":99,"requirement_source":66,"confidence":67},{"scenario":74,"context_length":75,"batch_size":76,"weight_gb":46,"kv_cache_gb":41,"vram_gb":101,"ram_gb":64,"disk_gb":102,"requirement_source":66,"confidence":67},{"minimum":108,"production":109,"recommended":110},{"scenario":61,"context_length":62,"batch_size":40,"weight_gb":46,"kv_cache_gb":41,"vram_gb":95,"ram_gb":64,"disk_gb":96,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":10,"weight_gb":46,"kv_cache_gb":41,"vram_gb":98,"ram_gb":64,"disk_gb":99,"requirement_source":66,"confidence":67},{"scenario":74,"context_length":75,"batch_size":76,"weight_gb":46,"kv_cache_gb":41,"vram_gb":101,"ram_gb":64,"disk_gb":102,"requirement_source":66,"confidence":67},{"recommended":112,"cheapest":129,"performance":132,"min_complexity":142,"cluster_options":150,"no_match":151,"required_vram_gb":101,"candidate_count":10,"usage":152},{"gpu_model":113,"gpu_model_id":114,"gpu_count":40,"total_vram_gb":44,"hour_price":115,"monthly_cost":116,"providers":117,"bandwidth_gbps":119,"fp16_tflops":114,"interconnect_score":120,"parallel_efficiency":40,"score":121,"tier":122,"perf_metric":123,"perf_label":124,"perf_value":125,"workload":15,"cost_metric":126},"RTX 5060 Ti",24,0.0719,52.49,[118],"vast",448,40,0.5388,"cheapest","vectors\u002Fs","89600 vec\u002Fs",89600,{"effective_value":127,"effective_unit":123,"capacity_factor":128},71680,0.8,{"gpu_model":113,"gpu_model_id":114,"gpu_count":40,"total_vram_gb":44,"hour_price":115,"monthly_cost":116,"providers":130,"bandwidth_gbps":119,"fp16_tflops":114,"interconnect_score":120,"parallel_efficiency":40,"score":121,"tier":122,"perf_metric":123,"perf_label":124,"perf_value":125,"workload":15,"cost_metric":131},[118],{"effective_value":127,"effective_unit":123,"capacity_factor":128},{"gpu_model":113,"gpu_model_id":114,"gpu_count":44,"total_vram_gb":11,"hour_price":115,"monthly_cost":133,"providers":134,"bandwidth_gbps":119,"fp16_tflops":114,"interconnect_score":120,"parallel_efficiency":135,"score":136,"tier":137,"perf_metric":123,"perf_label":138,"perf_value":139,"workload":15,"cost_metric":140},419.9,[118],0.4,0.3615,"balanced","716800 vec\u002Fs",716800,{"effective_value":141,"effective_unit":123,"capacity_factor":128},573440,{"gpu_model":113,"gpu_model_id":114,"gpu_count":76,"total_vram_gb":39,"hour_price":115,"monthly_cost":143,"providers":144,"bandwidth_gbps":119,"fp16_tflops":114,"interconnect_score":120,"parallel_efficiency":128,"score":145,"tier":122,"perf_metric":123,"perf_label":146,"perf_value":147,"workload":15,"cost_metric":148},104.97,[118],0.4504,"179200 vec\u002Fs",179200,{"effective_value":149,"effective_unit":123,"capacity_factor":128},143360,[],false,"value",[154,161,169,176,183],{"name":155,"vram_gb":39,"cheapest_hour":156,"cost":157,"provider":118,"estimated_tps":41},"Tesla V100",0.0289,{"hourly":156,"daily":158,"monthly":159,"yearly":160},0.69,21.1,253.16,{"name":162,"vram_gb":114,"cheapest_hour":163,"cost":164,"provider":118,"estimated_tps":168},"RTX 3090",0.0622,{"hourly":163,"daily":165,"monthly":166,"yearly":167},1.49,45.41,544.87,93600,{"name":170,"vram_gb":39,"cheapest_hour":171,"cost":172,"provider":118,"estimated_tps":41},"RTX 4070S Ti",0.0678,{"hourly":171,"daily":173,"monthly":174,"yearly":175},1.63,49.49,593.93,{"name":177,"vram_gb":39,"cheapest_hour":178,"cost":179,"provider":118,"estimated_tps":41},"RTX 4080S",0.0685,{"hourly":178,"daily":180,"monthly":181,"yearly":182},1.64,50.01,600.06,{"name":113,"vram_gb":44,"cheapest_hour":115,"cost":184,"provider":118,"estimated_tps":187},{"hourly":115,"daily":185,"monthly":116,"yearly":186},1.73,629.84,44800,{"posts_count":41,"benchmarks_count":41}]