[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-bge-small-en-v1-5":3},{"id":4,"slug":5,"name":6,"hf_id":6,"organization":7,"architecture":8,"architecture_config":9,"parameters":13,"parameters_label":14,"context_length":15,"license":16,"downloads":17,"likes":18,"status":19,"description":16,"huggingface_url":20,"created_at":21,"seo":22,"category":25,"family":28,"score":31,"variants":32,"quantizations":33,"requirements":60,"gpu_recommendations":117,"fitting_gpus":160,"community":192},433,"bge-small-en-v1-5","BAAI\u002Fbge-small-en-v1.5","BAAI","bert",{"layers":10,"head_dim":11,"hidden_size":12},12,32,384,32954112,"33M",512,null,72653415,532,"published","https:\u002F\u002Fhuggingface.co\u002FBAAI\u002Fbge-small-en-v1.5","2026-08-13T18:00:37+00:00",{"title":6,"description":23,"image":16,"robots":24,"canonical_url":16},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":26,"slug":27},"Embedding \u002F RAG","embedding",{"name":29,"slug":30},"BGE","bge",0.123,[],[34,39,44,49,53,57],{"format":35,"bits":36,"size_factor":37,"quality_loss":38},"AWQ",4,0.263,0.03,{"format":40,"bits":41,"size_factor":42,"quality_loss":43},"FP16",16,1,0,{"format":45,"bits":46,"size_factor":47,"quality_loss":48},"FP8",8,0.5,0.01,{"format":50,"bits":36,"size_factor":51,"quality_loss":52},"GGUF",0.25,0.05,{"format":54,"bits":36,"size_factor":55,"quality_loss":56},"GPTQ",0.275,0.04,{"format":58,"bits":46,"size_factor":47,"quality_loss":59},"INT8",0.02,{"FP16":61,"FP8":81,"INT8":91,"AWQ":95,"GPTQ":105,"GGUF":113},{"minimum":62,"production":70,"recommended":75},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":65,"kv_cache_gb":43,"vram_gb":66,"ram_gb":11,"disk_gb":67,"requirement_source":68,"confidence":69},"minimum",4096,0.06,1.3,39.1,"inferred",0.95,{"scenario":71,"context_length":72,"batch_size":36,"weight_gb":65,"kv_cache_gb":43,"vram_gb":73,"ram_gb":11,"disk_gb":74,"requirement_source":68,"confidence":69},"production",16384,1.78,104.1,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":65,"kv_cache_gb":43,"vram_gb":79,"ram_gb":11,"disk_gb":80,"requirement_source":68,"confidence":69},"recommended",8192,2,1.47,65.1,{"minimum":82,"production":85,"recommended":88},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":38,"kv_cache_gb":43,"vram_gb":83,"ram_gb":11,"disk_gb":84,"requirement_source":68,"confidence":69},1.25,39.05,{"scenario":71,"context_length":72,"batch_size":36,"weight_gb":38,"kv_cache_gb":43,"vram_gb":86,"ram_gb":11,"disk_gb":87,"requirement_source":68,"confidence":69},1.73,104.05,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":38,"kv_cache_gb":43,"vram_gb":89,"ram_gb":11,"disk_gb":90,"requirement_source":68,"confidence":69},1.43,65.05,{"minimum":92,"production":93,"recommended":94},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":38,"kv_cache_gb":43,"vram_gb":83,"ram_gb":11,"disk_gb":84,"requirement_source":68,"confidence":69},{"scenario":71,"context_length":72,"batch_size":36,"weight_gb":38,"kv_cache_gb":43,"vram_gb":86,"ram_gb":11,"disk_gb":87,"requirement_source":68,"confidence":69},{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":38,"kv_cache_gb":43,"vram_gb":89,"ram_gb":11,"disk_gb":90,"requirement_source":68,"confidence":69},{"minimum":96,"production":99,"recommended":102},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":59,"kv_cache_gb":43,"vram_gb":97,"ram_gb":11,"disk_gb":98,"requirement_source":68,"confidence":69},1.23,39.02,{"scenario":71,"context_length":72,"batch_size":36,"weight_gb":59,"kv_cache_gb":43,"vram_gb":100,"ram_gb":11,"disk_gb":101,"requirement_source":68,"confidence":69},1.7,104.02,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":59,"kv_cache_gb":43,"vram_gb":103,"ram_gb":11,"disk_gb":104,"requirement_source":68,"confidence":69},1.4,65.02,{"minimum":106,"production":108,"recommended":111},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":59,"kv_cache_gb":43,"vram_gb":97,"ram_gb":11,"disk_gb":107,"requirement_source":68,"confidence":69},39.03,{"scenario":71,"context_length":72,"batch_size":36,"weight_gb":59,"kv_cache_gb":43,"vram_gb":109,"ram_gb":11,"disk_gb":110,"requirement_source":68,"confidence":69},1.71,104.03,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":59,"kv_cache_gb":43,"vram_gb":103,"ram_gb":11,"disk_gb":112,"requirement_source":68,"confidence":69},65.03,{"minimum":114,"production":115,"recommended":116},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":59,"kv_cache_gb":43,"vram_gb":97,"ram_gb":11,"disk_gb":107,"requirement_source":68,"confidence":69},{"scenario":71,"context_length":72,"batch_size":36,"weight_gb":59,"kv_cache_gb":43,"vram_gb":109,"ram_gb":11,"disk_gb":110,"requirement_source":68,"confidence":69},{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":59,"kv_cache_gb":43,"vram_gb":103,"ram_gb":11,"disk_gb":112,"requirement_source":68,"confidence":69},{"recommended":118,"cheapest":135,"performance":138,"min_complexity":149,"cluster_options":157,"no_match":158,"required_vram_gb":103,"candidate_count":36,"usage":159},{"gpu_model":119,"gpu_model_id":120,"gpu_count":42,"total_vram_gb":46,"hour_price":121,"monthly_cost":122,"providers":123,"bandwidth_gbps":125,"fp16_tflops":120,"interconnect_score":126,"parallel_efficiency":42,"score":127,"tier":128,"perf_metric":129,"perf_label":130,"perf_value":131,"workload":16,"cost_metric":132},"RTX 5060 Ti",24,0.0554,40.44,[124],"vast",448,40,0.539,"cheapest","vectors\u002Fs","44800 vec\u002Fs",44800,{"effective_value":133,"effective_unit":129,"capacity_factor":134},35840,0.8,{"gpu_model":119,"gpu_model_id":120,"gpu_count":42,"total_vram_gb":46,"hour_price":121,"monthly_cost":122,"providers":136,"bandwidth_gbps":125,"fp16_tflops":120,"interconnect_score":126,"parallel_efficiency":42,"score":127,"tier":128,"perf_metric":129,"perf_label":130,"perf_value":131,"workload":16,"cost_metric":137},[124],{"effective_value":133,"effective_unit":129,"capacity_factor":134},{"gpu_model":119,"gpu_model_id":120,"gpu_count":46,"total_vram_gb":139,"hour_price":121,"monthly_cost":140,"providers":141,"bandwidth_gbps":125,"fp16_tflops":120,"interconnect_score":126,"parallel_efficiency":142,"score":143,"tier":144,"perf_metric":129,"perf_label":145,"perf_value":146,"workload":16,"cost_metric":147},64,323.54,[124],0.4,0.3635,"balanced","358400 vec\u002Fs",358400,{"effective_value":148,"effective_unit":129,"capacity_factor":134},286720,{"gpu_model":119,"gpu_model_id":120,"gpu_count":78,"total_vram_gb":41,"hour_price":121,"monthly_cost":150,"providers":151,"bandwidth_gbps":125,"fp16_tflops":120,"interconnect_score":126,"parallel_efficiency":134,"score":152,"tier":128,"perf_metric":129,"perf_label":153,"perf_value":154,"workload":16,"cost_metric":155},80.88,[124],0.4509,"89600 vec\u002Fs",89600,{"effective_value":156,"effective_unit":129,"capacity_factor":134},71680,[],false,"value",[161,168,173,181,184],{"name":162,"vram_gb":41,"cheapest_hour":163,"cost":164,"provider":124,"estimated_tps":43},"Tesla V100",0.0272,{"hourly":163,"daily":165,"monthly":166,"yearly":167},0.65,19.86,238.27,{"name":119,"vram_gb":46,"cheapest_hour":121,"cost":169,"provider":124,"estimated_tps":172},{"hourly":121,"daily":170,"monthly":122,"yearly":171},1.33,485.3,22400,{"name":174,"vram_gb":120,"cheapest_hour":175,"cost":176,"provider":124,"estimated_tps":180},"RTX 3090",0.0678,{"hourly":175,"daily":177,"monthly":178,"yearly":179},1.63,49.49,593.93,46800,{"name":182,"vram_gb":41,"cheapest_hour":175,"cost":183,"provider":124,"estimated_tps":43},"RTX 4070S Ti",{"hourly":175,"daily":177,"monthly":178,"yearly":179},{"name":185,"vram_gb":10,"cheapest_hour":186,"cost":187,"provider":124,"estimated_tps":191},"RTX 5070",0.0804,{"hourly":186,"daily":188,"monthly":189,"yearly":190},1.93,58.69,704.3,33600,{"posts_count":43,"benchmarks_count":43}]