[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-bge-small-en-v1-5":3},{"id":4,"slug":5,"name":6,"hf_id":6,"organization":7,"architecture":8,"architecture_config":9,"parameters":13,"parameters_label":14,"context_length":15,"license":16,"downloads":17,"likes":18,"status":19,"description":16,"huggingface_url":20,"created_at":21,"seo":22,"category":25,"family":28,"score":31,"variants":32,"quantizations":33,"requirements":60,"gpu_recommendations":117,"fitting_gpus":160,"community":194},433,"bge-small-en-v1-5","BAAI\u002Fbge-small-en-v1.5","BAAI","bert",{"layers":10,"head_dim":11,"hidden_size":12},12,32,384,32954112,"33M",512,null,72653415,532,"published","https:\u002F\u002Fhuggingface.co\u002FBAAI\u002Fbge-small-en-v1.5","2026-08-13T18:00:37+00:00",{"title":6,"description":23,"image":16,"robots":24,"canonical_url":16},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":26,"slug":27},"Embedding \u002F RAG","embedding",{"name":29,"slug":30},"BGE","bge",0.123,[],[34,39,44,49,53,57],{"format":35,"bits":36,"size_factor":37,"quality_loss":38},"AWQ",4,0.263,0.03,{"format":40,"bits":41,"size_factor":42,"quality_loss":43},"FP16",16,1,0,{"format":45,"bits":46,"size_factor":47,"quality_loss":48},"FP8",8,0.5,0.01,{"format":50,"bits":36,"size_factor":51,"quality_loss":52},"GGUF",0.25,0.05,{"format":54,"bits":36,"size_factor":55,"quality_loss":56},"GPTQ",0.275,0.04,{"format":58,"bits":46,"size_factor":47,"quality_loss":59},"INT8",0.02,{"FP16":61,"FP8":81,"INT8":91,"AWQ":95,"GPTQ":105,"GGUF":113},{"minimum":62,"production":70,"recommended":75},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":65,"kv_cache_gb":43,"vram_gb":66,"ram_gb":11,"disk_gb":67,"requirement_source":68,"confidence":69},"minimum",4096,0.06,1.3,39.1,"inferred",0.95,{"scenario":71,"context_length":72,"batch_size":36,"weight_gb":65,"kv_cache_gb":43,"vram_gb":73,"ram_gb":11,"disk_gb":74,"requirement_source":68,"confidence":69},"production",16384,1.78,104.1,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":65,"kv_cache_gb":43,"vram_gb":79,"ram_gb":11,"disk_gb":80,"requirement_source":68,"confidence":69},"recommended",8192,2,1.47,65.1,{"minimum":82,"production":85,"recommended":88},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":38,"kv_cache_gb":43,"vram_gb":83,"ram_gb":11,"disk_gb":84,"requirement_source":68,"confidence":69},1.25,39.05,{"scenario":71,"context_length":72,"batch_size":36,"weight_gb":38,"kv_cache_gb":43,"vram_gb":86,"ram_gb":11,"disk_gb":87,"requirement_source":68,"confidence":69},1.73,104.05,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":38,"kv_cache_gb":43,"vram_gb":89,"ram_gb":11,"disk_gb":90,"requirement_source":68,"confidence":69},1.43,65.05,{"minimum":92,"production":93,"recommended":94},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":38,"kv_cache_gb":43,"vram_gb":83,"ram_gb":11,"disk_gb":84,"requirement_source":68,"confidence":69},{"scenario":71,"context_length":72,"batch_size":36,"weight_gb":38,"kv_cache_gb":43,"vram_gb":86,"ram_gb":11,"disk_gb":87,"requirement_source":68,"confidence":69},{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":38,"kv_cache_gb":43,"vram_gb":89,"ram_gb":11,"disk_gb":90,"requirement_source":68,"confidence":69},{"minimum":96,"production":99,"recommended":102},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":59,"kv_cache_gb":43,"vram_gb":97,"ram_gb":11,"disk_gb":98,"requirement_source":68,"confidence":69},1.23,39.02,{"scenario":71,"context_length":72,"batch_size":36,"weight_gb":59,"kv_cache_gb":43,"vram_gb":100,"ram_gb":11,"disk_gb":101,"requirement_source":68,"confidence":69},1.7,104.02,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":59,"kv_cache_gb":43,"vram_gb":103,"ram_gb":11,"disk_gb":104,"requirement_source":68,"confidence":69},1.4,65.02,{"minimum":106,"production":108,"recommended":111},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":59,"kv_cache_gb":43,"vram_gb":97,"ram_gb":11,"disk_gb":107,"requirement_source":68,"confidence":69},39.03,{"scenario":71,"context_length":72,"batch_size":36,"weight_gb":59,"kv_cache_gb":43,"vram_gb":109,"ram_gb":11,"disk_gb":110,"requirement_source":68,"confidence":69},1.71,104.03,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":59,"kv_cache_gb":43,"vram_gb":103,"ram_gb":11,"disk_gb":112,"requirement_source":68,"confidence":69},65.03,{"minimum":114,"production":115,"recommended":116},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":59,"kv_cache_gb":43,"vram_gb":97,"ram_gb":11,"disk_gb":107,"requirement_source":68,"confidence":69},{"scenario":71,"context_length":72,"batch_size":36,"weight_gb":59,"kv_cache_gb":43,"vram_gb":109,"ram_gb":11,"disk_gb":110,"requirement_source":68,"confidence":69},{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":59,"kv_cache_gb":43,"vram_gb":103,"ram_gb":11,"disk_gb":112,"requirement_source":68,"confidence":69},{"recommended":118,"cheapest":135,"performance":138,"min_complexity":149,"cluster_options":157,"no_match":158,"required_vram_gb":103,"candidate_count":36,"usage":159},{"gpu_model":119,"gpu_model_id":120,"gpu_count":42,"total_vram_gb":46,"hour_price":121,"monthly_cost":122,"providers":123,"bandwidth_gbps":125,"fp16_tflops":120,"interconnect_score":126,"parallel_efficiency":42,"score":127,"tier":128,"perf_metric":129,"perf_label":130,"perf_value":131,"workload":16,"cost_metric":132},"RTX 5060 Ti",24,0.0719,52.49,[124],"vast",448,40,0.5388,"cheapest","vectors\u002Fs","44800 vec\u002Fs",44800,{"effective_value":133,"effective_unit":129,"capacity_factor":134},35840,0.8,{"gpu_model":119,"gpu_model_id":120,"gpu_count":42,"total_vram_gb":46,"hour_price":121,"monthly_cost":122,"providers":136,"bandwidth_gbps":125,"fp16_tflops":120,"interconnect_score":126,"parallel_efficiency":42,"score":127,"tier":128,"perf_metric":129,"perf_label":130,"perf_value":131,"workload":16,"cost_metric":137},[124],{"effective_value":133,"effective_unit":129,"capacity_factor":134},{"gpu_model":119,"gpu_model_id":120,"gpu_count":46,"total_vram_gb":139,"hour_price":121,"monthly_cost":140,"providers":141,"bandwidth_gbps":125,"fp16_tflops":120,"interconnect_score":126,"parallel_efficiency":142,"score":143,"tier":144,"perf_metric":129,"perf_label":145,"perf_value":146,"workload":16,"cost_metric":147},64,419.9,[124],0.4,0.3615,"balanced","358400 vec\u002Fs",358400,{"effective_value":148,"effective_unit":129,"capacity_factor":134},286720,{"gpu_model":119,"gpu_model_id":120,"gpu_count":78,"total_vram_gb":41,"hour_price":121,"monthly_cost":150,"providers":151,"bandwidth_gbps":125,"fp16_tflops":120,"interconnect_score":126,"parallel_efficiency":134,"score":152,"tier":128,"perf_metric":129,"perf_label":153,"perf_value":154,"workload":16,"cost_metric":155},104.97,[124],0.4504,"89600 vec\u002Fs",89600,{"effective_value":156,"effective_unit":129,"capacity_factor":134},71680,[],false,"value",[161,168,176,183,190],{"name":162,"vram_gb":41,"cheapest_hour":163,"cost":164,"provider":124,"estimated_tps":43},"Tesla V100",0.0289,{"hourly":163,"daily":165,"monthly":166,"yearly":167},0.69,21.1,253.16,{"name":169,"vram_gb":120,"cheapest_hour":170,"cost":171,"provider":124,"estimated_tps":175},"RTX 3090",0.0622,{"hourly":170,"daily":172,"monthly":173,"yearly":174},1.49,45.41,544.87,46800,{"name":177,"vram_gb":41,"cheapest_hour":178,"cost":179,"provider":124,"estimated_tps":43},"RTX 4070S Ti",0.0678,{"hourly":178,"daily":180,"monthly":181,"yearly":182},1.63,49.49,593.93,{"name":184,"vram_gb":41,"cheapest_hour":185,"cost":186,"provider":124,"estimated_tps":43},"RTX 4080S",0.0685,{"hourly":185,"daily":187,"monthly":188,"yearly":189},1.64,50.01,600.06,{"name":119,"vram_gb":46,"cheapest_hour":121,"cost":191,"provider":124,"estimated_tps":193},{"hourly":121,"daily":86,"monthly":122,"yearly":192},629.84,22400,{"posts_count":43,"benchmarks_count":43}]