[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-bge-base-en-v1-5":3},{"id":4,"slug":5,"name":6,"hf_id":6,"organization":7,"architecture":8,"architecture_config":9,"parameters":13,"parameters_label":14,"context_length":15,"license":16,"downloads":17,"likes":18,"status":19,"description":16,"huggingface_url":20,"created_at":21,"seo":22,"category":25,"family":28,"score":31,"variants":32,"quantizations":33,"requirements":60,"gpu_recommendations":116,"fitting_gpus":158,"community":190},456,"bge-base-en-v1-5","BAAI\u002Fbge-base-en-v1.5","BAAI","bert",{"layers":10,"head_dim":11,"hidden_size":12},12,64,768,108375552,"108.4M",512,null,11252114,464,"published","https:\u002F\u002Fhuggingface.co\u002FBAAI\u002Fbge-base-en-v1.5","2026-08-13T18:03:16+00:00",{"title":6,"description":23,"image":16,"robots":24,"canonical_url":16},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":26,"slug":27},"Embedding \u002F RAG","embedding",{"name":29,"slug":30},"BGE","bge",0.0481,[],[34,39,44,49,53,57],{"format":35,"bits":36,"size_factor":37,"quality_loss":38},"AWQ",4,0.263,0.03,{"format":40,"bits":41,"size_factor":42,"quality_loss":43},"FP16",16,1,0,{"format":45,"bits":46,"size_factor":47,"quality_loss":48},"FP8",8,0.5,0.01,{"format":50,"bits":36,"size_factor":51,"quality_loss":52},"GGUF",0.25,0.05,{"format":54,"bits":36,"size_factor":55,"quality_loss":56},"GPTQ",0.275,0.04,{"format":58,"bits":46,"size_factor":47,"quality_loss":59},"INT8",0.02,{"FP16":61,"FP8":82,"INT8":93,"AWQ":97,"GPTQ":107,"GGUF":112},{"minimum":62,"production":71,"recommended":76},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":65,"kv_cache_gb":43,"vram_gb":66,"ram_gb":67,"disk_gb":68,"requirement_source":69,"confidence":70},"minimum",4096,0.2,1.5,32,39.31,"inferred",0.95,{"scenario":72,"context_length":73,"batch_size":36,"weight_gb":65,"kv_cache_gb":43,"vram_gb":74,"ram_gb":67,"disk_gb":75,"requirement_source":69,"confidence":70},"production",16384,1.99,104.31,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":65,"kv_cache_gb":43,"vram_gb":80,"ram_gb":67,"disk_gb":81,"requirement_source":69,"confidence":70},"recommended",8192,2,1.68,65.31,{"minimum":83,"production":87,"recommended":90},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":84,"kv_cache_gb":43,"vram_gb":85,"ram_gb":67,"disk_gb":86,"requirement_source":69,"confidence":70},0.1,1.35,39.16,{"scenario":72,"context_length":73,"batch_size":36,"weight_gb":84,"kv_cache_gb":43,"vram_gb":88,"ram_gb":67,"disk_gb":89,"requirement_source":69,"confidence":70},1.84,104.16,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":84,"kv_cache_gb":43,"vram_gb":91,"ram_gb":67,"disk_gb":92,"requirement_source":69,"confidence":70},1.53,65.16,{"minimum":94,"production":95,"recommended":96},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":84,"kv_cache_gb":43,"vram_gb":85,"ram_gb":67,"disk_gb":86,"requirement_source":69,"confidence":70},{"scenario":72,"context_length":73,"batch_size":36,"weight_gb":84,"kv_cache_gb":43,"vram_gb":88,"ram_gb":67,"disk_gb":89,"requirement_source":69,"confidence":70},{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":84,"kv_cache_gb":43,"vram_gb":91,"ram_gb":67,"disk_gb":92,"requirement_source":69,"confidence":70},{"minimum":98,"production":101,"recommended":104},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":52,"kv_cache_gb":43,"vram_gb":99,"ram_gb":67,"disk_gb":100,"requirement_source":69,"confidence":70},1.28,39.08,{"scenario":72,"context_length":73,"batch_size":36,"weight_gb":52,"kv_cache_gb":43,"vram_gb":102,"ram_gb":67,"disk_gb":103,"requirement_source":69,"confidence":70},1.76,104.08,{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":52,"kv_cache_gb":43,"vram_gb":105,"ram_gb":67,"disk_gb":106,"requirement_source":69,"confidence":70},1.46,65.08,{"minimum":108,"production":110,"recommended":111},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":52,"kv_cache_gb":43,"vram_gb":109,"ram_gb":67,"disk_gb":100,"requirement_source":69,"confidence":70},1.29,{"scenario":72,"context_length":73,"batch_size":36,"weight_gb":52,"kv_cache_gb":43,"vram_gb":102,"ram_gb":67,"disk_gb":103,"requirement_source":69,"confidence":70},{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":52,"kv_cache_gb":43,"vram_gb":105,"ram_gb":67,"disk_gb":106,"requirement_source":69,"confidence":70},{"minimum":113,"production":114,"recommended":115},{"scenario":63,"context_length":64,"batch_size":42,"weight_gb":52,"kv_cache_gb":43,"vram_gb":109,"ram_gb":67,"disk_gb":100,"requirement_source":69,"confidence":70},{"scenario":72,"context_length":73,"batch_size":36,"weight_gb":52,"kv_cache_gb":43,"vram_gb":102,"ram_gb":67,"disk_gb":103,"requirement_source":69,"confidence":70},{"scenario":77,"context_length":78,"batch_size":79,"weight_gb":52,"kv_cache_gb":43,"vram_gb":105,"ram_gb":67,"disk_gb":106,"requirement_source":69,"confidence":70},{"recommended":117,"cheapest":134,"performance":137,"min_complexity":147,"cluster_options":155,"no_match":156,"required_vram_gb":105,"candidate_count":36,"usage":157},{"gpu_model":118,"gpu_model_id":119,"gpu_count":42,"total_vram_gb":46,"hour_price":120,"monthly_cost":121,"providers":122,"bandwidth_gbps":124,"fp16_tflops":119,"interconnect_score":125,"parallel_efficiency":42,"score":126,"tier":127,"perf_metric":128,"perf_label":129,"perf_value":130,"workload":16,"cost_metric":131},"RTX 5060 Ti",24,0.0554,40.44,[123],"vast",448,40,0.541,"cheapest","vectors\u002Fs","17920 vec\u002Fs",17920,{"effective_value":132,"effective_unit":128,"capacity_factor":133},14336,0.8,{"gpu_model":118,"gpu_model_id":119,"gpu_count":42,"total_vram_gb":46,"hour_price":120,"monthly_cost":121,"providers":135,"bandwidth_gbps":124,"fp16_tflops":119,"interconnect_score":125,"parallel_efficiency":42,"score":126,"tier":127,"perf_metric":128,"perf_label":129,"perf_value":130,"workload":16,"cost_metric":136},[123],{"effective_value":132,"effective_unit":128,"capacity_factor":133},{"gpu_model":118,"gpu_model_id":119,"gpu_count":46,"total_vram_gb":11,"hour_price":120,"monthly_cost":138,"providers":139,"bandwidth_gbps":124,"fp16_tflops":119,"interconnect_score":125,"parallel_efficiency":140,"score":141,"tier":142,"perf_metric":128,"perf_label":143,"perf_value":144,"workload":16,"cost_metric":145},323.54,[123],0.4,0.3635,"balanced","143360 vec\u002Fs",143360,{"effective_value":146,"effective_unit":128,"capacity_factor":133},114688,{"gpu_model":118,"gpu_model_id":119,"gpu_count":79,"total_vram_gb":41,"hour_price":120,"monthly_cost":148,"providers":149,"bandwidth_gbps":124,"fp16_tflops":119,"interconnect_score":125,"parallel_efficiency":133,"score":150,"tier":127,"perf_metric":128,"perf_label":151,"perf_value":152,"workload":16,"cost_metric":153},80.88,[123],0.4509,"35840 vec\u002Fs",35840,{"effective_value":154,"effective_unit":128,"capacity_factor":133},28672,[],false,"value",[159,166,171,179,182],{"name":160,"vram_gb":41,"cheapest_hour":161,"cost":162,"provider":123,"estimated_tps":43},"Tesla V100",0.0272,{"hourly":161,"daily":163,"monthly":164,"yearly":165},0.65,19.86,238.27,{"name":118,"vram_gb":46,"cheapest_hour":120,"cost":167,"provider":123,"estimated_tps":170},{"hourly":120,"daily":168,"monthly":121,"yearly":169},1.33,485.3,8960,{"name":172,"vram_gb":119,"cheapest_hour":173,"cost":174,"provider":123,"estimated_tps":178},"RTX 3090",0.0678,{"hourly":173,"daily":175,"monthly":176,"yearly":177},1.63,49.49,593.93,18720,{"name":180,"vram_gb":41,"cheapest_hour":173,"cost":181,"provider":123,"estimated_tps":43},"RTX 4070S Ti",{"hourly":173,"daily":175,"monthly":176,"yearly":177},{"name":183,"vram_gb":10,"cheapest_hour":184,"cost":185,"provider":123,"estimated_tps":189},"RTX 5070",0.0804,{"hourly":184,"daily":186,"monthly":187,"yearly":188},1.93,58.69,704.3,13440,{"posts_count":43,"benchmarks_count":43}]