[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-all-minilm-l6-v2":3},{"id":4,"slug":5,"name":6,"hf_id":6,"organization":7,"architecture":8,"architecture_config":9,"parameters":13,"parameters_label":14,"context_length":15,"license":16,"downloads":17,"likes":18,"status":19,"description":16,"huggingface_url":20,"created_at":21,"seo":22,"category":25,"family":28,"score":30,"variants":31,"quantizations":32,"requirements":59,"gpu_recommendations":111,"fitting_gpus":154,"community":189},430,"all-minilm-l6-v2","sentence-transformers\u002Fall-MiniLM-L6-v2","sentence-transformers","bert",{"layers":10,"head_dim":11,"hidden_size":12},6,32,384,22337280,"22.3M",512,null,254892100,5207,"published","https:\u002F\u002Fhuggingface.co\u002Fsentence-transformers\u002Fall-MiniLM-L6-v2","2026-08-13T18:00:16+00:00",{"title":6,"description":23,"image":16,"robots":24,"canonical_url":16},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":26,"slug":27},"Embedding \u002F RAG","embedding",{"name":29,"slug":7},"Sentence-transformers",0.4576,[],[33,38,43,48,52,56],{"format":34,"bits":35,"size_factor":36,"quality_loss":37},"AWQ",4,0.263,0.03,{"format":39,"bits":40,"size_factor":41,"quality_loss":42},"FP16",16,1,0,{"format":44,"bits":45,"size_factor":46,"quality_loss":47},"FP8",8,0.5,0.01,{"format":49,"bits":35,"size_factor":50,"quality_loss":51},"GGUF",0.25,0.05,{"format":53,"bits":35,"size_factor":54,"quality_loss":55},"GPTQ",0.275,0.04,{"format":57,"bits":45,"size_factor":46,"quality_loss":58},"INT8",0.02,{"FP16":60,"FP8":79,"INT8":89,"AWQ":93,"GPTQ":103,"GGUF":107},{"minimum":61,"production":68,"recommended":73},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":55,"kv_cache_gb":42,"vram_gb":64,"ram_gb":11,"disk_gb":65,"requirement_source":66,"confidence":67},"minimum",4096,1.27,39.06,"inferred",0.95,{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":55,"kv_cache_gb":42,"vram_gb":71,"ram_gb":11,"disk_gb":72,"requirement_source":66,"confidence":67},"production",16384,1.74,104.06,{"scenario":74,"context_length":75,"batch_size":76,"weight_gb":55,"kv_cache_gb":42,"vram_gb":77,"ram_gb":11,"disk_gb":78,"requirement_source":66,"confidence":67},"recommended",8192,2,1.44,65.06,{"minimum":80,"production":83,"recommended":86},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":58,"kv_cache_gb":42,"vram_gb":81,"ram_gb":11,"disk_gb":82,"requirement_source":66,"confidence":67},1.24,39.03,{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":58,"kv_cache_gb":42,"vram_gb":84,"ram_gb":11,"disk_gb":85,"requirement_source":66,"confidence":67},1.71,104.03,{"scenario":74,"context_length":75,"batch_size":76,"weight_gb":58,"kv_cache_gb":42,"vram_gb":87,"ram_gb":11,"disk_gb":88,"requirement_source":66,"confidence":67},1.41,65.03,{"minimum":90,"production":91,"recommended":92},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":58,"kv_cache_gb":42,"vram_gb":81,"ram_gb":11,"disk_gb":82,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":58,"kv_cache_gb":42,"vram_gb":84,"ram_gb":11,"disk_gb":85,"requirement_source":66,"confidence":67},{"scenario":74,"context_length":75,"batch_size":76,"weight_gb":58,"kv_cache_gb":42,"vram_gb":87,"ram_gb":11,"disk_gb":88,"requirement_source":66,"confidence":67},{"minimum":94,"production":97,"recommended":100},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":47,"kv_cache_gb":42,"vram_gb":95,"ram_gb":11,"disk_gb":96,"requirement_source":66,"confidence":67},1.23,39.02,{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":47,"kv_cache_gb":42,"vram_gb":98,"ram_gb":11,"disk_gb":99,"requirement_source":66,"confidence":67},1.7,104.02,{"scenario":74,"context_length":75,"batch_size":76,"weight_gb":47,"kv_cache_gb":42,"vram_gb":101,"ram_gb":11,"disk_gb":102,"requirement_source":66,"confidence":67},1.4,65.02,{"minimum":104,"production":105,"recommended":106},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":47,"kv_cache_gb":42,"vram_gb":95,"ram_gb":11,"disk_gb":96,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":47,"kv_cache_gb":42,"vram_gb":98,"ram_gb":11,"disk_gb":99,"requirement_source":66,"confidence":67},{"scenario":74,"context_length":75,"batch_size":76,"weight_gb":47,"kv_cache_gb":42,"vram_gb":101,"ram_gb":11,"disk_gb":102,"requirement_source":66,"confidence":67},{"minimum":108,"production":109,"recommended":110},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":47,"kv_cache_gb":42,"vram_gb":95,"ram_gb":11,"disk_gb":96,"requirement_source":66,"confidence":67},{"scenario":69,"context_length":70,"batch_size":35,"weight_gb":47,"kv_cache_gb":42,"vram_gb":98,"ram_gb":11,"disk_gb":99,"requirement_source":66,"confidence":67},{"scenario":74,"context_length":75,"batch_size":76,"weight_gb":47,"kv_cache_gb":42,"vram_gb":101,"ram_gb":11,"disk_gb":102,"requirement_source":66,"confidence":67},{"recommended":112,"cheapest":129,"performance":132,"min_complexity":143,"cluster_options":151,"no_match":152,"required_vram_gb":101,"candidate_count":35,"usage":153},{"gpu_model":113,"gpu_model_id":114,"gpu_count":41,"total_vram_gb":45,"hour_price":115,"monthly_cost":116,"providers":117,"bandwidth_gbps":119,"fp16_tflops":114,"interconnect_score":120,"parallel_efficiency":41,"score":121,"tier":122,"perf_metric":123,"perf_label":124,"perf_value":125,"workload":16,"cost_metric":126},"RTX 5060 Ti",24,0.0719,52.49,[118],"vast",448,40,0.5388,"cheapest","vectors\u002Fs","89600 vec\u002Fs",89600,{"effective_value":127,"effective_unit":123,"capacity_factor":128},71680,0.8,{"gpu_model":113,"gpu_model_id":114,"gpu_count":41,"total_vram_gb":45,"hour_price":115,"monthly_cost":116,"providers":130,"bandwidth_gbps":119,"fp16_tflops":114,"interconnect_score":120,"parallel_efficiency":41,"score":121,"tier":122,"perf_metric":123,"perf_label":124,"perf_value":125,"workload":16,"cost_metric":131},[118],{"effective_value":127,"effective_unit":123,"capacity_factor":128},{"gpu_model":113,"gpu_model_id":114,"gpu_count":45,"total_vram_gb":133,"hour_price":115,"monthly_cost":134,"providers":135,"bandwidth_gbps":119,"fp16_tflops":114,"interconnect_score":120,"parallel_efficiency":136,"score":137,"tier":138,"perf_metric":123,"perf_label":139,"perf_value":140,"workload":16,"cost_metric":141},64,419.9,[118],0.4,0.3615,"balanced","716800 vec\u002Fs",716800,{"effective_value":142,"effective_unit":123,"capacity_factor":128},573440,{"gpu_model":113,"gpu_model_id":114,"gpu_count":76,"total_vram_gb":40,"hour_price":115,"monthly_cost":144,"providers":145,"bandwidth_gbps":119,"fp16_tflops":114,"interconnect_score":120,"parallel_efficiency":128,"score":146,"tier":122,"perf_metric":123,"perf_label":147,"perf_value":148,"workload":16,"cost_metric":149},104.97,[118],0.4504,"179200 vec\u002Fs",179200,{"effective_value":150,"effective_unit":123,"capacity_factor":128},143360,[],false,"value",[155,162,170,177,184],{"name":156,"vram_gb":40,"cheapest_hour":157,"cost":158,"provider":118,"estimated_tps":42},"Tesla V100",0.0289,{"hourly":157,"daily":159,"monthly":160,"yearly":161},0.69,21.1,253.16,{"name":163,"vram_gb":114,"cheapest_hour":164,"cost":165,"provider":118,"estimated_tps":169},"RTX 3090",0.0622,{"hourly":164,"daily":166,"monthly":167,"yearly":168},1.49,45.41,544.87,93600,{"name":171,"vram_gb":40,"cheapest_hour":172,"cost":173,"provider":118,"estimated_tps":42},"RTX 4070S Ti",0.0678,{"hourly":172,"daily":174,"monthly":175,"yearly":176},1.63,49.49,593.93,{"name":178,"vram_gb":40,"cheapest_hour":179,"cost":180,"provider":118,"estimated_tps":42},"RTX 4080S",0.0685,{"hourly":179,"daily":181,"monthly":182,"yearly":183},1.64,50.01,600.06,{"name":113,"vram_gb":45,"cheapest_hour":115,"cost":185,"provider":118,"estimated_tps":188},{"hourly":115,"daily":186,"monthly":116,"yearly":187},1.73,629.84,44800,{"posts_count":42,"benchmarks_count":42}]