[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-nomic-embed-text-v1-5":3},{"id":4,"slug":5,"name":6,"hf_id":6,"organization":7,"architecture":8,"architecture_config":9,"parameters":13,"parameters_label":14,"context_length":15,"license":16,"downloads":17,"likes":18,"status":19,"description":16,"huggingface_url":20,"created_at":21,"seo":22,"category":25,"family":28,"score":30,"variants":31,"quantizations":32,"requirements":59,"gpu_recommendations":115,"fitting_gpus":157,"community":189},446,"nomic-embed-text-v1-5","nomic-ai\u002Fnomic-embed-text-v1.5","nomic-ai","nomic_bert",{"layers":10,"head_dim":11,"hidden_size":12},12,64,768,108380160,"108.4M",2048,null,15667197,891,"published","https:\u002F\u002Fhuggingface.co\u002Fnomic-ai\u002Fnomic-embed-text-v1.5","2026-08-13T18:02:06+00:00",{"title":6,"description":23,"image":16,"robots":24,"canonical_url":16},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":26,"slug":27},"Embedding \u002F RAG","embedding",{"name":29,"slug":7},"Nomic-ai",0.0344,[],[33,38,43,48,52,56],{"format":34,"bits":35,"size_factor":36,"quality_loss":37},"AWQ",4,0.263,0.03,{"format":39,"bits":40,"size_factor":41,"quality_loss":42},"FP16",16,1,0,{"format":44,"bits":45,"size_factor":46,"quality_loss":47},"FP8",8,0.5,0.01,{"format":49,"bits":35,"size_factor":50,"quality_loss":51},"GGUF",0.25,0.05,{"format":53,"bits":35,"size_factor":54,"quality_loss":55},"GPTQ",0.275,0.04,{"format":57,"bits":45,"size_factor":46,"quality_loss":58},"INT8",0.02,{"FP16":60,"FP8":81,"INT8":92,"AWQ":96,"GPTQ":106,"GGUF":111},{"minimum":61,"production":70,"recommended":75},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":64,"kv_cache_gb":42,"vram_gb":65,"ram_gb":66,"disk_gb":67,"requirement_source":68,"confidence":69},"minimum",4096,0.2,1.5,32,39.31,"inferred",0.95,{"scenario":71,"context_length":72,"batch_size":35,"weight_gb":64,"kv_cache_gb":42,"vram_gb":73,"ram_gb":66,"disk_gb":74,"requirement_source":68,"confidence":69},"production",16384,1.99,104.31,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":64,"kv_cache_gb":42,"vram_gb":79,"ram_gb":66,"disk_gb":80,"requirement_source":68,"confidence":69},"recommended",8192,2,1.68,65.31,{"minimum":82,"production":86,"recommended":89},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":83,"kv_cache_gb":42,"vram_gb":84,"ram_gb":66,"disk_gb":85,"requirement_source":68,"confidence":69},0.1,1.35,39.16,{"scenario":71,"context_length":72,"batch_size":35,"weight_gb":83,"kv_cache_gb":42,"vram_gb":87,"ram_gb":66,"disk_gb":88,"requirement_source":68,"confidence":69},1.84,104.16,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":83,"kv_cache_gb":42,"vram_gb":90,"ram_gb":66,"disk_gb":91,"requirement_source":68,"confidence":69},1.53,65.16,{"minimum":93,"production":94,"recommended":95},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":83,"kv_cache_gb":42,"vram_gb":84,"ram_gb":66,"disk_gb":85,"requirement_source":68,"confidence":69},{"scenario":71,"context_length":72,"batch_size":35,"weight_gb":83,"kv_cache_gb":42,"vram_gb":87,"ram_gb":66,"disk_gb":88,"requirement_source":68,"confidence":69},{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":83,"kv_cache_gb":42,"vram_gb":90,"ram_gb":66,"disk_gb":91,"requirement_source":68,"confidence":69},{"minimum":97,"production":100,"recommended":103},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":51,"kv_cache_gb":42,"vram_gb":98,"ram_gb":66,"disk_gb":99,"requirement_source":68,"confidence":69},1.28,39.08,{"scenario":71,"context_length":72,"batch_size":35,"weight_gb":51,"kv_cache_gb":42,"vram_gb":101,"ram_gb":66,"disk_gb":102,"requirement_source":68,"confidence":69},1.76,104.08,{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":51,"kv_cache_gb":42,"vram_gb":104,"ram_gb":66,"disk_gb":105,"requirement_source":68,"confidence":69},1.46,65.08,{"minimum":107,"production":109,"recommended":110},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":51,"kv_cache_gb":42,"vram_gb":108,"ram_gb":66,"disk_gb":99,"requirement_source":68,"confidence":69},1.29,{"scenario":71,"context_length":72,"batch_size":35,"weight_gb":51,"kv_cache_gb":42,"vram_gb":101,"ram_gb":66,"disk_gb":102,"requirement_source":68,"confidence":69},{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":51,"kv_cache_gb":42,"vram_gb":104,"ram_gb":66,"disk_gb":105,"requirement_source":68,"confidence":69},{"minimum":112,"production":113,"recommended":114},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":51,"kv_cache_gb":42,"vram_gb":108,"ram_gb":66,"disk_gb":99,"requirement_source":68,"confidence":69},{"scenario":71,"context_length":72,"batch_size":35,"weight_gb":51,"kv_cache_gb":42,"vram_gb":101,"ram_gb":66,"disk_gb":102,"requirement_source":68,"confidence":69},{"scenario":76,"context_length":77,"batch_size":78,"weight_gb":51,"kv_cache_gb":42,"vram_gb":104,"ram_gb":66,"disk_gb":105,"requirement_source":68,"confidence":69},{"recommended":116,"cheapest":133,"performance":136,"min_complexity":146,"cluster_options":154,"no_match":155,"required_vram_gb":104,"candidate_count":35,"usage":156},{"gpu_model":117,"gpu_model_id":118,"gpu_count":41,"total_vram_gb":45,"hour_price":119,"monthly_cost":120,"providers":121,"bandwidth_gbps":123,"fp16_tflops":118,"interconnect_score":124,"parallel_efficiency":41,"score":125,"tier":126,"perf_metric":127,"perf_label":128,"perf_value":129,"workload":16,"cost_metric":130},"RTX 5060 Ti",24,0.0554,40.44,[122],"vast",448,40,0.541,"cheapest","vectors\u002Fs","17920 vec\u002Fs",17920,{"effective_value":131,"effective_unit":127,"capacity_factor":132},14336,0.8,{"gpu_model":117,"gpu_model_id":118,"gpu_count":41,"total_vram_gb":45,"hour_price":119,"monthly_cost":120,"providers":134,"bandwidth_gbps":123,"fp16_tflops":118,"interconnect_score":124,"parallel_efficiency":41,"score":125,"tier":126,"perf_metric":127,"perf_label":128,"perf_value":129,"workload":16,"cost_metric":135},[122],{"effective_value":131,"effective_unit":127,"capacity_factor":132},{"gpu_model":117,"gpu_model_id":118,"gpu_count":45,"total_vram_gb":11,"hour_price":119,"monthly_cost":137,"providers":138,"bandwidth_gbps":123,"fp16_tflops":118,"interconnect_score":124,"parallel_efficiency":139,"score":140,"tier":141,"perf_metric":127,"perf_label":142,"perf_value":143,"workload":16,"cost_metric":144},323.54,[122],0.4,0.3635,"balanced","143360 vec\u002Fs",143360,{"effective_value":145,"effective_unit":127,"capacity_factor":132},114688,{"gpu_model":117,"gpu_model_id":118,"gpu_count":78,"total_vram_gb":40,"hour_price":119,"monthly_cost":147,"providers":148,"bandwidth_gbps":123,"fp16_tflops":118,"interconnect_score":124,"parallel_efficiency":132,"score":149,"tier":126,"perf_metric":127,"perf_label":150,"perf_value":151,"workload":16,"cost_metric":152},80.88,[122],0.4509,"35840 vec\u002Fs",35840,{"effective_value":153,"effective_unit":127,"capacity_factor":132},28672,[],false,"value",[158,165,170,178,181],{"name":159,"vram_gb":40,"cheapest_hour":160,"cost":161,"provider":122,"estimated_tps":42},"Tesla V100",0.0272,{"hourly":160,"daily":162,"monthly":163,"yearly":164},0.65,19.86,238.27,{"name":117,"vram_gb":45,"cheapest_hour":119,"cost":166,"provider":122,"estimated_tps":169},{"hourly":119,"daily":167,"monthly":120,"yearly":168},1.33,485.3,8960,{"name":171,"vram_gb":118,"cheapest_hour":172,"cost":173,"provider":122,"estimated_tps":177},"RTX 3090",0.0678,{"hourly":172,"daily":174,"monthly":175,"yearly":176},1.63,49.49,593.93,18720,{"name":179,"vram_gb":40,"cheapest_hour":172,"cost":180,"provider":122,"estimated_tps":42},"RTX 4070S Ti",{"hourly":172,"daily":174,"monthly":175,"yearly":176},{"name":182,"vram_gb":10,"cheapest_hour":183,"cost":184,"provider":122,"estimated_tps":188},"RTX 5070",0.0804,{"hourly":183,"daily":185,"monthly":186,"yearly":187},1.93,58.69,704.3,13440,{"posts_count":42,"benchmarks_count":42}]