[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-nomic-embed-text-v1":3},{"id":4,"slug":5,"name":6,"hf_id":6,"organization":7,"architecture":8,"architecture_config":9,"parameters":13,"parameters_label":14,"context_length":15,"license":16,"downloads":17,"likes":18,"status":19,"description":16,"huggingface_url":20,"created_at":21,"seo":22,"category":25,"family":28,"score":30,"variants":31,"quantizations":32,"requirements":59,"gpu_recommendations":114,"fitting_gpus":156,"community":188},495,"nomic-embed-text-v1","nomic-ai\u002Fnomic-embed-text-v1","nomic-ai","nomic_bert",{"layers":10,"head_dim":11,"hidden_size":12},12,64,768,108380160,"108.4M",8192,null,4686997,580,"published","https:\u002F\u002Fhuggingface.co\u002Fnomic-ai\u002Fnomic-embed-text-v1","2026-08-13T18:11:30+00:00",{"title":6,"description":23,"image":16,"robots":24,"canonical_url":16},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":26,"slug":27},"Embedding \u002F RAG","embedding",{"name":29,"slug":7},"Nomic-ai",0.0138,[],[33,38,43,48,52,56],{"format":34,"bits":35,"size_factor":36,"quality_loss":37},"AWQ",4,0.263,0.03,{"format":39,"bits":40,"size_factor":41,"quality_loss":42},"FP16",16,1,0,{"format":44,"bits":45,"size_factor":46,"quality_loss":47},"FP8",8,0.5,0.01,{"format":49,"bits":35,"size_factor":50,"quality_loss":51},"GGUF",0.25,0.05,{"format":53,"bits":35,"size_factor":54,"quality_loss":55},"GPTQ",0.275,0.04,{"format":57,"bits":45,"size_factor":46,"quality_loss":58},"INT8",0.02,{"FP16":60,"FP8":80,"INT8":91,"AWQ":95,"GPTQ":105,"GGUF":110},{"minimum":61,"production":70,"recommended":75},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":64,"kv_cache_gb":42,"vram_gb":65,"ram_gb":66,"disk_gb":67,"requirement_source":68,"confidence":69},"minimum",4096,0.2,1.5,32,39.31,"inferred",0.95,{"scenario":71,"context_length":72,"batch_size":35,"weight_gb":64,"kv_cache_gb":42,"vram_gb":73,"ram_gb":66,"disk_gb":74,"requirement_source":68,"confidence":69},"production",16384,1.99,104.31,{"scenario":76,"context_length":15,"batch_size":77,"weight_gb":64,"kv_cache_gb":42,"vram_gb":78,"ram_gb":66,"disk_gb":79,"requirement_source":68,"confidence":69},"recommended",2,1.68,65.31,{"minimum":81,"production":85,"recommended":88},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":82,"kv_cache_gb":42,"vram_gb":83,"ram_gb":66,"disk_gb":84,"requirement_source":68,"confidence":69},0.1,1.35,39.16,{"scenario":71,"context_length":72,"batch_size":35,"weight_gb":82,"kv_cache_gb":42,"vram_gb":86,"ram_gb":66,"disk_gb":87,"requirement_source":68,"confidence":69},1.84,104.16,{"scenario":76,"context_length":15,"batch_size":77,"weight_gb":82,"kv_cache_gb":42,"vram_gb":89,"ram_gb":66,"disk_gb":90,"requirement_source":68,"confidence":69},1.53,65.16,{"minimum":92,"production":93,"recommended":94},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":82,"kv_cache_gb":42,"vram_gb":83,"ram_gb":66,"disk_gb":84,"requirement_source":68,"confidence":69},{"scenario":71,"context_length":72,"batch_size":35,"weight_gb":82,"kv_cache_gb":42,"vram_gb":86,"ram_gb":66,"disk_gb":87,"requirement_source":68,"confidence":69},{"scenario":76,"context_length":15,"batch_size":77,"weight_gb":82,"kv_cache_gb":42,"vram_gb":89,"ram_gb":66,"disk_gb":90,"requirement_source":68,"confidence":69},{"minimum":96,"production":99,"recommended":102},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":51,"kv_cache_gb":42,"vram_gb":97,"ram_gb":66,"disk_gb":98,"requirement_source":68,"confidence":69},1.28,39.08,{"scenario":71,"context_length":72,"batch_size":35,"weight_gb":51,"kv_cache_gb":42,"vram_gb":100,"ram_gb":66,"disk_gb":101,"requirement_source":68,"confidence":69},1.76,104.08,{"scenario":76,"context_length":15,"batch_size":77,"weight_gb":51,"kv_cache_gb":42,"vram_gb":103,"ram_gb":66,"disk_gb":104,"requirement_source":68,"confidence":69},1.46,65.08,{"minimum":106,"production":108,"recommended":109},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":51,"kv_cache_gb":42,"vram_gb":107,"ram_gb":66,"disk_gb":98,"requirement_source":68,"confidence":69},1.29,{"scenario":71,"context_length":72,"batch_size":35,"weight_gb":51,"kv_cache_gb":42,"vram_gb":100,"ram_gb":66,"disk_gb":101,"requirement_source":68,"confidence":69},{"scenario":76,"context_length":15,"batch_size":77,"weight_gb":51,"kv_cache_gb":42,"vram_gb":103,"ram_gb":66,"disk_gb":104,"requirement_source":68,"confidence":69},{"minimum":111,"production":112,"recommended":113},{"scenario":62,"context_length":63,"batch_size":41,"weight_gb":51,"kv_cache_gb":42,"vram_gb":107,"ram_gb":66,"disk_gb":98,"requirement_source":68,"confidence":69},{"scenario":71,"context_length":72,"batch_size":35,"weight_gb":51,"kv_cache_gb":42,"vram_gb":100,"ram_gb":66,"disk_gb":101,"requirement_source":68,"confidence":69},{"scenario":76,"context_length":15,"batch_size":77,"weight_gb":51,"kv_cache_gb":42,"vram_gb":103,"ram_gb":66,"disk_gb":104,"requirement_source":68,"confidence":69},{"recommended":115,"cheapest":132,"performance":135,"min_complexity":145,"cluster_options":153,"no_match":154,"required_vram_gb":103,"candidate_count":35,"usage":155},{"gpu_model":116,"gpu_model_id":117,"gpu_count":41,"total_vram_gb":45,"hour_price":118,"monthly_cost":119,"providers":120,"bandwidth_gbps":122,"fp16_tflops":117,"interconnect_score":123,"parallel_efficiency":41,"score":124,"tier":125,"perf_metric":126,"perf_label":127,"perf_value":128,"workload":16,"cost_metric":129},"RTX 5060 Ti",24,0.0554,40.44,[121],"vast",448,40,0.541,"cheapest","vectors\u002Fs","17920 vec\u002Fs",17920,{"effective_value":130,"effective_unit":126,"capacity_factor":131},14336,0.8,{"gpu_model":116,"gpu_model_id":117,"gpu_count":41,"total_vram_gb":45,"hour_price":118,"monthly_cost":119,"providers":133,"bandwidth_gbps":122,"fp16_tflops":117,"interconnect_score":123,"parallel_efficiency":41,"score":124,"tier":125,"perf_metric":126,"perf_label":127,"perf_value":128,"workload":16,"cost_metric":134},[121],{"effective_value":130,"effective_unit":126,"capacity_factor":131},{"gpu_model":116,"gpu_model_id":117,"gpu_count":45,"total_vram_gb":11,"hour_price":118,"monthly_cost":136,"providers":137,"bandwidth_gbps":122,"fp16_tflops":117,"interconnect_score":123,"parallel_efficiency":138,"score":139,"tier":140,"perf_metric":126,"perf_label":141,"perf_value":142,"workload":16,"cost_metric":143},323.54,[121],0.4,0.3635,"balanced","143360 vec\u002Fs",143360,{"effective_value":144,"effective_unit":126,"capacity_factor":131},114688,{"gpu_model":116,"gpu_model_id":117,"gpu_count":77,"total_vram_gb":40,"hour_price":118,"monthly_cost":146,"providers":147,"bandwidth_gbps":122,"fp16_tflops":117,"interconnect_score":123,"parallel_efficiency":131,"score":148,"tier":125,"perf_metric":126,"perf_label":149,"perf_value":150,"workload":16,"cost_metric":151},80.88,[121],0.4509,"35840 vec\u002Fs",35840,{"effective_value":152,"effective_unit":126,"capacity_factor":131},28672,[],false,"value",[157,164,169,177,180],{"name":158,"vram_gb":40,"cheapest_hour":159,"cost":160,"provider":121,"estimated_tps":42},"Tesla V100",0.0272,{"hourly":159,"daily":161,"monthly":162,"yearly":163},0.65,19.86,238.27,{"name":116,"vram_gb":45,"cheapest_hour":118,"cost":165,"provider":121,"estimated_tps":168},{"hourly":118,"daily":166,"monthly":119,"yearly":167},1.33,485.3,8960,{"name":170,"vram_gb":117,"cheapest_hour":171,"cost":172,"provider":121,"estimated_tps":176},"RTX 3090",0.0678,{"hourly":171,"daily":173,"monthly":174,"yearly":175},1.63,49.49,593.93,18720,{"name":178,"vram_gb":40,"cheapest_hour":171,"cost":179,"provider":121,"estimated_tps":42},"RTX 4070S Ti",{"hourly":171,"daily":173,"monthly":174,"yearly":175},{"name":181,"vram_gb":10,"cheapest_hour":182,"cost":183,"provider":121,"estimated_tps":187},"RTX 5070",0.0804,{"hourly":182,"daily":184,"monthly":185,"yearly":186},1.93,58.69,704.3,13440,{"posts_count":42,"benchmarks_count":42}]