[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"model-qwen3-embedding-0-6b":3},{"id":4,"slug":5,"name":6,"hf_id":6,"organization":7,"architecture":8,"architecture_config":9,"parameters":14,"parameters_label":15,"context_length":16,"license":17,"downloads":18,"likes":19,"status":20,"description":17,"huggingface_url":21,"created_at":22,"seo":23,"category":26,"family":29,"score":31,"variants":32,"quantizations":33,"requirements":59,"gpu_recommendations":123,"fitting_gpus":166,"community":201},472,"qwen3-embedding-0-6b","Qwen\u002FQwen3-Embedding-0.6B","Qwen","qwen3",{"layers":10,"head_dim":11,"kv_heads":12,"hidden_size":13},28,128,8,1024,507630592,"507.6M",32768,null,8074610,1151,"published","https:\u002F\u002Fhuggingface.co\u002FQwen\u002FQwen3-Embedding-0.6B","2026-08-13T18:06:36+00:00",{"title":6,"description":24,"image":17,"robots":25,"canonical_url":17},"AI 模型部署决策引擎——开源 AI 的 GPU 需求、云端价格与成本分析。","index, follow",{"name":27,"slug":28},"Embedding \u002F RAG","embedding",{"name":7,"slug":30},"qwen",0.0254,[],[34,39,44,48,52,56],{"format":35,"bits":36,"size_factor":37,"quality_loss":38},"AWQ",4,0.263,0.03,{"format":40,"bits":41,"size_factor":42,"quality_loss":43},"FP16",16,1,0,{"format":45,"bits":12,"size_factor":46,"quality_loss":47},"FP8",0.5,0.01,{"format":49,"bits":36,"size_factor":50,"quality_loss":51},"GGUF",0.25,0.05,{"format":53,"bits":36,"size_factor":54,"quality_loss":55},"GPTQ",0.275,0.04,{"format":57,"bits":12,"size_factor":46,"quality_loss":58},"INT8",0.02,{"FP16":60,"FP8":80,"INT8":91,"AWQ":95,"GPTQ":105,"GGUF":113},{"minimum":61,"production":69,"recommended":74},{"scenario":62,"context_length":63,"batch_size":42,"weight_gb":64,"kv_cache_gb":43,"vram_gb":65,"ram_gb":66,"disk_gb":67,"requirement_source":68,"confidence":64},"minimum",4096,0.95,2.56,32,40.48,"inferred",{"scenario":70,"context_length":71,"batch_size":36,"weight_gb":64,"kv_cache_gb":43,"vram_gb":72,"ram_gb":66,"disk_gb":73,"requirement_source":68,"confidence":64},"production",16384,3.16,105.48,{"scenario":75,"context_length":76,"batch_size":77,"weight_gb":64,"kv_cache_gb":43,"vram_gb":78,"ram_gb":66,"disk_gb":79,"requirement_source":68,"confidence":64},"recommended",8192,2,2.79,66.48,{"minimum":81,"production":85,"recommended":88},{"scenario":62,"context_length":63,"batch_size":42,"weight_gb":82,"kv_cache_gb":43,"vram_gb":83,"ram_gb":66,"disk_gb":84,"requirement_source":68,"confidence":64},0.47,1.89,39.74,{"scenario":70,"context_length":71,"batch_size":36,"weight_gb":82,"kv_cache_gb":43,"vram_gb":86,"ram_gb":66,"disk_gb":87,"requirement_source":68,"confidence":64},2.42,104.74,{"scenario":75,"context_length":76,"batch_size":77,"weight_gb":82,"kv_cache_gb":43,"vram_gb":89,"ram_gb":66,"disk_gb":90,"requirement_source":68,"confidence":64},2.09,65.74,{"minimum":92,"production":93,"recommended":94},{"scenario":62,"context_length":63,"batch_size":42,"weight_gb":82,"kv_cache_gb":43,"vram_gb":83,"ram_gb":66,"disk_gb":84,"requirement_source":68,"confidence":64},{"scenario":70,"context_length":71,"batch_size":36,"weight_gb":82,"kv_cache_gb":43,"vram_gb":86,"ram_gb":66,"disk_gb":87,"requirement_source":68,"confidence":64},{"scenario":75,"context_length":76,"batch_size":77,"weight_gb":82,"kv_cache_gb":43,"vram_gb":89,"ram_gb":66,"disk_gb":90,"requirement_source":68,"confidence":64},{"minimum":96,"production":99,"recommended":102},{"scenario":62,"context_length":63,"batch_size":42,"weight_gb":50,"kv_cache_gb":43,"vram_gb":97,"ram_gb":66,"disk_gb":98,"requirement_source":68,"confidence":64},1.56,39.38,{"scenario":70,"context_length":71,"batch_size":36,"weight_gb":50,"kv_cache_gb":43,"vram_gb":100,"ram_gb":66,"disk_gb":101,"requirement_source":68,"confidence":64},2.06,104.38,{"scenario":75,"context_length":76,"batch_size":77,"weight_gb":50,"kv_cache_gb":43,"vram_gb":103,"ram_gb":66,"disk_gb":104,"requirement_source":68,"confidence":64},1.75,65.38,{"minimum":106,"production":108,"recommended":111},{"scenario":62,"context_length":63,"batch_size":42,"weight_gb":50,"kv_cache_gb":43,"vram_gb":97,"ram_gb":66,"disk_gb":107,"requirement_source":68,"confidence":64},39.39,{"scenario":70,"context_length":71,"batch_size":36,"weight_gb":50,"kv_cache_gb":43,"vram_gb":109,"ram_gb":66,"disk_gb":110,"requirement_source":68,"confidence":64},2.07,104.39,{"scenario":75,"context_length":76,"batch_size":77,"weight_gb":50,"kv_cache_gb":43,"vram_gb":103,"ram_gb":66,"disk_gb":112,"requirement_source":68,"confidence":64},65.39,{"minimum":114,"production":117,"recommended":120},{"scenario":62,"context_length":63,"batch_size":42,"weight_gb":50,"kv_cache_gb":43,"vram_gb":115,"ram_gb":66,"disk_gb":116,"requirement_source":68,"confidence":64},1.57,39.4,{"scenario":70,"context_length":71,"batch_size":36,"weight_gb":50,"kv_cache_gb":43,"vram_gb":118,"ram_gb":66,"disk_gb":119,"requirement_source":68,"confidence":64},2.08,104.4,{"scenario":75,"context_length":76,"batch_size":77,"weight_gb":50,"kv_cache_gb":43,"vram_gb":121,"ram_gb":66,"disk_gb":122,"requirement_source":68,"confidence":64},1.76,65.4,{"recommended":124,"cheapest":141,"performance":144,"min_complexity":155,"cluster_options":163,"no_match":164,"required_vram_gb":103,"candidate_count":36,"usage":165},{"gpu_model":125,"gpu_model_id":126,"gpu_count":42,"total_vram_gb":12,"hour_price":127,"monthly_cost":128,"providers":129,"bandwidth_gbps":131,"fp16_tflops":126,"interconnect_score":132,"parallel_efficiency":42,"score":133,"tier":134,"perf_metric":135,"perf_label":136,"perf_value":137,"workload":17,"cost_metric":138},"RTX 5060 Ti",24,0.0719,52.49,[130],"vast",448,40,0.5504,"cheapest","vectors\u002Fs","3584 vec\u002Fs",3584,{"effective_value":139,"effective_unit":135,"capacity_factor":140},2867.2,0.8,{"gpu_model":125,"gpu_model_id":126,"gpu_count":42,"total_vram_gb":12,"hour_price":127,"monthly_cost":128,"providers":142,"bandwidth_gbps":131,"fp16_tflops":126,"interconnect_score":132,"parallel_efficiency":42,"score":133,"tier":134,"perf_metric":135,"perf_label":136,"perf_value":137,"workload":17,"cost_metric":143},[130],{"effective_value":139,"effective_unit":135,"capacity_factor":140},{"gpu_model":125,"gpu_model_id":126,"gpu_count":12,"total_vram_gb":145,"hour_price":127,"monthly_cost":146,"providers":147,"bandwidth_gbps":131,"fp16_tflops":126,"interconnect_score":132,"parallel_efficiency":148,"score":149,"tier":150,"perf_metric":135,"perf_label":151,"perf_value":152,"workload":17,"cost_metric":153},64,419.9,[130],0.4,0.3615,"balanced","28672 vec\u002Fs",28672,{"effective_value":154,"effective_unit":135,"capacity_factor":140},22937.6,{"gpu_model":125,"gpu_model_id":126,"gpu_count":77,"total_vram_gb":41,"hour_price":127,"monthly_cost":156,"providers":157,"bandwidth_gbps":131,"fp16_tflops":126,"interconnect_score":132,"parallel_efficiency":140,"score":158,"tier":134,"perf_metric":135,"perf_label":159,"perf_value":160,"workload":17,"cost_metric":161},104.97,[130],0.4504,"7168 vec\u002Fs",7168,{"effective_value":162,"effective_unit":135,"capacity_factor":140},5734.4,[],false,"value",[167,174,182,189,196],{"name":168,"vram_gb":41,"cheapest_hour":169,"cost":170,"provider":130,"estimated_tps":43},"Tesla V100",0.0289,{"hourly":169,"daily":171,"monthly":172,"yearly":173},0.69,21.1,253.16,{"name":175,"vram_gb":126,"cheapest_hour":176,"cost":177,"provider":130,"estimated_tps":181},"RTX 3090",0.0622,{"hourly":176,"daily":178,"monthly":179,"yearly":180},1.49,45.41,544.87,3744,{"name":183,"vram_gb":41,"cheapest_hour":184,"cost":185,"provider":130,"estimated_tps":43},"RTX 4070S Ti",0.0678,{"hourly":184,"daily":186,"monthly":187,"yearly":188},1.63,49.49,593.93,{"name":190,"vram_gb":41,"cheapest_hour":191,"cost":192,"provider":130,"estimated_tps":43},"RTX 4080S",0.0685,{"hourly":191,"daily":193,"monthly":194,"yearly":195},1.64,50.01,600.06,{"name":125,"vram_gb":12,"cheapest_hour":127,"cost":197,"provider":130,"estimated_tps":200},{"hourly":127,"daily":198,"monthly":128,"yearly":199},1.73,629.84,1792,{"posts_count":43,"benchmarks_count":43}]