<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">
  <url>
    <loc>https://aideploy.com.cn/</loc>
    <changefreq>daily</changefreq>
    <priority>1.0</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/</loc>
    <changefreq>daily</changefreq>
    <priority>0.9</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/about/</loc>
    <changefreq>monthly</changefreq>
    <priority>0.6</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/faq/</loc>
    <changefreq>monthly</changefreq>
    <priority>0.5</priority>
  </url>
  
  <url>
    <loc>https://aideploy.com.cn/blog/ai-推理平台-2025-年综合排名国内用户如何选择-vllmreplicate-与-modal.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-推理平台排行榜基于吞吐量成本与易用性的-2025-年综合评分.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-推理平台的供应商锁定风险评估如何设计可迁移的部署架构.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-推理平台的性能基准测试框架构建可重复可比较的评测标准.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-推理平台的技术支持质量横评工单响应社区论坛与文档更新频率.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-推理平台的灾难恢复演练模拟区域故障时的切换与恢复流程.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-推理平台的退出策略如何将模型和数据从平台无缝迁移.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-推理平台选型决策树根据模型大小qps-与预算快速锁定方案.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-推理延迟优化全景从网络序列化到推理引擎的每一毫秒.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-推理请求的排队与批处理优化如何在延迟和吞吐之间取得平衡.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-推理请求的缓存策略语义缓存精确匹配缓存与结果预计算.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-模型-ab-测试部署架构在-vllm-后端实现流量分割与金丝雀发布.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-模型部署中的合规性检查数据驻留gdpr-与个人信息保护法.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-模型部署中的成本归因如何按部门项目或-api-key-拆分账单.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-模型部署中的模型加密与知识产权保护方案.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-模型部署中的流量预测与容量规划基于历史数据的自动扩缩容.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-模型部署安全清单api-鉴权速率限制与模型防盗用策略.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-模型部署对比裸金属kubernetesserverless-三种架构的适用场景.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-模型部署的-mock-测试如何在无-gpu-环境下测试-api-逻辑.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-模型部署的容量预留策略如何保证大促期间的推理资源.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/ai-部署-saas-平台评估清单安全合规sla-与技术支持怎么考.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-a-full-spectrum-of-cold-start-mitigation-strategies-for-serverless-inference-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-a-performance-benchmarking-framework-for-ai-inference-platforms-building-repe.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-ab-testing-deployment-architecture-for-ai-models-traffic-splitting-and-canary.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-ai-deployment-saas-evaluation-checklist-security-compliance-sla-and-technical.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-ai-inference-platform-decision-tree-quickly-lock-in-a-solution-by-model-size-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-ai-inference-platform-leaderboard-a-2025-composite-score-based-on-throughput-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-ai-inference-platform-rankings-2025-vllm-vs-replicate-vs-modal-for-global-tea.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-ai-model-deployment-comparison-bare-metal-kubernetes-and-serverless-architect.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-ai-model-deployment-security-checklist-api-authentication-rate-limiting-and-m.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-api-cost-accounting-by-call-volume-comparing-openai-replicate-and-self-hosted.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-api-fying-open-source-models-building-an-openai-compatible-endpoint-with-vllm.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-api-rate-limiting-for-self-hosted-inference-services-token-bucket-sliding-win.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-api-version-management-for-self-hosted-inference-iterating-without-breaking-c.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-auto-generating-api-documentation-for-self-hosted-inference-an-implementation.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-auto-scaling-a-self-hosted-inference-cluster-implementation-with-kubernetes-a.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-backup-and-disaster-recovery-for-self-hosted-inference-high-availability-desi.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-benchmarking-methodology-for-vllm-deployments-performance-evaluation-with-sha.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-building-a-model-inference-api-from-scratch-best-practices-with-docker-fastap.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-building-a-self-hosted-inference-server-from-bare-metal-setup-to-vllm-service.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-caching-strategies-for-ai-inference-requests-semantic-cache-exact-match-cache.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-capacity-reservation-strategies-for-ai-model-deployment-ensuring-inference-re.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-carbon-emissions-considerations-for-gpu-cloud-model-deployment-strategies-for.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-cicd-pipelines-for-self-hosted-inference-services-achieving-zero-downtime-mod.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-common-vllm-deployment-errors-and-fixes-oom-cuda-version-conflicts-and-token-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-compliance-and-audit-in-gpu-cloud-selection-soc2-iso27001-and-global-certific.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-compliance-in-ai-model-deployment-data-residency-gdpr-and-global-privacy-regu.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-container-orchestration-for-vllm-deployment-kubernetes-deployment-service-and.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-containerization-best-practices-for-vllm-multi-stage-builds-non-root-users-an.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-cost-attribution-in-ai-model-deployment-splitting-bills-by-department-project.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-cpu-and-memory-requirements-for-vllm-deployment-what-resources-are-needed-bey.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-cross-cloud-price-comparison-tools-for-gpu-rental-one-click-comparison-of-aws.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-custom-container-deployment-on-modal-running-non-python-inference-services.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-dependency-management-for-vllm-deployment-version-locking-strategies-with-poe.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-deploying-qwen-25-with-vllm-a-step-by-step-tutorial-from-weight-download-to-o.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-disaster-recovery-drills-for-ai-inference-platforms-simulating-regional-failu.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-distributed-inference-on-modal-processing-large-batches-in-parallel-using-the.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-exit-strategy-for-ai-inference-platforms-seamlessly-migrating-models-and-data.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-fault-recovery-mechanisms-for-vllm-deployments-health-checks-auto-restart-and.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-fp8-quantization-on-h100-with-vllm-in-practice-the-trade-off-between-throughp.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-from-docker-to-production-api-building-a-horizontally-scalable-model-inferenc.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-from-jupyter-notebook-to-production-api-bridging-the-engineering-gap-in-model.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-gpu-cloud-bill-analysis-and-optimization-finding-idle-resources-duplicate-sto.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-gpu-cloud-contracts-and-negotiation-how-to-secure-discounts-and-dedicated-sup.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-gpu-cloud-hidden-costs-revealed-data-transfer-storage-snapshots-and-static-ip.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-gpu-cloud-network-bandwidth-deep-dive-the-real-impact-of-cross-region-inferen.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-gpu-cloud-provider-sla-comparison-uptime-guarantees-compensation-mechanisms-a.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-gpu-cloud-service-selection-comparing-on-demand-reserved-and-spot-instance-co.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-gpu-rental-long-term-contract-vs-on-demand-a-cost-simulator-for-stable-infere.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-gpu-rental-market-outlook-2025-cost-efficiency-analysis-of-h100-b200-and-emer.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-gpu-rental-pitfalls-to-avoid-spot-instance-preemption-regional-stock-and-perf.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-gpu-rental-vs-serverless-cost-calculation-real-hourly-expenses-from-a100-to-h.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-gpu-temperature-and-power-monitoring-for-self-hosted-inference-a-prometheus-n.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-gpu-virtualization-for-self-hosted-inference-mig-vgpu-and-time-sharing-techno.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-handling-traffic-spikes-with-serverless-inference-cold-start-pools-reserved-c.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-hot-model-reloading-for-self-hosted-inference-switching-lora-or-base-models-w.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-build-a-cost-dashboard-for-ai-inference-tracking-spending-per-model-an.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-build-a-multi-model-unified-api-gateway-with-vllm-and-litellm.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-build-a-streaming-inference-endpoint-with-vllm-and-fastapi-sse-and-web.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-build-an-openai-fully-compatible-api-gateway-for-open-source-models.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-choose-a-deployment-region-latency-tests-from-north-america-europe-and.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-choose-an-inference-framework-for-open-source-llms-comparing-vllm-tgi-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-choose-an-overseas-gpu-cloud-a-horizontal-review-of-runpod-lambda-labs.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-deploy-code-generation-models-with-vllm-fim-inference-configuration-fo.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-deploy-embedding-and-reranking-models-for-rag-applications.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-deploy-embedding-models-with-vllm-building-text-vectorization-services.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-deploy-inference-services-for-edge-devices-model-adaptation-from-cloud.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-deploy-multimodal-models-with-vllm-inference-service-configuration-for.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-deploy-open-source-models-to-production-a-practical-handbook-covering-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-deploy-private-ai-inference-services-for-regulated-industries-like-hea.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-deploy-speech-recognition-models-with-vllm-streaming-and-batch-inferen.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-design-inference-infrastructure-for-agent-applications-tool-calling-mu.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-design-isolation-and-billing-for-multi-tenant-saas-inference-services.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-evaluate-the-price-performance-ratio-of-ai-inference-platforms-a-compo.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-evaluate-the-total-cost-of-ownership-for-model-deployment-hardware-ban.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-how-to-speed-up-rag-pipelines-by-deploying-embedding-and-reranking-models-wit.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-hybrid-architecture-of-serverless-and-container-deployments-when-to-shift-tra.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-image-registry-management-for-self-hosted-inference-integrating-harbor-ecr-an.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-ip-whitelisting-and-firewalling-for-serverless-gpu-platforms-security-practic.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-log-management-for-self-hosted-inference-clusters-applying-elk-loki-and-cloud.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-long-term-stability-test-of-serverless-gpu-platforms-a-7-day-uninterrupted-in.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-lora-hot-loading-on-modal-building-cost-effective-multi-tenant-model-microser.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-low-price-strategies-of-serverless-gpu-platforms-free-tiers-sign-up-credits-a.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-mock-testing-for-ai-model-deployment-testing-api-logic-without-a-gpu-environm.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-modal-cold-start-optimization-reducing-time-to-first-byte-with-warm-container.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-modal-cron-job-feature-automating-periodic-model-evaluation-with-serverless.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-modal-cross-region-deployment-serving-traffic-simultaneously-from-multiple-gl.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-modal-environment-variables-and-secrets-management-securely-injecting-api-key.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-modal-gpu-memory-limits-and-oom-handling-gracefully-catching-and-retrying.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-modal-gpu-model-selection-performance-pricing-and-use-cases-from-t4-to-h100.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-modal-gpu-time-slice-scheduling-how-short-tasks-avoid-queuing-and-complete-qu.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-modal-parallel-execution-model-achieving-hundreds-of-concurrent-inferences-wi.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-modal-real-time-log-streaming-and-debugging-quickly-locating-anomalies-in-inf.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-modal-review-the-pros-and-cons-of-a-python-native-serverless-platform-for-ai-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-modal-scheduled-tasks-and-workflows-building-an-automated-pipeline-for-daily-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-modal-secrets-management-and-environment-injection-standard-methods-for-passi.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-modal-storage-volume-performance-tuning-best-configuration-for-readwrite-band.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-modal-volume-snapshots-explained-reducing-model-loading-time-from-minutes-to-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-modal-vs-aws-lambda-gpu-choosing-serverless-inference-in-the-python-ecosystem.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-modal-vs-replicate-developer-experience-documentation-quality-sdk-usability-a.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-model-deployment-cost-control-handbook-quantization-caching-and-request-batch.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-model-encryption-and-intellectual-property-protection-in-ai-model-deployment.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-monitoring-and-observability-for-vllm-deployments-prometheus-metrics-grafana-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-multi-user-isolation-for-vllm-deployment-namespaces-resource-quotas-and-reque.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-network-configuration-for-vllm-deployment-load-balancing-tls-termination-and-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-night-and-weekend-discounts-for-gpu-rental-cutting-batch-inference-costs-usin.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-open-source-llm-production-deployment-a-full-guide-from-docker-image-to-api-e.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-optimizing-ai-inference-request-queuing-and-batching-balancing-latency-and-th.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-prometheus-exporter-configuration-for-vllm-which-metrics-to-expose-and-how-to.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-regional-stock-issues-in-gpu-cloud-selection-alternatives-when-the-target-gpu.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-api-rate-limiting-and-retry-strategies-best-practices-for-building-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-cog-tool-in-practice-packaging-any-python-model-into-a-production-g.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-model-analytics-dashboard-interpreting-call-volume-latency-distribu.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-model-cards-and-documentation-writing-high-quality-model-descriptio.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-model-deprecation-and-sunset-policy-how-to-handle-the-sudden-unavai.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-model-hotfix-updating-model-weights-without-service-downtime.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-model-marketplace-analysis-which-public-models-are-ready-for-produc.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-model-privacy-settings-public-private-and-unlisted-visibility-expla.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-model-security-scanning-ensuring-public-models-are-free-of-maliciou.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-model-sharing-and-team-collaboration-managing-model-access-within-a.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-model-usage-analytics-optimizing-call-patterns-through-api-log-anal.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-model-versioning-and-rollback-safely-updating-models-in-a-productio.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-pricing-model-fully-explained-per-second-billing-cold-starts-and-da.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-private-endpoint-feature-securing-data-transmission-via-vpc-peering.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-public-models-vs-private-deployment-pricing-when-to-migrate-from-ap.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-training-and-fine-tuning-review-cost-and-speed-of-lora-training-on-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-user-guide-how-to-package-and-publish-custom-models-using-cog.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-vs-runpod-cost-comparison-monthly-bill-simulation-for-the-same-mode.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-replicate-webhooks-and-asynchronous-inference-building-event-driven-ai-workfl.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-reserved-concurrency-and-provisioned-capacity-for-serverless-gpu-ensuring-zer.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-reserved-instances-and-savings-plans-for-gpu-rental-are-1-year-commitment-dis.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-api-and-cli-tools-automating-gpu-instance-management-with-scripts.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-billing-and-invoicing-for-international-users-a-complete-guide-to-comp.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-community-ecosystem-a-roundup-of-third-party-tools-templates-and-autom.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-console-and-payment-methods-explained-a-guide-for-international-users.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-data-center-network-architecture-quality-of-private-lines-peering-and-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-enterprise-features-explained-sso-audit-logs-and-dedicated-resource-gr.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-global-node-distribution-how-to-choose-the-data-center-closest-to-your.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-instance-type-selection-differences-between-community-cloud-secure-clo.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-invoicing-and-tax-how-international-users-obtain-compliant-tax-documen.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-network-optimization-for-global-users-achieving-the-lowest-latency-wor.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-network-storage-performance-test-throughput-comparison-of-nvme-hdd-and.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-pay-per-use-and-monthly-instance-mix-a-cost-saving-combo-for-base-and-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-serverless-concurrency-limits-and-scaling-behavior-stress-test-data-vs.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-serverless-gpu-in-depth-review-how-much-can-you-really-save-with-per-s.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-spot-instance-tips-running-non-real-time-inference-tasks-at-a-70-disco.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-startup-scripts-and-initialization-automating-environment-setup-model-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-team-management-sub-accounts-permission-roles-and-resource-quota-alloc.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-templates-and-community-images-quickly-launching-stable-diffusion-and-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-vs-salad-trade-offs-between-decentralized-gpu-networks-and-centralized.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-runpod-vs-vastai-reliability-and-cost-effectiveness-of-community-marketplace-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-self-hosted-vs-serverless-inference-cost-a-line-by-line-breakdown-with-llama-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-serverless-gpu-cold-start-benchmarks-modal-runpod-and-replicate-response-time.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-serverless-gpu-cold-start-deep-analysis-impact-of-image-size-model-loading-an.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-serverless-gpu-cold-start-time-leaderboard-startup-speed-comparison-across-pl.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-serverless-gpu-for-batch-inference-best-practices-for-large-scale-text-classi.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-serverless-gpu-for-real-time-speech-recognition-cost-and-latency-benchmarks-f.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-serverless-gpu-for-video-understanding-cost-analysis-for-deploying-models-lik.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-serverless-gpu-network-egress-fees-explained-the-true-cost-of-cross-region-da.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-serverless-gpu-platform-latency-test-by-region-ping-values-from-major-global-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-serverless-gpu-platform-selection-matrix-cold-start-max-vram-and-regional-ava.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-serverless-gpu-tested-in-practice-finding-the-sweet-spot-between-cold-start-a.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-serverless-inference-billing-traps-real-cases-of-minimum-billing-units-idle-c.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-ssl-certificate-automation-for-self-hosted-inference-certbot-and-acme-protoco.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-startup-time-optimization-for-vllm-deployment-model-warm-up-kernel-fusion-and.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-storage-choices-for-vllm-deployment-local-nvme-network-block-storage-and-obje.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-stress-testing-self-hosted-inference-services-simulating-real-user-load-with-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-technical-support-quality-review-for-ai-inference-platforms-ticket-response-c.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-the-break-even-point-for-gpu-rental-hourly-vs-monthly-billing-mathematical-mo.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-the-economics-of-serverless-inference-why-pay-per-use-wins-when-traffic-is-hi.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-the-financialization-of-gpu-rental-pricing-models-for-compute-futures-options.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-the-full-picture-of-ai-inference-latency-optimization-every-millisecond-from-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-the-second-hand-market-and-compute-reselling-for-gpu-rental-compliance-risks-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-the-total-cost-of-ownership-model-for-gpu-cloud-including-labor-power-colocat.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-the-ultimate-decision-checklist-for-gpu-cloud-selection-30-questions-to-lock-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-tls-certificate-management-for-self-hosted-inference-lets-encrypt-cert-manage.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-traffic-prediction-and-capacity-planning-for-ai-model-deployment-auto-scaling.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vendor-lock-in-risk-assessment-for-ai-inference-platforms-designing-a-migrata.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vllm-asynchronous-output-handling-efficiently-processing-results-with-streami.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vllm-block-size-tuning-an-experiment-on-the-impact-of-block-size-on-throughpu.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vllm-cuda-graph-optimization-reducing-kernel-launch-overhead-with-computation.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vllm-deployment-guide-from-docker-to-production-api-on-a-single-gpu.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vllm-deployment-tutorial-configuring-production-grade-inference-on-aws-gcp-an.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vllm-logging-levels-and-formats-structured-logging-json-output-and-log-aggreg.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vllm-long-context-support-vram-and-performance-tuning-for-processing-128k-tok.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vllm-lora-adapter-management-dynamic-loading-unloading-and-concurrent-serving.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vllm-multi-gpu-deployment-a-detailed-configuration-guide-for-tensor-pipeline-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vllm-on-consumer-gpus-extreme-tuning-for-running-7b-models-on-an-rtx-4090.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vllm-openai-compatible-api-in-detail-supported-parameters-and-limitations.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vllm-prefix-caching-in-practice-how-to-halve-the-cost-of-long-conversation-in.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vllm-production-tuning-continuous-batching-pagedattention-and-quantization-st.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vllm-quantization-deployment-guide-awq-gptq-and-fp8-performance-benchmarked-o.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vllm-request-scheduling-visualization-real-time-monitoring-of-queue-length-an.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vllm-scheduling-policy-explained-first-come-first-served-priority-queues-and-.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/en-vllm-speculative-decoding-implementation-doubling-inference-speed-with-a-draf.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/从-docker-到生产-api-的完整部署指南构建可水平扩展的模型推理服务.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/从-jupyter-notebook-到生产-api模型部署的工程化鸿沟如何跨越.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/从零构建模型推理-apidockerfastapi-与-vllm-的组合最佳实践.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/国内用户如何选择海外-gpu-云runpodlambda-labs-与-vastai-横向评测.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何为-agent-应用设计推理基础设施工具调用多轮对话与状态管理.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何为-rag-应用部署嵌入与重排序模型的推理服务.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何为医疗金融等合规行业部署私有化-ai-推理服务.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何为多租户-saas-产品设计推理服务的隔离与计费方案.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何为开源-llm-选择推理框架vllmtgitriton-与-ray-serve-对比.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何为开源模型构建与-openai-完全兼容的-api-网关.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何为边缘设备部署推理服务从云端到-jetson-的模型适配.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何构建-ai-推理的成本仪表板实时追踪每个模型每个版本的支出.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何用-vllm-和-fastapi-构建流式推理端点sse-与-websocket-实现.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何用-vllm-和-litellm-构建多模型统一-api-网关.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何用-vllm-部署代码生成模型deepseek-coder-的-fim-推理配置.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何用-vllm-部署多模态模型llavaqwen-vl-的推理服务配置.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何用-vllm-部署嵌入模型从-bge-到-e5-的文本向量化服务搭建.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何用-vllm-部署嵌入模型和重排序模型为-rag-管道提速.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何用-vllm-部署语音识别模型whisper-的流式与批量推理方案.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何评估-ai-推理平台的性价比构建包含延迟吞吐与成本的综合指标.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何评估模型部署方案的总拥有成本硬件带宽运维与机会成本.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何选择模型部署的地域中国大陆香港新加坡与美西的延迟测试.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/如何部署开源模型到生产环境一份涵盖-vllmtgi-与-triton-的实操手册.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/开源-llm-生产化部署方案选型从-docker-镜像到生产-api-全流程.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/开源模型-api-化部署使用-vllm-构建兼容-openai-接口的推理端点.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/按调用量算账openaireplicate-与自建-vllm-的-api-成本拆解.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/模型部署成本控制手册量化缓存与请求合并的降本三板斧.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/用-vllm-部署千问-25从权重下载到-openai-兼容-api-的分步教程.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/自托管-vs-serverless-推理成本对比以-llama-3-70b-为例逐项拆解.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/自托管推理方案的备份与灾备模型权重配置与日志的高可用设计.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/自托管推理服务器搭建实录从裸金属装机到-vllm-服务上线.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/自托管推理服务的-api-文档自动生成基于-openapi-与-swagger-的实现.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/自托管推理服务的-api-版本管理如何在不破坏客户端的情况下迭代.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/自托管推理服务的-api-限流令牌桶滑动窗口与分布式限流实现.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/自托管推理服务的-cicd-流水线模型更新零停机部署的实现.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/自托管推理服务的-tls-证书管理lets-encryptcert-manager-与自动续签.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/自托管推理服务的压力测试用-locust-和-k6-模拟真实用户负载.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/自托管推理的-gpu-温度与功耗监控prometheus-nvidia-dcgm-方案.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/自托管推理的-gpu-虚拟化方案migvgpu-与时分复用技术选型.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/自托管推理的-ssl-证书自动化certbot-与-acme-协议在私有网络中的应用.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/自托管推理的模型热更新无需重启服务即可切换-lora-或基础模型.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/自托管推理的镜像仓库管理harborecr-与安全扫描集成.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/自托管推理集群的日志管理elkloki-与云原生方案的应用.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
  <url>
    <loc>https://aideploy.com.cn/blog/自托管推理集群的自动扩缩容基于-kubernetes-与-prometheus-的实现.md/</loc>
    <lastmod>2026-07-28</lastmod>
    <changefreq>monthly</changefreq>
    <priority>0.8</priority>
  </url>
</urlset>