<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9" xmlns:news="http://www.google.com/schemas/sitemap-news/0.9">
  <url>
    <loc>https://theservingdesk.com/local-inference-break-even-mac-cloud</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-08T03:18:32+00:00</news:publication_date>
      <news:title>Mac local inference break-even: when hardware beats cloud per-token pricing</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/punto-equilibrio-inferencia-local-mac-nube</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-08T03:18:32+00:00</news:publication_date>
      <news:title>Punto de equilibrio de inferencia local en Mac: cuándo el hardware supera el precio por token en la nube</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/chatgpt-caida-spike-serving-check</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-08T03:17:17+00:00</news:publication_date>
      <news:title>ChatGPT caída spikes: check error rate, p95, queue, cost</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/picos-caida-chatgpt-tasa-error-p95-cola</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-08T03:17:17+00:00</news:publication_date>
      <news:title>Picos de caída de ChatGPT: revisa tasa de error, p95, cola y coste</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/local-ai-camera-stack-sizing</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-08T03:13:59+00:00</news:publication_date>
      <news:title>Sizing a local AI camera stack: model, RTSP, Docker, latency, cost</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/pila-camaras-ia-local-rtsp-docker-latencia</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-08T03:13:59+00:00</news:publication_date>
      <news:title>Dimensionar una pila de cámaras con IA local: modelo, RTSP, Docker, latencia y coste</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/ai-coding-token-cost-routing</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-08T03:13:24+00:00</news:publication_date>
      <news:title>Cut AI coding token bills with bulk-read routing and context caps</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/reducir-costes-tokens-ia-enrutamiento-modelos</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-08T03:13:24+00:00</news:publication_date>
      <news:title>Reduce los costes de tokens de IA con enrutamiento de lecturas masivas y límites de contexto</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/gemini-3-7-flash-price-cut-routing</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-08T03:06:39+00:00</news:publication_date>
      <news:title>Gemini 3.7 Flash Price Cut: Fix Caching and Routing</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/gemini-3-7-flash-recorte-precios-cache-enrutamiento</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-08T03:06:39+00:00</news:publication_date>
      <news:title>Recorte de precios de Gemini 3.7 Flash: corrige la caché y el enrutamiento</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/openrouter-gpt-6-astra-endpoint-pricing</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-08T03:06:19+00:00</news:publication_date>
      <news:title>How to Price a Model Endpoint: GPT-6 Astra on OpenRouter</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/precio-endpoint-gpt6-astra-openrouter</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-08T03:06:19+00:00</news:publication_date>
      <news:title>Cómo fijar el precio de un endpoint de modelo: GPT-6 Astra en OpenRouter</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/sub-dollar-inference-cost-model</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-08T03:05:52+00:00</news:publication_date>
      <news:title>Build a Sub-$1 Inference Cost Model</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/modelo-costos-inferencia-por-debajo-1-dolar</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-08T03:05:52+00:00</news:publication_date>
      <news:title>Construir un modelo de costos de inferencia por debajo de $1</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/ai-serving-cost-zai-inference</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-08T03:05:36+00:00</news:publication_date>
      <news:title>Estimate AI Serving Cost in Five Numbers: Z.AI&#039;s 100,000-Accelerator Run</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/coste-inferencia-ia-zai-aceleradores</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-08T03:05:36+00:00</news:publication_date>
      <news:title>Estima el coste de inferencia de IA con cinco cifras: la ejecución de 100.000 aceleradores de Z.AI</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/foundation-model-serving-stack-test</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-08T03:05:27+00:00</news:publication_date>
      <news:title>Choose the model your serving stack can run</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/elegir-modelo-inferencia-stack-servicio</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-08T03:05:27+00:00</news:publication_date>
      <news:title>Elige el modelo que tu infraestructura de inferencia puede ejecutar</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/gemini-4-model-serving-budget</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-08T03:05:14+00:00</news:publication_date>
      <news:title>Turn a frontier model update into a serving budget: the Gemini 4 example</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/presupuesto-servicio-gemini-4-inferencia</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-08T03:05:14+00:00</news:publication_date>
      <news:title>Convertir una actualización de modelo frontera en un presupuesto de servicio: el caso Gemini 4</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/ai-agent-inference-cost-budget</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-08T03:04:55+00:00</news:publication_date>
      <news:title>Build a Monthly Inference Budget for an AI Agent</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/presupuesto-mensual-inferencia-agente-ia</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-08T03:04:55+00:00</news:publication_date>
      <news:title>Construye un presupuesto mensual de inferencia para un agente de IA</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/vmware-private-ai-cloud-cost-test</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-08T03:04:33+00:00</news:publication_date>
      <news:title>Private AI Cloud Cost Test for Enterprise Inference</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/prueba-costos-nube-privada-ia-inferencia-empresarial</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-08T03:04:33+00:00</news:publication_date>
      <news:title>Prueba de costos de nube privada de IA para inferencia empresarial</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/open-weights-model-routing-pipeline</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-08T03:04:21+00:00</news:publication_date>
      <news:title>Open-Weights-First Model Routing Pipeline</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/enrutamiento-modelos-pesos-abiertos-por-defecto</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-08T03:04:21+00:00</news:publication_date>
      <news:title>Pipeline de enrutamiento de modelos con pesos abiertos por defecto</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/inference-exchange-vs-cloud-apis-self-hosted</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-08T03:03:13+00:00</news:publication_date>
      <news:title>Score cloud APIs, self-hosted models, and inference exchanges</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/comparar-apis-nube-modelos-autoalojados-inferencia</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-08T03:03:13+00:00</news:publication_date>
      <news:title>Evalúa APIs en la nube, modelos autoalojados e intercambios de inferencia</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/llm-inference-cost-latency-frontier</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-07T07:07:23+00:00</news:publication_date>
      <news:title>Pick the Cheapest LLM Stack That Meets Your p95 Latency</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/stack-llm-mas-barato-latencia-p95</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-07T07:07:23+00:00</news:publication_date>
      <news:title>Elige el stack de LLM más barato que cumpla tu latencia p95</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/vcf-inference-vs-managed-api-cost-latency</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-07T07:03:09+00:00</news:publication_date>
      <news:title>Use VCF for Inference Only If vLLM Latency, GPU Scheduling, and $/M Tokens Beat the Managed API</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/vcf-inferencia-latencia-gpu-tokens-api</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-07T07:03:09+00:00</news:publication_date>
      <news:title>Usa VCF solo para inferencia si la latencia de vLLM, la programación de GPU y los $/M tokens superan a la API gestionada</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/gpu-sizing-by-tokens-per-second-dollar</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-07T06:56:46+00:00</news:publication_date>
      <news:title>Size inference GPUs by tokens per second per dollar, not by model size;</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/dimensionar-gpus-inferencia-tokens-por-segundo-por-dolar</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-07T06:56:46+00:00</news:publication_date>
      <news:title>Dimensiona GPUs de inferencia por tokens por segundo por dólar, no por tamaño de modelo</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/token-cost-worksheet-inference-price-floor</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-07T06:54:48+00:00</news:publication_date>
      <news:title>Build a Token-Cost Worksheet That Sets Your Inference Price Floor</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/planilla-costo-por-token-piso-precio-inferencia</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-07T06:54:48+00:00</news:publication_date>
      <news:title>Crea una planilla de costo por token que fije el piso de precio de inferencia</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/local-mac-inference-cost-break-even</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-07T06:51:02+00:00</news:publication_date>
      <news:title>Local Mac Inference: 12 tok/s, 8k Prefill, Cost Break-Even</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/inferencia-local-mac-12-tok-s-prefill-8k</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-07T06:51:02+00:00</news:publication_date>
      <news:title>Inferencia local en Mac: 12 tok/s, prefill de 8k y punto de equilibrio de costos</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/budget-ttft-prefill-batching-quantization-routing</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-07T06:47:30+00:00</news:publication_date>
      <news:title>Budget TTFT: prefill, batching, quantization, and routing for streaming LLM products</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/presupuesto-ttft-llm-streaming-prefill</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-07T06:47:30+00:00</news:publication_date>
      <news:title>Presupuesto de TTFT: prefill, agrupación por lotes, cuantización y enrutamiento para productos de LLM en streaming</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/2-4b-slm-vs-frontier-model-choice</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-07T06:37:39+00:00</news:publication_date>
      <news:title>Pick a 2.4B SLM when your workflow clears the cost, latency, and privacy gates; otherwise pay for frontier</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/cuando-usar-slm-2-4b-costo-latencia-privacidad</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-07T06:37:39+00:00</news:publication_date>
      <news:title>Elige un SLM de 2,4B cuando tu flujo de trabajo supere los umbrales de costo, latencia y privacidad; de lo contrario, paga por un modelo frontera</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/calculate-outage-tax-gpu-cluster-expansion</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-07T06:35:34+00:00</news:publication_date>
      <news:title>Calculate the outage tax before your next GPU cluster expansion</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/impuesto-caidas-gpu-cluster-expansion</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-07T06:35:34+00:00</news:publication_date>
      <news:title>Calcula el impuesto por caídas antes de tu próxima expansión de clúster GPU</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/production-llm-selection-six-point-eval</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-07T06:28:03+00:00</news:publication_date>
      <news:title>Pick a Production LLM With a 6-Point Eval</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/elegir-llm-produccion-evaluacion-seis-puntos</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-07T06:28:03+00:00</news:publication_date>
      <news:title>Elige un LLM de producción con una evaluación de 6 puntos</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/ai-serving-outage-downtime-cost-slo-fallbacks</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-09-07T06:11:34+00:00</news:publication_date>
      <news:title>Cut AI Outage Cost: SLOs, Fallbacks, Downtime Budgets</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://theservingdesk.com/es/reducir-costo-caidas-ia-slo-respaldo-presupuesto</loc>
    <news:news>
      <news:publication><news:name>The Serving Desk</news:name><news:language>es</news:language></news:publication>
      <news:publication_date>2026-09-07T06:11:34+00:00</news:publication_date>
      <news:title>Reduce el costo de las caídas de IA: SLOs, respaldos y presupuestos de inactividad</news:title>
    </news:news>
  </url>
</urlset>