<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9" xmlns:news="http://www.google.com/schemas/sitemap-news/0.9">
  <url>
    <loc>https://xonoai.com/unmasking-inference-latency-vllm-tensorrt-llm-ollama-benchmarks/</loc>
    <news:news>
      <news:publication>
        <news:name>XonoAI</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-10-11T14:57:56+00:00</news:publication_date>
      <news:title>Unmasking Inference Latency: Real-World Benchmarking of vLLM, TensorRT-LLM, and Ollama in Production Clusters</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://xonoai.com/unpacking-the-vla-bottleneck-vision-language-action-models-edge/</loc>
    <news:news>
      <news:publication>
        <news:name>XonoAI</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-10-11T09:19:09+00:00</news:publication_date>
      <news:title>Unpacking the VLA Bottleneck: Why Vision-Language-Action Models Fail at Edge Actuation</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://xonoai.com/custom-silicon-asics-optical-interconnects-ai-superclusters/</loc>
    <news:news>
      <news:publication>
        <news:name>XonoAI</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-10-10T10:00:24+00:00</news:publication_date>
      <news:title>Unseating the GPU Monopolies: How Custom Silicon ASICs and Optical Interconnects are Rewriting AI Supercluster Economics</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://xonoai.com/deterministic-agentic-runtimes-fault-tolerant-mcp-topologies/</loc>
    <news:news>
      <news:publication>
        <news:name>XonoAI</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-10-10T09:33:14+00:00</news:publication_date>
      <news:title>Deterministic Agentic Runtimes: Engineering Fault-Tolerant Multi-Agent Topologies with MCP</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://xonoai.com/decoupling-test-time-computation-moe-reasoning-synthesis-engine/</loc>
    <news:news>
      <news:publication>
        <news:name>XonoAI</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-10-10T08:30:40+00:00</news:publication_date>
      <news:title>Decoupling Test-Time Computation: Inside XonoAI’s Dynamic MoE-Reasoning Synthesis Engine</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://xonoai.com/deepseek-v3-moe-architecture-mla-auxiliary-loss-free-load-balancing-fp8/</loc>
    <news:news>
      <news:publication>
        <news:name>XonoAI</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-10-10T08:23:29+00:00</news:publication_date>
      <news:title>DeepSeek-V3 MoE Architecture: Inside Multi-Head Latent Attention (MLA), Auxiliary-Loss-Free Load Balancing, and FP8 Mixed Precision Training</news:title>
    </news:news>
  </url>
</urlset>
