<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"
        xmlns:news="http://www.google.com/schemas/sitemap-news/0.9">
  <url>
    <loc>https://tpsreport.news/news/gemini-3-5-transcribe-speech-to-text-model</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-26T17:36:26.646Z</news:publication_date>
      <news:title>Google Launches Gemini 3.5 Transcribe, a Speech-to-Text Model That Cleans Up Rambling Speech</news:title>
      <news:keywords>Google, Gemini, speech-to-text, transcription, audio AI, Android</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/gemini-live-spark-gmail-integrations</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-26T17:35:57.401Z</news:publication_date>
      <news:title>Google Adds Spark, Gmail, and Daily Brief Integrations to Gemini Live</news:title>
      <news:keywords>Gemini Live, Google, Spark, Gmail, productivity, voice assistant, Daily Brief</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/gemini-3-5-transcribe-speech-to-text-1</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-26T17:21:03.743Z</news:publication_date>
      <news:title>Google DeepMind Launches Gemini 3.5 Transcribe, Claims 2.6% Word Error Rate in Testing</news:title>
      <news:keywords>Gemini, Google DeepMind, speech-to-text, transcription, voice AI, Gemini API</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/gemini-3-5-transcribe-speech-to-text</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-26T17:05:56.096Z</news:publication_date>
      <news:title>Google Launches Gemini 3.5 Transcribe with 2.6% Word Error Rate, Powers Gboard Rambler</news:title>
      <news:keywords>Gemini, Google, speech-to-text, transcription, Gboard, Chrome, voice AI</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/qwen3-8-flash-next-cost-efficient-moe</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-26T14:51:39.580Z</news:publication_date>
      <news:title>Alibaba Releases Qwen3.8-Flash-Next: 125B-Parameter MoE Model Matches Larger Rivals at $0.16/$0.47 per Million Tokens</news:title>
      <news:keywords>Qwen, Alibaba, Qwen3.8-Flash-Next, mixture-of-experts, open weights, AI pricing, agentic coding, Qwen4</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/transformers-v5-16-1-glm-5-3-flash-support</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-26T14:51:00.858Z</news:publication_date>
      <news:title>Hugging Face Transformers v5.16.1 Adds Support for GLM-5.3-Flash, a 320B-Parameter Multimodal MoE Model</news:title>
      <news:keywords>transformers, GLM-5.3-Flash, Hugging Face, Zhipu AI, mixture-of-experts, multimodal, open source</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/glm-5-3-flash-qwen3-8-flash-next-huggingface-roundup</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-26T14:37:44.406Z</news:publication_date>
      <news:title>GLM-5.3-Flash and Qwen3.8-Flash-Next Appear on Hugging Face With No Model Cards or Benchmarks Yet Published</news:title>
      <news:keywords>GLM-5.3-Flash, Qwen3.8-Flash-Next, Zhipu AI, Alibaba Qwen, Hugging Face, open-weight models, unsloth, roundup, trending</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/glm-5-3-flash-multimodal-moe-release</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-26T14:37:15.351Z</news:publication_date>
      <news:title>Zhipu AI Releases GLM-5.3-Flash: First Multimodal Model in GLM-5 Series, 320B Parameters with Only 18B Active</news:title>
      <news:keywords>GLM-5.3-Flash, Zhipu AI, zai-org, multimodal model, mixture-of-experts, open weights, Hugging Face</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/z-ai-ox-alpha-model-confirmed</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-26T14:36:23.441Z</news:publication_date>
      <news:title>Z.ai Confirmed as Creator of Chart-Topping 'Ox Alpha' Model, Weights Coming Wednesday</news:title>
      <news:keywords>Z.ai, GLM, open-weight, Ox Alpha, OpenRouter, China AI, reasoning models</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/zai-glm-5-3-flash-1m-context</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-26T14:35:46.191Z</news:publication_date>
      <news:title>Z.ai Launches GLM-5.3-Flash With 1M-Token Context and Hybrid Attention Architecture</news:title>
      <news:keywords>GLM-5.3-Flash, Z.ai, Zhipu AI, OpenRouter, multimodal, long-context, coding models, AI agents</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/qwen3-8-flash-next-hybrid-architecture-gguf</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-26T14:06:01.116Z</news:publication_date>
      <news:title>Qwen3.8-Flash-Next Debuts with 125B-Parameter Hybrid Architecture, Previews Qwen4 Design</news:title>
      <news:keywords>Qwen, Qwen3.8-Flash-Next, Alibaba, GGUF, Unsloth, open-weight, MoE, long context</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/qwen3-8-flash-next-architecture-preview</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-26T12:36:21.873Z</news:publication_date>
      <news:title>Alibaba Releases Qwen3.8-Flash-Next, a 125B-Parameter Preview of Qwen4's Architecture</news:title>
      <news:keywords>Qwen, Alibaba, Qwen4, open-weight, sparse attention, MoE, long context, agentic AI</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/ibm-granite-4-2-open-weight-agentic-models</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-26T10:50:48.057Z</news:publication_date>
      <news:title>IBM Releases Granite 4.2 Open-Weight Models With Agentic RL Training and 512K Context</news:title>
      <news:keywords>IBM, Granite, open-weight, agentic AI, Apache 2.0, speech recognition, tool calling</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/cline-desktop-v0-0-19-memory-leak-fix</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-26T09:35:47.849Z</news:publication_date>
      <news:title>Cline Desktop v0.0.19 Fixes Memory Leak That Ballooned Process to Tens of Gigabytes</news:title>
      <news:keywords>cline, desktop-app, changelog, bug-fix, model-routing</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/ollama-v0-33-0-claude-desktop-cache-fixes</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-26T00:35:37.206Z</news:publication_date>
      <news:title>Ollama v0.33.0 Fixes KV Cache Bug That Forced Full Reprocessing of 46K-Token Prompts</news:title>
      <news:keywords>ollama, product-update, claude-desktop, caching, kv-cache, changelog</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/chatgpt-task-scheduling-free-accounts</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-25T21:35:39.543Z</news:publication_date>
      <news:title>OpenAI Expands ChatGPT Task Scheduling to Free Accounts, Adds Gmail/Slack/GitHub Triggers for Paid Tiers</news:title>
      <news:keywords>ChatGPT, OpenAI, task scheduling, product update, automation, Gmail integration, Slack integration, GitHub integration</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/openai-admin-plugin-chatgpt-work-codex</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-25T21:05:39.604Z</news:publication_date>
      <news:title>OpenAI Launches Admin Plugin for ChatGPT Work and Codex Workspace Management</news:title>
      <news:keywords>OpenAI, ChatGPT, ChatGPT Work, Codex, enterprise AI, admin tools, product update</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/claude-cowork-chat-unified-memory</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-25T18:05:40.607Z</news:publication_date>
      <news:title>Anthropic Unifies Memory Across Claude Chat and Cowork</news:title>
      <news:keywords>Anthropic, Claude, Claude Cowork, product update, memory, AI agents</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/anthropic-unifies-memory-claude-cowork-chat</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-25T17:35:48.797Z</news:publication_date>
      <news:title>Anthropic Unifies Memory Between Claude Cowork and Claude Chat</news:title>
      <news:keywords>Anthropic, Claude, Claude Cowork, memory, product update</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/anthropic-claude-cowork-shared-memory</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-25T17:05:45.587Z</news:publication_date>
      <news:title>Anthropic Merges Claude Chat and Cowork Memory Into a Single Shared System</news:title>
      <news:keywords>Anthropic, Claude, Claude Cowork, AI memory, privacy, product update</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/google-gemini-enterprise-legal-launch</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-25T16:35:43.931Z</news:publication_date>
      <news:title>Google Launches Gemini Enterprise for Legal, Connecting AI to Contract and Research Platforms</news:title>
      <news:keywords>Google, Gemini, legal tech, enterprise AI, Google Cloud, MCP, product update</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/ibm-granite-4-2-reasoning-llms</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-25T15:21:02.257Z</news:publication_date>
      <news:title>IBM Releases Granite 4.2, Its First Reasoning-Focused LLM Family in 3B, 8B, and 30B Sizes</news:title>
      <news:keywords>IBM, Granite, open-source-llm, reasoning-model, agentic-ai, apache-2.0</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/granite-speech-5-0-turbo-ctc-470m</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-25T15:05:57.428Z</news:publication_date>
      <news:title>IBM Releases Granite Speech 5.0 Turbo CTC: 470M-Parameter ASR Model Hits 12,600x Real-Time Speed</news:title>
      <news:keywords>IBM, Granite Speech, speech recognition, ASR, open source, Apache 2.0, transformers, CTC</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/openai-jalapeno-chip-inference-benchmarks</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-25T14:35:45.661Z</news:publication_date>
      <news:title>OpenAI's Jalapeño Inference Chip Beats Nvidia Blackwell on Throughput and Power Efficiency, Company Claims</news:title>
      <news:keywords>OpenAI, Broadcom, Jalapeño, inference chip, AI hardware, Nvidia Blackwell, semiconductors</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/meta-hatch-ai-agent-watermelon-model</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-25T13:50:45.594Z</news:publication_date>
      <news:title>Meta Preps Paid AI Agent 'Hatch' for Launch, Plans New Model 'Watermelon' for October</news:title>
      <news:keywords>Meta, AI Agent, Hatch, Watermelon, Muse, WhatsApp, product launch</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/nvidia-groq-3-lpx-cerebras-benchmark-comparison</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-25T12:35:53.669Z</news:publication_date>
      <news:title>Nvidia Claims Groq 3 LPX Hits 3,400 Tokens/Sec, 4x Cerebras — But Needs 64 Chips to Do It</news:title>
      <news:keywords>Nvidia, Groq, Cerebras, inference chips, benchmark, AI hardware, Vera Rubin, Hot Chips 2026</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/sensenova-u1-5-8b-mot-multimodal-model</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-25T05:35:48.046Z</news:publication_date>
      <news:title>SenseNova Releases U1.5-8B-MoT, an Open-Weight Unified Model for Image Generation and Editing</news:title>
      <news:keywords>SenseNova, multimodal, image generation, image editing, open-weight, Apache 2.0, Hugging Face</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://tpsreport.news/news/openai-restores-5-hour-limit-chatgpt-plus</loc>
    <news:news>
      <news:publication>
        <news:name>TPS — Tokens Per Second</news:name>
        <news:language>en</news:language>
      </news:publication>
      <news:publication_date>2026-08-25T02:05:39.717Z</news:publication_date>
      <news:title>OpenAI Reinstates 5-Hour Usage Limit for ChatGPT Plus Codex and Work Tiers</news:title>
      <news:keywords>OpenAI, ChatGPT, Codex, ChatGPT Plus, usage limits, rate limits</news:keywords>
    </news:news>
  </url>
</urlset>