<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9" xmlns:news="http://www.google.com/schemas/sitemap-news/0.9">
  <url>
    <loc>https://modelsatwork.news/article/aws-agentic-value-model-business-case-hours-saved-misleads</loc>
    <news:news>
      <news:publication><news:name>Models at Work</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-10-11T17:00:00.000Z</news:publication_date>
      <news:title>AWS: 'hours saved' overstates agent ROI; price exceptions and oversight too</news:title>
      <news:keywords>AWS, Amazon Quick, Business case, ROI, Agentic automation, Vendor-published</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://modelsatwork.news/article/vals-ai-agent-teams-vibe-code-bench-cost-1-8-to-5-1x-one-significant-gain</loc>
    <news:news>
      <news:publication><news:name>Models at Work</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-10-11T16:00:00.000Z</news:publication_date>
      <news:title>Vals AI: agent teams cost 1.8-5.1x more, and only 1 of 4 gains was significant</news:title>
      <news:keywords>Vals AI, Vibe Code Bench, Multi-agent, GPT 6 Sol, Claude Opus 5.5, Agent cost</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://modelsatwork.news/article/incarna-blockrun-agentcore-payments-agents-pay-per-inference-x402</loc>
    <news:news>
      <news:publication><news:name>Models at Work</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-10-11T15:00:00.000Z</news:publication_date>
      <news:title>Incarna's agents pay BlockRun per inference call via AWS AgentCore payments</news:title>
      <news:keywords>Incarna, BlockRun, Amazon Bedrock AgentCore, x402, Agent payments, Vendor-published</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://modelsatwork.news/article/anthropic-cyber-verification-program-three-access-tiers</loc>
    <news:news>
      <news:publication><news:name>Models at Work</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-10-11T11:30:00.000Z</news:publication_date>
      <news:title>Anthropic's Cyber Verification Program now has three access tiers</news:title>
      <news:keywords>Anthropic, Cyber Verification Program, Security, Claude Opus 5.5, Project Glasswing, Data retention, Enterprise Frontier Safeguards</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://modelsatwork.news/article/microsoft-decision-1-model-routing-classification-foundry-openrouter</loc>
    <news:news>
      <news:publication><news:name>Models at Work</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-10-10T22:00:00.000Z</news:publication_date>
      <news:title>Microsoft-Decision-1: a 9B model that scores choices for $0.042 per million tokens</news:title>
      <news:keywords>Microsoft, Decision-1, Decision models, Routing, Classification, Azure Foundry, OpenRouter, Small model</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://modelsatwork.news/article/epoch-innovationeval-ai-agents-ml-research-misleading-claims-seed-farming</loc>
    <news:news>
      <news:publication><news:name>Models at Work</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-10-10T21:00:00.000Z</news:publication_date>
      <news:title>Epoch AI: frontier agents fail to rediscover an ML technique and overstate results</news:title>
      <news:keywords>Epoch AI, InnovationEval, Evals, Agents, AI R&amp;D, Reward hacking, Verification</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://modelsatwork.news/article/barclays-claude-16000-staff-knowledge-assistant-120000-emails-a-day</loc>
    <news:news>
      <news:publication><news:name>Models at Work</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-10-10T19:00:00.000Z</news:publication_date>
      <news:title>Barclays on Claude: 16,000 staff use a knowledge assistant, 120,000 emails a day sorted</news:title>
      <news:keywords>Barclays, Anthropic, Claude, Banking, RAG, Email triage, Claude Code, Vendor case study</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://modelsatwork.news/article/aws-rag-access-control-query-time-acl-checks-quick-bedrock</loc>
    <news:news>
      <news:publication><news:name>Models at Work</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-10-10T17:20:00.000Z</news:publication_date>
      <news:title>AWS: copied permissions go stale in enterprise RAG, so Amazon Quick now re-checks them with the source at query time</news:title>
      <news:keywords>RAG, Access control, Amazon Quick, Amazon Bedrock, Enterprise search, Vendor-published</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://modelsatwork.news/article/openai-misalignment-reports-grader-wrecked-environment-bypassed-get-only-proxy</loc>
    <news:news>
      <news:publication><news:name>Models at Work</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-10-10T16:00:00.000Z</news:publication_date>
      <news:title>OpenAI publishes two incident reports: a grader wrecked its own sandbox, and models bypassed a GET-only proxy</news:title>
      <news:keywords>OpenAI, Agent security, Monitoring, Sandboxing, Misalignment reports, Egress controls</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://modelsatwork.news/article/talorys-open-source-personal-agent-cloudflare-free-tier</loc>
    <news:news>
      <news:publication><news:name>Models at Work</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-10-10T14:57:00.000Z</news:publication_date>
      <news:title>Talorys: an open-source personal agent that runs in your own Cloudflare account, on the free tier</news:title>
      <news:keywords>Talorys, Open source, Cloudflare Workers, Durable Objects, Personal agents, Show HN</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://modelsatwork.news/article/postman-agent-mode-170-tools-15-context-bedrock</loc>
    <news:news>
      <news:publication><news:name>Models at Work</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-10-10T14:30:00.000Z</news:publication_date>
      <news:title>Postman: past about 40 visible tools, its agent's tool choices got worse; it now shows the model about 15 of 170</news:title>
      <news:keywords>Postman, Amazon Bedrock, Agents, Tool use, Prompt caching, Vendor-published</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://modelsatwork.news/article/anthropic-unintended-model-actions-evals-live-internet-off</loc>
    <news:news>
      <news:publication><news:name>Models at Work</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-10-10T15:00:00.000Z</news:publication_date>
      <news:title>Anthropic reports Claude acted on real websites during evals, and turns off live internet for all internal evaluations</news:title>
      <news:keywords>Anthropic, Claude, Agent safety, Evaluations, Security</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://modelsatwork.news/article/anthropic-dynamic-workflows-managed-agents-beta</loc>
    <news:news>
      <news:publication><news:name>Models at Work</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-10-09T20:17:07.000Z</news:publication_date>
      <news:title>Anthropic lets one Claude agent write and run a workflow of up to 1,000 agents</news:title>
      <news:keywords>Anthropic, Claude, Managed Agents, Multi-agent, Agents</news:keywords>
    </news:news>
  </url>
  <url>
    <loc>https://modelsatwork.news/briefing/2026-10-11</loc>
    <news:news>
      <news:publication><news:name>Models at Work</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-10-11T19:20:00.000Z</news:publication_date>
      <news:title>The briefing, 2026-10-11: Vals AI: agent teams cost 1.8 to 5.1 times more, and only one of four gains was significant</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://modelsatwork.news/briefing/2026-10-10</loc>
    <news:news>
      <news:publication><news:name>Models at Work</news:name><news:language>en</news:language></news:publication>
      <news:publication_date>2026-10-10T18:30:00.000Z</news:publication_date>
      <news:title>The briefing, 2026-10-10: Anthropic says Claude acted on real third-party websites during evaluations, including submitting a police tip</news:title>
    </news:news>
  </url>
</urlset>
