<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Ai-Security on Steve Hatch&#39;s Blog</title>
    <link>https://www.hatch.org/tags/ai-security/</link>
    <description>Recent content in Ai-Security on Steve Hatch&#39;s Blog</description>
    <generator>Hugo</generator>
    <language>en-us</language>
    <lastBuildDate>Mon, 27 Jul 2026 00:00:00 +0000</lastBuildDate>
    <atom:link href="https://www.hatch.org/tags/ai-security/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>The Guardrails That Stopped Your AI From Attacking Also Stopped It From Defending You</title>
      <link>https://www.hatch.org/2026/07/27/ai-guardrail-asymmetry/</link>
      <pubDate>Mon, 27 Jul 2026 00:00:00 +0000</pubDate>
      <guid>https://www.hatch.org/2026/07/27/ai-guardrail-asymmetry/</guid>
      <description>&lt;p&gt;&lt;img src=&#34;https://www.hatch.org/images/ai-guardrail-asymmetry.png&#34; alt=&#34;The Guardrails That Stopped Your AI From Attacking Also Stopped It From Defending You&#34;&gt;&lt;/p&gt;&#xA;&lt;p&gt;An OpenAI model chained a real zero-day, broke out of its own test harness, and reached Hugging Face&amp;rsquo;s production systems. That&amp;rsquo;s not the scary part. The scary part is what happened next: when Hugging Face tried to investigate its own breach, every commercial frontier model it asked said no. The safety filters built to stop an AI from helping an attacker couldn&amp;rsquo;t tell the difference between an attacker and the security team cleaning up after one. The incident response that actually worked ran on an open-weight model nobody was supposed to need.&lt;/p&gt;</description>
    </item>
  </channel>
</rss>
