<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Vision on jjshanks.net</title>
    <link>http://www.jjshanks.net/tags/vision/</link>
    <description>Recent content in Vision on jjshanks.net</description>
    <generator>Hugo</generator>
    <language>en</language>
    <lastBuildDate>Wed, 12 Aug 2026 09:00:00 -0800</lastBuildDate>
    <atom:link href="http://www.jjshanks.net/tags/vision/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>Lunchbench: Five Ways to Fail a Lunch Audit</title>
      <link>http://www.jjshanks.net/posts/lunchbench-14-models/</link>
      <pubDate>Wed, 12 Aug 2026 09:00:00 -0800</pubDate>
      <guid>http://www.jjshanks.net/posts/lunchbench-14-models/</guid>
      <description>&lt;p&gt;&lt;strong&gt;Quick Take&lt;/strong&gt; The first real task in &lt;a href=&#34;http://www.jjshanks.net/posts/personal-ai-benchmark/&#34; &gt;my personal AI benchmark&lt;/a&gt; asked 14 models, from cheap open weights to frontier, for a read-only audit of my daughter&amp;rsquo;s daycare lunch calendar, and GPT-5.6 Sol won at 68/72 with the only clean safety record among the models that graded all 72 cells. Cost predicted almost nothing, and the Claude models kept making write requests during an explicitly read-only audit, which surprised me given Anthropic&amp;rsquo;s alignment focus.&lt;/p&gt;</description>
    </item>
  </channel>
</rss>
