<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Ai-Platform on oceanheart.ai</title>
    <link>https://www.oceanheart.ai/tags/ai-platform/</link>
    <description>Recent content in Ai-Platform on oceanheart.ai</description>
    <generator>Hugo</generator>
    <language>en-gb</language>
    <lastBuildDate>Sat, 28 Mar 2026 00:00:00 +0000</lastBuildDate>
    <atom:link href="https://www.oceanheart.ai/tags/ai-platform/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>The Pit</title>
      <link>https://www.oceanheart.ai/projects/the-pit/</link>
      <pubDate>Sat, 28 Mar 2026 00:00:00 +0000</pubDate>
      <guid>https://www.oceanheart.ai/projects/the-pit/</guid>
      <description>&lt;a class=&#34;claude-coach&#34; href=&#34;https://claude.ai/new?q=Read&amp;#43;https%3A%2F%2Fwww.oceanheart.ai%2Fprojects%2Fthe-pit%2F&amp;#43;and&amp;#43;discuss&amp;#43;the&amp;#43;evaluation&amp;#43;design&amp;#43;with&amp;#43;me%3A&amp;#43;what&amp;#43;structured&amp;#43;agent&amp;#43;contests&amp;#43;can&amp;#43;and&amp;#43;cannot&amp;#43;prove%2C&amp;#43;and&amp;#43;how&amp;#43;you&amp;#43;would&amp;#43;extend&amp;#43;the&amp;#43;scoring&amp;#43;and&amp;#43;failure&amp;#43;taxonomy.&#34; target=&#34;_blank&#34; rel=&#34;noopener noreferrer&#34; aria-label=&#34;Open a prepared Claude conversation&#34;&gt;&#xA;  &lt;span class=&#34;claude-coach-mark&#34; aria-hidden=&#34;true&#34;&gt;&amp;#10035;&lt;/span&gt;&#xA;  &lt;span class=&#34;claude-coach-copy&#34;&gt;&#xA;    &lt;span class=&#34;claude-coach-eyebrow&#34;&gt;Claude coach&lt;/span&gt;&#xA;    &lt;strong class=&#34;claude-coach-title&#34;&gt;Interrogate this architecture&lt;/strong&gt;&#xA;    &lt;span class=&#34;claude-coach-description&#34;&gt;Open a Claude conversation primed to discuss what structured agent contests prove, and where the evaluation design could go next.&lt;/span&gt;&#xA;    &lt;span class=&#34;claude-coach-action&#34;&gt;Discuss it with Claude &lt;span aria-hidden=&#34;true&#34;&gt;&amp;#8599;&lt;/span&gt;&lt;/span&gt;&#xA;  &lt;/span&gt;&#xA;&lt;/a&gt;&#xA;&#xA;&lt;h2 id=&#34;what-it-is&#34;&gt;What it is&lt;/h2&gt;&#xA;&lt;p&gt;A multi-agent AI evaluation platform built to make agent performance legible, comparable, governable, and economically inspectable. Structured contests between agent configurations with observable traces, explicit scoring, failure tagging, and cost visibility.&lt;/p&gt;&#xA;&lt;p&gt;The platform demonstrates seven skill domains end-to-end: specification precision, evaluation and quality judgment, decomposition and orchestration, failure pattern recognition, trust and guardrail design, context architecture, and token/cost economics.&lt;/p&gt;</description>
    </item>
  </channel>
</rss>
