<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom" xmlns:dc="http://purl.org/dc/elements/1.1/">
  <channel>
    <title>Taktile Labs</title>
    <link>https://labs.taktile.com</link>
    <description>Open benchmarks, evaluation frameworks, and applied research for deploying AI in regulated financial institutions.</description>
    <language>en</language>
    <lastBuildDate>Tue, 29 Sep 2026 00:00:00 GMT</lastBuildDate>
    <atom:link href="https://labs.taktile.com/feed.xml" rel="self" type="application/rss+xml"/>
    <item>
      <title>UBO-Bench: Evaluating AI Agents for Beneficial Ownership Checks</title>
      <link>https://labs.taktile.com/benchmarks/ubo-bench</link>
      <guid isPermaLink="true">https://labs.taktile.com/benchmarks/ubo-bench</guid>
      <category>Benchmark</category>
      <pubDate>Tue, 29 Sep 2026 00:00:00 GMT</pubDate>
      <dc:creator>Moritz Geist, David Ahn, Maximilian Eber, PhD</dc:creator>
      <description>A benchmark of AI agents that trace ownership structures for beneficial ownership checks on UK and German registers. Measures tracing accuracy across frontier models, importance of country specific configuration, and how much of the work agents complete on their own.</description>
    </item>
    <item>
      <title>KYBench update: 25 models, GPT-6 Astra leads on cost per business</title>
      <link>https://labs.taktile.com/benchmarks/kybench</link>
      <guid isPermaLink="false">https://labs.taktile.com/benchmarks/kybench#update-2026-09-07</guid>
      <category>Update</category>
      <pubDate>Mon, 07 Sep 2026 00:00:00 GMT</pubDate>
    </item>
    <item>
      <title>PIBench: Prompt Injection Resistance in Agentic Underwriting</title>
      <link>https://labs.taktile.com/benchmarks/pibench</link>
      <guid isPermaLink="true">https://labs.taktile.com/benchmarks/pibench</guid>
      <category>Benchmark</category>
      <pubDate>Tue, 30 Jun 2026 00:00:00 GMT</pubDate>
      <dc:creator>Koen Roelofs, Jakob Schmitt, Maximilian Eber, PhD</dc:creator>
      <description>The first benchmark of prompt-injection resistance for agentic underwriting. Measures defense success across 16 frontier models, three providers, and five attack vectors — with and without untrusted-content tagging.</description>
    </item>
    <item>
      <title>KYBench: Evaluating AI Agents for Adverse Media Research</title>
      <link>https://labs.taktile.com/benchmarks/kybench</link>
      <guid isPermaLink="true">https://labs.taktile.com/benchmarks/kybench</guid>
      <category>Benchmark</category>
      <pubDate>Thu, 02 Apr 2026 00:00:00 GMT</pubDate>
      <dc:creator>David Ahn, Maximilian Eber, PhD, Sahith Jagarlamudi</dc:creator>
      <description>The first public benchmark of AI-driven adverse media investigation. Evaluates detection accuracy, evidence quality, reliability across agent runs, and cost efficiency across frontier models.</description>
    </item>
    <item>
      <title>FinSpread-Bench: Evaluating Agentic AI for Financial Spreading</title>
      <link>https://labs.taktile.com/benchmarks/finspread</link>
      <guid isPermaLink="true">https://labs.taktile.com/benchmarks/finspread</guid>
      <category>Benchmark</category>
      <pubDate>Tue, 10 Mar 2026 00:00:00 GMT</pubDate>
      <dc:creator>Nico Klees, Maximilian Eber, PhD</dc:creator>
      <description>The first public benchmark for agentic financial document processing. Evaluates extraction accuracy, cross-document reasoning, calculation correctness, and structured output quality across seven frontier models. Built on anonymized production data from financial institutions.</description>
    </item>
    <item>
      <title>AI in AML: A guide to governance and implementation</title>
      <link>https://labs.taktile.com/publications/ai-in-aml</link>
      <guid isPermaLink="true">https://labs.taktile.com/publications/ai-in-aml</guid>
      <category>Paper</category>
      <pubDate>Sun, 01 Mar 2026 00:00:00 GMT</pubDate>
      <dc:creator>Dustin Eaton, Maximilian Eber, PhD</dc:creator>
      <description>Why AML teams must now apply model risk management standards to AI systems. Published in ACAMS Today, exploring how regulators are extending MRM frameworks to AI deployed in compliance functions — and what institutions need to do to prepare.</description>
    </item>
  </channel>
</rss>
