<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Synjuku Research</title>
    <link>https://synjuku.ai/blog/</link>
    <atom:link href="https://synjuku.ai/blog/feed.xml" rel="self" type="application/rss+xml" />
    <description>Measured findings on embodied-AI data quality, capture, QC, and deployment, from the team running the loop in production.</description>
    <language>en</language>
    <lastBuildDate>Sat, 01 Aug 2026 12:00:00 GMT</lastBuildDate>
    <item>
      <title>We hand-checked 240 auto-generated captions in a production robotics dataset. One in three was wrong.</title>
      <link>https://synjuku.ai/blog/caption-audit</link>
      <guid isPermaLink="true">https://synjuku.ai/blog/caption-audit</guid>
      <pubDate>Sat, 01 Aug 2026 12:00:00 GMT</pubDate>
      <description>Everyone is benchmarking whether AI can write video labels. We asked whether the labels already shipping in embodied-AI datasets are actually true. On a 136-hour egocentric manipulation dataset, roughly a third of captions failed a human check, the errors concentrate in the verbs, and a carefully calibrated VLM judge catches about half of them with 90% precision, for $3 per hour of footage.</description>
    </item>
  </channel>
</rss>
