{
  "schema_version": "paper_public_manifest_v1",
  "paper_id": "psi-zero_2026_07",
  "slug": "psi-zero",
  "title": "Stacking 10x data still loses 40% -- Psi-Zero says mixed training for humanoid robots is fundamentally wrong",
  "authors": [],
  "source": {
    "arxiv_id": "2603.12263",
    "pdf_url": "https://arxiv.org/pdf/2603.12263",
    "project_url": "",
    "github_url": "",
    "huggingface_url": "",
    "original_source": "https://arxiv.org/abs/2603.12263"
  },
  "site": {
    "post_url": "/posts/psi-zero",
    "canonical_url": "https://haiguangboy.com/posts/psi-zero",
    "cover_image": "https://static.haiguangboy.com/papers/psi-zero/cover.webp"
  },
  "taxonomy": {
    "domain": "embodied_ai",
    "track": "vla",
    "tasks": [
      "embodied_ai",
      "vla",
      "action_generation",
      "robotics",
      "Embodied intelligence",
      "Humanoid robots",
      "Robot learning"
    ],
    "related_topics": []
  },
  "analyst_take": {
    "type": "author_opinion",
    "text": "The truly valuable part of Ψ0 is not that it builds another humanoid robot foundation model, but that it clearly explains why small amounts of robot data can be effective: first use human videos to learn visual representations and task semantics, then use robot data to learn action dynamics, rather than forcibly blending the two distributions into a single end-to-end objective.\n\nThis will influence future VLA engineering trade-offs. Data scale certainly matters, but what matters more is the role of data in the training objective: human videos are suited for shaping representations, robot data for calibrating actions. By separating these responsibilities, small amounts of real data can be amplified."
  },
  "ruling": {
    "importance_score": 3.0,
    "one_sentence": "Policy learning with limited robot data"
  },
  "asset_base_url": "https://static.haiguangboy.com/papers/psi-zero",
  "assets": [
    {
      "type": "pdf_screenshot",
      "object_key": "papers/psi-zero/page_01.webp",
      "content_type": "image/webp",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/psi-zero/page_01.webp",
      "role": "paper_first_page",
      "size_bytes": 159626
    },
    {
      "type": "pdf_screenshot",
      "object_key": "papers/psi-zero/key_figure.webp",
      "content_type": "image/webp",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/psi-zero/key_figure.webp",
      "role": "method_figure",
      "size_bytes": 249104
    },
    {
      "type": "cover_image",
      "object_key": "papers/psi-zero/cover.webp",
      "content_type": "image/webp",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/psi-zero/cover.webp",
      "role": "post_cover",
      "size_bytes": 82470
    },
    {
      "type": "public_brief",
      "object_key": "papers/psi-zero/public_brief.md",
      "content_type": "text/markdown; charset=utf-8",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/psi-zero/public_brief.md",
      "role": "public_brief",
      "size_bytes": 2618
    },
    {
      "type": "public_manifest",
      "object_key": "papers/psi-zero/public_manifest.json",
      "content_type": "application/json; charset=utf-8",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/psi-zero/public_manifest.json",
      "role": "public_manifest",
      "size_bytes": 3571
    }
  ],
  "published_at": "2026-07-07T19:22:32+08:00",
  "created_at": "2026-07-07T19:22:32+08:00",
  "updated_at": "2026-09-02T10:55:06+08:00"
}
