{
  "schema_version": "paper_public_manifest_v1",
  "paper_id": "transformer-transformer-codesign_2026_08",
  "slug": "transformer-transformer-codesign",
  "title": "Robot failures may not be due to an insufficiently large policy",
  "authors": [],
  "source": {
    "arxiv_id": "",
    "pdf_url": "",
    "project_url": "https://transformer-transformer.github.io",
    "github_url": "",
    "huggingface_url": "",
    "original_source": "https://transformer-transformer.github.io"
  },
  "site": {
    "post_url": "/posts/transformer-transformer-codesign",
    "canonical_url": "https://haiguangboy.com/posts/transformer-transformer-codesign",
    "cover_image": "https://static.haiguangboy.com/papers/transformer-transformer-codesign/cover.webp"
  },
  "taxonomy": {
    "domain": "embodied_ai",
    "track": "world_model",
    "tasks": [
      "embodied_ai",
      "world_model",
      "vla",
      "action_generation",
      "robotics",
      "state_prediction",
      "Embodied intelligence",
      "Robotics",
      "Morphology design",
      "World models"
    ],
    "related_topics": [
      {
        "paper_id": "wx_界面新闻_20260605_2026_06",
        "title": "wx_界面新闻_20260605",
        "url": "https://haiguangboy.com/posts/wx_界面新闻_20260605",
        "relation": "contrast",
        "summary": "Route claim: Task failures should not be defaulted to policy issues; morphology, controller, and policy should enter the same optimization problem contradicts Route bet: Focus on the brain, not the body—betting on the foundation model company identity and rejecting the locomotion capability track",
        "strength": "strong"
      },
      {
        "paper_id": "learning_a_thousand_tasks_in_a_day_2026_08",
        "title": "1,000 tasks in 1 day, relying on inductive biases",
        "url": "https://haiguangboy.com/posts/mt3-thousand-tasks",
        "relation": "same_track",
        "summary": "The design space lacks complex grids, scenes, and contact targets, leaving key gaps for assembly and dexterous contact validates Perception dependency: Vision-only, single camera, no tactile sensing, relying on accurate segmentation",
        "strength": "strong"
      },
      {
        "paper_id": "rl_100_performant_robotic_manipulation_with_real_world_reinforcement_learning_2026_08",
        "title": "RL should not be learned from scratch; it should be post-trained",
        "url": "https://haiguangboy.com/posts/rl-100",
        "relation": "same_track",
        "summary": "Dynamics Self-Guidance: Backpropagating reward gradients from predicting full dynamics to morphology tokens validates Modeling denoising as a two-level MDP, where K denoising steps share the same environment-level advantage, solving weak credit assignment",
        "strength": "strong"
      },
      {
        "paper_id": "latepost_xuhuazhe_202603_2026_03",
        "title": "latepost_xuhuazhe_202603",
        "url": "https://haiguangboy.com/posts/latepost_xuhuazhe_202603",
        "relation": "same_track",
        "summary": "Problem definition: Given end-effector goal motion and reward function, jointly generate full morphology and controller validates Route bet: Behavior/action components must be a unified model, opposing modular assembly",
        "strength": "strong"
      },
      {
        "paper_id": "decompose_and_reorganize_planning_with_primitives_and_visuomotor_policies_learne_2026_08",
        "title": "Switching logic should not be learned by the policy",
        "url": "https://haiguangboy.com/posts/dr-lfd",
        "relation": "same_track",
        "summary": "Route claim: Task failures should not be defaulted to policy issues; morphology, controller, and policy should enter the same optimization problem validates Problem diagnosis: End-to-end policies are forced to learn both 'how to do' and 'when to switch,' causing data requirements to explode combinatorially",
        "strength": "strong"
      },
      {
        "paper_id": "x_manual_pi_rl_20260726_20260727_2026_07",
        "title": "The bottleneck in robot RL is sampling, not algorithms",
        "url": "https://haiguangboy.com/posts/pi-rl-chelsea-finn-talk",
        "relation": "same_track",
        "summary": "'Seconds vs. hours' only holds in the amortized inference phase; new design spaces still incur high expert data costs validates Engineering reading: The difference between robot RL and LLM post-training is the order-of-magnitude gap in sampling cost, not a contest of algorithmic superiority",
        "strength": "strong"
      }
    ]
  },
  "ruling": {
    "importance_score": 3.0,
    "one_sentence": "Robot failures may not be due to an insufficiently large policy"
  },
  "asset_base_url": "https://static.haiguangboy.com/papers/transformer-transformer-codesign",
  "assets": [
    {
      "type": "public_brief",
      "object_key": "papers/transformer-transformer-codesign/public_brief.md",
      "content_type": "text/markdown; charset=utf-8",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/transformer-transformer-codesign/public_brief.md",
      "role": "public_brief",
      "size_bytes": 5285
    },
    {
      "type": "public_manifest",
      "object_key": "papers/transformer-transformer-codesign/public_manifest.json",
      "content_type": "application/json; charset=utf-8",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/transformer-transformer-codesign/public_manifest.json",
      "role": "public_manifest",
      "size_bytes": 5315
    }
  ],
  "published_at": "2026-08-08T12:56:40+08:00",
  "created_at": "2026-08-08T12:56:40+08:00",
  "updated_at": "2026-09-02T10:55:06+08:00",
  "analyst_take": {
    "type": "author_opinion",
    "text": "This challenges the 'focus on the brain, not the body' route. It argues that many failures are not policy-level issues but stem from the body-controller-task distribution not being co-designed. For general-purpose robot companies, this may be too heavy; but for fixed-station, fixed-task-distribution industrial systems, first adjusting mounting angles, linkages, tool geometry, and actuators, then post-training the policy, may be more correct than scaling up the policy model."
  }
}
