{
  "schema_version": "paper_public_manifest_v1",
  "paper_id": "joyai-ra-dual-alignment_2026_08",
  "slug": "joyai-ra-dual-alignment",
  "title": "Massive human videos go unused, and data from different robots can't be combined—this paper solves both problems at once",
  "authors": [],
  "source": {
    "arxiv_id": "2608.05674",
    "pdf_url": "https://arxiv.org/pdf/2608.05674",
    "project_url": "",
    "github_url": "",
    "huggingface_url": "",
    "original_source": "https://arxiv.org/abs/2608.05674"
  },
  "site": {
    "post_url": "/posts/joyai-ra-dual-alignment",
    "canonical_url": "https://haiguangboy.com/posts/joyai-ra-dual-alignment",
    "cover_image": "https://static.haiguangboy.com/papers/joyai-ra-dual-alignment/cover.webp"
  },
  "taxonomy": {
    "domain": "embodied_ai",
    "track": "world_model",
    "tasks": [
      "embodied_ai",
      "world_model",
      "vla",
      "action_generation",
      "robotics",
      "state_prediction",
      "Embodied intelligence",
      "Heterogeneous data"
    ],
    "related_topics": [
      {
        "paper_id": "latent_action_pretraining_through_world_modeling_2026_07",
        "title": "LAWM: Why action labels become a burden",
        "url": "https://haiguangboy.com/posts/latent_action_pretraining_through_world_modeling",
        "relation": "contrast",
        "summary": "LAWM: Why action labels become a burden",
        "strength": "strong"
      },
      {
        "paper_id": "latepost_xuhuazhe_202603_2026_03",
        "title": "latepost_xuhuazhe_202603",
        "url": "https://haiguangboy.com/posts/latepost_xuhuazhe_202603",
        "relation": "contrast",
        "summary": "Architecture: VLM handles semantics, frozen LAC-WM handles dynamics, late fusion feeds into flow-matching action expert contradicts route bet: behavior/action parts must be a unified model, opposing modular assembly",
        "strength": "strong"
      },
      {
        "paper_id": "wx_星河频率_20260718_2026_07",
        "title": "Robots begin to 'stand in the light': Lingchu Intelligence enters optical module production lines",
        "url": "https://haiguangboy.com/posts/lingchu-optical",
        "relation": "contrast",
        "summary": "Pretraining corpus: 53K+ hours of human videos, 11K+ hours of simulation, 8K+ hours of real robot data, covering dual-arm and single-arm multiple embodiments contradicts 'native human data' pyramid claim: pretraining is dominated by human data, real robot data only for post-training adaptation",
        "strength": "strong"
      },
      {
        "paper_id": "an_open_foundation_model_towards_2026_07",
        "title": "An_Open_Foundation_Model_Towards",
        "url": "https://haiguangboy.com/posts/an_open_foundation_model_towards",
        "relation": "same_track",
        "summary": "Core claim: reframing 'insufficient robot data' as 'inconsistent supervision formats across heterogeneous data sources', routing by available supervision type rather than forcing uniform formats validates joint co-training of heterogeneous human-robot data is a structurally suboptimal approach",
        "strength": "strong"
      },
      {
        "paper_id": "beyond_action_residuals_real_world_robot_policy_steering_via_bottleneck_latent_r_2026_08",
        "title": "At which layer should RL intervene",
        "url": "https://haiguangboy.com/posts/zprl",
        "relation": "same_track",
        "summary": "Dual-loop reinforcement learning: fast inner loop at the edge adapts to current tasks, central asynchronous outer loop continuously improves the base model and syncs back to the edge validates problem reframing: the key to RL post-training isn't just 'how much to change', but 'at which layer to intervene'",
        "strength": "strong"
      },
      {
        "paper_id": "t_rex_tactile_reactive_dexterous_manipulation_2026_07",
        "title": "T-Rex: Why touch should be modeled separately",
        "url": "https://haiguangboy.com/posts/t_rex_tactile_reactive_dexterous_manipulation",
        "relation": "same_track",
        "summary": "T-Rex: Why touch should be modeled separately",
        "strength": "strong"
      }
    ]
  },
  "analyst_take": {
    "type": "author_opinion",
    "text": "The strongest point of this paper is the reframing of the problem itself—changing 'insufficient data' to 'inconsistent supervision formats'. This judgment aligns with the direction of several other works, especially the philosophy of 'massive non-action data pretraining + small action data alignment', for which this paper provides a concrete engineering implementation.\n\nBut there's one point worth thinking about further. This paper's ablation shows that removing only the implicit alignment of unlabeled videos doesn't cause the largest drop in generalization—suggesting labeled data is more critical for accuracy. Yet another work's counterintuitive finding is that world model pretraining entirely without action labels can match or even exceed supervised pretraining with labels. Both are empirical results, but in opposite directions. The possible difference lies in how the two works use 'world model pretraining'—one uses it to fill the gap in unseen generalization, the other uses it to entirely replace labeled pretraining, so they're not actually asking the same question."
  },
  "ruling": {
    "importance_score": 3.0,
    "one_sentence": "The problem with heterogeneous data isn't quantity, it's inconsistent supervision formats"
  },
  "asset_base_url": "https://static.haiguangboy.com/papers/joyai-ra-dual-alignment",
  "assets": [
    {
      "type": "pdf_screenshot",
      "object_key": "papers/joyai-ra-dual-alignment/page_01.webp",
      "content_type": "image/webp",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/joyai-ra-dual-alignment/page_01.webp",
      "role": "paper_first_page",
      "size_bytes": 163978
    },
    {
      "type": "pdf_screenshot",
      "object_key": "papers/joyai-ra-dual-alignment/key_figure.webp",
      "content_type": "image/webp",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/joyai-ra-dual-alignment/key_figure.webp",
      "role": "method_figure",
      "size_bytes": 142014
    },
    {
      "type": "cover_image",
      "object_key": "papers/joyai-ra-dual-alignment/cover.webp",
      "content_type": "image/webp",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/joyai-ra-dual-alignment/cover.webp",
      "role": "post_cover",
      "size_bytes": 93866
    },
    {
      "type": "public_brief",
      "object_key": "papers/joyai-ra-dual-alignment/public_brief.md",
      "content_type": "text/markdown; charset=utf-8",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/joyai-ra-dual-alignment/public_brief.md",
      "role": "public_brief",
      "size_bytes": 4650
    },
    {
      "type": "public_manifest",
      "object_key": "papers/joyai-ra-dual-alignment/public_manifest.json",
      "content_type": "application/json; charset=utf-8",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/joyai-ra-dual-alignment/public_manifest.json",
      "role": "public_manifest",
      "size_bytes": 7070
    }
  ],
  "published_at": "2026-08-09T13:29:52+08:00",
  "created_at": "2026-08-09T13:29:52+08:00",
  "updated_at": "2026-09-02T10:55:06+08:00"
}
