{
  "schema_version": "paper_public_manifest_v1",
  "paper_id": "rl-100_2026_08",
  "slug": "rl-100",
  "title": "RL Should Not Be Learned from Scratch, but Should Be Post-Trained",
  "authors": [],
  "source": {
    "arxiv_id": "2510.14830",
    "pdf_url": "https://arxiv.org/pdf/2510.14830",
    "project_url": "",
    "github_url": "",
    "huggingface_url": "",
    "original_source": "https://arxiv.org/abs/2510.14830"
  },
  "site": {
    "post_url": "/posts/rl-100",
    "canonical_url": "https://haiguangboy.com/posts/rl-100",
    "cover_image": "https://static.haiguangboy.com/papers/rl-100/cover.webp"
  },
  "taxonomy": {
    "domain": "embodied_ai",
    "track": "action_generation",
    "tasks": [
      "embodied_ai",
      "action_generation",
      "robotics",
      "Embodied Intelligence",
      "Reinforcement Learning",
      "Robot Policy"
    ],
    "related_topics": [
      {
        "paper_id": "learning_a_thousand_tasks_in_a_day_2026_08",
        "title": "1,000 Tasks in 1 Day, Thanks to Inductive Bias",
        "url": "https://haiguangboy.com/posts/mt3-thousand-tasks",
        "relation": "same_track",
        "summary": "Data budget: human demonstrations average only 115 episodes per task (1.8 hours), accounting for less than 13% of the total data collection budget, validating the main result of the controlled experiment: MT3 with 3 demonstrations outperforms other methods with 50.",
        "strength": "strong"
      },
      {
        "paper_id": "latent_action_pretraining_through_world_modeling_2026_07",
        "title": "LAWM: Why Action Labels Become a Burden",
        "url": "https://haiguangboy.com/posts/latent_action_pretraining_through_world_modeling",
        "relation": "same_track",
        "summary": "LAWM: Why Action Labels Become a Burden",
        "strength": "strong"
      },
      {
        "paper_id": "beyond_action_residuals_real_world_robot_policy_steering_via_bottleneck_latent_r_2026_08",
        "title": "At Which Level Should RL Intervene",
        "url": "https://haiguangboy.com/posts/zprl",
        "relation": "same_track",
        "summary": "In the offline phase, an approximate model Q-function is used as a gate: only when the prediction shows clear improvement is the behavior policy advanced; otherwise, the update is rejected, validating the core claim: the choice of intervention level is itself a key design variable, not an implementation detail.",
        "strength": "strong"
      },
      {
        "paper_id": "orca_2026_07",
        "title": "When π0.5 fails to grab a spoon, it trembles in place, but Orca goes further with physical intuition learned from watching videos",
        "url": "https://haiguangboy.com/posts/orca",
        "relation": "same_track",
        "summary": "The key to a world model is a readable state",
        "strength": "strong"
      },
      {
        "paper_id": "decompose_and_reorganize_planning_with_primitives_and_visuomotor_policies_learne_2026_08",
        "title": "Switching logic should not be learned by the policy",
        "url": "https://haiguangboy.com/posts/dr-lfd",
        "relation": "same_track",
        "summary": "Data budget: human demonstrations average only 115 episodes per task (1.8 hours), accounting for less than 13% of the total data collection budget, validating the results: 100% under simulated peg-in-hole ID vs. 44% for ACT and 54% for DP; on DexMimicGen, 100 demonstrations outperform the baseline's 1,000.",
        "strength": "strong"
      },
      {
        "paper_id": "fast-wam_2026_07",
        "title": "No need to imagine the future at inference time—the robot still reaches 91.8%! Fast-WAM debunks WAM's core assumption",
        "url": "https://haiguangboy.com/posts/fast-wam",
        "relation": "same_track",
        "summary": "Training video objectives matter more than imagining the future at test time",
        "strength": "strong"
      }
    ]
  },
  "ruling": {
    "importance_score": 3.0,
    "one_sentence": "RL Should Not Be Learned from Scratch, but Should Be Post-Trained"
  },
  "asset_base_url": "https://static.haiguangboy.com/papers/rl-100",
  "assets": [
    {
      "type": "pdf_screenshot",
      "object_key": "papers/rl-100/page_01.webp",
      "content_type": "image/webp",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/rl-100/page_01.webp",
      "role": "paper_first_page",
      "size_bytes": 188422
    },
    {
      "type": "pdf_screenshot",
      "object_key": "papers/rl-100/key_figure.webp",
      "content_type": "image/webp",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/rl-100/key_figure.webp",
      "role": "method_figure",
      "size_bytes": 249360
    },
    {
      "type": "cover_image",
      "object_key": "papers/rl-100/cover.webp",
      "content_type": "image/webp",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/rl-100/cover.webp",
      "role": "post_cover",
      "size_bytes": 69508
    },
    {
      "type": "public_brief",
      "object_key": "papers/rl-100/public_brief.md",
      "content_type": "text/markdown; charset=utf-8",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/rl-100/public_brief.md",
      "role": "public_brief",
      "size_bytes": 4709
    },
    {
      "type": "public_manifest",
      "object_key": "papers/rl-100/public_manifest.json",
      "content_type": "application/json; charset=utf-8",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/rl-100/public_manifest.json",
      "role": "public_manifest",
      "size_bytes": 5527
    }
  ],
  "published_at": "2026-08-07T16:03:15+08:00",
  "created_at": "2026-08-07T16:03:15+08:00",
  "updated_at": "2026-09-02T10:55:06+08:00",
  "analyst_take": {
    "type": "author_opinion",
    "text": "The core claim of this work—that RL should be post-trained around deployment metrics rather than learned from scratch or purely approximating demonstrations—is independently corroborated by two other findings. One is simulation-first, with pretraining barely touching the real robot and a small amount of real-robot RL filling gaps after deployment; the other inverts the data pyramid, using human data as the foundation rather than internet video. The three approaches start from different points but converge to the same shape: strong priors plus lightweight real-robot post-training.\n\nThe offline gating also connects to another recent work—whose conclusion is that the level at which RL intervenes is itself a design variable as important as how much to change. The gating mechanism in this work is a concrete implementation of that 'intervention-level choice.'"
  }
}
