{
  "schema_version": "paper_public_manifest_v1",
  "paper_id": "smoothrl_online_reinforcement_learning_during_asynchronous_execution_2026_09",
  "slug": "smoothrl_online_reinforcement_learning_during_asynchronous_execution",
  "title": "SmoothRL:只对真正执行的动作算账",
  "authors": [],
  "source": {
    "arxiv_id": "2608.29768",
    "pdf_url": "https://arxiv.org/pdf/2608.29768",
    "project_url": "",
    "github_url": "",
    "huggingface_url": "",
    "original_source": "https://arxiv.org/pdf/2608.29768"
  },
  "site": {
    "post_url": "/posts/smoothrl_online_reinforcement_learning_during_asynchronous_execution",
    "canonical_url": "https://haiguangboy.com/posts/smoothrl_online_reinforcement_learning_during_asynchronous_execution",
    "cover_image": "https://static.haiguangboy.com/papers/smoothrl_online_reinforcement_learning_during_asynchronous_execution/cover.webp"
  },
  "taxonomy": {
    "domain": "embodied_ai",
    "track": "vla",
    "tasks": [
      "embodied_ai",
      "vla",
      "action_generation",
      "robotics",
      "state_prediction",
      "具身智能",
      "强化学习"
    ],
    "related_topics": [
      {
        "paper_id": "rl_100_performant_robotic_manipulation_with_real_world_reinforcement_learning_2026_08",
        "title": "RL 不该从零学，该做后训练",
        "url": "https://haiguangboy.com/posts/rl-100",
        "relation": "contrast",
        "summary": "真机三项任务,250轮rollout在线强化学习后相对冻结基座策略均大幅提升:动态投掷39%到94%,套笔帽8%到83%,开箱30%到90%;基座策略的失败是系统性的固定偏差(不是随机误差),在线RL主要在修正这些系统性偏差 contradicts 真机八任务 1000/1000 次试验 100% 成功;离线 RL 阶段贡献了大部分提升,在线阶段只清理残余…",
        "strength": "strong"
      },
      {
        "paper_id": "beyond_action_residuals_real_world_robot_policy_steering_via_bottleneck_latent_r_2026_08",
        "title": "RL 该在哪一层介入",
        "url": "https://haiguangboy.com/posts/zprl",
        "relation": "contrast",
        "summary": "预训练策略的平滑性来自示范数据,一旦价值目标开始在执行段上推高Q值,这个平滑性保证就不再自动成立——必须把平滑性重新作为显式约束(逐帧速度/加速度/加加速度的边界惩罚项)加回优化目标 contradicts 机制解释:引导是在重映射状态-动作关联,而不是发明新动作",
        "strength": "strong"
      },
      {
        "paper_id": "n_0_vtla_scaling_vision_tactile_language_action_model_with_latent_tactile_tokens_2026_07",
        "title": "触觉当预测目标而非观测输入",
        "url": "https://haiguangboy.com/posts/n_0_vtla_scaling_vision_tactile_language_action_model_with_latent_tactile_tokens",
        "relation": "contrast",
        "summary": "触觉当预测目标而非观测输入",
        "strength": "strong"
      },
      {
        "paper_id": "an_open_foundation_model_towards_2026_07",
        "title": "An_Open_Foundation_Model_Towards",
        "url": "https://haiguangboy.com/posts/an_open_foundation_model_towards",
        "relation": "same_track",
        "summary": "异步推理循环在训练时的rollout过程里就实际运行,不是只在部署阶段才用——这样回放数据里记录的是机器人在和目标函数完全一致的时序和执行调度下真正执行过的动作,确保策略永远不会在部署时不会遇到的动力学条件下被优化 validates 训练时RTC：遮掩前d个动作token使模型学会平滑续接",
        "strength": "strong"
      },
      {
        "paper_id": "mathcaln_0_foundation_towards_the_age_of_tactile_intelligence_2026_09",
        "title": "N0-Foundation:开启触觉新时代",
        "url": "https://haiguangboy.com/posts/mathcaln_0_foundation_towards_the_age_of_tactile_intelligence",
        "relation": "same_track",
        "summary": "N0-Foundation:开启触觉新时代",
        "strength": "strong"
      },
      {
        "paper_id": "learning_native_continuation_for_action_chunking_flow_policies_2026_09",
        "title": "VLA换动作块不再卡顿",
        "url": "https://haiguangboy.com/posts/learning_native_continuation_for_action_chunking_flow_policies",
        "relation": "same_track",
        "summary": "VLA换动作块不再卡顿",
        "strength": "strong"
      }
    ]
  },
  "analyst_take": {
    "type": "author_opinion",
    "text": "这篇和已经解读过的RL-100形成一个值得记的分歧：RL-100自己的发现是离线RL阶段贡献了大部分提升，在线阶段只是清理残余失败；这篇却把在线RL当成修正系统性偏差（不是残余噪声）的核心手段。两边对\"在线RL到底该扛多重的活\"给出不同答案，可能取决于预训练阶段本身把策略带到了多接近成功。另外，这篇\"训练时把异步循环也跑起来，让优化目标和部署时的真实时序对齐\"这个原则，和刚解读过的Legato是同一个方向——两篇论文都在强调，训练阶段绝不能对部署时会遇到的时序结构视而不见。"
  },
  "ruling": {
    "importance_score": 3.0,
    "one_sentence": "SmoothRL:只对真正执行的动作算账"
  },
  "asset_base_url": "https://static.haiguangboy.com/papers/smoothrl_online_reinforcement_learning_during_asynchronous_execution",
  "assets": [
    {
      "type": "pdf_screenshot",
      "object_key": "papers/smoothrl_online_reinforcement_learning_during_asynchronous_execution/page_01.webp",
      "content_type": "image/webp",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/smoothrl_online_reinforcement_learning_during_asynchronous_execution/page_01.webp",
      "role": "paper_first_page",
      "size_bytes": 157610
    },
    {
      "type": "pdf_screenshot",
      "object_key": "papers/smoothrl_online_reinforcement_learning_during_asynchronous_execution/key_figure.webp",
      "content_type": "image/webp",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/smoothrl_online_reinforcement_learning_during_asynchronous_execution/key_figure.webp",
      "role": "method_figure",
      "size_bytes": 176936
    },
    {
      "type": "cover_image",
      "object_key": "papers/smoothrl_online_reinforcement_learning_during_asynchronous_execution/cover.webp",
      "content_type": "image/webp",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/smoothrl_online_reinforcement_learning_during_asynchronous_execution/cover.webp",
      "role": "post_cover",
      "size_bytes": 98260
    },
    {
      "type": "public_brief",
      "object_key": "papers/smoothrl_online_reinforcement_learning_during_asynchronous_execution/public_brief.md",
      "content_type": "text/markdown; charset=utf-8",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/smoothrl_online_reinforcement_learning_during_asynchronous_execution/public_brief.md",
      "role": "public_brief",
      "size_bytes": 5513
    },
    {
      "type": "public_brief",
      "object_key": "papers/smoothrl_online_reinforcement_learning_during_asynchronous_execution/public_brief.en.md",
      "content_type": "text/markdown; charset=utf-8",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/smoothrl_online_reinforcement_learning_during_asynchronous_execution/public_brief.en.md",
      "role": "public_brief_en",
      "size_bytes": 7286
    },
    {
      "type": "public_manifest",
      "object_key": "papers/smoothrl_online_reinforcement_learning_during_asynchronous_execution/public_manifest.en.json",
      "content_type": "application/json; charset=utf-8",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/smoothrl_online_reinforcement_learning_during_asynchronous_execution/public_manifest.en.json",
      "role": "public_manifest_en",
      "size_bytes": 8349
    },
    {
      "type": "public_manifest",
      "object_key": "papers/smoothrl_online_reinforcement_learning_during_asynchronous_execution/public_manifest.json",
      "content_type": "application/json; charset=utf-8",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/smoothrl_online_reinforcement_learning_during_asynchronous_execution/public_manifest.json",
      "role": "public_manifest",
      "size_bytes": 8894
    }
  ],
  "published_at": "2026-09-13T08:06:20+08:00",
  "created_at": "2026-09-13T08:06:20+08:00",
  "updated_at": "2026-09-13T08:06:34+08:00"
}
