{
  "schema_version": "paper_public_manifest_v1",
  "paper_id": "towards_generalizable_visually_grounded_exploration_of_household_devices_2026_09",
  "slug": "towards_generalizable_visually_grounded_exploration_of_household_devices",
  "title": "VGEBench:可泛化视觉探索",
  "authors": [],
  "source": {
    "arxiv_id": "2609.00845",
    "pdf_url": "https://arxiv.org/pdf/2609.00845",
    "project_url": "",
    "github_url": "",
    "huggingface_url": "",
    "original_source": "https://arxiv.org/abs/2609.00845"
  },
  "site": {
    "post_url": "/posts/towards_generalizable_visually_grounded_exploration_of_household_devices",
    "canonical_url": "https://haiguangboy.com/posts/towards_generalizable_visually_grounded_exploration_of_household_devices",
    "cover_image": "https://static.haiguangboy.com/papers/towards_generalizable_visually_grounded_exploration_of_household_devices/cover.webp"
  },
  "taxonomy": {
    "domain": "embodied_ai",
    "track": "world_model",
    "tasks": [
      "embodied_ai",
      "world_model",
      "action_generation",
      "robotics",
      "state_prediction",
      "具身智能",
      "vlm智能体",
      "视觉定位"
    ],
    "related_topics": [
      {
        "paper_id": "learning_a_thousand_tasks_in_a_day_2026_08",
        "title": "1 天 1000 任务，靠的是归纳偏置",
        "url": "https://haiguangboy.com/posts/mt3-thousand-tasks",
        "relation": "same_track",
        "summary": "作者判断:导航探索能力和精细视觉定位能力是两种可分离的能力,前者强不代表后者也强,而这类任务的成败由后者决定 validates 感知依赖:仅视觉、单相机,无触觉,依赖准确分割",
        "strength": "strong"
      },
      {
        "paper_id": "mathcaln_0_foundation_towards_the_age_of_tactile_intelligence_2026_09",
        "title": "N0-Foundation:开启触觉新时代",
        "url": "https://haiguangboy.com/posts/mathcaln_0_foundation_towards_the_age_of_tactile_intelligence",
        "relation": "same_track",
        "summary": "N0-Foundation:开启触觉新时代",
        "strength": "strong"
      },
      {
        "paper_id": "transformer-transformer_blog_transformer_transformer_a_unified_20260807_2026_08",
        "title": "Transformer Transformer: A Unified Model for Motion-Conditioned Robot Co-design",
        "url": "https://haiguangboy.com/posts/transformer-transformer-codesign",
        "relation": "same_track",
        "summary": "核心定位判断:无说明书的可泛化视觉引导式探索是VLM智能体现存的一个关键能力缺口,既不是传统模仿/强化学习机器人基准能覆盖的,也不是依赖显式API/说明文档的LLM工具调用范式能覆盖的 validates 输入仍是目标末端轨迹和手写奖励，不是从语言与场景直接理解任务",
        "strength": "strong"
      },
      {
        "paper_id": "wx_elsewhere别处发生_20260722_2026_07",
        "title": "梁文锋四小时投资人会议实录",
        "url": "https://haiguangboy.com/posts/liangwenfeng-world-model",
        "relation": "same_track",
        "summary": "核心定位判断:无说明书的可泛化视觉引导式探索是VLM智能体现存的一个关键能力缺口,既不是传统模仿/强化学习机器人基准能覆盖的,也不是依赖显式API/说明文档的LLM工具调用范式能覆盖的 validates ★★核心判断：「世界模型跟智能上限无关」的隐含前提是「数据管道已通」——LLM通了所以是绕路，机器人没通所以是桥",
        "strength": "strong"
      },
      {
        "paper_id": "what_matters_for_latent_actions_in_robot_learning_2026_08",
        "title": "潜在动作:光流是负资产",
        "url": "https://haiguangboy.com/posts/what_matters_for_latent_actions_in_robot_learning",
        "relation": "same_track",
        "summary": "潜在动作:光流是负资产",
        "strength": "strong"
      },
      {
        "paper_id": "real_time_semg_based_telecontrol_of_an_assistive_robotic_arm_using_a_1d_convolut_2026_07",
        "title": "神经信号：它的价值和它的脆弱，来自同一处",
        "url": "https://haiguangboy.com/posts/real_time_semg_based_telecontrol_of_an_assistive_robotic_arm_using_a_1d_convolut",
        "relation": "same_track",
        "summary": "神经信号：它的价值和它的脆弱，来自同一处",
        "strength": "strong"
      }
    ]
  },
  "analyst_take": {
    "type": "author_opinion",
    "text": "最有意思的一条线：\"给系统更多信息/更长上下文不是无条件的好事\"这个判断，在完全不同的领域反复被独立证实。这篇发现更长的交互历史只对强模型有用，弱模型反而在短历史下表现更好——因为弱模型分不清有用反馈和历史噪声。这和之前解读过的几篇工作是同一个方向：What Matters for Latent Actions里，更懂运动的光流表示反而拖累效果；TacWAM发现让辅助模态\"可见\"比缺少信息损失更大；FLEX-π强调关键机制是强制重建缺失流本身，不只是\"看过更多模态\"。跨语言模型、触觉表示、潜在动作、VLM智能体四个不同领域，同一个判断反复出现：接收方有没有能力筛选和整合信息，比给多少信息更重要。另外，这篇对\"没手册的探索能力\"的定位判断，也呼应了另一条更大的判断——世界模型在机器人领域比在语言模型领域更被需要，隐含的前提正是机器人的\"数据管道\"还没打通，需要额外的桥，而不是像LLM那样可以绕过去。"
  },
  "ruling": {
    "importance_score": 3.0,
    "one_sentence": "VGEBench:可泛化视觉探索"
  },
  "asset_base_url": "https://static.haiguangboy.com/papers/towards_generalizable_visually_grounded_exploration_of_household_devices",
  "assets": [
    {
      "type": "pdf_screenshot",
      "object_key": "papers/towards_generalizable_visually_grounded_exploration_of_household_devices/page_01.webp",
      "content_type": "image/webp",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/towards_generalizable_visually_grounded_exploration_of_household_devices/page_01.webp",
      "role": "paper_first_page",
      "size_bytes": 218862
    },
    {
      "type": "pdf_screenshot",
      "object_key": "papers/towards_generalizable_visually_grounded_exploration_of_household_devices/key_figure.webp",
      "content_type": "image/webp",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/towards_generalizable_visually_grounded_exploration_of_household_devices/key_figure.webp",
      "role": "method_figure",
      "size_bytes": 205140
    },
    {
      "type": "cover_image",
      "object_key": "papers/towards_generalizable_visually_grounded_exploration_of_household_devices/cover.webp",
      "content_type": "image/webp",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/towards_generalizable_visually_grounded_exploration_of_household_devices/cover.webp",
      "role": "post_cover",
      "size_bytes": 106570
    },
    {
      "type": "public_brief",
      "object_key": "papers/towards_generalizable_visually_grounded_exploration_of_household_devices/public_brief.md",
      "content_type": "text/markdown; charset=utf-8",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/towards_generalizable_visually_grounded_exploration_of_household_devices/public_brief.md",
      "role": "public_brief",
      "size_bytes": 5250
    },
    {
      "type": "public_brief",
      "object_key": "papers/towards_generalizable_visually_grounded_exploration_of_household_devices/public_brief.en.md",
      "content_type": "text/markdown; charset=utf-8",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/towards_generalizable_visually_grounded_exploration_of_household_devices/public_brief.en.md",
      "role": "public_brief_en",
      "size_bytes": 6550
    },
    {
      "type": "public_manifest",
      "object_key": "papers/towards_generalizable_visually_grounded_exploration_of_household_devices/public_manifest.en.json",
      "content_type": "application/json; charset=utf-8",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/towards_generalizable_visually_grounded_exploration_of_household_devices/public_manifest.en.json",
      "role": "public_manifest_en",
      "size_bytes": 8781
    },
    {
      "type": "public_manifest",
      "object_key": "papers/towards_generalizable_visually_grounded_exploration_of_household_devices/public_manifest.json",
      "content_type": "application/json; charset=utf-8",
      "upload_status": "uploaded",
      "bucket": "paper-assets",
      "url": "https://static.haiguangboy.com/papers/towards_generalizable_visually_grounded_exploration_of_household_devices/public_manifest.json",
      "role": "public_manifest",
      "size_bytes": 9381
    }
  ],
  "published_at": "2026-09-07T09:31:05+08:00",
  "created_at": "2026-09-07T09:31:05+08:00",
  "updated_at": "2026-09-07T09:31:22+08:00"
}
