From 7519309a275e8f88aa53f4acabc08cc1c385a1ba Mon Sep 17 00:00:00 2001 From: cen617-code <1057290604@qq.com> Date: Thu, 24 Sep 2026 17:02:10 +0800 Subject: [PATCH] =?UTF-8?q?fix:=20release=20v1.0.3=20=E9=A2=84=E8=AE=BE?= =?UTF-8?q?=E6=B8=B2=E6=9F=93=E4=B8=8E=E8=AF=AD=E8=A8=80=E6=90=AC=E8=BF=90?= =?UTF-8?q?=E7=9B=91=E7=9D=A3=E4=BF=AE=E5=A4=8D?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- CHANGELOG.md | 12 ++ decision_server/providers/jev.py | 123 ++++++++++++++++-- decision_server/tests/test_jev_context.py | 75 +++++++++++ docs/website-deployment.md | 2 +- docs/website-language-render-fix.md | 85 ++++++++++++ package-lock.json | 4 +- package.json | 2 +- .../e2e/lekiwi.agent.workspace.spec.ts | 36 +++++ web_platform/e2e/website.live.spec.ts | 91 ++++++++++++- .../playwright.website-live.config.ts | 7 +- .../mobile/agent/AgentTaskController.test.ts | 22 ++++ .../src/mobile/agent/AgentTaskController.ts | 3 +- .../src/mobile/agent/TaskSupervisor.test.ts | 26 ++++ .../src/mobile/agent/TaskSupervisor.ts | 17 ++- web_platform/src/viewer/MuJoCoViewer.ts | 28 ++-- web_platform/src/viewer/renderQuality.test.ts | 72 ++++++++++ web_platform/src/viewer/renderQuality.ts | 45 +++++++ 17 files changed, 623 insertions(+), 27 deletions(-) create mode 100644 decision_server/tests/test_jev_context.py create mode 100644 docs/website-language-render-fix.md create mode 100644 web_platform/src/viewer/renderQuality.test.ts create mode 100644 web_platform/src/viewer/renderQuality.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index a8b8ed66..36fcc017 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,18 @@ ## [未发布] +## [1.0.3] - 2026-09-24 + +### 预设场景清晰度与语言搬运监督修复 + +- 将已上线的预设渲染与语言搬运修复归档为 `v1.0.3`,同步 npm 包与锁文件版本;提交仅包含源码、测试及文档,不纳入密钥、运行预算、构建产物和测试日志。 +- 提交前复验:TypeScript、137 文件/675 项前端单测、48 项后端测试(另有 3 项既有可选门禁跳过)、定向 TS/Python lint 和变更文件格式检查通过;本次归档不重复部署或付费推理。 + +- 主视口和腕部/前置相机采用 2048² 场景适配柔和阴影、纹理 mipmap/各向异性过滤及有上限的 GPU 超采样;修复 DPR 同时放大 CSS 画布导致裁切、抵消子视口清晰度的问题。保留全部 CAD 网格、物理参数和软件 WebGL 的控制延迟预算。 +- Jev 请求补充下一技能含义、阶段前置条件、相邻技能完成上下文和等待期间物理冻结语义,避免将抓取前空爪、抬升验证前未 secure、放置后恢复支撑误当故障。保留官方 choice 与 deny/uncertain/stop,不重排概率、不强行放行。Noul/Choice 停止显示中文原因兜底,终态保留被否决的决策以便诊断。 +- 新增默认预设真实 API 专项入口(显式授权、最多 2 次 LLM/14 次 Jev、无自动重试)、高 DPI 双相机及监督语义回归。675 项前端单测通过;硬件浏览器真实 DeepSeek+Jev 完成默认 B 区搬运 0.728882 m。中途模型误停、上游网络中断及软件渲染并行测试超时如实保留,非稳定性/多种子证明;详见 `docs/website-language-render-fix.md`。 +- 用户授权后同步发布网站前后端为 `20260924T081555Z-language-render-fix`,上一版 `20260924T065821Z-language-v2` 保留。发布包/线上静态资源及运行时 Jev 源码哈希一致、容器健康;公网非付费检查 6/6、RTX 高 DPI 双相机验证通过。密钥、持久预算和 iframe 配置保持不变;本次发布验证没有新增付费推理,随后按用户要求归档为 Git 版本 `v1.0.3`。 + ## [1.0.2] - 2026-09-24 ### LeKiwi 语言控制、内置机器人与网站嵌入 diff --git a/decision_server/providers/jev.py b/decision_server/providers/jev.py index 500f0525..ac7854fe 100644 --- a/decision_server/providers/jev.py +++ b/decision_server/providers/jev.py @@ -6,15 +6,70 @@ Protocol shape informed by MIT-licensed jev-libero / embodied-jev; see THIRD_PAR import json import math -from ..protocol import LANGUAGE_VERSION, DecisionError, schema_for, validate_jev +from ..protocol import ( + LANGUAGE_SKILLS, + LANGUAGE_VERSION, + SKILLS, + DecisionError, + schema_for, + validate_jev, +) from .http import post, usage +# Criteria are the decision model's option semantics, not just display labels. +# In particular, `close` means close the gripper, NOT close/terminate the task. +SKILL_CRITERIA = { + "stow": "Raise the empty arm to its safe travel pose before approaching the source.", + "approach": ( + "After completedSkill=stow with failure=none, the arm has reached its safe travel pose. " + "Proceed to dock the base at the source; the gripper should still be empty." + ), + "open": "Open the gripper before grasping a supported block (also safe recovery).", + "pregrasp": "Move the TCP above the block while the base is stopped.", + "descend": "Lower the open gripper from above to align its TCP with the block.", + "close": ( + "Close the gripper fingers around the aligned, supported block; TCP-object distance " + "must be <=0.012 m. Zero finger forces and secure=false BEFORE closing are expected." + ), + "verify": ( + "Lift to verify grasp AFTER closing, with both finger forces >=0.2 N. " + "secure=false is expected BEFORE this lift; this skill establishes verified grasp." + ), + "carry": "Transport the block with verified grasp: evidence.secure=true is required.", + "stop-base": "Stop at the target dock while maintaining evidence.secure=true.", + "place": "Lower the transported block onto the target support with the base stopped.", + "release": ( + "Open the fingers after evidence.onGoalSupport=true. secure may already be false " + "because the block is now supported; this is expected, not slipping." + ), + "retreat": "Withdraw the empty gripper after releasing the supported block.", + "settle": "Hold still to measure released block stability on the target support.", + "stop": ( + "Choose stop when failure indicates a hard fault, safety flags are present, or the " + "next skill's physical precondition is contradicted or cannot be established; " + "NOT merely because the whole task is unfinished or the gripper is empty before grasp." + ), +} +STAGE_CONTEXT = ( + "Judge the offered NEXT skill, not completion of the whole task. observation.phase names " + "the current/just-completed skill; candidates names the next permitted skill or stop. " + "transition.completedSkill means the local executor completed that skill and its exit " + "checks without failure; null means initial entry or recovery, not completion. " + "The local simulator freezes physics while waiting for this response; time is simulation " + "seconds, not a wall-clock timestamp. The stamped observation is the current snapshot. " + "An empty gripper before close is normal. verify establishes secure grasp by lifting; " + "only carry/stop-base require an already verified grasp. During place/release the object " + "returns to support, so secure=false there is not by itself a lost grasp. " + "Never ignore an explicit failure, safety flag or missing required evidence. " + "The local controller independently checks physical preconditions before any actuation. " +) -def question(options, instruction): + +def question(options, instruction, criteria=None): return { "type": "choice", "instructions": instruction, - "criteria": {option: option for option in options}, + "criteria": {option: (criteria or {}).get(option, option) for option in options}, } @@ -44,8 +99,8 @@ async def decide(session, conn, request): options["choice"] = request["candidates"] prompts = { "choice": ( - "Choose an offered next skill if safe. Stop for hard safety faults. " - "Only the local controller can determine success or permit actuation." + STAGE_CONTEXT + "Choose the safe offered next skill according to its criteria. " + "Stop when its required evidence is absent or a hard safety fault is present." ), "grasp": ( "secure requires two finger forces >=0.2 N, verified lift and stable grasp evidence. " @@ -68,8 +123,9 @@ async def decide(session, conn, request): prompts.update( { "noul": ( - "Veto unsafe actions: allow only with fresh sufficient evidence and no hard " - "safety fault; deny on danger, uncertain on missing evidence. " + STAGE_CONTEXT + "Veto unsafe NEXT skills, not incomplete tasks: allow with " + "sufficient evidence for the offered stage and no hard safety fault; " + "deny on danger, uncertain on missing evidence required for that stage. " "This is not physical authorization." ), "score": ( @@ -83,6 +139,46 @@ async def decide(session, conn, request): ), } ) + observation = request["observation"] + candidates = [skill for skill in request["candidates"] if skill != "stop"] + next_skill = candidates[0] if len(candidates) == 1 else None + failure = request.get("failure", "none") + sequence = LANGUAGE_SKILLS if version == LANGUAGE_VERSION else SKILLS + adjacent = ( + next_skill in sequence + and sequence.index(next_skill) == sequence.index(observation["phase"]) + 1 + ) + # This is the caller's skill-boundary protocol, not a new safety authorization. + # Preserve raw observations and vetoes; never infer grasp from task progress. + transition = { + "nextSkill": next_skill, + "completedSkill": ( + observation["phase"] + if adjacent and failure == "none" and not observation["safety"] + else None + ), + } + criteria = { + "choice": SKILL_CRITERIA, + "grasp": { + "secure": "evidence.secure=true and both fingerForces >=0.2 N; verified lifted grasp.", + "empty": "No held block; normal before closing and after release.", + "slipping": "Previously held block is being lost, not intentionally placed on support.", + "uncertain": "Grasp not yet verified: fingers closed but lift verification pending.", + }, + "noul": { + "allow": "The offered next skill is safe with sufficient evidence for THAT stage.", + "deny": "Explicit danger, safety fault or violated precondition for the next skill.", + "uncertain": "Evidence required for the next skill is missing or ambiguous.", + }, + "reason": { + "none": "No safety concern and no specific progress finding.", + "unsafe": "Explicit danger or safety fault motivates a veto/stop.", + "insufficient_evidence": "Missing stage evidence motivates uncertainty or a stop.", + "tracking_error": "Measured alignment or tracking failure.", + "verified_progress": "Physical evidence supports progress at this skill boundary.", + }, + } result = await post( session, conn, @@ -95,11 +191,20 @@ async def decide(session, conn, request): else {} ), "state": json.dumps( - {"observation": request["observation"], "failure": request.get("failure", "none")}, + { + "observation": observation, + "failure": failure, + "transition": transition, + "candidates": request["candidates"], + "nextSkillCriteria": { + skill: SKILL_CRITERIA[skill] for skill in request["candidates"] + }, + }, ensure_ascii=False, ), "questions": { - name: question(values, prompts[name]) for name, values in options.items() + name: question(values, prompts[name], criteria.get(name)) + for name, values in options.items() }, }, ) diff --git a/decision_server/tests/test_jev_context.py b/decision_server/tests/test_jev_context.py new file mode 100644 index 00000000..57f0ea0c --- /dev/null +++ b/decision_server/tests/test_jev_context.py @@ -0,0 +1,75 @@ +import json +import unittest +from types import SimpleNamespace +from unittest.mock import AsyncMock, patch + +from decision_server.protocol import LANGUAGE_SKILLS, LANGUAGE_VERSION +from decision_server.providers import jev +from decision_server.tests.test_service import observation + + +class JevContextTests(unittest.IsolatedAsyncioTestCase): + async def decide(self, candidate="close", phase="descend", failure="none", **changes): + obs = observation() + obs.update(version=LANGUAGE_VERSION, phase=phase) + values = ( + dict( + choice=candidate, + grasp="empty", + diagnosis="none", + recovery="continue", + noul="allow", + score="partial", + reason="none", + ) + | changes + ) + upstream = {"answers": {k: {"choice": v} for k, v in values.items()}} + request = dict(observation=obs, candidates=[candidate, "stop"], failure=failure) + with patch.object(jev, "post", AsyncMock(return_value=upstream)) as post: + result, _ = await jev.decide( + None, SimpleNamespace(model="fixture", protocol="openrouter-decisions"), request + ) + return result, post.call_args.args[3], request + + async def test_next_skill_context_and_descriptive_criteria(self): + result, body, request = await self.decide() + state = json.loads(body["state"]) + self.assertEqual(state["observation"], request["observation"]) + self.assertEqual(state["candidates"], ["close", "stop"]) + self.assertEqual(state["transition"], {"nextSkill": "close", "completedSkill": "descend"}) + self.assertEqual(result["choice"], "close") + choice = body["questions"]["choice"] + self.assertEqual(set(choice["criteria"]), {"close", "stop"}) + self.assertIn("0.012 m", choice["criteria"]["close"]) + self.assertIn("Zero finger forces", choice["criteria"]["close"]) + self.assertIn("freezes physics", choice["instructions"]) + self.assertIn("NEXT", body["questions"]["noul"]["instructions"]) + self.assertEqual(body["provider"], {"allow_fallbacks": False}) + for skill in LANGUAGE_SKILLS: + self.assertGreater(len(jev.SKILL_CRITERIA[skill]), len(skill)) + self.assertIn("secure=false is expected", jev.SKILL_CRITERIA["verify"]) + self.assertIn("secure=true", jev.SKILL_CRITERIA["carry"]) + self.assertIn("onGoalSupport=true", jev.SKILL_CRITERIA["release"]) + + async def test_initial_entry_and_failed_recovery_are_not_completed_skills(self): + for candidate, phase, failure in [ + ("stow", "stow", "none"), + ("open", "close", "empty_grasp"), + ("carry", "stow", "none"), + ]: + _, body, _ = await self.decide(candidate, phase, failure) + self.assertIsNone(json.loads(body["state"])["transition"]["completedSkill"]) + _, body, _ = await self.decide("approach", "stow") + self.assertEqual(json.loads(body["state"])["transition"]["completedSkill"], "stow") + + async def test_never_rewrites_veto_or_stop_even_with_reason_none(self): + for changes in [dict(noul="deny"), dict(noul="uncertain"), dict(choice="stop")]: + with self.subTest(changes=changes): + result, _, _ = await self.decide(**changes) + for key, value in changes.items(): + self.assertEqual(result[key], value) + self.assertEqual(result["reason"], "none") + + def test_legacy_question_shape_is_unchanged(self): + self.assertEqual(jev.question(["ok"], "Choose ok.")["criteria"], {"ok": "ok"}) diff --git a/docs/website-deployment.md b/docs/website-deployment.md index dffa04dd..bb9a8d3e 100644 --- a/docs/website-deployment.md +++ b/docs/website-deployment.md @@ -2,7 +2,7 @@ 目标:`https://cadworld-sim.robotquan.com`,主机 `root@47.93.31.109`。复用 1Panel 的 OpenResty,独立容器仅发布 `127.0.0.1:8768`。不安装训练/GPU 服务,不改系统 Python,不需要访问者开启终端。 -当前在线版本:`20260924T065821Z-language-v2`,上一版本 `20260924T052730Z-dual-camera`。前后端共同发布,公网非付费检查通过;详见 [发布记录](lekiwi-language-v2-release.md)。 +当前在线版本:`20260924T081555Z-language-render-fix`,上一版本 `20260924T065821Z-language-v2`。前后端共同发布,修复预设阴影/高 DPI 渲染与 Jev 技能上下文;公网非付费检查 6/6 及硬件双相机验证通过。详见 [修复与发布记录](website-language-render-fix.md),其中包含保留 iframe 热更新的回滚步骤。 iframe 嵌入已通过响应头热更新放行 `https://cadworld.robotquan.com`,未重启应用;接入示例、验证范围与配置回滚见 [嵌入说明](website-embedding.md)。重新发布旧包会覆盖该热更新,应从当前源码重新构建。 diff --git a/docs/website-language-render-fix.md b/docs/website-language-render-fix.md new file mode 100644 index 00000000..9aa4f708 --- /dev/null +++ b/docs/website-language-render-fix.md @@ -0,0 +1,85 @@ +# 预设渲染与默认语言搬运修复(2026-09-24) + +## 状态 + +**已按用户授权同步发布前后端:`20260924T081555Z-language-render-fix`。** 网站为 ,上一版本 `20260924T065821Z-language-v2`。本次发布重启了模型服务,旧访客会话失效;没有修改服务器密钥或删除共享预算。部署时未创建 Git 提交/标签,随后按用户要求将源码、测试与文档归档为 `v1.0.3`,同步 npm 版本;版本归档不重复部署或付费推理。 + +## 发布验收 + +- 发布前复验 TypeScript、675 项前端单测、51 项后端测试(48 通过、3 跳过)与网站构建。 +- 显式白名单发布包 86 个文件,远端逐文件 SHA-256 通过;锁定基础镜像离线构建,先在无网络/无密钥挂载的容器中验证后端导入与 v2 契约,再切换线上。 +- 新容器健康,OpenResty 配置检查/reload 通过;28 个线上静态文件与 manifest 一致,公网首页、主应用和 MuJoCoViewer 资源哈希匹配新包,运行容器内 Jev 源码哈希匹配本地修复。 +- 公网 Playwright **6/6 通过**:预设/双相机、Python、会话与配置保护、允许/拒绝 iframe 来源及 CSRF。另在 RTX 5080、DPR=2 的真实浏览器检查前置/腕部双相机:绘图缓冲 1728×1912,CSS 与宿主均为 864×956,4×MSAA,无页面异常。 +- 密钥文件和预算 SQLite 的 inode、权限、UID/GID 保持不变;已批准 iframe 配置哈希不变。**本次发布验收新增付费推理为 0**;下文真实 API 成功结果来自发布前本地源码接真实模型,不冒充发布后的公网付费回合。 + +发布证据:`build/website-deployment/language-render-release/`,含基线、构建/发布日志、远端/公网哈希校验、6 项浏览器结果与高 DPI 双相机截图。 + +- 发布包 SHA-256:`c2fa2e5dfea8b2fa60f2ee757d94ff32f4ead1a6e8f77cfa3cabfa68141a9501` +- 公网首页 SHA-256:`f3af1b1b42e73ad093e00cce356617e5efa2da3fda9b7a29b8966a26b0cdbcd1` + +必要时前后端成对回滚到上一版本;旧包早于 iframe 热更新,回滚后须恢复本次保存的实际线上配置: + +```bash +bash /opt/cadworld-sim/releases/20260924T065821Z-language-v2/deploy/publish.sh \ + 20260924T065821Z-language-v2 +cp /opt/cadworld-sim/backups/20260924T081555Z-language-render-fix/openresty.conf \ + /opt/1panel/apps/openresty/openresty/conf/conf.d/cadworld-sim.robotquan.com.conf +docker exec 1Panel-openresty-m72w nginx -t && \ + docker exec 1Panel-openresty-m72w nginx -s reload +``` + +本次未执行回滚演练;旧版本及实际线上配置备份均保留,不得删除 secrets/budget。 + +## 定位与修改 + +- RTX 5080 实际浏览器重现腕部画面的块状阴影。此前使用默认 512² 阴影贴图及固定阴影范围,且 `DataTexture` 默认最近邻过滤。现为 2048²、PCFSoft、按场景中心/尺度适配阴影视锥和偏移、纹理 mipmap/各向异性过滤;GPU 像素比 1.5–2,软件 WebGL 保留原预算。不减面,不改变相机安装、接触或物理参数。 +- `renderer.setSize(..., false)` 未设置 CSS 尺寸,DPR>1 时画布实际在页面上变大并被裁切,腕部视口也丢失高 DPI 收益。改为逻辑 CSS 尺寸与绘图缓冲分离,双相机继续共用同一画布/布局槽位。 +- 公网默认指令 API 正常,但在 `descend → close` 收到 `choice=stop,noul=allow,reason=none`。原请求仅提供技能名作为 criteria,未说明 close 是闭爪、verify 建立抓持证据、放置时 secure 可能自然变为 false,也未明确当前/下一阶段关系。 +- 请求现补充具名技能含义和前置条件、候选下一技能、相邻正常边界的 completedSkill、仿真等待时冻结的时钟语义。初始入口、跳阶段和失败恢复不冒充已完成;不修改原始观测,不用模型评分替代本地物理验收。 +- Choice、名为 Noul/Score 的离散监督问题沿用现有 v2 契约。各问题独立判定,reason 不是可验证的模型解释,因此 `none` 不再直接展示为停止原因;显示“模型拒绝继续/无法确认安全,未提供具体原因”,同时终态保存否决响应。**deny/uncertain/stop 仍然停止,不私自用概率最高项覆盖官方 choice。** + +## 真实 API 实测(经本次用户授权) + +全部真实调用使用固定 `deepseek-flash` 与 `typesafe/jev-1.13`,未切换 mock、修改物理、自动重试或在同一任务内覆盖停止。每个测试会话最终销毁;密钥仅由后端从明确指定的文件读取,不进入浏览器、日志或证据。 + +| 阶段 | 环境 | 结果 | 尝试调用 LLM/Jev | +| ------------------ | --------------------- | ------------------------------------------------------------ | ---------------- | +| 线上复现 | 公网旧版、SwiftShader | `descend → close` 被 Choice 停止 | 2/7 | +| 补充技能语义 | 本地源码、RTX | 完整成功,搬运 0.728882 m | 2/14 | +| 回归发现剩余歧义 | 本地源码、SwiftShader | `stow → approach` 被 Choice 停止 | 2/3 | +| 补齐阶段完成上下文 | 本地源码、SwiftShader | 已完成 carry,下一次调用 `upstream_transport_error` 安全停止 | 2/10 | +| 隔离硬件复验 | 本地源码、RTX | 13 个技能全部通过,最终稳定放置成功 | 2/14 | + +合计 **10 次 LLM / 48 次 Jev 调用尝试**,包含一次上游传输失败;不是完整费用账单。不把中间成功或最终单轮成功描述为多轮可靠性证明,模型与网络仍可能触发安全停止。 + +最终回合:`state=succeeded`、`phase=settle`、`transported=0.7288822577190641 m`、`onGoalSupport=true`、`succeeded=true`;`/command` 1 次、`/plan` 1 次、`/decide` 13 次,合计 2 LLM/14 Jev(含意图审核)。 + +主要本机证据(不入 Git): + +- `build/website-deployment/language-repro/`:公网原始请求、响应及终态。 +- `build/website-deployment/render-hardware-{before,after}/before.png`:同一初始场景硬件渲染对比。 +- `build/website-deployment/language-fix-final-hardware/{receipts.json,result.json,after.png}`:最终真实模型完整闭环与截图。 +- `build/website-deployment/language-fix-live-{regression,transition}/`:中间模型停止/网络失败,不覆盖失败记录。 + +## 非付费验证 + +- TypeScript、定向 TS/Python lint、网站生产构建通过;保留既有 Pyodide 浏览器外部模块和大 chunk 提示。 +- Vitest **137 文件 / 675 项**通过;决策服务 **51 项(48 通过、3 个既有可选门禁跳过)**。 +- 真实 WASM:连续 B 区→自由坐标搬运、锁开夹爪失败、目标边界 IK 标定通过;高 DPI 双相机尺寸回归通过,零推理。 +- 主工作台假上游连续两次搬运在并行运行两个重 CAD 软件渲染浏览器时超时,隔离复跑 **1/1 通过**。其余上述四项首轮通过,不能将首轮整个 suite 说成一次全绿。 + +## 可复验入口 + +现有完整运动实测保持不变;新增预设专项,显式选择后不会再跑完整运动实测: + +```bash +CADWORLD_LIVE_API=1 CADWORLD_LIVE_PRESET=1 \ + npx playwright test -c web_platform/playwright.website-live.config.ts +``` + +默认访问已更新的公网网站,**此命令会产生新的付费推理,发布验收未执行**。开发服务已正确配置时,可显式加 `CADWORLD_LIVE_URL=http://127.0.0.1:5173` 测本地源码接真实模型。专项上限为 2 LLM/14 Jev,每轮零自动重试;普通 CI 不启用。 + +## Sources + +- [Jev 官方教程:各问题独立判定、不能读取彼此答案](https://openrouter.ai/docs/guides/community/jev-tutorial) +- [Decisions API 契约](https://openrouter.ai/docs/api/api-reference/alphadecisions/submit-a-decisions-questions-and-answers-request) diff --git a/package-lock.json b/package-lock.json index 280f58cc..be1b044f 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "mujoco-web-platform", - "version": "1.0.2", + "version": "1.0.3", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "mujoco-web-platform", - "version": "1.0.2", + "version": "1.0.3", "license": "Apache-2.0", "dependencies": { "@monaco-editor/react": "^4.7.0", diff --git a/package.json b/package.json index d52afea1..8ca7466e 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "mujoco-web-platform", - "version": "1.0.2", + "version": "1.0.3", "description": "基于 MuJoCo WebAssembly 的本地机器人仿真与控制平台", "private": true, "type": "module", diff --git a/web_platform/e2e/lekiwi.agent.workspace.spec.ts b/web_platform/e2e/lekiwi.agent.workspace.spec.ts index e71176f7..bc33b9ec 100644 --- a/web_platform/e2e/lekiwi.agent.workspace.spec.ts +++ b/web_platform/e2e/lekiwi.agent.workspace.spec.ts @@ -30,3 +30,39 @@ test('主工作台:加载预设、语言命名目标及连续自由坐标抓 await expect(page.getByRole('button', { name: '强化学习任务', exact: true })).toBeVisible(); await page.screenshot({ path: info.outputPath('language-workbench.png') }); }); + +test.describe('预设画布高 DPI 回归', () => { + test.use({ deviceScaleFactor: 2 }); + test('主画布不被 DPR 放大裁切,双相机共用高分辨率绘图缓冲', async ({ page }) => { + let inference = 0; + await page.route('**/api/decision/v1/**', async (route) => { + if (/\/(command|plan|decide|test)$/.test(new URL(route.request().url()).pathname)) { + inference++; + await route.abort(); + } else await route.continue(); + }); + await page.goto('/'); + await page.getByRole('tab', { name: '控制台', exact: true }).click(); + await page.getByRole('button', { name: '机器人语言控制', exact: true }).click(); + const panel = page.getByLabel('机器人语言控制', { exact: true }); + await panel.getByRole('button', { name: '载入预设场景' }).click(); + await expect(panel.getByRole('status')).toContainText('预设已载入并暂停', { timeout: 90000 }); + for (const name of ['腕部摄像头', '前置摄像头']) { + await page.getByRole('combobox', { name: '摄像头视角' }).selectOption({ label: name }); + const size = await page + .locator('canvas') + .first() + .evaluate((canvas: HTMLCanvasElement) => ({ + css: [canvas.clientWidth, canvas.clientHeight], + host: [canvas.parentElement!.clientWidth, canvas.parentElement!.clientHeight], + pixels: [canvas.width, canvas.height], + })); + expect(size.css).toEqual(size.host); + for (let i = 0; i < 2; i++) { + expect(size.pixels[i] / size.css[i]).toBeGreaterThanOrEqual(1.74); + expect(size.pixels[i] / size.css[i]).toBeLessThanOrEqual(2); + } + } + expect(inference).toBe(0); + }); +}); diff --git a/web_platform/e2e/website.live.spec.ts b/web_platform/e2e/website.live.spec.ts index 6ed6b996..5c6ec855 100644 --- a/web_platform/e2e/website.live.spec.ts +++ b/web_platform/e2e/website.live.spec.ts @@ -1,9 +1,12 @@ import { expect, test } from '@playwright/test'; import { writeFile } from 'node:fs/promises'; -import { loadWebsiteRobot, runWebsiteCommand } from './website.helpers'; +import { collectRobotResults, loadWebsiteRobot, runWebsiteCommand } from './website.helpers'; test('显式授权:公网内置模型,前进/转向/抓放,无手填密钥或自动重试', async ({ page }, info) => { - test.skip(process.env.CADWORLD_LIVE_API !== '1', '仅经用户授权后运行,会消耗服务器公共额度'); + test.skip( + process.env.CADWORLD_LIVE_API !== '1' || process.env.CADWORLD_LIVE_PRESET === '1', + '仅经用户授权后运行;预设专项测试不重复执行运动测试', + ); test.setTimeout(360000); const calls = { command: 0, plan: 0, decide: 0, blocked: 0 }; let csrf = '', @@ -86,3 +89,87 @@ test('显式授权:公网内置模型,前进/转向/抓放,无手填密钥 } expect(cleanup).toBe(true); }); + +test('显式授权:空工作台默认预设指令完成真实模型抓放', async ({ page }, info) => { + test.skip( + process.env.CADWORLD_LIVE_API !== '1' || process.env.CADWORLD_LIVE_PRESET !== '1', + '仅在显式选择预设专项实测时调用付费 API', + ); + test.setTimeout(360000); + const calls = { command: 0, plan: 0, decide: 0, blocked: 0 }; + let csrf = '', + cleanup = false; + const captures: Promise[] = []; + page.on('response', (response) => { + if (response.url().endsWith('/session') && response.request().method() === 'POST') + captures.push( + response.json().then((value) => { + csrf = value.csrfToken; + }), + ); + }); + await page.route('**/api/decision/v1/**', async (route) => { + const endpoint = new URL(route.request().url()).pathname.split('/').at(-1); + if (endpoint === 'command' || endpoint === 'plan' || endpoint === 'decide') { + if (calls[endpoint] >= { command: 1, plan: 1, decide: 13 }[endpoint]) { + calls.blocked++; + await route.abort(); + return; + } + calls[endpoint]++; + } else if (endpoint === 'test') { + calls.blocked++; + await route.abort(); + return; + } + await route.continue(); + }); + try { + await collectRobotResults(page); + await page.goto('/'); + await page.getByRole('tab', { name: '控制台', exact: true }).click(); + await page.getByRole('button', { name: '机器人语言控制', exact: true }).click(); + const panel = page.getByLabel('机器人语言控制', { exact: true }); + await panel.getByRole('button', { name: '载入预设场景' }).click(); + await expect(panel.getByRole('status')).toContainText('预设已载入并暂停', { timeout: 120000 }); + await expect(panel.getByRole('textbox', { name: '机器人指令' })).toHaveValue('把方块搬到 B 区'); + const result = await runWebsiteCommand(page, '把方块搬到 B 区', info, 'live-default-preset'); + expect(result.goal).toEqual([0.257, 0.75, 0.128]); + expect(result.transported).toBeGreaterThan(0.7); + expect(result.sample.onGoalSupport).toBe(true); + expect(result.llmCalls).toBe(2); + expect(result.jevCalls).toBe(14); + expect(calls).toEqual({ command: 1, plan: 1, decide: 13, blocked: 0 }); + await page.screenshot({ path: info.outputPath('live-default-preset.png') }); + } finally { + await Promise.all(captures); + if (csrf) + cleanup = await page + .evaluate( + async (token) => + ( + await fetch('/api/decision/v1/session', { + method: 'DELETE', + headers: { 'Content-Type': 'application/json', 'X-CSRF-Token': token }, + body: '{}', + }) + ).status === 200, + csrf, + ) + .catch(() => false); + await writeFile( + info.outputPath('live-default-summary.json'), + JSON.stringify( + { + calls, + sessionDestroyed: cleanup, + maximumUpstreamCalls: { llm: 2, jev: 14 }, + retries: 0, + }, + null, + 2, + ), + ); + } + expect(cleanup).toBe(true); +}); diff --git a/web_platform/playwright.website-live.config.ts b/web_platform/playwright.website-live.config.ts index 59a45e7c..882df420 100644 --- a/web_platform/playwright.website-live.config.ts +++ b/web_platform/playwright.website-live.config.ts @@ -5,5 +5,10 @@ export default defineConfig(production, { testMatch: 'website.live.spec.ts', retries: 0, timeout: 360_000, - use: { trace: 'off', video: 'off', screenshot: 'off' }, + use: { + ...(process.env.CADWORLD_LIVE_URL ? { baseURL: process.env.CADWORLD_LIVE_URL } : {}), + trace: 'off', + video: 'off', + screenshot: 'off', + }, }); diff --git a/web_platform/src/mobile/agent/AgentTaskController.test.ts b/web_platform/src/mobile/agent/AgentTaskController.test.ts index b34e23c3..d282ae6f 100644 --- a/web_platform/src/mobile/agent/AgentTaskController.test.ts +++ b/web_platform/src/mobile/agent/AgentTaskController.test.ts @@ -3,6 +3,7 @@ import { AgentTaskController, type SkillExecutor } from './AgentTaskController'; import { MockDecisionProvider } from './MockDecisionProvider'; import { PickPlaceEvaluator, type PhysicalSample } from './PhysicalEvidence'; import { SKILLS, type Skill } from './protocol'; +import { LANGUAGE_VERSION } from './LanguageScene'; class FakeSkills implements SkillExecutor { phase: Skill = 'open'; @@ -99,6 +100,27 @@ describe('模型闭环调度(技能为 mock,物理另有真实 WASM 验收 expect(driver.results).toHaveLength(11); driver.dispose(); }); + it('v2 veto remains in terminal telemetry and never authorizes motion', async () => { + const skills = new FakeSkills(); + const provider = new MockDecisionProvider(); + const normal = provider.decide.bind(provider); + provider.decide = async (...args) => { + const reply = await normal(...args); + reply.value.noul = 'uncertain'; + reply.value.reason = 'none'; + return reply; + }; + const driver = new AgentTaskController(skills, provider, '搬运方块', { + version: LANGUAGE_VERSION, + }); + await run(driver); + expect(driver.state).toBe('failed'); + expect(driver.error).toContain('模型无法确认下一技能安全'); + expect(driver.snapshot().decision).toMatchObject({ noul: 'uncertain', reason: 'none' }); + expect(skills.authorized).toEqual([]); + expect(skills.sample.time).toBe(0); + driver.dispose(); + }); it('支持可恢复空抓,但跨重规划也不能重置每技能 2 次上限', async () => { const skills = new FakeSkills(); skills.fault = 'empty_grasp'; diff --git a/web_platform/src/mobile/agent/AgentTaskController.ts b/web_platform/src/mobile/agent/AgentTaskController.ts index 02c9f8cb..bcc531ac 100644 --- a/web_platform/src/mobile/agent/AgentTaskController.ts +++ b/web_platform/src/mobile/agent/AgentTaskController.ts @@ -136,7 +136,8 @@ export class AgentTaskController implements AgentDriver { phase: this.skills.phase, candidate: this.skills.candidate, plan: this.plan, - decision: this.lastDecision, + // Preserve even a vetoed response for diagnostics; it never authorizes a skill. + decision: this.supervisor.last, llmCalls: this.llmCalls, jevCalls: this.jevCalls, replans: this.replans, diff --git a/web_platform/src/mobile/agent/TaskSupervisor.test.ts b/web_platform/src/mobile/agent/TaskSupervisor.test.ts index b7bcb135..b5502c4f 100644 --- a/web_platform/src/mobile/agent/TaskSupervisor.test.ts +++ b/web_platform/src/mobile/agent/TaskSupervisor.test.ts @@ -23,6 +23,32 @@ describe('Choice / Noul / Score', () => { new TaskSupervisor().inspect({ ...decision, noul }, ['carry'], LANGUAGE_VERSION, true), ).toThrow('Noul'); }); + it.each(['deny', 'uncertain'] as const)( + 'explains %s even when upstream reason is none', + (noul) => { + const supervisor = new TaskSupervisor(); + const rejected = { ...decision, noul, reason: 'none' }; + expect(() => supervisor.inspect(rejected, ['carry'], LANGUAGE_VERSION, true)).toThrow( + noul === 'uncertain' ? '模型无法确认下一技能安全' : '模型拒绝继续', + ); + expect(supervisor.last).toEqual(rejected); + expect(() => supervisor.inspect(rejected, ['carry'], LANGUAGE_VERSION, true)).not.toThrow( + 'Noul 拦截:none', + ); + }, + ); + it('does not let an allow score override choice/recovery stops', () => { + const supervisor = new TaskSupervisor(); + for (const stop of [{ choice: 'stop' }, { recovery: 'stop' }]) + expect(() => + supervisor.inspect( + { ...decision, ...stop, reason: 'none' }, + ['carry', 'stop'], + LANGUAGE_VERSION, + true, + ), + ).toThrow('Jev 请求安全停止:模型拒绝继续,未提供具体原因'); + }); it('rejects illegal candidate, missing fields and arbitrary scores', () => { const s = new TaskSupervisor(); expect(() => s.inspect(decision, ['open'], LANGUAGE_VERSION, true)).toThrow(); diff --git a/web_platform/src/mobile/agent/TaskSupervisor.ts b/web_platform/src/mobile/agent/TaskSupervisor.ts index 305fa3c6..c293ce01 100644 --- a/web_platform/src/mobile/agent/TaskSupervisor.ts +++ b/web_platform/src/mobile/agent/TaskSupervisor.ts @@ -2,6 +2,19 @@ import { LANGUAGE_VERSION } from './LanguageScene'; import { validateJevDecision, type AgentVersion, type JevDecision, type Skill } from './protocol'; export const QUALITY_POINTS = { poor: 0, partial: 25, good: 75, excellent: 100 } as const; +const REASONS = { + unsafe: '检测到不安全因素', + insufficient_evidence: '当前阶段证据不足', + tracking_error: '动作跟踪或对齐异常', + verified_progress: '模型报告已有进展,但仍拒绝继续', +} as const; +function stopReason(decision: JevDecision): string { + const reason = decision.reason; + if (reason && reason !== 'none') return REASONS[reason]; + return decision.noul === 'uncertain' + ? '模型无法确认下一技能安全,未提供具体原因' + : '模型拒绝继续,未提供具体原因'; +} /** Supervisory advice can veto but cannot grant physical authority. */ export class TaskSupervisor { last?: JevDecision; @@ -18,9 +31,9 @@ export class TaskSupervisor { const decision = validateJevDecision(value, candidates, version); this.last = decision; if (version === LANGUAGE_VERSION && decision.noul !== 'allow') - throw new Error(`Noul 拦截:${decision.reason ?? 'insufficient_evidence'}`); + throw new Error(`Noul 拦截(${decision.noul}):${stopReason(decision)}`); if (decision.choice === 'stop' || decision.recovery === 'stop') - throw new Error('Jev 请求安全停止'); + throw new Error(`Jev 请求安全停止:${stopReason(decision)}`); if (decision.grasp === 'secure' && !secure) throw new Error('Jev 抓持判断与本地物理证据矛盾'); if ( ['carry', 'stop-base'].includes(decision.choice) && diff --git a/web_platform/src/viewer/MuJoCoViewer.ts b/web_platform/src/viewer/MuJoCoViewer.ts index 8f9ed268..d3d2bd8e 100644 --- a/web_platform/src/viewer/MuJoCoViewer.ts +++ b/web_platform/src/viewer/MuJoCoViewer.ts @@ -24,6 +24,7 @@ import type { EditableMapDocument, MapEditorTransformMode } from '../map/editor/ import type { MapEditorDraftPreviewInstance } from '../map/mapSceneDraft'; import { MapEditorLayer } from './MapEditorLayer'; import { ParametricMapPreviewLayer } from './ParametricMapPreviewLayer'; +import { filterSurfaceTexture, fitSceneShadow, viewerPixelRatio } from './renderQuality'; export type InteractionMode = 'select' | 'joint' | 'force'; export type ViewerTheme = 'light' | 'dark'; @@ -144,6 +145,7 @@ export class MuJoCoViewer { private mapEditorLayer: MapEditorLayer; private grid: THREE.GridHelper; private hemisphere: THREE.HemisphereLight; + private readonly sun = new THREE.DirectionalLight(0xffffff, 2); private themeTransition?: { started: number; fromBackground: THREE.Color; @@ -177,8 +179,9 @@ export class MuJoCoViewer { String(gl.getParameter(debug.UNMASKED_RENDERER_WEBGL)), ) : false; - this.renderer.setPixelRatio(Math.min(devicePixelRatio, 1.75)); + this.renderer.setPixelRatio(viewerPixelRatio(devicePixelRatio, this.softwareRendering)); this.renderer.shadowMap.enabled = true; + this.renderer.shadowMap.type = THREE.PCFSoftShadowMap; this.renderer.outputColorSpace = THREE.SRGBColorSpace; host.append(this.renderer.domElement); this.camera.up.set(0, 0, 1); @@ -209,10 +212,9 @@ export class MuJoCoViewer { this.scene.background = new THREE.Color(0x0b1220); this.hemisphere = new THREE.HemisphereLight(0xffffff, 0x223344, 1.3); this.scene.add(this.hemisphere); - const light = new THREE.DirectionalLight(0xffffff, 2); - light.position.set(4, -3, 7); - light.castShadow = true; - this.scene.add(light); + this.sun.castShadow = true; + fitSceneShadow(this.sun, [0, 0, 0], 2); + this.scene.add(this.sun, this.sun.target); this.grid = new THREE.GridHelper(20, 40, 0x3b82f6, 0x253047).rotateX(Math.PI / 2); this.grid.visible = this.displayOptions.showGrid; this.scene.add(this.grid); @@ -628,6 +630,7 @@ export class MuJoCoViewer { this.cameraFocusTransition = undefined; const { extent, center } = session.geometryBounds(); this.modelExtent = extent; + fitSceneShadow(this.sun, center, extent); this.controls.target.set(center[0], center[1], center[2]); this.camera.position.set( center[0] + extent * 1.5, @@ -651,10 +654,19 @@ export class MuJoCoViewer { private resize(): void { const w = Math.max(1, this.host.clientWidth), h = Math.max(1, this.host.clientHeight); - if (w === this.resizeWidth && h === this.resizeHeight) return; + const ratio = viewerPixelRatio(devicePixelRatio, this.softwareRendering); + if ( + w === this.resizeWidth && + h === this.resizeHeight && + ratio === this.renderer.getPixelRatio() + ) + return; + this.renderer.setPixelRatio(ratio); this.resizeWidth = w; this.resizeHeight = h; - this.renderer.setSize(w, h, false); + // CSS size must stay in logical pixels: otherwise DPR also enlarges/clips the canvas + // and cancels the sensor viewport's resolution gain. + this.renderer.setSize(w, h); this.camera.aspect = w / h; this.camera.updateProjectionMatrix(); } @@ -977,7 +989,7 @@ export class MuJoCoViewer { found = new THREE.DataTexture(data, w, h, THREE.RGBAFormat); found.colorSpace = THREE.SRGBColorSpace; found.flipY = true; - found.needsUpdate = true; + filterSurfaceTexture(found, this.renderer.capabilities.getMaxAnisotropy()); this.textures.set(id, found); return found; } diff --git a/web_platform/src/viewer/renderQuality.test.ts b/web_platform/src/viewer/renderQuality.test.ts new file mode 100644 index 00000000..82ffc227 --- /dev/null +++ b/web_platform/src/viewer/renderQuality.test.ts @@ -0,0 +1,72 @@ +import * as THREE from 'three'; +import { filterSurfaceTexture, fitSceneShadow, viewerPixelRatio } from './renderQuality'; +import { MuJoCoViewer } from './MuJoCoViewer'; + +it('GPU supersampling is bounded; software rendering keeps its existing budget', () => { + expect(viewerPixelRatio(1, false)).toBe(1.5); + expect(viewerPixelRatio(1.75, false)).toBe(1.75); + expect(viewerPixelRatio(3, false)).toBe(2); + expect(viewerPixelRatio(1, true)).toBe(1); + expect(viewerPixelRatio(3, true)).toBe(1.75); + expect(viewerPixelRatio(NaN, false)).toBe(1.5); +}); + +it('fits a 2048px biased shadow to the loaded scene, including translated scenes', () => { + const light = new THREE.DirectionalLight(); + fitSceneShadow(light, [12, -3, 2], 1.5); + expect(light.shadow.mapSize.toArray()).toEqual([2048, 2048]); + expect(light.target.position.toArray()).toEqual([12, -3, 2]); + expect(light.position.distanceTo(light.target.position)).toBeCloseTo(3); + expect(light.shadow.camera.left).toBe(-1.5); + expect(light.shadow.camera.right).toBe(1.5); + expect(light.shadow.camera.far).toBe(6); + expect(light.shadow.bias).toBeLessThan(0); + expect(light.shadow.normalBias).toBeGreaterThan(0); + fitSceneShadow(light, [0, 0, 0], 0.2); + expect(light.shadow.camera.right).toBe(1); + light.dispose(); +}); + +it('filters magnified and oblique textures without changing their pixels', () => { + const pixels = new Uint8Array([255, 127, 0, 255]); + const texture = new THREE.DataTexture(pixels, 1, 1); + filterSurfaceTexture(texture, 16); + expect(texture.magFilter).toBe(THREE.LinearFilter); + expect(texture.minFilter).toBe(THREE.LinearMipmapLinearFilter); + expect(texture.generateMipmaps).toBe(true); + expect(texture.anisotropy).toBe(8); + expect(texture.image.data).toBe(pixels); + filterSurfaceTexture(texture, 2); + expect(texture.anisotropy).toBe(2); + texture.dispose(); +}); + +it('resizes logical CSS dimensions independently of DPR, including DPR-only changes', () => { + vi.stubGlobal('devicePixelRatio', 2); + try { + let pixelRatio = 1; + const renderer = { + getPixelRatio: () => pixelRatio, + setPixelRatio: vi.fn((value: number) => { + pixelRatio = value; + }), + setSize: vi.fn(), + }; + const viewer = Object.assign(Object.create(MuJoCoViewer.prototype), { + host: { clientWidth: 800, clientHeight: 600 }, + renderer, + camera: new THREE.PerspectiveCamera(), + softwareRendering: false, + resizeWidth: 800, + resizeHeight: 600, + }) as { resize(): void }; + viewer.resize(); + expect(renderer.setPixelRatio).toHaveBeenCalledWith(2); + // Do not pass updateStyle=false: the DOM canvas would become 1600x1200 CSS pixels. + expect(renderer.setSize).toHaveBeenCalledWith(800, 600); + viewer.resize(); + expect(renderer.setSize).toHaveBeenCalledTimes(1); + } finally { + vi.unstubAllGlobals(); + } +}); diff --git a/web_platform/src/viewer/renderQuality.ts b/web_platform/src/viewer/renderQuality.ts new file mode 100644 index 00000000..872b6083 --- /dev/null +++ b/web_platform/src/viewer/renderQuality.ts @@ -0,0 +1,45 @@ +import * as THREE from 'three'; + +/** Supersample low-DPI GPU displays too; software WebGL keeps its latency budget. */ +export function viewerPixelRatio(dpr: number, software: boolean): number { + const ratio = Number.isFinite(dpr) && dpr > 0 ? dpr : 1; + return software ? Math.min(ratio, 1.75) : Math.min(Math.max(ratio, 1.5), 2); +} + +/** Fit shadow texels to the loaded scene instead of a default 10 m / 512 px map. */ +export function fitSceneShadow( + light: THREE.DirectionalLight, + center: readonly number[], + extent: number, +): void { + const radius = Math.max(1, extent); + light.target.position.set(center[0], center[1], center[2]); + light.position + .set(4, -3, 7) + .normalize() + .multiplyScalar(radius * 2) + .add(light.target.position); + const shadow = light.shadow; + shadow.mapSize.set(2048, 2048); + Object.assign(shadow.camera, { + left: -radius, + right: radius, + top: radius, + bottom: -radius, + near: radius * 0.01, + far: radius * 4, + }); + shadow.bias = -0.00005; + shadow.normalBias = radius / 4096; + shadow.camera.updateProjectionMatrix(); + shadow.needsUpdate = true; +} + +/** DataTexture defaults to nearest-neighbour; oblique camera views need mipmaps. */ +export function filterSurfaceTexture(texture: THREE.DataTexture, maxAnisotropy: number): void { + texture.magFilter = THREE.LinearFilter; + texture.minFilter = THREE.LinearMipmapLinearFilter; + texture.generateMipmaps = true; + texture.anisotropy = Math.min(8, maxAnisotropy); + texture.needsUpdate = true; +}