Refine course progression and runtime safety

This commit is contained in:
Haoran
2026-08-11 15:13:13 +08:00
parent b36dbcd84f
commit ab35e59672
83 changed files with 5291 additions and 2267 deletions

166
tests/test_web_scenarios.py Normal file
View File

@@ -0,0 +1,166 @@
from __future__ import annotations
import asyncio
import importlib.util
import json
import re
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
SCENARIOS = ROOT / "web" / "src" / "data" / "scenarios"
GENERATED_VERSIONS = ROOT / "web" / "src" / "data" / "generated" / "versions.json"
def load_scenario(lesson: str) -> dict:
return json.loads((SCENARIOS / f"{lesson}.json").read_text())
def load_lesson(name: str, script: Path):
spec = importlib.util.spec_from_file_location(name, script)
if spec is None or spec.loader is None:
raise RuntimeError(f"unable to load {script}")
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module
def test_s15_scenario_uses_the_real_plan_protocol() -> None:
steps = load_scenario("s15")["steps"]
spawn = next(
step for step in steps
if step.get("toolName") == "spawn_teammate"
and '"name":"backend"' in step.get("content", "")
)
claim_index = next(
index for index, step in enumerate(steps)
if "claim_next_task(backend)" in step.get("content", "")
)
request_index = next(
index for index, step in enumerate(steps)
if step.get("toolName") == "request_plan"
)
review_index = next(
index for index, step in enumerate(steps)
if step.get("toolName") == "review_plan"
)
response_index = next(
index for index, step in enumerate(steps)
if "plan_approval_response" in step.get("content", "")
)
review = json.loads(steps[review_index]["content"])
assert json.loads(spawn["content"])["require_plan"] is True
assert claim_index < request_index < review_index < response_index
assert review["request_id"] == "req_000007"
assert re.fullmatch(r"req_\d{6}", review["request_id"])
assert review["approve"] is True
assert "approved" not in review
def test_s17_scenario_calls_the_discovered_mcp_tool() -> None:
steps = load_scenario("s17")["steps"]
bash_index = next(
index for index, step in enumerate(steps)
if step.get("toolName") == "bash"
)
approval_index = next(
index for index, step in enumerate(steps)
if "permission: user approved" in step.get("content", "")
)
connect_index = next(
index for index, step in enumerate(steps)
if step.get("toolName") == "connect_mcp"
)
status_index = next(
index for index, step in enumerate(steps)
if step.get("toolName") == "mcp__deploy__status"
and step["type"] == "tool_call"
)
result_index = next(
index for index, step in enumerate(steps)
if step.get("toolName") == "mcp__deploy__status"
and step["type"] == "tool_result"
)
notification_index = next(
index for index, step in enumerate(steps)
if "task_notification(status=completed)" in step.get("content", "")
)
bash_call = json.loads(steps[bash_index]["content"])
assert bash_call == {
"command": "python -m unittest tests.test_agent_teams_runtime",
"run_in_background": True,
}
assert bash_index < approval_index < notification_index
assert connect_index < status_index < result_index
def test_s17_runtime_discovers_and_dispatches_mcp_tools(
tmp_path: Path, monkeypatch
) -> None:
monkeypatch.setenv("MODEL_ID", "test-model")
harness = load_lesson(
"integrated_mcp_scenario_test",
ROOT / "s17_integrated_harness" / "code.py",
)
harness.WORKDIR = tmp_path
_, handlers_before = harness.assemble_tool_pool()
assert "mcp__deploy__status" not in handlers_before
assert "Connected to MCP server 'deploy'" in harness.connect_mcp("deploy")
tools_after, handlers_after = harness.assemble_tool_pool()
assert "mcp__deploy__status" in {tool["name"] for tool in tools_after}
assert handlers_after["mcp__deploy__status"](service="web") == (
"[deploy] web: running (v1.4.2)"
)
def test_s18_scenario_matches_the_deterministic_runtime(tmp_path: Path) -> None:
scenario = load_scenario("s18")
workflow_call = next(
step for step in scenario["steps"]
if step.get("toolName") == "Workflow" and step["type"] == "tool_call"
)
workflow_result = next(
step for step in scenario["steps"]
if step.get("toolName") == "Workflow" and step["type"] == "tool_result"
)
call_input = json.loads(workflow_call["content"])
shown_result = json.loads(workflow_result["content"])
workflow = load_lesson(
"workflow_scenario_test", ROOT / "s18_workflow_runtime" / "code.py"
)
workflow.STORE = tmp_path
workflow.create_run_id = lambda _meta: "wf_review-changes_0000000000001a7b"
actual = asyncio.run(workflow.run_workflow(**call_input))
assert set(call_input) <= set(workflow.WORKFLOW_TOOL["input_schema"]["properties"])
assert shown_result == actual
def test_generated_s18_metadata_extends_s17_without_registry_false_positives() -> None:
versions = json.loads(GENERATED_VERSIONS.read_text())
by_id = {version["id"]: version for version in versions["versions"]}
s17 = by_id["s17"]
s18 = by_id["s18"]
assert set(s17["tools"]) < set(s18["tools"])
assert s18["newTools"] == ["Workflow"]
assert "Workflow" in s18["tools"]
assert "review-changes" not in s18["tools"]
chapter_dirs = {
path.name.split("_", 1)[0]: path
for path in ROOT.glob("s[0-9][0-9]_*")
}
for lesson_id in ("s13", "s14", "s15", "s16", "s17", "s18"):
assert by_id[lesson_id]["source"] == (
chapter_dirs[lesson_id] / "code.py"
).read_text()
signatures = {
function["name"]: function["signature"]
for function in s18["functions"]
}
assert signatures["run_workflow"].startswith("async def run_workflow(")