Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
36 changes: 36 additions & 0 deletions .github/workflows/game-agents.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,36 @@
name: Game Agents

on:
push:
branches: [main, feature/stardew-minecraft]
pull_request:

permissions:
contents: read

jobs:
test:
strategy:
matrix:
os: [ubuntu-latest, windows-latest]
runs-on: ${{ matrix.os }}
env:
LITELLM_LOCAL_MODEL_COST_MAP: "True"
CUSTOM_TIKTOKEN_CACHE_DIR: .test-token-cache
TIKTOKEN_CACHE_DIR: .test-token-cache
steps:
- if: runner.os == 'Windows'
run: git config --global core.longpaths true
- uses: actions/checkout@v4
- uses: actions/setup-python@v5
with:
python-version: "3.12"
cache: pip
- run: python -m pip install -e . "pytest>=9,<10" "pytest-asyncio>=1.3,<2" ruff
- run: python -c "import tiktoken; tiktoken.get_encoding('cl100k_base')"
- run: python -m ruff check --ignore N999 PhyAgentOS/game_agents PhyAgentOS/cli/general_game_commands.py tests/game_agents
- run: python -m pytest tests
- run: paos general-game --help
- run: paos minecraft warmup --help
- run: paos minecraft benchmark --help
- run: python -m pip wheel --no-deps --wheel-dir dist .
3 changes: 2 additions & 1 deletion .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@ build/
.venv/
venv/
__pycache__/
node_modules/
poetry.lock
.pytest_cache/
botpy.log
Expand All @@ -30,4 +31,4 @@ workspaces
# External StarDojo checkout used by the Stardew adapter. Do not commit upstream code.
PhyAgentOS/runtime/adapters/stardewvalley/stardojo/

screen_shot_buffer/
screen_shot_buffer/
7 changes: 6 additions & 1 deletion PhyAgentOS/benchmarks/minecraft/techtree/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,11 @@
inventory_count,
inventory_counts,
)
from PhyAgentOS.benchmarks.minecraft.techtree.harness import BenchmarkResult, run_task
from PhyAgentOS.benchmarks.minecraft.techtree.harness import (
BenchmarkResult,
run_task,
run_task_spec,
)
from PhyAgentOS.benchmarks.minecraft.techtree.loader import list_tasks, load_manifest, load_task
from PhyAgentOS.benchmarks.minecraft.techtree.schema import (
DEFAULT_ARENA_BOUNDARY_BLOCK,
Expand Down Expand Up @@ -41,4 +45,5 @@
"load_manifest",
"load_task",
"run_task",
"run_task_spec",
]
24 changes: 23 additions & 1 deletion PhyAgentOS/benchmarks/minecraft/techtree/harness.py
Original file line number Diff line number Diff line change
Expand Up @@ -70,6 +70,23 @@ def run_task(
"""

task = load_task(task_id, manifest_path)
return run_task_spec(task, agent_fn, world_adapter)


def run_task_spec(
task: TechTreeTask,
agent_fn: AgentFn,
world_adapter: WorldAdapter,
*,
metadata: Mapping[str, Any] | None = None,
) -> BenchmarkResult:
"""Run an already loaded task.

``run_task`` remains the stable manifest-backed API. This companion is
used by the fixed warm-up curriculum, whose targets deliberately do not
appear in the benchmark manifest.
"""

started = _utc_now()
initial_observation: Mapping[str, Any] | None = None
final_observation: Mapping[str, Any] | None = None
Expand Down Expand Up @@ -108,7 +125,12 @@ def run_task(
final_observation=final_observation,
agent_result=agent_result,
error=error,
metadata={"benchmark": "minecraft_techtree", "tier": task.tier, "family": task.family},
metadata={
"benchmark": "minecraft_techtree",
"tier": task.tier,
"family": task.family,
**dict(metadata or {}),
},
)


Expand Down
4 changes: 4 additions & 0 deletions PhyAgentOS/cli/commands.py
Original file line number Diff line number Diff line change
Expand Up @@ -1000,6 +1000,10 @@ async def _trigger():

app.add_typer(stardew_app, name="stardew")

from PhyAgentOS.cli.general_game_commands import run_general_game

app.command("general-game")(run_general_game)


if __name__ == "__main__":
app()
144 changes: 144 additions & 0 deletions PhyAgentOS/cli/general_game_commands.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,144 @@
"""Run Planner/Actor sessions through the existing runtime and target contracts."""

from __future__ import annotations

import json
import math
import os
from contextlib import asynccontextmanager
from importlib.resources import files
from pathlib import Path

import typer
import yaml

from PhyAgentOS.game_agents.stardew import register_general_game
from PhyAgentOS.providers.custom_provider import CustomProvider
from PhyAgentOS.runtime.preflight.runtime_compatibility_preflight import (
RuntimeCompatibilityPreflight,
)
from PhyAgentOS.runtime.schemas import SessionSpec, SkillRuntimeSpec, TargetSpec
from PhyAgentOS.runtime.sessions.session_runner import SessionRunner
from PhyAgentOS.runtime.watchdog.runtime_registry import SkillRuntimeRegistry, TargetRuntimeRegistry
from PhyAgentOS.runtime.watchdog.scheduler import ScheduledSession


def success_check(expected):
if not isinstance(expected, dict) or not expected:
raise ValueError("success checks must be a nonempty {observation.path: value} object")
for path, value in expected.items():
if not isinstance(path, str) or not all(path.split(".")):
raise ValueError("success checks require nonempty observation paths")
if isinstance(value, dict) and "$lt" in value:
limit = value["$lt"]
if set(value) != {"$lt"} or type(limit) not in (int, float) or not math.isfinite(limit):
raise ValueError("$lt requires a single finite numeric upper bound")

def verify(observation, feedback):
for path, expected_value in expected.items():
value = observation
for key in path.split("."):
if not isinstance(value, dict) or key not in value:
return False
value = value[key]
if isinstance(expected_value, dict) and "$lt" in expected_value:
if (
type(value) not in (int, float)
or not math.isfinite(value)
or not value < expected_value["$lt"]
):
return False
elif value != expected_value:
return False
return True

return verify


def run_general_game(
workspace: Path = typer.Option(..., help="Session workspace directory"),
target: Path = typer.Option(..., help="TargetSpec YAML"),
session: Path = typer.Option(..., help="SessionSpec YAML"),
actions: Path = typer.Option(..., help="Action catalog JSON"),
model: str = typer.Option(..., help="Model name"),
api_base: str = typer.Option("http://localhost:8000/v1", help="OpenAI-compatible API base"),
evolve: bool = typer.Option(False, help="Record unverified memory candidates after execution"),
success_checks: Path | None = typer.Option(None, help="Observed completion conditions JSON"),
) -> None:
"""Execute a configured game task with bounded Planner/Actor rounds."""
# Core resolves relative contract paths through this directory, even on first run.
workspace.mkdir(parents=True, exist_ok=True)
target = TargetSpec.model_validate(yaml.safe_load(target.read_text(encoding="utf-8")))
session = SessionSpec.model_validate(yaml.safe_load(session.read_text(encoding="utf-8")))
skill = SkillRuntimeSpec.model_validate(
yaml.safe_load(
files("PhyAgentOS")
.joinpath("templates/configs/skillruntimes/general_game.yaml")
.read_text(encoding="utf-8"),
)
)
if session.skillruntime_ref.removeprefix("skillruntime://") != skill.id:
raise typer.BadParameter("session.skillruntime_ref must reference general_game")
if session.target_ref.removeprefix("target://") != target.id:
raise typer.BadParameter("session.target_ref must match the supplied target")
verify = (
success_check(json.loads(success_checks.read_text(encoding="utf-8")))
if success_checks
else None
)
if (
target.runtime.target_runtime
in {
"StardewValleyTargetRuntime",
"MinecraftTargetRuntime",
}
and verify is None
):
raise typer.BadParameter(
"native game targets require --success-checks to verify task completion"
)

@asynccontextmanager
async def provider_factory():
provider = CustomProvider(
api_key=os.environ.get("GAME_AGENT_API_KEY", "no-key"),
api_base=api_base,
default_model=model,
)
# CustomProvider owns this client; Core has no provider-wide close API.
async with provider._client:
yield provider

register_general_game(
provider_factory,
model=model,
action_catalog=json.loads(actions.read_text(encoding="utf-8")),
memory_workspace=workspace,
evolve=evolve,
verify=verify,
)
scheduled = ScheduledSession(session, target, skill, target.id, skill.id)
preflight = RuntimeCompatibilityPreflight(workspace).check(scheduled)
if preflight.verdict != "accepted":
raise typer.BadParameter(preflight.model_dump_json(indent=2))
runner = SessionRunner(
session=session,
target_spec=target,
skillruntime_spec=skill,
adapter_plan=preflight.adapter_plan,
target=TargetRuntimeRegistry().build(
target,
target_endpoint=session.routing.target_endpoint or target.runtime.target_endpoint,
),
skill_runtime=SkillRuntimeRegistry().build(skill.runtime),
policy_client=None,
perception_runtime=None,
perception_plan=None,
)
try:
result = runner.start()
print(result.model_dump_json(indent=2))
if not result.success:
raise typer.Exit(1)
finally:
runner.close()
73 changes: 73 additions & 0 deletions PhyAgentOS/cli/minecraft_commands.py
Original file line number Diff line number Diff line change
Expand Up @@ -47,6 +47,7 @@
"place: {x,y,z,face} 面编号0=下1=上2=北3=南4=西5=东\n"
"collect: {block_type,count} 自动寻找并采集\n"
"craft: {recipe_id,count} 合成(需附近有工作台)\n"
"smelt: {input,fuel,count} 使用附近熔炉烧炼\n"
"select_slot: {slot:0-8} 切换快捷键\n"
"equip: {item,destination:\"hand\"|\"torso\"|...} 装备指定物品(如 {\"item\":\"wooden_shovel\"})\n"
"drop: {slot} 丢弃物品\n"
Expand Down Expand Up @@ -434,3 +435,75 @@ def minecraft_tp(
target.step({"type": "move", "params": {"dx": x, "dy": y, "dz": z, "absolute": True}})
console.print(f"[green]✓[/green] bot 已传送到 ({x}, {y}, {z})")
target.close()


@minecraft_app.command("warmup")
def minecraft_warmup(
output_dir: str = typer.Option(..., "--output-dir", "-o", help="Skill Graph 输出根目录"),
bridge_url: str = typer.Option("http://127.0.0.1:3001", "--url", "-u"),
):
"""固定运行 W01-W07,每个任务一个 trial,并保存冻结图谱和 benchmark 可写副本。"""
from PhyAgentOS.game_agents.minecraft import run_warmup
from PhyAgentOS.runtime.benchmark.minecraft_glue import MinecraftTargetWorldAdapter
from PhyAgentOS.runtime.targets.game.minecraft_target import MinecraftTarget

target = MinecraftTarget({"bridge_url": bridge_url, "verify_ssl": False})
try:
target.build()
result = run_warmup(MinecraftTargetWorldAdapter(target), output_dir)
except Exception as exc:
console.print(f"[red]预热失败: {exc}[/red]")
raise typer.Exit(1) from exc
finally:
target.close()
console.print_json(data=result)


@minecraft_app.command("benchmark")
def minecraft_benchmark(
output_dir: str = typer.Option(..., "--output-dir", "-o", help="episode 结果目录"),
graph_dir: str = typer.Option(..., "--graph-dir", help="预热产生的 benchmark_graph 目录"),
tasks: str = typer.Option("wooden.obtain_oak_log", "--tasks", help="逗号分隔 task id"),
all_tasks: bool = typer.Option(False, "--all", help="运行 manifest 中全部任务"),
trials: int = typer.Option(1, "--trials", min=1),
run_id: str | None = typer.Option(None, "--run-id", help="可复现的批次 ID;默认自动生成"),
bridge_url: str = typer.Option("http://127.0.0.1:3001", "--url", "-u"),
):
"""串行执行 benchmark,并在每个 episode 后同步沉淀 Skill Graph。"""
from PhyAgentOS.game_agents.minecraft import (
build_scripted_agent,
run_benchmark_tasks,
)
from PhyAgentOS.benchmarks.minecraft.techtree import list_tasks
from PhyAgentOS.runtime.benchmark.minecraft_glue import MinecraftTargetWorldAdapter
from PhyAgentOS.runtime.targets.game.minecraft_target import MinecraftTarget

task_ids = (
[task.id for task in list_tasks()]
if all_tasks
else [task.strip() for task in tasks.split(",") if task.strip()]
)
if not task_ids:
raise typer.BadParameter("--tasks must contain at least one task id")
target = MinecraftTarget({"bridge_url": bridge_url, "verify_ssl": False})
try:
target.build()
results = run_benchmark_tasks(
task_ids,
build_scripted_agent,
MinecraftTargetWorldAdapter(target),
graph_dir=graph_dir,
results_dir=output_dir,
trials=trials,
run_id=run_id,
)
except Exception as exc:
console.print(f"[red]benchmark 失败: {exc}[/red]")
raise typer.Exit(1) from exc
finally:
target.close()
passed = sum(result.success for result in results)
actual_run_id = results[0].metadata.get("run_id") if results else run_id
console.print(
f"[green]完成[/green]: {passed}/{len(results)} episodes passed (run_id={actual_run_id})"
)
23 changes: 23 additions & 0 deletions PhyAgentOS/game_agents/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
# Game Agents

Independent game-agent workflows within the Core package.

```text
game_agents/
├── stardew/ # Planner–Actor runtime, decisions, receipts and role memory
└── minecraft/ # Skill Graph, evidence store, warm-up and benchmark runner
```

| Module | Command | Documentation |
|:-------|:--------|:--------------|
| Stardew | `paos general-game` | [Module](stardew/README.md), [English](../../docs/en/general-game.md), [中文](../../docs/zh/general-game.md) |
| Minecraft | `paos minecraft warmup` / `benchmark` | [Module](minecraft/README.md), [Guide](../../docs/scenarios/game/minecraft/4_benchmark.md) |

The modules do not import each other. Stardew keeps its existing `GeneralGameSkillRuntime`
and registration API; Minecraft keeps its `AgentFn`, `WorldAdapter` and graph APIs.
Game-specific target clients and bridges remain in their existing Core locations.

Model-generated Stardew memory candidates remain unverified. Minecraft graph claims follow
its documented single-observation verification policy. These policies are independent.

Run `python -m pytest tests/game_agents` to exercise both modules without a live game server.
1 change: 1 addition & 0 deletions PhyAgentOS/game_agents/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
"""Independent game-agent workflows hosted by PhyAgentOS Core."""
Loading
Loading