Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 3 additions & 2 deletions plugins/openclaw/slash_sleep.py
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,7 @@
import json
import os
import shutil
import subprocess
import sys
from pathlib import Path
from datetime import datetime
Expand Down Expand Up @@ -104,8 +105,8 @@ def run_category(category: str, *, dry_run: bool = False) -> int:

print(f"=== /sleep run {category}{' (dry-run)' if dry_run else ''} ===")
print(f" cmd: {' '.join(cmd)}")
rc = os.system(" ".join(f'"{c}"' for c in cmd))
return rc
result = subprocess.run(cmd)
return result.returncode


def run_all(*, dry_run: bool = False) -> int:
Expand Down
6 changes: 4 additions & 2 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -38,10 +38,12 @@ dependencies = [
alfworld = ["alfworld>=0.4.0", "gymnasium>=0.29.0"]
# Claude model backend
claude = ["claude-agent-sdk>=0.1.0", "json_repair>=0.61.0"]
# Codex model backend (via OpenAI Codex SDK)
codex = ["openai-codex-sdk>=0.1.0"]
# Qwen local model backend (via vLLM)
qwen = ["vllm>=0.4.0", "json_repair>=0.61.0"]
qwen = ["vllm>=0.8.4", "json_repair>=0.61.0"]
# SearchQA data materialization
searchqa = ["datasets>=2.18.0"]
searchqa = ["datasets>=3.0"]
# Documentation site
docs = ["mkdocs-material>=9.5.0", "mkdocstrings[python]>=0.24.0"]
# WebUI dashboard
Expand Down
4 changes: 2 additions & 2 deletions requirements.txt
Original file line number Diff line number Diff line change
Expand Up @@ -15,7 +15,7 @@ httpx>=0.27.0
# claude-agent-sdk>=0.1.0

# ── Optional: Qwen local model (via vLLM) ────────
# vllm>=0.4.0
# vllm>=0.8.4

# ── Optional: tolerant JSON repair for free-form output from non-OpenAI
# backends (Claude/Qwen). Without it extract_json() falls back safely and
Expand All @@ -24,7 +24,7 @@ httpx>=0.27.0
# json_repair>=0.61.0

# ── Optional: WebUI dashboard ────────────────────
# gradio>=4.0.0
# gradio>=5.50.0

# ── Optional: Documentation site ─────────────────
# mkdocs-material>=9.5.0
Expand Down
3 changes: 2 additions & 1 deletion skillopt/envs/spreadsheetbench/rollout.py
Original file line number Diff line number Diff line change
Expand Up @@ -314,7 +314,8 @@ def process_one(
# ── Stage 1: run ReAct agent on test case 1 ─────────────────────
result["phase"] = "agent"

work_dir = tempfile.mkdtemp(prefix=f"react_{task_id}_")
safe_task_id = "".join(c if c.isalnum() or c in "-_" else "_" for c in str(task_id))
work_dir = tempfile.mkdtemp(prefix=f"react_{safe_task_id}_")
try:
# Copy input so agent works in an isolated directory
work_input = os.path.join(work_dir, os.path.basename(ip1))
Expand Down
123 changes: 83 additions & 40 deletions skillopt_webui/app.py
Original file line number Diff line number Diff line change
Expand Up @@ -46,6 +46,58 @@ def load_config(path: str) -> dict:
return yaml.safe_load(f)


def scan_outputs(out_dir: str) -> list:
"""Digest experiment results strictly under PROJECT_ROOT.

The Output Explorer callback. Any path that escapes PROJECT_ROOT is
rejected (empty result) at the point data is read, so a traversal arg
can never read files outside the project.
"""
rows = []
if not out_dir:
return rows
base = (PROJECT_ROOT / out_dir).resolve()
project_resolved = PROJECT_ROOT.resolve()
try:
base.relative_to(project_resolved)
except ValueError:
return rows
if not base.exists() or not base.is_dir():
return rows
for bench_dir in sorted(base.iterdir()):
if not bench_dir.is_dir():
continue
for run_dir in sorted(bench_dir.iterdir()):
if not run_dir.is_dir():
continue
cfg_file = run_dir / "config.yaml"
score = "—"
steps = "—"
if cfg_file.exists():
try:
c = yaml.safe_load(cfg_file.read_text())
steps = str(c.get("train", {}).get("num_steps", "—"))
except Exception:
pass
# Try to find best score from logs
for log_f in run_dir.glob("**/*.jsonl"):
try:
with open(log_f) as f:
for line in f:
d = json.loads(line)
if "score" in d:
score = f"{d['score']:.4f}"
except Exception:
pass
rows.append([
run_dir.name,
bench_dir.name,
score,
steps,
])
return rows


def config_to_display(cfg: dict) -> str:
"""Pretty-print config for display."""
return yaml.dump(cfg, default_flow_style=False, sort_keys=False)
Expand Down Expand Up @@ -602,46 +654,6 @@ def on_refresh():
label="Experiments",
)

def scan_outputs(out_dir):
rows = []
if not out_dir:
return rows
base = PROJECT_ROOT / out_dir
if not base.exists():
return rows
for bench_dir in sorted(base.iterdir()):
if not bench_dir.is_dir():
continue
for run_dir in sorted(bench_dir.iterdir()):
if not run_dir.is_dir():
continue
cfg_file = run_dir / "config.yaml"
score = "—"
steps = "—"
if cfg_file.exists():
try:
c = yaml.safe_load(cfg_file.read_text())
steps = str(c.get("train", {}).get("num_steps", "—"))
except Exception:
pass
# Try to find best score from logs
for log_f in run_dir.glob("**/*.jsonl"):
try:
with open(log_f) as f:
for line in f:
d = json.loads(line)
if "score" in d:
score = f"{d['score']:.4f}"
except Exception:
pass
rows.append([
run_dir.name,
bench_dir.name,
score,
steps,
])
return rows

scan_btn.click(scan_outputs, output_dir, results_table)

return app
Expand All @@ -668,6 +680,10 @@ def main():
parser.add_argument("--host", type=str, default="127.0.0.1",
help="Server host. Default is localhost; use 0.0.0.0 "
"to expose publicly (no auth, use with care).")
parser.add_argument("--auth-user", type=str, default=None,
help="Username for basic auth (or set SKILLOPT_WEBUI_USER).")
parser.add_argument("--auth-pass", type=str, default=None,
help="Password for basic auth (or set SKILLOPT_WEBUI_PASS).")
args = parser.parse_args()

if args.host and args.host not in ("127.0.0.1", "localhost", "::1"):
Expand All @@ -679,8 +695,35 @@ def main():
file=sys.stderr,
)

if args.share:
print(
"⚠ warning: --share creates a public tunnel (gradio.live) with no "
"authentication by default. Anyone with the URL can start/stop "
"training and browse the filesystem via Output Explorer. "
"Use --auth-user / --auth-pass (or SKILLOPT_WEBUI_USER / "
"SKILLOPT_WEBUI_PASS) to require login.",
file=sys.stderr,
)

auth_user = args.auth_user or os.environ.get("SKILLOPT_WEBUI_USER")
auth_pass = args.auth_pass or os.environ.get("SKILLOPT_WEBUI_PASS")
# Fail-closed: authentication requires BOTH credentials. Supplying only a
# username or only a password must not silently launch the UI unauthenticated
# (a deployment could expose the training controls without login).
if bool(auth_user) != bool(auth_pass):
print(
"SKILLOPT_WEBUI authentication requires BOTH --auth-user and "
"--auth-pass (or SKILLOPT_WEBUI_USER and SKILLOPT_WEBUI_PASS). "
"Refusing to start with incomplete credentials.",
file=sys.stderr,
)
sys.exit(1)
auth = (auth_user, auth_pass) if auth_user else None

app = build_ui()
launch_kwargs = build_launch_kwargs(args.host, args.port, args.share)
if auth:
launch_kwargs["auth"] = auth
app.launch(**launch_kwargs)


Expand Down
153 changes: 153 additions & 0 deletions tests/test_webui_security.py
Original file line number Diff line number Diff line change
Expand Up @@ -55,3 +55,156 @@ def test_main_warns_on_public_host(webui, monkeypatch, capsys):
assert "warning" in captured.err.lower()
_args, kwargs = launcher.call_args
assert kwargs["server_name"] == "0.0.0.0"


def test_main_warns_on_share(webui, monkeypatch, capsys):
"""--share must emit a public-tunnel warning."""
webui_mod = webui
launcher = mock.MagicMock()
app_mock = mock.MagicMock()
app_mock.launch = launcher
monkeypatch.setattr(webui_mod, "build_ui", lambda: app_mock)
monkeypatch.setattr(sys, "argv", ["app.py", "--share"])

webui_mod.main()

captured = capsys.readouterr()
assert "share" in captured.err.lower()
assert "public" in captured.err.lower() or "tunnel" in captured.err.lower()


def test_main_auth_via_cli_args(webui, monkeypatch):
"""--auth-user and --auth-pass must enable Gradio basic auth."""
webui_mod = webui
launcher = mock.MagicMock()
app_mock = mock.MagicMock()
app_mock.launch = launcher
monkeypatch.setattr(webui_mod, "build_ui", lambda: app_mock)
monkeypatch.setattr(sys, "argv", ["app.py", "--auth-user", "admin", "--auth-pass", "s3cret"])

webui_mod.main()

_args, kwargs = launcher.call_args
assert kwargs.get("auth") == ("admin", "s3cret")


def test_main_auth_via_env(webui, monkeypatch):
"""SKILLOPT_WEBUI_USER / SKILLOPT_WEBUI_PASS must enable auth without CLI args."""
webui_mod = webui
launcher = mock.MagicMock()
app_mock = mock.MagicMock()
app_mock.launch = launcher
monkeypatch.setattr(webui_mod, "build_ui", lambda: app_mock)
monkeypatch.setattr(sys, "argv", ["app.py"])
monkeypatch.setenv("SKILLOPT_WEBUI_USER", "envuser")
monkeypatch.setenv("SKILLOPT_WEBUI_PASS", "envpass")

webui_mod.main()

_args, kwargs = launcher.call_args
assert kwargs.get("auth") == ("envuser", "envpass")


def test_main_no_auth_by_default(webui, monkeypatch):
"""Without auth args or env vars, no auth must be configured."""
webui_mod = webui
launcher = mock.MagicMock()
app_mock = mock.MagicMock()
app_mock.launch = launcher
monkeypatch.setattr(webui_mod, "build_ui", lambda: app_mock)
monkeypatch.setattr(sys, "argv", ["app.py"])
monkeypatch.delenv("SKILLOPT_WEBUI_USER", raising=False)
monkeypatch.delenv("SKILLOPT_WEBUI_PASS", raising=False)

webui_mod.main()

_args, kwargs = launcher.call_args
assert "auth" not in kwargs or kwargs["auth"] is None


def test_scan_outputs_rejects_path_traversal(webui, tmp_path, monkeypatch):
"""The scan_outputs callback must not enumerate directories outside PROJECT_ROOT."""
monkeypatch.setattr(webui, "PROJECT_ROOT", tmp_path)
(tmp_path / "outputs").mkdir()
# Every traversal / escape form is denied at consumption: no rows, no reads.
for bad in ("/../../etc/passwd", "../outside", "outputs/../../../etc", "C:\\Windows"):
assert webui.scan_outputs(bad) == [], f"traversal {bad!r} must be denied"


def test_scan_outputs_allows_valid_subdir(webui, tmp_path, monkeypatch):
"""scan_outputs must accept directories within PROJECT_ROOT."""
monkeypatch.setattr(webui, "PROJECT_ROOT", tmp_path)
(tmp_path / "outputs" / "bench1" / "run1").mkdir(parents=True)
(tmp_path / "outputs" / "bench1" / "run1" / "config.yaml").write_text("a: 1\n", encoding="utf-8")
rows = webui.scan_outputs("outputs")
assert rows, "valid in-tree output area must be digested"


def test_scan_outputs_callback_consumes_within_project(webui, tmp_path, monkeypatch):
"""The registered scan_outputs callback must digest data only inside PROJECT_ROOT
at the point data is actually read (traversal denied, in-tree consumed)."""
monkeypatch.setattr(webui, "PROJECT_ROOT", tmp_path)
# A traversal arg must be denied at consumption: no rows, no data read.
assert webui.scan_outputs("/../../etc/passwd") == []
assert webui.scan_outputs("../outside") == []
# A valid in-tree output area is digested (config.yaml read per run dir).
(tmp_path / "outputs/bench1/run1").mkdir(parents=True)
(tmp_path / "outputs/bench1/run1/config.yaml").write_text("alpha: 1\n", encoding="utf-8")
rows = webui.scan_outputs("outputs")
assert rows, f"expected rows from a valid in-tree output area, got {rows!r}"


def test_main_rejects_incomplete_cli_auth_user_only(webui, monkeypatch):
"""--auth-user without --auth-pass must fail closed (never launch)."""
webui_mod = webui
launcher = mock.MagicMock()
app_mock = mock.MagicMock()
app_mock.launch = launcher
monkeypatch.setattr(webui_mod, "build_ui", lambda: app_mock)
monkeypatch.setattr(sys, "argv", ["app.py", "--host", "0.0.0.0", "--auth-user", "admin"])
with pytest.raises(SystemExit):
webui_mod.main()
launcher.assert_not_called()


def test_main_rejects_incomplete_cli_auth_pass_only(webui, monkeypatch):
"""--auth-pass without --auth-user must fail closed (never launch)."""
webui_mod = webui
launcher = mock.MagicMock()
app_mock = mock.MagicMock()
app_mock.launch = launcher
monkeypatch.setattr(webui_mod, "build_ui", lambda: app_mock)
monkeypatch.setattr(sys, "argv", ["app.py", "--host", "0.0.0.0", "--auth-pass", "s3cret"])
with pytest.raises(SystemExit):
webui_mod.main()
launcher.assert_not_called()


def test_main_rejects_incomplete_env_auth_user_only(webui, monkeypatch):
"""Only SKILLOPT_WEBUI_USER set must fail closed (never launch)."""
webui_mod = webui
launcher = mock.MagicMock()
app_mock = mock.MagicMock()
app_mock.launch = launcher
monkeypatch.setattr(webui_mod, "build_ui", lambda: app_mock)
monkeypatch.setattr(sys, "argv", ["app.py", "--host", "0.0.0.0"])
monkeypatch.setenv("SKILLOPT_WEBUI_USER", "envuser")
monkeypatch.delenv("SKILLOPT_WEBUI_PASS", raising=False)
with pytest.raises(SystemExit):
webui_mod.main()
launcher.assert_not_called()


def test_main_rejects_incomplete_env_auth_pass_only(webui, monkeypatch):
"""Only SKILLOPT_WEBUI_PASS set must fail closed (never launch)."""
webui_mod = webui
launcher = mock.MagicMock()
app_mock = mock.MagicMock()
app_mock.launch = launcher
monkeypatch.setattr(webui_mod, "build_ui", lambda: app_mock)
monkeypatch.setattr(sys, "argv", ["app.py", "--host", "0.0.0.0"])
monkeypatch.setenv("SKILLOPT_WEBUI_PASS", "envpass")
monkeypatch.delenv("SKILLOPT_WEBUI_USER", raising=False)
with pytest.raises(SystemExit):
webui_mod.main()
launcher.assert_not_called()