Skip to content

Commit 32a29ad

Browse files
committed
test(webui): exercise scan_outputs callback at the data-consumption point
Lift scan_outputs out of the build_ui closure so the Output Explorer callback is directly testable, and add callback-level tests that call it with traversal args (denied, returns []) and a valid in-tree output area (digested, reads config.yaml). This replaces the prior approximation tests that only re-checked relative_to() in isolation.
1 parent d417c4c commit 32a29ad

2 files changed

Lines changed: 76 additions & 74 deletions

File tree

‎skillopt_webui/app.py‎

Lines changed: 52 additions & 45 deletions
Original file line numberDiff line numberDiff line change
@@ -46,6 +46,58 @@ def load_config(path: str) -> dict:
4646
return yaml.safe_load(f)
4747

4848

49+
def scan_outputs(out_dir: str) -> list:
50+
"""Digest experiment results strictly under PROJECT_ROOT.
51+
52+
The Output Explorer callback. Any path that escapes PROJECT_ROOT is
53+
rejected (empty result) at the point data is read, so a traversal arg
54+
can never read files outside the project.
55+
"""
56+
rows = []
57+
if not out_dir:
58+
return rows
59+
base = (PROJECT_ROOT / out_dir).resolve()
60+
project_resolved = PROJECT_ROOT.resolve()
61+
try:
62+
base.relative_to(project_resolved)
63+
except ValueError:
64+
return rows
65+
if not base.exists() or not base.is_dir():
66+
return rows
67+
for bench_dir in sorted(base.iterdir()):
68+
if not bench_dir.is_dir():
69+
continue
70+
for run_dir in sorted(bench_dir.iterdir()):
71+
if not run_dir.is_dir():
72+
continue
73+
cfg_file = run_dir / "config.yaml"
74+
score = "—"
75+
steps = "—"
76+
if cfg_file.exists():
77+
try:
78+
c = yaml.safe_load(cfg_file.read_text())
79+
steps = str(c.get("train", {}).get("num_steps", "—"))
80+
except Exception:
81+
pass
82+
# Try to find best score from logs
83+
for log_f in run_dir.glob("**/*.jsonl"):
84+
try:
85+
with open(log_f) as f:
86+
for line in f:
87+
d = json.loads(line)
88+
if "score" in d:
89+
score = f"{d['score']:.4f}"
90+
except Exception:
91+
pass
92+
rows.append([
93+
run_dir.name,
94+
bench_dir.name,
95+
score,
96+
steps,
97+
])
98+
return rows
99+
100+
49101
def config_to_display(cfg: dict) -> str:
50102
"""Pretty-print config for display."""
51103
return yaml.dump(cfg, default_flow_style=False, sort_keys=False)
@@ -602,51 +654,6 @@ def on_refresh():
602654
label="Experiments",
603655
)
604656

605-
def scan_outputs(out_dir):
606-
rows = []
607-
if not out_dir:
608-
return rows
609-
base = (PROJECT_ROOT / out_dir).resolve()
610-
project_resolved = PROJECT_ROOT.resolve()
611-
try:
612-
base.relative_to(project_resolved)
613-
except ValueError:
614-
return rows
615-
if not base.exists() or not base.is_dir():
616-
return rows
617-
for bench_dir in sorted(base.iterdir()):
618-
if not bench_dir.is_dir():
619-
continue
620-
for run_dir in sorted(bench_dir.iterdir()):
621-
if not run_dir.is_dir():
622-
continue
623-
cfg_file = run_dir / "config.yaml"
624-
score = "—"
625-
steps = "—"
626-
if cfg_file.exists():
627-
try:
628-
c = yaml.safe_load(cfg_file.read_text())
629-
steps = str(c.get("train", {}).get("num_steps", "—"))
630-
except Exception:
631-
pass
632-
# Try to find best score from logs
633-
for log_f in run_dir.glob("**/*.jsonl"):
634-
try:
635-
with open(log_f) as f:
636-
for line in f:
637-
d = json.loads(line)
638-
if "score" in d:
639-
score = f"{d['score']:.4f}"
640-
except Exception:
641-
pass
642-
rows.append([
643-
run_dir.name,
644-
bench_dir.name,
645-
score,
646-
steps,
647-
])
648-
return rows
649-
650657
scan_btn.click(scan_outputs, output_dir, results_table)
651658

652659
return app

‎tests/test_webui_security.py‎

Lines changed: 24 additions & 29 deletions
Original file line numberDiff line numberDiff line change
@@ -123,40 +123,35 @@ def test_main_no_auth_by_default(webui, monkeypatch):
123123

124124

125125
def test_scan_outputs_rejects_path_traversal(webui, tmp_path, monkeypatch):
126-
"""scan_outputs must not enumerate directories outside PROJECT_ROOT."""
127-
webui_mod = webui
128-
monkeypatch.setattr(webui_mod, "PROJECT_ROOT", tmp_path)
126+
"""The scan_outputs callback must not enumerate directories outside PROJECT_ROOT."""
127+
monkeypatch.setattr(webui, "PROJECT_ROOT", tmp_path)
129128
(tmp_path / "outputs").mkdir()
130-
131-
outside = tmp_path / "outputs"
132-
result = webui_mod.build_ui.__wrapped__ if hasattr(webui_mod.build_ui, "__wrapped__") else None
133-
134-
from pathlib import Path
135-
base = (tmp_path / "outputs" / "../../etc").resolve()
136-
project_resolved = tmp_path.resolve()
137-
try:
138-
base.relative_to(project_resolved)
139-
escaped = False
140-
except ValueError:
141-
escaped = True
142-
assert escaped, "Path traversal via Output Explorer must be blocked"
129+
# Every traversal / escape form is denied at consumption: no rows, no reads.
130+
for bad in ("/../../etc/passwd", "../outside", "outputs/../../../etc", "C:\\Windows"):
131+
assert webui.scan_outputs(bad) == [], f"traversal {bad!r} must be denied"
143132

144133

145134
def test_scan_outputs_allows_valid_subdir(webui, tmp_path, monkeypatch):
146135
"""scan_outputs must accept directories within PROJECT_ROOT."""
147-
from pathlib import Path
148-
project = tmp_path
149-
monkeypatch.setattr(webui, "PROJECT_ROOT", project)
150-
(project / "outputs" / "bench1" / "run1").mkdir(parents=True)
151-
152-
base = (project / "outputs").resolve()
153-
project_resolved = project.resolve()
154-
try:
155-
base.relative_to(project_resolved)
156-
contained = True
157-
except ValueError:
158-
contained = False
159-
assert contained, "Valid subdirectory must pass containment check"
136+
monkeypatch.setattr(webui, "PROJECT_ROOT", tmp_path)
137+
(tmp_path / "outputs" / "bench1" / "run1").mkdir(parents=True)
138+
(tmp_path / "outputs" / "bench1" / "run1" / "config.yaml").write_text("a: 1\n", encoding="utf-8")
139+
rows = webui.scan_outputs("outputs")
140+
assert rows, "valid in-tree output area must be digested"
141+
142+
143+
def test_scan_outputs_callback_consumes_within_project(webui, tmp_path, monkeypatch):
144+
"""The registered scan_outputs callback must digest data only inside PROJECT_ROOT
145+
at the point data is actually read (traversal denied, in-tree consumed)."""
146+
monkeypatch.setattr(webui, "PROJECT_ROOT", tmp_path)
147+
# A traversal arg must be denied at consumption: no rows, no data read.
148+
assert webui.scan_outputs("/../../etc/passwd") == []
149+
assert webui.scan_outputs("../outside") == []
150+
# A valid in-tree output area is digested (config.yaml read per run dir).
151+
(tmp_path / "outputs/bench1/run1").mkdir(parents=True)
152+
(tmp_path / "outputs/bench1/run1/config.yaml").write_text("alpha: 1\n", encoding="utf-8")
153+
rows = webui.scan_outputs("outputs")
154+
assert rows, f"expected rows from a valid in-tree output area, got {rows!r}"
160155

161156

162157
def test_main_rejects_incomplete_cli_auth_user_only(webui, monkeypatch):

0 commit comments

Comments
 (0)