Skip to content

Commit b8e7304

Browse files
committed
feat(yolo-2026): add runtime performance instrumentation with aggregate stats
PerfTracker instruments every pipeline stage: - file_read: frame path check - inference: model prediction - postprocess: bbox extraction + class filtering - emit: JSON serialization Emits perf_stats event every 50 frames with avg/min/max/p50/p95/p99. Also tracks one-time model_load_ms and coreml_export_ms.
1 parent fe1f66a commit b8e7304

1 file changed

Lines changed: 135 additions & 9 deletions

File tree

  • skills/detection/yolo-detection-2026/scripts

‎skills/detection/yolo-detection-2026/scripts/detect.py‎

Lines changed: 135 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -7,6 +7,7 @@
77
stdout: {"event": "detections", "frame_id": N, "camera_id": "...", "objects": [...]}
88
99
On Apple Silicon (MPS), auto-converts to CoreML for ~2x faster inference via ANE.
10+
Emits periodic performance statistics via "perf_stats" events.
1011
1112
Usage:
1213
python detect.py --config config.json
@@ -17,6 +18,7 @@
1718
import json
1819
import argparse
1920
import signal
21+
import time
2022
from pathlib import Path
2123

2224

@@ -28,6 +30,92 @@
2830
"large": "yolo26l",
2931
}
3032

33+
# How often to emit aggregate perf stats (every N frames)
34+
PERF_STATS_INTERVAL = 50
35+
36+
37+
# ───────────────────────────────────────────────────────────────────────────────
38+
# Performance tracker — collects per-frame timings, emits aggregate stats
39+
# ───────────────────────────────────────────────────────────────────────────────
40+
41+
class PerfTracker:
42+
"""Tracks timing for each pipeline stage and emits periodic statistics."""
43+
44+
def __init__(self, interval: int = PERF_STATS_INTERVAL):
45+
self.interval = interval
46+
self.frame_count = 0
47+
self.total_frames = 0
48+
self.error_count = 0
49+
50+
# One-time timings (ms)
51+
self.model_load_ms = 0.0
52+
self.coreml_export_ms = 0.0
53+
54+
# Per-frame accumulators (ms)
55+
self._timings: dict[str, list[float]] = {
56+
"file_read": [], # frame_path existence check + file I/O
57+
"inference": [], # model(frame_path, ...)
58+
"postprocess": [], # bbox extraction + filtering
59+
"emit": [], # JSON serialization + print
60+
"total": [], # end-to-end per frame
61+
}
62+
63+
def record(self, stage: str, duration_ms: float):
64+
"""Record a timing for a pipeline stage."""
65+
if stage in self._timings:
66+
self._timings[stage].append(duration_ms)
67+
68+
def record_frame(self):
69+
"""Increment frame counter and emit stats if interval reached."""
70+
self.frame_count += 1
71+
self.total_frames += 1
72+
if self.frame_count >= self.interval:
73+
self.emit_stats()
74+
self.frame_count = 0
75+
76+
def emit_stats(self):
77+
"""Emit aggregate statistics as a JSONL event."""
78+
stats = {
79+
"event": "perf_stats",
80+
"total_frames": self.total_frames,
81+
"window_size": len(self._timings["total"]) or 1,
82+
"errors": self.error_count,
83+
"model_load_ms": round(self.model_load_ms, 1),
84+
"timings_ms": {},
85+
}
86+
87+
if self.coreml_export_ms > 0:
88+
stats["coreml_export_ms"] = round(self.coreml_export_ms, 1)
89+
90+
for stage, values in self._timings.items():
91+
if not values:
92+
continue
93+
sorted_v = sorted(values)
94+
n = len(sorted_v)
95+
stats["timings_ms"][stage] = {
96+
"avg": round(sum(sorted_v) / n, 2),
97+
"min": round(sorted_v[0], 2),
98+
"max": round(sorted_v[-1], 2),
99+
"p50": round(sorted_v[n // 2], 2),
100+
"p95": round(sorted_v[int(n * 0.95)], 2),
101+
"p99": round(sorted_v[int(n * 0.99)], 2),
102+
}
103+
104+
emit(stats)
105+
106+
# Reset per-frame accumulators for next window
107+
for key in self._timings:
108+
self._timings[key].clear()
109+
110+
def emit_final(self):
111+
"""Emit remaining stats on shutdown."""
112+
if self._timings["total"]:
113+
self.emit_stats()
114+
115+
116+
# ───────────────────────────────────────────────────────────────────────────────
117+
# Helpers
118+
# ───────────────────────────────────────────────────────────────────────────────
31119

32120
def parse_args():
33121
parser = argparse.ArgumentParser(description="YOLO 2026 Detection Skill")
@@ -81,7 +169,6 @@ def select_device(preference: str) -> str:
81169
return "cuda"
82170
if hasattr(torch.backends, "mps") and torch.backends.mps.is_available():
83171
return "mps"
84-
# ROCm exposes as CUDA in PyTorch with ROCm builds
85172
except ImportError:
86173
pass
87174
return "cpu"
@@ -97,8 +184,8 @@ def log(msg: str):
97184
print(f"[YOLO-2026] {msg}", file=sys.stderr, flush=True)
98185

99186

100-
def try_coreml_export(model, model_name: str) -> "Path | None":
101-
"""Export PyTorch model to CoreML. Returns path to .mlpackage or None on failure."""
187+
def try_coreml_export(model, model_name: str, perf: PerfTracker) -> "Path | None":
188+
"""Export PyTorch model to CoreML. Returns path to .mlpackage or None."""
102189
coreml_path = Path(f"{model_name}.mlpackage")
103190

104191
# Already exported
@@ -108,10 +195,12 @@ def try_coreml_export(model, model_name: str) -> "Path | None":
108195

109196
try:
110197
log(f"Exporting {model_name}.pt → CoreML (one-time, ~30s)...")
198+
t0 = time.perf_counter()
111199
exported = model.export(format="coreml", half=True, nms=False)
200+
perf.coreml_export_ms = (time.perf_counter() - t0) * 1000
112201
exported_path = Path(exported)
113202
if exported_path.exists():
114-
log(f"CoreML export complete: {exported_path}")
203+
log(f"CoreML export complete: {exported_path} ({perf.coreml_export_ms:.0f}ms)")
115204
return exported_path
116205
log(f"CoreML export returned path {exported} but file not found")
117206
except Exception as e:
@@ -120,36 +209,44 @@ def try_coreml_export(model, model_name: str) -> "Path | None":
120209
return None
121210

122211

123-
def load_model(model_name: str, device: str, use_coreml: bool):
212+
def load_model(model_name: str, device: str, use_coreml: bool, perf: PerfTracker):
124213
"""Load YOLO model — CoreML on MPS if available, PyTorch otherwise."""
125214
from ultralytics import YOLO
126215

127216
model_format = "pytorch"
217+
t0 = time.perf_counter()
128218

129219
# Try CoreML on Apple Silicon
130220
if device == "mps" and use_coreml:
131221
pt_model = YOLO(f"{model_name}.pt")
132-
coreml_path = try_coreml_export(pt_model, model_name)
222+
coreml_path = try_coreml_export(pt_model, model_name, perf)
133223

134224
if coreml_path:
135225
try:
136226
model = YOLO(str(coreml_path))
137227
model_format = "coreml"
138-
log(f"Loaded CoreML model ({coreml_path})")
228+
perf.model_load_ms = (time.perf_counter() - t0) * 1000
229+
log(f"Loaded CoreML model ({coreml_path}) in {perf.model_load_ms:.0f}ms")
139230
return model, model_format
140231
except Exception as e:
141232
log(f"CoreML load failed, falling back to PyTorch MPS: {e}")
142233

143234
# Fallback: use the already-loaded PyTorch model on MPS
144235
pt_model.to(device)
236+
perf.model_load_ms = (time.perf_counter() - t0) * 1000
145237
return pt_model, model_format
146238

147239
# Non-CoreML path: standard PyTorch
148240
model = YOLO(f"{model_name}.pt")
149241
model.to(device)
242+
perf.model_load_ms = (time.perf_counter() - t0) * 1000
150243
return model, model_format
151244

152245

246+
# ───────────────────────────────────────────────────────────────────────────────
247+
# Main loop
248+
# ───────────────────────────────────────────────────────────────────────────────
249+
153250
def main():
154251
args = parse_args()
155252
config = load_config(args)
@@ -172,9 +269,12 @@ def main():
172269
if isinstance(target_classes, str):
173270
target_classes = [c.strip() for c in target_classes.split(",")]
174271

272+
# Performance tracker
273+
perf = PerfTracker(interval=PERF_STATS_INTERVAL)
274+
175275
# Load YOLO model (with CoreML auto-conversion on MPS)
176276
try:
177-
model, model_format = load_model(model_name, device, use_coreml)
277+
model, model_format = load_model(model_name, device, use_coreml, perf)
178278
emit({
179279
"event": "ready",
180280
"model": f"yolo2026{model_size[0]}",
@@ -183,6 +283,7 @@ def main():
183283
"format": model_format,
184284
"classes": len(model.names),
185285
"fps": fps,
286+
"model_load_ms": round(perf.model_load_ms, 1),
186287
"available_sizes": list(MODEL_SIZE_MAP.keys()),
187288
})
188289
except Exception as e:
@@ -215,23 +316,34 @@ def handle_signal(signum, frame):
215316
break
216317

217318
if msg.get("event") == "frame":
319+
t_frame_start = time.perf_counter()
320+
218321
frame_path = msg.get("frame_path")
219322
frame_id = msg.get("frame_id")
220323
camera_id = msg.get("camera_id", "unknown")
221324
timestamp = msg.get("timestamp", "")
222325

326+
# ── File check ──
327+
t0 = time.perf_counter()
223328
if not frame_path or not Path(frame_path).exists():
224329
emit({
225330
"event": "error",
226331
"frame_id": frame_id,
227332
"message": f"Frame not found: {frame_path}",
228333
"retriable": True,
229334
})
335+
perf.error_count += 1
230336
continue
337+
perf.record("file_read", (time.perf_counter() - t0) * 1000)
231338

232-
# Run inference
339+
# ── Inference ──
233340
try:
341+
t0 = time.perf_counter()
234342
results = model(frame_path, conf=confidence, verbose=False)
343+
perf.record("inference", (time.perf_counter() - t0) * 1000)
344+
345+
# ── Postprocess ──
346+
t0 = time.perf_counter()
235347
objects = []
236348
for r in results:
237349
for box in r.boxes:
@@ -244,21 +356,35 @@ def handle_signal(signum, frame):
244356
"confidence": round(float(box.conf[0]), 3),
245357
"bbox": [int(x1), int(y1), int(x2), int(y2)],
246358
})
359+
perf.record("postprocess", (time.perf_counter() - t0) * 1000)
247360

361+
# ── Emit ──
362+
t0 = time.perf_counter()
248363
emit({
249364
"event": "detections",
250365
"frame_id": frame_id,
251366
"camera_id": camera_id,
252367
"timestamp": timestamp,
253368
"objects": objects,
254369
})
370+
perf.record("emit", (time.perf_counter() - t0) * 1000)
371+
255372
except Exception as e:
256373
emit({
257374
"event": "error",
258375
"frame_id": frame_id,
259376
"message": f"Inference error: {e}",
260377
"retriable": True,
261378
})
379+
perf.error_count += 1
380+
continue
381+
382+
# ── Total frame time ──
383+
perf.record("total", (time.perf_counter() - t_frame_start) * 1000)
384+
perf.record_frame()
385+
386+
# Emit final stats on shutdown
387+
perf.emit_final()
262388

263389

264390
if __name__ == "__main__":

0 commit comments

Comments
 (0)