Repository navigation
Expand file tree
/
Copy pathroofline_suite.py
More file actions
240 lines (207 loc) · 11.4 KB
/
Copy pathroofline_suite.py
File metadata and controls
240 lines (207 loc) · 11.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
#!/usr/bin/env python3
"""Roofline suite: one command that fingerprints the machine, runs the full
roofline battery, and emits a single merged report keyed to that machine.
Design (agreed): this SITS ABOVE the existing bench scripts rather than forking
them. The performance rooflines already live in dedicated scripts, each writing
its own canonical results/*.json that the paper cites - so the suite subprocesses
those (leaving their JSONs untouched and citable) and collects them by reference.
The one plane that has no script yet - the numeric CORRECTNESS cliffs from issue
#115 and docs/cross-chip.md - runs in-process via bench/numeric_cliffs.py.
The merged report carries the machine fingerprint (bench/_machine.py), so many
contributors on different Apple Silicon machines can each commit a
non-colliding roofline-<chip>-<model>-<hwhash>-<runid>.json, and a later
aggregator can pool them by hardware_hash without mixing an M2 Air with an M2
Mac mini.
Run:
PYTHONPATH=. python3 bench/roofline_suite.py # fingerprint + numeric cliffs (fast, no sudo)
PYTHONPATH=. python3 bench/roofline_suite.py --perf # + fast headline perf (a few min). Prompts once
# for sudo to read watts -- REQUIRED; aborts if refused.
PYTHONPATH=. python3 bench/roofline_suite.py --perf-full # + full paper-grade battery (~30 min)
"""
from __future__ import annotations
import argparse
import json
import os
import subprocess
import sys
import threading
import time
from pathlib import Path
os.environ.setdefault("KMP_DUPLICATE_LIB_OK", "TRUE")
REPO = Path(__file__).resolve().parents[1]
if str(REPO) not in sys.path:
sys.path.insert(0, str(REPO))
sys.path.insert(0, str(Path(__file__).resolve().parent)) # sibling bench modules
import _machine # noqa: E402
import numeric_cliffs # noqa: E402
RESULTS = REPO / "bench" / "results"
ROOFLINES_DIR = RESULTS / "rooflines" # one immutable JSON per submission; PR'd in
# Perf scripts run in dependency order (roofline_analysis reads the saturation +
# bandwidth JSONs, so it runs last). Each entry: (script, result_file, fast_args).
#
# FAST is the contributor default (`--perf`): only the scripts the headline table
# needs, with --quick sweeps and short power windows -> a few minutes. FULL is the
# complete paper-grade battery (`--perf-full`): every script at full sampling.
PERF_FAST = [
("device_saturation_sweep.py", "device_saturation_sweep_results.json", ["--quick"]),
("device_bandwidth_roofline.py", "device_bandwidth_roofline_results.json", ["--quick", "--window", "2"]),
("decode_measurement.py", "decode_measurement_results.json", ["--quick"]),
("roofline_analysis.py", "roofline_analysis_results.json", []),
]
PERF_FULL = [
("device_saturation_sweep.py", "device_saturation_sweep_results.json", []),
("device_bandwidth_roofline.py", "device_bandwidth_roofline_results.json", []),
("device_serving_sweep.py", "device_serving_sweep_results.json", []),
("decode_measurement.py", "decode_measurement_results.json", []),
("device_compare_wattcomplete.py", "device_compare_wattcomplete_results.json", []),
("roofline_analysis.py", "roofline_analysis_results.json", []),
]
def _run_perf_script(script: str, result_file: str, extra_args: list[str]) -> dict:
"""Subprocess one existing bench script and collect its canonical JSON by reference.
Does NOT restore the result file here: some scripts read earlier scripts'
outputs (roofline_analysis synthesizes from the saturation + bandwidth JSONs),
so restoring per-script would feed the synthesis stale committed data instead
of this run's fresh numbers. The whole batch is snapshotted and restored around
the loop in main() instead, keeping the dependency chain intact.
"""
path = REPO / "bench" / script
out_path = RESULTS / result_file
t0 = time.perf_counter()
proc = subprocess.run(
[sys.executable, str(path), *extra_args],
cwd=str(REPO), env={**os.environ, "PYTHONPATH": str(REPO)},
capture_output=True, text=True,
)
dt = time.perf_counter() - t0
entry = {"script": script, "result_file": result_file,
"returncode": proc.returncode, "seconds": round(dt, 1)}
if proc.returncode != 0:
entry["error"] = proc.stderr.strip().splitlines()[-1:] or ["(no stderr)"]
else:
try:
entry["summary"] = json.loads(out_path.read_text())
except Exception as e:
entry["error"] = f"could not read {result_file}: {e}"
return entry
def _snapshot(result_files: list[str]) -> dict[str, bytes | None]:
"""Prior bytes of each reference JSON (None == did not exist)."""
return {rf: (RESULTS / rf).read_bytes() if (RESULTS / rf).exists() else None
for rf in result_files}
def _restore(snap: dict[str, bytes | None]) -> None:
"""Put each reference JSON back exactly as it was before the batch ran."""
for rf, prior in snap.items():
p = RESULTS / rf
if prior is not None:
p.write_bytes(prior)
elif p.exists():
p.unlink()
def _authorize_sudo() -> bool:
"""Interactively cache sudo credentials (prompts once) so the perf scripts'
`sudo -n powermetrics` calls succeed without passwordless sudo. Needs a TTY."""
try:
return subprocess.run(["sudo", "-v"]).returncode == 0 # inherits stdio -> prompts
except Exception:
return False
def _sudo_keepalive_start() -> threading.Event:
"""Refresh the sudo timestamp every 60s so it does not expire mid-run (the long
saturation sweep can outlast the default 5-min sudo timeout between power reads)."""
stop = threading.Event()
def loop():
while not stop.wait(60):
subprocess.run(["sudo", "-n", "true"], capture_output=True)
threading.Thread(target=loop, daemon=True).start()
return stop
def main():
ap = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--perf", action="store_true",
help="fast headline perf run (a few min; --quick sweeps, short power windows)")
ap.add_argument("--perf-full", dest="perf_full", action="store_true",
help="full paper-grade perf battery (all scripts, full sampling; ~30 min)")
ap.add_argument("--out", default=None, help="explicit output path (default: fingerprinted name)")
ap.add_argument("--contributor", default=None,
help="your GitHub handle to be credited (overrides auto-detection)")
args = ap.parse_args()
keepalive = None
if args.perf or args.perf_full:
# A perf run reads per-rail power with `sudo powermetrics` -- authorize it FIRST,
# before any measurement, and ABORT if refused. A perf run without watts is not a
# valid submission, so there is no opt-out. Passwordless sudo authorizes silently;
# otherwise this prompts once for your login password.
print("This perf run reads per-rail power and needs sudo. It will run, repeatedly:")
print(" sudo powermetrics --samplers ane_power,cpu_power,gpu_power")
print("(read-only power sampling; nothing is modified.)")
if not _authorize_sudo():
print("\nsudo not granted -- aborting. (Run without --perf for the cliffs-only "
"submission, which needs no sudo.)", file=sys.stderr)
sys.exit(1)
keepalive = _sudo_keepalive_start()
fp = _machine.fingerprint()
print("machine:", fp["hardware"]["chip"], fp["hardware"]["model_identifier"],
" hwhash", fp["hardware_hash"], " sudo", fp["environment"]["have_sudo"])
# Credit: explicit flag wins; otherwise auto-detect from a GitHub noreply email.
contributor = args.contributor or _machine.github_handle()
contributor = contributor.lstrip("@") if contributor else None
if contributor:
src = "flag" if args.contributor else "auto-detected from git noreply email"
print(f"contributor: @{contributor} ({src})")
else:
print("contributor: none (git email is not a GitHub noreply; pass --contributor to be credited)")
report = {
"suite": "roofline",
"schema_version": 2,
"contributor": contributor,
"machine": fp,
"numeric_cliffs": None,
"perf_rooflines": None,
}
print("\n[numeric cliffs] correctness rooflines (issue #115, docs/cross-chip.md) ...")
report["numeric_cliffs"] = numeric_cliffs.run()
mm = report["numeric_cliffs"]["matmul_saturation"]["by_K"][0].get("cliff")
sl = report["numeric_cliffs"]["slice_saturation"].get("cliff")
rex = report["numeric_cliffs"]["reduce_exactness"].get("last_all_exact")
print(f" matmul inf-cliff ~{mm} | slice cliff {sl} (None=exact/A16+) | reduce exact <= {rex}")
if args.perf or args.perf_full:
scripts = PERF_FULL if args.perf_full else PERF_FAST
label = "full paper-grade battery (~30 min)" if args.perf_full else "fast headline run (a few min)"
print(f"\n[perf] {label}") # sudo already authorized up front
report["machine"]["environment"]["sudo_authorized"] = True
pw = fp["environment"]["power"]
if pw.get("is_laptop") and pw.get("source") == "battery":
mode = (pw.get("energy_mode") or {}).get("mode")
if mode == "high_power":
print(f"[perf] note: on battery ({pw.get('battery_pct')}%) but in High Power mode "
"-> close to AC over short bench windows; the state is recorded in the report.")
else:
print(f"[perf] WARNING: on battery ({pw.get('battery_pct')}%), energy mode "
f"'{mode}' -> clocks may throttle; High Power mode or AC gives cleaner perf "
"rooflines. (Numeric cliffs are unaffected.) The state is recorded either way.")
# Snapshot every reference JSON once, run the whole batch (so roofline_analysis
# reads THIS run's fresh saturation/bandwidth), then restore them all at the end.
snap = _snapshot([rf for _, rf, _ in scripts])
collected = []
try:
for script, result_file, fast_args in scripts:
print(f"\n[perf] {script} ...", flush=True)
entry = _run_perf_script(script, result_file, fast_args)
state = "ok" if entry["returncode"] == 0 else f"FAILED rc={entry['returncode']}"
print(f" {state} in {entry.get('seconds')}s")
collected.append(entry)
finally:
_restore(snap)
if keepalive is not None:
keepalive.set()
report["perf_rooflines"] = collected
# where the perf/W's ANE power came from (issue #289): powermetrics, or the SMC rail
sat = next((e.get("summary") for e in collected
if e.get("script") == "device_saturation_sweep.py" and e.get("summary")), None)
report["ane_power_source"] = ((sat or {}).get("meta") or {}).get("ane_power_source")
else:
print("\n[perf] skipped (pass --perf for the fast headline run, --perf-full for everything)")
out = Path(args.out) if args.out else ROOFLINES_DIR / _machine.result_filename(fp)
out.parent.mkdir(parents=True, exist_ok=True)
out.write_text(json.dumps(report, indent=2))
print(f"\nwrote merged report -> {out}")
print("to publish: commit this file and run `python3 bench/aggregate_rooflines.py`, then open a PR")
if __name__ == "__main__":
main()