# -*- coding: utf-8 -*- """Hardware and environment calibration for paper 08. One script, identical on every arm, no dependencies beyond the standard library. It measures the things a wall-clock comparison between operating systems would otherwise silently confound, so a reader can discount the timings themselves rather than take them on trust. Six measures: cpu a fixed integer and float loop. Pure core speed bulk_write large sequential write. Disk and cache throughput small_files create, write and close many tiny files. This is the one that separates filesystems, and it is what an agent actually does stat stat many files. Directory traversal cost shell_spawn run one trivial command through THIS ARM'S shell. The unit of a tool call, and the measure most likely to differ by platform fsync_honoured whether fsync costs anything here at all. A diagnostic, not a score, and the reason the two measures above do not fsync net TCP and TLS handshake to the API host. No request is sent, no key is used, nothing is billed Each measure repeats and reports both median and min. Min is the cleanest estimate of the hardware because it is the sample least disturbed by everything else. Usage: python calibrate.py --arm L-bash --shell "/bin/sh -c" python calibrate.py --arm W-gitbash --shell "sh -c" python calibrate.py --arm W-ps51 --shell "powershell -NoProfile -Command" """ import argparse import json import os import platform import shutil import socket import ssl import statistics import subprocess import sys import tempfile import time REPS = 5 def timed(fn, reps=REPS): """Run fn reps times, return median and min in milliseconds.""" out = [] for _ in range(reps): t0 = time.perf_counter() fn() out.append((time.perf_counter() - t0) * 1000.0) return {'median_ms': round(statistics.median(out), 3), 'min_ms': round(min(out), 3), 'reps': reps} def m_cpu(): def work(): acc = 0 f = 1.0 for i in range(1, 400001): acc += i * i % 97 f = f * 1.0000001 + 0.5 return acc, f return timed(work) def m_bulk_write(d, mb=64): blob = b'x' * (1024 * 1024) def work(): p = os.path.join(d, 'bulk.bin') with open(p, 'wb') as fh: for _ in range(mb): fh.write(blob) fh.flush() os.remove(p) r = timed(work, reps=3) r['mb'] = mb for k in ('median_ms', 'min_ms'): r[k.replace('_ms', '_mb_s')] = round(mb / (r[k] / 1000.0), 1) return r def m_small_files(d, n=500): def work(): sub = os.path.join(d, 'many') os.mkdir(sub) for i in range(n): p = os.path.join(sub, 'f%04d.txt' % i) with open(p, 'wb') as fh: fh.write(b'the quick brown fox\n') fh.flush() shutil.rmtree(sub) r = timed(work, reps=3) r['n'] = n for k in ('median_ms', 'min_ms'): r[k.replace('_ms', '_files_s')] = round(n / (r[k] / 1000.0), 1) return r def m_fsync_honoured(d, n=200): """Does fsync cost anything on this arm? Added after the L-bash run reported 81,000 fsync'd file creations per second, which is not a believable number for any storage device. It was not: on that machine fsync and no fsync take the same time, so the disk measures were timing the page cache and the drive's volatile cache. That matters because it is not uniform across arms. An arm whose fsync is real and an arm whose fsync is a no-op are not comparable, and putting them in the same table would have produced a large fake filesystem difference of exactly the kind this paper exists to catch. So the disk measures no longer fsync at all, on any arm, and this records whether durability was even measurable here. """ out = {} for label, sync in (('no_fsync', False), ('fsync', True)): sub = os.path.join(d, 'fs_' + label) os.mkdir(sub) t0 = time.perf_counter() for i in range(n): fh = open(os.path.join(sub, 'f%04d' % i), 'wb') fh.write(b'x' * 20) fh.flush() if sync: os.fsync(fh.fileno()) fh.close() out[label + '_ms'] = round((time.perf_counter() - t0) * 1000.0, 3) shutil.rmtree(sub) ratio = out['fsync_ms'] / out['no_fsync_ms'] if out['no_fsync_ms'] else 0.0 out['ratio'] = round(ratio, 2) out['n'] = n out['honoured'] = ratio > 1.5 out['note'] = ('fsync appears real on this arm' if ratio > 1.5 else 'fsync is a no-op here, so the disk measures are cache throughput ' 'rather than durability. A property of the arm, reported not corrected') return out def m_stat(d, n=2000): sub = os.path.join(d, 'stats') os.mkdir(sub) paths = [] for i in range(n): p = os.path.join(sub, 's%04d' % i) open(p, 'wb').close() paths.append(p) def work(): for p in paths: os.stat(p) r = timed(work) shutil.rmtree(sub) r['n'] = n for k in ('median_ms', 'min_ms'): r[k.replace('_ms', '_stats_s')] = round(n / (r[k] / 1000.0), 1) return r def m_shell_spawn(shell, n=20): """Time n trivial commands through this arm's shell. The shell is part of the arm definition, so measuring the arm's own shell is correct by design rather than a confound. This is the unit of a tool call. """ parts = shell.split() samples = [] for _ in range(n): t0 = time.perf_counter() subprocess.call(parts + ['exit 0'], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) samples.append((time.perf_counter() - t0) * 1000.0) return {'median_ms': round(statistics.median(samples), 3), 'min_ms': round(min(samples), 3), 'max_ms': round(max(samples), 3), 'n': n, 'shell': shell} def m_net(host='api.anthropic.com', port=443, n=5): """TCP connect plus TLS handshake. No request, no key, nothing billed.""" ctx = ssl.create_default_context() tcp, tls, errs = [], [], 0 for _ in range(n): try: t0 = time.perf_counter() s = socket.create_connection((host, port), timeout=10) t1 = time.perf_counter() w = ctx.wrap_socket(s, server_hostname=host) t2 = time.perf_counter() w.close() tcp.append((t1 - t0) * 1000.0) tls.append((t2 - t1) * 1000.0) except Exception: errs += 1 if not tcp: return {'error': 'unreachable', 'failures': errs} return {'tcp_median_ms': round(statistics.median(tcp), 2), 'tcp_min_ms': round(min(tcp), 2), 'tls_median_ms': round(statistics.median(tls), 2), 'n': n, 'failures': errs, 'host': host} def default_shell(): if os.name == 'nt' and not os.environ.get('SHELL'): return 'cmd /c' return '/bin/sh -c' def main(): ap = argparse.ArgumentParser() ap.add_argument('--arm', required=True) ap.add_argument('--shell', default=default_shell()) ap.add_argument('--workdir', default=None, help='where the disk measures run. Defaults to a temp dir, ' 'which on WSL2 must be set explicitly to measure the ' 'boundary rather than the VM disk') a = ap.parse_args() d = a.workdir or tempfile.mkdtemp(prefix='cal08_') made = a.workdir is None if a.workdir and not os.path.isdir(a.workdir): os.makedirs(a.workdir) res = { 'arm': a.arm, 'when': time.strftime('%Y-%m-%dT%H:%M:%S'), 'host': platform.node(), 'system': platform.system(), 'release': platform.release(), 'machine': platform.machine(), 'processor': platform.processor() or 'unknown', 'cpu_count': os.cpu_count(), 'python': sys.version.split()[0], 'workdir': d, } try: res['loadavg'] = [round(x, 2) for x in os.getloadavg()] except (OSError, AttributeError): res['loadavg'] = None print('calibrating %s on %s ...' % (a.arm, res['host']), file=sys.stderr) res['cpu'] = m_cpu() res['bulk_write'] = m_bulk_write(d) res['small_files'] = m_small_files(d) res['fsync_honoured'] = m_fsync_honoured(d) res['stat'] = m_stat(d) res['shell_spawn'] = m_shell_spawn(a.shell) res['net'] = m_net() if made: shutil.rmtree(d, ignore_errors=True) print(json.dumps(res, indent=2)) if __name__ == '__main__': main()