Download example.py from zeechimp/hv-falsification-os: direct link, hf CLI and curl.
- Browser
- Download file 5.38 kB
-
https://huggingface.co/zeechimp/hv-falsification-os/resolve/main/example.py
- Command line
-
hf download hf://zeechimp/hv-falsification-os/example.py
-
curl -L -o example.py https://huggingface.co/zeechimp/hv-falsification-os/resolve/main/example.py
5.38 kB
| #!/usr/bin/env python3 | |
| """ | |
| example.py — Usage demonstrations for hv_falsification_os. | |
| NumPy only. | |
| """ | |
| import numpy as np | |
| from hv_falsification_os import ( | |
| FalsificationOS, | |
| make_synthetic_workload, | |
| make_two_workloads, | |
| make_amortized_workload, | |
| ) | |
| def demo_cv(): | |
| print("=" * 72) | |
| print("Demo 1 — measure the CV of a metric") | |
| print("=" * 72) | |
| print() | |
| os_ = FalsificationOS() | |
| fn = make_synthetic_workload(base=1.0, cv=0.05, seed_offset=0) | |
| r = os_.measure_cv(fn, n_trials=500, seed=0) | |
| print(f" mean: {r['mean']:.4f}") | |
| print(f" std: {r['std']:.4f}") | |
| print(f" CV: {r['cv']:.4f}") | |
| print(f" 95% CI: +/- {r['ci95_halfwidth']:.4f}") | |
| print() | |
| print(" Interpretation: a delta smaller than " | |
| f"{2*r['cv']:.2%} is indistinguishable from noise.") | |
| print() | |
| def demo_same_code(): | |
| print("=" * 72) | |
| print("Demo 2 — is this delta real?") | |
| print("=" * 72) | |
| print() | |
| os_ = FalsificationOS() | |
| for delta in [0.01, 0.05, 0.10, 0.15, 0.50]: | |
| fn_a, fn_b = make_two_workloads(delta, cv=0.05, seed_offset=0) | |
| r = os_.same_code_test(fn_a, fn_b, n_trials=500, seed=0) | |
| print(f" delta={delta:>5.2f} ratio={r['ratio_to_cv']:>5.2f} " | |
| f"verdict={r['verdict']:<10s} " | |
| f"(design CV = 5%)") | |
| print() | |
| def demo_amortization(): | |
| print("=" * 72) | |
| print("Demo 3 — amortization sweep") | |
| print("=" * 72) | |
| print() | |
| os_ = FalsificationOS() | |
| fn = make_amortized_workload(peak_param=1000.0, peak_speedup=0.78) | |
| params = [1, 10, 100, 500, 1000, 2000, 5000, 10000, 50000] | |
| r = os_.amortization_sweep(fn, params, n_trials=20, seed=0) | |
| print(f" {'param':>8s} {'value':>10s} {'bar'}") | |
| print(" " + "-" * 50) | |
| for p, m in zip(r['param_values'], r['means']): | |
| bar = "#" * int(m * 40) | |
| print(f" {p:>8.0f} {m:>10.4f} {bar}") | |
| print() | |
| print(f" Peak: param = {r['peak_param']:.0f}, " | |
| f"value = {r['peak_value']:.3f}") | |
| if r['peak_value'] < 1.0: | |
| print(" Verdict: peak loses to baseline.") | |
| print() | |
| def demo_precondition(): | |
| print("=" * 72) | |
| print("Demo 4 — precondition testing") | |
| print("=" * 72) | |
| print() | |
| os_ = FalsificationOS() | |
| def clustered(seed): | |
| rng = np.random.default_rng(seed) | |
| K, d, N = 5, 64, 500 | |
| centers = rng.standard_normal((K, d)) | |
| centers /= np.linalg.norm(centers, axis=1, keepdims=True) | |
| a = rng.integers(0, K, size=N) | |
| items = centers[a] + 0.1 * rng.standard_normal((N, d)) | |
| items /= np.linalg.norm(items, axis=1, keepdims=True) | |
| i = rng.integers(0, N, size=2000) | |
| j = rng.integers(0, N, size=2000) | |
| return float((np.einsum('ij,ij->i', items[i], items[j]) > 0.5).mean()) > 0.05 | |
| def uniform(seed): | |
| rng = np.random.default_rng(seed) | |
| N, d = 500, 64 | |
| items = rng.standard_normal((N, d)) | |
| items /= np.linalg.norm(items, axis=1, keepdims=True) | |
| i = rng.integers(0, N, size=2000) | |
| j = rng.integers(0, N, size=2000) | |
| return float((np.einsum('ij,ij->i', items[i], items[j]) > 0.5).mean()) > 0.05 | |
| for name, fn in [('clustered', clustered), ('uniform', uniform)]: | |
| r = os_.precondition_test(fn, n_trials=30, seed=0) | |
| print(f" {name:<10s} pass_rate = {r['pass_rate']:.1%} " | |
| f"holds = {r['holds']}") | |
| print() | |
| print(" The same test passes on clustered data and fails on uniform.") | |
| print(" That is the precondition check.") | |
| print() | |
| def demo_four_question(): | |
| print("=" * 72) | |
| print("Demo 5 — four-question probe") | |
| print("=" * 72) | |
| print() | |
| os_ = FalsificationOS() | |
| rng = np.random.default_rng(42) | |
| N, d = 1000, 64 | |
| print(f" Case A: low-rank task in random H") | |
| H = rng.standard_normal((N, d)) | |
| Y = H[:, :3].sum(axis=1) + 0.5 * rng.standard_normal(N) | |
| r = os_.four_question_probe(H, Y) | |
| print(f" effective_rank_H = {r['effective_rank_H']:.2f}") | |
| print(f" cross_cov_rank = {r['cross_cov_rank']:.2f}") | |
| print(f" fresh_probe_r2 = {r['fresh_probe_r2']:.4f}") | |
| print() | |
| print(f" Case B: task unrelated to H") | |
| Y = rng.standard_normal(N) | |
| r = os_.four_question_probe(H, Y) | |
| print(f" effective_rank_H = {r['effective_rank_H']:.2f}") | |
| print(f" cross_cov_rank = {r['cross_cov_rank']:.2f}") | |
| print(f" fresh_probe_r2 = {r['fresh_probe_r2']:.4f}") | |
| print() | |
| print(" In case A the task signal is recoverable (R² > 0.9).") | |
| print(" In case B it is not, even though H is high-rank.") | |
| print() | |
| def demo_reversal(): | |
| print("=" * 72) | |
| print("Demo 6 — reversal detection") | |
| print("=" * 72) | |
| print() | |
| os_ = FalsificationOS() | |
| cases = [ | |
| ('clear winner', (1.00, 1.10), (1.02, 1.11)), | |
| ('flipping winner', (1.00, 0.98), (0.99, 1.01)), | |
| ('tiny gap', (1.000, 1.001), (1.002, 1.000)), | |
| ] | |
| for name, r1, r2 in cases: | |
| r = os_.reversal_detection(r1, r2) | |
| status = "REVERSED" if r['reversal'] else "stable" | |
| print(f" {name:<18s} run1={r['winner_1']} run2={r['winner_2']} " | |
| f"→ {status}") | |
| print() | |
| def main(): | |
| demo_cv() | |
| demo_same_code() | |
| demo_amortization() | |
| demo_precondition() | |
| demo_four_question() | |
| demo_reversal() | |
| if __name__ == "__main__": | |
| main() |