Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
98 changes: 89 additions & 9 deletions integrity_check.py
Original file line number Diff line number Diff line change
Expand Up @@ -15,7 +15,10 @@
slug/filename mismatches, and physically-impossible single>multi benchmarks.
The statistical cross-source/era outliers stay advisory (a heterogeneous catalog
of server + desktop + mobile parts legitimately produces many ratio outliers), so
they are printed for review but never fail the gate.
they are printed for review but never fail the gate. The CPU cross-source ratio
check regresses out the core-count trend so it compares each part against the
ratio expected for its own core count instead of a desktop-dominated global
median (see mad_outliers).
``--hard-report`` writes a JSON list of hard anomalies for baseline comparison.
"""
from __future__ import annotations
Expand Down Expand Up @@ -91,20 +94,88 @@ def load(comp):
recs.append((p, fn[:-5], json.load(open(p, encoding="utf-8"))))
return recs

def mad_outliers(pairs, lo=0.34, hi=3.0):
"""pairs: list of (label, a, b); flag log(a/b) outliers via median±3*MAD."""
rs = [(l, math.log(a / b)) for l, a, b in pairs if a and b]
if len(rs) < 8:
# Cross-source ratio outliers: same wrong-variant detector as the era rule, but
# statistical. The original computed ONE global median±MAD over the whole CPU
# catalog for ratios like cinebench_r23_multi/geekbench_multi. That ratio is not
# scale-free across the catalog — it is confounded by core/thread count, because
# the two benchmarks scale differently with parallelism: Cinebench R23 multi
# scales near-linearly with cores while Geekbench multicore compresses at high
# core counts, so the R23/GB ratio climbs monotonically with thread count
# (measured on live data: ~1.05 at 1-4T rising to ~1.48 at 65T+, Pearson
# corr(threads, log-ratio) ≈ +0.52; the PassMark/R23 ratio falls with threads,
# corr ≈ -0.59). A single global median therefore flags entire legitimate
# core-count strata — every many-core EPYC/Threadripper/Xeon and, at the other
# end, the low-core parts — as "contamination" (90 of 739 R23/GB pairs, median
# 56 threads vs 16 overall, all with genuine scores). That is the same shape of
# bug as the flat era ceiling: a fixed reference blind to a variable that
# legitimately shifts what "normal" looks like.
#
# Fix (mirrors PR #111's cores-aware era rule, which divided the score by thread
# count): regress the confounder out. When a per-part covariate is supplied we
# fit a robust (Theil–Sen) line of log-ratio against log(covariate) and run the
# median±MAD test on the *residuals*, so a part is measured against the ratio
# expected for its own core count rather than a desktop-dominated global median.
# The systematic core-count gradient no longer flags; a part whose ratio is
# anomalous for its own class — the real wrong-variant signal — still does
# (live R23/GB flags drop 90 → 10, and the survivors are genuine per-class
# outliers). Coarse banding was rejected: it removes the between-band trend but
# shrinks the within-band MAD envelope, netting *more* false positives.
# Callers with no meaningful covariate (e.g. GPUs) omit it and get the original
# single-population behaviour unchanged.
def _theil_sen(xs: list[float], ys: list[float]) -> tuple[float, float]:
"""Robust slope/intercept via median of pairwise slopes (Theil–Sen)."""
slopes = [
(ys[j] - ys[i]) / (xs[j] - xs[i])
for i in range(len(xs)) for j in range(i + 1, len(xs))
if xs[j] != xs[i]
]
slope = statistics.median(slopes) if slopes else 0.0
intercept = statistics.median(y - slope * x for x, y in zip(xs, ys, strict=True))
return slope, intercept

def mad_outliers(pairs):
"""Flag log(a/b) outliers via median±4*MAD.

``pairs`` is a list of ``(label, a, b)`` or ``(label, a, b, covariate)``.
With a positive per-part ``covariate`` (e.g. thread count) the log-ratio is
first detrended against ``log(covariate)`` with a robust Theil–Sen fit and
the outlier test runs on the residuals, so a variable that legitimately
shifts the ratio does not turn a whole stratum into false positives. Fewer
than 8 usable points returns nothing — too few to estimate a robust
median/MAD. Omitting the covariate reproduces the original global test.
"""
rows: list[tuple[str, float, float | None]] = []
for item in pairs:
label, a, b = item[0], item[1], item[2]
cov = item[3] if len(item) > 3 else None
if a and b:
rows.append((label, math.log(a / b), cov))
if len(rows) < 8:
return []
med = statistics.median(r for _, r in rs)
mad = statistics.median(abs(r - med) for _, r in rs) or 1e-9
return [(l, round(math.exp(r), 2)) for l, r in rs if abs(r - med) > 4 * mad]
ys = [r for _, r, _ in rows]
xs = [math.log(c) for _, _, c in rows if c and c > 0]
if len(xs) == len(rows) and len({round(x, 9) for x in xs}) > 1:
slope, intercept = _theil_sen(xs, ys)
scores = [y - (slope * x + intercept) for x, y in zip(xs, ys, strict=True)]
else:
scores = ys # no usable covariate -> original global behaviour
med = statistics.median(scores)
mad = statistics.median(abs(s - med) for s in scores) or 1e-9
return [
(rows[i][0], round(math.exp(ys[i]), 2))
for i in range(len(rows)) if abs(scores[i] - med) > 4 * mad
]

def section(t): print(f"\n### {t}")

def collect(recs, fa, fb):
return [(d["name"], d[fa], d[fb]) for p, fn, d in recs if d.get(fa) and d.get(fb)]

def collect_cpu(recs, fa, fb):
"""Like ``collect`` but tags each pair with its thread-count covariate."""
return [(d["name"], d[fa], d[fb], d.get("threads") or d.get("cores") or 1)
for p, fn, d in recs if d.get(fa) and d.get(fb)]

def main() -> None:
records = {category: load(category) for category in CATEGORIES}
cpus = records["cpu"]; gpus = records["gpu"]
Expand Down Expand Up @@ -157,16 +228,25 @@ def main() -> None:
print(msg)

# --- 5. cross-source correlation outliers (KEY contamination detector) ---
# Thread-count-aware (see mad_outliers): the ratio between two CPU benchmarks
# is confounded by core count, so the core-count trend is regressed out and
# each part is judged against the ratio expected for its own core count
# rather than a desktop-dominated global median.
section("CPU cross-source ratio outliers (possible wrong-variant)")
for fa, fb in [("passmark_cpu_mark","cinebench_r23_multi"),
("passmark_cpu_mark","geekbench_multi"),
("cinebench_r23_multi","geekbench_multi"),
("cinebench_2024_multi","cinebench_r23_multi")]:
out = mad_outliers(collect(cpus, fa, fb))
out = mad_outliers(collect_cpu(cpus, fa, fb))
for label, ratio in out:
print(f" [{fa}/{fb}] {label!r}: ratio={ratio}")

# --- 6. GPU cross-source + sanity ---
# Left as a single population on purpose: unlike the CPU thread-count
# confounder, the GPU ratios mix a theoretical spec (fp32_tflops) with
# empirical benchmarks across gaming vs. compute cards and many hardware
# eras, so there is no single clean stratifying variable. These stay
# advisory-only and are surfaced for human review rather than gated.
section("GPU cross-source ratio outliers + sanity")
for fa, fb in [("passmark_g3d_mark","timespy_score"),
("timespy_score","blender_score"),
Expand Down
123 changes: 123 additions & 0 deletions tests/unit/test_integrity_cross_source.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,123 @@
"""Unit tests for the core-count-aware cross-source ratio outlier detector in
``integrity_check``.

Cross-source ratios (e.g. cinebench_r23_multi / geekbench_multi) are the key
wrong-variant contamination signal, but the raw ratio is confounded by core
count: the two benchmarks scale differently with parallelism, so the ratio drifts
monotonically with thread count. The original single global median±MAD therefore
flagged entire legitimate core-count strata (every many-core EPYC/Threadripper/
Xeon) as outliers — the same shape of bug as the flat era ceiling that PR #111
fixed. The fix regresses the core-count trend out and runs the outlier test on
the residuals. These tests pin that behaviour: a whole core-count band that only
follows the natural trend must not flag, while a part that is anomalous *for its
own core count* still must, and callers without a covariate keep the original
global behaviour.
"""

from __future__ import annotations

import math

import integrity_check as ic


def _clean_pair_at(
threads: int, trend_slope: float = 0.11, base: float = 0.05
) -> tuple[float, float]:
"""Build an (a, b) pair whose log-ratio sits exactly on the core-count trend.

log(a/b) = base + trend_slope * log(threads); b fixed at 1000.
"""
log_ratio = base + trend_slope * math.log(threads)
b = 1000.0
a = b * math.exp(log_ratio)
return a, b


# A realistic thread-count ladder spanning desktop -> HEDT -> server, each part's
# ratio lying on the same gentle upward core-count trend (the confounder). The
# population is desktop-dominated (like the real catalog: median ~16 threads), so
# a global median is pulled toward the low-core parts and the legitimate high-core
# trend-followers look like outliers to it.
TREND_THREADS = (
[4] * 6 + [8] * 8 + [12] * 10 + [16] * 12 + [24] * 6 + [32] * 4
+ [48, 64, 96, 128, 192, 256]
)


def test_pairs_on_the_core_count_trend_do_not_flag() -> None:
# Every pair follows the same log-ratio-vs-log-threads line, so after the
# trend is regressed out the residuals are ~0 and nothing is an outlier --
# even though the raw ratios span a wide range (the old global test flagged
# the extremes).
pairs = [
(f"part-{t}-{i}", *_clean_pair_at(t), t)
for i, t in enumerate(TREND_THREADS)
]
assert ic.mad_outliers(pairs) == []


def test_raw_global_test_would_have_flagged_the_trend_extremes() -> None:
# Guard/contrast: the SAME data, fed WITHOUT the covariate, reproduces the
# old behaviour and flags the high-core extremes. This documents that the fix
# -- not a change in the data -- is what removes the false positives.
pairs_no_cov = [(f"part-{t}-{i}", *_clean_pair_at(t)) for i, t in enumerate(TREND_THREADS)]
flagged = [label for label, _ in ic.mad_outliers(pairs_no_cov)]
# The many-core trend-followers (which are perfectly legitimate) are what the
# covariate-free test wrongly flags.
assert flagged, "global (covariate-free) test should still flag trend extremes"
assert any("part-256" in f or "part-192" in f for f in flagged)


def test_part_anomalous_for_its_own_core_count_still_flags() -> None:
# A part whose ratio is far off the trend line *for its thread count* is a
# genuine wrong-variant candidate and must survive the detrending.
pairs = [
(f"part-{t}-{i}", *_clean_pair_at(t), t)
for i, t in enumerate(TREND_THREADS)
]
# Inject a 32-thread part with a wildly wrong ratio (e.g. a swapped variant):
bogus_a, bogus_b = _clean_pair_at(32)
pairs.append(("Swapped-variant 32T", bogus_a * 4.0, bogus_b, 32))
flagged = [label for label, _ in ic.mad_outliers(pairs)]
assert "Swapped-variant 32T" in flagged


def test_missing_covariate_recovers_global_behaviour() -> None:
# Three-tuples (no covariate) must behave exactly like the original detector:
# a uniform cluster with one gross outlier flags only the outlier.
pairs = [(f"p{i}", 1000.0 + i, 1000.0) for i in range(20)]
pairs.append(("gross", 100000.0, 1000.0))
flagged = [label for label, _ in ic.mad_outliers(pairs)]
assert flagged == ["gross"]


def test_fewer_than_eight_points_never_flags() -> None:
pairs = [(f"p{i}", 1000.0 * (i + 1), 1000.0, 8) for i in range(7)]
assert ic.mad_outliers(pairs) == []


def test_zero_or_missing_values_are_skipped() -> None:
# a or b of 0/None must not raise and must be dropped from the population.
pairs = [(f"p{i}", 1000.0, 1000.0, 8) for i in range(10)]
pairs += [("zero-a", 0, 1000.0, 8), ("none-b", 1000.0, None, 8)]
# No real outlier among the valid points -> empty, and no exception.
assert ic.mad_outliers(pairs) == []


def test_returned_ratio_is_the_raw_ratio_not_the_residual() -> None:
# The reported number must remain the human-readable a/b ratio so reviewers
# can eyeball it, even though the test runs on residuals.
pairs = [(f"p{i}", *_clean_pair_at(t), t) for i, t in enumerate(TREND_THREADS)]
a, b = _clean_pair_at(32)
pairs.append(("Swapped-variant 32T", a * 4.0, b, 32))
result = dict(ic.mad_outliers(pairs))
assert math.isclose(result["Swapped-variant 32T"], round((a * 4.0) / b, 2), rel_tol=1e-6)


def test_theil_sen_recovers_a_known_slope() -> None:
xs = [math.log(t) for t in (1, 2, 4, 8, 16, 32, 64)]
ys = [0.5 + 0.3 * x for x in xs] # perfect line, slope 0.3
slope, intercept = ic._theil_sen(xs, ys)
assert math.isclose(slope, 0.3, rel_tol=1e-9)
assert math.isclose(intercept, 0.5, rel_tol=1e-9)
Loading