Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 6 additions & 1 deletion app/ingest/enrich.py
Original file line number Diff line number Diff line change
Expand Up @@ -149,7 +149,12 @@ def enrich(
break
processed += 1
name = rec.get("name", "")
out = resolver(client, name, overrides.get(name))
if resolver is topcpu.resolve or resolver is topcpu.resolve_gpu:
out = resolver(
client, name, overrides.get(name), architecture=rec.get("architecture")
)
else:
out = resolver(client, name, overrides.get(name))
if sleep:
time.sleep(sleep)
if out is None:
Expand Down
53 changes: 44 additions & 9 deletions app/ingest/sources/topcpu.py
Original file line number Diff line number Diff line change
Expand Up @@ -60,6 +60,32 @@
_DIGITS = re.compile(r"[^0-9]")
_NUM = re.compile(r"[\d,]+\.?\d*")

# Architecture families, including versioned labels such as Tesla 2.0 and
# TeraScale 3. These cannot provide the DX12 driver required by Time Spy (FL11_0).
# AMD's Windows driver support matrix separates HD5000/6000 (DX11) from GCN:
# https://www.amd.com/en/resources/support-articles/faqs/GPU-615.html
# NVIDIA's DX12 support starts at Fermi, INCLUDING Fermi and early Kepler:
# https://nvidia.custhelp.com/app/answers/detail/a_id/3711
# https://support.benchmarks.ul.com/support/solutions/articles/44002136075
_PRE_DX12_ARCHITECTURE = re.compile(
r"\b(?:TeraScale|Tesla|Curie|Rankine|Kelvin|Celsius|Ultra[- ]Threaded SE|"
r"R100|R200|R300|R400)\b",
re.IGNORECASE,
)
_DX12_FIELDS = frozenset({"timespy_score", "timespy_extreme_score", "speedway_score"})


def is_pre_dx12(architecture: str | None) -> bool:
"""Recognize known unsupported architectures; unknown labels are not guessed.

Use the record's microarchitecture, never its release year or product brand
(e.g. a Tesla-branded card may have a newer, DX12-capable architecture).
This is a pre-DX12 exclusion, not a complete Speed Way/Ultimate eligibility
check. OctaneBench uses CUDA, and FP32 is a spec, so neither is DX12-filtered.
"""
return bool(architecture and _PRE_DX12_ARCHITECTURE.search(architecture))


# Cached normalized score maps, keyed by (url, normalizer name).
_caches: dict[str, dict[str, float]] = {}

Expand Down Expand Up @@ -110,9 +136,15 @@ def reset_cache() -> None:


def resolve(
client: httpx.Client, name: str, id_override: str | None = None
client: httpx.Client,
name: str,
id_override: str | None = None,
*,
architecture: str | None = None,
) -> tuple[dict[str, int], str] | None:
"""GPU Time Spy resolver: ``({"timespy_score": score}, url)`` or None."""
if is_pre_dx12(architecture):
return None
hit = _load_map(client, TIMESPY_URL, normalize_gpu).get(normalize_gpu(name))
if hit is None:
return None
Expand Down Expand Up @@ -140,23 +172,26 @@ def resolve_cpu(


def resolve_gpu(
client: httpx.Client, name: str, id_override: str | None = None
client: httpx.Client,
name: str,
id_override: str | None = None,
*,
architecture: str | None = None,
) -> tuple[dict[str, float], str] | None:
"""GPU breadth resolver: Time Spy Extreme / Speed Way / OctaneBench / FP32.

WARNING: topcpu publishes unreliable *estimated* 3DMark/Octane scores for
pre-DX12 cards that can't actually run them (e.g. Radeon HD 5670 "Time Spy"
3897 — physically impossible; contradicts its PassMark G3D). The same applies
to ``resolve`` (Time Spy). When enriching, GUARD on DX12 capability
(release year >= 2011 / GCN/Kepler+) before writing timespy*/speedway/
octanebench — only fp32_tflops (a spec) is era-safe. See
TechAPI/.claude/benchmark_fill_progress.md pt.7.
topcpu publishes estimated DX12 scores for cards that cannot run those
benchmarks. Pass the record's architecture to suppress Time Spy Extreme
and Speed Way for known pre-DX12 families, regardless of release year.
Non-DX12 dimensions retain their existing behavior.
"""
key = normalize_gpu(name)
if not key:
return None
scores: dict[str, float] = {}
for url, field, as_float in _GPU_FAMILIES:
if field in _DX12_FIELDS and is_pre_dx12(architecture):
continue
v = _load_map(client, url, normalize_gpu, as_float=as_float).get(key)
if v is not None:
scores[field] = v
Expand Down
98 changes: 98 additions & 0 deletions tests/unit/test_gpu_sources.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,12 @@
import io
import json
import zipfile
from pathlib import Path

import httpx
import pytest

from app.ingest import enrich as enrich_mod
from app.ingest.sources import blender, topcpu, videocardbenchmark

# --- shared GPU name normalization (variant safety) ---------------------------
Expand Down Expand Up @@ -238,3 +243,96 @@ def test_topcpu_gpu_breadth_int_and_float() -> None:
}
assert "gpu-r" in url
assert topcpu.resolve_gpu(_RoutingClient(routes), "Radeon RX 9999") is None


@pytest.mark.parametrize("architecture", [
"TeraScale", "TeraScale 2", "terascale 3", "Tesla", "Tesla 2.0",
"Tesla 2.0 | Tesla", "Curie", "Rankine", "Kelvin", "Celsius",
"Ultra-Threaded SE", "R100", "R200", "R300", "R400",
])
def test_topcpu_rejects_pre_dx12_time_spy(architecture: str) -> None:
topcpu.reset_cache()
client = _HtmlClient(_gpu_row("Legacy GPU", "3897"))
assert topcpu.resolve(client, "Legacy GPU", architecture=architecture) is None


@pytest.mark.parametrize("architecture", [
"Fermi", "Fermi 2.0", "Kepler", "Kepler 2.0", "GCN 1.0", "RDNA 2", "Ampere",
])
def test_topcpu_keeps_dx12_time_spy(architecture: str) -> None:
topcpu.reset_cache()
client = _HtmlClient(_gpu_row("Supported GPU", "1000"))
assert topcpu.resolve(client, "Supported GPU", architecture=architecture) == (
{"timespy_score": 1000}, topcpu.TIMESPY_URL,
)


def test_topcpu_unknown_architecture_is_not_guessed() -> None:
assert not topcpu.is_pre_dx12(None)
assert not topcpu.is_pre_dx12("")
assert not topcpu.is_pre_dx12("Unclassified")


@pytest.mark.parametrize("resolver,primary_field", [
(topcpu.resolve, "timespy_score"), (topcpu.resolve_gpu, None),
])
def test_topcpu_enrichment_uses_architecture_not_year(
tmp_path: Path, monkeypatch, resolver, primary_field,
) -> None:
topcpu.reset_cache()
# GCN and TeraScale coexisted in 2012; Fermi can run Time Spy despite 2010.
records = [
("Radeon HD 7350", "TeraScale 2", "2012-01-01", False),
("Radeon HD 7970", "GCN 1.0", "2012-01-01", True),
("GeForce GTX 480", "Fermi", "2010-03-26", True),
("Tesla K20", "Kepler", "2012-11-12", True),
("GeForce GT 330", "Tesla 2.0", "2024-01-01", False),
]
gpu_dir = tmp_path / "gpu"
gpu_dir.mkdir()
for i, (name, architecture, release_date, _) in enumerate(records):
(gpu_dir / f"{i}.json").write_text(json.dumps({
"name": name, "architecture": architecture, "release_date": release_date,
"timespy_score": None, "timespy_extreme_score": None, "speedway_score": None,
"octanebench_score": 55, "fp32_tflops": None, "source_urls": [],
}), encoding="utf-8")

def handle(request: httpx.Request) -> httpx.Response:
value = "0.25" if "fp32-float" in str(request.url) else "1000"
return httpx.Response(200, text="".join(_gpu_row(r[0], value) for r in records))

monkeypatch.setattr(enrich_mod, "make_client", lambda: httpx.Client(
transport=httpx.MockTransport(handle),
))
result = enrich_mod.enrich(
data_root=tmp_path, component="gpu", resolver=resolver,
primary_field=primary_field, sleep=0,
)
for i, (_, _, _, supported) in enumerate(records):
written = json.loads((gpu_dir / f"{i}.json").read_text(encoding="utf-8"))
if resolver is topcpu.resolve:
assert written["timespy_score"] == (1000 if supported else None)
else:
assert written["timespy_extreme_score"] == (1000 if supported else None)
assert written["speedway_score"] == (1000 if supported else None)
assert written["fp32_tflops"] == 0.25
# Fill-only-nulls must still preserve existing data.
assert written["octanebench_score"] == 55
if resolver is topcpu.resolve:
assert result.unresolved == ["Radeon HD 7350", "GeForce GT 330"]
else:
assert result.unresolved == []


def test_topcpu_pre_dx12_breadth_keeps_non_dx12_dimensions() -> None:
topcpu.reset_cache()
name = "GeForce GT 330"
routes = {
"3dmark-time-spy-extreme": _gpu_row(name, "2000"),
"3dmark-speed-way": _gpu_row(name, "3000"),
"octanebench": _gpu_row(name, "20"),
"fp32-float": _gpu_row(name, "0.25"),
}
out = topcpu.resolve_gpu(_RoutingClient(routes), name, architecture="Tesla 2.0")
assert out is not None
assert out[0] == {"octanebench_score": 20, "fp32_tflops": 0.25}
Loading