Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 16 additions & 1 deletion .github/workflows/validate-data.yml
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,11 @@ on:
type: string
required: false
default: "main"
base-sha:
description: "PR base SHA; when set only records changed since it are validated"
type: string
required: false
default: ""
workflow_dispatch:
inputs:
data-ref:
Expand All @@ -27,10 +32,20 @@ jobs:
repository: GetTechAPI/TechAPI
ref: ${{ inputs.data-ref }}
path: TechAPI
# Scoped runs need the diff back to the merge base, not the full history.
fetch-depth: ${{ inputs.base-sha != '' && 100 || 1 }}
- uses: actions/setup-python@v6
with:
python-version: "3.12"
- name: Validate seed JSON (§9.3, §15.3)
env:
TECHAPI_DATA_DIR: ${{ github.workspace }}/TechAPI/data
run: python -m app.validate
run: |
if [ -n "${{ inputs.base-sha }}" ]; then
git -C TechAPI fetch --no-tags --depth=100 origin "${{ inputs.base-sha }}"
git -C TechAPI merge-base "${{ inputs.base-sha }}" HEAD >/dev/null 2>&1 \
|| git -C TechAPI fetch --no-tags --unshallow origin
python -m app.validate --changed-since "${{ inputs.base-sha }}"
else
python -m app.validate
fi
73 changes: 67 additions & 6 deletions app/validate.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,8 +10,10 @@

from __future__ import annotations

import argparse
import json
import re
import subprocess
import sys
from pathlib import Path
from typing import Any
Expand Down Expand Up @@ -148,6 +150,52 @@ def _load(subdir: str, only: set[str] | None = None) -> list[tuple[str, dict[str
]


# Categories other records reference by slug (brand/soc/cpu/gpu). A scoped run
# still loads these whole (~9k files) so foreign-key checks stay exact.
FK_CATEGORIES = ("brand", "soc", "cpu", "gpu")


def changed_paths(base: str) -> set[str]:
"""Seed paths (``data/`` stripped) changed between ``base`` and HEAD."""
out = subprocess.run(
["git", "diff", "--name-only", f"{base}...HEAD", "--", "data/"],
capture_output=True, text=True, check=True, cwd=DATA_DIR.parent,
).stdout
return {
line[len("data/"):] for line in map(str.strip, out.splitlines())
if line.startswith("data/") and line.endswith(".json")
}


def _check_unique_slugs_scoped(
category: str,
records: list[tuple[str, dict[str, Any]]],
errors: list[str],
) -> None:
"""Scoped uniqueness: changed records vs each other and vs every file *name*.

Filenames equal slugs by convention, so listing names (no JSON parsing) finds
a clash with an unchanged record. A clash where the name differs from the slug
is only caught by the full run (push to develop/main, nightly).
"""
changed = {fname for fname, _ in records}
stems: dict[str, list[str]] = {}
for f in (DATA_DIR / category).rglob("*.json"):
stems.setdefault(f.stem, []).append(str(f.relative_to(DATA_DIR)))
seen: dict[str, str] = {}
for fname, rec in records:
slug = rec.get("slug")
if not isinstance(slug, str):
continue
if slug in seen:
errors.append(f"{fname}: duplicate {category} slug '{slug}' (also in {seen[slug]})")
continue
seen[slug] = fname
for other in stems.get(slug, []):
if other not in changed:
errors.append(f"{fname}: duplicate {category} slug '{slug}' (also in {other})")


def _check_required(
name: str, record: dict[str, Any], required: set[str], errors: list[str]
) -> None:
Expand Down Expand Up @@ -234,10 +282,14 @@ def _check_variant_path(
errors.append(f"{fname}: filename must match slug '{rec.get('slug')}'")


def validate() -> list[str]:
def validate(only: set[str] | None = None) -> list[str]:
"""Validate the seed data; with ``only``, just those paths (+ whole FK targets)."""
errors: list[str] = []

loaded = {category: _load(category) for category in CATEGORIES}
loaded = {
category: _load(category, None if category in FK_CATEGORIES else only)
for category in CATEGORIES
}
(brands, socs, phones, tablets, watches, pdas, gpus, cpus,
laptops, monitors, software, websites) = (loaded[category] for category in CATEGORIES)

Expand All @@ -247,7 +299,10 @@ def validate() -> list[str]:
gpu_slugs = {rec["slug"] for _, rec in gpus if "slug" in rec}

for category, records in loaded.items():
_check_unique_slugs(category, records, errors)
if only is not None and category not in FK_CATEGORIES:
_check_unique_slugs_scoped(category, records, errors)
else:
_check_unique_slugs(category, records, errors)

for fname, rec in brands:
_check_required(fname, rec, BRAND_REQUIRED, errors)
Expand Down Expand Up @@ -413,13 +468,13 @@ def validate() -> list[str]:
return errors


def run() -> int:
def run(only: set[str] | None = None) -> int:
# The ✅/❌ status glyphs must not crash on legacy consoles (e.g. cp949).
try:
sys.stdout.reconfigure(encoding="utf-8") # type: ignore[union-attr]
except Exception:
pass
errors = validate()
errors = validate(only)
if errors:
print(f"❌ Data validation failed ({len(errors)} issue(s)):")
for err in errors:
Expand All @@ -430,4 +485,10 @@ def run() -> int:


if __name__ == "__main__":
sys.exit(run())
parser = argparse.ArgumentParser(description="Validate TechAPI seed data")
parser.add_argument(
"--changed-since", metavar="BASE",
help="validate only records changed vs BASE (full run when omitted)",
)
args = parser.parse_args()
sys.exit(run(changed_paths(args.changed_since) if args.changed_since else None))
43 changes: 43 additions & 0 deletions tests/unit/test_validate_scoped.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,43 @@
"""Scoped validation: FK targets stay whole, duplicates are still caught."""

import json

from app import validate

BRAND = {"slug": "acme", "name": "Acme", "country": "US",
"categories": ["smartphone-oem"], "source_urls": ["https://example.com"]}


def _put(root, rel, rec):
f = root / rel
f.parent.mkdir(parents=True, exist_ok=True)
f.write_text(json.dumps(rec), encoding="utf-8")


def _site(tmp_path, monkeypatch):
monkeypatch.setattr(validate, "DATA_DIR", tmp_path)
_put(tmp_path, "brand/us/acme.json", BRAND)
_put(tmp_path, "website/a/one.json", {"slug": "one"})
_put(tmp_path, "website/a/two.json", {"slug": "two"})


def test_scoped_only_checks_changed_files(tmp_path, monkeypatch):
_site(tmp_path, monkeypatch)
full = validate.validate()
scoped = validate.validate({"website/a/one.json"})
# full run flags both incomplete websites, scoped run only the changed one
assert any("two.json" in e for e in full)
assert not any("two.json" in e for e in scoped)
assert any("one.json" in e for e in scoped)


def test_scoped_finds_duplicate_slug_in_unchanged_file(tmp_path, monkeypatch):
_site(tmp_path, monkeypatch)
_put(tmp_path, "website/b/one.json", {"slug": "one"})
errs = validate.validate({"website/b/one.json"})
assert any("duplicate website slug 'one'" in e for e in errs)


def test_scoped_empty_change_set_passes_fk_categories_only(tmp_path, monkeypatch):
_site(tmp_path, monkeypatch)
assert not any("website" in e for e in validate.validate(set()))
Loading