Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 3 additions & 4 deletions src/submission_checker/checker.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@

from . import layout
from .drafters import ApprovedDrafter, DrafterListError, load_approved_drafters
from .messages import Invalid
from .models import (
AccuracyResult,
CheckResult,
Expand Down Expand Up @@ -971,10 +972,8 @@ def _derive_regions(

try:
regions = compute_regions(c_max, c_min)
except ValueError as exc:
results.append(
_err("region-computation", "fail", sd_path or model_dir, detail=str(exc))
)
except Invalid as exc:
results.append(_err(exc.rule, exc.key, sd_path or model_dir, **exc.params))
return None, results
return regions, results

Expand Down
73 changes: 71 additions & 2 deletions src/submission_checker/data/messages.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -24,11 +24,46 @@ _shared:
io-error:
text: 'IO error reading {file}: {error}'
field-invalid:
text: 'Validation error in {file}: {field} — {problem}'
text: '{file}, field `{field}`: {problem}'
file-invalid:
text: '{file}: {problem}'
unworded-error:
text: '{message}'
score-not-fraction:
text: 'r{concurrency}: `{dataset}` score {value} is not a fraction in [0, 1], which is what the
reference scorer reports'
spec: §4.3
_pydantic:
# Pydantic's built-in validation errors, keyed by its error type. Templates may use
# the error's context values (see pydantic_core's list_all_errors) and {input}, the
# offending value. The test suite fails when this list and the types the file
# models can raise drift apart.
title: Field validation
spec: ''
messages:
missing: required, but missing
extra_forbidden:
text: not a recognised field
fix: Remove it, or check its spelling against the spec's template.
model_type: expected an object, got {input}
model_attributes_type: expected an object, got {input}
dict_type: expected an object, got {input}
list_type: expected a list, got {input}
too_short: needs at least {min_length} item(s), got {actual_length}
string_type: expected a string, got {input}
string_too_short: needs at least {min_length} character(s), got {input}
int_type: expected a whole number, got {input}
int_parsing: expected a whole number, got {input}
int_from_float: expected a whole number, got {input}
float_type: expected a number, got {input}
float_parsing: expected a number, got {input}
bool_type: expected true or false, got {input}
bool_parsing: expected true or false, got {input}
greater_than: must be greater than {gt}, got {input}
greater_than_equal: must be at least {ge}, got {input}
less_than_equal: must be at most {le}, got {input}
literal_error: must be {expected}, got {input}
enum: must be {expected}, got {input}
accuracy-coverage:
title: Accuracy at the mandatory points
spec: §5.3
Expand Down Expand Up @@ -75,6 +110,14 @@ accuracy-valid:
spec: §4.3
messages:
fail: accuracy_result.json is empty
scores-not-collection: accuracy_scores must be a mapping of datasets or a list of entries
entry-not-mapping: each accuracy_scores entry must be an object
entry-unnamed: each accuracy_scores entry needs a non-empty dataset_name
dataset-duplicate: dataset {name!r} appears more than once
score-missing: dataset {name!r} has no score
alias-conflict:
text: dataset {name!r} gives {native} and {canonical} different values
fix: They are two spellings of one field; give one, or make them agree.
agentic-accuracy:
title: Agentic accuracy
spec: §3.2
Expand Down Expand Up @@ -383,6 +426,14 @@ point-cap:
spec: §2, §8
messages:
fail: '{n} points exceed the {max_points}-point cap'
point-config-valid:
title: Point configuration (point.yaml)
spec: §8.3
messages:
warmup-completed-exceeds-issued: warmup requests_completed ({completed}) exceeds requests_issued
({issued})
warmup-concurrency-zero: warmup concurrency must be positive when duration or requests are
nonzero
point-count:
title: Number of points
spec: §2, §8
Expand Down Expand Up @@ -429,6 +480,10 @@ power-descriptor:
title: Power descriptor (system_power.json)
spec: §4.5.2
messages:
value-count: give exactly one of value_w, value_kw or value_pj
mlc-default-source: an mlc_default source names its Appendix D subsection ({subsections}), not
{source!r}
source-not-url: a {source_type} source must be a resolvable URL, not {source!r}
pass:
text: No system_power.json; power normalisation is not required for the {division} division
spec: §4.5
Expand Down Expand Up @@ -502,7 +557,8 @@ region-computation:
title: Region boundaries
spec: §5.5
messages:
fail: '{detail}'
c-min-range: Minimum concurrency must be between 1 and {limit} (inclusive), got {c_min}
c-max-too-low: Maximum Supported Concurrency must be greater than {limit}, got {c_max}
region-declared:
title: Region declared
spec: §8.3
Expand All @@ -523,6 +579,12 @@ required-dir:
fail:
text: 'Missing required directory: {name}/{hint}'
fix: Create {name}/ at the submission root, laid out as §8.1 describes.
result-file-valid:
title: Result summary (result_summary.json)
spec: §8.3
messages:
percentile-key: '{key!r} is not a percentile between 0 and 100'
percentile-conflict: two keys for percentile {percentile} give different values
result-summary-present:
title: Result summary present
spec: §1
Expand Down Expand Up @@ -675,6 +737,13 @@ system-description-valid:
messages:
pass: System description valid for {rel}
fail: system_desc.json is not readable as a JSON object
core-count-missing: a node type must give host_processor_core_count or host_processor_vcpu_count
division-unknown: unknown division {value!r}; must be one of standardized, serviced, rdi
availability-unknown: unknown availability {value!r}; must be one of available, preview, rdi
availability-conflict:
text: the availability fields disagree ({values})
fix: publication_status, system_availability_status and availability_status are spellings of
one field; give one, or make them agree.
system-results-dir:
title: System results directory
spec: §1
Expand Down
31 changes: 31 additions & 0 deletions src/submission_checker/messages.py
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,13 @@
Templates use :meth:`str.format` syntax restricted to plain names: ``{model}``,
``{value!r}`` and ``{ratio:.2f}`` work, ``{point.name}`` and ``{scores[0]}`` do not.
The code formats anything richer before passing it in.

Validation inside the file models words its findings the same way. A validator
raises :class:`Invalid` with a rule, a key and values rather than a hand-written
``ValueError``; the loaders render it from the catalog, located at the field that
failed. Pydantic's own errors ("Field required", "Input should be a valid
integer") are worded by the :data:`PYDANTIC` section, keyed by Pydantic's error
type, so a library upgrade cannot change what a submitter reads.
"""

from __future__ import annotations
Expand All @@ -37,6 +44,9 @@
import yaml

__all__ = [
"PYDANTIC",
"SHARED",
"Invalid",
"MessageCatalogError",
"Rendered",
"catalog",
Expand All @@ -52,6 +62,11 @@
#: field-level errors every file loader reports under its own rule.
SHARED = "_shared"

#: Section of the catalog wording Pydantic's built-in validation errors, keyed by
#: Pydantic's error type (``missing``, ``int_parsing``, ...). Its templates can use
#: the error's context values and ``input``, the offending value.
PYDANTIC = "_pydantic"

#: When True, a missing key, a missing parameter or a template error raises
#: instead of degrading to a generic message. The test suite turns this on, so a
#: message that cannot render is a failing test rather than a quiet fallback.
Expand All @@ -62,6 +77,22 @@ class MessageCatalogError(RuntimeError):
"""The catalog is malformed, or a result names a message it does not hold."""


class Invalid(ValueError):
"""A finding raised rather than returned, worded by the catalog.

File-model validators raise it where they would raise ``ValueError``: Pydantic
carries it through to the loader, which reports it under the field that failed.
Code outside a model, such as the region computation, raises it for its caller
to report. ``str()`` of it is the rendered message.
"""

def __init__(self, rule: str, key: str, /, **params: object) -> None:
self.rule = rule
self.key = key
self.params = params
super().__init__(render(rule, key, params).text)


@dataclass(frozen=True)
class Message:
"""One catalog entry: its template, optional fix, and spec section."""
Expand Down
19 changes: 13 additions & 6 deletions src/submission_checker/models/file/accuracy.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@

from pydantic import PrivateAttr, RootModel, model_validator

from ...messages import Invalid
from ..results import CheckResult, err

__all__ = ["AccuracyResult"]
Expand Down Expand Up @@ -38,20 +39,20 @@ def _read_native_report(cls, data: Any) -> Any:
if isinstance(data, dict) and "accuracy_scores" in data:
data = data["accuracy_scores"]
if not isinstance(data, (dict, list)):
raise ValueError("accuracy_scores must be a dataset mapping or list")
raise Invalid("accuracy-valid", "scores-not-collection")
if not isinstance(data, list):
return data
indexed: dict[str, dict[str, Any]] = {}
for entry in data:
if not isinstance(entry, dict):
raise ValueError("Each accuracy_scores entry must be a dictionary")
raise Invalid("accuracy-valid", "entry-not-mapping")
name = entry.get("dataset_name")
if not isinstance(name, str) or not name.strip():
raise ValueError("Each accuracy_scores entry must have a non-empty dataset_name")
raise Invalid("accuracy-valid", "entry-unnamed")
if name in indexed:
raise ValueError(f"Duplicate accuracy dataset_name: {name!r}")
raise Invalid("accuracy-valid", "dataset-duplicate", name=name)
if "score" not in entry:
raise ValueError(f"Native accuracy entry {name!r} is missing score")
raise Invalid("accuracy-valid", "score-missing", name=name)
normalized = dict(entry)
# Native scorers name these fields differently. Keep the originals
# and expose aliases used by the existing sample-count/weight gates.
Expand All @@ -61,7 +62,13 @@ def _read_native_report(cls, data: Any) -> Any:
):
if native in entry:
if canonical in entry and entry[canonical] != entry[native]:
raise ValueError(f"Conflicting {native} and {canonical} for {name!r}")
raise Invalid(
"accuracy-valid",
"alias-conflict",
name=name,
native=native,
canonical=canonical,
)
normalized[canonical] = entry[native]
indexed[name] = normalized
return indexed
Expand Down
14 changes: 7 additions & 7 deletions src/submission_checker/models/file/point_config.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@
import re
from pathlib import Path

from ...messages import fragment
from ...messages import Invalid, fragment

__all__ = ["DpShortfall", "NodesUsed", "PointConfig", "RuntimeSettings", "WarmupSpec"]

Expand Down Expand Up @@ -93,14 +93,14 @@ def is_disabled(self) -> bool:
@model_validator(mode="after")
def _check_completed_le_issued(self) -> WarmupSpec:
if self.requests_completed > self.requests_issued:
raise ValueError(
f"requests_completed ({self.requests_completed})"
f" > requests_issued ({self.requests_issued})"
raise Invalid(
"point-config-valid",
"warmup-completed-exceeds-issued",
completed=self.requests_completed,
issued=self.requests_issued,
)
if self.concurrency == 0 and not self.is_disabled:
raise ValueError(
"Warmup concurrency must be positive when duration or requests are nonzero"
)
raise Invalid("point-config-valid", "warmup-concurrency-zero")
return self


Expand Down
11 changes: 8 additions & 3 deletions src/submission_checker/models/file/point_summary.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,8 @@

from pydantic import BaseModel, ConfigDict, Field, computed_field, field_validator

from ...messages import Invalid

__all__ = ["PercentileStats", "PointSummary"]


Expand All @@ -28,12 +30,15 @@ def _normalize_percentile_keys(cls, values: dict[str, float]) -> dict[str, float
"""Treat native decimal keys and integer keys as the same percentile."""
normalized: dict[str, float] = {}
for key, value in values.items():
percentile = float(key)
try:
percentile = float(key)
except ValueError:
percentile = math.nan
if not math.isfinite(percentile) or not 0 <= percentile <= 100:
raise ValueError(f"Invalid percentile key: {key!r}")
raise Invalid("result-file-valid", "percentile-key", key=key)
canonical = str(int(percentile)) if percentile.is_integer() else str(percentile)
if canonical in normalized and normalized[canonical] != value:
raise ValueError(f"Conflicting values for percentile {canonical}")
raise Invalid("result-file-valid", "percentile-conflict", percentile=canonical)
normalized[canonical] = value
return normalized

Expand Down
13 changes: 6 additions & 7 deletions src/submission_checker/models/file/system.py
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,8 @@
model_validator,
)

from ...messages import Invalid

__all__ = [
"AcceleratorInfo",
"ConfigSummary",
Expand Down Expand Up @@ -130,10 +132,7 @@ def _lift_flat_accelerator_fields(cls, data: object) -> object:
def _require_core_or_vcpu_count(self) -> NodeType:
"""A node must disclose at least one of physical core count or vCPU count."""
if self.host_processor_core_count is None and self.host_processor_vcpu_count is None:
raise ValueError(
"node_types entry must specify host_processor_core_count or"
" host_processor_vcpu_count"
)
raise Invalid("system-description-valid", "core-count-missing")
return self


Expand Down Expand Up @@ -268,7 +267,7 @@ def _coerce_division(cls, v: object) -> object:
normalized = mapping.get(v.strip().lower())
if normalized is not None:
return normalized
raise ValueError(f"Unknown division {v!r}. Must be one of: standardized, serviced, rdi")
raise Invalid("system-description-valid", "division-unknown", value=v)
return v

@model_validator(mode="before")
Expand All @@ -291,7 +290,7 @@ def _reconcile_availability_spellings(cls, data: object) -> object:
distinct = {str(v).strip().lower() for v in present.values()}
if len(distinct) > 1:
pairs = ", ".join(f"{k}={v!r}" for k, v in sorted(present.items()))
raise ValueError(f"Conflicting availability values: {pairs}")
raise Invalid("system-description-valid", "availability-conflict", values=pairs)
if data.get("publication_status") in (None, ""):
data = {**data, "publication_status": next(iter(present.values()))}
return data
Expand All @@ -304,7 +303,7 @@ def _coerce_availability(cls, v: object) -> object:
normalized = mapping.get(v.strip().lower())
if normalized is not None:
return normalized
raise ValueError(f"Unknown availability {v!r}. Must be one of: available, preview, rdi")
raise Invalid("system-description-valid", "availability-unknown", value=v)
return v

@field_validator("input_token_average", "output_token_average", mode="before")
Expand Down
19 changes: 12 additions & 7 deletions src/submission_checker/models/file/system_power.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,7 @@

from pydantic import BaseModel, ConfigDict, Field, model_validator

from ...messages import fragment
from ...messages import Invalid, fragment
from ...power_defaults import (
PowerDefault,
accelerator_default,
Expand Down Expand Up @@ -116,16 +116,21 @@ class SourcedValue(BaseModel):
def _one_value_and_a_real_source(self) -> SourcedValue:
values = [v for v in (self.value_w, self.value_kw, self.value_pj) if v is not None]
if len(values) != 1:
raise ValueError("exactly one of value_w, value_kw or value_pj is required")
raise Invalid("power-descriptor", "value-count")
if self.source_type == "mlc_default":
if self.source not in _APPENDIX_D:
raise ValueError(
f"an mlc_default source names its Appendix D subsection"
f" ({', '.join(sorted(_APPENDIX_D))}), not {self.source!r}"
raise Invalid(
"power-descriptor",
"mlc-default-source",
subsections=", ".join(sorted(_APPENDIX_D)),
source=self.source,
)
elif not self.source.startswith(("https://", "http://")):
raise ValueError(
f"a {self.source_type} source must be a resolvable URL, not {self.source!r}"
raise Invalid(
"power-descriptor",
"source-not-url",
source_type=self.source_type,
source=self.source,
)
return self

Expand Down
Loading
Loading