Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 5 additions & 11 deletions loopx/capabilities/benchmark_toolkit/concurrency_envelope.py
Original file line number Diff line number Diff line change
@@ -1,12 +1,15 @@
from __future__ import annotations

import json
import re
from collections.abc import Mapping
from datetime import UTC, datetime, timedelta
from pathlib import Path
from typing import Any

from .experiment_identity import (
ARM_ROLES,
experiment_token_text as _token,
)
from ...domain_state import default_domain_state_file_path
from ...file_lock import exclusive_file_lock
from ...registry import atomic_write_json
Expand All @@ -17,8 +20,6 @@
)
BENCHMARK_CONCURRENCY_ENVELOPE_FILENAME = "concurrency-envelope.json"

_TOKEN_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.:@+-]{0,127}$")
_ARM_ROLES = {"baseline", "control", "treatment", "explore"}
_RESOURCE_HEADROOM_KINDS = {
"file_descriptors",
"memory",
Expand Down Expand Up @@ -69,13 +70,6 @@ def _timestamp(value: Any, *, field: str) -> str:
return parsed.isoformat().replace("+00:00", "Z")


def _token(value: Any, *, field: str) -> str:
text = str(value or "").strip()
if not _TOKEN_RE.fullmatch(text):
raise ValueError(f"{field} must be a compact public-safe token")
return text


def _bounded_int(value: Any, *, field: str, minimum: int = 0) -> int:
if isinstance(value, bool) or not isinstance(value, int) or value < minimum:
qualifier = "positive" if minimum == 1 else "non-negative"
Expand Down Expand Up @@ -275,7 +269,7 @@ def _normalize_active_run(value: Mapping[str, Any]) -> dict[str, str]:
raise TypeError("active run must be an object")
_reject_unknown_fields(value, allowed=_ACTIVE_RUN_FIELDS, field="active_run")
arm_role = _token(value.get("arm_role"), field="arm_role")
if arm_role not in _ARM_ROLES:
if arm_role not in ARM_ROLES:
raise ValueError("arm_role is unsupported")
return {
"run_id": _token(value.get("run_id"), field="run_id"),
Expand Down
20 changes: 7 additions & 13 deletions loopx/capabilities/benchmark_toolkit/experiment_board.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,13 +2,16 @@

import json
import math
import re
from collections.abc import Iterable, Mapping
from datetime import datetime
from pathlib import Path
from typing import Any

from ...domain_state import default_domain_state_file_path, upsert_domain_state_jsonl
from .experiment_identity import (
ARM_ROLES,
experiment_token_text as _token,
)
from .factorial_contrast import (
build_benchmark_factorial_contrasts,
build_benchmark_metric_delta,
Expand All @@ -18,8 +21,6 @@
BENCHMARK_EXPERIMENT_BOARD_SCHEMA_VERSION = "benchmark_experiment_board_v0"
BENCHMARK_EXPERIMENT_BOARD_LEDGER_FILENAME = "experiment-board.jsonl"

_TOKEN_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.:@+-]{0,127}$")
_ARM_ROLES = {"baseline", "control", "treatment", "explore"}
_RUN_STATUSES = {"planned", "running", "completed", "runner_invalid", "cancelled"}
_RUN_STATUS_TRANSITIONS = {
"planned": _RUN_STATUSES,
Expand Down Expand Up @@ -78,13 +79,6 @@ def _reject_unknown_fields(
raise ValueError(f"{field} contains unsupported fields: {', '.join(unknown)}")


def _token(value: Any, *, field: str) -> str:
text = str(value or "").strip()
if not _TOKEN_RE.fullmatch(text):
raise ValueError(f"{field} must be a compact public-safe token")
return text


def _optional_token(value: Any, *, field: str) -> str | None:
if value in (None, ""):
return None
Expand Down Expand Up @@ -267,7 +261,7 @@ def normalize_benchmark_experiment_board_row(
raise ValueError("benchmark experiment board row schema mismatch")

arm_role = _token(payload.get("arm_role"), field="arm_role")
if arm_role not in _ARM_ROLES:
if arm_role not in ARM_ROLES:
raise ValueError("arm_role is unsupported")
status = _token(payload.get("status"), field="status")
if status not in _RUN_STATUSES:
Expand Down Expand Up @@ -744,12 +738,12 @@ def build_benchmark_experiment_board(
]
role_counts = {
role: sum(1 for row in normalized if row["arm_role"] == role)
for role in sorted(_ARM_ROLES)
for role in sorted(ARM_ROLES)
}
comparison_arm_role_counts = _comparison_lane_counts(
comparisons,
field="candidate_arm_role",
values=_ARM_ROLES - {"baseline"},
values=ARM_ROLES - {"baseline"},
)
comparison_claim_scope_counts = _comparison_lane_counts(
comparisons,
Expand Down
33 changes: 33 additions & 0 deletions loopx/capabilities/benchmark_toolkit/experiment_identity.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,33 @@
"""One owner for the two vocabularies a benchmark experiment identity uses.

An arm role and a public-safe token are validated at the package boundary, at
the study projection, at the concurrency envelope and at the CLI that admits a
case slot. Each of those places restated the answer, so widening a token in one
file left the others rejecting the same value, and the CLI's ``--arm-role``
choices could disagree with the envelope that admits the run.

``ARM_ROLES`` is the set every checker compares against. ``ARM_ROLE_CHOICES``
is the same four names in the order ``argparse`` shows them to a human; the
order is part of the help text, so it is stated rather than derived.

``experiment_token_text`` is the reject path that goes with the token shape.
Five modules carried a byte-identical private ``_token`` helper, so the shape and
its rejection message travelled together in five copies.
"""

from __future__ import annotations

import re
from typing import Any

ARM_ROLES = frozenset({"baseline", "control", "treatment", "explore"})
ARM_ROLE_CHOICES = ("baseline", "control", "treatment", "explore")

EXPERIMENT_TOKEN_PATTERN = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.:@+-]{0,127}$")


def experiment_token_text(value: Any, *, field: str) -> str:
text = str(value or "").strip()
if not EXPERIMENT_TOKEN_PATTERN.fullmatch(text):
raise ValueError(f"{field} must be a compact public-safe token")
return text
10 changes: 1 addition & 9 deletions loopx/capabilities/benchmark_toolkit/factorial_contrast.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,10 +2,10 @@

from __future__ import annotations

import re
from collections.abc import Iterable, Mapping
from typing import Any

from .experiment_identity import experiment_token_text as _token
from .four_arm_contract import (
BENCHMARK_FOUR_ARM_CONTRACT_SCHEMA_VERSION,
BENCHMARK_FOUR_ARM_QUALIFICATION_SCOPE,
Expand All @@ -14,17 +14,9 @@

BENCHMARK_FACTORIAL_CONTRAST_SCHEMA_VERSION = "benchmark_factorial_contrast_v0"

_TOKEN_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.:@+-]{0,127}$")
_FACTOR_CELLS = {(False, False), (True, False), (False, True), (True, True)}


def _token(value: Any, *, field: str) -> str:
text = str(value or "").strip()
if not _TOKEN_RE.fullmatch(text):
raise ValueError(f"{field} must be a compact public-safe token")
return text


def _optional_token(value: Any, *, field: str) -> str | None:
if value in (None, ""):
return None
Expand Down
16 changes: 5 additions & 11 deletions loopx/capabilities/benchmark_toolkit/study_projection.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,6 @@
import json
import math
import os
import re
import statistics
import tempfile
from collections.abc import Iterable, Mapping
Expand All @@ -24,6 +23,10 @@
normalize_benchmark_experiment_board_row,
preview_benchmark_experiment_board_upsert,
)
from .experiment_identity import (
ARM_ROLES,
experiment_token_text as _token,
)
from .four_arm_contract import BENCHMARK_FOUR_ARM_CONTRACT_SCHEMA_VERSION
from .runtime_observation import (
BENCHMARK_RUNTIME_OBSERVATION_SCHEMA_VERSION,
Expand All @@ -41,8 +44,6 @@
)
BENCHMARK_STUDY_DASHBOARD_SCHEMA_VERSION = "benchmark_study_dashboard_v0"

_TOKEN_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.:@+-]{0,127}$")
_ARM_ROLES = {"baseline", "control", "treatment", "explore"}
_METRIC_ROLES = {"primary", "guardrail", "supporting"}
_RECORD_KINDS = {
"study_manifest",
Expand All @@ -63,13 +64,6 @@ def _reject_unknown_fields(
raise ValueError(f"{field} contains unsupported fields: {', '.join(unknown)}")


def _token(value: Any, *, field: str) -> str:
text = str(value or "").strip()
if not _TOKEN_RE.fullmatch(text):
raise ValueError(f"{field} must be a compact public-safe token")
return text


def _optional_token(value: Any, *, field: str) -> str | None:
if value in (None, ""):
return None
Expand Down Expand Up @@ -230,7 +224,7 @@ def normalize_benchmark_study_manifest(
)
arm_id = _token(raw_arm.get("arm_id"), field="arm.arm_id")
arm_role = _token(raw_arm.get("arm_role"), field="arm.arm_role")
if arm_role not in _ARM_ROLES:
if arm_role not in ARM_ROLES:
raise ValueError("arm.arm_role is unsupported")
raw_assignments = raw_arm.get("factor_assignments")
if not isinstance(raw_assignments, Mapping):
Expand Down
10 changes: 1 addition & 9 deletions loopx/capabilities/benchmark_toolkit/traex_evidence.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,25 +4,17 @@

import hashlib
import json
import re
from collections.abc import Iterable, Mapping
from pathlib import Path
from typing import Any

from .experiment_identity import experiment_token_text as _token
from ...registry import atomic_write_json

TRAE_BENCHMARK_EVIDENCE_SCHEMA_VERSION = "benchmark_trae_evidence_capture_v0"
BENCHMARK_MODEL_ROUTE_RECEIPT_SCHEMA_VERSION = "benchmark_model_route_receipt_v0"
ATIF_SCHEMA_VERSION = "ATIF-v1.7"

_PUBLIC_TOKEN = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.:@+-]{0,127}$")


def _token(value: Any, *, field: str) -> str:
text = str(value or "").strip()
if not _PUBLIC_TOKEN.fullmatch(text):
raise ValueError(f"{field} must be a compact public-safe token")
return text


def _canonical_json(value: Any) -> str:
Expand Down
3 changes: 2 additions & 1 deletion loopx/cli_commands/benchmark_concurrency.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
from pathlib import Path
from typing import Any

from ..capabilities.benchmark_toolkit.experiment_identity import ARM_ROLE_CHOICES
from ..capabilities.benchmark_toolkit import (
admit_benchmark_case,
build_benchmark_adaptive_concurrency_policy,
Expand Down Expand Up @@ -102,7 +103,7 @@ def register_benchmark_concurrency_commands(
admit_parser.add_argument(
"--arm-role",
required=True,
choices=["baseline", "control", "treatment", "explore"],
choices=ARM_ROLE_CHOICES,
)
admit_parser.add_argument(
"--resource-headroom-json",
Expand Down
Loading
Loading