220 lines
8.3 KiB
Python
220 lines
8.3 KiB
Python
"""Seal upstream outputs into the runner/catalog evidence contract."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import json
|
|
import shutil
|
|
import sys
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
RUN_DIR = Path("/run")
|
|
REQUIRED_EVALUATION = (
|
|
"phrase_duration_evidence",
|
|
"release_negative_hours",
|
|
"release_false_activations",
|
|
"speaker_macro_recall",
|
|
"p95_detection_latency_ms",
|
|
)
|
|
|
|
|
|
def _sha256(path: Path) -> str:
|
|
return hashlib.sha256(path.read_bytes()).hexdigest()
|
|
|
|
|
|
def _load(path: Path) -> dict[str, Any]:
|
|
value = json.loads(path.read_text(encoding="utf-8"))
|
|
if not isinstance(value, dict):
|
|
raise ValueError(f"{path.name} must be a JSON object")
|
|
return value
|
|
|
|
|
|
def _canonical(value: dict[str, Any]) -> bytes:
|
|
return json.dumps(value, sort_keys=True, separators=(",", ":")).encode("utf-8")
|
|
|
|
|
|
def _validate_onnx(path: Path) -> None:
|
|
import onnx
|
|
|
|
model = onnx.load(path)
|
|
onnx.checker.check_model(model)
|
|
if not any(item.version == 18 for item in model.opset_import):
|
|
raise ValueError("classifier must use ONNX opset 18")
|
|
inputs = {item.name: item for item in model.graph.input}
|
|
outputs = {item.name for item in model.graph.output}
|
|
embeddings = inputs.get("embeddings")
|
|
if embeddings is None or "score" not in outputs:
|
|
raise ValueError("classifier must expose embeddings input and score output")
|
|
dimensions = embeddings.type.tensor_type.shape.dim
|
|
if embeddings.type.tensor_type.elem_type != onnx.TensorProto.FLOAT or [
|
|
item.dim_value for item in dimensions[-2:]
|
|
] != [16, 96]:
|
|
raise ValueError("classifier must accept FP32 (batch,16,96) embeddings")
|
|
|
|
|
|
def _upstream_eval(output_dir: Path) -> dict[str, Any] | None:
|
|
candidates = sorted(output_dir.rglob("*_eval.json"))
|
|
if len(candidates) != 1:
|
|
return None
|
|
return _load(candidates[0])
|
|
|
|
|
|
def _evaluation(
|
|
run_kind: str, primary_phrase: str, variants: list[str], output_dir: Path
|
|
) -> dict[str, Any]:
|
|
if run_kind == "commercial":
|
|
value = _load(RUN_DIR / "release-evaluation.json")
|
|
if any(key not in value for key in REQUIRED_EVALUATION):
|
|
raise ValueError("release-evaluation.json is incomplete")
|
|
phrases = [primary_phrase, *variants]
|
|
durations = value["phrase_duration_evidence"]
|
|
if not isinstance(durations, dict) or any(
|
|
phrase not in durations
|
|
or not isinstance(durations[phrase], list)
|
|
or not durations[phrase]
|
|
or any(
|
|
not isinstance(item, (int, float)) or not 0 < item <= 1.8
|
|
for item in durations[phrase]
|
|
)
|
|
for phrase in phrases
|
|
):
|
|
raise ValueError("phrase duration evidence is incomplete or exceeds 1.80 seconds")
|
|
return {**value, "schema_version": 1}
|
|
result = {
|
|
"schema_version": 1,
|
|
"phrase_duration_evidence": {phrase: [1.0] for phrase in [primary_phrase, *variants]},
|
|
"release_negative_hours": 0.0,
|
|
"release_false_activations": 0,
|
|
"speaker_macro_recall": 0.0,
|
|
"p95_detection_latency_ms": 0.0,
|
|
"quarantined": run_kind == "quarantined_smoke",
|
|
}
|
|
upstream = _upstream_eval(output_dir)
|
|
if upstream is not None:
|
|
result["upstream_metrics"] = upstream
|
|
return result
|
|
|
|
|
|
def _upstream_evaluation_artifacts(output_dir: Path, *, commercial: bool) -> list[dict[str, str]]:
|
|
"""Record the raw artefacts emitted by LiveKit's ``eval`` stage.
|
|
|
|
The catalog consumes our compact JSON evidence, but keeping hashes of the
|
|
upstream output makes that evidence independently reviewable. We do not
|
|
copy or reinterpret the files: LiveKit keeps its native filenames and
|
|
format under the sealed output directory.
|
|
"""
|
|
|
|
candidates = [
|
|
path
|
|
for path in output_dir.rglob("*")
|
|
if path.is_file()
|
|
and path.name not in {"manifest.json", "evaluation.json", "classifier.onnx"}
|
|
and ("eval" in path.name.lower() or "det" in path.name.lower())
|
|
]
|
|
if commercial and not candidates:
|
|
raise ValueError("commercial output is missing an upstream LiveKit evaluation artifact")
|
|
return [
|
|
{
|
|
"path": path.relative_to(output_dir).as_posix(),
|
|
"sha256": _sha256(path),
|
|
}
|
|
for path in sorted(candidates)
|
|
]
|
|
|
|
|
|
def main(config_path: Path) -> int:
|
|
request = _load(RUN_DIR / "request.json")["request"]
|
|
if not isinstance(request, dict):
|
|
raise ValueError("request payload is invalid")
|
|
run_kind = str(request["run_kind"])
|
|
if run_kind not in {"commercial", "local_experiment", "quarantined_smoke"}:
|
|
raise ValueError("request run_kind is invalid")
|
|
config = _load(config_path)
|
|
family_slug = config.get("family_slug")
|
|
version = config.get("version")
|
|
threshold = config.get("threshold")
|
|
if (
|
|
not isinstance(family_slug, str)
|
|
or not isinstance(version, int)
|
|
or not (isinstance(threshold, (int, float)) or threshold == "auto")
|
|
):
|
|
raise ValueError("effective config requires family_slug, version, and threshold")
|
|
raw_variants = request["variants"]
|
|
if not isinstance(raw_variants, list) or not all(
|
|
isinstance(item, str) for item in raw_variants
|
|
):
|
|
raise ValueError("request variants are invalid")
|
|
variants = list(raw_variants)
|
|
primary_phrase = str(request["primary_phrase"])
|
|
output_dir = RUN_DIR / "output"
|
|
evaluation = _evaluation(run_kind, primary_phrase, variants, output_dir)
|
|
if threshold == "auto":
|
|
upstream_metrics = evaluation.get("upstream_metrics")
|
|
if not isinstance(upstream_metrics, dict) or not isinstance(
|
|
upstream_metrics.get("optimal_threshold"), (int, float)
|
|
):
|
|
raise ValueError("automatic threshold requires one upstream evaluation result")
|
|
threshold = float(upstream_metrics["optimal_threshold"])
|
|
classifiers = [path for path in output_dir.rglob("*.onnx") if path.is_file()]
|
|
if len(classifiers) != 1:
|
|
raise ValueError("upstream output must contain exactly one ONNX classifier")
|
|
_validate_onnx(classifiers[0])
|
|
classifier = output_dir / "classifier.onnx"
|
|
if classifiers[0] != classifier:
|
|
shutil.copyfile(classifiers[0], classifier)
|
|
image = _load(RUN_DIR / "image-provenance.json")
|
|
source = _load(RUN_DIR / "sources.lock.json")
|
|
upstream_artifacts = _upstream_evaluation_artifacts(
|
|
output_dir, commercial=run_kind == "commercial"
|
|
)
|
|
evaluation["upstream_cli_artifacts"] = upstream_artifacts
|
|
publication_status = {
|
|
"commercial": "review_required",
|
|
"local_experiment": "local_only",
|
|
"quarantined_smoke": "quarantined",
|
|
}[run_kind]
|
|
evaluation["promotion"] = {
|
|
"eligible": False,
|
|
"status": publication_status,
|
|
}
|
|
manifest = {
|
|
"schema_version": 1,
|
|
"family_slug": family_slug,
|
|
"version": version,
|
|
"primary_phrase": primary_phrase,
|
|
"phrase_variants": variants,
|
|
"language": "en",
|
|
"threshold": float(threshold),
|
|
"debounce_ms": 2000,
|
|
"runtime_compatibility": {
|
|
"format": "onnx", "architecture": "conv_attention", "precision": "fp32", "opset": 18
|
|
},
|
|
"runner_provenance": {
|
|
"image_digest": request["runner_image_digest"],
|
|
"livekit_wakeword_version": "0.2.1",
|
|
"source_revision": "1ec7f680df30ff4ca0ebae6b5983441e94b10980",
|
|
"dependency_lock_sha256": image["dependency_lock"]["sha256"],
|
|
},
|
|
"source_provenance": {
|
|
"source_lock_sha256": _sha256(RUN_DIR / "sources.lock.json"),
|
|
"approved": run_kind == "commercial",
|
|
"inventory": [
|
|
item.get("source_id", item.get("id", "unidentified"))
|
|
for item in source.get("sources", [])
|
|
if isinstance(item, dict)
|
|
],
|
|
},
|
|
"evaluation_summary": evaluation,
|
|
"artifacts": {"onnx_sha256": _sha256(classifier), "evaluation_sha256": ""},
|
|
}
|
|
evaluation_path = output_dir / "evaluation.json"
|
|
evaluation_path.write_bytes(_canonical(evaluation))
|
|
manifest["artifacts"]["evaluation_sha256"] = _sha256(evaluation_path)
|
|
(RUN_DIR / "output" / "manifest.json").write_bytes(_canonical(manifest))
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main(Path(sys.argv[1])))
|